Index: vendor/llvm/dist-release_80/cmake/modules/AddLLVM.cmake =================================================================== --- vendor/llvm/dist-release_80/cmake/modules/AddLLVM.cmake (revision 343793) +++ vendor/llvm/dist-release_80/cmake/modules/AddLLVM.cmake (revision 343794) @@ -1,1742 +1,1741 @@ include(LLVMProcessSources) include(LLVM-Config) include(DetermineGCCCompatible) function(llvm_update_compile_flags name) get_property(sources TARGET ${name} PROPERTY SOURCES) if("${sources}" MATCHES "\\.c(;|$)") set(update_src_props ON) endif() # LLVM_REQUIRES_EH is an internal flag that individual targets can use to # force EH if(LLVM_REQUIRES_EH OR LLVM_ENABLE_EH) if(NOT (LLVM_REQUIRES_RTTI OR LLVM_ENABLE_RTTI)) message(AUTHOR_WARNING "Exception handling requires RTTI. Enabling RTTI for ${name}") set(LLVM_REQUIRES_RTTI ON) endif() if(MSVC) list(APPEND LLVM_COMPILE_FLAGS "/EHsc") endif() else() if(LLVM_COMPILER_IS_GCC_COMPATIBLE) list(APPEND LLVM_COMPILE_FLAGS "-fno-exceptions") elseif(MSVC) list(APPEND LLVM_COMPILE_DEFINITIONS _HAS_EXCEPTIONS=0) list(APPEND LLVM_COMPILE_FLAGS "/EHs-c-") endif() endif() # LLVM_REQUIRES_RTTI is an internal flag that individual # targets can use to force RTTI set(LLVM_CONFIG_HAS_RTTI YES CACHE INTERNAL "") if(NOT (LLVM_REQUIRES_RTTI OR LLVM_ENABLE_RTTI)) set(LLVM_CONFIG_HAS_RTTI NO CACHE INTERNAL "") list(APPEND LLVM_COMPILE_DEFINITIONS GTEST_HAS_RTTI=0) if (LLVM_COMPILER_IS_GCC_COMPATIBLE) list(APPEND LLVM_COMPILE_FLAGS "-fno-rtti") elseif (MSVC) list(APPEND LLVM_COMPILE_FLAGS "/GR-") endif () elseif(MSVC) list(APPEND LLVM_COMPILE_FLAGS "/GR") endif() # Assume that; # - LLVM_COMPILE_FLAGS is list. # - PROPERTY COMPILE_FLAGS is string. string(REPLACE ";" " " target_compile_flags " ${LLVM_COMPILE_FLAGS}") if(update_src_props) foreach(fn ${sources}) get_filename_component(suf ${fn} EXT) if("${suf}" STREQUAL ".cpp") set_property(SOURCE ${fn} APPEND_STRING PROPERTY COMPILE_FLAGS "${target_compile_flags}") endif() endforeach() else() # Update target props, since all sources are C++. set_property(TARGET ${name} APPEND_STRING PROPERTY COMPILE_FLAGS "${target_compile_flags}") endif() set_property(TARGET ${name} APPEND PROPERTY COMPILE_DEFINITIONS ${LLVM_COMPILE_DEFINITIONS}) endfunction() function(add_llvm_symbol_exports target_name export_file) if(${CMAKE_SYSTEM_NAME} MATCHES "Darwin") set(native_export_file "${target_name}.exports") add_custom_command(OUTPUT ${native_export_file} COMMAND sed -e "s/^/_/" < ${export_file} > ${native_export_file} DEPENDS ${export_file} VERBATIM COMMENT "Creating export file for ${target_name}") set_property(TARGET ${target_name} APPEND_STRING PROPERTY LINK_FLAGS " -Wl,-exported_symbols_list,\"${CMAKE_CURRENT_BINARY_DIR}/${native_export_file}\"") elseif(${CMAKE_SYSTEM_NAME} MATCHES "AIX") set_property(TARGET ${target_name} APPEND_STRING PROPERTY LINK_FLAGS " -Wl,-bE:${export_file}") elseif(LLVM_HAVE_LINK_VERSION_SCRIPT) # Gold and BFD ld require a version script rather than a plain list. set(native_export_file "${target_name}.exports") # FIXME: Don't write the "local:" line on OpenBSD. # in the export file, also add a linker script to version LLVM symbols (form: LLVM_N.M) add_custom_command(OUTPUT ${native_export_file} COMMAND echo "LLVM_${LLVM_VERSION_MAJOR} {" > ${native_export_file} COMMAND grep -q "[[:alnum:]]" ${export_file} && echo " global:" >> ${native_export_file} || : COMMAND sed -e "s/$/;/" -e "s/^/ /" < ${export_file} >> ${native_export_file} COMMAND echo " local: *;" >> ${native_export_file} COMMAND echo "};" >> ${native_export_file} DEPENDS ${export_file} VERBATIM COMMENT "Creating export file for ${target_name}") if (${LLVM_LINKER_IS_SOLARISLD}) set_property(TARGET ${target_name} APPEND_STRING PROPERTY LINK_FLAGS " -Wl,-M,\"${CMAKE_CURRENT_BINARY_DIR}/${native_export_file}\"") else() set_property(TARGET ${target_name} APPEND_STRING PROPERTY LINK_FLAGS " -Wl,--version-script,\"${CMAKE_CURRENT_BINARY_DIR}/${native_export_file}\"") endif() else() set(native_export_file "${target_name}.def") add_custom_command(OUTPUT ${native_export_file} COMMAND ${PYTHON_EXECUTABLE} -c "import sys;print(''.join(['EXPORTS\\n']+sys.stdin.readlines(),))" < ${export_file} > ${native_export_file} DEPENDS ${export_file} VERBATIM COMMENT "Creating export file for ${target_name}") set(export_file_linker_flag "${CMAKE_CURRENT_BINARY_DIR}/${native_export_file}") if(MSVC) set(export_file_linker_flag "/DEF:\"${export_file_linker_flag}\"") endif() set_property(TARGET ${target_name} APPEND_STRING PROPERTY LINK_FLAGS " ${export_file_linker_flag}") endif() add_custom_target(${target_name}_exports DEPENDS ${native_export_file}) set_target_properties(${target_name}_exports PROPERTIES FOLDER "Misc") get_property(srcs TARGET ${target_name} PROPERTY SOURCES) foreach(src ${srcs}) get_filename_component(extension ${src} EXT) if(extension STREQUAL ".cpp") set(first_source_file ${src}) break() endif() endforeach() # Force re-linking when the exports file changes. Actually, it # forces recompilation of the source file. The LINK_DEPENDS target # property only works for makefile-based generators. # FIXME: This is not safe because this will create the same target # ${native_export_file} in several different file: # - One where we emitted ${target_name}_exports # - One where we emitted the build command for the following object. # set_property(SOURCE ${first_source_file} APPEND PROPERTY # OBJECT_DEPENDS ${CMAKE_CURRENT_BINARY_DIR}/${native_export_file}) set_property(DIRECTORY APPEND PROPERTY ADDITIONAL_MAKE_CLEAN_FILES ${native_export_file}) add_dependencies(${target_name} ${target_name}_exports) # Add dependency to *_exports later -- CMake issue 14747 list(APPEND LLVM_COMMON_DEPENDS ${target_name}_exports) set(LLVM_COMMON_DEPENDS ${LLVM_COMMON_DEPENDS} PARENT_SCOPE) endfunction(add_llvm_symbol_exports) if(APPLE) execute_process( COMMAND "${CMAKE_LINKER}" -v ERROR_VARIABLE stderr ) set(LLVM_LINKER_DETECTED YES) if("${stderr}" MATCHES "PROJECT:ld64") set(LLVM_LINKER_IS_LD64 YES) message(STATUS "Linker detection: ld64") else() set(LLVM_LINKER_DETECTED NO) message(STATUS "Linker detection: unknown") endif() elseif(NOT WIN32) # Detect what linker we have here if( LLVM_USE_LINKER ) set(command ${CMAKE_C_COMPILER} -fuse-ld=${LLVM_USE_LINKER} -Wl,--version) else() separate_arguments(flags UNIX_COMMAND "${CMAKE_EXE_LINKER_FLAGS}") set(command ${CMAKE_C_COMPILER} ${flags} -Wl,--version) endif() execute_process( COMMAND ${command} OUTPUT_VARIABLE stdout ERROR_VARIABLE stderr ) set(LLVM_LINKER_DETECTED YES) if("${stdout}" MATCHES "GNU gold") set(LLVM_LINKER_IS_GOLD YES) message(STATUS "Linker detection: GNU Gold") elseif("${stdout}" MATCHES "^LLD") set(LLVM_LINKER_IS_LLD YES) message(STATUS "Linker detection: LLD") elseif("${stdout}" MATCHES "GNU ld") set(LLVM_LINKER_IS_GNULD YES) message(STATUS "Linker detection: GNU ld") elseif("${stderr}" MATCHES "Solaris Link Editors" OR "${stdout}" MATCHES "Solaris Link Editors") set(LLVM_LINKER_IS_SOLARISLD YES) message(STATUS "Linker detection: Solaris ld") else() set(LLVM_LINKER_DETECTED NO) message(STATUS "Linker detection: unknown") endif() endif() function(add_link_opts target_name) # Don't use linker optimizations in debug builds since it slows down the # linker in a context where the optimizations are not important. if (NOT uppercase_CMAKE_BUILD_TYPE STREQUAL "DEBUG") # Pass -O3 to the linker. This enabled different optimizations on different # linkers. if(NOT (${CMAKE_SYSTEM_NAME} MATCHES "Darwin|SunOS|AIX" OR WIN32)) set_property(TARGET ${target_name} APPEND_STRING PROPERTY LINK_FLAGS " -Wl,-O3") endif() if(LLVM_LINKER_IS_GOLD) # With gold gc-sections is always safe. set_property(TARGET ${target_name} APPEND_STRING PROPERTY LINK_FLAGS " -Wl,--gc-sections") # Note that there is a bug with -Wl,--icf=safe so it is not safe # to enable. See https://sourceware.org/bugzilla/show_bug.cgi?id=17704. endif() if(NOT LLVM_NO_DEAD_STRIP) if(${CMAKE_SYSTEM_NAME} MATCHES "Darwin") # ld64's implementation of -dead_strip breaks tools that use plugins. set_property(TARGET ${target_name} APPEND_STRING PROPERTY LINK_FLAGS " -Wl,-dead_strip") elseif(${CMAKE_SYSTEM_NAME} MATCHES "SunOS") set_property(TARGET ${target_name} APPEND_STRING PROPERTY LINK_FLAGS " -Wl,-z -Wl,discard-unused=sections") elseif(NOT WIN32 AND NOT LLVM_LINKER_IS_GOLD AND NOT ${CMAKE_SYSTEM_NAME} MATCHES "OpenBSD") # Object files are compiled with -ffunction-data-sections. # Versions of bfd ld < 2.23.1 have a bug in --gc-sections that breaks # tools that use plugins. Always pass --gc-sections once we require # a newer linker. set_property(TARGET ${target_name} APPEND_STRING PROPERTY LINK_FLAGS " -Wl,--gc-sections") endif() endif() endif() endfunction(add_link_opts) # Set each output directory according to ${CMAKE_CONFIGURATION_TYPES}. # Note: Don't set variables CMAKE_*_OUTPUT_DIRECTORY any more, # or a certain builder, for eaxample, msbuild.exe, would be confused. function(set_output_directory target) cmake_parse_arguments(ARG "" "BINARY_DIR;LIBRARY_DIR" "" ${ARGN}) # module_dir -- corresponding to LIBRARY_OUTPUT_DIRECTORY. # It affects output of add_library(MODULE). if(WIN32 OR CYGWIN) # DLL platform set(module_dir ${ARG_BINARY_DIR}) else() set(module_dir ${ARG_LIBRARY_DIR}) endif() if(NOT "${CMAKE_CFG_INTDIR}" STREQUAL ".") foreach(build_mode ${CMAKE_CONFIGURATION_TYPES}) string(TOUPPER "${build_mode}" CONFIG_SUFFIX) if(ARG_BINARY_DIR) string(REPLACE ${CMAKE_CFG_INTDIR} ${build_mode} bi ${ARG_BINARY_DIR}) set_target_properties(${target} PROPERTIES "RUNTIME_OUTPUT_DIRECTORY_${CONFIG_SUFFIX}" ${bi}) endif() if(ARG_LIBRARY_DIR) string(REPLACE ${CMAKE_CFG_INTDIR} ${build_mode} li ${ARG_LIBRARY_DIR}) set_target_properties(${target} PROPERTIES "ARCHIVE_OUTPUT_DIRECTORY_${CONFIG_SUFFIX}" ${li}) endif() if(module_dir) string(REPLACE ${CMAKE_CFG_INTDIR} ${build_mode} mi ${module_dir}) set_target_properties(${target} PROPERTIES "LIBRARY_OUTPUT_DIRECTORY_${CONFIG_SUFFIX}" ${mi}) endif() endforeach() else() if(ARG_BINARY_DIR) set_target_properties(${target} PROPERTIES RUNTIME_OUTPUT_DIRECTORY ${ARG_BINARY_DIR}) endif() if(ARG_LIBRARY_DIR) set_target_properties(${target} PROPERTIES ARCHIVE_OUTPUT_DIRECTORY ${ARG_LIBRARY_DIR}) endif() if(module_dir) set_target_properties(${target} PROPERTIES LIBRARY_OUTPUT_DIRECTORY ${module_dir}) endif() endif() endfunction() # If on Windows and building with MSVC, add the resource script containing the # VERSIONINFO data to the project. This embeds version resource information # into the output .exe or .dll. # TODO: Enable for MinGW Windows builds too. # function(add_windows_version_resource_file OUT_VAR) set(sources ${ARGN}) if (MSVC AND CMAKE_HOST_SYSTEM_NAME STREQUAL "Windows") set(resource_file ${LLVM_SOURCE_DIR}/resources/windows_version_resource.rc) if(EXISTS ${resource_file}) set(sources ${sources} ${resource_file}) source_group("Resource Files" ${resource_file}) set(windows_resource_file ${resource_file} PARENT_SCOPE) endif() endif(MSVC AND CMAKE_HOST_SYSTEM_NAME STREQUAL "Windows") set(${OUT_VAR} ${sources} PARENT_SCOPE) endfunction(add_windows_version_resource_file) # set_windows_version_resource_properties(name resource_file... # VERSION_MAJOR int # Optional major version number (defaults to LLVM_VERSION_MAJOR) # VERSION_MINOR int # Optional minor version number (defaults to LLVM_VERSION_MINOR) # VERSION_PATCHLEVEL int # Optional patchlevel version number (defaults to LLVM_VERSION_PATCH) # VERSION_STRING # Optional version string (defaults to PACKAGE_VERSION) # PRODUCT_NAME # Optional product name string (defaults to "LLVM") # ) function(set_windows_version_resource_properties name resource_file) cmake_parse_arguments(ARG "" "VERSION_MAJOR;VERSION_MINOR;VERSION_PATCHLEVEL;VERSION_STRING;PRODUCT_NAME" "" ${ARGN}) if (NOT DEFINED ARG_VERSION_MAJOR) set(ARG_VERSION_MAJOR ${LLVM_VERSION_MAJOR}) endif() if (NOT DEFINED ARG_VERSION_MINOR) set(ARG_VERSION_MINOR ${LLVM_VERSION_MINOR}) endif() if (NOT DEFINED ARG_VERSION_PATCHLEVEL) set(ARG_VERSION_PATCHLEVEL ${LLVM_VERSION_PATCH}) endif() if (NOT DEFINED ARG_VERSION_STRING) set(ARG_VERSION_STRING ${PACKAGE_VERSION}) endif() if (NOT DEFINED ARG_PRODUCT_NAME) set(ARG_PRODUCT_NAME "LLVM") endif() set_property(SOURCE ${resource_file} PROPERTY COMPILE_FLAGS /nologo) set_property(SOURCE ${resource_file} PROPERTY COMPILE_DEFINITIONS "RC_VERSION_FIELD_1=${ARG_VERSION_MAJOR}" "RC_VERSION_FIELD_2=${ARG_VERSION_MINOR}" "RC_VERSION_FIELD_3=${ARG_VERSION_PATCHLEVEL}" "RC_VERSION_FIELD_4=0" "RC_FILE_VERSION=\"${ARG_VERSION_STRING}\"" "RC_INTERNAL_NAME=\"${name}\"" "RC_PRODUCT_NAME=\"${ARG_PRODUCT_NAME}\"" "RC_PRODUCT_VERSION=\"${ARG_VERSION_STRING}\"") endfunction(set_windows_version_resource_properties) # llvm_add_library(name sources... # SHARED;STATIC # STATIC by default w/o BUILD_SHARED_LIBS. # SHARED by default w/ BUILD_SHARED_LIBS. # OBJECT # Also create an OBJECT library target. Default if STATIC && SHARED. # MODULE # Target ${name} might not be created on unsupported platforms. # Check with "if(TARGET ${name})". # DISABLE_LLVM_LINK_LLVM_DYLIB # Do not link this library to libLLVM, even if # LLVM_LINK_LLVM_DYLIB is enabled. # OUTPUT_NAME name # Corresponds to OUTPUT_NAME in target properties. # DEPENDS targets... # Same semantics as add_dependencies(). # LINK_COMPONENTS components... # Same as the variable LLVM_LINK_COMPONENTS. # LINK_LIBS lib_targets... # Same semantics as target_link_libraries(). # ADDITIONAL_HEADERS # May specify header files for IDE generators. # SONAME # Should set SONAME link flags and create symlinks # NO_INSTALL_RPATH # Suppress default RPATH settings in shared libraries. # PLUGIN_TOOL # The tool (i.e. cmake target) that this plugin will link against # ) function(llvm_add_library name) cmake_parse_arguments(ARG "MODULE;SHARED;STATIC;OBJECT;DISABLE_LLVM_LINK_LLVM_DYLIB;SONAME;NO_INSTALL_RPATH" "OUTPUT_NAME;PLUGIN_TOOL" "ADDITIONAL_HEADERS;DEPENDS;LINK_COMPONENTS;LINK_LIBS;OBJLIBS" ${ARGN}) list(APPEND LLVM_COMMON_DEPENDS ${ARG_DEPENDS}) if(ARG_ADDITIONAL_HEADERS) # Pass through ADDITIONAL_HEADERS. set(ARG_ADDITIONAL_HEADERS ADDITIONAL_HEADERS ${ARG_ADDITIONAL_HEADERS}) endif() if(ARG_OBJLIBS) set(ALL_FILES ${ARG_OBJLIBS}) else() llvm_process_sources(ALL_FILES ${ARG_UNPARSED_ARGUMENTS} ${ARG_ADDITIONAL_HEADERS}) endif() if(ARG_MODULE) if(ARG_SHARED OR ARG_STATIC) message(WARNING "MODULE with SHARED|STATIC doesn't make sense.") endif() # Plugins that link against a tool are allowed even when plugins in general are not if(NOT LLVM_ENABLE_PLUGINS AND NOT (ARG_PLUGIN_TOOL AND LLVM_EXPORT_SYMBOLS_FOR_PLUGINS)) message(STATUS "${name} ignored -- Loadable modules not supported on this platform.") return() endif() else() if(ARG_PLUGIN_TOOL) message(WARNING "PLUGIN_TOOL without MODULE doesn't make sense.") endif() if(BUILD_SHARED_LIBS AND NOT ARG_STATIC) set(ARG_SHARED TRUE) endif() if(NOT ARG_SHARED) set(ARG_STATIC TRUE) endif() endif() # Generate objlib if((ARG_SHARED AND ARG_STATIC) OR ARG_OBJECT) # Generate an obj library for both targets. set(obj_name "obj.${name}") add_library(${obj_name} OBJECT EXCLUDE_FROM_ALL ${ALL_FILES} ) llvm_update_compile_flags(${obj_name}) set(ALL_FILES "$") # Do add_dependencies(obj) later due to CMake issue 14747. list(APPEND objlibs ${obj_name}) set_target_properties(${obj_name} PROPERTIES FOLDER "Object Libraries") endif() if(ARG_SHARED AND ARG_STATIC) # static set(name_static "${name}_static") if(ARG_OUTPUT_NAME) set(output_name OUTPUT_NAME "${ARG_OUTPUT_NAME}") endif() # DEPENDS has been appended to LLVM_COMMON_LIBS. llvm_add_library(${name_static} STATIC ${output_name} OBJLIBS ${ALL_FILES} # objlib LINK_LIBS ${ARG_LINK_LIBS} LINK_COMPONENTS ${ARG_LINK_COMPONENTS} ) # FIXME: Add name_static to anywhere in TARGET ${name}'s PROPERTY. set(ARG_STATIC) endif() if(ARG_MODULE) add_library(${name} MODULE ${ALL_FILES}) elseif(ARG_SHARED) add_windows_version_resource_file(ALL_FILES ${ALL_FILES}) add_library(${name} SHARED ${ALL_FILES}) else() add_library(${name} STATIC ${ALL_FILES}) endif() if(NOT ARG_NO_INSTALL_RPATH) if(ARG_MODULE OR ARG_SHARED) llvm_setup_rpath(${name}) endif() endif() setup_dependency_debugging(${name} ${LLVM_COMMON_DEPENDS}) if(DEFINED windows_resource_file) set_windows_version_resource_properties(${name} ${windows_resource_file}) set(windows_resource_file ${windows_resource_file} PARENT_SCOPE) endif() set_output_directory(${name} BINARY_DIR ${LLVM_RUNTIME_OUTPUT_INTDIR} LIBRARY_DIR ${LLVM_LIBRARY_OUTPUT_INTDIR}) # $ doesn't require compile flags. if(NOT obj_name) llvm_update_compile_flags(${name}) endif() add_link_opts( ${name} ) if(ARG_OUTPUT_NAME) set_target_properties(${name} PROPERTIES OUTPUT_NAME ${ARG_OUTPUT_NAME} ) endif() if(ARG_MODULE) set_target_properties(${name} PROPERTIES PREFIX "" SUFFIX ${LLVM_PLUGIN_EXT} ) endif() if(ARG_SHARED) if(WIN32) set_target_properties(${name} PROPERTIES PREFIX "" ) endif() # Set SOVERSION on shared libraries that lack explicit SONAME # specifier, on *nix systems that are not Darwin. if(UNIX AND NOT APPLE AND NOT ARG_SONAME) set_target_properties(${name} PROPERTIES # Since 4.0.0, the ABI version is indicated by the major version SOVERSION ${LLVM_VERSION_MAJOR}${LLVM_VERSION_SUFFIX} VERSION ${LLVM_VERSION_MAJOR}${LLVM_VERSION_SUFFIX}) endif() endif() if(ARG_MODULE OR ARG_SHARED) # Do not add -Dname_EXPORTS to the command-line when building files in this # target. Doing so is actively harmful for the modules build because it # creates extra module variants, and not useful because we don't use these # macros. set_target_properties( ${name} PROPERTIES DEFINE_SYMBOL "" ) if (LLVM_EXPORTED_SYMBOL_FILE) add_llvm_symbol_exports( ${name} ${LLVM_EXPORTED_SYMBOL_FILE} ) endif() endif() if(ARG_SHARED AND UNIX) if(NOT APPLE AND ARG_SONAME) get_target_property(output_name ${name} OUTPUT_NAME) if(${output_name} STREQUAL "output_name-NOTFOUND") set(output_name ${name}) endif() set(library_name ${output_name}-${LLVM_VERSION_MAJOR}${LLVM_VERSION_SUFFIX}) set(api_name ${output_name}-${LLVM_VERSION_MAJOR}.${LLVM_VERSION_MINOR}.${LLVM_VERSION_PATCH}${LLVM_VERSION_SUFFIX}) set_target_properties(${name} PROPERTIES OUTPUT_NAME ${library_name}) llvm_install_library_symlink(${api_name} ${library_name} SHARED COMPONENT ${name} ALWAYS_GENERATE) llvm_install_library_symlink(${output_name} ${library_name} SHARED COMPONENT ${name} ALWAYS_GENERATE) endif() endif() if(ARG_MODULE AND LLVM_EXPORT_SYMBOLS_FOR_PLUGINS AND ARG_PLUGIN_TOOL AND (WIN32 OR CYGWIN)) # On DLL platforms symbols are imported from the tool by linking against it. set(llvm_libs ${ARG_PLUGIN_TOOL}) elseif (DEFINED LLVM_LINK_COMPONENTS OR DEFINED ARG_LINK_COMPONENTS) if (LLVM_LINK_LLVM_DYLIB AND NOT ARG_DISABLE_LLVM_LINK_LLVM_DYLIB) set(llvm_libs LLVM) else() llvm_map_components_to_libnames(llvm_libs ${ARG_LINK_COMPONENTS} ${LLVM_LINK_COMPONENTS} ) endif() else() # Components have not been defined explicitly in CMake, so add the # dependency information for this library as defined by LLVMBuild. # # It would be nice to verify that we have the dependencies for this library # name, but using get_property(... SET) doesn't suffice to determine if a # property has been set to an empty value. get_property(lib_deps GLOBAL PROPERTY LLVMBUILD_LIB_DEPS_${name}) endif() if(ARG_STATIC) set(libtype INTERFACE) else() # We can use PRIVATE since SO knows its dependent libs. set(libtype PRIVATE) endif() target_link_libraries(${name} ${libtype} ${ARG_LINK_LIBS} ${lib_deps} ${llvm_libs} ) if(LLVM_COMMON_DEPENDS) add_dependencies(${name} ${LLVM_COMMON_DEPENDS}) # Add dependencies also to objlibs. # CMake issue 14747 -- add_dependencies() might be ignored to objlib's user. foreach(objlib ${objlibs}) add_dependencies(${objlib} ${LLVM_COMMON_DEPENDS}) endforeach() endif() if(ARG_SHARED OR ARG_MODULE) llvm_externalize_debuginfo(${name}) llvm_codesign(${name}) endif() endfunction() function(add_llvm_install_targets target) cmake_parse_arguments(ARG "" "COMPONENT;PREFIX" "DEPENDS" ${ARGN}) if(ARG_COMPONENT) set(component_option -DCMAKE_INSTALL_COMPONENT="${ARG_COMPONENT}") endif() if(ARG_PREFIX) set(prefix_option -DCMAKE_INSTALL_PREFIX="${ARG_PREFIX}") endif() add_custom_target(${target} DEPENDS ${ARG_DEPENDS} COMMAND "${CMAKE_COMMAND}" ${component_option} ${prefix_option} -P "${CMAKE_BINARY_DIR}/cmake_install.cmake" USES_TERMINAL) add_custom_target(${target}-stripped DEPENDS ${ARG_DEPENDS} COMMAND "${CMAKE_COMMAND}" ${component_option} ${prefix_option} -DCMAKE_INSTALL_DO_STRIP=1 -P "${CMAKE_BINARY_DIR}/cmake_install.cmake" USES_TERMINAL) endfunction() macro(add_llvm_library name) cmake_parse_arguments(ARG "SHARED;BUILDTREE_ONLY;MODULE" "" "" ${ARGN}) if(ARG_MODULE) llvm_add_library(${name} MODULE ${ARG_UNPARSED_ARGUMENTS}) elseif( BUILD_SHARED_LIBS OR ARG_SHARED ) llvm_add_library(${name} SHARED ${ARG_UNPARSED_ARGUMENTS}) else() llvm_add_library(${name} ${ARG_UNPARSED_ARGUMENTS}) endif() # Libraries that are meant to only be exposed via the build tree only are # never installed and are only exported as a target in the special build tree # config file. if (NOT ARG_BUILDTREE_ONLY AND NOT ARG_MODULE) set_property( GLOBAL APPEND PROPERTY LLVM_LIBS ${name} ) endif() if (ARG_MODULE AND NOT TARGET ${name}) # Add empty "phony" target add_custom_target(${name}) elseif( EXCLUDE_FROM_ALL ) set_target_properties( ${name} PROPERTIES EXCLUDE_FROM_ALL ON) elseif(ARG_BUILDTREE_ONLY) set_property(GLOBAL APPEND PROPERTY LLVM_EXPORTS_BUILDTREE_ONLY ${name}) else() if (NOT LLVM_INSTALL_TOOLCHAIN_ONLY OR ${name} STREQUAL "LTO" OR ${name} STREQUAL "OptRemarks" OR (LLVM_LINK_LLVM_DYLIB AND ${name} STREQUAL "LLVM")) set(install_dir lib${LLVM_LIBDIR_SUFFIX}) if(ARG_MODULE OR ARG_SHARED OR BUILD_SHARED_LIBS) if(WIN32 OR CYGWIN OR MINGW) set(install_type RUNTIME) set(install_dir bin) else() set(install_type LIBRARY) endif() else() set(install_type ARCHIVE) endif() if (ARG_MODULE) set(install_type LIBRARY) endif() if(${name} IN_LIST LLVM_DISTRIBUTION_COMPONENTS OR NOT LLVM_DISTRIBUTION_COMPONENTS) set(export_to_llvmexports EXPORT LLVMExports) set_property(GLOBAL PROPERTY LLVM_HAS_EXPORTS True) endif() install(TARGETS ${name} ${export_to_llvmexports} ${install_type} DESTINATION ${install_dir} COMPONENT ${name}) if (NOT LLVM_ENABLE_IDE) add_llvm_install_targets(install-${name} DEPENDS ${name} COMPONENT ${name}) endif() endif() set_property(GLOBAL APPEND PROPERTY LLVM_EXPORTS ${name}) endif() if (ARG_MODULE) set_target_properties(${name} PROPERTIES FOLDER "Loadable modules") else() set_target_properties(${name} PROPERTIES FOLDER "Libraries") endif() endmacro(add_llvm_library name) macro(add_llvm_executable name) cmake_parse_arguments(ARG "DISABLE_LLVM_LINK_LLVM_DYLIB;IGNORE_EXTERNALIZE_DEBUGINFO;NO_INSTALL_RPATH" "ENTITLEMENTS" "DEPENDS" ${ARGN}) llvm_process_sources( ALL_FILES ${ARG_UNPARSED_ARGUMENTS} ) list(APPEND LLVM_COMMON_DEPENDS ${ARG_DEPENDS}) # Generate objlib if(LLVM_ENABLE_OBJLIB) # Generate an obj library for both targets. set(obj_name "obj.${name}") add_library(${obj_name} OBJECT EXCLUDE_FROM_ALL ${ALL_FILES} ) llvm_update_compile_flags(${obj_name}) set(ALL_FILES "$") set_target_properties(${obj_name} PROPERTIES FOLDER "Object Libraries") endif() add_windows_version_resource_file(ALL_FILES ${ALL_FILES}) if(XCODE) # Note: the dummy.cpp source file provides no definitions. However, # it forces Xcode to properly link the static library. list(APPEND ALL_FILES "${LLVM_MAIN_SRC_DIR}/cmake/dummy.cpp") endif() if( EXCLUDE_FROM_ALL ) add_executable(${name} EXCLUDE_FROM_ALL ${ALL_FILES}) else() add_executable(${name} ${ALL_FILES}) endif() setup_dependency_debugging(${name} ${LLVM_COMMON_DEPENDS}) if(NOT ARG_NO_INSTALL_RPATH) llvm_setup_rpath(${name}) endif() if(DEFINED windows_resource_file) set_windows_version_resource_properties(${name} ${windows_resource_file}) endif() # $ doesn't require compile flags. if(NOT LLVM_ENABLE_OBJLIB) llvm_update_compile_flags(${name}) endif() add_link_opts( ${name} ) # Do not add -Dname_EXPORTS to the command-line when building files in this # target. Doing so is actively harmful for the modules build because it # creates extra module variants, and not useful because we don't use these # macros. set_target_properties( ${name} PROPERTIES DEFINE_SYMBOL "" ) if (LLVM_EXPORTED_SYMBOL_FILE) add_llvm_symbol_exports( ${name} ${LLVM_EXPORTED_SYMBOL_FILE} ) endif(LLVM_EXPORTED_SYMBOL_FILE) if (LLVM_LINK_LLVM_DYLIB AND NOT ARG_DISABLE_LLVM_LINK_LLVM_DYLIB) set(USE_SHARED USE_SHARED) endif() set(EXCLUDE_FROM_ALL OFF) set_output_directory(${name} BINARY_DIR ${LLVM_RUNTIME_OUTPUT_INTDIR} LIBRARY_DIR ${LLVM_LIBRARY_OUTPUT_INTDIR}) llvm_config( ${name} ${USE_SHARED} ${LLVM_LINK_COMPONENTS} ) if( LLVM_COMMON_DEPENDS ) add_dependencies( ${name} ${LLVM_COMMON_DEPENDS} ) endif( LLVM_COMMON_DEPENDS ) if(NOT ARG_IGNORE_EXTERNALIZE_DEBUGINFO) llvm_externalize_debuginfo(${name}) endif() if (LLVM_PTHREAD_LIB) # libpthreads overrides some standard library symbols, so main # executable must be linked with it in order to provide consistent # API for all shared libaries loaded by this executable. target_link_libraries(${name} PRIVATE ${LLVM_PTHREAD_LIB}) endif() llvm_codesign(${name} ENTITLEMENTS ${ARG_ENTITLEMENTS}) endmacro(add_llvm_executable name) function(export_executable_symbols target) if (LLVM_EXPORTED_SYMBOL_FILE) # The symbol file should contain the symbols we want the executable to # export set_target_properties(${target} PROPERTIES ENABLE_EXPORTS 1) elseif (LLVM_EXPORT_SYMBOLS_FOR_PLUGINS) # Extract the symbols to export from the static libraries that the # executable links against. set_target_properties(${target} PROPERTIES ENABLE_EXPORTS 1) set(exported_symbol_file ${CMAKE_CURRENT_BINARY_DIR}/${CMAKE_CFG_INTDIR}/${target}.symbols) # We need to consider not just the direct link dependencies, but also the # transitive link dependencies. Do this by starting with the set of direct # dependencies, then the dependencies of those dependencies, and so on. get_target_property(new_libs ${target} LINK_LIBRARIES) set(link_libs ${new_libs}) while(NOT "${new_libs}" STREQUAL "") foreach(lib ${new_libs}) if(TARGET ${lib}) get_target_property(lib_type ${lib} TYPE) if("${lib_type}" STREQUAL "STATIC_LIBRARY") list(APPEND static_libs ${lib}) else() list(APPEND other_libs ${lib}) endif() get_target_property(transitive_libs ${lib} INTERFACE_LINK_LIBRARIES) foreach(transitive_lib ${transitive_libs}) list(FIND link_libs ${transitive_lib} idx) if(TARGET ${transitive_lib} AND idx EQUAL -1) list(APPEND newer_libs ${transitive_lib}) list(APPEND link_libs ${transitive_lib}) endif() endforeach(transitive_lib) endif() endforeach(lib) set(new_libs ${newer_libs}) set(newer_libs "") endwhile() if (MSVC) set(mangling microsoft) else() set(mangling itanium) endif() add_custom_command(OUTPUT ${exported_symbol_file} COMMAND ${PYTHON_EXECUTABLE} ${LLVM_MAIN_SRC_DIR}/utils/extract_symbols.py --mangling=${mangling} ${static_libs} -o ${exported_symbol_file} WORKING_DIRECTORY ${LLVM_LIBRARY_OUTPUT_INTDIR} DEPENDS ${LLVM_MAIN_SRC_DIR}/utils/extract_symbols.py ${static_libs} VERBATIM COMMENT "Generating export list for ${target}") add_llvm_symbol_exports( ${target} ${exported_symbol_file} ) # If something links against this executable then we want a # transitive link against only the libraries whose symbols # we aren't exporting. set_target_properties(${target} PROPERTIES INTERFACE_LINK_LIBRARIES "${other_libs}") # The default import library suffix that cmake uses for cygwin/mingw is # ".dll.a", but for clang.exe that causes a collision with libclang.dll, # where the import libraries of both get named libclang.dll.a. Use a suffix # of ".exe.a" to avoid this. if(CYGWIN OR MINGW) set_target_properties(${target} PROPERTIES IMPORT_SUFFIX ".exe.a") endif() elseif(NOT (WIN32 OR CYGWIN)) # On Windows auto-exporting everything doesn't work because of the limit on # the size of the exported symbol table, but on other platforms we can do # it without any trouble. set_target_properties(${target} PROPERTIES ENABLE_EXPORTS 1) if (APPLE) set_property(TARGET ${target} APPEND_STRING PROPERTY LINK_FLAGS " -rdynamic") endif() endif() endfunction() if(NOT LLVM_TOOLCHAIN_TOOLS) set (LLVM_TOOLCHAIN_TOOLS llvm-ar llvm-ranlib llvm-lib llvm-objdump llvm-rc ) endif() macro(add_llvm_tool name) if( NOT LLVM_BUILD_TOOLS ) set(EXCLUDE_FROM_ALL ON) endif() add_llvm_executable(${name} ${ARGN}) if ( ${name} IN_LIST LLVM_TOOLCHAIN_TOOLS OR NOT LLVM_INSTALL_TOOLCHAIN_ONLY) if( LLVM_BUILD_TOOLS ) if(${name} IN_LIST LLVM_DISTRIBUTION_COMPONENTS OR NOT LLVM_DISTRIBUTION_COMPONENTS) set(export_to_llvmexports EXPORT LLVMExports) set_property(GLOBAL PROPERTY LLVM_HAS_EXPORTS True) endif() install(TARGETS ${name} ${export_to_llvmexports} RUNTIME DESTINATION ${LLVM_TOOLS_INSTALL_DIR} COMPONENT ${name}) if (NOT LLVM_ENABLE_IDE) add_llvm_install_targets(install-${name} DEPENDS ${name} COMPONENT ${name}) endif() endif() endif() if( LLVM_BUILD_TOOLS ) set_property(GLOBAL APPEND PROPERTY LLVM_EXPORTS ${name}) endif() set_target_properties(${name} PROPERTIES FOLDER "Tools") endmacro(add_llvm_tool name) macro(add_llvm_example name) if( NOT LLVM_BUILD_EXAMPLES ) set(EXCLUDE_FROM_ALL ON) endif() add_llvm_executable(${name} ${ARGN}) if( LLVM_BUILD_EXAMPLES ) install(TARGETS ${name} RUNTIME DESTINATION examples) endif() set_target_properties(${name} PROPERTIES FOLDER "Examples") endmacro(add_llvm_example name) # This is a macro that is used to create targets for executables that are needed # for development, but that are not intended to be installed by default. macro(add_llvm_utility name) if ( NOT LLVM_BUILD_UTILS ) set(EXCLUDE_FROM_ALL ON) endif() add_llvm_executable(${name} DISABLE_LLVM_LINK_LLVM_DYLIB ${ARGN}) set_target_properties(${name} PROPERTIES FOLDER "Utils") if( LLVM_INSTALL_UTILS AND LLVM_BUILD_UTILS ) install (TARGETS ${name} RUNTIME DESTINATION ${LLVM_UTILS_INSTALL_DIR} COMPONENT ${name}) if (NOT LLVM_ENABLE_IDE) add_llvm_install_targets(install-${name} DEPENDS ${name} COMPONENT ${name}) endif() set_property(GLOBAL APPEND PROPERTY LLVM_EXPORTS ${name}) elseif( LLVM_BUILD_UTILS ) set_property(GLOBAL APPEND PROPERTY LLVM_EXPORTS_BUILDTREE_ONLY ${name}) endif() endmacro(add_llvm_utility name) macro(add_llvm_fuzzer name) cmake_parse_arguments(ARG "" "DUMMY_MAIN" "" ${ARGN}) if( LLVM_LIB_FUZZING_ENGINE ) set(LLVM_OPTIONAL_SOURCES ${ARG_DUMMY_MAIN}) add_llvm_executable(${name} ${ARG_UNPARSED_ARGUMENTS}) target_link_libraries(${name} PRIVATE ${LLVM_LIB_FUZZING_ENGINE}) set_target_properties(${name} PROPERTIES FOLDER "Fuzzers") elseif( LLVM_USE_SANITIZE_COVERAGE ) set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -fsanitize=fuzzer") set(LLVM_OPTIONAL_SOURCES ${ARG_DUMMY_MAIN}) add_llvm_executable(${name} ${ARG_UNPARSED_ARGUMENTS}) set_target_properties(${name} PROPERTIES FOLDER "Fuzzers") elseif( ARG_DUMMY_MAIN ) add_llvm_executable(${name} ${ARG_DUMMY_MAIN} ${ARG_UNPARSED_ARGUMENTS}) set_target_properties(${name} PROPERTIES FOLDER "Fuzzers") endif() endmacro() macro(add_llvm_target target_name) include_directories(BEFORE ${CMAKE_CURRENT_BINARY_DIR} ${CMAKE_CURRENT_SOURCE_DIR}) add_llvm_library(LLVM${target_name} ${ARGN}) set( CURRENT_LLVM_TARGET LLVM${target_name} ) endmacro(add_llvm_target) function(canonicalize_tool_name name output) string(REPLACE "${CMAKE_CURRENT_SOURCE_DIR}/" "" nameStrip ${name}) string(REPLACE "-" "_" nameUNDERSCORE ${nameStrip}) string(TOUPPER ${nameUNDERSCORE} nameUPPER) set(${output} "${nameUPPER}" PARENT_SCOPE) endfunction(canonicalize_tool_name) # Custom add_subdirectory wrapper # Takes in a project name (i.e. LLVM), the subdirectory name, and an optional # path if it differs from the name. function(add_llvm_subdirectory project type name) set(add_llvm_external_dir "${ARGN}") if("${add_llvm_external_dir}" STREQUAL "") set(add_llvm_external_dir ${name}) endif() canonicalize_tool_name(${name} nameUPPER) set(canonical_full_name ${project}_${type}_${nameUPPER}) get_property(already_processed GLOBAL PROPERTY ${canonical_full_name}_PROCESSED) if(already_processed) return() endif() set_property(GLOBAL PROPERTY ${canonical_full_name}_PROCESSED YES) if(EXISTS ${CMAKE_CURRENT_SOURCE_DIR}/${add_llvm_external_dir}/CMakeLists.txt) # Treat it as in-tree subproject. option(${canonical_full_name}_BUILD "Whether to build ${name} as part of ${project}" On) mark_as_advanced(${project}_${type}_${name}_BUILD) if(${canonical_full_name}_BUILD) add_subdirectory(${CMAKE_CURRENT_SOURCE_DIR}/${add_llvm_external_dir} ${add_llvm_external_dir}) endif() else() set(LLVM_EXTERNAL_${nameUPPER}_SOURCE_DIR "${LLVM_EXTERNAL_${nameUPPER}_SOURCE_DIR}" CACHE PATH "Path to ${name} source directory") set(${canonical_full_name}_BUILD_DEFAULT ON) if(NOT LLVM_EXTERNAL_${nameUPPER}_SOURCE_DIR OR NOT EXISTS ${LLVM_EXTERNAL_${nameUPPER}_SOURCE_DIR}) set(${canonical_full_name}_BUILD_DEFAULT OFF) endif() if("${LLVM_EXTERNAL_${nameUPPER}_BUILD}" STREQUAL "OFF") set(${canonical_full_name}_BUILD_DEFAULT OFF) endif() option(${canonical_full_name}_BUILD "Whether to build ${name} as part of LLVM" ${${canonical_full_name}_BUILD_DEFAULT}) if (${canonical_full_name}_BUILD) if(EXISTS ${LLVM_EXTERNAL_${nameUPPER}_SOURCE_DIR}) add_subdirectory(${LLVM_EXTERNAL_${nameUPPER}_SOURCE_DIR} ${add_llvm_external_dir}) elseif(NOT "${LLVM_EXTERNAL_${nameUPPER}_SOURCE_DIR}" STREQUAL "") message(WARNING "Nonexistent directory for ${name}: ${LLVM_EXTERNAL_${nameUPPER}_SOURCE_DIR}") endif() endif() endif() endfunction() # Add external project that may want to be built as part of llvm such as Clang, # lld, and Polly. This adds two options. One for the source directory of the # project, which defaults to ${CMAKE_CURRENT_SOURCE_DIR}/${name}. Another to # enable or disable building it with everything else. # Additional parameter can be specified as the name of directory. macro(add_llvm_external_project name) add_llvm_subdirectory(LLVM TOOL ${name} ${ARGN}) endmacro() macro(add_llvm_tool_subdirectory name) add_llvm_external_project(${name}) endmacro(add_llvm_tool_subdirectory) function(get_project_name_from_src_var var output) string(REGEX MATCH "LLVM_EXTERNAL_(.*)_SOURCE_DIR" MACHED_TOOL "${var}") if(MACHED_TOOL) set(${output} ${CMAKE_MATCH_1} PARENT_SCOPE) else() set(${output} PARENT_SCOPE) endif() endfunction() function(create_subdirectory_options project type) file(GLOB sub-dirs "${CMAKE_CURRENT_SOURCE_DIR}/*") foreach(dir ${sub-dirs}) if(IS_DIRECTORY "${dir}" AND EXISTS "${dir}/CMakeLists.txt") canonicalize_tool_name(${dir} name) option(${project}_${type}_${name}_BUILD "Whether to build ${name} as part of ${project}" On) mark_as_advanced(${project}_${type}_${name}_BUILD) endif() endforeach() endfunction(create_subdirectory_options) function(create_llvm_tool_options) create_subdirectory_options(LLVM TOOL) endfunction(create_llvm_tool_options) function(llvm_add_implicit_projects project) set(list_of_implicit_subdirs "") file(GLOB sub-dirs "${CMAKE_CURRENT_SOURCE_DIR}/*") foreach(dir ${sub-dirs}) if(IS_DIRECTORY "${dir}" AND EXISTS "${dir}/CMakeLists.txt") canonicalize_tool_name(${dir} name) if (${project}_TOOL_${name}_BUILD) get_filename_component(fn "${dir}" NAME) list(APPEND list_of_implicit_subdirs "${fn}") endif() endif() endforeach() foreach(external_proj ${list_of_implicit_subdirs}) add_llvm_subdirectory(${project} TOOL "${external_proj}" ${ARGN}) endforeach() endfunction(llvm_add_implicit_projects) function(add_llvm_implicit_projects) llvm_add_implicit_projects(LLVM) endfunction(add_llvm_implicit_projects) # Generic support for adding a unittest. function(add_unittest test_suite test_name) if( NOT LLVM_BUILD_TESTS ) set(EXCLUDE_FROM_ALL ON) endif() # Our current version of gtest does not properly recognize C++11 support # with MSVC, so it falls back to tr1 / experimental classes. Since LLVM # itself requires C++11, we can safely force it on unconditionally so that # we don't have to fight with the buggy gtest check. add_definitions(-DGTEST_LANG_CXX11=1) add_definitions(-DGTEST_HAS_TR1_TUPLE=0) include_directories(${LLVM_MAIN_SRC_DIR}/utils/unittest/googletest/include) include_directories(${LLVM_MAIN_SRC_DIR}/utils/unittest/googlemock/include) if (NOT LLVM_ENABLE_THREADS) list(APPEND LLVM_COMPILE_DEFINITIONS GTEST_HAS_PTHREAD=0) endif () if (SUPPORTS_VARIADIC_MACROS_FLAG) list(APPEND LLVM_COMPILE_FLAGS "-Wno-variadic-macros") endif () # Some parts of gtest rely on this GNU extension, don't warn on it. if(SUPPORTS_GNU_ZERO_VARIADIC_MACRO_ARGUMENTS_FLAG) list(APPEND LLVM_COMPILE_FLAGS "-Wno-gnu-zero-variadic-macro-arguments") endif() set(LLVM_REQUIRES_RTTI OFF) list(APPEND LLVM_LINK_COMPONENTS Support) # gtest needs it for raw_ostream add_llvm_executable(${test_name} IGNORE_EXTERNALIZE_DEBUGINFO NO_INSTALL_RPATH ${ARGN}) set(outdir ${CMAKE_CURRENT_BINARY_DIR}/${CMAKE_CFG_INTDIR}) set_output_directory(${test_name} BINARY_DIR ${outdir} LIBRARY_DIR ${outdir}) # libpthreads overrides some standard library symbols, so main # executable must be linked with it in order to provide consistent # API for all shared libaries loaded by this executable. target_link_libraries(${test_name} PRIVATE gtest_main gtest ${LLVM_PTHREAD_LIB}) add_dependencies(${test_suite} ${test_name}) get_target_property(test_suite_folder ${test_suite} FOLDER) if (NOT ${test_suite_folder} STREQUAL "NOTFOUND") set_property(TARGET ${test_name} PROPERTY FOLDER "${test_suite_folder}") endif () endfunction() # Use for test binaries that call llvm::getInputFileDirectory(). Use of this # is discouraged. function(add_unittest_with_input_files test_suite test_name) set(LLVM_UNITTEST_SOURCE_DIR ${CMAKE_CURRENT_SOURCE_DIR}) configure_file( ${LLVM_MAIN_SRC_DIR}/unittests/unittest.cfg.in ${CMAKE_CURRENT_BINARY_DIR}/llvm.srcdir.txt) add_unittest(${test_suite} ${test_name} ${ARGN}) endfunction() # Generic support for adding a benchmark. function(add_benchmark benchmark_name) if( NOT LLVM_BUILD_BENCHMARKS ) set(EXCLUDE_FROM_ALL ON) endif() add_llvm_executable(${benchmark_name} IGNORE_EXTERNALIZE_DEBUGINFO NO_INSTALL_RPATH ${ARGN}) set(outdir ${CMAKE_CURRENT_BINARY_DIR}/${CMAKE_CFG_INTDIR}) set_output_directory(${benchmark_name} BINARY_DIR ${outdir} LIBRARY_DIR ${outdir}) set_property(TARGET ${benchmark_name} PROPERTY FOLDER "Utils") target_link_libraries(${benchmark_name} PRIVATE benchmark) endfunction() function(llvm_add_go_executable binary pkgpath) cmake_parse_arguments(ARG "ALL" "" "DEPENDS;GOFLAGS" ${ARGN}) if(LLVM_BINDINGS MATCHES "go") # FIXME: This should depend only on the libraries Go needs. get_property(llvmlibs GLOBAL PROPERTY LLVM_LIBS) set(binpath ${CMAKE_BINARY_DIR}/bin/${binary}${CMAKE_EXECUTABLE_SUFFIX}) set(cc "${CMAKE_C_COMPILER} ${CMAKE_C_COMPILER_ARG1}") set(cxx "${CMAKE_CXX_COMPILER} ${CMAKE_CXX_COMPILER_ARG1}") set(cppflags "") get_property(include_dirs DIRECTORY PROPERTY INCLUDE_DIRECTORIES) foreach(d ${include_dirs}) set(cppflags "${cppflags} -I${d}") endforeach(d) set(ldflags "${CMAKE_EXE_LINKER_FLAGS}") add_custom_command(OUTPUT ${binpath} COMMAND ${CMAKE_BINARY_DIR}/bin/llvm-go "go=${GO_EXECUTABLE}" "cc=${cc}" "cxx=${cxx}" "cppflags=${cppflags}" "ldflags=${ldflags}" "packages=${LLVM_GO_PACKAGES}" ${ARG_GOFLAGS} build -o ${binpath} ${pkgpath} DEPENDS llvm-config ${CMAKE_BINARY_DIR}/bin/llvm-go${CMAKE_EXECUTABLE_SUFFIX} ${llvmlibs} ${ARG_DEPENDS} COMMENT "Building Go executable ${binary}" VERBATIM) if (ARG_ALL) add_custom_target(${binary} ALL DEPENDS ${binpath}) else() add_custom_target(${binary} DEPENDS ${binpath}) endif() endif() endfunction() # This function canonicalize the CMake variables passed by names # from CMake boolean to 0/1 suitable for passing into Python or C++, # in place. function(llvm_canonicalize_cmake_booleans) foreach(var ${ARGN}) if(${var}) set(${var} 1 PARENT_SCOPE) else() set(${var} 0 PARENT_SCOPE) endif() endforeach() endfunction(llvm_canonicalize_cmake_booleans) macro(set_llvm_build_mode) # Configuration-time: See Unit/lit.site.cfg.in if (CMAKE_CFG_INTDIR STREQUAL ".") set(LLVM_BUILD_MODE ".") else () set(LLVM_BUILD_MODE "%(build_mode)s") endif () endmacro() # This function provides an automatic way to 'configure'-like generate a file # based on a set of common and custom variables, specifically targeting the # variables needed for the 'lit.site.cfg' files. This function bundles the # common variables that any Lit instance is likely to need, and custom # variables can be passed in. function(configure_lit_site_cfg site_in site_out) cmake_parse_arguments(ARG "" "" "MAIN_CONFIG;OUTPUT_MAPPING" ${ARGN}) if ("${ARG_MAIN_CONFIG}" STREQUAL "") get_filename_component(INPUT_DIR ${site_in} DIRECTORY) set(ARG_MAIN_CONFIG "${INPUT_DIR}/lit.cfg") endif() if ("${ARG_OUTPUT_MAPPING}" STREQUAL "") set(ARG_OUTPUT_MAPPING "${site_out}") endif() foreach(c ${LLVM_TARGETS_TO_BUILD}) set(TARGETS_BUILT "${TARGETS_BUILT} ${c}") endforeach(c) set(TARGETS_TO_BUILD ${TARGETS_BUILT}) set(SHLIBEXT "${LTDL_SHLIB_EXT}") set_llvm_build_mode() # They below might not be the build tree but provided binary tree. set(LLVM_SOURCE_DIR ${LLVM_MAIN_SRC_DIR}) set(LLVM_BINARY_DIR ${LLVM_BINARY_DIR}) string(REPLACE "${CMAKE_CFG_INTDIR}" "${LLVM_BUILD_MODE}" LLVM_TOOLS_DIR "${LLVM_TOOLS_BINARY_DIR}") string(REPLACE ${CMAKE_CFG_INTDIR} ${LLVM_BUILD_MODE} LLVM_LIBS_DIR "${LLVM_LIBRARY_DIR}") # SHLIBDIR points the build tree. string(REPLACE "${CMAKE_CFG_INTDIR}" "${LLVM_BUILD_MODE}" SHLIBDIR "${LLVM_SHLIB_OUTPUT_INTDIR}") set(PYTHON_EXECUTABLE ${PYTHON_EXECUTABLE}) # FIXME: "ENABLE_SHARED" doesn't make sense, since it is used just for # plugins. We may rename it. if(LLVM_ENABLE_PLUGINS) set(ENABLE_SHARED "1") else() set(ENABLE_SHARED "0") endif() if(LLVM_ENABLE_ASSERTIONS AND NOT MSVC_IDE) set(ENABLE_ASSERTIONS "1") else() set(ENABLE_ASSERTIONS "0") endif() set(HOST_OS ${CMAKE_SYSTEM_NAME}) set(HOST_ARCH ${CMAKE_SYSTEM_PROCESSOR}) set(HOST_CC "${CMAKE_C_COMPILER} ${CMAKE_C_COMPILER_ARG1}") set(HOST_CXX "${CMAKE_CXX_COMPILER} ${CMAKE_CXX_COMPILER_ARG1}") set(HOST_LDFLAGS "${CMAKE_EXE_LINKER_FLAGS}") set(LIT_SITE_CFG_IN_HEADER "## Autogenerated from ${site_in}\n## Do not edit!") # Override config_target_triple (and the env) if(LLVM_TARGET_TRIPLE_ENV) # This is expanded into the heading. string(CONCAT LIT_SITE_CFG_IN_HEADER "${LIT_SITE_CFG_IN_HEADER}\n\n" "import os\n" "target_env = \"${LLVM_TARGET_TRIPLE_ENV}\"\n" "config.target_triple = config.environment[target_env] = os.environ.get(target_env, \"${TARGET_TRIPLE}\")\n" ) # This is expanded to; config.target_triple = ""+config.target_triple+"" set(TARGET_TRIPLE "\"+config.target_triple+\"") endif() configure_file(${site_in} ${site_out} @ONLY) if (EXISTS "${ARG_MAIN_CONFIG}") set(PYTHON_STATEMENT "map_config('${ARG_MAIN_CONFIG}', '${site_out}')") get_property(LLVM_LIT_CONFIG_MAP GLOBAL PROPERTY LLVM_LIT_CONFIG_MAP) set(LLVM_LIT_CONFIG_MAP "${LLVM_LIT_CONFIG_MAP}\n${PYTHON_STATEMENT}") set_property(GLOBAL PROPERTY LLVM_LIT_CONFIG_MAP ${LLVM_LIT_CONFIG_MAP}) endif() endfunction() function(dump_all_cmake_variables) get_cmake_property(_variableNames VARIABLES) foreach (_variableName ${_variableNames}) message(STATUS "${_variableName}=${${_variableName}}") endforeach() endfunction() function(get_llvm_lit_path base_dir file_name) cmake_parse_arguments(ARG "ALLOW_EXTERNAL" "" "" ${ARGN}) if (ARG_ALLOW_EXTERNAL) - set(LLVM_DEFAULT_EXTERNAL_LIT "${LLVM_EXTERNAL_LIT}") set (LLVM_EXTERNAL_LIT "" CACHE STRING "Command used to spawn lit") if ("${LLVM_EXTERNAL_LIT}" STREQUAL "") set(LLVM_EXTERNAL_LIT "${LLVM_DEFAULT_EXTERNAL_LIT}") endif() if (NOT "${LLVM_EXTERNAL_LIT}" STREQUAL "") if (EXISTS ${LLVM_EXTERNAL_LIT}) get_filename_component(LIT_FILE_NAME ${LLVM_EXTERNAL_LIT} NAME) get_filename_component(LIT_BASE_DIR ${LLVM_EXTERNAL_LIT} DIRECTORY) set(${file_name} ${LIT_FILE_NAME} PARENT_SCOPE) set(${base_dir} ${LIT_BASE_DIR} PARENT_SCOPE) return() else() message(WARN "LLVM_EXTERNAL_LIT set to ${LLVM_EXTERNAL_LIT}, but the path does not exist.") endif() endif() endif() set(lit_file_name "llvm-lit") if (CMAKE_HOST_WIN32 AND NOT CYGWIN) # llvm-lit needs suffix.py for multiprocess to find a main module. set(lit_file_name "${lit_file_name}.py") endif () set(${file_name} ${lit_file_name} PARENT_SCOPE) get_property(LLVM_LIT_BASE_DIR GLOBAL PROPERTY LLVM_LIT_BASE_DIR) if (NOT "${LLVM_LIT_BASE_DIR}" STREQUAL "") set(${base_dir} ${LLVM_LIT_BASE_DIR} PARENT_SCOPE) endif() # Allow individual projects to provide an override if (NOT "${LLVM_LIT_OUTPUT_DIR}" STREQUAL "") set(LLVM_LIT_BASE_DIR ${LLVM_LIT_OUTPUT_DIR}) elseif(NOT "${LLVM_RUNTIME_OUTPUT_INTDIR}" STREQUAL "") set(LLVM_LIT_BASE_DIR ${LLVM_RUNTIME_OUTPUT_INTDIR}) else() set(LLVM_LIT_BASE_DIR "") endif() # Cache this so we don't have to do it again and have subsequent calls # potentially disagree on the value. set_property(GLOBAL PROPERTY LLVM_LIT_BASE_DIR ${LLVM_LIT_BASE_DIR}) set(${base_dir} ${LLVM_LIT_BASE_DIR} PARENT_SCOPE) endfunction() # A raw function to create a lit target. This is used to implement the testuite # management functions. function(add_lit_target target comment) cmake_parse_arguments(ARG "" "" "PARAMS;DEPENDS;ARGS" ${ARGN}) set(LIT_ARGS "${ARG_ARGS} ${LLVM_LIT_ARGS}") separate_arguments(LIT_ARGS) if (NOT CMAKE_CFG_INTDIR STREQUAL ".") list(APPEND LIT_ARGS --param build_mode=${CMAKE_CFG_INTDIR}) endif () # Get the path to the lit to *run* tests with. This can be overriden by # the user by specifying -DLLVM_EXTERNAL_LIT= get_llvm_lit_path( lit_base_dir lit_file_name ALLOW_EXTERNAL ) set(LIT_COMMAND "${PYTHON_EXECUTABLE};${lit_base_dir}/${lit_file_name}") list(APPEND LIT_COMMAND ${LIT_ARGS}) foreach(param ${ARG_PARAMS}) list(APPEND LIT_COMMAND --param ${param}) endforeach() if (ARG_UNPARSED_ARGUMENTS) add_custom_target(${target} COMMAND ${LIT_COMMAND} ${ARG_UNPARSED_ARGUMENTS} COMMENT "${comment}" USES_TERMINAL ) else() add_custom_target(${target} COMMAND ${CMAKE_COMMAND} -E echo "${target} does nothing, no tools built.") message(STATUS "${target} does nothing.") endif() if (ARG_DEPENDS) add_dependencies(${target} ${ARG_DEPENDS}) endif() # Tests should be excluded from "Build Solution". set_target_properties(${target} PROPERTIES EXCLUDE_FROM_DEFAULT_BUILD ON) endfunction() # A function to add a set of lit test suites to be driven through 'check-*' targets. function(add_lit_testsuite target comment) cmake_parse_arguments(ARG "" "" "PARAMS;DEPENDS;ARGS" ${ARGN}) # EXCLUDE_FROM_ALL excludes the test ${target} out of check-all. if(NOT EXCLUDE_FROM_ALL) # Register the testsuites, params and depends for the global check rule. set_property(GLOBAL APPEND PROPERTY LLVM_LIT_TESTSUITES ${ARG_UNPARSED_ARGUMENTS}) set_property(GLOBAL APPEND PROPERTY LLVM_LIT_PARAMS ${ARG_PARAMS}) set_property(GLOBAL APPEND PROPERTY LLVM_LIT_DEPENDS ${ARG_DEPENDS}) set_property(GLOBAL APPEND PROPERTY LLVM_LIT_EXTRA_ARGS ${ARG_ARGS}) endif() # Produce a specific suffixed check rule. add_lit_target(${target} ${comment} ${ARG_UNPARSED_ARGUMENTS} PARAMS ${ARG_PARAMS} DEPENDS ${ARG_DEPENDS} ARGS ${ARG_ARGS} ) endfunction() function(add_lit_testsuites project directory) if (NOT LLVM_ENABLE_IDE) cmake_parse_arguments(ARG "" "" "PARAMS;DEPENDS;ARGS" ${ARGN}) # Search recursively for test directories by assuming anything not # in a directory called Inputs contains tests. file(GLOB_RECURSE to_process LIST_DIRECTORIES true ${directory}/*) foreach(lit_suite ${to_process}) if(NOT IS_DIRECTORY ${lit_suite}) continue() endif() string(FIND ${lit_suite} Inputs is_inputs) string(FIND ${lit_suite} Output is_output) if (NOT (is_inputs EQUAL -1 AND is_output EQUAL -1)) continue() endif() # Create a check- target for the directory. string(REPLACE ${directory} "" name_slash ${lit_suite}) if (name_slash) string(REPLACE "/" "-" name_slash ${name_slash}) string(REPLACE "\\" "-" name_dashes ${name_slash}) string(TOLOWER "${project}${name_dashes}" name_var) add_lit_target("check-${name_var}" "Running lit suite ${lit_suite}" ${lit_suite} PARAMS ${ARG_PARAMS} DEPENDS ${ARG_DEPENDS} ARGS ${ARG_ARGS} ) endif() endforeach() endif() endfunction() function(llvm_install_library_symlink name dest type) cmake_parse_arguments(ARG "ALWAYS_GENERATE" "COMPONENT" "" ${ARGN}) foreach(path ${CMAKE_MODULE_PATH}) if(EXISTS ${path}/LLVMInstallSymlink.cmake) set(INSTALL_SYMLINK ${path}/LLVMInstallSymlink.cmake) break() endif() endforeach() set(component ${ARG_COMPONENT}) if(NOT component) set(component ${name}) endif() set(full_name ${CMAKE_${type}_LIBRARY_PREFIX}${name}${CMAKE_${type}_LIBRARY_SUFFIX}) set(full_dest ${CMAKE_${type}_LIBRARY_PREFIX}${dest}${CMAKE_${type}_LIBRARY_SUFFIX}) set(output_dir lib${LLVM_LIBDIR_SUFFIX}) if(WIN32 AND "${type}" STREQUAL "SHARED") set(output_dir bin) endif() install(SCRIPT ${INSTALL_SYMLINK} CODE "install_symlink(${full_name} ${full_dest} ${output_dir})" COMPONENT ${component}) if (NOT LLVM_ENABLE_IDE AND NOT ARG_ALWAYS_GENERATE) add_llvm_install_targets(install-${name} DEPENDS ${name} ${dest} install-${dest} COMPONENT ${name}) endif() endfunction() function(llvm_install_symlink name dest) cmake_parse_arguments(ARG "ALWAYS_GENERATE" "COMPONENT" "" ${ARGN}) foreach(path ${CMAKE_MODULE_PATH}) if(EXISTS ${path}/LLVMInstallSymlink.cmake) set(INSTALL_SYMLINK ${path}/LLVMInstallSymlink.cmake) break() endif() endforeach() if(ARG_COMPONENT) set(component ${ARG_COMPONENT}) else() if(ARG_ALWAYS_GENERATE) set(component ${dest}) else() set(component ${name}) endif() endif() set(full_name ${name}${CMAKE_EXECUTABLE_SUFFIX}) set(full_dest ${dest}${CMAKE_EXECUTABLE_SUFFIX}) install(SCRIPT ${INSTALL_SYMLINK} CODE "install_symlink(${full_name} ${full_dest} ${LLVM_TOOLS_INSTALL_DIR})" COMPONENT ${component}) if (NOT LLVM_ENABLE_IDE AND NOT ARG_ALWAYS_GENERATE) add_llvm_install_targets(install-${name} DEPENDS ${name} ${dest} install-${dest} COMPONENT ${name}) endif() endfunction() function(add_llvm_tool_symlink link_name target) cmake_parse_arguments(ARG "ALWAYS_GENERATE" "OUTPUT_DIR" "" ${ARGN}) set(dest_binary "$") # This got a bit gross... For multi-configuration generators the target # properties return the resolved value of the string, not the build system # expression. To reconstruct the platform-agnostic path we have to do some # magic. First we grab one of the types, and a type-specific path. Then from # the type-specific path we find the last occurrence of the type in the path, # and replace it with CMAKE_CFG_INTDIR. This allows the build step to be type # agnostic again. if(NOT ARG_OUTPUT_DIR) # If you're not overriding the OUTPUT_DIR, we can make the link relative in # the same directory. if(CMAKE_HOST_UNIX) set(dest_binary "$") endif() if(CMAKE_CONFIGURATION_TYPES) list(GET CMAKE_CONFIGURATION_TYPES 0 first_type) string(TOUPPER ${first_type} first_type_upper) set(first_type_suffix _${first_type_upper}) endif() get_target_property(target_type ${target} TYPE) if(${target_type} STREQUAL "STATIC_LIBRARY") get_target_property(ARG_OUTPUT_DIR ${target} ARCHIVE_OUTPUT_DIRECTORY${first_type_suffix}) elseif(UNIX AND ${target_type} STREQUAL "SHARED_LIBRARY") get_target_property(ARG_OUTPUT_DIR ${target} LIBRARY_OUTPUT_DIRECTORY${first_type_suffix}) else() get_target_property(ARG_OUTPUT_DIR ${target} RUNTIME_OUTPUT_DIRECTORY${first_type_suffix}) endif() if(CMAKE_CONFIGURATION_TYPES) string(FIND "${ARG_OUTPUT_DIR}" "/${first_type}/" type_start REVERSE) string(SUBSTRING "${ARG_OUTPUT_DIR}" 0 ${type_start} path_prefix) string(SUBSTRING "${ARG_OUTPUT_DIR}" ${type_start} -1 path_suffix) string(REPLACE "/${first_type}/" "/${CMAKE_CFG_INTDIR}/" path_suffix ${path_suffix}) set(ARG_OUTPUT_DIR ${path_prefix}${path_suffix}) endif() endif() if(CMAKE_HOST_UNIX) set(LLVM_LINK_OR_COPY create_symlink) else() set(LLVM_LINK_OR_COPY copy) endif() set(output_path "${ARG_OUTPUT_DIR}/${link_name}${CMAKE_EXECUTABLE_SUFFIX}") set(target_name ${link_name}) if(TARGET ${link_name}) set(target_name ${link_name}-link) endif() if(ARG_ALWAYS_GENERATE) set_property(DIRECTORY APPEND PROPERTY ADDITIONAL_MAKE_CLEAN_FILES ${dest_binary}) add_custom_command(TARGET ${target} POST_BUILD COMMAND ${CMAKE_COMMAND} -E ${LLVM_LINK_OR_COPY} "${dest_binary}" "${output_path}") else() add_custom_command(OUTPUT ${output_path} COMMAND ${CMAKE_COMMAND} -E ${LLVM_LINK_OR_COPY} "${dest_binary}" "${output_path}" DEPENDS ${target}) add_custom_target(${target_name} ALL DEPENDS ${target} ${output_path}) set_target_properties(${target_name} PROPERTIES FOLDER Tools) # Make sure both the link and target are toolchain tools if (${link_name} IN_LIST LLVM_TOOLCHAIN_TOOLS AND ${target} IN_LIST LLVM_TOOLCHAIN_TOOLS) set(TOOL_IS_TOOLCHAIN ON) endif() if ((TOOL_IS_TOOLCHAIN OR NOT LLVM_INSTALL_TOOLCHAIN_ONLY) AND LLVM_BUILD_TOOLS) llvm_install_symlink(${link_name} ${target}) endif() endif() endfunction() function(llvm_externalize_debuginfo name) if(NOT LLVM_EXTERNALIZE_DEBUGINFO) return() endif() if(NOT LLVM_EXTERNALIZE_DEBUGINFO_SKIP_STRIP) if(APPLE) if(NOT CMAKE_STRIP) set(CMAKE_STRIP xcrun strip) endif() set(strip_command COMMAND ${CMAKE_STRIP} -Sxl $) else() set(strip_command COMMAND ${CMAKE_STRIP} -g -x $) endif() endif() if(LLVM_EXTERNALIZE_DEBUGINFO_OUTPUT_DIR) if(APPLE) set(output_name "$.dSYM") set(output_path "-o=${LLVM_EXTERNALIZE_DEBUGINFO_OUTPUT_DIR}/${output_name}") endif() endif() if(APPLE) if(CMAKE_CXX_FLAGS MATCHES "-flto" OR CMAKE_CXX_FLAGS_${uppercase_CMAKE_BUILD_TYPE} MATCHES "-flto") set(lto_object ${CMAKE_CURRENT_BINARY_DIR}/${CMAKE_CFG_INTDIR}/${name}-lto.o) set_property(TARGET ${name} APPEND_STRING PROPERTY LINK_FLAGS " -Wl,-object_path_lto,${lto_object}") endif() if(NOT CMAKE_DSYMUTIL) set(CMAKE_DSYMUTIL xcrun dsymutil) endif() add_custom_command(TARGET ${name} POST_BUILD COMMAND ${CMAKE_DSYMUTIL} ${output_path} $ ${strip_command} ) else() add_custom_command(TARGET ${name} POST_BUILD COMMAND ${CMAKE_OBJCOPY} --only-keep-debug $ $.debug ${strip_command} -R .gnu_debuglink COMMAND ${CMAKE_OBJCOPY} --add-gnu-debuglink=$.debug $ ) endif() endfunction() # Usage: llvm_codesign(name [ENTITLEMENTS file]) function(llvm_codesign name) cmake_parse_arguments(ARG "" "ENTITLEMENTS" "" ${ARGN}) if(NOT LLVM_CODESIGNING_IDENTITY) return() endif() if(CMAKE_GENERATOR STREQUAL "Xcode") set_target_properties(${name} PROPERTIES XCODE_ATTRIBUTE_CODE_SIGN_IDENTITY ${LLVM_CODESIGNING_IDENTITY} ) if(DEFINED ARG_ENTITLEMENTS) set_target_properties(${name} PROPERTIES XCODE_ATTRIBUTE_CODE_SIGN_ENTITLEMENTS ${ARG_ENTITLEMENTS} ) endif() elseif(APPLE) if(NOT CMAKE_CODESIGN) set(CMAKE_CODESIGN xcrun codesign) endif() if(NOT CMAKE_CODESIGN_ALLOCATE) execute_process( COMMAND xcrun -f codesign_allocate OUTPUT_STRIP_TRAILING_WHITESPACE OUTPUT_VARIABLE CMAKE_CODESIGN_ALLOCATE ) endif() if(DEFINED ARG_ENTITLEMENTS) set(pass_entitlements --entitlements ${ARG_ENTITLEMENTS}) endif() add_custom_command( TARGET ${name} POST_BUILD COMMAND ${CMAKE_COMMAND} -E env CODESIGN_ALLOCATE=${CMAKE_CODESIGN_ALLOCATE} ${CMAKE_CODESIGN} -s ${LLVM_CODESIGNING_IDENTITY} ${pass_entitlements} $ ) endif() endfunction() function(llvm_setup_rpath name) if(CMAKE_INSTALL_RPATH) return() endif() if(LLVM_INSTALL_PREFIX AND NOT (LLVM_INSTALL_PREFIX STREQUAL CMAKE_INSTALL_PREFIX)) set(extra_libdir ${LLVM_LIBRARY_DIR}) elseif(LLVM_BUILD_LIBRARY_DIR) set(extra_libdir ${LLVM_LIBRARY_DIR}) endif() if (APPLE) set(_install_name_dir INSTALL_NAME_DIR "@rpath") set(_install_rpath "@loader_path/../lib" ${extra_libdir}) elseif(UNIX) set(_install_rpath "\$ORIGIN/../lib${LLVM_LIBDIR_SUFFIX}" ${extra_libdir}) if(${CMAKE_SYSTEM_NAME} MATCHES "(FreeBSD|DragonFly)") set_property(TARGET ${name} APPEND_STRING PROPERTY LINK_FLAGS " -Wl,-z,origin ") endif() if(LLVM_LINKER_IS_GNULD) # $ORIGIN is not interpreted at link time by ld.bfd set_property(TARGET ${name} APPEND_STRING PROPERTY LINK_FLAGS " -Wl,-rpath-link,${LLVM_LIBRARY_OUTPUT_INTDIR} ") endif() else() return() endif() set_target_properties(${name} PROPERTIES BUILD_WITH_INSTALL_RPATH On INSTALL_RPATH "${_install_rpath}" ${_install_name_dir}) endfunction() function(setup_dependency_debugging name) if(NOT LLVM_DEPENDENCY_DEBUGGING) return() endif() if("intrinsics_gen" IN_LIST ARGN) return() endif() set(deny_attributes_inc "(deny file* (literal \"${LLVM_BINARY_DIR}/include/llvm/IR/Attributes.inc\"))") set(deny_intrinsics_inc "(deny file* (literal \"${LLVM_BINARY_DIR}/include/llvm/IR/Intrinsics.inc\"))") set(sandbox_command "sandbox-exec -p '(version 1) (allow default) ${deny_attributes_inc} ${deny_intrinsics_inc}'") set_target_properties(${name} PROPERTIES RULE_LAUNCH_COMPILE ${sandbox_command}) endfunction() # Figure out if we can track VC revisions. function(find_first_existing_file out_var) foreach(file ${ARGN}) if(EXISTS "${file}") set(${out_var} "${file}" PARENT_SCOPE) return() endif() endforeach() endfunction() macro(find_first_existing_vc_file out_var path) find_program(git_executable NAMES git git.exe git.cmd) # Run from a subdirectory to force git to print an absolute path. execute_process(COMMAND ${git_executable} rev-parse --git-dir WORKING_DIRECTORY ${path}/cmake RESULT_VARIABLE git_result OUTPUT_VARIABLE git_dir ERROR_QUIET) if(git_result EQUAL 0) string(STRIP "${git_dir}" git_dir) set(${out_var} "${git_dir}/logs/HEAD") # some branchless cases (e.g. 'repo') may not yet have .git/logs/HEAD if (NOT EXISTS "${git_dir}/logs/HEAD") file(WRITE "${git_dir}/logs/HEAD" "") endif() else() find_first_existing_file(${out_var} "${path}/.svn/wc.db" # SVN 1.7 "${path}/.svn/entries" # SVN 1.6 ) endif() endmacro() Index: vendor/llvm/dist-release_80/docs/ReleaseNotes.rst =================================================================== --- vendor/llvm/dist-release_80/docs/ReleaseNotes.rst (revision 343793) +++ vendor/llvm/dist-release_80/docs/ReleaseNotes.rst (revision 343794) @@ -1,140 +1,182 @@ ======================== LLVM 8.0.0 Release Notes ======================== .. contents:: :local: .. warning:: These are in-progress notes for the upcoming LLVM 8 release. Release notes for previous releases can be found on `the Download Page `_. Introduction ============ This document contains the release notes for the LLVM Compiler Infrastructure, release 8.0.0. Here we describe the status of LLVM, including major improvements from the previous release, improvements in various subprojects of LLVM, and some of the current users of the code. All LLVM releases may be downloaded from the `LLVM releases web site `_. For more information about LLVM, including information about the latest release, please check out the `main LLVM web site `_. If you have questions or comments, the `LLVM Developer's Mailing List `_ is a good place to send them. Note that if you are reading this file from a Subversion checkout or the main LLVM web page, this document applies to the *next* release, not the current one. To see the release notes for a specific release, please see the `releases page `_. Non-comprehensive list of changes in this release ================================================= .. NOTE For small 1-3 sentence descriptions, just add an entry at the end of this list. If your description won't fit comfortably in one bullet point (e.g. maybe you would like to give an example of the functionality, or simply have a lot to talk about), see the `NOTE` below for adding a new subsection. * The **llvm-cov** tool can now export lcov trace files using the `-format=lcov` option of the `export` command. * The add_llvm_loadable_module CMake macro has been removed. The add_llvm_library macro with the MODULE argument now provides the same functionality. See `Writing an LLVM Pass `_. +* For MinGW, references to data variables that might need to be imported + from a dll are accessed via a stub, to allow the linker to convert it to + a dllimport if needed. + +* Added support for labels as offsets in ``.reloc`` directive. + .. NOTE If you would like to document a larger change, then you can add a subsection about it right here. You can copy the following boilerplate and un-indent it (the indentation causes it to be inside this comment). Special New Feature ------------------- Makes programs 10x faster by doing Special New Thing. Changes to the LLVM IR ---------------------- +Changes to the AArch64 Target +----------------------------- + +* Added support for the ``.arch_extension`` assembler directive, just like + on ARM. + + Changes to the ARM Backend -------------------------- During this release ... +Changes to the Hexagon Target +-------------------------- + +* Added support for Hexagon/HVX V66 ISA. + Changes to the MIPS Target -------------------------- - During this release ... +* Improved support of GlobalISel instruction selection framework. +* Implemented emission of ``R_MIPS_JALR`` and ``R_MICROMIPS_JALR`` + relocations. These relocations provide hints to a linker for optimization + of jumps to protected symbols. +* ORC JIT has been supported for MIPS and MIPS64 architectures. + +* Assembler now suggests alternative MIPS instruction mnemonics when + an invalid one is specified. + +* Improved support for MIPS N32 ABI. + +* Added new instructions (``pll.ps``, ``plu.ps``, ``cvt.s.pu``, + ``cvt.s.pl``, ``cvt.ps``, ``sigrie``). + +* Numerous bug fixes and code cleanups. + Changes to the PowerPC Target ----------------------------- During this release ... Changes to the X86 Target ------------------------- * Machine model for AMD bdver2 (Piledriver) CPU was added. It is used to support instruction scheduling and other instruction cost heuristics. Changes to the AMDGPU Target ----------------------------- During this release ... Changes to the AVR Target ----------------------------- During this release ... Changes to the WebAssembly Target --------------------------------- The WebAssembly target is no longer "experimental"! It's now built by default, rather than needing to be enabled with LLVM_EXPERIMENTAL_TARGETS_TO_BUILD. The object file format and core C ABI are now considered stable. That said, the object file format has an ABI versioning capability, and one anticipated use for it will be to add support for returning small structs as multiple return values, once the underlying WebAssembly platform itself supports it. Additionally, multithreading support is not yet included in the stable ABI. Changes to the OCaml bindings ----------------------------- Changes to the C API -------------------- Changes to the DAG infrastructure --------------------------------- External Open Source Projects Using LLVM 8 ========================================== -* A project... +Zig Programming Language +------------------------ + +`Zig `_ is a system programming language intended to be +an alternative to C. It provides high level features such as generics, compile +time function execution, and partial evaluation, while exposing low level LLVM +IR features such as aliases and intrinsics. Zig uses Clang to provide automatic +import of .h symbols, including inline functions and simple macros. Zig uses +LLD combined with lazily building compiler-rt to provide out-of-the-box +cross-compiling for all supported targets. Additional Information ====================== A wide variety of additional information is available on the `LLVM web page `_, in particular in the `documentation `_ section. The web page also contains versions of the API documentation which is up-to-date with the Subversion version of the source code. You can access versions of these documents specific to this release by going into the ``llvm/docs/`` directory in the LLVM tree. If you have any questions or comments about LLVM, please feel free to contact us via the `mailing lists `_. Index: vendor/llvm/dist-release_80/include/llvm/Support/JSON.h =================================================================== --- vendor/llvm/dist-release_80/include/llvm/Support/JSON.h (revision 343793) +++ vendor/llvm/dist-release_80/include/llvm/Support/JSON.h (revision 343794) @@ -1,711 +1,712 @@ //===--- JSON.h - JSON values, parsing and serialization -------*- C++ -*-===// // // The LLVM Compiler Infrastructure // // This file is distributed under the University of Illinois Open Source // License. See LICENSE.TXT for details. // //===---------------------------------------------------------------------===// /// /// \file /// This file supports working with JSON data. /// /// It comprises: /// /// - classes which hold dynamically-typed parsed JSON structures /// These are value types that can be composed, inspected, and modified. /// See json::Value, and the related types json::Object and json::Array. /// /// - functions to parse JSON text into Values, and to serialize Values to text. /// See parse(), operator<<, and format_provider. /// /// - a convention and helpers for mapping between json::Value and user-defined /// types. See fromJSON(), ObjectMapper, and the class comment on Value. /// /// Typically, JSON data would be read from an external source, parsed into /// a Value, and then converted into some native data structure before doing /// real work on it. (And vice versa when writing). /// /// Other serialization mechanisms you may consider: /// /// - YAML is also text-based, and more human-readable than JSON. It's a more /// complex format and data model, and YAML parsers aren't ubiquitous. /// YAMLParser.h is a streaming parser suitable for parsing large documents /// (including JSON, as YAML is a superset). It can be awkward to use /// directly. YAML I/O (YAMLTraits.h) provides data mapping that is more /// declarative than the toJSON/fromJSON conventions here. /// /// - LLVM bitstream is a space- and CPU- efficient binary format. Typically it /// encodes LLVM IR ("bitcode"), but it can be a container for other data. /// Low-level reader/writer libraries are in Bitcode/Bitstream*.h /// //===---------------------------------------------------------------------===// #ifndef LLVM_SUPPORT_JSON_H #define LLVM_SUPPORT_JSON_H #include "llvm/ADT/DenseMap.h" #include "llvm/ADT/SmallVector.h" #include "llvm/ADT/StringRef.h" #include "llvm/Support/Error.h" #include "llvm/Support/FormatVariadic.h" #include "llvm/Support/raw_ostream.h" #include namespace llvm { namespace json { // === String encodings === // // JSON strings are character sequences (not byte sequences like std::string). // We need to know the encoding, and for simplicity only support UTF-8. // // - When parsing, invalid UTF-8 is a syntax error like any other // // - When creating Values from strings, callers must ensure they are UTF-8. // with asserts on, invalid UTF-8 will crash the program // with asserts off, we'll substitute the replacement character (U+FFFD) // Callers can use json::isUTF8() and json::fixUTF8() for validation. // // - When retrieving strings from Values (e.g. asString()), the result will // always be valid UTF-8. /// Returns true if \p S is valid UTF-8, which is required for use as JSON. /// If it returns false, \p Offset is set to a byte offset near the first error. bool isUTF8(llvm::StringRef S, size_t *ErrOffset = nullptr); /// Replaces invalid UTF-8 sequences in \p S with the replacement character /// (U+FFFD). The returned string is valid UTF-8. /// This is much slower than isUTF8, so test that first. std::string fixUTF8(llvm::StringRef S); class Array; class ObjectKey; class Value; template Value toJSON(const llvm::Optional &Opt); /// An Object is a JSON object, which maps strings to heterogenous JSON values. /// It simulates DenseMap. ObjectKey is a maybe-owned string. class Object { using Storage = DenseMap>; Storage M; public: using key_type = ObjectKey; using mapped_type = Value; using value_type = Storage::value_type; using iterator = Storage::iterator; using const_iterator = Storage::const_iterator; explicit Object() = default; // KV is a trivial key-value struct for list-initialization. // (using std::pair forces extra copies). struct KV; explicit Object(std::initializer_list Properties); iterator begin() { return M.begin(); } const_iterator begin() const { return M.begin(); } iterator end() { return M.end(); } const_iterator end() const { return M.end(); } bool empty() const { return M.empty(); } size_t size() const { return M.size(); } void clear() { M.clear(); } std::pair insert(KV E); template std::pair try_emplace(const ObjectKey &K, Ts &&... Args) { return M.try_emplace(K, std::forward(Args)...); } template std::pair try_emplace(ObjectKey &&K, Ts &&... Args) { return M.try_emplace(std::move(K), std::forward(Args)...); } iterator find(StringRef K) { return M.find_as(K); } const_iterator find(StringRef K) const { return M.find_as(K); } // operator[] acts as if Value was default-constructible as null. Value &operator[](const ObjectKey &K); Value &operator[](ObjectKey &&K); // Look up a property, returning nullptr if it doesn't exist. Value *get(StringRef K); const Value *get(StringRef K) const; // Typed accessors return None/nullptr if // - the property doesn't exist // - or it has the wrong type llvm::Optional getNull(StringRef K) const; llvm::Optional getBoolean(StringRef K) const; llvm::Optional getNumber(StringRef K) const; llvm::Optional getInteger(StringRef K) const; llvm::Optional getString(StringRef K) const; const json::Object *getObject(StringRef K) const; json::Object *getObject(StringRef K); const json::Array *getArray(StringRef K) const; json::Array *getArray(StringRef K); }; bool operator==(const Object &LHS, const Object &RHS); inline bool operator!=(const Object &LHS, const Object &RHS) { return !(LHS == RHS); } /// An Array is a JSON array, which contains heterogeneous JSON values. /// It simulates std::vector. class Array { std::vector V; public: using value_type = Value; using iterator = std::vector::iterator; using const_iterator = std::vector::const_iterator; explicit Array() = default; explicit Array(std::initializer_list Elements); template explicit Array(const Collection &C) { for (const auto &V : C) emplace_back(V); } Value &operator[](size_t I) { return V[I]; } const Value &operator[](size_t I) const { return V[I]; } Value &front() { return V.front(); } const Value &front() const { return V.front(); } Value &back() { return V.back(); } const Value &back() const { return V.back(); } Value *data() { return V.data(); } const Value *data() const { return V.data(); } iterator begin() { return V.begin(); } const_iterator begin() const { return V.begin(); } iterator end() { return V.end(); } const_iterator end() const { return V.end(); } bool empty() const { return V.empty(); } size_t size() const { return V.size(); } void clear() { V.clear(); } void push_back(const Value &E) { V.push_back(E); } void push_back(Value &&E) { V.push_back(std::move(E)); } template void emplace_back(Args &&... A) { V.emplace_back(std::forward(A)...); } void pop_back() { V.pop_back(); } // FIXME: insert() takes const_iterator since C++11, old libstdc++ disagrees. iterator insert(iterator P, const Value &E) { return V.insert(P, E); } iterator insert(iterator P, Value &&E) { return V.insert(P, std::move(E)); } template iterator insert(iterator P, It A, It Z) { return V.insert(P, A, Z); } template iterator emplace(const_iterator P, Args &&... A) { return V.emplace(P, std::forward(A)...); } friend bool operator==(const Array &L, const Array &R) { return L.V == R.V; } }; inline bool operator!=(const Array &L, const Array &R) { return !(L == R); } /// A Value is an JSON value of unknown type. /// They can be copied, but should generally be moved. /// /// === Composing values === /// /// You can implicitly construct Values from: /// - strings: std::string, SmallString, formatv, StringRef, char* /// (char*, and StringRef are references, not copies!) /// - numbers /// - booleans /// - null: nullptr /// - arrays: {"foo", 42.0, false} /// - serializable things: types with toJSON(const T&)->Value, found by ADL /// /// They can also be constructed from object/array helpers: /// - json::Object is a type like map /// - json::Array is a type like vector /// These can be list-initialized, or used to build up collections in a loop. /// json::ary(Collection) converts all items in a collection to Values. /// /// === Inspecting values === /// /// Each Value is one of the JSON kinds: /// null (nullptr_t) /// boolean (bool) /// number (double or int64) /// string (StringRef) /// array (json::Array) /// object (json::Object) /// /// The kind can be queried directly, or implicitly via the typed accessors: /// if (Optional S = E.getAsString() /// assert(E.kind() == Value::String); /// /// Array and Object also have typed indexing accessors for easy traversal: /// Expected E = parse(R"( {"options": {"font": "sans-serif"}} )"); /// if (Object* O = E->getAsObject()) /// if (Object* Opts = O->getObject("options")) /// if (Optional Font = Opts->getString("font")) /// assert(Opts->at("font").kind() == Value::String); /// /// === Converting JSON values to C++ types === /// /// The convention is to have a deserializer function findable via ADL: /// fromJSON(const json::Value&, T&)->bool /// Deserializers are provided for: /// - bool /// - int and int64_t /// - double /// - std::string /// - vector, where T is deserializable /// - map, where T is deserializable /// - Optional, where T is deserializable /// ObjectMapper can help writing fromJSON() functions for object types. /// /// For conversion in the other direction, the serializer function is: /// toJSON(const T&) -> json::Value /// If this exists, then it also allows constructing Value from T, and can /// be used to serialize vector, map, and Optional. /// /// === Serialization === /// /// Values can be serialized to JSON: /// 1) raw_ostream << Value // Basic formatting. /// 2) raw_ostream << formatv("{0}", Value) // Basic formatting. /// 3) raw_ostream << formatv("{0:2}", Value) // Pretty-print with indent 2. /// /// And parsed: /// Expected E = json::parse("[1, 2, null]"); /// assert(E && E->kind() == Value::Array); class Value { public: enum Kind { Null, Boolean, /// Number values can store both int64s and doubles at full precision, /// depending on what they were constructed/parsed from. Number, String, Array, Object, }; // It would be nice to have Value() be null. But that would make {} null too. Value(const Value &M) { copyFrom(M); } Value(Value &&M) { moveFrom(std::move(M)); } Value(std::initializer_list Elements); Value(json::Array &&Elements) : Type(T_Array) { create(std::move(Elements)); } template Value(const std::vector &C) : Value(json::Array(C)) {} Value(json::Object &&Properties) : Type(T_Object) { create(std::move(Properties)); } template Value(const std::map &C) : Value(json::Object(C)) {} // Strings: types with value semantics. Must be valid UTF-8. Value(std::string V) : Type(T_String) { if (LLVM_UNLIKELY(!isUTF8(V))) { assert(false && "Invalid UTF-8 in value used as JSON"); V = fixUTF8(std::move(V)); } create(std::move(V)); } Value(const llvm::SmallVectorImpl &V) : Value(std::string(V.begin(), V.end())){}; Value(const llvm::formatv_object_base &V) : Value(V.str()){}; // Strings: types with reference semantics. Must be valid UTF-8. Value(StringRef V) : Type(T_StringRef) { create(V); if (LLVM_UNLIKELY(!isUTF8(V))) { assert(false && "Invalid UTF-8 in value used as JSON"); *this = Value(fixUTF8(V)); } } Value(const char *V) : Value(StringRef(V)) {} Value(std::nullptr_t) : Type(T_Null) {} // Boolean (disallow implicit conversions). // (The last template parameter is a dummy to keep templates distinct.) template < typename T, typename = typename std::enable_if::value>::type, bool = false> Value(T B) : Type(T_Boolean) { create(B); } // Integers (except boolean). Must be non-narrowing convertible to int64_t. template < typename T, typename = typename std::enable_if::value>::type, typename = typename std::enable_if::value>::type> Value(T I) : Type(T_Integer) { create(int64_t{I}); } // Floating point. Must be non-narrowing convertible to double. template ::value>::type, double * = nullptr> Value(T D) : Type(T_Double) { create(double{D}); } // Serializable types: with a toJSON(const T&)->Value function, found by ADL. template ::value>, Value * = nullptr> Value(const T &V) : Value(toJSON(V)) {} Value &operator=(const Value &M) { destroy(); copyFrom(M); return *this; } Value &operator=(Value &&M) { destroy(); moveFrom(std::move(M)); return *this; } ~Value() { destroy(); } Kind kind() const { switch (Type) { case T_Null: return Null; case T_Boolean: return Boolean; case T_Double: case T_Integer: return Number; case T_String: case T_StringRef: return String; case T_Object: return Object; case T_Array: return Array; } llvm_unreachable("Unknown kind"); } // Typed accessors return None/nullptr if the Value is not of this type. llvm::Optional getAsNull() const { if (LLVM_LIKELY(Type == T_Null)) return nullptr; return llvm::None; } llvm::Optional getAsBoolean() const { if (LLVM_LIKELY(Type == T_Boolean)) return as(); return llvm::None; } llvm::Optional getAsNumber() const { if (LLVM_LIKELY(Type == T_Double)) return as(); if (LLVM_LIKELY(Type == T_Integer)) return as(); return llvm::None; } // Succeeds if the Value is a Number, and exactly representable as int64_t. llvm::Optional getAsInteger() const { if (LLVM_LIKELY(Type == T_Integer)) return as(); if (LLVM_LIKELY(Type == T_Double)) { double D = as(); if (LLVM_LIKELY(std::modf(D, &D) == 0.0 && D >= double(std::numeric_limits::min()) && D <= double(std::numeric_limits::max()))) return D; } return llvm::None; } llvm::Optional getAsString() const { if (Type == T_String) return llvm::StringRef(as()); if (LLVM_LIKELY(Type == T_StringRef)) return as(); return llvm::None; } const json::Object *getAsObject() const { return LLVM_LIKELY(Type == T_Object) ? &as() : nullptr; } json::Object *getAsObject() { return LLVM_LIKELY(Type == T_Object) ? &as() : nullptr; } const json::Array *getAsArray() const { return LLVM_LIKELY(Type == T_Array) ? &as() : nullptr; } json::Array *getAsArray() { return LLVM_LIKELY(Type == T_Array) ? &as() : nullptr; } /// Serializes this Value to JSON, writing it to the provided stream. /// The formatting is compact (no extra whitespace) and deterministic. /// For pretty-printing, use the formatv() format_provider below. friend llvm::raw_ostream &operator<<(llvm::raw_ostream &, const Value &); private: void destroy(); void copyFrom(const Value &M); // We allow moving from *const* Values, by marking all members as mutable! // This hack is needed to support initializer-list syntax efficiently. // (std::initializer_list is a container of const T). void moveFrom(const Value &&M); friend class Array; friend class Object; template void create(U &&... V) { new (reinterpret_cast(Union.buffer)) T(std::forward(V)...); } template T &as() const { // Using this two-step static_cast via void * instead of reinterpret_cast // silences a -Wstrict-aliasing false positive from GCC6 and earlier. void *Storage = static_cast(Union.buffer); return *static_cast(Storage); } template void print(llvm::raw_ostream &, const Indenter &) const; friend struct llvm::format_provider; enum ValueType : char { T_Null, T_Boolean, T_Double, T_Integer, T_StringRef, T_String, T_Object, T_Array, }; // All members mutable, see moveFrom(). mutable ValueType Type; mutable llvm::AlignedCharArrayUnion Union; + friend bool operator==(const Value &, const Value &); }; bool operator==(const Value &, const Value &); inline bool operator!=(const Value &L, const Value &R) { return !(L == R); } llvm::raw_ostream &operator<<(llvm::raw_ostream &, const Value &); /// ObjectKey is a used to capture keys in Object. Like Value but: /// - only strings are allowed /// - it's optimized for the string literal case (Owned == nullptr) /// Like Value, strings must be UTF-8. See isUTF8 documentation for details. class ObjectKey { public: ObjectKey(const char *S) : ObjectKey(StringRef(S)) {} ObjectKey(std::string S) : Owned(new std::string(std::move(S))) { if (LLVM_UNLIKELY(!isUTF8(*Owned))) { assert(false && "Invalid UTF-8 in value used as JSON"); *Owned = fixUTF8(std::move(*Owned)); } Data = *Owned; } ObjectKey(llvm::StringRef S) : Data(S) { if (LLVM_UNLIKELY(!isUTF8(Data))) { assert(false && "Invalid UTF-8 in value used as JSON"); *this = ObjectKey(fixUTF8(S)); } } ObjectKey(const llvm::SmallVectorImpl &V) : ObjectKey(std::string(V.begin(), V.end())) {} ObjectKey(const llvm::formatv_object_base &V) : ObjectKey(V.str()) {} ObjectKey(const ObjectKey &C) { *this = C; } ObjectKey(ObjectKey &&C) : ObjectKey(static_cast(C)) {} ObjectKey &operator=(const ObjectKey &C) { if (C.Owned) { Owned.reset(new std::string(*C.Owned)); Data = *Owned; } else { Data = C.Data; } return *this; } ObjectKey &operator=(ObjectKey &&) = default; operator llvm::StringRef() const { return Data; } std::string str() const { return Data.str(); } private: // FIXME: this is unneccesarily large (3 pointers). Pointer + length + owned // could be 2 pointers at most. std::unique_ptr Owned; llvm::StringRef Data; }; inline bool operator==(const ObjectKey &L, const ObjectKey &R) { return llvm::StringRef(L) == llvm::StringRef(R); } inline bool operator!=(const ObjectKey &L, const ObjectKey &R) { return !(L == R); } inline bool operator<(const ObjectKey &L, const ObjectKey &R) { return StringRef(L) < StringRef(R); } struct Object::KV { ObjectKey K; Value V; }; inline Object::Object(std::initializer_list Properties) { for (const auto &P : Properties) { auto R = try_emplace(P.K, nullptr); if (R.second) R.first->getSecond().moveFrom(std::move(P.V)); } } inline std::pair Object::insert(KV E) { return try_emplace(std::move(E.K), std::move(E.V)); } // Standard deserializers are provided for primitive types. // See comments on Value. inline bool fromJSON(const Value &E, std::string &Out) { if (auto S = E.getAsString()) { Out = *S; return true; } return false; } inline bool fromJSON(const Value &E, int &Out) { if (auto S = E.getAsInteger()) { Out = *S; return true; } return false; } inline bool fromJSON(const Value &E, int64_t &Out) { if (auto S = E.getAsInteger()) { Out = *S; return true; } return false; } inline bool fromJSON(const Value &E, double &Out) { if (auto S = E.getAsNumber()) { Out = *S; return true; } return false; } inline bool fromJSON(const Value &E, bool &Out) { if (auto S = E.getAsBoolean()) { Out = *S; return true; } return false; } template bool fromJSON(const Value &E, llvm::Optional &Out) { if (E.getAsNull()) { Out = llvm::None; return true; } T Result; if (!fromJSON(E, Result)) return false; Out = std::move(Result); return true; } template bool fromJSON(const Value &E, std::vector &Out) { if (auto *A = E.getAsArray()) { Out.clear(); Out.resize(A->size()); for (size_t I = 0; I < A->size(); ++I) if (!fromJSON((*A)[I], Out[I])) return false; return true; } return false; } template bool fromJSON(const Value &E, std::map &Out) { if (auto *O = E.getAsObject()) { Out.clear(); for (const auto &KV : *O) if (!fromJSON(KV.second, Out[llvm::StringRef(KV.first)])) return false; return true; } return false; } // Allow serialization of Optional for supported T. template Value toJSON(const llvm::Optional &Opt) { return Opt ? Value(*Opt) : Value(nullptr); } /// Helper for mapping JSON objects onto protocol structs. /// /// Example: /// \code /// bool fromJSON(const Value &E, MyStruct &R) { /// ObjectMapper O(E); /// if (!O || !O.map("mandatory_field", R.MandatoryField)) /// return false; /// O.map("optional_field", R.OptionalField); /// return true; /// } /// \endcode class ObjectMapper { public: ObjectMapper(const Value &E) : O(E.getAsObject()) {} /// True if the expression is an object. /// Must be checked before calling map(). operator bool() { return O; } /// Maps a property to a field, if it exists. template bool map(StringRef Prop, T &Out) { assert(*this && "Must check this is an object before calling map()"); if (const Value *E = O->get(Prop)) return fromJSON(*E, Out); return false; } /// Maps a property to a field, if it exists. /// (Optional requires special handling, because missing keys are OK). template bool map(StringRef Prop, llvm::Optional &Out) { assert(*this && "Must check this is an object before calling map()"); if (const Value *E = O->get(Prop)) return fromJSON(*E, Out); Out = llvm::None; return true; } private: const Object *O; }; /// Parses the provided JSON source, or returns a ParseError. /// The returned Value is self-contained and owns its strings (they do not refer /// to the original source). llvm::Expected parse(llvm::StringRef JSON); class ParseError : public llvm::ErrorInfo { const char *Msg; unsigned Line, Column, Offset; public: static char ID; ParseError(const char *Msg, unsigned Line, unsigned Column, unsigned Offset) : Msg(Msg), Line(Line), Column(Column), Offset(Offset) {} void log(llvm::raw_ostream &OS) const override { OS << llvm::formatv("[{0}:{1}, byte={2}]: {3}", Line, Column, Offset, Msg); } std::error_code convertToErrorCode() const override { return llvm::inconvertibleErrorCode(); } }; } // namespace json /// Allow printing json::Value with formatv(). /// The default style is basic/compact formatting, like operator<<. /// A format string like formatv("{0:2}", Value) pretty-prints with indent 2. template <> struct format_provider { static void format(const llvm::json::Value &, raw_ostream &, StringRef); }; } // namespace llvm #endif Index: vendor/llvm/dist-release_80/include/llvm/Transforms/Utils/FunctionImportUtils.h =================================================================== --- vendor/llvm/dist-release_80/include/llvm/Transforms/Utils/FunctionImportUtils.h (revision 343793) +++ vendor/llvm/dist-release_80/include/llvm/Transforms/Utils/FunctionImportUtils.h (revision 343794) @@ -1,122 +1,127 @@ //===- FunctionImportUtils.h - Importing support utilities -----*- C++ -*-===// // // The LLVM Compiler Infrastructure // // This file is distributed under the University of Illinois Open Source // License. See LICENSE.TXT for details. // //===----------------------------------------------------------------------===// // // This file defines the FunctionImportGlobalProcessing class which is used // to perform the necessary global value handling for function importing. // //===----------------------------------------------------------------------===// #ifndef LLVM_TRANSFORMS_UTILS_FUNCTIONIMPORTUTILS_H #define LLVM_TRANSFORMS_UTILS_FUNCTIONIMPORTUTILS_H #include "llvm/ADT/SetVector.h" #include "llvm/IR/ModuleSummaryIndex.h" namespace llvm { class Module; /// Class to handle necessary GlobalValue changes required by ThinLTO /// function importing, including linkage changes and any necessary renaming. class FunctionImportGlobalProcessing { /// The Module which we are exporting or importing functions from. Module &M; /// Module summary index passed in for function importing/exporting handling. const ModuleSummaryIndex &ImportIndex; /// Globals to import from this module, all other functions will be /// imported as declarations instead of definitions. SetVector *GlobalsToImport; /// Set to true if the given ModuleSummaryIndex contains any functions /// from this source module, in which case we must conservatively assume /// that any of its functions may be imported into another module /// as part of a different backend compilation process. bool HasExportedFunctions = false; /// Set of llvm.*used values, in order to validate that we don't try /// to promote any non-renamable values. SmallPtrSet Used; + /// Keep track of any COMDATs that require renaming (because COMDAT + /// leader was promoted and renamed). Maps from original COMDAT to one + /// with new name. + DenseMap RenamedComdats; + /// Check if we should promote the given local value to global scope. bool shouldPromoteLocalToGlobal(const GlobalValue *SGV); #ifndef NDEBUG /// Check if the given value is a local that can't be renamed (promoted). /// Only used in assertion checking, and disabled under NDEBUG since the Used /// set will not be populated. bool isNonRenamableLocal(const GlobalValue &GV) const; #endif /// Helper methods to check if we are importing from or potentially /// exporting from the current source module. bool isPerformingImport() const { return GlobalsToImport != nullptr; } bool isModuleExporting() const { return HasExportedFunctions; } /// If we are importing from the source module, checks if we should /// import SGV as a definition, otherwise import as a declaration. bool doImportAsDefinition(const GlobalValue *SGV); /// Get the name for SGV that should be used in the linked destination /// module. Specifically, this handles the case where we need to rename /// a local that is being promoted to global scope, which it will always /// do when \p DoPromote is true (or when importing a local). std::string getName(const GlobalValue *SGV, bool DoPromote); /// Process globals so that they can be used in ThinLTO. This includes /// promoting local variables so that they can be reference externally by /// thin lto imported globals and converting strong external globals to /// available_externally. void processGlobalsForThinLTO(); void processGlobalForThinLTO(GlobalValue &GV); /// Get the new linkage for SGV that should be used in the linked destination /// module. Specifically, for ThinLTO importing or exporting it may need /// to be adjusted. When \p DoPromote is true then we must adjust the /// linkage for a required promotion of a local to global scope. GlobalValue::LinkageTypes getLinkage(const GlobalValue *SGV, bool DoPromote); public: FunctionImportGlobalProcessing( Module &M, const ModuleSummaryIndex &Index, SetVector *GlobalsToImport = nullptr) : M(M), ImportIndex(Index), GlobalsToImport(GlobalsToImport) { // If we have a ModuleSummaryIndex but no function to import, // then this is the primary module being compiled in a ThinLTO // backend compilation, and we need to see if it has functions that // may be exported to another backend compilation. if (!GlobalsToImport) HasExportedFunctions = ImportIndex.hasExportedFunctions(M); #ifndef NDEBUG // First collect those in the llvm.used set. collectUsedGlobalVariables(M, Used, /*CompilerUsed*/ false); // Next collect those in the llvm.compiler.used set. collectUsedGlobalVariables(M, Used, /*CompilerUsed*/ true); #endif } bool run(); static bool doImportAsDefinition(const GlobalValue *SGV, SetVector *GlobalsToImport); }; /// Perform in-place global value handling on the given Module for /// exported local functions renamed and promoted for ThinLTO. bool renameModuleForThinLTO( Module &M, const ModuleSummaryIndex &Index, SetVector *GlobalsToImport = nullptr); /// Compute synthetic function entry counts. void computeSyntheticCounts(ModuleSummaryIndex &Index); } // End llvm namespace #endif Index: vendor/llvm/dist-release_80/lib/CodeGen/AsmPrinter/CodeViewDebug.cpp =================================================================== --- vendor/llvm/dist-release_80/lib/CodeGen/AsmPrinter/CodeViewDebug.cpp (revision 343793) +++ vendor/llvm/dist-release_80/lib/CodeGen/AsmPrinter/CodeViewDebug.cpp (revision 343794) @@ -1,3011 +1,3014 @@ //===- llvm/lib/CodeGen/AsmPrinter/CodeViewDebug.cpp ----------------------===// // // The LLVM Compiler Infrastructure // // This file is distributed under the University of Illinois Open Source // License. See LICENSE.TXT for details. // //===----------------------------------------------------------------------===// // // This file contains support for writing Microsoft CodeView debug info. // //===----------------------------------------------------------------------===// #include "CodeViewDebug.h" #include "DwarfExpression.h" #include "llvm/ADT/APSInt.h" #include "llvm/ADT/ArrayRef.h" #include "llvm/ADT/DenseMap.h" #include "llvm/ADT/DenseSet.h" #include "llvm/ADT/MapVector.h" #include "llvm/ADT/None.h" #include "llvm/ADT/Optional.h" #include "llvm/ADT/STLExtras.h" #include "llvm/ADT/SmallString.h" #include "llvm/ADT/SmallVector.h" #include "llvm/ADT/StringRef.h" #include "llvm/ADT/TinyPtrVector.h" #include "llvm/ADT/Triple.h" #include "llvm/ADT/Twine.h" #include "llvm/BinaryFormat/COFF.h" #include "llvm/BinaryFormat/Dwarf.h" #include "llvm/CodeGen/AsmPrinter.h" #include "llvm/CodeGen/LexicalScopes.h" #include "llvm/CodeGen/MachineFrameInfo.h" #include "llvm/CodeGen/MachineFunction.h" #include "llvm/CodeGen/MachineInstr.h" #include "llvm/CodeGen/MachineModuleInfo.h" #include "llvm/CodeGen/MachineOperand.h" #include "llvm/CodeGen/TargetFrameLowering.h" #include "llvm/CodeGen/TargetRegisterInfo.h" #include "llvm/CodeGen/TargetSubtargetInfo.h" #include "llvm/Config/llvm-config.h" #include "llvm/DebugInfo/CodeView/CVTypeVisitor.h" #include "llvm/DebugInfo/CodeView/CodeView.h" #include "llvm/DebugInfo/CodeView/ContinuationRecordBuilder.h" #include "llvm/DebugInfo/CodeView/DebugInlineeLinesSubsection.h" #include "llvm/DebugInfo/CodeView/EnumTables.h" #include "llvm/DebugInfo/CodeView/Line.h" #include "llvm/DebugInfo/CodeView/SymbolRecord.h" #include "llvm/DebugInfo/CodeView/TypeDumpVisitor.h" #include "llvm/DebugInfo/CodeView/TypeIndex.h" #include "llvm/DebugInfo/CodeView/TypeRecord.h" #include "llvm/DebugInfo/CodeView/TypeTableCollection.h" #include "llvm/IR/Constants.h" #include "llvm/IR/DataLayout.h" #include "llvm/IR/DebugInfoMetadata.h" #include "llvm/IR/DebugLoc.h" #include "llvm/IR/Function.h" #include "llvm/IR/GlobalValue.h" #include "llvm/IR/GlobalVariable.h" #include "llvm/IR/Metadata.h" #include "llvm/IR/Module.h" #include "llvm/MC/MCAsmInfo.h" #include "llvm/MC/MCContext.h" #include "llvm/MC/MCSectionCOFF.h" #include "llvm/MC/MCStreamer.h" #include "llvm/MC/MCSymbol.h" #include "llvm/Support/BinaryByteStream.h" #include "llvm/Support/BinaryStreamReader.h" #include "llvm/Support/Casting.h" #include "llvm/Support/CommandLine.h" #include "llvm/Support/Compiler.h" #include "llvm/Support/Endian.h" #include "llvm/Support/Error.h" #include "llvm/Support/ErrorHandling.h" #include "llvm/Support/FormatVariadic.h" #include "llvm/Support/Path.h" #include "llvm/Support/SMLoc.h" #include "llvm/Support/ScopedPrinter.h" #include "llvm/Target/TargetLoweringObjectFile.h" #include "llvm/Target/TargetMachine.h" #include #include #include #include #include #include #include #include #include #include using namespace llvm; using namespace llvm::codeview; static CPUType mapArchToCVCPUType(Triple::ArchType Type) { switch (Type) { case Triple::ArchType::x86: return CPUType::Pentium3; case Triple::ArchType::x86_64: return CPUType::X64; case Triple::ArchType::thumb: return CPUType::Thumb; case Triple::ArchType::aarch64: return CPUType::ARM64; default: report_fatal_error("target architecture doesn't map to a CodeView CPUType"); } } CodeViewDebug::CodeViewDebug(AsmPrinter *AP) : DebugHandlerBase(AP), OS(*Asm->OutStreamer), TypeTable(Allocator) { // If module doesn't have named metadata anchors or COFF debug section // is not available, skip any debug info related stuff. if (!MMI->getModule()->getNamedMetadata("llvm.dbg.cu") || !AP->getObjFileLowering().getCOFFDebugSymbolsSection()) { Asm = nullptr; MMI->setDebugInfoAvailability(false); return; } // Tell MMI that we have debug info. MMI->setDebugInfoAvailability(true); TheCPU = mapArchToCVCPUType(Triple(MMI->getModule()->getTargetTriple()).getArch()); collectGlobalVariableInfo(); // Check if we should emit type record hashes. ConstantInt *GH = mdconst::extract_or_null( MMI->getModule()->getModuleFlag("CodeViewGHash")); EmitDebugGlobalHashes = GH && !GH->isZero(); } StringRef CodeViewDebug::getFullFilepath(const DIFile *File) { std::string &Filepath = FileToFilepathMap[File]; if (!Filepath.empty()) return Filepath; StringRef Dir = File->getDirectory(), Filename = File->getFilename(); // If this is a Unix-style path, just use it as is. Don't try to canonicalize // it textually because one of the path components could be a symlink. if (Dir.startswith("/") || Filename.startswith("/")) { if (llvm::sys::path::is_absolute(Filename, llvm::sys::path::Style::posix)) return Filename; Filepath = Dir; if (Dir.back() != '/') Filepath += '/'; Filepath += Filename; return Filepath; } // Clang emits directory and relative filename info into the IR, but CodeView // operates on full paths. We could change Clang to emit full paths too, but // that would increase the IR size and probably not needed for other users. // For now, just concatenate and canonicalize the path here. if (Filename.find(':') == 1) Filepath = Filename; else Filepath = (Dir + "\\" + Filename).str(); // Canonicalize the path. We have to do it textually because we may no longer // have access the file in the filesystem. // First, replace all slashes with backslashes. std::replace(Filepath.begin(), Filepath.end(), '/', '\\'); // Remove all "\.\" with "\". size_t Cursor = 0; while ((Cursor = Filepath.find("\\.\\", Cursor)) != std::string::npos) Filepath.erase(Cursor, 2); // Replace all "\XXX\..\" with "\". Don't try too hard though as the original // path should be well-formatted, e.g. start with a drive letter, etc. Cursor = 0; while ((Cursor = Filepath.find("\\..\\", Cursor)) != std::string::npos) { // Something's wrong if the path starts with "\..\", abort. if (Cursor == 0) break; size_t PrevSlash = Filepath.rfind('\\', Cursor - 1); if (PrevSlash == std::string::npos) // Something's wrong, abort. break; Filepath.erase(PrevSlash, Cursor + 3 - PrevSlash); // The next ".." might be following the one we've just erased. Cursor = PrevSlash; } // Remove all duplicate backslashes. Cursor = 0; while ((Cursor = Filepath.find("\\\\", Cursor)) != std::string::npos) Filepath.erase(Cursor, 1); return Filepath; } unsigned CodeViewDebug::maybeRecordFile(const DIFile *F) { StringRef FullPath = getFullFilepath(F); unsigned NextId = FileIdMap.size() + 1; auto Insertion = FileIdMap.insert(std::make_pair(FullPath, NextId)); if (Insertion.second) { // We have to compute the full filepath and emit a .cv_file directive. ArrayRef ChecksumAsBytes; FileChecksumKind CSKind = FileChecksumKind::None; if (F->getChecksum()) { std::string Checksum = fromHex(F->getChecksum()->Value); void *CKMem = OS.getContext().allocate(Checksum.size(), 1); memcpy(CKMem, Checksum.data(), Checksum.size()); ChecksumAsBytes = ArrayRef( reinterpret_cast(CKMem), Checksum.size()); switch (F->getChecksum()->Kind) { case DIFile::CSK_MD5: CSKind = FileChecksumKind::MD5; break; case DIFile::CSK_SHA1: CSKind = FileChecksumKind::SHA1; break; } } bool Success = OS.EmitCVFileDirective(NextId, FullPath, ChecksumAsBytes, static_cast(CSKind)); (void)Success; assert(Success && ".cv_file directive failed"); } return Insertion.first->second; } CodeViewDebug::InlineSite & CodeViewDebug::getInlineSite(const DILocation *InlinedAt, const DISubprogram *Inlinee) { auto SiteInsertion = CurFn->InlineSites.insert({InlinedAt, InlineSite()}); InlineSite *Site = &SiteInsertion.first->second; if (SiteInsertion.second) { unsigned ParentFuncId = CurFn->FuncId; if (const DILocation *OuterIA = InlinedAt->getInlinedAt()) ParentFuncId = getInlineSite(OuterIA, InlinedAt->getScope()->getSubprogram()) .SiteFuncId; Site->SiteFuncId = NextFuncId++; OS.EmitCVInlineSiteIdDirective( Site->SiteFuncId, ParentFuncId, maybeRecordFile(InlinedAt->getFile()), InlinedAt->getLine(), InlinedAt->getColumn(), SMLoc()); Site->Inlinee = Inlinee; InlinedSubprograms.insert(Inlinee); getFuncIdForSubprogram(Inlinee); } return *Site; } static StringRef getPrettyScopeName(const DIScope *Scope) { StringRef ScopeName = Scope->getName(); if (!ScopeName.empty()) return ScopeName; switch (Scope->getTag()) { case dwarf::DW_TAG_enumeration_type: case dwarf::DW_TAG_class_type: case dwarf::DW_TAG_structure_type: case dwarf::DW_TAG_union_type: return ""; case dwarf::DW_TAG_namespace: return "`anonymous namespace'"; } return StringRef(); } static const DISubprogram *getQualifiedNameComponents( const DIScope *Scope, SmallVectorImpl &QualifiedNameComponents) { const DISubprogram *ClosestSubprogram = nullptr; while (Scope != nullptr) { if (ClosestSubprogram == nullptr) ClosestSubprogram = dyn_cast(Scope); StringRef ScopeName = getPrettyScopeName(Scope); if (!ScopeName.empty()) QualifiedNameComponents.push_back(ScopeName); Scope = Scope->getScope().resolve(); } return ClosestSubprogram; } static std::string getQualifiedName(ArrayRef QualifiedNameComponents, StringRef TypeName) { std::string FullyQualifiedName; for (StringRef QualifiedNameComponent : llvm::reverse(QualifiedNameComponents)) { FullyQualifiedName.append(QualifiedNameComponent); FullyQualifiedName.append("::"); } FullyQualifiedName.append(TypeName); return FullyQualifiedName; } static std::string getFullyQualifiedName(const DIScope *Scope, StringRef Name) { SmallVector QualifiedNameComponents; getQualifiedNameComponents(Scope, QualifiedNameComponents); return getQualifiedName(QualifiedNameComponents, Name); } struct CodeViewDebug::TypeLoweringScope { TypeLoweringScope(CodeViewDebug &CVD) : CVD(CVD) { ++CVD.TypeEmissionLevel; } ~TypeLoweringScope() { // Don't decrement TypeEmissionLevel until after emitting deferred types, so // inner TypeLoweringScopes don't attempt to emit deferred types. if (CVD.TypeEmissionLevel == 1) CVD.emitDeferredCompleteTypes(); --CVD.TypeEmissionLevel; } CodeViewDebug &CVD; }; static std::string getFullyQualifiedName(const DIScope *Ty) { const DIScope *Scope = Ty->getScope().resolve(); return getFullyQualifiedName(Scope, getPrettyScopeName(Ty)); } TypeIndex CodeViewDebug::getScopeIndex(const DIScope *Scope) { // No scope means global scope and that uses the zero index. if (!Scope || isa(Scope)) return TypeIndex(); assert(!isa(Scope) && "shouldn't make a namespace scope for a type"); // Check if we've already translated this scope. auto I = TypeIndices.find({Scope, nullptr}); if (I != TypeIndices.end()) return I->second; // Build the fully qualified name of the scope. std::string ScopeName = getFullyQualifiedName(Scope); StringIdRecord SID(TypeIndex(), ScopeName); auto TI = TypeTable.writeLeafType(SID); return recordTypeIndexForDINode(Scope, TI); } TypeIndex CodeViewDebug::getFuncIdForSubprogram(const DISubprogram *SP) { assert(SP); // Check if we've already translated this subprogram. auto I = TypeIndices.find({SP, nullptr}); if (I != TypeIndices.end()) return I->second; // The display name includes function template arguments. Drop them to match // MSVC. StringRef DisplayName = SP->getName().split('<').first; const DIScope *Scope = SP->getScope().resolve(); TypeIndex TI; if (const auto *Class = dyn_cast_or_null(Scope)) { // If the scope is a DICompositeType, then this must be a method. Member // function types take some special handling, and require access to the // subprogram. TypeIndex ClassType = getTypeIndex(Class); MemberFuncIdRecord MFuncId(ClassType, getMemberFunctionType(SP, Class), DisplayName); TI = TypeTable.writeLeafType(MFuncId); } else { // Otherwise, this must be a free function. TypeIndex ParentScope = getScopeIndex(Scope); FuncIdRecord FuncId(ParentScope, getTypeIndex(SP->getType()), DisplayName); TI = TypeTable.writeLeafType(FuncId); } return recordTypeIndexForDINode(SP, TI); } static bool isTrivial(const DICompositeType *DCTy) { return ((DCTy->getFlags() & DINode::FlagTrivial) == DINode::FlagTrivial); } static FunctionOptions getFunctionOptions(const DISubroutineType *Ty, const DICompositeType *ClassTy = nullptr, StringRef SPName = StringRef("")) { FunctionOptions FO = FunctionOptions::None; const DIType *ReturnTy = nullptr; if (auto TypeArray = Ty->getTypeArray()) { if (TypeArray.size()) ReturnTy = TypeArray[0].resolve(); } if (auto *ReturnDCTy = dyn_cast_or_null(ReturnTy)) { if (!isTrivial(ReturnDCTy)) FO |= FunctionOptions::CxxReturnUdt; } // DISubroutineType is unnamed. Use DISubprogram's i.e. SPName in comparison. if (ClassTy && !isTrivial(ClassTy) && SPName == ClassTy->getName()) { FO |= FunctionOptions::Constructor; // TODO: put the FunctionOptions::ConstructorWithVirtualBases flag. } return FO; } TypeIndex CodeViewDebug::getMemberFunctionType(const DISubprogram *SP, const DICompositeType *Class) { // Always use the method declaration as the key for the function type. The // method declaration contains the this adjustment. if (SP->getDeclaration()) SP = SP->getDeclaration(); assert(!SP->getDeclaration() && "should use declaration as key"); // Key the MemberFunctionRecord into the map as {SP, Class}. It won't collide // with the MemberFuncIdRecord, which is keyed in as {SP, nullptr}. auto I = TypeIndices.find({SP, Class}); if (I != TypeIndices.end()) return I->second; // Make sure complete type info for the class is emitted *after* the member // function type, as the complete class type is likely to reference this // member function type. TypeLoweringScope S(*this); const bool IsStaticMethod = (SP->getFlags() & DINode::FlagStaticMember) != 0; FunctionOptions FO = getFunctionOptions(SP->getType(), Class, SP->getName()); TypeIndex TI = lowerTypeMemberFunction( SP->getType(), Class, SP->getThisAdjustment(), IsStaticMethod, FO); return recordTypeIndexForDINode(SP, TI, Class); } TypeIndex CodeViewDebug::recordTypeIndexForDINode(const DINode *Node, TypeIndex TI, const DIType *ClassTy) { auto InsertResult = TypeIndices.insert({{Node, ClassTy}, TI}); (void)InsertResult; assert(InsertResult.second && "DINode was already assigned a type index"); return TI; } unsigned CodeViewDebug::getPointerSizeInBytes() { return MMI->getModule()->getDataLayout().getPointerSizeInBits() / 8; } void CodeViewDebug::recordLocalVariable(LocalVariable &&Var, const LexicalScope *LS) { if (const DILocation *InlinedAt = LS->getInlinedAt()) { // This variable was inlined. Associate it with the InlineSite. const DISubprogram *Inlinee = Var.DIVar->getScope()->getSubprogram(); InlineSite &Site = getInlineSite(InlinedAt, Inlinee); Site.InlinedLocals.emplace_back(Var); } else { // This variable goes into the corresponding lexical scope. ScopeVariables[LS].emplace_back(Var); } } static void addLocIfNotPresent(SmallVectorImpl &Locs, const DILocation *Loc) { auto B = Locs.begin(), E = Locs.end(); if (std::find(B, E, Loc) == E) Locs.push_back(Loc); } void CodeViewDebug::maybeRecordLocation(const DebugLoc &DL, const MachineFunction *MF) { // Skip this instruction if it has the same location as the previous one. if (!DL || DL == PrevInstLoc) return; const DIScope *Scope = DL.get()->getScope(); if (!Scope) return; // Skip this line if it is longer than the maximum we can record. LineInfo LI(DL.getLine(), DL.getLine(), /*IsStatement=*/true); if (LI.getStartLine() != DL.getLine() || LI.isAlwaysStepInto() || LI.isNeverStepInto()) return; ColumnInfo CI(DL.getCol(), /*EndColumn=*/0); if (CI.getStartColumn() != DL.getCol()) return; if (!CurFn->HaveLineInfo) CurFn->HaveLineInfo = true; unsigned FileId = 0; if (PrevInstLoc.get() && PrevInstLoc->getFile() == DL->getFile()) FileId = CurFn->LastFileId; else FileId = CurFn->LastFileId = maybeRecordFile(DL->getFile()); PrevInstLoc = DL; unsigned FuncId = CurFn->FuncId; if (const DILocation *SiteLoc = DL->getInlinedAt()) { const DILocation *Loc = DL.get(); // If this location was actually inlined from somewhere else, give it the ID // of the inline call site. FuncId = getInlineSite(SiteLoc, Loc->getScope()->getSubprogram()).SiteFuncId; // Ensure we have links in the tree of inline call sites. bool FirstLoc = true; while ((SiteLoc = Loc->getInlinedAt())) { InlineSite &Site = getInlineSite(SiteLoc, Loc->getScope()->getSubprogram()); if (!FirstLoc) addLocIfNotPresent(Site.ChildSites, Loc); FirstLoc = false; Loc = SiteLoc; } addLocIfNotPresent(CurFn->ChildSites, Loc); } OS.EmitCVLocDirective(FuncId, FileId, DL.getLine(), DL.getCol(), /*PrologueEnd=*/false, /*IsStmt=*/false, DL->getFilename(), SMLoc()); } void CodeViewDebug::emitCodeViewMagicVersion() { OS.EmitValueToAlignment(4); OS.AddComment("Debug section magic"); OS.EmitIntValue(COFF::DEBUG_SECTION_MAGIC, 4); } void CodeViewDebug::endModule() { if (!Asm || !MMI->hasDebugInfo()) return; assert(Asm != nullptr); // The COFF .debug$S section consists of several subsections, each starting // with a 4-byte control code (e.g. 0xF1, 0xF2, etc) and then a 4-byte length // of the payload followed by the payload itself. The subsections are 4-byte // aligned. // Use the generic .debug$S section, and make a subsection for all the inlined // subprograms. switchToDebugSectionForSymbol(nullptr); MCSymbol *CompilerInfo = beginCVSubsection(DebugSubsectionKind::Symbols); emitCompilerInformation(); endCVSubsection(CompilerInfo); emitInlineeLinesSubsection(); // Emit per-function debug information. for (auto &P : FnDebugInfo) if (!P.first->isDeclarationForLinker()) emitDebugInfoForFunction(P.first, *P.second); // Emit global variable debug information. setCurrentSubprogram(nullptr); emitDebugInfoForGlobals(); // Emit retained types. emitDebugInfoForRetainedTypes(); // Switch back to the generic .debug$S section after potentially processing // comdat symbol sections. switchToDebugSectionForSymbol(nullptr); // Emit UDT records for any types used by global variables. if (!GlobalUDTs.empty()) { MCSymbol *SymbolsEnd = beginCVSubsection(DebugSubsectionKind::Symbols); emitDebugInfoForUDTs(GlobalUDTs); endCVSubsection(SymbolsEnd); } // This subsection holds a file index to offset in string table table. OS.AddComment("File index to string table offset subsection"); OS.EmitCVFileChecksumsDirective(); // This subsection holds the string table. OS.AddComment("String table"); OS.EmitCVStringTableDirective(); // Emit S_BUILDINFO, which points to LF_BUILDINFO. Put this in its own symbol // subsection in the generic .debug$S section at the end. There is no // particular reason for this ordering other than to match MSVC. emitBuildInfo(); // Emit type information and hashes last, so that any types we translate while // emitting function info are included. emitTypeInformation(); if (EmitDebugGlobalHashes) emitTypeGlobalHashes(); clear(); } static void emitNullTerminatedSymbolName(MCStreamer &OS, StringRef S, unsigned MaxFixedRecordLength = 0xF00) { // The maximum CV record length is 0xFF00. Most of the strings we emit appear // after a fixed length portion of the record. The fixed length portion should // always be less than 0xF00 (3840) bytes, so truncate the string so that the // overall record size is less than the maximum allowed. SmallString<32> NullTerminatedString( S.take_front(MaxRecordLength - MaxFixedRecordLength - 1)); NullTerminatedString.push_back('\0'); OS.EmitBytes(NullTerminatedString); } void CodeViewDebug::emitTypeInformation() { if (TypeTable.empty()) return; // Start the .debug$T or .debug$P section with 0x4. OS.SwitchSection(Asm->getObjFileLowering().getCOFFDebugTypesSection()); emitCodeViewMagicVersion(); SmallString<8> CommentPrefix; if (OS.isVerboseAsm()) { CommentPrefix += '\t'; CommentPrefix += Asm->MAI->getCommentString(); CommentPrefix += ' '; } TypeTableCollection Table(TypeTable.records()); Optional B = Table.getFirst(); while (B) { // This will fail if the record data is invalid. CVType Record = Table.getType(*B); if (OS.isVerboseAsm()) { // Emit a block comment describing the type record for readability. SmallString<512> CommentBlock; raw_svector_ostream CommentOS(CommentBlock); ScopedPrinter SP(CommentOS); SP.setPrefix(CommentPrefix); TypeDumpVisitor TDV(Table, &SP, false); Error E = codeview::visitTypeRecord(Record, *B, TDV); if (E) { logAllUnhandledErrors(std::move(E), errs(), "error: "); llvm_unreachable("produced malformed type record"); } // emitRawComment will insert its own tab and comment string before // the first line, so strip off our first one. It also prints its own // newline. OS.emitRawComment( CommentOS.str().drop_front(CommentPrefix.size() - 1).rtrim()); } OS.EmitBinaryData(Record.str_data()); B = Table.getNext(*B); } } void CodeViewDebug::emitTypeGlobalHashes() { if (TypeTable.empty()) return; // Start the .debug$H section with the version and hash algorithm, currently // hardcoded to version 0, SHA1. OS.SwitchSection(Asm->getObjFileLowering().getCOFFGlobalTypeHashesSection()); OS.EmitValueToAlignment(4); OS.AddComment("Magic"); OS.EmitIntValue(COFF::DEBUG_HASHES_SECTION_MAGIC, 4); OS.AddComment("Section Version"); OS.EmitIntValue(0, 2); OS.AddComment("Hash Algorithm"); OS.EmitIntValue(uint16_t(GlobalTypeHashAlg::SHA1_8), 2); TypeIndex TI(TypeIndex::FirstNonSimpleIndex); for (const auto &GHR : TypeTable.hashes()) { if (OS.isVerboseAsm()) { // Emit an EOL-comment describing which TypeIndex this hash corresponds // to, as well as the stringified SHA1 hash. SmallString<32> Comment; raw_svector_ostream CommentOS(Comment); CommentOS << formatv("{0:X+} [{1}]", TI.getIndex(), GHR); OS.AddComment(Comment); ++TI; } assert(GHR.Hash.size() == 8); StringRef S(reinterpret_cast(GHR.Hash.data()), GHR.Hash.size()); OS.EmitBinaryData(S); } } static SourceLanguage MapDWLangToCVLang(unsigned DWLang) { switch (DWLang) { case dwarf::DW_LANG_C: case dwarf::DW_LANG_C89: case dwarf::DW_LANG_C99: case dwarf::DW_LANG_C11: case dwarf::DW_LANG_ObjC: return SourceLanguage::C; case dwarf::DW_LANG_C_plus_plus: case dwarf::DW_LANG_C_plus_plus_03: case dwarf::DW_LANG_C_plus_plus_11: case dwarf::DW_LANG_C_plus_plus_14: return SourceLanguage::Cpp; case dwarf::DW_LANG_Fortran77: case dwarf::DW_LANG_Fortran90: case dwarf::DW_LANG_Fortran03: case dwarf::DW_LANG_Fortran08: return SourceLanguage::Fortran; case dwarf::DW_LANG_Pascal83: return SourceLanguage::Pascal; case dwarf::DW_LANG_Cobol74: case dwarf::DW_LANG_Cobol85: return SourceLanguage::Cobol; case dwarf::DW_LANG_Java: return SourceLanguage::Java; case dwarf::DW_LANG_D: return SourceLanguage::D; default: // There's no CodeView representation for this language, and CV doesn't // have an "unknown" option for the language field, so we'll use MASM, // as it's very low level. return SourceLanguage::Masm; } } namespace { struct Version { int Part[4]; }; } // end anonymous namespace // Takes a StringRef like "clang 4.0.0.0 (other nonsense 123)" and parses out // the version number. static Version parseVersion(StringRef Name) { Version V = {{0}}; int N = 0; for (const char C : Name) { if (isdigit(C)) { V.Part[N] *= 10; V.Part[N] += C - '0'; } else if (C == '.') { ++N; if (N >= 4) return V; } else if (N > 0) return V; } return V; } void CodeViewDebug::emitCompilerInformation() { MCSymbol *CompilerEnd = beginSymbolRecord(SymbolKind::S_COMPILE3); uint32_t Flags = 0; NamedMDNode *CUs = MMI->getModule()->getNamedMetadata("llvm.dbg.cu"); const MDNode *Node = *CUs->operands().begin(); const auto *CU = cast(Node); // The low byte of the flags indicates the source language. Flags = MapDWLangToCVLang(CU->getSourceLanguage()); // TODO: Figure out which other flags need to be set. OS.AddComment("Flags and language"); OS.EmitIntValue(Flags, 4); OS.AddComment("CPUType"); OS.EmitIntValue(static_cast(TheCPU), 2); StringRef CompilerVersion = CU->getProducer(); Version FrontVer = parseVersion(CompilerVersion); OS.AddComment("Frontend version"); for (int N = 0; N < 4; ++N) OS.EmitIntValue(FrontVer.Part[N], 2); // Some Microsoft tools, like Binscope, expect a backend version number of at // least 8.something, so we'll coerce the LLVM version into a form that // guarantees it'll be big enough without really lying about the version. int Major = 1000 * LLVM_VERSION_MAJOR + 10 * LLVM_VERSION_MINOR + LLVM_VERSION_PATCH; // Clamp it for builds that use unusually large version numbers. Major = std::min(Major, std::numeric_limits::max()); Version BackVer = {{ Major, 0, 0, 0 }}; OS.AddComment("Backend version"); for (int N = 0; N < 4; ++N) OS.EmitIntValue(BackVer.Part[N], 2); OS.AddComment("Null-terminated compiler version string"); emitNullTerminatedSymbolName(OS, CompilerVersion); endSymbolRecord(CompilerEnd); } static TypeIndex getStringIdTypeIdx(GlobalTypeTableBuilder &TypeTable, StringRef S) { StringIdRecord SIR(TypeIndex(0x0), S); return TypeTable.writeLeafType(SIR); } void CodeViewDebug::emitBuildInfo() { // First, make LF_BUILDINFO. It's a sequence of strings with various bits of // build info. The known prefix is: // - Absolute path of current directory // - Compiler path // - Main source file path, relative to CWD or absolute // - Type server PDB file // - Canonical compiler command line // If frontend and backend compilation are separated (think llc or LTO), it's // not clear if the compiler path should refer to the executable for the // frontend or the backend. Leave it blank for now. TypeIndex BuildInfoArgs[BuildInfoRecord::MaxArgs] = {}; NamedMDNode *CUs = MMI->getModule()->getNamedMetadata("llvm.dbg.cu"); const MDNode *Node = *CUs->operands().begin(); // FIXME: Multiple CUs. const auto *CU = cast(Node); const DIFile *MainSourceFile = CU->getFile(); BuildInfoArgs[BuildInfoRecord::CurrentDirectory] = getStringIdTypeIdx(TypeTable, MainSourceFile->getDirectory()); BuildInfoArgs[BuildInfoRecord::SourceFile] = getStringIdTypeIdx(TypeTable, MainSourceFile->getFilename()); // FIXME: Path to compiler and command line. PDB is intentionally blank unless // we implement /Zi type servers. BuildInfoRecord BIR(BuildInfoArgs); TypeIndex BuildInfoIndex = TypeTable.writeLeafType(BIR); // Make a new .debug$S subsection for the S_BUILDINFO record, which points // from the module symbols into the type stream. MCSymbol *BISubsecEnd = beginCVSubsection(DebugSubsectionKind::Symbols); MCSymbol *BIEnd = beginSymbolRecord(SymbolKind::S_BUILDINFO); OS.AddComment("LF_BUILDINFO index"); OS.EmitIntValue(BuildInfoIndex.getIndex(), 4); endSymbolRecord(BIEnd); endCVSubsection(BISubsecEnd); } void CodeViewDebug::emitInlineeLinesSubsection() { if (InlinedSubprograms.empty()) return; OS.AddComment("Inlinee lines subsection"); MCSymbol *InlineEnd = beginCVSubsection(DebugSubsectionKind::InlineeLines); // We emit the checksum info for files. This is used by debuggers to // determine if a pdb matches the source before loading it. Visual Studio, // for instance, will display a warning that the breakpoints are not valid if // the pdb does not match the source. OS.AddComment("Inlinee lines signature"); OS.EmitIntValue(unsigned(InlineeLinesSignature::Normal), 4); for (const DISubprogram *SP : InlinedSubprograms) { assert(TypeIndices.count({SP, nullptr})); TypeIndex InlineeIdx = TypeIndices[{SP, nullptr}]; OS.AddBlankLine(); unsigned FileId = maybeRecordFile(SP->getFile()); OS.AddComment("Inlined function " + SP->getName() + " starts at " + SP->getFilename() + Twine(':') + Twine(SP->getLine())); OS.AddBlankLine(); OS.AddComment("Type index of inlined function"); OS.EmitIntValue(InlineeIdx.getIndex(), 4); OS.AddComment("Offset into filechecksum table"); OS.EmitCVFileChecksumOffsetDirective(FileId); OS.AddComment("Starting line number"); OS.EmitIntValue(SP->getLine(), 4); } endCVSubsection(InlineEnd); } void CodeViewDebug::emitInlinedCallSite(const FunctionInfo &FI, const DILocation *InlinedAt, const InlineSite &Site) { assert(TypeIndices.count({Site.Inlinee, nullptr})); TypeIndex InlineeIdx = TypeIndices[{Site.Inlinee, nullptr}]; // SymbolRecord MCSymbol *InlineEnd = beginSymbolRecord(SymbolKind::S_INLINESITE); OS.AddComment("PtrParent"); OS.EmitIntValue(0, 4); OS.AddComment("PtrEnd"); OS.EmitIntValue(0, 4); OS.AddComment("Inlinee type index"); OS.EmitIntValue(InlineeIdx.getIndex(), 4); unsigned FileId = maybeRecordFile(Site.Inlinee->getFile()); unsigned StartLineNum = Site.Inlinee->getLine(); OS.EmitCVInlineLinetableDirective(Site.SiteFuncId, FileId, StartLineNum, FI.Begin, FI.End); endSymbolRecord(InlineEnd); emitLocalVariableList(FI, Site.InlinedLocals); // Recurse on child inlined call sites before closing the scope. for (const DILocation *ChildSite : Site.ChildSites) { auto I = FI.InlineSites.find(ChildSite); assert(I != FI.InlineSites.end() && "child site not in function inline site map"); emitInlinedCallSite(FI, ChildSite, I->second); } // Close the scope. emitEndSymbolRecord(SymbolKind::S_INLINESITE_END); } void CodeViewDebug::switchToDebugSectionForSymbol(const MCSymbol *GVSym) { // If we have a symbol, it may be in a section that is COMDAT. If so, find the // comdat key. A section may be comdat because of -ffunction-sections or // because it is comdat in the IR. MCSectionCOFF *GVSec = GVSym ? dyn_cast(&GVSym->getSection()) : nullptr; const MCSymbol *KeySym = GVSec ? GVSec->getCOMDATSymbol() : nullptr; MCSectionCOFF *DebugSec = cast( Asm->getObjFileLowering().getCOFFDebugSymbolsSection()); DebugSec = OS.getContext().getAssociativeCOFFSection(DebugSec, KeySym); OS.SwitchSection(DebugSec); // Emit the magic version number if this is the first time we've switched to // this section. if (ComdatDebugSections.insert(DebugSec).second) emitCodeViewMagicVersion(); } // Emit an S_THUNK32/S_END symbol pair for a thunk routine. // The only supported thunk ordinal is currently the standard type. void CodeViewDebug::emitDebugInfoForThunk(const Function *GV, FunctionInfo &FI, const MCSymbol *Fn) { std::string FuncName = GlobalValue::dropLLVMManglingEscape(GV->getName()); const ThunkOrdinal ordinal = ThunkOrdinal::Standard; // Only supported kind. OS.AddComment("Symbol subsection for " + Twine(FuncName)); MCSymbol *SymbolsEnd = beginCVSubsection(DebugSubsectionKind::Symbols); // Emit S_THUNK32 MCSymbol *ThunkRecordEnd = beginSymbolRecord(SymbolKind::S_THUNK32); OS.AddComment("PtrParent"); OS.EmitIntValue(0, 4); OS.AddComment("PtrEnd"); OS.EmitIntValue(0, 4); OS.AddComment("PtrNext"); OS.EmitIntValue(0, 4); OS.AddComment("Thunk section relative address"); OS.EmitCOFFSecRel32(Fn, /*Offset=*/0); OS.AddComment("Thunk section index"); OS.EmitCOFFSectionIndex(Fn); OS.AddComment("Code size"); OS.emitAbsoluteSymbolDiff(FI.End, Fn, 2); OS.AddComment("Ordinal"); OS.EmitIntValue(unsigned(ordinal), 1); OS.AddComment("Function name"); emitNullTerminatedSymbolName(OS, FuncName); // Additional fields specific to the thunk ordinal would go here. endSymbolRecord(ThunkRecordEnd); // Local variables/inlined routines are purposely omitted here. The point of // marking this as a thunk is so Visual Studio will NOT stop in this routine. // Emit S_PROC_ID_END emitEndSymbolRecord(SymbolKind::S_PROC_ID_END); endCVSubsection(SymbolsEnd); } void CodeViewDebug::emitDebugInfoForFunction(const Function *GV, FunctionInfo &FI) { // For each function there is a separate subsection which holds the PC to // file:line table. const MCSymbol *Fn = Asm->getSymbol(GV); assert(Fn); // Switch to the to a comdat section, if appropriate. switchToDebugSectionForSymbol(Fn); std::string FuncName; auto *SP = GV->getSubprogram(); assert(SP); setCurrentSubprogram(SP); if (SP->isThunk()) { emitDebugInfoForThunk(GV, FI, Fn); return; } // If we have a display name, build the fully qualified name by walking the // chain of scopes. if (!SP->getName().empty()) FuncName = getFullyQualifiedName(SP->getScope().resolve(), SP->getName()); // If our DISubprogram name is empty, use the mangled name. if (FuncName.empty()) FuncName = GlobalValue::dropLLVMManglingEscape(GV->getName()); // Emit FPO data, but only on 32-bit x86. No other platforms use it. if (Triple(MMI->getModule()->getTargetTriple()).getArch() == Triple::x86) OS.EmitCVFPOData(Fn); // Emit a symbol subsection, required by VS2012+ to find function boundaries. OS.AddComment("Symbol subsection for " + Twine(FuncName)); MCSymbol *SymbolsEnd = beginCVSubsection(DebugSubsectionKind::Symbols); { SymbolKind ProcKind = GV->hasLocalLinkage() ? SymbolKind::S_LPROC32_ID : SymbolKind::S_GPROC32_ID; MCSymbol *ProcRecordEnd = beginSymbolRecord(ProcKind); // These fields are filled in by tools like CVPACK which run after the fact. OS.AddComment("PtrParent"); OS.EmitIntValue(0, 4); OS.AddComment("PtrEnd"); OS.EmitIntValue(0, 4); OS.AddComment("PtrNext"); OS.EmitIntValue(0, 4); // This is the important bit that tells the debugger where the function // code is located and what's its size: OS.AddComment("Code size"); OS.emitAbsoluteSymbolDiff(FI.End, Fn, 4); OS.AddComment("Offset after prologue"); OS.EmitIntValue(0, 4); OS.AddComment("Offset before epilogue"); OS.EmitIntValue(0, 4); OS.AddComment("Function type index"); OS.EmitIntValue(getFuncIdForSubprogram(GV->getSubprogram()).getIndex(), 4); OS.AddComment("Function section relative address"); OS.EmitCOFFSecRel32(Fn, /*Offset=*/0); OS.AddComment("Function section index"); OS.EmitCOFFSectionIndex(Fn); OS.AddComment("Flags"); OS.EmitIntValue(0, 1); // Emit the function display name as a null-terminated string. OS.AddComment("Function name"); // Truncate the name so we won't overflow the record length field. emitNullTerminatedSymbolName(OS, FuncName); endSymbolRecord(ProcRecordEnd); MCSymbol *FrameProcEnd = beginSymbolRecord(SymbolKind::S_FRAMEPROC); // Subtract out the CSR size since MSVC excludes that and we include it. OS.AddComment("FrameSize"); OS.EmitIntValue(FI.FrameSize - FI.CSRSize, 4); OS.AddComment("Padding"); OS.EmitIntValue(0, 4); OS.AddComment("Offset of padding"); OS.EmitIntValue(0, 4); OS.AddComment("Bytes of callee saved registers"); OS.EmitIntValue(FI.CSRSize, 4); OS.AddComment("Exception handler offset"); OS.EmitIntValue(0, 4); OS.AddComment("Exception handler section"); OS.EmitIntValue(0, 2); OS.AddComment("Flags (defines frame register)"); OS.EmitIntValue(uint32_t(FI.FrameProcOpts), 4); endSymbolRecord(FrameProcEnd); emitLocalVariableList(FI, FI.Locals); emitGlobalVariableList(FI.Globals); emitLexicalBlockList(FI.ChildBlocks, FI); // Emit inlined call site information. Only emit functions inlined directly // into the parent function. We'll emit the other sites recursively as part // of their parent inline site. for (const DILocation *InlinedAt : FI.ChildSites) { auto I = FI.InlineSites.find(InlinedAt); assert(I != FI.InlineSites.end() && "child site not in function inline site map"); emitInlinedCallSite(FI, InlinedAt, I->second); } for (auto Annot : FI.Annotations) { MCSymbol *Label = Annot.first; MDTuple *Strs = cast(Annot.second); MCSymbol *AnnotEnd = beginSymbolRecord(SymbolKind::S_ANNOTATION); OS.EmitCOFFSecRel32(Label, /*Offset=*/0); // FIXME: Make sure we don't overflow the max record size. OS.EmitCOFFSectionIndex(Label); OS.EmitIntValue(Strs->getNumOperands(), 2); for (Metadata *MD : Strs->operands()) { // MDStrings are null terminated, so we can do EmitBytes and get the // nice .asciz directive. StringRef Str = cast(MD)->getString(); assert(Str.data()[Str.size()] == '\0' && "non-nullterminated MDString"); OS.EmitBytes(StringRef(Str.data(), Str.size() + 1)); } endSymbolRecord(AnnotEnd); } if (SP != nullptr) emitDebugInfoForUDTs(LocalUDTs); // We're done with this function. emitEndSymbolRecord(SymbolKind::S_PROC_ID_END); } endCVSubsection(SymbolsEnd); // We have an assembler directive that takes care of the whole line table. OS.EmitCVLinetableDirective(FI.FuncId, Fn, FI.End); } CodeViewDebug::LocalVarDefRange CodeViewDebug::createDefRangeMem(uint16_t CVRegister, int Offset) { LocalVarDefRange DR; DR.InMemory = -1; DR.DataOffset = Offset; assert(DR.DataOffset == Offset && "truncation"); DR.IsSubfield = 0; DR.StructOffset = 0; DR.CVRegister = CVRegister; return DR; } void CodeViewDebug::collectVariableInfoFromMFTable( DenseSet &Processed) { const MachineFunction &MF = *Asm->MF; const TargetSubtargetInfo &TSI = MF.getSubtarget(); const TargetFrameLowering *TFI = TSI.getFrameLowering(); const TargetRegisterInfo *TRI = TSI.getRegisterInfo(); for (const MachineFunction::VariableDbgInfo &VI : MF.getVariableDbgInfo()) { if (!VI.Var) continue; assert(VI.Var->isValidLocationForIntrinsic(VI.Loc) && "Expected inlined-at fields to agree"); Processed.insert(InlinedEntity(VI.Var, VI.Loc->getInlinedAt())); LexicalScope *Scope = LScopes.findLexicalScope(VI.Loc); // If variable scope is not found then skip this variable. if (!Scope) continue; // If the variable has an attached offset expression, extract it. // FIXME: Try to handle DW_OP_deref as well. int64_t ExprOffset = 0; if (VI.Expr) if (!VI.Expr->extractIfOffset(ExprOffset)) continue; // Get the frame register used and the offset. unsigned FrameReg = 0; int FrameOffset = TFI->getFrameIndexReference(*Asm->MF, VI.Slot, FrameReg); uint16_t CVReg = TRI->getCodeViewRegNum(FrameReg); // Calculate the label ranges. LocalVarDefRange DefRange = createDefRangeMem(CVReg, FrameOffset + ExprOffset); for (const InsnRange &Range : Scope->getRanges()) { const MCSymbol *Begin = getLabelBeforeInsn(Range.first); const MCSymbol *End = getLabelAfterInsn(Range.second); End = End ? End : Asm->getFunctionEnd(); DefRange.Ranges.emplace_back(Begin, End); } LocalVariable Var; Var.DIVar = VI.Var; Var.DefRanges.emplace_back(std::move(DefRange)); recordLocalVariable(std::move(Var), Scope); } } static bool canUseReferenceType(const DbgVariableLocation &Loc) { return !Loc.LoadChain.empty() && Loc.LoadChain.back() == 0; } static bool needsReferenceType(const DbgVariableLocation &Loc) { return Loc.LoadChain.size() == 2 && Loc.LoadChain.back() == 0; } void CodeViewDebug::calculateRanges( LocalVariable &Var, const DbgValueHistoryMap::InstrRanges &Ranges) { const TargetRegisterInfo *TRI = Asm->MF->getSubtarget().getRegisterInfo(); // Calculate the definition ranges. for (auto I = Ranges.begin(), E = Ranges.end(); I != E; ++I) { const InsnRange &Range = *I; const MachineInstr *DVInst = Range.first; assert(DVInst->isDebugValue() && "Invalid History entry"); // FIXME: Find a way to represent constant variables, since they are // relatively common. Optional Location = DbgVariableLocation::extractFromMachineInstruction(*DVInst); if (!Location) continue; // CodeView can only express variables in register and variables in memory // at a constant offset from a register. However, for variables passed // indirectly by pointer, it is common for that pointer to be spilled to a // stack location. For the special case of one offseted load followed by a // zero offset load (a pointer spilled to the stack), we change the type of // the local variable from a value type to a reference type. This tricks the // debugger into doing the load for us. if (Var.UseReferenceType) { // We're using a reference type. Drop the last zero offset load. if (canUseReferenceType(*Location)) Location->LoadChain.pop_back(); else continue; } else if (needsReferenceType(*Location)) { // This location can't be expressed without switching to a reference type. // Start over using that. Var.UseReferenceType = true; Var.DefRanges.clear(); calculateRanges(Var, Ranges); return; } // We can only handle a register or an offseted load of a register. if (Location->Register == 0 || Location->LoadChain.size() > 1) continue; { LocalVarDefRange DR; DR.CVRegister = TRI->getCodeViewRegNum(Location->Register); DR.InMemory = !Location->LoadChain.empty(); DR.DataOffset = !Location->LoadChain.empty() ? Location->LoadChain.back() : 0; if (Location->FragmentInfo) { DR.IsSubfield = true; DR.StructOffset = Location->FragmentInfo->OffsetInBits / 8; } else { DR.IsSubfield = false; DR.StructOffset = 0; } if (Var.DefRanges.empty() || Var.DefRanges.back().isDifferentLocation(DR)) { Var.DefRanges.emplace_back(std::move(DR)); } } // Compute the label range. const MCSymbol *Begin = getLabelBeforeInsn(Range.first); const MCSymbol *End = getLabelAfterInsn(Range.second); if (!End) { // This range is valid until the next overlapping bitpiece. In the // common case, ranges will not be bitpieces, so they will overlap. auto J = std::next(I); const DIExpression *DIExpr = DVInst->getDebugExpression(); while (J != E && !DIExpr->fragmentsOverlap(J->first->getDebugExpression())) ++J; if (J != E) End = getLabelBeforeInsn(J->first); else End = Asm->getFunctionEnd(); } // If the last range end is our begin, just extend the last range. // Otherwise make a new range. SmallVectorImpl> &R = Var.DefRanges.back().Ranges; if (!R.empty() && R.back().second == Begin) R.back().second = End; else R.emplace_back(Begin, End); // FIXME: Do more range combining. } } void CodeViewDebug::collectVariableInfo(const DISubprogram *SP) { DenseSet Processed; // Grab the variable info that was squirreled away in the MMI side-table. collectVariableInfoFromMFTable(Processed); for (const auto &I : DbgValues) { InlinedEntity IV = I.first; if (Processed.count(IV)) continue; const DILocalVariable *DIVar = cast(IV.first); const DILocation *InlinedAt = IV.second; // Instruction ranges, specifying where IV is accessible. const auto &Ranges = I.second; LexicalScope *Scope = nullptr; if (InlinedAt) Scope = LScopes.findInlinedScope(DIVar->getScope(), InlinedAt); else Scope = LScopes.findLexicalScope(DIVar->getScope()); // If variable scope is not found then skip this variable. if (!Scope) continue; LocalVariable Var; Var.DIVar = DIVar; calculateRanges(Var, Ranges); recordLocalVariable(std::move(Var), Scope); } } void CodeViewDebug::beginFunctionImpl(const MachineFunction *MF) { const TargetSubtargetInfo &TSI = MF->getSubtarget(); const TargetRegisterInfo *TRI = TSI.getRegisterInfo(); const MachineFrameInfo &MFI = MF->getFrameInfo(); const Function &GV = MF->getFunction(); auto Insertion = FnDebugInfo.insert({&GV, llvm::make_unique()}); assert(Insertion.second && "function already has info"); CurFn = Insertion.first->second.get(); CurFn->FuncId = NextFuncId++; CurFn->Begin = Asm->getFunctionBegin(); // The S_FRAMEPROC record reports the stack size, and how many bytes of // callee-saved registers were used. For targets that don't use a PUSH // instruction (AArch64), this will be zero. CurFn->CSRSize = MFI.getCVBytesOfCalleeSavedRegisters(); CurFn->FrameSize = MFI.getStackSize(); CurFn->OffsetAdjustment = MFI.getOffsetAdjustment(); CurFn->HasStackRealignment = TRI->needsStackRealignment(*MF); // For this function S_FRAMEPROC record, figure out which codeview register // will be the frame pointer. CurFn->EncodedParamFramePtrReg = EncodedFramePtrReg::None; // None. CurFn->EncodedLocalFramePtrReg = EncodedFramePtrReg::None; // None. if (CurFn->FrameSize > 0) { if (!TSI.getFrameLowering()->hasFP(*MF)) { CurFn->EncodedLocalFramePtrReg = EncodedFramePtrReg::StackPtr; CurFn->EncodedParamFramePtrReg = EncodedFramePtrReg::StackPtr; } else { // If there is an FP, parameters are always relative to it. CurFn->EncodedParamFramePtrReg = EncodedFramePtrReg::FramePtr; if (CurFn->HasStackRealignment) { // If the stack needs realignment, locals are relative to SP or VFRAME. CurFn->EncodedLocalFramePtrReg = EncodedFramePtrReg::StackPtr; } else { // Otherwise, locals are relative to EBP, and we probably have VLAs or // other stack adjustments. CurFn->EncodedLocalFramePtrReg = EncodedFramePtrReg::FramePtr; } } } // Compute other frame procedure options. FrameProcedureOptions FPO = FrameProcedureOptions::None; if (MFI.hasVarSizedObjects()) FPO |= FrameProcedureOptions::HasAlloca; if (MF->exposesReturnsTwice()) FPO |= FrameProcedureOptions::HasSetJmp; // FIXME: Set HasLongJmp if we ever track that info. if (MF->hasInlineAsm()) FPO |= FrameProcedureOptions::HasInlineAssembly; if (GV.hasPersonalityFn()) { if (isAsynchronousEHPersonality( classifyEHPersonality(GV.getPersonalityFn()))) FPO |= FrameProcedureOptions::HasStructuredExceptionHandling; else FPO |= FrameProcedureOptions::HasExceptionHandling; } if (GV.hasFnAttribute(Attribute::InlineHint)) FPO |= FrameProcedureOptions::MarkedInline; if (GV.hasFnAttribute(Attribute::Naked)) FPO |= FrameProcedureOptions::Naked; if (MFI.hasStackProtectorIndex()) FPO |= FrameProcedureOptions::SecurityChecks; FPO |= FrameProcedureOptions(uint32_t(CurFn->EncodedLocalFramePtrReg) << 14U); FPO |= FrameProcedureOptions(uint32_t(CurFn->EncodedParamFramePtrReg) << 16U); if (Asm->TM.getOptLevel() != CodeGenOpt::None && !GV.optForSize() && !GV.hasFnAttribute(Attribute::OptimizeNone)) FPO |= FrameProcedureOptions::OptimizedForSpeed; // FIXME: Set GuardCfg when it is implemented. CurFn->FrameProcOpts = FPO; OS.EmitCVFuncIdDirective(CurFn->FuncId); // Find the end of the function prolog. First known non-DBG_VALUE and // non-frame setup location marks the beginning of the function body. // FIXME: is there a simpler a way to do this? Can we just search // for the first instruction of the function, not the last of the prolog? DebugLoc PrologEndLoc; bool EmptyPrologue = true; for (const auto &MBB : *MF) { for (const auto &MI : MBB) { if (!MI.isMetaInstruction() && !MI.getFlag(MachineInstr::FrameSetup) && MI.getDebugLoc()) { PrologEndLoc = MI.getDebugLoc(); break; } else if (!MI.isMetaInstruction()) { EmptyPrologue = false; } } } // Record beginning of function if we have a non-empty prologue. if (PrologEndLoc && !EmptyPrologue) { DebugLoc FnStartDL = PrologEndLoc.getFnDebugLoc(); maybeRecordLocation(FnStartDL, MF); } } static bool shouldEmitUdt(const DIType *T) { if (!T) return false; // MSVC does not emit UDTs for typedefs that are scoped to classes. if (T->getTag() == dwarf::DW_TAG_typedef) { if (DIScope *Scope = T->getScope().resolve()) { switch (Scope->getTag()) { case dwarf::DW_TAG_structure_type: case dwarf::DW_TAG_class_type: case dwarf::DW_TAG_union_type: return false; } } } while (true) { if (!T || T->isForwardDecl()) return false; const DIDerivedType *DT = dyn_cast(T); if (!DT) return true; T = DT->getBaseType().resolve(); } return true; } void CodeViewDebug::addToUDTs(const DIType *Ty) { // Don't record empty UDTs. if (Ty->getName().empty()) return; if (!shouldEmitUdt(Ty)) return; SmallVector QualifiedNameComponents; const DISubprogram *ClosestSubprogram = getQualifiedNameComponents( Ty->getScope().resolve(), QualifiedNameComponents); std::string FullyQualifiedName = getQualifiedName(QualifiedNameComponents, getPrettyScopeName(Ty)); if (ClosestSubprogram == nullptr) { GlobalUDTs.emplace_back(std::move(FullyQualifiedName), Ty); } else if (ClosestSubprogram == CurrentSubprogram) { LocalUDTs.emplace_back(std::move(FullyQualifiedName), Ty); } // TODO: What if the ClosestSubprogram is neither null or the current // subprogram? Currently, the UDT just gets dropped on the floor. // // The current behavior is not desirable. To get maximal fidelity, we would // need to perform all type translation before beginning emission of .debug$S // and then make LocalUDTs a member of FunctionInfo } TypeIndex CodeViewDebug::lowerType(const DIType *Ty, const DIType *ClassTy) { // Generic dispatch for lowering an unknown type. switch (Ty->getTag()) { case dwarf::DW_TAG_array_type: return lowerTypeArray(cast(Ty)); case dwarf::DW_TAG_typedef: return lowerTypeAlias(cast(Ty)); case dwarf::DW_TAG_base_type: return lowerTypeBasic(cast(Ty)); case dwarf::DW_TAG_pointer_type: if (cast(Ty)->getName() == "__vtbl_ptr_type") return lowerTypeVFTableShape(cast(Ty)); LLVM_FALLTHROUGH; case dwarf::DW_TAG_reference_type: case dwarf::DW_TAG_rvalue_reference_type: return lowerTypePointer(cast(Ty)); case dwarf::DW_TAG_ptr_to_member_type: return lowerTypeMemberPointer(cast(Ty)); case dwarf::DW_TAG_restrict_type: case dwarf::DW_TAG_const_type: case dwarf::DW_TAG_volatile_type: // TODO: add support for DW_TAG_atomic_type here return lowerTypeModifier(cast(Ty)); case dwarf::DW_TAG_subroutine_type: if (ClassTy) { // The member function type of a member function pointer has no // ThisAdjustment. return lowerTypeMemberFunction(cast(Ty), ClassTy, /*ThisAdjustment=*/0, /*IsStaticMethod=*/false); } return lowerTypeFunction(cast(Ty)); case dwarf::DW_TAG_enumeration_type: return lowerTypeEnum(cast(Ty)); case dwarf::DW_TAG_class_type: case dwarf::DW_TAG_structure_type: return lowerTypeClass(cast(Ty)); case dwarf::DW_TAG_union_type: return lowerTypeUnion(cast(Ty)); case dwarf::DW_TAG_unspecified_type: if (Ty->getName() == "decltype(nullptr)") return TypeIndex::NullptrT(); return TypeIndex::None(); default: // Use the null type index. return TypeIndex(); } } TypeIndex CodeViewDebug::lowerTypeAlias(const DIDerivedType *Ty) { DITypeRef UnderlyingTypeRef = Ty->getBaseType(); TypeIndex UnderlyingTypeIndex = getTypeIndex(UnderlyingTypeRef); StringRef TypeName = Ty->getName(); addToUDTs(Ty); if (UnderlyingTypeIndex == TypeIndex(SimpleTypeKind::Int32Long) && TypeName == "HRESULT") return TypeIndex(SimpleTypeKind::HResult); if (UnderlyingTypeIndex == TypeIndex(SimpleTypeKind::UInt16Short) && TypeName == "wchar_t") return TypeIndex(SimpleTypeKind::WideCharacter); return UnderlyingTypeIndex; } TypeIndex CodeViewDebug::lowerTypeArray(const DICompositeType *Ty) { DITypeRef ElementTypeRef = Ty->getBaseType(); TypeIndex ElementTypeIndex = getTypeIndex(ElementTypeRef); // IndexType is size_t, which depends on the bitness of the target. TypeIndex IndexType = getPointerSizeInBytes() == 8 ? TypeIndex(SimpleTypeKind::UInt64Quad) : TypeIndex(SimpleTypeKind::UInt32Long); uint64_t ElementSize = getBaseTypeSize(ElementTypeRef) / 8; // Add subranges to array type. DINodeArray Elements = Ty->getElements(); for (int i = Elements.size() - 1; i >= 0; --i) { const DINode *Element = Elements[i]; assert(Element->getTag() == dwarf::DW_TAG_subrange_type); const DISubrange *Subrange = cast(Element); assert(Subrange->getLowerBound() == 0 && "codeview doesn't support subranges with lower bounds"); int64_t Count = -1; if (auto *CI = Subrange->getCount().dyn_cast()) Count = CI->getSExtValue(); // Forward declarations of arrays without a size and VLAs use a count of -1. // Emit a count of zero in these cases to match what MSVC does for arrays // without a size. MSVC doesn't support VLAs, so it's not clear what we // should do for them even if we could distinguish them. if (Count == -1) Count = 0; // Update the element size and element type index for subsequent subranges. ElementSize *= Count; // If this is the outermost array, use the size from the array. It will be // more accurate if we had a VLA or an incomplete element type size. uint64_t ArraySize = (i == 0 && ElementSize == 0) ? Ty->getSizeInBits() / 8 : ElementSize; StringRef Name = (i == 0) ? Ty->getName() : ""; ArrayRecord AR(ElementTypeIndex, IndexType, ArraySize, Name); ElementTypeIndex = TypeTable.writeLeafType(AR); } return ElementTypeIndex; } TypeIndex CodeViewDebug::lowerTypeBasic(const DIBasicType *Ty) { TypeIndex Index; dwarf::TypeKind Kind; uint32_t ByteSize; Kind = static_cast(Ty->getEncoding()); ByteSize = Ty->getSizeInBits() / 8; SimpleTypeKind STK = SimpleTypeKind::None; switch (Kind) { case dwarf::DW_ATE_address: // FIXME: Translate break; case dwarf::DW_ATE_boolean: switch (ByteSize) { case 1: STK = SimpleTypeKind::Boolean8; break; case 2: STK = SimpleTypeKind::Boolean16; break; case 4: STK = SimpleTypeKind::Boolean32; break; case 8: STK = SimpleTypeKind::Boolean64; break; case 16: STK = SimpleTypeKind::Boolean128; break; } break; case dwarf::DW_ATE_complex_float: switch (ByteSize) { case 2: STK = SimpleTypeKind::Complex16; break; case 4: STK = SimpleTypeKind::Complex32; break; case 8: STK = SimpleTypeKind::Complex64; break; case 10: STK = SimpleTypeKind::Complex80; break; case 16: STK = SimpleTypeKind::Complex128; break; } break; case dwarf::DW_ATE_float: switch (ByteSize) { case 2: STK = SimpleTypeKind::Float16; break; case 4: STK = SimpleTypeKind::Float32; break; case 6: STK = SimpleTypeKind::Float48; break; case 8: STK = SimpleTypeKind::Float64; break; case 10: STK = SimpleTypeKind::Float80; break; case 16: STK = SimpleTypeKind::Float128; break; } break; case dwarf::DW_ATE_signed: switch (ByteSize) { case 1: STK = SimpleTypeKind::SignedCharacter; break; case 2: STK = SimpleTypeKind::Int16Short; break; case 4: STK = SimpleTypeKind::Int32; break; case 8: STK = SimpleTypeKind::Int64Quad; break; case 16: STK = SimpleTypeKind::Int128Oct; break; } break; case dwarf::DW_ATE_unsigned: switch (ByteSize) { case 1: STK = SimpleTypeKind::UnsignedCharacter; break; case 2: STK = SimpleTypeKind::UInt16Short; break; case 4: STK = SimpleTypeKind::UInt32; break; case 8: STK = SimpleTypeKind::UInt64Quad; break; case 16: STK = SimpleTypeKind::UInt128Oct; break; } break; case dwarf::DW_ATE_UTF: switch (ByteSize) { case 2: STK = SimpleTypeKind::Character16; break; case 4: STK = SimpleTypeKind::Character32; break; } break; case dwarf::DW_ATE_signed_char: if (ByteSize == 1) STK = SimpleTypeKind::SignedCharacter; break; case dwarf::DW_ATE_unsigned_char: if (ByteSize == 1) STK = SimpleTypeKind::UnsignedCharacter; break; default: break; } // Apply some fixups based on the source-level type name. if (STK == SimpleTypeKind::Int32 && Ty->getName() == "long int") STK = SimpleTypeKind::Int32Long; if (STK == SimpleTypeKind::UInt32 && Ty->getName() == "long unsigned int") STK = SimpleTypeKind::UInt32Long; if (STK == SimpleTypeKind::UInt16Short && (Ty->getName() == "wchar_t" || Ty->getName() == "__wchar_t")) STK = SimpleTypeKind::WideCharacter; if ((STK == SimpleTypeKind::SignedCharacter || STK == SimpleTypeKind::UnsignedCharacter) && Ty->getName() == "char") STK = SimpleTypeKind::NarrowCharacter; return TypeIndex(STK); } TypeIndex CodeViewDebug::lowerTypePointer(const DIDerivedType *Ty, PointerOptions PO) { TypeIndex PointeeTI = getTypeIndex(Ty->getBaseType()); // Pointers to simple types without any options can use SimpleTypeMode, rather // than having a dedicated pointer type record. if (PointeeTI.isSimple() && PO == PointerOptions::None && PointeeTI.getSimpleMode() == SimpleTypeMode::Direct && Ty->getTag() == dwarf::DW_TAG_pointer_type) { SimpleTypeMode Mode = Ty->getSizeInBits() == 64 ? SimpleTypeMode::NearPointer64 : SimpleTypeMode::NearPointer32; return TypeIndex(PointeeTI.getSimpleKind(), Mode); } PointerKind PK = Ty->getSizeInBits() == 64 ? PointerKind::Near64 : PointerKind::Near32; PointerMode PM = PointerMode::Pointer; switch (Ty->getTag()) { default: llvm_unreachable("not a pointer tag type"); case dwarf::DW_TAG_pointer_type: PM = PointerMode::Pointer; break; case dwarf::DW_TAG_reference_type: PM = PointerMode::LValueReference; break; case dwarf::DW_TAG_rvalue_reference_type: PM = PointerMode::RValueReference; break; } if (Ty->isObjectPointer()) PO |= PointerOptions::Const; PointerRecord PR(PointeeTI, PK, PM, PO, Ty->getSizeInBits() / 8); return TypeTable.writeLeafType(PR); } static PointerToMemberRepresentation translatePtrToMemberRep(unsigned SizeInBytes, bool IsPMF, unsigned Flags) { // SizeInBytes being zero generally implies that the member pointer type was // incomplete, which can happen if it is part of a function prototype. In this // case, use the unknown model instead of the general model. if (IsPMF) { switch (Flags & DINode::FlagPtrToMemberRep) { case 0: return SizeInBytes == 0 ? PointerToMemberRepresentation::Unknown : PointerToMemberRepresentation::GeneralFunction; case DINode::FlagSingleInheritance: return PointerToMemberRepresentation::SingleInheritanceFunction; case DINode::FlagMultipleInheritance: return PointerToMemberRepresentation::MultipleInheritanceFunction; case DINode::FlagVirtualInheritance: return PointerToMemberRepresentation::VirtualInheritanceFunction; } } else { switch (Flags & DINode::FlagPtrToMemberRep) { case 0: return SizeInBytes == 0 ? PointerToMemberRepresentation::Unknown : PointerToMemberRepresentation::GeneralData; case DINode::FlagSingleInheritance: return PointerToMemberRepresentation::SingleInheritanceData; case DINode::FlagMultipleInheritance: return PointerToMemberRepresentation::MultipleInheritanceData; case DINode::FlagVirtualInheritance: return PointerToMemberRepresentation::VirtualInheritanceData; } } llvm_unreachable("invalid ptr to member representation"); } TypeIndex CodeViewDebug::lowerTypeMemberPointer(const DIDerivedType *Ty, PointerOptions PO) { assert(Ty->getTag() == dwarf::DW_TAG_ptr_to_member_type); TypeIndex ClassTI = getTypeIndex(Ty->getClassType()); TypeIndex PointeeTI = getTypeIndex(Ty->getBaseType(), Ty->getClassType()); PointerKind PK = getPointerSizeInBytes() == 8 ? PointerKind::Near64 : PointerKind::Near32; bool IsPMF = isa(Ty->getBaseType()); PointerMode PM = IsPMF ? PointerMode::PointerToMemberFunction : PointerMode::PointerToDataMember; assert(Ty->getSizeInBits() / 8 <= 0xff && "pointer size too big"); uint8_t SizeInBytes = Ty->getSizeInBits() / 8; MemberPointerInfo MPI( ClassTI, translatePtrToMemberRep(SizeInBytes, IsPMF, Ty->getFlags())); PointerRecord PR(PointeeTI, PK, PM, PO, SizeInBytes, MPI); return TypeTable.writeLeafType(PR); } /// Given a DWARF calling convention, get the CodeView equivalent. If we don't /// have a translation, use the NearC convention. static CallingConvention dwarfCCToCodeView(unsigned DwarfCC) { switch (DwarfCC) { case dwarf::DW_CC_normal: return CallingConvention::NearC; case dwarf::DW_CC_BORLAND_msfastcall: return CallingConvention::NearFast; case dwarf::DW_CC_BORLAND_thiscall: return CallingConvention::ThisCall; case dwarf::DW_CC_BORLAND_stdcall: return CallingConvention::NearStdCall; case dwarf::DW_CC_BORLAND_pascal: return CallingConvention::NearPascal; case dwarf::DW_CC_LLVM_vectorcall: return CallingConvention::NearVector; } return CallingConvention::NearC; } TypeIndex CodeViewDebug::lowerTypeModifier(const DIDerivedType *Ty) { ModifierOptions Mods = ModifierOptions::None; PointerOptions PO = PointerOptions::None; bool IsModifier = true; const DIType *BaseTy = Ty; while (IsModifier && BaseTy) { // FIXME: Need to add DWARF tags for __unaligned and _Atomic switch (BaseTy->getTag()) { case dwarf::DW_TAG_const_type: Mods |= ModifierOptions::Const; PO |= PointerOptions::Const; break; case dwarf::DW_TAG_volatile_type: Mods |= ModifierOptions::Volatile; PO |= PointerOptions::Volatile; break; case dwarf::DW_TAG_restrict_type: // Only pointer types be marked with __restrict. There is no known flag // for __restrict in LF_MODIFIER records. PO |= PointerOptions::Restrict; break; default: IsModifier = false; break; } if (IsModifier) BaseTy = cast(BaseTy)->getBaseType().resolve(); } // Check if the inner type will use an LF_POINTER record. If so, the // qualifiers will go in the LF_POINTER record. This comes up for types like // 'int *const' and 'int *__restrict', not the more common cases like 'const // char *'. if (BaseTy) { switch (BaseTy->getTag()) { case dwarf::DW_TAG_pointer_type: case dwarf::DW_TAG_reference_type: case dwarf::DW_TAG_rvalue_reference_type: return lowerTypePointer(cast(BaseTy), PO); case dwarf::DW_TAG_ptr_to_member_type: return lowerTypeMemberPointer(cast(BaseTy), PO); default: break; } } TypeIndex ModifiedTI = getTypeIndex(BaseTy); // Return the base type index if there aren't any modifiers. For example, the // metadata could contain restrict wrappers around non-pointer types. if (Mods == ModifierOptions::None) return ModifiedTI; ModifierRecord MR(ModifiedTI, Mods); return TypeTable.writeLeafType(MR); } TypeIndex CodeViewDebug::lowerTypeFunction(const DISubroutineType *Ty) { SmallVector ReturnAndArgTypeIndices; for (DITypeRef ArgTypeRef : Ty->getTypeArray()) ReturnAndArgTypeIndices.push_back(getTypeIndex(ArgTypeRef)); // MSVC uses type none for variadic argument. if (ReturnAndArgTypeIndices.size() > 1 && ReturnAndArgTypeIndices.back() == TypeIndex::Void()) { ReturnAndArgTypeIndices.back() = TypeIndex::None(); } TypeIndex ReturnTypeIndex = TypeIndex::Void(); ArrayRef ArgTypeIndices = None; if (!ReturnAndArgTypeIndices.empty()) { auto ReturnAndArgTypesRef = makeArrayRef(ReturnAndArgTypeIndices); ReturnTypeIndex = ReturnAndArgTypesRef.front(); ArgTypeIndices = ReturnAndArgTypesRef.drop_front(); } ArgListRecord ArgListRec(TypeRecordKind::ArgList, ArgTypeIndices); TypeIndex ArgListIndex = TypeTable.writeLeafType(ArgListRec); CallingConvention CC = dwarfCCToCodeView(Ty->getCC()); FunctionOptions FO = getFunctionOptions(Ty); ProcedureRecord Procedure(ReturnTypeIndex, CC, FO, ArgTypeIndices.size(), ArgListIndex); return TypeTable.writeLeafType(Procedure); } TypeIndex CodeViewDebug::lowerTypeMemberFunction(const DISubroutineType *Ty, const DIType *ClassTy, int ThisAdjustment, bool IsStaticMethod, FunctionOptions FO) { // Lower the containing class type. TypeIndex ClassType = getTypeIndex(ClassTy); DITypeRefArray ReturnAndArgs = Ty->getTypeArray(); unsigned Index = 0; SmallVector ArgTypeIndices; - TypeIndex ReturnTypeIndex = getTypeIndex(ReturnAndArgs[Index++]); + TypeIndex ReturnTypeIndex = TypeIndex::Void(); + if (ReturnAndArgs.size() > Index) { + ReturnTypeIndex = getTypeIndex(ReturnAndArgs[Index++]); + } // If the first argument is a pointer type and this isn't a static method, // treat it as the special 'this' parameter, which is encoded separately from // the arguments. TypeIndex ThisTypeIndex; if (!IsStaticMethod && ReturnAndArgs.size() > Index) { if (const DIDerivedType *PtrTy = dyn_cast_or_null(ReturnAndArgs[Index].resolve())) { if (PtrTy->getTag() == dwarf::DW_TAG_pointer_type) { ThisTypeIndex = getTypeIndexForThisPtr(PtrTy, Ty); Index++; } } } while (Index < ReturnAndArgs.size()) ArgTypeIndices.push_back(getTypeIndex(ReturnAndArgs[Index++])); // MSVC uses type none for variadic argument. if (!ArgTypeIndices.empty() && ArgTypeIndices.back() == TypeIndex::Void()) ArgTypeIndices.back() = TypeIndex::None(); ArgListRecord ArgListRec(TypeRecordKind::ArgList, ArgTypeIndices); TypeIndex ArgListIndex = TypeTable.writeLeafType(ArgListRec); CallingConvention CC = dwarfCCToCodeView(Ty->getCC()); MemberFunctionRecord MFR(ReturnTypeIndex, ClassType, ThisTypeIndex, CC, FO, ArgTypeIndices.size(), ArgListIndex, ThisAdjustment); return TypeTable.writeLeafType(MFR); } TypeIndex CodeViewDebug::lowerTypeVFTableShape(const DIDerivedType *Ty) { unsigned VSlotCount = Ty->getSizeInBits() / (8 * Asm->MAI->getCodePointerSize()); SmallVector Slots(VSlotCount, VFTableSlotKind::Near); VFTableShapeRecord VFTSR(Slots); return TypeTable.writeLeafType(VFTSR); } static MemberAccess translateAccessFlags(unsigned RecordTag, unsigned Flags) { switch (Flags & DINode::FlagAccessibility) { case DINode::FlagPrivate: return MemberAccess::Private; case DINode::FlagPublic: return MemberAccess::Public; case DINode::FlagProtected: return MemberAccess::Protected; case 0: // If there was no explicit access control, provide the default for the tag. return RecordTag == dwarf::DW_TAG_class_type ? MemberAccess::Private : MemberAccess::Public; } llvm_unreachable("access flags are exclusive"); } static MethodOptions translateMethodOptionFlags(const DISubprogram *SP) { if (SP->isArtificial()) return MethodOptions::CompilerGenerated; // FIXME: Handle other MethodOptions. return MethodOptions::None; } static MethodKind translateMethodKindFlags(const DISubprogram *SP, bool Introduced) { if (SP->getFlags() & DINode::FlagStaticMember) return MethodKind::Static; switch (SP->getVirtuality()) { case dwarf::DW_VIRTUALITY_none: break; case dwarf::DW_VIRTUALITY_virtual: return Introduced ? MethodKind::IntroducingVirtual : MethodKind::Virtual; case dwarf::DW_VIRTUALITY_pure_virtual: return Introduced ? MethodKind::PureIntroducingVirtual : MethodKind::PureVirtual; default: llvm_unreachable("unhandled virtuality case"); } return MethodKind::Vanilla; } static TypeRecordKind getRecordKind(const DICompositeType *Ty) { switch (Ty->getTag()) { case dwarf::DW_TAG_class_type: return TypeRecordKind::Class; case dwarf::DW_TAG_structure_type: return TypeRecordKind::Struct; } llvm_unreachable("unexpected tag"); } /// Return ClassOptions that should be present on both the forward declaration /// and the defintion of a tag type. static ClassOptions getCommonClassOptions(const DICompositeType *Ty) { ClassOptions CO = ClassOptions::None; // MSVC always sets this flag, even for local types. Clang doesn't always // appear to give every type a linkage name, which may be problematic for us. // FIXME: Investigate the consequences of not following them here. if (!Ty->getIdentifier().empty()) CO |= ClassOptions::HasUniqueName; // Put the Nested flag on a type if it appears immediately inside a tag type. // Do not walk the scope chain. Do not attempt to compute ContainsNestedClass // here. That flag is only set on definitions, and not forward declarations. const DIScope *ImmediateScope = Ty->getScope().resolve(); if (ImmediateScope && isa(ImmediateScope)) CO |= ClassOptions::Nested; // Put the Scoped flag on function-local types. MSVC puts this flag for enum // type only when it has an immediate function scope. Clang never puts enums // inside DILexicalBlock scopes. Enum types, as generated by clang, are // always in function, class, or file scopes. if (Ty->getTag() == dwarf::DW_TAG_enumeration_type) { if (ImmediateScope && isa(ImmediateScope)) CO |= ClassOptions::Scoped; } else { for (const DIScope *Scope = ImmediateScope; Scope != nullptr; Scope = Scope->getScope().resolve()) { if (isa(Scope)) { CO |= ClassOptions::Scoped; break; } } } return CO; } void CodeViewDebug::addUDTSrcLine(const DIType *Ty, TypeIndex TI) { switch (Ty->getTag()) { case dwarf::DW_TAG_class_type: case dwarf::DW_TAG_structure_type: case dwarf::DW_TAG_union_type: case dwarf::DW_TAG_enumeration_type: break; default: return; } if (const auto *File = Ty->getFile()) { StringIdRecord SIDR(TypeIndex(0x0), getFullFilepath(File)); TypeIndex SIDI = TypeTable.writeLeafType(SIDR); UdtSourceLineRecord USLR(TI, SIDI, Ty->getLine()); TypeTable.writeLeafType(USLR); } } TypeIndex CodeViewDebug::lowerTypeEnum(const DICompositeType *Ty) { ClassOptions CO = getCommonClassOptions(Ty); TypeIndex FTI; unsigned EnumeratorCount = 0; if (Ty->isForwardDecl()) { CO |= ClassOptions::ForwardReference; } else { ContinuationRecordBuilder ContinuationBuilder; ContinuationBuilder.begin(ContinuationRecordKind::FieldList); for (const DINode *Element : Ty->getElements()) { // We assume that the frontend provides all members in source declaration // order, which is what MSVC does. if (auto *Enumerator = dyn_cast_or_null(Element)) { EnumeratorRecord ER(MemberAccess::Public, APSInt::getUnsigned(Enumerator->getValue()), Enumerator->getName()); ContinuationBuilder.writeMemberType(ER); EnumeratorCount++; } } FTI = TypeTable.insertRecord(ContinuationBuilder); } std::string FullName = getFullyQualifiedName(Ty); EnumRecord ER(EnumeratorCount, CO, FTI, FullName, Ty->getIdentifier(), getTypeIndex(Ty->getBaseType())); TypeIndex EnumTI = TypeTable.writeLeafType(ER); addUDTSrcLine(Ty, EnumTI); return EnumTI; } //===----------------------------------------------------------------------===// // ClassInfo //===----------------------------------------------------------------------===// struct llvm::ClassInfo { struct MemberInfo { const DIDerivedType *MemberTypeNode; uint64_t BaseOffset; }; // [MemberInfo] using MemberList = std::vector; using MethodsList = TinyPtrVector; // MethodName -> MethodsList using MethodsMap = MapVector; /// Base classes. std::vector Inheritance; /// Direct members. MemberList Members; // Direct overloaded methods gathered by name. MethodsMap Methods; TypeIndex VShapeTI; std::vector NestedTypes; }; void CodeViewDebug::clear() { assert(CurFn == nullptr); FileIdMap.clear(); FnDebugInfo.clear(); FileToFilepathMap.clear(); LocalUDTs.clear(); GlobalUDTs.clear(); TypeIndices.clear(); CompleteTypeIndices.clear(); ScopeGlobals.clear(); } void CodeViewDebug::collectMemberInfo(ClassInfo &Info, const DIDerivedType *DDTy) { if (!DDTy->getName().empty()) { Info.Members.push_back({DDTy, 0}); return; } // An unnamed member may represent a nested struct or union. Attempt to // interpret the unnamed member as a DICompositeType possibly wrapped in // qualifier types. Add all the indirect fields to the current record if that // succeeds, and drop the member if that fails. assert((DDTy->getOffsetInBits() % 8) == 0 && "Unnamed bitfield member!"); uint64_t Offset = DDTy->getOffsetInBits(); const DIType *Ty = DDTy->getBaseType().resolve(); bool FullyResolved = false; while (!FullyResolved) { switch (Ty->getTag()) { case dwarf::DW_TAG_const_type: case dwarf::DW_TAG_volatile_type: // FIXME: we should apply the qualifier types to the indirect fields // rather than dropping them. Ty = cast(Ty)->getBaseType().resolve(); break; default: FullyResolved = true; break; } } const DICompositeType *DCTy = dyn_cast(Ty); if (!DCTy) return; ClassInfo NestedInfo = collectClassInfo(DCTy); for (const ClassInfo::MemberInfo &IndirectField : NestedInfo.Members) Info.Members.push_back( {IndirectField.MemberTypeNode, IndirectField.BaseOffset + Offset}); } ClassInfo CodeViewDebug::collectClassInfo(const DICompositeType *Ty) { ClassInfo Info; // Add elements to structure type. DINodeArray Elements = Ty->getElements(); for (auto *Element : Elements) { // We assume that the frontend provides all members in source declaration // order, which is what MSVC does. if (!Element) continue; if (auto *SP = dyn_cast(Element)) { Info.Methods[SP->getRawName()].push_back(SP); } else if (auto *DDTy = dyn_cast(Element)) { if (DDTy->getTag() == dwarf::DW_TAG_member) { collectMemberInfo(Info, DDTy); } else if (DDTy->getTag() == dwarf::DW_TAG_inheritance) { Info.Inheritance.push_back(DDTy); } else if (DDTy->getTag() == dwarf::DW_TAG_pointer_type && DDTy->getName() == "__vtbl_ptr_type") { Info.VShapeTI = getTypeIndex(DDTy); } else if (DDTy->getTag() == dwarf::DW_TAG_typedef) { Info.NestedTypes.push_back(DDTy); } else if (DDTy->getTag() == dwarf::DW_TAG_friend) { // Ignore friend members. It appears that MSVC emitted info about // friends in the past, but modern versions do not. } } else if (auto *Composite = dyn_cast(Element)) { Info.NestedTypes.push_back(Composite); } // Skip other unrecognized kinds of elements. } return Info; } static bool shouldAlwaysEmitCompleteClassType(const DICompositeType *Ty) { // This routine is used by lowerTypeClass and lowerTypeUnion to determine // if a complete type should be emitted instead of a forward reference. return Ty->getName().empty() && Ty->getIdentifier().empty() && !Ty->isForwardDecl(); } TypeIndex CodeViewDebug::lowerTypeClass(const DICompositeType *Ty) { // Emit the complete type for unnamed structs. C++ classes with methods // which have a circular reference back to the class type are expected to // be named by the front-end and should not be "unnamed". C unnamed // structs should not have circular references. if (shouldAlwaysEmitCompleteClassType(Ty)) { // If this unnamed complete type is already in the process of being defined // then the description of the type is malformed and cannot be emitted // into CodeView correctly so report a fatal error. auto I = CompleteTypeIndices.find(Ty); if (I != CompleteTypeIndices.end() && I->second == TypeIndex()) report_fatal_error("cannot debug circular reference to unnamed type"); return getCompleteTypeIndex(Ty); } // First, construct the forward decl. Don't look into Ty to compute the // forward decl options, since it might not be available in all TUs. TypeRecordKind Kind = getRecordKind(Ty); ClassOptions CO = ClassOptions::ForwardReference | getCommonClassOptions(Ty); std::string FullName = getFullyQualifiedName(Ty); ClassRecord CR(Kind, 0, CO, TypeIndex(), TypeIndex(), TypeIndex(), 0, FullName, Ty->getIdentifier()); TypeIndex FwdDeclTI = TypeTable.writeLeafType(CR); if (!Ty->isForwardDecl()) DeferredCompleteTypes.push_back(Ty); return FwdDeclTI; } TypeIndex CodeViewDebug::lowerCompleteTypeClass(const DICompositeType *Ty) { // Construct the field list and complete type record. TypeRecordKind Kind = getRecordKind(Ty); ClassOptions CO = getCommonClassOptions(Ty); TypeIndex FieldTI; TypeIndex VShapeTI; unsigned FieldCount; bool ContainsNestedClass; std::tie(FieldTI, VShapeTI, FieldCount, ContainsNestedClass) = lowerRecordFieldList(Ty); if (ContainsNestedClass) CO |= ClassOptions::ContainsNestedClass; std::string FullName = getFullyQualifiedName(Ty); uint64_t SizeInBytes = Ty->getSizeInBits() / 8; ClassRecord CR(Kind, FieldCount, CO, FieldTI, TypeIndex(), VShapeTI, SizeInBytes, FullName, Ty->getIdentifier()); TypeIndex ClassTI = TypeTable.writeLeafType(CR); addUDTSrcLine(Ty, ClassTI); addToUDTs(Ty); return ClassTI; } TypeIndex CodeViewDebug::lowerTypeUnion(const DICompositeType *Ty) { // Emit the complete type for unnamed unions. if (shouldAlwaysEmitCompleteClassType(Ty)) return getCompleteTypeIndex(Ty); ClassOptions CO = ClassOptions::ForwardReference | getCommonClassOptions(Ty); std::string FullName = getFullyQualifiedName(Ty); UnionRecord UR(0, CO, TypeIndex(), 0, FullName, Ty->getIdentifier()); TypeIndex FwdDeclTI = TypeTable.writeLeafType(UR); if (!Ty->isForwardDecl()) DeferredCompleteTypes.push_back(Ty); return FwdDeclTI; } TypeIndex CodeViewDebug::lowerCompleteTypeUnion(const DICompositeType *Ty) { ClassOptions CO = ClassOptions::Sealed | getCommonClassOptions(Ty); TypeIndex FieldTI; unsigned FieldCount; bool ContainsNestedClass; std::tie(FieldTI, std::ignore, FieldCount, ContainsNestedClass) = lowerRecordFieldList(Ty); if (ContainsNestedClass) CO |= ClassOptions::ContainsNestedClass; uint64_t SizeInBytes = Ty->getSizeInBits() / 8; std::string FullName = getFullyQualifiedName(Ty); UnionRecord UR(FieldCount, CO, FieldTI, SizeInBytes, FullName, Ty->getIdentifier()); TypeIndex UnionTI = TypeTable.writeLeafType(UR); addUDTSrcLine(Ty, UnionTI); addToUDTs(Ty); return UnionTI; } std::tuple CodeViewDebug::lowerRecordFieldList(const DICompositeType *Ty) { // Manually count members. MSVC appears to count everything that generates a // field list record. Each individual overload in a method overload group // contributes to this count, even though the overload group is a single field // list record. unsigned MemberCount = 0; ClassInfo Info = collectClassInfo(Ty); ContinuationRecordBuilder ContinuationBuilder; ContinuationBuilder.begin(ContinuationRecordKind::FieldList); // Create base classes. for (const DIDerivedType *I : Info.Inheritance) { if (I->getFlags() & DINode::FlagVirtual) { // Virtual base. unsigned VBPtrOffset = I->getVBPtrOffset(); // FIXME: Despite the accessor name, the offset is really in bytes. unsigned VBTableIndex = I->getOffsetInBits() / 4; auto RecordKind = (I->getFlags() & DINode::FlagIndirectVirtualBase) == DINode::FlagIndirectVirtualBase ? TypeRecordKind::IndirectVirtualBaseClass : TypeRecordKind::VirtualBaseClass; VirtualBaseClassRecord VBCR( RecordKind, translateAccessFlags(Ty->getTag(), I->getFlags()), getTypeIndex(I->getBaseType()), getVBPTypeIndex(), VBPtrOffset, VBTableIndex); ContinuationBuilder.writeMemberType(VBCR); MemberCount++; } else { assert(I->getOffsetInBits() % 8 == 0 && "bases must be on byte boundaries"); BaseClassRecord BCR(translateAccessFlags(Ty->getTag(), I->getFlags()), getTypeIndex(I->getBaseType()), I->getOffsetInBits() / 8); ContinuationBuilder.writeMemberType(BCR); MemberCount++; } } // Create members. for (ClassInfo::MemberInfo &MemberInfo : Info.Members) { const DIDerivedType *Member = MemberInfo.MemberTypeNode; TypeIndex MemberBaseType = getTypeIndex(Member->getBaseType()); StringRef MemberName = Member->getName(); MemberAccess Access = translateAccessFlags(Ty->getTag(), Member->getFlags()); if (Member->isStaticMember()) { StaticDataMemberRecord SDMR(Access, MemberBaseType, MemberName); ContinuationBuilder.writeMemberType(SDMR); MemberCount++; continue; } // Virtual function pointer member. if ((Member->getFlags() & DINode::FlagArtificial) && Member->getName().startswith("_vptr$")) { VFPtrRecord VFPR(getTypeIndex(Member->getBaseType())); ContinuationBuilder.writeMemberType(VFPR); MemberCount++; continue; } // Data member. uint64_t MemberOffsetInBits = Member->getOffsetInBits() + MemberInfo.BaseOffset; if (Member->isBitField()) { uint64_t StartBitOffset = MemberOffsetInBits; if (const auto *CI = dyn_cast_or_null(Member->getStorageOffsetInBits())) { MemberOffsetInBits = CI->getZExtValue() + MemberInfo.BaseOffset; } StartBitOffset -= MemberOffsetInBits; BitFieldRecord BFR(MemberBaseType, Member->getSizeInBits(), StartBitOffset); MemberBaseType = TypeTable.writeLeafType(BFR); } uint64_t MemberOffsetInBytes = MemberOffsetInBits / 8; DataMemberRecord DMR(Access, MemberBaseType, MemberOffsetInBytes, MemberName); ContinuationBuilder.writeMemberType(DMR); MemberCount++; } // Create methods for (auto &MethodItr : Info.Methods) { StringRef Name = MethodItr.first->getString(); std::vector Methods; for (const DISubprogram *SP : MethodItr.second) { TypeIndex MethodType = getMemberFunctionType(SP, Ty); bool Introduced = SP->getFlags() & DINode::FlagIntroducedVirtual; unsigned VFTableOffset = -1; if (Introduced) VFTableOffset = SP->getVirtualIndex() * getPointerSizeInBytes(); Methods.push_back(OneMethodRecord( MethodType, translateAccessFlags(Ty->getTag(), SP->getFlags()), translateMethodKindFlags(SP, Introduced), translateMethodOptionFlags(SP), VFTableOffset, Name)); MemberCount++; } assert(!Methods.empty() && "Empty methods map entry"); if (Methods.size() == 1) ContinuationBuilder.writeMemberType(Methods[0]); else { // FIXME: Make this use its own ContinuationBuilder so that // MethodOverloadList can be split correctly. MethodOverloadListRecord MOLR(Methods); TypeIndex MethodList = TypeTable.writeLeafType(MOLR); OverloadedMethodRecord OMR(Methods.size(), MethodList, Name); ContinuationBuilder.writeMemberType(OMR); } } // Create nested classes. for (const DIType *Nested : Info.NestedTypes) { NestedTypeRecord R(getTypeIndex(DITypeRef(Nested)), Nested->getName()); ContinuationBuilder.writeMemberType(R); MemberCount++; } TypeIndex FieldTI = TypeTable.insertRecord(ContinuationBuilder); return std::make_tuple(FieldTI, Info.VShapeTI, MemberCount, !Info.NestedTypes.empty()); } TypeIndex CodeViewDebug::getVBPTypeIndex() { if (!VBPType.getIndex()) { // Make a 'const int *' type. ModifierRecord MR(TypeIndex::Int32(), ModifierOptions::Const); TypeIndex ModifiedTI = TypeTable.writeLeafType(MR); PointerKind PK = getPointerSizeInBytes() == 8 ? PointerKind::Near64 : PointerKind::Near32; PointerMode PM = PointerMode::Pointer; PointerOptions PO = PointerOptions::None; PointerRecord PR(ModifiedTI, PK, PM, PO, getPointerSizeInBytes()); VBPType = TypeTable.writeLeafType(PR); } return VBPType; } TypeIndex CodeViewDebug::getTypeIndex(DITypeRef TypeRef, DITypeRef ClassTyRef) { const DIType *Ty = TypeRef.resolve(); const DIType *ClassTy = ClassTyRef.resolve(); // The null DIType is the void type. Don't try to hash it. if (!Ty) return TypeIndex::Void(); // Check if we've already translated this type. Don't try to do a // get-or-create style insertion that caches the hash lookup across the // lowerType call. It will update the TypeIndices map. auto I = TypeIndices.find({Ty, ClassTy}); if (I != TypeIndices.end()) return I->second; TypeLoweringScope S(*this); TypeIndex TI = lowerType(Ty, ClassTy); return recordTypeIndexForDINode(Ty, TI, ClassTy); } codeview::TypeIndex CodeViewDebug::getTypeIndexForThisPtr(const DIDerivedType *PtrTy, const DISubroutineType *SubroutineTy) { assert(PtrTy->getTag() == dwarf::DW_TAG_pointer_type && "this type must be a pointer type"); PointerOptions Options = PointerOptions::None; if (SubroutineTy->getFlags() & DINode::DIFlags::FlagLValueReference) Options = PointerOptions::LValueRefThisPointer; else if (SubroutineTy->getFlags() & DINode::DIFlags::FlagRValueReference) Options = PointerOptions::RValueRefThisPointer; // Check if we've already translated this type. If there is no ref qualifier // on the function then we look up this pointer type with no associated class // so that the TypeIndex for the this pointer can be shared with the type // index for other pointers to this class type. If there is a ref qualifier // then we lookup the pointer using the subroutine as the parent type. auto I = TypeIndices.find({PtrTy, SubroutineTy}); if (I != TypeIndices.end()) return I->second; TypeLoweringScope S(*this); TypeIndex TI = lowerTypePointer(PtrTy, Options); return recordTypeIndexForDINode(PtrTy, TI, SubroutineTy); } TypeIndex CodeViewDebug::getTypeIndexForReferenceTo(DITypeRef TypeRef) { DIType *Ty = TypeRef.resolve(); PointerRecord PR(getTypeIndex(Ty), getPointerSizeInBytes() == 8 ? PointerKind::Near64 : PointerKind::Near32, PointerMode::LValueReference, PointerOptions::None, Ty->getSizeInBits() / 8); return TypeTable.writeLeafType(PR); } TypeIndex CodeViewDebug::getCompleteTypeIndex(DITypeRef TypeRef) { const DIType *Ty = TypeRef.resolve(); // The null DIType is the void type. Don't try to hash it. if (!Ty) return TypeIndex::Void(); // Look through typedefs when getting the complete type index. Call // getTypeIndex on the typdef to ensure that any UDTs are accumulated and are // emitted only once. if (Ty->getTag() == dwarf::DW_TAG_typedef) (void)getTypeIndex(Ty); while (Ty->getTag() == dwarf::DW_TAG_typedef) Ty = cast(Ty)->getBaseType().resolve(); // If this is a non-record type, the complete type index is the same as the // normal type index. Just call getTypeIndex. switch (Ty->getTag()) { case dwarf::DW_TAG_class_type: case dwarf::DW_TAG_structure_type: case dwarf::DW_TAG_union_type: break; default: return getTypeIndex(Ty); } // Check if we've already translated the complete record type. const auto *CTy = cast(Ty); auto InsertResult = CompleteTypeIndices.insert({CTy, TypeIndex()}); if (!InsertResult.second) return InsertResult.first->second; TypeLoweringScope S(*this); // Make sure the forward declaration is emitted first. It's unclear if this // is necessary, but MSVC does it, and we should follow suit until we can show // otherwise. // We only emit a forward declaration for named types. if (!CTy->getName().empty() || !CTy->getIdentifier().empty()) { TypeIndex FwdDeclTI = getTypeIndex(CTy); // Just use the forward decl if we don't have complete type info. This // might happen if the frontend is using modules and expects the complete // definition to be emitted elsewhere. if (CTy->isForwardDecl()) return FwdDeclTI; } TypeIndex TI; switch (CTy->getTag()) { case dwarf::DW_TAG_class_type: case dwarf::DW_TAG_structure_type: TI = lowerCompleteTypeClass(CTy); break; case dwarf::DW_TAG_union_type: TI = lowerCompleteTypeUnion(CTy); break; default: llvm_unreachable("not a record"); } // Update the type index associated with this CompositeType. This cannot // use the 'InsertResult' iterator above because it is potentially // invalidated by map insertions which can occur while lowering the class // type above. CompleteTypeIndices[CTy] = TI; return TI; } /// Emit all the deferred complete record types. Try to do this in FIFO order, /// and do this until fixpoint, as each complete record type typically /// references /// many other record types. void CodeViewDebug::emitDeferredCompleteTypes() { SmallVector TypesToEmit; while (!DeferredCompleteTypes.empty()) { std::swap(DeferredCompleteTypes, TypesToEmit); for (const DICompositeType *RecordTy : TypesToEmit) getCompleteTypeIndex(RecordTy); TypesToEmit.clear(); } } void CodeViewDebug::emitLocalVariableList(const FunctionInfo &FI, ArrayRef Locals) { // Get the sorted list of parameters and emit them first. SmallVector Params; for (const LocalVariable &L : Locals) if (L.DIVar->isParameter()) Params.push_back(&L); llvm::sort(Params, [](const LocalVariable *L, const LocalVariable *R) { return L->DIVar->getArg() < R->DIVar->getArg(); }); for (const LocalVariable *L : Params) emitLocalVariable(FI, *L); // Next emit all non-parameters in the order that we found them. for (const LocalVariable &L : Locals) if (!L.DIVar->isParameter()) emitLocalVariable(FI, L); } /// Only call this on endian-specific types like ulittle16_t and little32_t, or /// structs composed of them. template static void copyBytesForDefRange(SmallString<20> &BytePrefix, SymbolKind SymKind, const T &DefRangeHeader) { BytePrefix.resize(2 + sizeof(T)); ulittle16_t SymKindLE = ulittle16_t(SymKind); memcpy(&BytePrefix[0], &SymKindLE, 2); memcpy(&BytePrefix[2], &DefRangeHeader, sizeof(T)); } void CodeViewDebug::emitLocalVariable(const FunctionInfo &FI, const LocalVariable &Var) { // LocalSym record, see SymbolRecord.h for more info. MCSymbol *LocalEnd = beginSymbolRecord(SymbolKind::S_LOCAL); LocalSymFlags Flags = LocalSymFlags::None; if (Var.DIVar->isParameter()) Flags |= LocalSymFlags::IsParameter; if (Var.DefRanges.empty()) Flags |= LocalSymFlags::IsOptimizedOut; OS.AddComment("TypeIndex"); TypeIndex TI = Var.UseReferenceType ? getTypeIndexForReferenceTo(Var.DIVar->getType()) : getCompleteTypeIndex(Var.DIVar->getType()); OS.EmitIntValue(TI.getIndex(), 4); OS.AddComment("Flags"); OS.EmitIntValue(static_cast(Flags), 2); // Truncate the name so we won't overflow the record length field. emitNullTerminatedSymbolName(OS, Var.DIVar->getName()); endSymbolRecord(LocalEnd); // Calculate the on disk prefix of the appropriate def range record. The // records and on disk formats are described in SymbolRecords.h. BytePrefix // should be big enough to hold all forms without memory allocation. SmallString<20> BytePrefix; for (const LocalVarDefRange &DefRange : Var.DefRanges) { BytePrefix.clear(); if (DefRange.InMemory) { int Offset = DefRange.DataOffset; unsigned Reg = DefRange.CVRegister; // 32-bit x86 call sequences often use PUSH instructions, which disrupt // ESP-relative offsets. Use the virtual frame pointer, VFRAME or $T0, // instead. In frames without stack realignment, $T0 will be the CFA. if (RegisterId(Reg) == RegisterId::ESP) { Reg = unsigned(RegisterId::VFRAME); Offset += FI.OffsetAdjustment; } // If we can use the chosen frame pointer for the frame and this isn't a // sliced aggregate, use the smaller S_DEFRANGE_FRAMEPOINTER_REL record. // Otherwise, use S_DEFRANGE_REGISTER_REL. EncodedFramePtrReg EncFP = encodeFramePtrReg(RegisterId(Reg), TheCPU); if (!DefRange.IsSubfield && EncFP != EncodedFramePtrReg::None && (bool(Flags & LocalSymFlags::IsParameter) ? (EncFP == FI.EncodedParamFramePtrReg) : (EncFP == FI.EncodedLocalFramePtrReg))) { little32_t FPOffset = little32_t(Offset); copyBytesForDefRange(BytePrefix, S_DEFRANGE_FRAMEPOINTER_REL, FPOffset); } else { uint16_t RegRelFlags = 0; if (DefRange.IsSubfield) { RegRelFlags = DefRangeRegisterRelSym::IsSubfieldFlag | (DefRange.StructOffset << DefRangeRegisterRelSym::OffsetInParentShift); } DefRangeRegisterRelSym::Header DRHdr; DRHdr.Register = Reg; DRHdr.Flags = RegRelFlags; DRHdr.BasePointerOffset = Offset; copyBytesForDefRange(BytePrefix, S_DEFRANGE_REGISTER_REL, DRHdr); } } else { assert(DefRange.DataOffset == 0 && "unexpected offset into register"); if (DefRange.IsSubfield) { DefRangeSubfieldRegisterSym::Header DRHdr; DRHdr.Register = DefRange.CVRegister; DRHdr.MayHaveNoName = 0; DRHdr.OffsetInParent = DefRange.StructOffset; copyBytesForDefRange(BytePrefix, S_DEFRANGE_SUBFIELD_REGISTER, DRHdr); } else { DefRangeRegisterSym::Header DRHdr; DRHdr.Register = DefRange.CVRegister; DRHdr.MayHaveNoName = 0; copyBytesForDefRange(BytePrefix, S_DEFRANGE_REGISTER, DRHdr); } } OS.EmitCVDefRangeDirective(DefRange.Ranges, BytePrefix); } } void CodeViewDebug::emitLexicalBlockList(ArrayRef Blocks, const FunctionInfo& FI) { for (LexicalBlock *Block : Blocks) emitLexicalBlock(*Block, FI); } /// Emit an S_BLOCK32 and S_END record pair delimiting the contents of a /// lexical block scope. void CodeViewDebug::emitLexicalBlock(const LexicalBlock &Block, const FunctionInfo& FI) { MCSymbol *RecordEnd = beginSymbolRecord(SymbolKind::S_BLOCK32); OS.AddComment("PtrParent"); OS.EmitIntValue(0, 4); // PtrParent OS.AddComment("PtrEnd"); OS.EmitIntValue(0, 4); // PtrEnd OS.AddComment("Code size"); OS.emitAbsoluteSymbolDiff(Block.End, Block.Begin, 4); // Code Size OS.AddComment("Function section relative address"); OS.EmitCOFFSecRel32(Block.Begin, /*Offset=*/0); // Func Offset OS.AddComment("Function section index"); OS.EmitCOFFSectionIndex(FI.Begin); // Func Symbol OS.AddComment("Lexical block name"); emitNullTerminatedSymbolName(OS, Block.Name); // Name endSymbolRecord(RecordEnd); // Emit variables local to this lexical block. emitLocalVariableList(FI, Block.Locals); emitGlobalVariableList(Block.Globals); // Emit lexical blocks contained within this block. emitLexicalBlockList(Block.Children, FI); // Close the lexical block scope. emitEndSymbolRecord(SymbolKind::S_END); } /// Convenience routine for collecting lexical block information for a list /// of lexical scopes. void CodeViewDebug::collectLexicalBlockInfo( SmallVectorImpl &Scopes, SmallVectorImpl &Blocks, SmallVectorImpl &Locals, SmallVectorImpl &Globals) { for (LexicalScope *Scope : Scopes) collectLexicalBlockInfo(*Scope, Blocks, Locals, Globals); } /// Populate the lexical blocks and local variable lists of the parent with /// information about the specified lexical scope. void CodeViewDebug::collectLexicalBlockInfo( LexicalScope &Scope, SmallVectorImpl &ParentBlocks, SmallVectorImpl &ParentLocals, SmallVectorImpl &ParentGlobals) { if (Scope.isAbstractScope()) return; // Gather information about the lexical scope including local variables, // global variables, and address ranges. bool IgnoreScope = false; auto LI = ScopeVariables.find(&Scope); SmallVectorImpl *Locals = LI != ScopeVariables.end() ? &LI->second : nullptr; auto GI = ScopeGlobals.find(Scope.getScopeNode()); SmallVectorImpl *Globals = GI != ScopeGlobals.end() ? GI->second.get() : nullptr; const DILexicalBlock *DILB = dyn_cast(Scope.getScopeNode()); const SmallVectorImpl &Ranges = Scope.getRanges(); // Ignore lexical scopes which do not contain variables. if (!Locals && !Globals) IgnoreScope = true; // Ignore lexical scopes which are not lexical blocks. if (!DILB) IgnoreScope = true; // Ignore scopes which have too many address ranges to represent in the // current CodeView format or do not have a valid address range. // // For lexical scopes with multiple address ranges you may be tempted to // construct a single range covering every instruction where the block is // live and everything in between. Unfortunately, Visual Studio only // displays variables from the first matching lexical block scope. If the // first lexical block contains exception handling code or cold code which // is moved to the bottom of the routine creating a single range covering // nearly the entire routine, then it will hide all other lexical blocks // and the variables they contain. if (Ranges.size() != 1 || !getLabelAfterInsn(Ranges.front().second)) IgnoreScope = true; if (IgnoreScope) { // This scope can be safely ignored and eliminating it will reduce the // size of the debug information. Be sure to collect any variable and scope // information from the this scope or any of its children and collapse them // into the parent scope. if (Locals) ParentLocals.append(Locals->begin(), Locals->end()); if (Globals) ParentGlobals.append(Globals->begin(), Globals->end()); collectLexicalBlockInfo(Scope.getChildren(), ParentBlocks, ParentLocals, ParentGlobals); return; } // Create a new CodeView lexical block for this lexical scope. If we've // seen this DILexicalBlock before then the scope tree is malformed and // we can handle this gracefully by not processing it a second time. auto BlockInsertion = CurFn->LexicalBlocks.insert({DILB, LexicalBlock()}); if (!BlockInsertion.second) return; // Create a lexical block containing the variables and collect the the // lexical block information for the children. const InsnRange &Range = Ranges.front(); assert(Range.first && Range.second); LexicalBlock &Block = BlockInsertion.first->second; Block.Begin = getLabelBeforeInsn(Range.first); Block.End = getLabelAfterInsn(Range.second); assert(Block.Begin && "missing label for scope begin"); assert(Block.End && "missing label for scope end"); Block.Name = DILB->getName(); if (Locals) Block.Locals = std::move(*Locals); if (Globals) Block.Globals = std::move(*Globals); ParentBlocks.push_back(&Block); collectLexicalBlockInfo(Scope.getChildren(), Block.Children, Block.Locals, Block.Globals); } void CodeViewDebug::endFunctionImpl(const MachineFunction *MF) { const Function &GV = MF->getFunction(); assert(FnDebugInfo.count(&GV)); assert(CurFn == FnDebugInfo[&GV].get()); collectVariableInfo(GV.getSubprogram()); // Build the lexical block structure to emit for this routine. if (LexicalScope *CFS = LScopes.getCurrentFunctionScope()) collectLexicalBlockInfo(*CFS, CurFn->ChildBlocks, CurFn->Locals, CurFn->Globals); // Clear the scope and variable information from the map which will not be // valid after we have finished processing this routine. This also prepares // the map for the subsequent routine. ScopeVariables.clear(); // Don't emit anything if we don't have any line tables. // Thunks are compiler-generated and probably won't have source correlation. if (!CurFn->HaveLineInfo && !GV.getSubprogram()->isThunk()) { FnDebugInfo.erase(&GV); CurFn = nullptr; return; } CurFn->Annotations = MF->getCodeViewAnnotations(); CurFn->End = Asm->getFunctionEnd(); CurFn = nullptr; } void CodeViewDebug::beginInstruction(const MachineInstr *MI) { DebugHandlerBase::beginInstruction(MI); // Ignore DBG_VALUE and DBG_LABEL locations and function prologue. if (!Asm || !CurFn || MI->isDebugInstr() || MI->getFlag(MachineInstr::FrameSetup)) return; // If the first instruction of a new MBB has no location, find the first // instruction with a location and use that. DebugLoc DL = MI->getDebugLoc(); if (!DL && MI->getParent() != PrevInstBB) { for (const auto &NextMI : *MI->getParent()) { if (NextMI.isDebugInstr()) continue; DL = NextMI.getDebugLoc(); if (DL) break; } } PrevInstBB = MI->getParent(); // If we still don't have a debug location, don't record a location. if (!DL) return; maybeRecordLocation(DL, Asm->MF); } MCSymbol *CodeViewDebug::beginCVSubsection(DebugSubsectionKind Kind) { MCSymbol *BeginLabel = MMI->getContext().createTempSymbol(), *EndLabel = MMI->getContext().createTempSymbol(); OS.EmitIntValue(unsigned(Kind), 4); OS.AddComment("Subsection size"); OS.emitAbsoluteSymbolDiff(EndLabel, BeginLabel, 4); OS.EmitLabel(BeginLabel); return EndLabel; } void CodeViewDebug::endCVSubsection(MCSymbol *EndLabel) { OS.EmitLabel(EndLabel); // Every subsection must be aligned to a 4-byte boundary. OS.EmitValueToAlignment(4); } static StringRef getSymbolName(SymbolKind SymKind) { for (const EnumEntry &EE : getSymbolTypeNames()) if (EE.Value == SymKind) return EE.Name; return ""; } MCSymbol *CodeViewDebug::beginSymbolRecord(SymbolKind SymKind) { MCSymbol *BeginLabel = MMI->getContext().createTempSymbol(), *EndLabel = MMI->getContext().createTempSymbol(); OS.AddComment("Record length"); OS.emitAbsoluteSymbolDiff(EndLabel, BeginLabel, 2); OS.EmitLabel(BeginLabel); if (OS.isVerboseAsm()) OS.AddComment("Record kind: " + getSymbolName(SymKind)); OS.EmitIntValue(unsigned(SymKind), 2); return EndLabel; } void CodeViewDebug::endSymbolRecord(MCSymbol *SymEnd) { // MSVC does not pad out symbol records to four bytes, but LLVM does to avoid // an extra copy of every symbol record in LLD. This increases object file // size by less than 1% in the clang build, and is compatible with the Visual // C++ linker. OS.EmitValueToAlignment(4); OS.EmitLabel(SymEnd); } void CodeViewDebug::emitEndSymbolRecord(SymbolKind EndKind) { OS.AddComment("Record length"); OS.EmitIntValue(2, 2); if (OS.isVerboseAsm()) OS.AddComment("Record kind: " + getSymbolName(EndKind)); OS.EmitIntValue(unsigned(EndKind), 2); // Record Kind } void CodeViewDebug::emitDebugInfoForUDTs( ArrayRef> UDTs) { for (const auto &UDT : UDTs) { const DIType *T = UDT.second; assert(shouldEmitUdt(T)); MCSymbol *UDTRecordEnd = beginSymbolRecord(SymbolKind::S_UDT); OS.AddComment("Type"); OS.EmitIntValue(getCompleteTypeIndex(T).getIndex(), 4); emitNullTerminatedSymbolName(OS, UDT.first); endSymbolRecord(UDTRecordEnd); } } void CodeViewDebug::collectGlobalVariableInfo() { DenseMap GlobalMap; for (const GlobalVariable &GV : MMI->getModule()->globals()) { SmallVector GVEs; GV.getDebugInfo(GVEs); for (const auto *GVE : GVEs) GlobalMap[GVE] = &GV; } NamedMDNode *CUs = MMI->getModule()->getNamedMetadata("llvm.dbg.cu"); for (const MDNode *Node : CUs->operands()) { const auto *CU = cast(Node); for (const auto *GVE : CU->getGlobalVariables()) { const auto *GV = GlobalMap.lookup(GVE); if (!GV || GV->isDeclarationForLinker()) continue; const DIGlobalVariable *DIGV = GVE->getVariable(); DIScope *Scope = DIGV->getScope(); SmallVector *VariableList; if (Scope && isa(Scope)) { // Locate a global variable list for this scope, creating one if // necessary. auto Insertion = ScopeGlobals.insert( {Scope, std::unique_ptr()}); if (Insertion.second) Insertion.first->second = llvm::make_unique(); VariableList = Insertion.first->second.get(); } else if (GV->hasComdat()) // Emit this global variable into a COMDAT section. VariableList = &ComdatVariables; else // Emit this globla variable in a single global symbol section. VariableList = &GlobalVariables; CVGlobalVariable CVGV = {DIGV, GV}; VariableList->emplace_back(std::move(CVGV)); } } } void CodeViewDebug::emitDebugInfoForGlobals() { // First, emit all globals that are not in a comdat in a single symbol // substream. MSVC doesn't like it if the substream is empty, so only open // it if we have at least one global to emit. switchToDebugSectionForSymbol(nullptr); if (!GlobalVariables.empty()) { OS.AddComment("Symbol subsection for globals"); MCSymbol *EndLabel = beginCVSubsection(DebugSubsectionKind::Symbols); emitGlobalVariableList(GlobalVariables); endCVSubsection(EndLabel); } // Second, emit each global that is in a comdat into its own .debug$S // section along with its own symbol substream. for (const CVGlobalVariable &CVGV : ComdatVariables) { MCSymbol *GVSym = Asm->getSymbol(CVGV.GV); OS.AddComment("Symbol subsection for " + Twine(GlobalValue::dropLLVMManglingEscape(CVGV.GV->getName()))); switchToDebugSectionForSymbol(GVSym); MCSymbol *EndLabel = beginCVSubsection(DebugSubsectionKind::Symbols); // FIXME: emitDebugInfoForGlobal() doesn't handle DIExpressions. emitDebugInfoForGlobal(CVGV.DIGV, CVGV.GV, GVSym); endCVSubsection(EndLabel); } } void CodeViewDebug::emitDebugInfoForRetainedTypes() { NamedMDNode *CUs = MMI->getModule()->getNamedMetadata("llvm.dbg.cu"); for (const MDNode *Node : CUs->operands()) { for (auto *Ty : cast(Node)->getRetainedTypes()) { if (DIType *RT = dyn_cast(Ty)) { getTypeIndex(RT); // FIXME: Add to global/local DTU list. } } } } // Emit each global variable in the specified array. void CodeViewDebug::emitGlobalVariableList(ArrayRef Globals) { for (const CVGlobalVariable &CVGV : Globals) { MCSymbol *GVSym = Asm->getSymbol(CVGV.GV); // FIXME: emitDebugInfoForGlobal() doesn't handle DIExpressions. emitDebugInfoForGlobal(CVGV.DIGV, CVGV.GV, GVSym); } } void CodeViewDebug::emitDebugInfoForGlobal(const DIGlobalVariable *DIGV, const GlobalVariable *GV, MCSymbol *GVSym) { // DataSym record, see SymbolRecord.h for more info. Thread local data // happens to have the same format as global data. SymbolKind DataSym = GV->isThreadLocal() ? (DIGV->isLocalToUnit() ? SymbolKind::S_LTHREAD32 : SymbolKind::S_GTHREAD32) : (DIGV->isLocalToUnit() ? SymbolKind::S_LDATA32 : SymbolKind::S_GDATA32); MCSymbol *DataEnd = beginSymbolRecord(DataSym); OS.AddComment("Type"); OS.EmitIntValue(getCompleteTypeIndex(DIGV->getType()).getIndex(), 4); OS.AddComment("DataOffset"); OS.EmitCOFFSecRel32(GVSym, /*Offset=*/0); OS.AddComment("Segment"); OS.EmitCOFFSectionIndex(GVSym); OS.AddComment("Name"); const unsigned LengthOfDataRecord = 12; emitNullTerminatedSymbolName(OS, DIGV->getName(), LengthOfDataRecord); endSymbolRecord(DataEnd); } Index: vendor/llvm/dist-release_80/lib/CodeGen/AsmPrinter/DwarfDebug.cpp =================================================================== --- vendor/llvm/dist-release_80/lib/CodeGen/AsmPrinter/DwarfDebug.cpp (revision 343793) +++ vendor/llvm/dist-release_80/lib/CodeGen/AsmPrinter/DwarfDebug.cpp (revision 343794) @@ -1,2754 +1,2756 @@ //===- llvm/CodeGen/DwarfDebug.cpp - Dwarf Debug Framework ----------------===// // // The LLVM Compiler Infrastructure // // This file is distributed under the University of Illinois Open Source // License. See LICENSE.TXT for details. // //===----------------------------------------------------------------------===// // // This file contains support for writing dwarf debug info into asm files. // //===----------------------------------------------------------------------===// #include "DwarfDebug.h" #include "ByteStreamer.h" #include "DIEHash.h" #include "DebugLocEntry.h" #include "DebugLocStream.h" #include "DwarfCompileUnit.h" #include "DwarfExpression.h" #include "DwarfFile.h" #include "DwarfUnit.h" #include "llvm/ADT/APInt.h" #include "llvm/ADT/DenseMap.h" #include "llvm/ADT/DenseSet.h" #include "llvm/ADT/MapVector.h" #include "llvm/ADT/STLExtras.h" #include "llvm/ADT/SmallVector.h" #include "llvm/ADT/StringRef.h" #include "llvm/ADT/Triple.h" #include "llvm/ADT/Twine.h" #include "llvm/BinaryFormat/Dwarf.h" #include "llvm/CodeGen/AccelTable.h" #include "llvm/CodeGen/AsmPrinter.h" #include "llvm/CodeGen/DIE.h" #include "llvm/CodeGen/LexicalScopes.h" #include "llvm/CodeGen/MachineBasicBlock.h" #include "llvm/CodeGen/MachineFunction.h" #include "llvm/CodeGen/MachineInstr.h" #include "llvm/CodeGen/MachineModuleInfo.h" #include "llvm/CodeGen/MachineOperand.h" #include "llvm/CodeGen/TargetInstrInfo.h" #include "llvm/CodeGen/TargetRegisterInfo.h" #include "llvm/CodeGen/TargetSubtargetInfo.h" #include "llvm/IR/Constants.h" #include "llvm/IR/DebugInfoMetadata.h" #include "llvm/IR/DebugLoc.h" #include "llvm/IR/Function.h" #include "llvm/IR/GlobalVariable.h" #include "llvm/IR/Module.h" #include "llvm/MC/MCAsmInfo.h" #include "llvm/MC/MCContext.h" #include "llvm/MC/MCDwarf.h" #include "llvm/MC/MCSection.h" #include "llvm/MC/MCStreamer.h" #include "llvm/MC/MCSymbol.h" #include "llvm/MC/MCTargetOptions.h" #include "llvm/MC/MachineLocation.h" #include "llvm/MC/SectionKind.h" #include "llvm/Pass.h" #include "llvm/Support/Casting.h" #include "llvm/Support/CommandLine.h" #include "llvm/Support/Debug.h" #include "llvm/Support/ErrorHandling.h" #include "llvm/Support/MD5.h" #include "llvm/Support/MathExtras.h" #include "llvm/Support/Timer.h" #include "llvm/Support/raw_ostream.h" #include "llvm/Target/TargetLoweringObjectFile.h" #include "llvm/Target/TargetMachine.h" #include "llvm/Target/TargetOptions.h" #include #include #include #include #include #include #include #include using namespace llvm; #define DEBUG_TYPE "dwarfdebug" static cl::opt DisableDebugInfoPrinting("disable-debug-info-print", cl::Hidden, cl::desc("Disable debug info printing")); static cl::opt UseDwarfRangesBaseAddressSpecifier( "use-dwarf-ranges-base-address-specifier", cl::Hidden, cl::desc("Use base address specifiers in debug_ranges"), cl::init(false)); static cl::opt GenerateARangeSection("generate-arange-section", cl::Hidden, cl::desc("Generate dwarf aranges"), cl::init(false)); static cl::opt GenerateDwarfTypeUnits("generate-type-units", cl::Hidden, cl::desc("Generate DWARF4 type units."), cl::init(false)); static cl::opt SplitDwarfCrossCuReferences( "split-dwarf-cross-cu-references", cl::Hidden, cl::desc("Enable cross-cu references in DWO files"), cl::init(false)); enum DefaultOnOff { Default, Enable, Disable }; static cl::opt UnknownLocations( "use-unknown-locations", cl::Hidden, cl::desc("Make an absence of debug location information explicit."), cl::values(clEnumVal(Default, "At top of block or after label"), clEnumVal(Enable, "In all cases"), clEnumVal(Disable, "Never")), cl::init(Default)); static cl::opt AccelTables( "accel-tables", cl::Hidden, cl::desc("Output dwarf accelerator tables."), cl::values(clEnumValN(AccelTableKind::Default, "Default", "Default for platform"), clEnumValN(AccelTableKind::None, "Disable", "Disabled."), clEnumValN(AccelTableKind::Apple, "Apple", "Apple"), clEnumValN(AccelTableKind::Dwarf, "Dwarf", "DWARF")), cl::init(AccelTableKind::Default)); static cl::opt DwarfInlinedStrings("dwarf-inlined-strings", cl::Hidden, cl::desc("Use inlined strings rather than string section."), cl::values(clEnumVal(Default, "Default for platform"), clEnumVal(Enable, "Enabled"), clEnumVal(Disable, "Disabled")), cl::init(Default)); static cl::opt NoDwarfRangesSection("no-dwarf-ranges-section", cl::Hidden, cl::desc("Disable emission .debug_ranges section."), cl::init(false)); static cl::opt DwarfSectionsAsReferences( "dwarf-sections-as-references", cl::Hidden, cl::desc("Use sections+offset as references rather than labels."), cl::values(clEnumVal(Default, "Default for platform"), clEnumVal(Enable, "Enabled"), clEnumVal(Disable, "Disabled")), cl::init(Default)); enum LinkageNameOption { DefaultLinkageNames, AllLinkageNames, AbstractLinkageNames }; static cl::opt DwarfLinkageNames("dwarf-linkage-names", cl::Hidden, cl::desc("Which DWARF linkage-name attributes to emit."), cl::values(clEnumValN(DefaultLinkageNames, "Default", "Default for platform"), clEnumValN(AllLinkageNames, "All", "All"), clEnumValN(AbstractLinkageNames, "Abstract", "Abstract subprograms")), cl::init(DefaultLinkageNames)); static const char *const DWARFGroupName = "dwarf"; static const char *const DWARFGroupDescription = "DWARF Emission"; static const char *const DbgTimerName = "writer"; static const char *const DbgTimerDescription = "DWARF Debug Writer"; void DebugLocDwarfExpression::emitOp(uint8_t Op, const char *Comment) { BS.EmitInt8( Op, Comment ? Twine(Comment) + " " + dwarf::OperationEncodingString(Op) : dwarf::OperationEncodingString(Op)); } void DebugLocDwarfExpression::emitSigned(int64_t Value) { BS.EmitSLEB128(Value, Twine(Value)); } void DebugLocDwarfExpression::emitUnsigned(uint64_t Value) { BS.EmitULEB128(Value, Twine(Value)); } bool DebugLocDwarfExpression::isFrameRegister(const TargetRegisterInfo &TRI, unsigned MachineReg) { // This information is not available while emitting .debug_loc entries. return false; } bool DbgVariable::isBlockByrefVariable() const { assert(getVariable() && "Invalid complex DbgVariable!"); return getVariable()->getType().resolve()->isBlockByrefStruct(); } const DIType *DbgVariable::getType() const { DIType *Ty = getVariable()->getType().resolve(); // FIXME: isBlockByrefVariable should be reformulated in terms of complex // addresses instead. if (Ty->isBlockByrefStruct()) { /* Byref variables, in Blocks, are declared by the programmer as "SomeType VarName;", but the compiler creates a __Block_byref_x_VarName struct, and gives the variable VarName either the struct, or a pointer to the struct, as its type. This is necessary for various behind-the-scenes things the compiler needs to do with by-reference variables in blocks. However, as far as the original *programmer* is concerned, the variable should still have type 'SomeType', as originally declared. The following function dives into the __Block_byref_x_VarName struct to find the original type of the variable. This will be passed back to the code generating the type for the Debug Information Entry for the variable 'VarName'. 'VarName' will then have the original type 'SomeType' in its debug information. The original type 'SomeType' will be the type of the field named 'VarName' inside the __Block_byref_x_VarName struct. NOTE: In order for this to not completely fail on the debugger side, the Debug Information Entry for the variable VarName needs to have a DW_AT_location that tells the debugger how to unwind through the pointers and __Block_byref_x_VarName struct to find the actual value of the variable. The function addBlockByrefType does this. */ DIType *subType = Ty; uint16_t tag = Ty->getTag(); if (tag == dwarf::DW_TAG_pointer_type) subType = resolve(cast(Ty)->getBaseType()); auto Elements = cast(subType)->getElements(); for (unsigned i = 0, N = Elements.size(); i < N; ++i) { auto *DT = cast(Elements[i]); if (getName() == DT->getName()) return resolve(DT->getBaseType()); } } return Ty; } ArrayRef DbgVariable::getFrameIndexExprs() const { if (FrameIndexExprs.size() == 1) return FrameIndexExprs; assert(llvm::all_of(FrameIndexExprs, [](const FrameIndexExpr &A) { return A.Expr->isFragment(); }) && "multiple FI expressions without DW_OP_LLVM_fragment"); llvm::sort(FrameIndexExprs, [](const FrameIndexExpr &A, const FrameIndexExpr &B) -> bool { return A.Expr->getFragmentInfo()->OffsetInBits < B.Expr->getFragmentInfo()->OffsetInBits; }); return FrameIndexExprs; } void DbgVariable::addMMIEntry(const DbgVariable &V) { assert(DebugLocListIndex == ~0U && !MInsn && "not an MMI entry"); assert(V.DebugLocListIndex == ~0U && !V.MInsn && "not an MMI entry"); assert(V.getVariable() == getVariable() && "conflicting variable"); assert(V.getInlinedAt() == getInlinedAt() && "conflicting inlined-at location"); assert(!FrameIndexExprs.empty() && "Expected an MMI entry"); assert(!V.FrameIndexExprs.empty() && "Expected an MMI entry"); // FIXME: This logic should not be necessary anymore, as we now have proper // deduplication. However, without it, we currently run into the assertion // below, which means that we are likely dealing with broken input, i.e. two // non-fragment entries for the same variable at different frame indices. if (FrameIndexExprs.size()) { auto *Expr = FrameIndexExprs.back().Expr; if (!Expr || !Expr->isFragment()) return; } for (const auto &FIE : V.FrameIndexExprs) // Ignore duplicate entries. if (llvm::none_of(FrameIndexExprs, [&](const FrameIndexExpr &Other) { return FIE.FI == Other.FI && FIE.Expr == Other.Expr; })) FrameIndexExprs.push_back(FIE); assert((FrameIndexExprs.size() == 1 || llvm::all_of(FrameIndexExprs, [](FrameIndexExpr &FIE) { return FIE.Expr && FIE.Expr->isFragment(); })) && "conflicting locations for variable"); } static AccelTableKind computeAccelTableKind(unsigned DwarfVersion, bool GenerateTypeUnits, DebuggerKind Tuning, const Triple &TT) { // Honor an explicit request. if (AccelTables != AccelTableKind::Default) return AccelTables; // Accelerator tables with type units are currently not supported. if (GenerateTypeUnits) return AccelTableKind::None; // Accelerator tables get emitted if targetting DWARF v5 or LLDB. DWARF v5 // always implies debug_names. For lower standard versions we use apple // accelerator tables on apple platforms and debug_names elsewhere. if (DwarfVersion >= 5) return AccelTableKind::Dwarf; if (Tuning == DebuggerKind::LLDB) return TT.isOSBinFormatMachO() ? AccelTableKind::Apple : AccelTableKind::Dwarf; return AccelTableKind::None; } DwarfDebug::DwarfDebug(AsmPrinter *A, Module *M) : DebugHandlerBase(A), DebugLocs(A->OutStreamer->isVerboseAsm()), InfoHolder(A, "info_string", DIEValueAllocator), SkeletonHolder(A, "skel_string", DIEValueAllocator), IsDarwin(A->TM.getTargetTriple().isOSDarwin()) { const Triple &TT = Asm->TM.getTargetTriple(); // Make sure we know our "debugger tuning." The target option takes // precedence; fall back to triple-based defaults. if (Asm->TM.Options.DebuggerTuning != DebuggerKind::Default) DebuggerTuning = Asm->TM.Options.DebuggerTuning; else if (IsDarwin) DebuggerTuning = DebuggerKind::LLDB; else if (TT.isPS4CPU()) DebuggerTuning = DebuggerKind::SCE; else DebuggerTuning = DebuggerKind::GDB; if (DwarfInlinedStrings == Default) UseInlineStrings = TT.isNVPTX(); else UseInlineStrings = DwarfInlinedStrings == Enable; UseLocSection = !TT.isNVPTX(); HasAppleExtensionAttributes = tuneForLLDB(); // Handle split DWARF. HasSplitDwarf = !Asm->TM.Options.MCOptions.SplitDwarfFile.empty(); // SCE defaults to linkage names only for abstract subprograms. if (DwarfLinkageNames == DefaultLinkageNames) UseAllLinkageNames = !tuneForSCE(); else UseAllLinkageNames = DwarfLinkageNames == AllLinkageNames; unsigned DwarfVersionNumber = Asm->TM.Options.MCOptions.DwarfVersion; unsigned DwarfVersion = DwarfVersionNumber ? DwarfVersionNumber : MMI->getModule()->getDwarfVersion(); // Use dwarf 4 by default if nothing is requested. For NVPTX, use dwarf 2. DwarfVersion = TT.isNVPTX() ? 2 : (DwarfVersion ? DwarfVersion : dwarf::DWARF_VERSION); UseRangesSection = !NoDwarfRangesSection && !TT.isNVPTX(); // Use sections as references. Force for NVPTX. if (DwarfSectionsAsReferences == Default) UseSectionsAsReferences = TT.isNVPTX(); else UseSectionsAsReferences = DwarfSectionsAsReferences == Enable; // Don't generate type units for unsupported object file formats. GenerateTypeUnits = A->TM.getTargetTriple().isOSBinFormatELF() && GenerateDwarfTypeUnits; TheAccelTableKind = computeAccelTableKind( DwarfVersion, GenerateTypeUnits, DebuggerTuning, A->TM.getTargetTriple()); // Work around a GDB bug. GDB doesn't support the standard opcode; // SCE doesn't support GNU's; LLDB prefers the standard opcode, which // is defined as of DWARF 3. // See GDB bug 11616 - DW_OP_form_tls_address is unimplemented // https://sourceware.org/bugzilla/show_bug.cgi?id=11616 UseGNUTLSOpcode = tuneForGDB() || DwarfVersion < 3; // GDB does not fully support the DWARF 4 representation for bitfields. UseDWARF2Bitfields = (DwarfVersion < 4) || tuneForGDB(); // The DWARF v5 string offsets table has - possibly shared - contributions // from each compile and type unit each preceded by a header. The string // offsets table used by the pre-DWARF v5 split-DWARF implementation uses // a monolithic string offsets table without any header. UseSegmentedStringOffsetsTable = DwarfVersion >= 5; Asm->OutStreamer->getContext().setDwarfVersion(DwarfVersion); } // Define out of line so we don't have to include DwarfUnit.h in DwarfDebug.h. DwarfDebug::~DwarfDebug() = default; static bool isObjCClass(StringRef Name) { return Name.startswith("+") || Name.startswith("-"); } static bool hasObjCCategory(StringRef Name) { if (!isObjCClass(Name)) return false; return Name.find(") ") != StringRef::npos; } static void getObjCClassCategory(StringRef In, StringRef &Class, StringRef &Category) { if (!hasObjCCategory(In)) { Class = In.slice(In.find('[') + 1, In.find(' ')); Category = ""; return; } Class = In.slice(In.find('[') + 1, In.find('(')); Category = In.slice(In.find('[') + 1, In.find(' ')); } static StringRef getObjCMethodName(StringRef In) { return In.slice(In.find(' ') + 1, In.find(']')); } // Add the various names to the Dwarf accelerator table names. void DwarfDebug::addSubprogramNames(const DICompileUnit &CU, const DISubprogram *SP, DIE &Die) { if (getAccelTableKind() != AccelTableKind::Apple && CU.getNameTableKind() == DICompileUnit::DebugNameTableKind::None) return; if (!SP->isDefinition()) return; if (SP->getName() != "") addAccelName(CU, SP->getName(), Die); // If the linkage name is different than the name, go ahead and output that as // well into the name table. Only do that if we are going to actually emit // that name. if (SP->getLinkageName() != "" && SP->getName() != SP->getLinkageName() && (useAllLinkageNames() || InfoHolder.getAbstractSPDies().lookup(SP))) addAccelName(CU, SP->getLinkageName(), Die); // If this is an Objective-C selector name add it to the ObjC accelerator // too. if (isObjCClass(SP->getName())) { StringRef Class, Category; getObjCClassCategory(SP->getName(), Class, Category); addAccelObjC(CU, Class, Die); if (Category != "") addAccelObjC(CU, Category, Die); // Also add the base method name to the name table. addAccelName(CU, getObjCMethodName(SP->getName()), Die); } } /// Check whether we should create a DIE for the given Scope, return true /// if we don't create a DIE (the corresponding DIE is null). bool DwarfDebug::isLexicalScopeDIENull(LexicalScope *Scope) { if (Scope->isAbstractScope()) return false; // We don't create a DIE if there is no Range. const SmallVectorImpl &Ranges = Scope->getRanges(); if (Ranges.empty()) return true; if (Ranges.size() > 1) return false; // We don't create a DIE if we have a single Range and the end label // is null. return !getLabelAfterInsn(Ranges.front().second); } template static void forBothCUs(DwarfCompileUnit &CU, Func F) { F(CU); if (auto *SkelCU = CU.getSkeleton()) if (CU.getCUNode()->getSplitDebugInlining()) F(*SkelCU); } bool DwarfDebug::shareAcrossDWOCUs() const { return SplitDwarfCrossCuReferences; } void DwarfDebug::constructAbstractSubprogramScopeDIE(DwarfCompileUnit &SrcCU, LexicalScope *Scope) { assert(Scope && Scope->getScopeNode()); assert(Scope->isAbstractScope()); assert(!Scope->getInlinedAt()); auto *SP = cast(Scope->getScopeNode()); // Find the subprogram's DwarfCompileUnit in the SPMap in case the subprogram // was inlined from another compile unit. if (useSplitDwarf() && !shareAcrossDWOCUs() && !SP->getUnit()->getSplitDebugInlining()) // Avoid building the original CU if it won't be used SrcCU.constructAbstractSubprogramScopeDIE(Scope); else { auto &CU = getOrCreateDwarfCompileUnit(SP->getUnit()); if (auto *SkelCU = CU.getSkeleton()) { (shareAcrossDWOCUs() ? CU : SrcCU) .constructAbstractSubprogramScopeDIE(Scope); if (CU.getCUNode()->getSplitDebugInlining()) SkelCU->constructAbstractSubprogramScopeDIE(Scope); } else CU.constructAbstractSubprogramScopeDIE(Scope); } } void DwarfDebug::constructCallSiteEntryDIEs(const DISubprogram &SP, DwarfCompileUnit &CU, DIE &ScopeDIE, const MachineFunction &MF) { // Add a call site-related attribute (DWARF5, Sec. 3.3.1.3). Do this only if // the subprogram is required to have one. if (!SP.areAllCallsDescribed() || !SP.isDefinition()) return; // Use DW_AT_call_all_calls to express that call site entries are present // for both tail and non-tail calls. Don't use DW_AT_call_all_source_calls // because one of its requirements is not met: call site entries for // optimized-out calls are elided. CU.addFlag(ScopeDIE, dwarf::DW_AT_call_all_calls); const TargetInstrInfo *TII = MF.getSubtarget().getInstrInfo(); assert(TII && "TargetInstrInfo not found: cannot label tail calls"); // Emit call site entries for each call or tail call in the function. for (const MachineBasicBlock &MBB : MF) { for (const MachineInstr &MI : MBB.instrs()) { // Skip instructions which aren't calls. Both calls and tail-calling jump // instructions (e.g TAILJMPd64) are classified correctly here. if (!MI.isCall()) continue; // TODO: Add support for targets with delay slots (see: beginInstruction). if (MI.hasDelaySlot()) return; // If this is a direct call, find the callee's subprogram. const MachineOperand &CalleeOp = MI.getOperand(0); if (!CalleeOp.isGlobal()) continue; const Function *CalleeDecl = dyn_cast(CalleeOp.getGlobal()); if (!CalleeDecl || !CalleeDecl->getSubprogram()) continue; // TODO: Omit call site entries for runtime calls (objc_msgSend, etc). // TODO: Add support for indirect calls. bool IsTail = TII->isTailCall(MI); // For tail calls, no return PC information is needed. For regular calls, // the return PC is needed to disambiguate paths in the call graph which // could lead to some target function. const MCExpr *PCOffset = IsTail ? nullptr : getFunctionLocalOffsetAfterInsn(&MI); assert((IsTail || PCOffset) && "Call without return PC information"); LLVM_DEBUG(dbgs() << "CallSiteEntry: " << MF.getName() << " -> " << CalleeDecl->getName() << (IsTail ? " [tail]" : "") << "\n"); CU.constructCallSiteEntryDIE(ScopeDIE, *CalleeDecl->getSubprogram(), IsTail, PCOffset); } } } void DwarfDebug::addGnuPubAttributes(DwarfCompileUnit &U, DIE &D) const { if (!U.hasDwarfPubSections()) return; U.addFlag(D, dwarf::DW_AT_GNU_pubnames); } void DwarfDebug::finishUnitAttributes(const DICompileUnit *DIUnit, DwarfCompileUnit &NewCU) { DIE &Die = NewCU.getUnitDie(); StringRef FN = DIUnit->getFilename(); StringRef Producer = DIUnit->getProducer(); StringRef Flags = DIUnit->getFlags(); if (!Flags.empty() && !useAppleExtensionAttributes()) { std::string ProducerWithFlags = Producer.str() + " " + Flags.str(); NewCU.addString(Die, dwarf::DW_AT_producer, ProducerWithFlags); } else NewCU.addString(Die, dwarf::DW_AT_producer, Producer); NewCU.addUInt(Die, dwarf::DW_AT_language, dwarf::DW_FORM_data2, DIUnit->getSourceLanguage()); NewCU.addString(Die, dwarf::DW_AT_name, FN); // Add DW_str_offsets_base to the unit DIE, except for split units. if (useSegmentedStringOffsetsTable() && !useSplitDwarf()) NewCU.addStringOffsetsStart(); if (!useSplitDwarf()) { NewCU.initStmtList(); // If we're using split dwarf the compilation dir is going to be in the // skeleton CU and so we don't need to duplicate it here. if (!CompilationDir.empty()) NewCU.addString(Die, dwarf::DW_AT_comp_dir, CompilationDir); addGnuPubAttributes(NewCU, Die); } if (useAppleExtensionAttributes()) { if (DIUnit->isOptimized()) NewCU.addFlag(Die, dwarf::DW_AT_APPLE_optimized); StringRef Flags = DIUnit->getFlags(); if (!Flags.empty()) NewCU.addString(Die, dwarf::DW_AT_APPLE_flags, Flags); if (unsigned RVer = DIUnit->getRuntimeVersion()) NewCU.addUInt(Die, dwarf::DW_AT_APPLE_major_runtime_vers, dwarf::DW_FORM_data1, RVer); } if (DIUnit->getDWOId()) { // This CU is either a clang module DWO or a skeleton CU. NewCU.addUInt(Die, dwarf::DW_AT_GNU_dwo_id, dwarf::DW_FORM_data8, DIUnit->getDWOId()); if (!DIUnit->getSplitDebugFilename().empty()) // This is a prefabricated skeleton CU. NewCU.addString(Die, dwarf::DW_AT_GNU_dwo_name, DIUnit->getSplitDebugFilename()); } } // Create new DwarfCompileUnit for the given metadata node with tag // DW_TAG_compile_unit. DwarfCompileUnit & DwarfDebug::getOrCreateDwarfCompileUnit(const DICompileUnit *DIUnit) { if (auto *CU = CUMap.lookup(DIUnit)) return *CU; CompilationDir = DIUnit->getDirectory(); auto OwnedUnit = llvm::make_unique( InfoHolder.getUnits().size(), DIUnit, Asm, this, &InfoHolder); DwarfCompileUnit &NewCU = *OwnedUnit; InfoHolder.addUnit(std::move(OwnedUnit)); for (auto *IE : DIUnit->getImportedEntities()) NewCU.addImportedEntity(IE); // LTO with assembly output shares a single line table amongst multiple CUs. // To avoid the compilation directory being ambiguous, let the line table // explicitly describe the directory of all files, never relying on the // compilation directory. if (!Asm->OutStreamer->hasRawTextSupport() || SingleCU) Asm->OutStreamer->emitDwarfFile0Directive( CompilationDir, DIUnit->getFilename(), NewCU.getMD5AsBytes(DIUnit->getFile()), DIUnit->getSource(), NewCU.getUniqueID()); if (useSplitDwarf()) { NewCU.setSkeleton(constructSkeletonCU(NewCU)); NewCU.setSection(Asm->getObjFileLowering().getDwarfInfoDWOSection()); } else { finishUnitAttributes(DIUnit, NewCU); NewCU.setSection(Asm->getObjFileLowering().getDwarfInfoSection()); } CUMap.insert({DIUnit, &NewCU}); CUDieMap.insert({&NewCU.getUnitDie(), &NewCU}); return NewCU; } void DwarfDebug::constructAndAddImportedEntityDIE(DwarfCompileUnit &TheCU, const DIImportedEntity *N) { if (isa(N->getScope())) return; if (DIE *D = TheCU.getOrCreateContextDIE(N->getScope())) D->addChild(TheCU.constructImportedEntityDIE(N)); } /// Sort and unique GVEs by comparing their fragment offset. static SmallVectorImpl & sortGlobalExprs(SmallVectorImpl &GVEs) { llvm::sort( GVEs, [](DwarfCompileUnit::GlobalExpr A, DwarfCompileUnit::GlobalExpr B) { // Sort order: first null exprs, then exprs without fragment // info, then sort by fragment offset in bits. // FIXME: Come up with a more comprehensive comparator so // the sorting isn't non-deterministic, and so the following // std::unique call works correctly. if (!A.Expr || !B.Expr) return !!B.Expr; auto FragmentA = A.Expr->getFragmentInfo(); auto FragmentB = B.Expr->getFragmentInfo(); if (!FragmentA || !FragmentB) return !!FragmentB; return FragmentA->OffsetInBits < FragmentB->OffsetInBits; }); GVEs.erase(std::unique(GVEs.begin(), GVEs.end(), [](DwarfCompileUnit::GlobalExpr A, DwarfCompileUnit::GlobalExpr B) { return A.Expr == B.Expr; }), GVEs.end()); return GVEs; } // Emit all Dwarf sections that should come prior to the content. Create // global DIEs and emit initial debug info sections. This is invoked by // the target AsmPrinter. void DwarfDebug::beginModule() { NamedRegionTimer T(DbgTimerName, DbgTimerDescription, DWARFGroupName, DWARFGroupDescription, TimePassesIsEnabled); if (DisableDebugInfoPrinting) { MMI->setDebugInfoAvailability(false); return; } const Module *M = MMI->getModule(); unsigned NumDebugCUs = std::distance(M->debug_compile_units_begin(), M->debug_compile_units_end()); // Tell MMI whether we have debug info. assert(MMI->hasDebugInfo() == (NumDebugCUs > 0) && "DebugInfoAvailabilty initialized unexpectedly"); SingleCU = NumDebugCUs == 1; DenseMap> GVMap; for (const GlobalVariable &Global : M->globals()) { SmallVector GVs; Global.getDebugInfo(GVs); for (auto *GVE : GVs) GVMap[GVE->getVariable()].push_back({&Global, GVE->getExpression()}); } // Create the symbol that designates the start of the unit's contribution // to the string offsets table. In a split DWARF scenario, only the skeleton // unit has the DW_AT_str_offsets_base attribute (and hence needs the symbol). if (useSegmentedStringOffsetsTable()) (useSplitDwarf() ? SkeletonHolder : InfoHolder) .setStringOffsetsStartSym(Asm->createTempSymbol("str_offsets_base")); // Create the symbols that designates the start of the DWARF v5 range list // and locations list tables. They are located past the table headers. if (getDwarfVersion() >= 5) { DwarfFile &Holder = useSplitDwarf() ? SkeletonHolder : InfoHolder; Holder.setRnglistsTableBaseSym( Asm->createTempSymbol("rnglists_table_base")); Holder.setLoclistsTableBaseSym( Asm->createTempSymbol("loclists_table_base")); if (useSplitDwarf()) InfoHolder.setRnglistsTableBaseSym( Asm->createTempSymbol("rnglists_dwo_table_base")); } // Create the symbol that points to the first entry following the debug // address table (.debug_addr) header. AddrPool.setLabel(Asm->createTempSymbol("addr_table_base")); for (DICompileUnit *CUNode : M->debug_compile_units()) { // FIXME: Move local imported entities into a list attached to the // subprogram, then this search won't be needed and a // getImportedEntities().empty() test should go below with the rest. bool HasNonLocalImportedEntities = llvm::any_of( CUNode->getImportedEntities(), [](const DIImportedEntity *IE) { return !isa(IE->getScope()); }); if (!HasNonLocalImportedEntities && CUNode->getEnumTypes().empty() && CUNode->getRetainedTypes().empty() && CUNode->getGlobalVariables().empty() && CUNode->getMacros().empty()) continue; DwarfCompileUnit &CU = getOrCreateDwarfCompileUnit(CUNode); // Global Variables. for (auto *GVE : CUNode->getGlobalVariables()) { // Don't bother adding DIGlobalVariableExpressions listed in the CU if we // already know about the variable and it isn't adding a constant // expression. auto &GVMapEntry = GVMap[GVE->getVariable()]; auto *Expr = GVE->getExpression(); if (!GVMapEntry.size() || (Expr && Expr->isConstant())) GVMapEntry.push_back({nullptr, Expr}); } DenseSet Processed; for (auto *GVE : CUNode->getGlobalVariables()) { DIGlobalVariable *GV = GVE->getVariable(); if (Processed.insert(GV).second) CU.getOrCreateGlobalVariableDIE(GV, sortGlobalExprs(GVMap[GV])); } for (auto *Ty : CUNode->getEnumTypes()) { // The enum types array by design contains pointers to // MDNodes rather than DIRefs. Unique them here. CU.getOrCreateTypeDIE(cast(Ty)); } for (auto *Ty : CUNode->getRetainedTypes()) { // The retained types array by design contains pointers to // MDNodes rather than DIRefs. Unique them here. if (DIType *RT = dyn_cast(Ty)) // There is no point in force-emitting a forward declaration. CU.getOrCreateTypeDIE(RT); } // Emit imported_modules last so that the relevant context is already // available. for (auto *IE : CUNode->getImportedEntities()) constructAndAddImportedEntityDIE(CU, IE); } } void DwarfDebug::finishEntityDefinitions() { for (const auto &Entity : ConcreteEntities) { DIE *Die = Entity->getDIE(); assert(Die); // FIXME: Consider the time-space tradeoff of just storing the unit pointer // in the ConcreteEntities list, rather than looking it up again here. // DIE::getUnit isn't simple - it walks parent pointers, etc. DwarfCompileUnit *Unit = CUDieMap.lookup(Die->getUnitDie()); assert(Unit); Unit->finishEntityDefinition(Entity.get()); } } void DwarfDebug::finishSubprogramDefinitions() { for (const DISubprogram *SP : ProcessedSPNodes) { assert(SP->getUnit()->getEmissionKind() != DICompileUnit::NoDebug); forBothCUs( getOrCreateDwarfCompileUnit(SP->getUnit()), [&](DwarfCompileUnit &CU) { CU.finishSubprogramDefinition(SP); }); } } void DwarfDebug::finalizeModuleInfo() { const TargetLoweringObjectFile &TLOF = Asm->getObjFileLowering(); finishSubprogramDefinitions(); finishEntityDefinitions(); // Include the DWO file name in the hash if there's more than one CU. // This handles ThinLTO's situation where imported CUs may very easily be // duplicate with the same CU partially imported into another ThinLTO unit. StringRef DWOName; if (CUMap.size() > 1) DWOName = Asm->TM.Options.MCOptions.SplitDwarfFile; // Handle anything that needs to be done on a per-unit basis after // all other generation. for (const auto &P : CUMap) { auto &TheCU = *P.second; if (TheCU.getCUNode()->isDebugDirectivesOnly()) continue; // Emit DW_AT_containing_type attribute to connect types with their // vtable holding type. TheCU.constructContainingTypeDIEs(); // Add CU specific attributes if we need to add any. // If we're splitting the dwarf out now that we've got the entire // CU then add the dwo id to it. auto *SkCU = TheCU.getSkeleton(); if (useSplitDwarf() && !empty(TheCU.getUnitDie().children())) { finishUnitAttributes(TheCU.getCUNode(), TheCU); TheCU.addString(TheCU.getUnitDie(), dwarf::DW_AT_GNU_dwo_name, Asm->TM.Options.MCOptions.SplitDwarfFile); SkCU->addString(SkCU->getUnitDie(), dwarf::DW_AT_GNU_dwo_name, Asm->TM.Options.MCOptions.SplitDwarfFile); // Emit a unique identifier for this CU. uint64_t ID = DIEHash(Asm).computeCUSignature(DWOName, TheCU.getUnitDie()); if (getDwarfVersion() >= 5) { TheCU.setDWOId(ID); SkCU->setDWOId(ID); } else { TheCU.addUInt(TheCU.getUnitDie(), dwarf::DW_AT_GNU_dwo_id, dwarf::DW_FORM_data8, ID); SkCU->addUInt(SkCU->getUnitDie(), dwarf::DW_AT_GNU_dwo_id, dwarf::DW_FORM_data8, ID); } if (getDwarfVersion() < 5 && !SkeletonHolder.getRangeLists().empty()) { const MCSymbol *Sym = TLOF.getDwarfRangesSection()->getBeginSymbol(); SkCU->addSectionLabel(SkCU->getUnitDie(), dwarf::DW_AT_GNU_ranges_base, Sym, Sym); } } else if (SkCU) { finishUnitAttributes(SkCU->getCUNode(), *SkCU); } // If we have code split among multiple sections or non-contiguous // ranges of code then emit a DW_AT_ranges attribute on the unit that will // remain in the .o file, otherwise add a DW_AT_low_pc. // FIXME: We should use ranges allow reordering of code ala // .subsections_via_symbols in mach-o. This would mean turning on // ranges for all subprogram DIEs for mach-o. DwarfCompileUnit &U = SkCU ? *SkCU : TheCU; // We don't keep track of which addresses are used in which CU so this // is a bit pessimistic under LTO. if (!AddrPool.isEmpty() && (getDwarfVersion() >= 5 || (SkCU && !empty(TheCU.getUnitDie().children())))) U.addAddrTableBase(); if (unsigned NumRanges = TheCU.getRanges().size()) { if (NumRanges > 1 && useRangesSection()) // A DW_AT_low_pc attribute may also be specified in combination with // DW_AT_ranges to specify the default base address for use in // location lists (see Section 2.6.2) and range lists (see Section // 2.17.3). U.addUInt(U.getUnitDie(), dwarf::DW_AT_low_pc, dwarf::DW_FORM_addr, 0); else U.setBaseAddress(TheCU.getRanges().front().getStart()); U.attachRangesOrLowHighPC(U.getUnitDie(), TheCU.takeRanges()); } if (getDwarfVersion() >= 5) { if (U.hasRangeLists()) U.addRnglistsBase(); if (!DebugLocs.getLists().empty() && !useSplitDwarf()) U.addLoclistsBase(); } auto *CUNode = cast(P.first); // If compile Unit has macros, emit "DW_AT_macro_info" attribute. if (CUNode->getMacros()) U.addSectionLabel(U.getUnitDie(), dwarf::DW_AT_macro_info, U.getMacroLabelBegin(), TLOF.getDwarfMacinfoSection()->getBeginSymbol()); } // Emit all frontend-produced Skeleton CUs, i.e., Clang modules. for (auto *CUNode : MMI->getModule()->debug_compile_units()) if (CUNode->getDWOId()) getOrCreateDwarfCompileUnit(CUNode); // Compute DIE offsets and sizes. InfoHolder.computeSizeAndOffsets(); if (useSplitDwarf()) SkeletonHolder.computeSizeAndOffsets(); } // Emit all Dwarf sections that should come after the content. void DwarfDebug::endModule() { assert(CurFn == nullptr); assert(CurMI == nullptr); // If we aren't actually generating debug info (check beginModule - // conditionalized on !DisableDebugInfoPrinting and the presence of the // llvm.dbg.cu metadata node) if (!MMI->hasDebugInfo()) return; // Finalize the debug info for the module. finalizeModuleInfo(); emitDebugStr(); if (useSplitDwarf()) emitDebugLocDWO(); else // Emit info into a debug loc section. emitDebugLoc(); // Corresponding abbreviations into a abbrev section. emitAbbreviations(); // Emit all the DIEs into a debug info section. emitDebugInfo(); // Emit info into a debug aranges section. if (GenerateARangeSection) emitDebugARanges(); // Emit info into a debug ranges section. emitDebugRanges(); // Emit info into a debug macinfo section. emitDebugMacinfo(); if (useSplitDwarf()) { emitDebugStrDWO(); emitDebugInfoDWO(); emitDebugAbbrevDWO(); emitDebugLineDWO(); emitDebugRangesDWO(); } emitDebugAddr(); // Emit info into the dwarf accelerator table sections. switch (getAccelTableKind()) { case AccelTableKind::Apple: emitAccelNames(); emitAccelObjC(); emitAccelNamespaces(); emitAccelTypes(); break; case AccelTableKind::Dwarf: emitAccelDebugNames(); break; case AccelTableKind::None: break; case AccelTableKind::Default: llvm_unreachable("Default should have already been resolved."); } // Emit the pubnames and pubtypes sections if requested. emitDebugPubSections(); // clean up. // FIXME: AbstractVariables.clear(); } void DwarfDebug::ensureAbstractEntityIsCreated(DwarfCompileUnit &CU, const DINode *Node, const MDNode *ScopeNode) { if (CU.getExistingAbstractEntity(Node)) return; CU.createAbstractEntity(Node, LScopes.getOrCreateAbstractScope( cast(ScopeNode))); } void DwarfDebug::ensureAbstractEntityIsCreatedIfScoped(DwarfCompileUnit &CU, const DINode *Node, const MDNode *ScopeNode) { if (CU.getExistingAbstractEntity(Node)) return; if (LexicalScope *Scope = LScopes.findAbstractScope(cast_or_null(ScopeNode))) CU.createAbstractEntity(Node, Scope); } // Collect variable information from side table maintained by MF. void DwarfDebug::collectVariableInfoFromMFTable( DwarfCompileUnit &TheCU, DenseSet &Processed) { SmallDenseMap MFVars; for (const auto &VI : Asm->MF->getVariableDbgInfo()) { if (!VI.Var) continue; assert(VI.Var->isValidLocationForIntrinsic(VI.Loc) && "Expected inlined-at fields to agree"); InlinedEntity Var(VI.Var, VI.Loc->getInlinedAt()); Processed.insert(Var); LexicalScope *Scope = LScopes.findLexicalScope(VI.Loc); // If variable scope is not found then skip this variable. if (!Scope) continue; ensureAbstractEntityIsCreatedIfScoped(TheCU, Var.first, Scope->getScopeNode()); auto RegVar = llvm::make_unique( cast(Var.first), Var.second); RegVar->initializeMMI(VI.Expr, VI.Slot); if (DbgVariable *DbgVar = MFVars.lookup(Var)) DbgVar->addMMIEntry(*RegVar); else if (InfoHolder.addScopeVariable(Scope, RegVar.get())) { MFVars.insert({Var, RegVar.get()}); ConcreteEntities.push_back(std::move(RegVar)); } } } // Get .debug_loc entry for the instruction range starting at MI. static DebugLocEntry::Value getDebugLocValue(const MachineInstr *MI) { const DIExpression *Expr = MI->getDebugExpression(); assert(MI->getNumOperands() == 4); if (MI->getOperand(0).isReg()) { auto RegOp = MI->getOperand(0); auto Op1 = MI->getOperand(1); // If the second operand is an immediate, this is a // register-indirect address. assert((!Op1.isImm() || (Op1.getImm() == 0)) && "unexpected offset"); MachineLocation MLoc(RegOp.getReg(), Op1.isImm()); return DebugLocEntry::Value(Expr, MLoc); } if (MI->getOperand(0).isImm()) return DebugLocEntry::Value(Expr, MI->getOperand(0).getImm()); if (MI->getOperand(0).isFPImm()) return DebugLocEntry::Value(Expr, MI->getOperand(0).getFPImm()); if (MI->getOperand(0).isCImm()) return DebugLocEntry::Value(Expr, MI->getOperand(0).getCImm()); llvm_unreachable("Unexpected 4-operand DBG_VALUE instruction!"); } /// If this and Next are describing different fragments of the same /// variable, merge them by appending Next's values to the current /// list of values. /// Return true if the merge was successful. bool DebugLocEntry::MergeValues(const DebugLocEntry &Next) { if (Begin == Next.Begin) { auto *FirstExpr = cast(Values[0].Expression); auto *FirstNextExpr = cast(Next.Values[0].Expression); if (!FirstExpr->isFragment() || !FirstNextExpr->isFragment()) return false; // We can only merge entries if none of the fragments overlap any others. // In doing so, we can take advantage of the fact that both lists are // sorted. for (unsigned i = 0, j = 0; i < Values.size(); ++i) { for (; j < Next.Values.size(); ++j) { int res = cast(Values[i].Expression)->fragmentCmp( cast(Next.Values[j].Expression)); if (res == 0) // The two expressions overlap, we can't merge. return false; // Values[i] is entirely before Next.Values[j], // so go back to the next entry of Values. else if (res == -1) break; // Next.Values[j] is entirely before Values[i], so go on to the // next entry of Next.Values. } } addValues(Next.Values); End = Next.End; return true; } return false; } /// Build the location list for all DBG_VALUEs in the function that /// describe the same variable. If the ranges of several independent /// fragments of the same variable overlap partially, split them up and /// combine the ranges. The resulting DebugLocEntries are will have /// strict monotonically increasing begin addresses and will never /// overlap. // // Input: // // Ranges History [var, loc, fragment ofs size] // 0 | [x, (reg0, fragment 0, 32)] // 1 | | [x, (reg1, fragment 32, 32)] <- IsFragmentOfPrevEntry // 2 | | ... // 3 | [clobber reg0] // 4 [x, (mem, fragment 0, 64)] <- overlapping with both previous fragments of // x. // // Output: // // [0-1] [x, (reg0, fragment 0, 32)] // [1-3] [x, (reg0, fragment 0, 32), (reg1, fragment 32, 32)] // [3-4] [x, (reg1, fragment 32, 32)] // [4- ] [x, (mem, fragment 0, 64)] void DwarfDebug::buildLocationList(SmallVectorImpl &DebugLoc, const DbgValueHistoryMap::InstrRanges &Ranges) { SmallVector OpenRanges; for (auto I = Ranges.begin(), E = Ranges.end(); I != E; ++I) { const MachineInstr *Begin = I->first; const MachineInstr *End = I->second; assert(Begin->isDebugValue() && "Invalid History entry"); // Check if a variable is inaccessible in this range. if (Begin->getNumOperands() > 1 && Begin->getOperand(0).isReg() && !Begin->getOperand(0).getReg()) { OpenRanges.clear(); continue; } // If this fragment overlaps with any open ranges, truncate them. const DIExpression *DIExpr = Begin->getDebugExpression(); auto Last = remove_if(OpenRanges, [&](DebugLocEntry::Value R) { return DIExpr->fragmentsOverlap(R.getExpression()); }); OpenRanges.erase(Last, OpenRanges.end()); const MCSymbol *StartLabel = getLabelBeforeInsn(Begin); assert(StartLabel && "Forgot label before DBG_VALUE starting a range!"); const MCSymbol *EndLabel; if (End != nullptr) EndLabel = getLabelAfterInsn(End); else if (std::next(I) == Ranges.end()) EndLabel = Asm->getFunctionEnd(); else EndLabel = getLabelBeforeInsn(std::next(I)->first); assert(EndLabel && "Forgot label after instruction ending a range!"); LLVM_DEBUG(dbgs() << "DotDebugLoc: " << *Begin << "\n"); auto Value = getDebugLocValue(Begin); // Omit entries with empty ranges as they do not have any effect in DWARF. if (StartLabel == EndLabel) { // If this is a fragment, we must still add the value to the list of // open ranges, since it may describe non-overlapping parts of the // variable. if (DIExpr->isFragment()) OpenRanges.push_back(Value); LLVM_DEBUG(dbgs() << "Omitting location list entry with empty range.\n"); continue; } DebugLocEntry Loc(StartLabel, EndLabel, Value); bool couldMerge = false; // If this is a fragment, it may belong to the current DebugLocEntry. if (DIExpr->isFragment()) { // Add this value to the list of open ranges. OpenRanges.push_back(Value); // Attempt to add the fragment to the last entry. if (!DebugLoc.empty()) if (DebugLoc.back().MergeValues(Loc)) couldMerge = true; } if (!couldMerge) { // Need to add a new DebugLocEntry. Add all values from still // valid non-overlapping fragments. if (OpenRanges.size()) Loc.addValues(OpenRanges); DebugLoc.push_back(std::move(Loc)); } // Attempt to coalesce the ranges of two otherwise identical // DebugLocEntries. auto CurEntry = DebugLoc.rbegin(); LLVM_DEBUG({ dbgs() << CurEntry->getValues().size() << " Values:\n"; for (auto &Value : CurEntry->getValues()) Value.dump(); dbgs() << "-----\n"; }); auto PrevEntry = std::next(CurEntry); if (PrevEntry != DebugLoc.rend() && PrevEntry->MergeRanges(*CurEntry)) DebugLoc.pop_back(); } } DbgEntity *DwarfDebug::createConcreteEntity(DwarfCompileUnit &TheCU, LexicalScope &Scope, const DINode *Node, const DILocation *Location, const MCSymbol *Sym) { ensureAbstractEntityIsCreatedIfScoped(TheCU, Node, Scope.getScopeNode()); if (isa(Node)) { ConcreteEntities.push_back( llvm::make_unique(cast(Node), Location)); InfoHolder.addScopeVariable(&Scope, cast(ConcreteEntities.back().get())); } else if (isa(Node)) { ConcreteEntities.push_back( llvm::make_unique(cast(Node), Location, Sym)); InfoHolder.addScopeLabel(&Scope, cast(ConcreteEntities.back().get())); } return ConcreteEntities.back().get(); } /// Determine whether a *singular* DBG_VALUE is valid for the entirety of its /// enclosing lexical scope. The check ensures there are no other instructions /// in the same lexical scope preceding the DBG_VALUE and that its range is /// either open or otherwise rolls off the end of the scope. static bool validThroughout(LexicalScopes &LScopes, const MachineInstr *DbgValue, const MachineInstr *RangeEnd) { assert(DbgValue->getDebugLoc() && "DBG_VALUE without a debug location"); auto MBB = DbgValue->getParent(); auto DL = DbgValue->getDebugLoc(); auto *LScope = LScopes.findLexicalScope(DL); // Scope doesn't exist; this is a dead DBG_VALUE. if (!LScope) return false; auto &LSRange = LScope->getRanges(); if (LSRange.size() == 0) return false; // Determine if the DBG_VALUE is valid at the beginning of its lexical block. const MachineInstr *LScopeBegin = LSRange.front().first; // Early exit if the lexical scope begins outside of the current block. if (LScopeBegin->getParent() != MBB) return false; MachineBasicBlock::const_reverse_iterator Pred(DbgValue); for (++Pred; Pred != MBB->rend(); ++Pred) { if (Pred->getFlag(MachineInstr::FrameSetup)) break; auto PredDL = Pred->getDebugLoc(); if (!PredDL || Pred->isMetaInstruction()) continue; // Check whether the instruction preceding the DBG_VALUE is in the same // (sub)scope as the DBG_VALUE. if (DL->getScope() == PredDL->getScope()) return false; auto *PredScope = LScopes.findLexicalScope(PredDL); if (!PredScope || LScope->dominates(PredScope)) return false; } // If the range of the DBG_VALUE is open-ended, report success. if (!RangeEnd) return true; // Fail if there are instructions belonging to our scope in another block. const MachineInstr *LScopeEnd = LSRange.back().second; if (LScopeEnd->getParent() != MBB) return false; // Single, constant DBG_VALUEs in the prologue are promoted to be live // throughout the function. This is a hack, presumably for DWARF v2 and not // necessarily correct. It would be much better to use a dbg.declare instead // if we know the constant is live throughout the scope. if (DbgValue->getOperand(0).isImm() && MBB->pred_empty()) return true; return false; } // Find variables for each lexical scope. void DwarfDebug::collectEntityInfo(DwarfCompileUnit &TheCU, const DISubprogram *SP, DenseSet &Processed) { // Grab the variable info that was squirreled away in the MMI side-table. collectVariableInfoFromMFTable(TheCU, Processed); for (const auto &I : DbgValues) { InlinedEntity IV = I.first; if (Processed.count(IV)) continue; // Instruction ranges, specifying where IV is accessible. const auto &Ranges = I.second; if (Ranges.empty()) continue; LexicalScope *Scope = nullptr; const DILocalVariable *LocalVar = cast(IV.first); if (const DILocation *IA = IV.second) Scope = LScopes.findInlinedScope(LocalVar->getScope(), IA); else Scope = LScopes.findLexicalScope(LocalVar->getScope()); // If variable scope is not found then skip this variable. if (!Scope) continue; Processed.insert(IV); DbgVariable *RegVar = cast(createConcreteEntity(TheCU, *Scope, LocalVar, IV.second)); const MachineInstr *MInsn = Ranges.front().first; assert(MInsn->isDebugValue() && "History must begin with debug value"); // Check if there is a single DBG_VALUE, valid throughout the var's scope. if (Ranges.size() == 1 && validThroughout(LScopes, MInsn, Ranges.front().second)) { RegVar->initializeDbgValue(MInsn); continue; } // Do not emit location lists if .debug_loc secton is disabled. if (!useLocSection()) continue; // Handle multiple DBG_VALUE instructions describing one variable. DebugLocStream::ListBuilder List(DebugLocs, TheCU, *Asm, *RegVar, *MInsn); // Build the location list for this variable. SmallVector Entries; buildLocationList(Entries, Ranges); // If the variable has a DIBasicType, extract it. Basic types cannot have // unique identifiers, so don't bother resolving the type with the // identifier map. const DIBasicType *BT = dyn_cast( static_cast(LocalVar->getType())); // Finalize the entry by lowering it into a DWARF bytestream. for (auto &Entry : Entries) Entry.finalize(*Asm, List, BT); } // For each InlinedEntity collected from DBG_LABEL instructions, convert to // DWARF-related DbgLabel. for (const auto &I : DbgLabels) { InlinedEntity IL = I.first; const MachineInstr *MI = I.second; if (MI == nullptr) continue; LexicalScope *Scope = nullptr; const DILabel *Label = cast(IL.first); // Get inlined DILocation if it is inlined label. if (const DILocation *IA = IL.second) Scope = LScopes.findInlinedScope(Label->getScope(), IA); else Scope = LScopes.findLexicalScope(Label->getScope()); // If label scope is not found then skip this label. if (!Scope) continue; Processed.insert(IL); /// At this point, the temporary label is created. /// Save the temporary label to DbgLabel entity to get the /// actually address when generating Dwarf DIE. MCSymbol *Sym = getLabelBeforeInsn(MI); createConcreteEntity(TheCU, *Scope, Label, IL.second, Sym); } // Collect info for variables/labels that were optimized out. for (const DINode *DN : SP->getRetainedNodes()) { if (!Processed.insert(InlinedEntity(DN, nullptr)).second) continue; LexicalScope *Scope = nullptr; if (auto *DV = dyn_cast(DN)) { Scope = LScopes.findLexicalScope(DV->getScope()); } else if (auto *DL = dyn_cast(DN)) { Scope = LScopes.findLexicalScope(DL->getScope()); } if (Scope) createConcreteEntity(TheCU, *Scope, DN, nullptr); } } // Process beginning of an instruction. void DwarfDebug::beginInstruction(const MachineInstr *MI) { DebugHandlerBase::beginInstruction(MI); assert(CurMI); const auto *SP = MI->getMF()->getFunction().getSubprogram(); if (!SP || SP->getUnit()->getEmissionKind() == DICompileUnit::NoDebug) return; // Check if source location changes, but ignore DBG_VALUE and CFI locations. // If the instruction is part of the function frame setup code, do not emit // any line record, as there is no correspondence with any user code. if (MI->isMetaInstruction() || MI->getFlag(MachineInstr::FrameSetup)) return; const DebugLoc &DL = MI->getDebugLoc(); // When we emit a line-0 record, we don't update PrevInstLoc; so look at // the last line number actually emitted, to see if it was line 0. unsigned LastAsmLine = Asm->OutStreamer->getContext().getCurrentDwarfLoc().getLine(); // Request a label after the call in order to emit AT_return_pc information // in call site entries. TODO: Add support for targets with delay slots. if (SP->areAllCallsDescribed() && MI->isCall() && !MI->hasDelaySlot()) requestLabelAfterInsn(MI); if (DL == PrevInstLoc) { // If we have an ongoing unspecified location, nothing to do here. if (!DL) return; // We have an explicit location, same as the previous location. // But we might be coming back to it after a line 0 record. if (LastAsmLine == 0 && DL.getLine() != 0) { // Reinstate the source location but not marked as a statement. const MDNode *Scope = DL.getScope(); recordSourceLine(DL.getLine(), DL.getCol(), Scope, /*Flags=*/0); } return; } if (!DL) { // We have an unspecified location, which might want to be line 0. // If we have already emitted a line-0 record, don't repeat it. if (LastAsmLine == 0) return; // If user said Don't Do That, don't do that. if (UnknownLocations == Disable) return; // See if we have a reason to emit a line-0 record now. // Reasons to emit a line-0 record include: // - User asked for it (UnknownLocations). // - Instruction has a label, so it's referenced from somewhere else, // possibly debug information; we want it to have a source location. // - Instruction is at the top of a block; we don't want to inherit the // location from the physically previous (maybe unrelated) block. if (UnknownLocations == Enable || PrevLabel || (PrevInstBB && PrevInstBB != MI->getParent())) { // Preserve the file and column numbers, if we can, to save space in // the encoded line table. // Do not update PrevInstLoc, it remembers the last non-0 line. const MDNode *Scope = nullptr; unsigned Column = 0; if (PrevInstLoc) { Scope = PrevInstLoc.getScope(); Column = PrevInstLoc.getCol(); } recordSourceLine(/*Line=*/0, Column, Scope, /*Flags=*/0); } return; } // We have an explicit location, different from the previous location. // Don't repeat a line-0 record, but otherwise emit the new location. // (The new location might be an explicit line 0, which we do emit.) if (PrevInstLoc && DL.getLine() == 0 && LastAsmLine == 0) return; unsigned Flags = 0; if (DL == PrologEndLoc) { Flags |= DWARF2_FLAG_PROLOGUE_END | DWARF2_FLAG_IS_STMT; PrologEndLoc = DebugLoc(); } // If the line changed, we call that a new statement; unless we went to // line 0 and came back, in which case it is not a new statement. unsigned OldLine = PrevInstLoc ? PrevInstLoc.getLine() : LastAsmLine; if (DL.getLine() && DL.getLine() != OldLine) Flags |= DWARF2_FLAG_IS_STMT; const MDNode *Scope = DL.getScope(); recordSourceLine(DL.getLine(), DL.getCol(), Scope, Flags); // If we're not at line 0, remember this location. if (DL.getLine()) PrevInstLoc = DL; } static DebugLoc findPrologueEndLoc(const MachineFunction *MF) { // First known non-DBG_VALUE and non-frame setup location marks // the beginning of the function body. for (const auto &MBB : *MF) for (const auto &MI : MBB) if (!MI.isMetaInstruction() && !MI.getFlag(MachineInstr::FrameSetup) && MI.getDebugLoc()) return MI.getDebugLoc(); return DebugLoc(); } // Gather pre-function debug information. Assumes being called immediately // after the function entry point has been emitted. void DwarfDebug::beginFunctionImpl(const MachineFunction *MF) { CurFn = MF; auto *SP = MF->getFunction().getSubprogram(); assert(LScopes.empty() || SP == LScopes.getCurrentFunctionScope()->getScopeNode()); if (SP->getUnit()->getEmissionKind() == DICompileUnit::NoDebug) return; DwarfCompileUnit &CU = getOrCreateDwarfCompileUnit(SP->getUnit()); // Set DwarfDwarfCompileUnitID in MCContext to the Compile Unit this function // belongs to so that we add to the correct per-cu line table in the // non-asm case. if (Asm->OutStreamer->hasRawTextSupport()) // Use a single line table if we are generating assembly. Asm->OutStreamer->getContext().setDwarfCompileUnitID(0); else Asm->OutStreamer->getContext().setDwarfCompileUnitID(CU.getUniqueID()); // Record beginning of function. PrologEndLoc = findPrologueEndLoc(MF); if (PrologEndLoc) { // We'd like to list the prologue as "not statements" but GDB behaves // poorly if we do that. Revisit this with caution/GDB (7.5+) testing. auto *SP = PrologEndLoc->getInlinedAtScope()->getSubprogram(); recordSourceLine(SP->getScopeLine(), 0, SP, DWARF2_FLAG_IS_STMT); } } void DwarfDebug::skippedNonDebugFunction() { // If we don't have a subprogram for this function then there will be a hole // in the range information. Keep note of this by setting the previously used // section to nullptr. PrevCU = nullptr; CurFn = nullptr; } // Gather and emit post-function debug information. void DwarfDebug::endFunctionImpl(const MachineFunction *MF) { const DISubprogram *SP = MF->getFunction().getSubprogram(); assert(CurFn == MF && "endFunction should be called with the same function as beginFunction"); // Set DwarfDwarfCompileUnitID in MCContext to default value. Asm->OutStreamer->getContext().setDwarfCompileUnitID(0); LexicalScope *FnScope = LScopes.getCurrentFunctionScope(); assert(!FnScope || SP == FnScope->getScopeNode()); DwarfCompileUnit &TheCU = *CUMap.lookup(SP->getUnit()); if (TheCU.getCUNode()->isDebugDirectivesOnly()) { PrevLabel = nullptr; CurFn = nullptr; return; } DenseSet Processed; collectEntityInfo(TheCU, SP, Processed); // Add the range of this function to the list of ranges for the CU. TheCU.addRange(RangeSpan(Asm->getFunctionBegin(), Asm->getFunctionEnd())); // Under -gmlt, skip building the subprogram if there are no inlined // subroutines inside it. But with -fdebug-info-for-profiling, the subprogram // is still needed as we need its source location. if (!TheCU.getCUNode()->getDebugInfoForProfiling() && TheCU.getCUNode()->getEmissionKind() == DICompileUnit::LineTablesOnly && LScopes.getAbstractScopesList().empty() && !IsDarwin) { assert(InfoHolder.getScopeVariables().empty()); PrevLabel = nullptr; CurFn = nullptr; return; } #ifndef NDEBUG size_t NumAbstractScopes = LScopes.getAbstractScopesList().size(); #endif // Construct abstract scopes. for (LexicalScope *AScope : LScopes.getAbstractScopesList()) { auto *SP = cast(AScope->getScopeNode()); for (const DINode *DN : SP->getRetainedNodes()) { if (!Processed.insert(InlinedEntity(DN, nullptr)).second) continue; const MDNode *Scope = nullptr; if (auto *DV = dyn_cast(DN)) Scope = DV->getScope(); else if (auto *DL = dyn_cast(DN)) Scope = DL->getScope(); else llvm_unreachable("Unexpected DI type!"); // Collect info for variables/labels that were optimized out. ensureAbstractEntityIsCreated(TheCU, DN, Scope); assert(LScopes.getAbstractScopesList().size() == NumAbstractScopes && "ensureAbstractEntityIsCreated inserted abstract scopes"); } constructAbstractSubprogramScopeDIE(TheCU, AScope); } ProcessedSPNodes.insert(SP); DIE &ScopeDIE = TheCU.constructSubprogramScopeDIE(SP, FnScope); if (auto *SkelCU = TheCU.getSkeleton()) if (!LScopes.getAbstractScopesList().empty() && TheCU.getCUNode()->getSplitDebugInlining()) SkelCU->constructSubprogramScopeDIE(SP, FnScope); // Construct call site entries. constructCallSiteEntryDIEs(*SP, TheCU, ScopeDIE, *MF); // Clear debug info // Ownership of DbgVariables is a bit subtle - ScopeVariables owns all the // DbgVariables except those that are also in AbstractVariables (since they // can be used cross-function) InfoHolder.getScopeVariables().clear(); InfoHolder.getScopeLabels().clear(); PrevLabel = nullptr; CurFn = nullptr; } // Register a source line with debug info. Returns the unique label that was // emitted and which provides correspondence to the source line list. void DwarfDebug::recordSourceLine(unsigned Line, unsigned Col, const MDNode *S, unsigned Flags) { StringRef Fn; unsigned FileNo = 1; unsigned Discriminator = 0; if (auto *Scope = cast_or_null(S)) { Fn = Scope->getFilename(); if (Line != 0 && getDwarfVersion() >= 4) if (auto *LBF = dyn_cast(Scope)) Discriminator = LBF->getDiscriminator(); unsigned CUID = Asm->OutStreamer->getContext().getDwarfCompileUnitID(); FileNo = static_cast(*InfoHolder.getUnits()[CUID]) .getOrCreateSourceID(Scope->getFile()); } Asm->OutStreamer->EmitDwarfLocDirective(FileNo, Line, Col, Flags, 0, Discriminator, Fn); } //===----------------------------------------------------------------------===// // Emit Methods //===----------------------------------------------------------------------===// // Emit the debug info section. void DwarfDebug::emitDebugInfo() { DwarfFile &Holder = useSplitDwarf() ? SkeletonHolder : InfoHolder; Holder.emitUnits(/* UseOffsets */ false); } // Emit the abbreviation section. void DwarfDebug::emitAbbreviations() { DwarfFile &Holder = useSplitDwarf() ? SkeletonHolder : InfoHolder; Holder.emitAbbrevs(Asm->getObjFileLowering().getDwarfAbbrevSection()); } void DwarfDebug::emitStringOffsetsTableHeader() { DwarfFile &Holder = useSplitDwarf() ? SkeletonHolder : InfoHolder; Holder.getStringPool().emitStringOffsetsTableHeader( *Asm, Asm->getObjFileLowering().getDwarfStrOffSection(), Holder.getStringOffsetsStartSym()); } template void DwarfDebug::emitAccel(AccelTableT &Accel, MCSection *Section, StringRef TableName) { Asm->OutStreamer->SwitchSection(Section); // Emit the full data. emitAppleAccelTable(Asm, Accel, TableName, Section->getBeginSymbol()); } void DwarfDebug::emitAccelDebugNames() { // Don't emit anything if we have no compilation units to index. if (getUnits().empty()) return; emitDWARF5AccelTable(Asm, AccelDebugNames, *this, getUnits()); } // Emit visible names into a hashed accelerator table section. void DwarfDebug::emitAccelNames() { emitAccel(AccelNames, Asm->getObjFileLowering().getDwarfAccelNamesSection(), "Names"); } // Emit objective C classes and categories into a hashed accelerator table // section. void DwarfDebug::emitAccelObjC() { emitAccel(AccelObjC, Asm->getObjFileLowering().getDwarfAccelObjCSection(), "ObjC"); } // Emit namespace dies into a hashed accelerator table. void DwarfDebug::emitAccelNamespaces() { emitAccel(AccelNamespace, Asm->getObjFileLowering().getDwarfAccelNamespaceSection(), "namespac"); } // Emit type dies into a hashed accelerator table. void DwarfDebug::emitAccelTypes() { emitAccel(AccelTypes, Asm->getObjFileLowering().getDwarfAccelTypesSection(), "types"); } // Public name handling. // The format for the various pubnames: // // dwarf pubnames - offset/name pairs where the offset is the offset into the CU // for the DIE that is named. // // gnu pubnames - offset/index value/name tuples where the offset is the offset // into the CU and the index value is computed according to the type of value // for the DIE that is named. // // For type units the offset is the offset of the skeleton DIE. For split dwarf // it's the offset within the debug_info/debug_types dwo section, however, the // reference in the pubname header doesn't change. /// computeIndexValue - Compute the gdb index value for the DIE and CU. static dwarf::PubIndexEntryDescriptor computeIndexValue(DwarfUnit *CU, const DIE *Die) { // Entities that ended up only in a Type Unit reference the CU instead (since // the pub entry has offsets within the CU there's no real offset that can be // provided anyway). As it happens all such entities (namespaces and types, // types only in C++ at that) are rendered as TYPE+EXTERNAL. If this turns out // not to be true it would be necessary to persist this information from the // point at which the entry is added to the index data structure - since by // the time the index is built from that, the original type/namespace DIE in a // type unit has already been destroyed so it can't be queried for properties // like tag, etc. if (Die->getTag() == dwarf::DW_TAG_compile_unit) return dwarf::PubIndexEntryDescriptor(dwarf::GIEK_TYPE, dwarf::GIEL_EXTERNAL); dwarf::GDBIndexEntryLinkage Linkage = dwarf::GIEL_STATIC; // We could have a specification DIE that has our most of our knowledge, // look for that now. if (DIEValue SpecVal = Die->findAttribute(dwarf::DW_AT_specification)) { DIE &SpecDIE = SpecVal.getDIEEntry().getEntry(); if (SpecDIE.findAttribute(dwarf::DW_AT_external)) Linkage = dwarf::GIEL_EXTERNAL; } else if (Die->findAttribute(dwarf::DW_AT_external)) Linkage = dwarf::GIEL_EXTERNAL; switch (Die->getTag()) { case dwarf::DW_TAG_class_type: case dwarf::DW_TAG_structure_type: case dwarf::DW_TAG_union_type: case dwarf::DW_TAG_enumeration_type: return dwarf::PubIndexEntryDescriptor( dwarf::GIEK_TYPE, CU->getLanguage() != dwarf::DW_LANG_C_plus_plus ? dwarf::GIEL_STATIC : dwarf::GIEL_EXTERNAL); case dwarf::DW_TAG_typedef: case dwarf::DW_TAG_base_type: case dwarf::DW_TAG_subrange_type: return dwarf::PubIndexEntryDescriptor(dwarf::GIEK_TYPE, dwarf::GIEL_STATIC); case dwarf::DW_TAG_namespace: return dwarf::GIEK_TYPE; case dwarf::DW_TAG_subprogram: return dwarf::PubIndexEntryDescriptor(dwarf::GIEK_FUNCTION, Linkage); case dwarf::DW_TAG_variable: return dwarf::PubIndexEntryDescriptor(dwarf::GIEK_VARIABLE, Linkage); case dwarf::DW_TAG_enumerator: return dwarf::PubIndexEntryDescriptor(dwarf::GIEK_VARIABLE, dwarf::GIEL_STATIC); default: return dwarf::GIEK_NONE; } } /// emitDebugPubSections - Emit visible names and types into debug pubnames and /// pubtypes sections. void DwarfDebug::emitDebugPubSections() { for (const auto &NU : CUMap) { DwarfCompileUnit *TheU = NU.second; if (!TheU->hasDwarfPubSections()) continue; bool GnuStyle = TheU->getCUNode()->getNameTableKind() == DICompileUnit::DebugNameTableKind::GNU; Asm->OutStreamer->SwitchSection( GnuStyle ? Asm->getObjFileLowering().getDwarfGnuPubNamesSection() : Asm->getObjFileLowering().getDwarfPubNamesSection()); emitDebugPubSection(GnuStyle, "Names", TheU, TheU->getGlobalNames()); Asm->OutStreamer->SwitchSection( GnuStyle ? Asm->getObjFileLowering().getDwarfGnuPubTypesSection() : Asm->getObjFileLowering().getDwarfPubTypesSection()); emitDebugPubSection(GnuStyle, "Types", TheU, TheU->getGlobalTypes()); } } void DwarfDebug::emitSectionReference(const DwarfCompileUnit &CU) { if (useSectionsAsReferences()) Asm->EmitDwarfOffset(CU.getSection()->getBeginSymbol(), CU.getDebugSectionOffset()); else Asm->emitDwarfSymbolReference(CU.getLabelBegin()); } void DwarfDebug::emitDebugPubSection(bool GnuStyle, StringRef Name, DwarfCompileUnit *TheU, const StringMap &Globals) { if (auto *Skeleton = TheU->getSkeleton()) TheU = Skeleton; // Emit the header. Asm->OutStreamer->AddComment("Length of Public " + Name + " Info"); MCSymbol *BeginLabel = Asm->createTempSymbol("pub" + Name + "_begin"); MCSymbol *EndLabel = Asm->createTempSymbol("pub" + Name + "_end"); Asm->EmitLabelDifference(EndLabel, BeginLabel, 4); Asm->OutStreamer->EmitLabel(BeginLabel); Asm->OutStreamer->AddComment("DWARF Version"); Asm->emitInt16(dwarf::DW_PUBNAMES_VERSION); Asm->OutStreamer->AddComment("Offset of Compilation Unit Info"); emitSectionReference(*TheU); Asm->OutStreamer->AddComment("Compilation Unit Length"); Asm->emitInt32(TheU->getLength()); // Emit the pubnames for this compilation unit. for (const auto &GI : Globals) { const char *Name = GI.getKeyData(); const DIE *Entity = GI.second; Asm->OutStreamer->AddComment("DIE offset"); Asm->emitInt32(Entity->getOffset()); if (GnuStyle) { dwarf::PubIndexEntryDescriptor Desc = computeIndexValue(TheU, Entity); Asm->OutStreamer->AddComment( Twine("Attributes: ") + dwarf::GDBIndexEntryKindString(Desc.Kind) + ", " + dwarf::GDBIndexEntryLinkageString(Desc.Linkage)); Asm->emitInt8(Desc.toBits()); } Asm->OutStreamer->AddComment("External Name"); Asm->OutStreamer->EmitBytes(StringRef(Name, GI.getKeyLength() + 1)); } Asm->OutStreamer->AddComment("End Mark"); Asm->emitInt32(0); Asm->OutStreamer->EmitLabel(EndLabel); } /// Emit null-terminated strings into a debug str section. void DwarfDebug::emitDebugStr() { MCSection *StringOffsetsSection = nullptr; if (useSegmentedStringOffsetsTable()) { emitStringOffsetsTableHeader(); StringOffsetsSection = Asm->getObjFileLowering().getDwarfStrOffSection(); } DwarfFile &Holder = useSplitDwarf() ? SkeletonHolder : InfoHolder; Holder.emitStrings(Asm->getObjFileLowering().getDwarfStrSection(), StringOffsetsSection, /* UseRelativeOffsets = */ true); } void DwarfDebug::emitDebugLocEntry(ByteStreamer &Streamer, const DebugLocStream::Entry &Entry) { auto &&Comments = DebugLocs.getComments(Entry); auto Comment = Comments.begin(); auto End = Comments.end(); for (uint8_t Byte : DebugLocs.getBytes(Entry)) Streamer.EmitInt8(Byte, Comment != End ? *(Comment++) : ""); } static void emitDebugLocValue(const AsmPrinter &AP, const DIBasicType *BT, const DebugLocEntry::Value &Value, DwarfExpression &DwarfExpr) { auto *DIExpr = Value.getExpression(); DIExpressionCursor ExprCursor(DIExpr); DwarfExpr.addFragmentOffset(DIExpr); // Regular entry. if (Value.isInt()) { if (BT && (BT->getEncoding() == dwarf::DW_ATE_signed || BT->getEncoding() == dwarf::DW_ATE_signed_char)) DwarfExpr.addSignedConstant(Value.getInt()); else DwarfExpr.addUnsignedConstant(Value.getInt()); } else if (Value.isLocation()) { MachineLocation Location = Value.getLoc(); if (Location.isIndirect()) DwarfExpr.setMemoryLocationKind(); DIExpressionCursor Cursor(DIExpr); const TargetRegisterInfo &TRI = *AP.MF->getSubtarget().getRegisterInfo(); if (!DwarfExpr.addMachineRegExpression(TRI, Cursor, Location.getReg())) return; return DwarfExpr.addExpression(std::move(Cursor)); } else if (Value.isConstantFP()) { APInt RawBytes = Value.getConstantFP()->getValueAPF().bitcastToAPInt(); DwarfExpr.addUnsignedConstant(RawBytes); } DwarfExpr.addExpression(std::move(ExprCursor)); } void DebugLocEntry::finalize(const AsmPrinter &AP, DebugLocStream::ListBuilder &List, const DIBasicType *BT) { assert(Begin != End && "unexpected location list entry with empty range"); DebugLocStream::EntryBuilder Entry(List, Begin, End); BufferByteStreamer Streamer = Entry.getStreamer(); DebugLocDwarfExpression DwarfExpr(AP.getDwarfVersion(), Streamer); const DebugLocEntry::Value &Value = Values[0]; if (Value.isFragment()) { // Emit all fragments that belong to the same variable and range. assert(llvm::all_of(Values, [](DebugLocEntry::Value P) { return P.isFragment(); }) && "all values are expected to be fragments"); assert(std::is_sorted(Values.begin(), Values.end()) && "fragments are expected to be sorted"); for (auto Fragment : Values) emitDebugLocValue(AP, BT, Fragment, DwarfExpr); } else { assert(Values.size() == 1 && "only fragments may have >1 value"); emitDebugLocValue(AP, BT, Value, DwarfExpr); } DwarfExpr.finalize(); } void DwarfDebug::emitDebugLocEntryLocation(const DebugLocStream::Entry &Entry) { // Emit the size. Asm->OutStreamer->AddComment("Loc expr size"); - Asm->emitInt16(DebugLocs.getBytes(Entry).size()); - + if (getDwarfVersion() >= 5) + Asm->EmitULEB128(DebugLocs.getBytes(Entry).size()); + else + Asm->emitInt16(DebugLocs.getBytes(Entry).size()); // Emit the entry. APByteStreamer Streamer(*Asm); emitDebugLocEntry(Streamer, Entry); } // Emit the common part of the DWARF 5 range/locations list tables header. static void emitListsTableHeaderStart(AsmPrinter *Asm, const DwarfFile &Holder, MCSymbol *TableStart, MCSymbol *TableEnd) { // Build the table header, which starts with the length field. Asm->OutStreamer->AddComment("Length"); Asm->EmitLabelDifference(TableEnd, TableStart, 4); Asm->OutStreamer->EmitLabel(TableStart); // Version number (DWARF v5 and later). Asm->OutStreamer->AddComment("Version"); Asm->emitInt16(Asm->OutStreamer->getContext().getDwarfVersion()); // Address size. Asm->OutStreamer->AddComment("Address size"); Asm->emitInt8(Asm->MAI->getCodePointerSize()); // Segment selector size. Asm->OutStreamer->AddComment("Segment selector size"); Asm->emitInt8(0); } // Emit the header of a DWARF 5 range list table list table. Returns the symbol // that designates the end of the table for the caller to emit when the table is // complete. static MCSymbol *emitRnglistsTableHeader(AsmPrinter *Asm, const DwarfFile &Holder) { MCSymbol *TableStart = Asm->createTempSymbol("debug_rnglist_table_start"); MCSymbol *TableEnd = Asm->createTempSymbol("debug_rnglist_table_end"); emitListsTableHeaderStart(Asm, Holder, TableStart, TableEnd); Asm->OutStreamer->AddComment("Offset entry count"); Asm->emitInt32(Holder.getRangeLists().size()); Asm->OutStreamer->EmitLabel(Holder.getRnglistsTableBaseSym()); for (const RangeSpanList &List : Holder.getRangeLists()) Asm->EmitLabelDifference(List.getSym(), Holder.getRnglistsTableBaseSym(), 4); return TableEnd; } // Emit the header of a DWARF 5 locations list table. Returns the symbol that // designates the end of the table for the caller to emit when the table is // complete. static MCSymbol *emitLoclistsTableHeader(AsmPrinter *Asm, const DwarfFile &Holder) { MCSymbol *TableStart = Asm->createTempSymbol("debug_loclist_table_start"); MCSymbol *TableEnd = Asm->createTempSymbol("debug_loclist_table_end"); emitListsTableHeaderStart(Asm, Holder, TableStart, TableEnd); // FIXME: Generate the offsets table and use DW_FORM_loclistx with the // DW_AT_loclists_base attribute. Until then set the number of offsets to 0. Asm->OutStreamer->AddComment("Offset entry count"); Asm->emitInt32(0); Asm->OutStreamer->EmitLabel(Holder.getLoclistsTableBaseSym()); return TableEnd; } // Emit locations into the .debug_loc/.debug_rnglists section. void DwarfDebug::emitDebugLoc() { if (DebugLocs.getLists().empty()) return; bool IsLocLists = getDwarfVersion() >= 5; MCSymbol *TableEnd = nullptr; if (IsLocLists) { Asm->OutStreamer->SwitchSection( Asm->getObjFileLowering().getDwarfLoclistsSection()); TableEnd = emitLoclistsTableHeader(Asm, useSplitDwarf() ? SkeletonHolder : InfoHolder); } else { Asm->OutStreamer->SwitchSection( Asm->getObjFileLowering().getDwarfLocSection()); } unsigned char Size = Asm->MAI->getCodePointerSize(); for (const auto &List : DebugLocs.getLists()) { Asm->OutStreamer->EmitLabel(List.Label); const DwarfCompileUnit *CU = List.CU; const MCSymbol *Base = CU->getBaseAddress(); for (const auto &Entry : DebugLocs.getEntries(List)) { if (Base) { // Set up the range. This range is relative to the entry point of the // compile unit. This is a hard coded 0 for low_pc when we're emitting // ranges, or the DW_AT_low_pc on the compile unit otherwise. if (IsLocLists) { Asm->OutStreamer->AddComment("DW_LLE_offset_pair"); Asm->OutStreamer->EmitIntValue(dwarf::DW_LLE_offset_pair, 1); Asm->OutStreamer->AddComment(" starting offset"); Asm->EmitLabelDifferenceAsULEB128(Entry.BeginSym, Base); Asm->OutStreamer->AddComment(" ending offset"); Asm->EmitLabelDifferenceAsULEB128(Entry.EndSym, Base); } else { Asm->EmitLabelDifference(Entry.BeginSym, Base, Size); Asm->EmitLabelDifference(Entry.EndSym, Base, Size); } emitDebugLocEntryLocation(Entry); continue; } // We have no base address. if (IsLocLists) { // TODO: Use DW_LLE_base_addressx + DW_LLE_offset_pair, or // DW_LLE_startx_length in case if there is only a single range. // That should reduce the size of the debug data emited. // For now just use the DW_LLE_startx_length for all cases. Asm->OutStreamer->AddComment("DW_LLE_startx_length"); Asm->emitInt8(dwarf::DW_LLE_startx_length); Asm->OutStreamer->AddComment(" start idx"); Asm->EmitULEB128(AddrPool.getIndex(Entry.BeginSym)); Asm->OutStreamer->AddComment(" length"); Asm->EmitLabelDifferenceAsULEB128(Entry.EndSym, Entry.BeginSym); } else { Asm->OutStreamer->EmitSymbolValue(Entry.BeginSym, Size); Asm->OutStreamer->EmitSymbolValue(Entry.EndSym, Size); } emitDebugLocEntryLocation(Entry); } if (IsLocLists) { // .debug_loclists section ends with DW_LLE_end_of_list. Asm->OutStreamer->AddComment("DW_LLE_end_of_list"); Asm->OutStreamer->EmitIntValue(dwarf::DW_LLE_end_of_list, 1); } else { // Terminate the .debug_loc list with two 0 values. Asm->OutStreamer->EmitIntValue(0, Size); Asm->OutStreamer->EmitIntValue(0, Size); } } if (TableEnd) Asm->OutStreamer->EmitLabel(TableEnd); } void DwarfDebug::emitDebugLocDWO() { Asm->OutStreamer->SwitchSection( Asm->getObjFileLowering().getDwarfLocDWOSection()); for (const auto &List : DebugLocs.getLists()) { Asm->OutStreamer->EmitLabel(List.Label); for (const auto &Entry : DebugLocs.getEntries(List)) { // GDB only supports startx_length in pre-standard split-DWARF. // (in v5 standard loclists, it currently* /only/ supports base_address + // offset_pair, so the implementations can't really share much since they // need to use different representations) // * as of October 2018, at least // Ideally/in v5, this could use SectionLabels to reuse existing addresses // in the address pool to minimize object size/relocations. Asm->emitInt8(dwarf::DW_LLE_startx_length); unsigned idx = AddrPool.getIndex(Entry.BeginSym); Asm->EmitULEB128(idx); Asm->EmitLabelDifference(Entry.EndSym, Entry.BeginSym, 4); emitDebugLocEntryLocation(Entry); } Asm->emitInt8(dwarf::DW_LLE_end_of_list); } } struct ArangeSpan { const MCSymbol *Start, *End; }; // Emit a debug aranges section, containing a CU lookup for any // address we can tie back to a CU. void DwarfDebug::emitDebugARanges() { // Provides a unique id per text section. MapVector> SectionMap; // Filter labels by section. for (const SymbolCU &SCU : ArangeLabels) { if (SCU.Sym->isInSection()) { // Make a note of this symbol and it's section. MCSection *Section = &SCU.Sym->getSection(); if (!Section->getKind().isMetadata()) SectionMap[Section].push_back(SCU); } else { // Some symbols (e.g. common/bss on mach-o) can have no section but still // appear in the output. This sucks as we rely on sections to build // arange spans. We can do it without, but it's icky. SectionMap[nullptr].push_back(SCU); } } DenseMap> Spans; for (auto &I : SectionMap) { MCSection *Section = I.first; SmallVector &List = I.second; if (List.size() < 1) continue; // If we have no section (e.g. common), just write out // individual spans for each symbol. if (!Section) { for (const SymbolCU &Cur : List) { ArangeSpan Span; Span.Start = Cur.Sym; Span.End = nullptr; assert(Cur.CU); Spans[Cur.CU].push_back(Span); } continue; } // Sort the symbols by offset within the section. std::stable_sort( List.begin(), List.end(), [&](const SymbolCU &A, const SymbolCU &B) { unsigned IA = A.Sym ? Asm->OutStreamer->GetSymbolOrder(A.Sym) : 0; unsigned IB = B.Sym ? Asm->OutStreamer->GetSymbolOrder(B.Sym) : 0; // Symbols with no order assigned should be placed at the end. // (e.g. section end labels) if (IA == 0) return false; if (IB == 0) return true; return IA < IB; }); // Insert a final terminator. List.push_back(SymbolCU(nullptr, Asm->OutStreamer->endSection(Section))); // Build spans between each label. const MCSymbol *StartSym = List[0].Sym; for (size_t n = 1, e = List.size(); n < e; n++) { const SymbolCU &Prev = List[n - 1]; const SymbolCU &Cur = List[n]; // Try and build the longest span we can within the same CU. if (Cur.CU != Prev.CU) { ArangeSpan Span; Span.Start = StartSym; Span.End = Cur.Sym; assert(Prev.CU); Spans[Prev.CU].push_back(Span); StartSym = Cur.Sym; } } } // Start the dwarf aranges section. Asm->OutStreamer->SwitchSection( Asm->getObjFileLowering().getDwarfARangesSection()); unsigned PtrSize = Asm->MAI->getCodePointerSize(); // Build a list of CUs used. std::vector CUs; for (const auto &it : Spans) { DwarfCompileUnit *CU = it.first; CUs.push_back(CU); } // Sort the CU list (again, to ensure consistent output order). llvm::sort(CUs, [](const DwarfCompileUnit *A, const DwarfCompileUnit *B) { return A->getUniqueID() < B->getUniqueID(); }); // Emit an arange table for each CU we used. for (DwarfCompileUnit *CU : CUs) { std::vector &List = Spans[CU]; // Describe the skeleton CU's offset and length, not the dwo file's. if (auto *Skel = CU->getSkeleton()) CU = Skel; // Emit size of content not including length itself. unsigned ContentSize = sizeof(int16_t) + // DWARF ARange version number sizeof(int32_t) + // Offset of CU in the .debug_info section sizeof(int8_t) + // Pointer Size (in bytes) sizeof(int8_t); // Segment Size (in bytes) unsigned TupleSize = PtrSize * 2; // 7.20 in the Dwarf specs requires the table to be aligned to a tuple. unsigned Padding = OffsetToAlignment(sizeof(int32_t) + ContentSize, TupleSize); ContentSize += Padding; ContentSize += (List.size() + 1) * TupleSize; // For each compile unit, write the list of spans it covers. Asm->OutStreamer->AddComment("Length of ARange Set"); Asm->emitInt32(ContentSize); Asm->OutStreamer->AddComment("DWARF Arange version number"); Asm->emitInt16(dwarf::DW_ARANGES_VERSION); Asm->OutStreamer->AddComment("Offset Into Debug Info Section"); emitSectionReference(*CU); Asm->OutStreamer->AddComment("Address Size (in bytes)"); Asm->emitInt8(PtrSize); Asm->OutStreamer->AddComment("Segment Size (in bytes)"); Asm->emitInt8(0); Asm->OutStreamer->emitFill(Padding, 0xff); for (const ArangeSpan &Span : List) { Asm->EmitLabelReference(Span.Start, PtrSize); // Calculate the size as being from the span start to it's end. if (Span.End) { Asm->EmitLabelDifference(Span.End, Span.Start, PtrSize); } else { // For symbols without an end marker (e.g. common), we // write a single arange entry containing just that one symbol. uint64_t Size = SymSize[Span.Start]; if (Size == 0) Size = 1; Asm->OutStreamer->EmitIntValue(Size, PtrSize); } } Asm->OutStreamer->AddComment("ARange terminator"); Asm->OutStreamer->EmitIntValue(0, PtrSize); Asm->OutStreamer->EmitIntValue(0, PtrSize); } } /// Emit a single range list. We handle both DWARF v5 and earlier. static void emitRangeList(DwarfDebug &DD, AsmPrinter *Asm, const RangeSpanList &List) { auto DwarfVersion = DD.getDwarfVersion(); // Emit our symbol so we can find the beginning of the range. Asm->OutStreamer->EmitLabel(List.getSym()); // Gather all the ranges that apply to the same section so they can share // a base address entry. MapVector> SectionRanges; // Size for our labels. auto Size = Asm->MAI->getCodePointerSize(); for (const RangeSpan &Range : List.getRanges()) SectionRanges[&Range.getStart()->getSection()].push_back(&Range); const DwarfCompileUnit &CU = List.getCU(); const MCSymbol *CUBase = CU.getBaseAddress(); bool BaseIsSet = false; for (const auto &P : SectionRanges) { // Don't bother with a base address entry if there's only one range in // this section in this range list - for example ranges for a CU will // usually consist of single regions from each of many sections // (-ffunction-sections, or just C++ inline functions) except under LTO // or optnone where there may be holes in a single CU's section // contributions. auto *Base = CUBase; if (!Base && (P.second.size() > 1 || DwarfVersion < 5) && (CU.getCUNode()->getRangesBaseAddress() || DwarfVersion >= 5)) { BaseIsSet = true; // FIXME/use care: This may not be a useful base address if it's not // the lowest address/range in this object. Base = P.second.front()->getStart(); if (DwarfVersion >= 5) { Base = DD.getSectionLabel(&Base->getSection()); Asm->OutStreamer->AddComment("DW_RLE_base_addressx"); Asm->OutStreamer->EmitIntValue(dwarf::DW_RLE_base_addressx, 1); Asm->OutStreamer->AddComment(" base address index"); Asm->EmitULEB128(DD.getAddressPool().getIndex(Base)); } else { Asm->OutStreamer->EmitIntValue(-1, Size); Asm->OutStreamer->AddComment(" base address"); Asm->OutStreamer->EmitSymbolValue(Base, Size); } } else if (BaseIsSet && DwarfVersion < 5) { BaseIsSet = false; assert(!Base); Asm->OutStreamer->EmitIntValue(-1, Size); Asm->OutStreamer->EmitIntValue(0, Size); } for (const auto *RS : P.second) { const MCSymbol *Begin = RS->getStart(); const MCSymbol *End = RS->getEnd(); assert(Begin && "Range without a begin symbol?"); assert(End && "Range without an end symbol?"); if (Base) { if (DwarfVersion >= 5) { // Emit DW_RLE_offset_pair when we have a base. Asm->OutStreamer->AddComment("DW_RLE_offset_pair"); Asm->OutStreamer->EmitIntValue(dwarf::DW_RLE_offset_pair, 1); Asm->OutStreamer->AddComment(" starting offset"); Asm->EmitLabelDifferenceAsULEB128(Begin, Base); Asm->OutStreamer->AddComment(" ending offset"); Asm->EmitLabelDifferenceAsULEB128(End, Base); } else { Asm->EmitLabelDifference(Begin, Base, Size); Asm->EmitLabelDifference(End, Base, Size); } } else if (DwarfVersion >= 5) { Asm->OutStreamer->AddComment("DW_RLE_startx_length"); Asm->OutStreamer->EmitIntValue(dwarf::DW_RLE_startx_length, 1); Asm->OutStreamer->AddComment(" start index"); Asm->EmitULEB128(DD.getAddressPool().getIndex(Begin)); Asm->OutStreamer->AddComment(" length"); Asm->EmitLabelDifferenceAsULEB128(End, Begin); } else { Asm->OutStreamer->EmitSymbolValue(Begin, Size); Asm->OutStreamer->EmitSymbolValue(End, Size); } } } if (DwarfVersion >= 5) { Asm->OutStreamer->AddComment("DW_RLE_end_of_list"); Asm->OutStreamer->EmitIntValue(dwarf::DW_RLE_end_of_list, 1); } else { // Terminate the list with two 0 values. Asm->OutStreamer->EmitIntValue(0, Size); Asm->OutStreamer->EmitIntValue(0, Size); } } static void emitDebugRangesImpl(DwarfDebug &DD, AsmPrinter *Asm, const DwarfFile &Holder, MCSymbol *TableEnd) { for (const RangeSpanList &List : Holder.getRangeLists()) emitRangeList(DD, Asm, List); if (TableEnd) Asm->OutStreamer->EmitLabel(TableEnd); } /// Emit address ranges into the .debug_ranges section or into the DWARF v5 /// .debug_rnglists section. void DwarfDebug::emitDebugRanges() { if (CUMap.empty()) return; const auto &Holder = useSplitDwarf() ? SkeletonHolder : InfoHolder; if (Holder.getRangeLists().empty()) return; assert(useRangesSection()); assert(llvm::none_of(CUMap, [](const decltype(CUMap)::value_type &Pair) { return Pair.second->getCUNode()->isDebugDirectivesOnly(); })); // Start the dwarf ranges section. MCSymbol *TableEnd = nullptr; if (getDwarfVersion() >= 5) { Asm->OutStreamer->SwitchSection( Asm->getObjFileLowering().getDwarfRnglistsSection()); TableEnd = emitRnglistsTableHeader(Asm, Holder); } else Asm->OutStreamer->SwitchSection( Asm->getObjFileLowering().getDwarfRangesSection()); emitDebugRangesImpl(*this, Asm, Holder, TableEnd); } void DwarfDebug::emitDebugRangesDWO() { assert(useSplitDwarf()); if (CUMap.empty()) return; const auto &Holder = InfoHolder; if (Holder.getRangeLists().empty()) return; assert(getDwarfVersion() >= 5); assert(useRangesSection()); assert(llvm::none_of(CUMap, [](const decltype(CUMap)::value_type &Pair) { return Pair.second->getCUNode()->isDebugDirectivesOnly(); })); // Start the dwarf ranges section. Asm->OutStreamer->SwitchSection( Asm->getObjFileLowering().getDwarfRnglistsDWOSection()); MCSymbol *TableEnd = emitRnglistsTableHeader(Asm, Holder); emitDebugRangesImpl(*this, Asm, Holder, TableEnd); } void DwarfDebug::handleMacroNodes(DIMacroNodeArray Nodes, DwarfCompileUnit &U) { for (auto *MN : Nodes) { if (auto *M = dyn_cast(MN)) emitMacro(*M); else if (auto *F = dyn_cast(MN)) emitMacroFile(*F, U); else llvm_unreachable("Unexpected DI type!"); } } void DwarfDebug::emitMacro(DIMacro &M) { Asm->EmitULEB128(M.getMacinfoType()); Asm->EmitULEB128(M.getLine()); StringRef Name = M.getName(); StringRef Value = M.getValue(); Asm->OutStreamer->EmitBytes(Name); if (!Value.empty()) { // There should be one space between macro name and macro value. Asm->emitInt8(' '); Asm->OutStreamer->EmitBytes(Value); } Asm->emitInt8('\0'); } void DwarfDebug::emitMacroFile(DIMacroFile &F, DwarfCompileUnit &U) { assert(F.getMacinfoType() == dwarf::DW_MACINFO_start_file); Asm->EmitULEB128(dwarf::DW_MACINFO_start_file); Asm->EmitULEB128(F.getLine()); Asm->EmitULEB128(U.getOrCreateSourceID(F.getFile())); handleMacroNodes(F.getElements(), U); Asm->EmitULEB128(dwarf::DW_MACINFO_end_file); } /// Emit macros into a debug macinfo section. void DwarfDebug::emitDebugMacinfo() { if (CUMap.empty()) return; if (llvm::all_of(CUMap, [](const decltype(CUMap)::value_type &Pair) { return Pair.second->getCUNode()->isDebugDirectivesOnly(); })) return; // Start the dwarf macinfo section. Asm->OutStreamer->SwitchSection( Asm->getObjFileLowering().getDwarfMacinfoSection()); for (const auto &P : CUMap) { auto &TheCU = *P.second; if (TheCU.getCUNode()->isDebugDirectivesOnly()) continue; auto *SkCU = TheCU.getSkeleton(); DwarfCompileUnit &U = SkCU ? *SkCU : TheCU; auto *CUNode = cast(P.first); DIMacroNodeArray Macros = CUNode->getMacros(); if (!Macros.empty()) { Asm->OutStreamer->EmitLabel(U.getMacroLabelBegin()); handleMacroNodes(Macros, U); } } Asm->OutStreamer->AddComment("End Of Macro List Mark"); Asm->emitInt8(0); } // DWARF5 Experimental Separate Dwarf emitters. void DwarfDebug::initSkeletonUnit(const DwarfUnit &U, DIE &Die, std::unique_ptr NewU) { if (!CompilationDir.empty()) NewU->addString(Die, dwarf::DW_AT_comp_dir, CompilationDir); addGnuPubAttributes(*NewU, Die); SkeletonHolder.addUnit(std::move(NewU)); } DwarfCompileUnit &DwarfDebug::constructSkeletonCU(const DwarfCompileUnit &CU) { auto OwnedUnit = llvm::make_unique( CU.getUniqueID(), CU.getCUNode(), Asm, this, &SkeletonHolder); DwarfCompileUnit &NewCU = *OwnedUnit; NewCU.setSection(Asm->getObjFileLowering().getDwarfInfoSection()); NewCU.initStmtList(); if (useSegmentedStringOffsetsTable()) NewCU.addStringOffsetsStart(); initSkeletonUnit(CU, NewCU.getUnitDie(), std::move(OwnedUnit)); return NewCU; } // Emit the .debug_info.dwo section for separated dwarf. This contains the // compile units that would normally be in debug_info. void DwarfDebug::emitDebugInfoDWO() { assert(useSplitDwarf() && "No split dwarf debug info?"); // Don't emit relocations into the dwo file. InfoHolder.emitUnits(/* UseOffsets */ true); } // Emit the .debug_abbrev.dwo section for separated dwarf. This contains the // abbreviations for the .debug_info.dwo section. void DwarfDebug::emitDebugAbbrevDWO() { assert(useSplitDwarf() && "No split dwarf?"); InfoHolder.emitAbbrevs(Asm->getObjFileLowering().getDwarfAbbrevDWOSection()); } void DwarfDebug::emitDebugLineDWO() { assert(useSplitDwarf() && "No split dwarf?"); SplitTypeUnitFileTable.Emit( *Asm->OutStreamer, MCDwarfLineTableParams(), Asm->getObjFileLowering().getDwarfLineDWOSection()); } void DwarfDebug::emitStringOffsetsTableHeaderDWO() { assert(useSplitDwarf() && "No split dwarf?"); InfoHolder.getStringPool().emitStringOffsetsTableHeader( *Asm, Asm->getObjFileLowering().getDwarfStrOffDWOSection(), InfoHolder.getStringOffsetsStartSym()); } // Emit the .debug_str.dwo section for separated dwarf. This contains the // string section and is identical in format to traditional .debug_str // sections. void DwarfDebug::emitDebugStrDWO() { if (useSegmentedStringOffsetsTable()) emitStringOffsetsTableHeaderDWO(); assert(useSplitDwarf() && "No split dwarf?"); MCSection *OffSec = Asm->getObjFileLowering().getDwarfStrOffDWOSection(); InfoHolder.emitStrings(Asm->getObjFileLowering().getDwarfStrDWOSection(), OffSec, /* UseRelativeOffsets = */ false); } // Emit address pool. void DwarfDebug::emitDebugAddr() { AddrPool.emit(*Asm, Asm->getObjFileLowering().getDwarfAddrSection()); } MCDwarfDwoLineTable *DwarfDebug::getDwoLineTable(const DwarfCompileUnit &CU) { if (!useSplitDwarf()) return nullptr; const DICompileUnit *DIUnit = CU.getCUNode(); SplitTypeUnitFileTable.maybeSetRootFile( DIUnit->getDirectory(), DIUnit->getFilename(), CU.getMD5AsBytes(DIUnit->getFile()), DIUnit->getSource()); return &SplitTypeUnitFileTable; } uint64_t DwarfDebug::makeTypeSignature(StringRef Identifier) { MD5 Hash; Hash.update(Identifier); // ... take the least significant 8 bytes and return those. Our MD5 // implementation always returns its results in little endian, so we actually // need the "high" word. MD5::MD5Result Result; Hash.final(Result); return Result.high(); } void DwarfDebug::addDwarfTypeUnitType(DwarfCompileUnit &CU, StringRef Identifier, DIE &RefDie, const DICompositeType *CTy) { // Fast path if we're building some type units and one has already used the // address pool we know we're going to throw away all this work anyway, so // don't bother building dependent types. if (!TypeUnitsUnderConstruction.empty() && AddrPool.hasBeenUsed()) return; auto Ins = TypeSignatures.insert(std::make_pair(CTy, 0)); if (!Ins.second) { CU.addDIETypeSignature(RefDie, Ins.first->second); return; } bool TopLevelType = TypeUnitsUnderConstruction.empty(); AddrPool.resetUsedFlag(); auto OwnedUnit = llvm::make_unique(CU, Asm, this, &InfoHolder, getDwoLineTable(CU)); DwarfTypeUnit &NewTU = *OwnedUnit; DIE &UnitDie = NewTU.getUnitDie(); TypeUnitsUnderConstruction.emplace_back(std::move(OwnedUnit), CTy); NewTU.addUInt(UnitDie, dwarf::DW_AT_language, dwarf::DW_FORM_data2, CU.getLanguage()); uint64_t Signature = makeTypeSignature(Identifier); NewTU.setTypeSignature(Signature); Ins.first->second = Signature; if (useSplitDwarf()) { MCSection *Section = getDwarfVersion() <= 4 ? Asm->getObjFileLowering().getDwarfTypesDWOSection() : Asm->getObjFileLowering().getDwarfInfoDWOSection(); NewTU.setSection(Section); } else { MCSection *Section = getDwarfVersion() <= 4 ? Asm->getObjFileLowering().getDwarfTypesSection(Signature) : Asm->getObjFileLowering().getDwarfInfoSection(Signature); NewTU.setSection(Section); // Non-split type units reuse the compile unit's line table. CU.applyStmtList(UnitDie); } // Add DW_AT_str_offsets_base to the type unit DIE, but not for split type // units. if (useSegmentedStringOffsetsTable() && !useSplitDwarf()) NewTU.addStringOffsetsStart(); NewTU.setType(NewTU.createTypeDIE(CTy)); if (TopLevelType) { auto TypeUnitsToAdd = std::move(TypeUnitsUnderConstruction); TypeUnitsUnderConstruction.clear(); // Types referencing entries in the address table cannot be placed in type // units. if (AddrPool.hasBeenUsed()) { // Remove all the types built while building this type. // This is pessimistic as some of these types might not be dependent on // the type that used an address. for (const auto &TU : TypeUnitsToAdd) TypeSignatures.erase(TU.second); // Construct this type in the CU directly. // This is inefficient because all the dependent types will be rebuilt // from scratch, including building them in type units, discovering that // they depend on addresses, throwing them out and rebuilding them. CU.constructTypeDIE(RefDie, cast(CTy)); return; } // If the type wasn't dependent on fission addresses, finish adding the type // and all its dependent types. for (auto &TU : TypeUnitsToAdd) { InfoHolder.computeSizeAndOffsetsForUnit(TU.first.get()); InfoHolder.emitUnit(TU.first.get(), useSplitDwarf()); } } CU.addDIETypeSignature(RefDie, Signature); } // Add the Name along with its companion DIE to the appropriate accelerator // table (for AccelTableKind::Dwarf it's always AccelDebugNames, for // AccelTableKind::Apple, we use the table we got as an argument). If // accelerator tables are disabled, this function does nothing. template void DwarfDebug::addAccelNameImpl(const DICompileUnit &CU, AccelTable &AppleAccel, StringRef Name, const DIE &Die) { if (getAccelTableKind() == AccelTableKind::None) return; if (getAccelTableKind() != AccelTableKind::Apple && CU.getNameTableKind() == DICompileUnit::DebugNameTableKind::None) return; DwarfFile &Holder = useSplitDwarf() ? SkeletonHolder : InfoHolder; DwarfStringPoolEntryRef Ref = Holder.getStringPool().getEntry(*Asm, Name); switch (getAccelTableKind()) { case AccelTableKind::Apple: AppleAccel.addName(Ref, Die); break; case AccelTableKind::Dwarf: AccelDebugNames.addName(Ref, Die); break; case AccelTableKind::Default: llvm_unreachable("Default should have already been resolved."); case AccelTableKind::None: llvm_unreachable("None handled above"); } } void DwarfDebug::addAccelName(const DICompileUnit &CU, StringRef Name, const DIE &Die) { addAccelNameImpl(CU, AccelNames, Name, Die); } void DwarfDebug::addAccelObjC(const DICompileUnit &CU, StringRef Name, const DIE &Die) { // ObjC names go only into the Apple accelerator tables. if (getAccelTableKind() == AccelTableKind::Apple) addAccelNameImpl(CU, AccelObjC, Name, Die); } void DwarfDebug::addAccelNamespace(const DICompileUnit &CU, StringRef Name, const DIE &Die) { addAccelNameImpl(CU, AccelNamespace, Name, Die); } void DwarfDebug::addAccelType(const DICompileUnit &CU, StringRef Name, const DIE &Die, char Flags) { addAccelNameImpl(CU, AccelTypes, Name, Die); } uint16_t DwarfDebug::getDwarfVersion() const { return Asm->OutStreamer->getContext().getDwarfVersion(); } void DwarfDebug::addSectionLabel(const MCSymbol *Sym) { SectionLabels.insert(std::make_pair(&Sym->getSection(), Sym)); } const MCSymbol *DwarfDebug::getSectionLabel(const MCSection *S) { return SectionLabels.find(S)->second; } Index: vendor/llvm/dist-release_80/lib/CodeGen/MachineInstr.cpp =================================================================== --- vendor/llvm/dist-release_80/lib/CodeGen/MachineInstr.cpp (revision 343793) +++ vendor/llvm/dist-release_80/lib/CodeGen/MachineInstr.cpp (revision 343794) @@ -1,2102 +1,2103 @@ //===- lib/CodeGen/MachineInstr.cpp ---------------------------------------===// // // The LLVM Compiler Infrastructure // // This file is distributed under the University of Illinois Open Source // License. See LICENSE.TXT for details. // //===----------------------------------------------------------------------===// // // Methods common to all machine instructions. // //===----------------------------------------------------------------------===// #include "llvm/CodeGen/MachineInstr.h" #include "llvm/ADT/APFloat.h" #include "llvm/ADT/ArrayRef.h" #include "llvm/ADT/FoldingSet.h" #include "llvm/ADT/Hashing.h" #include "llvm/ADT/None.h" #include "llvm/ADT/STLExtras.h" #include "llvm/ADT/SmallBitVector.h" #include "llvm/ADT/SmallString.h" #include "llvm/ADT/SmallVector.h" #include "llvm/Analysis/AliasAnalysis.h" #include "llvm/Analysis/Loads.h" #include "llvm/Analysis/MemoryLocation.h" #include "llvm/CodeGen/GlobalISel/RegisterBank.h" #include "llvm/CodeGen/MachineBasicBlock.h" #include "llvm/CodeGen/MachineFunction.h" #include "llvm/CodeGen/MachineInstrBuilder.h" #include "llvm/CodeGen/MachineInstrBundle.h" #include "llvm/CodeGen/MachineMemOperand.h" #include "llvm/CodeGen/MachineModuleInfo.h" #include "llvm/CodeGen/MachineOperand.h" #include "llvm/CodeGen/MachineRegisterInfo.h" #include "llvm/CodeGen/PseudoSourceValue.h" #include "llvm/CodeGen/TargetInstrInfo.h" #include "llvm/CodeGen/TargetRegisterInfo.h" #include "llvm/CodeGen/TargetSubtargetInfo.h" #include "llvm/Config/llvm-config.h" #include "llvm/IR/Constants.h" #include "llvm/IR/DebugInfoMetadata.h" #include "llvm/IR/DebugLoc.h" #include "llvm/IR/DerivedTypes.h" #include "llvm/IR/Function.h" #include "llvm/IR/InlineAsm.h" #include "llvm/IR/InstrTypes.h" #include "llvm/IR/Intrinsics.h" #include "llvm/IR/LLVMContext.h" #include "llvm/IR/Metadata.h" #include "llvm/IR/Module.h" #include "llvm/IR/ModuleSlotTracker.h" #include "llvm/IR/Type.h" #include "llvm/IR/Value.h" #include "llvm/IR/Operator.h" #include "llvm/MC/MCInstrDesc.h" #include "llvm/MC/MCRegisterInfo.h" #include "llvm/MC/MCSymbol.h" #include "llvm/Support/Casting.h" #include "llvm/Support/CommandLine.h" #include "llvm/Support/Compiler.h" #include "llvm/Support/Debug.h" #include "llvm/Support/ErrorHandling.h" #include "llvm/Support/LowLevelTypeImpl.h" #include "llvm/Support/MathExtras.h" #include "llvm/Support/raw_ostream.h" #include "llvm/Target/TargetIntrinsicInfo.h" #include "llvm/Target/TargetMachine.h" #include #include #include #include #include #include #include using namespace llvm; static const MachineFunction *getMFIfAvailable(const MachineInstr &MI) { if (const MachineBasicBlock *MBB = MI.getParent()) if (const MachineFunction *MF = MBB->getParent()) return MF; return nullptr; } // Try to crawl up to the machine function and get TRI and IntrinsicInfo from // it. static void tryToGetTargetInfo(const MachineInstr &MI, const TargetRegisterInfo *&TRI, const MachineRegisterInfo *&MRI, const TargetIntrinsicInfo *&IntrinsicInfo, const TargetInstrInfo *&TII) { if (const MachineFunction *MF = getMFIfAvailable(MI)) { TRI = MF->getSubtarget().getRegisterInfo(); MRI = &MF->getRegInfo(); IntrinsicInfo = MF->getTarget().getIntrinsicInfo(); TII = MF->getSubtarget().getInstrInfo(); } } void MachineInstr::addImplicitDefUseOperands(MachineFunction &MF) { if (MCID->ImplicitDefs) for (const MCPhysReg *ImpDefs = MCID->getImplicitDefs(); *ImpDefs; ++ImpDefs) addOperand(MF, MachineOperand::CreateReg(*ImpDefs, true, true)); if (MCID->ImplicitUses) for (const MCPhysReg *ImpUses = MCID->getImplicitUses(); *ImpUses; ++ImpUses) addOperand(MF, MachineOperand::CreateReg(*ImpUses, false, true)); } /// MachineInstr ctor - This constructor creates a MachineInstr and adds the /// implicit operands. It reserves space for the number of operands specified by /// the MCInstrDesc. MachineInstr::MachineInstr(MachineFunction &MF, const MCInstrDesc &tid, DebugLoc dl, bool NoImp) : MCID(&tid), debugLoc(std::move(dl)) { assert(debugLoc.hasTrivialDestructor() && "Expected trivial destructor"); // Reserve space for the expected number of operands. if (unsigned NumOps = MCID->getNumOperands() + MCID->getNumImplicitDefs() + MCID->getNumImplicitUses()) { CapOperands = OperandCapacity::get(NumOps); Operands = MF.allocateOperandArray(CapOperands); } if (!NoImp) addImplicitDefUseOperands(MF); } /// MachineInstr ctor - Copies MachineInstr arg exactly /// MachineInstr::MachineInstr(MachineFunction &MF, const MachineInstr &MI) : MCID(&MI.getDesc()), Info(MI.Info), debugLoc(MI.getDebugLoc()) { assert(debugLoc.hasTrivialDestructor() && "Expected trivial destructor"); CapOperands = OperandCapacity::get(MI.getNumOperands()); Operands = MF.allocateOperandArray(CapOperands); // Copy operands. for (const MachineOperand &MO : MI.operands()) addOperand(MF, MO); // Copy all the sensible flags. setFlags(MI.Flags); } /// getRegInfo - If this instruction is embedded into a MachineFunction, /// return the MachineRegisterInfo object for the current function, otherwise /// return null. MachineRegisterInfo *MachineInstr::getRegInfo() { if (MachineBasicBlock *MBB = getParent()) return &MBB->getParent()->getRegInfo(); return nullptr; } /// RemoveRegOperandsFromUseLists - Unlink all of the register operands in /// this instruction from their respective use lists. This requires that the /// operands already be on their use lists. void MachineInstr::RemoveRegOperandsFromUseLists(MachineRegisterInfo &MRI) { for (MachineOperand &MO : operands()) if (MO.isReg()) MRI.removeRegOperandFromUseList(&MO); } /// AddRegOperandsToUseLists - Add all of the register operands in /// this instruction from their respective use lists. This requires that the /// operands not be on their use lists yet. void MachineInstr::AddRegOperandsToUseLists(MachineRegisterInfo &MRI) { for (MachineOperand &MO : operands()) if (MO.isReg()) MRI.addRegOperandToUseList(&MO); } void MachineInstr::addOperand(const MachineOperand &Op) { MachineBasicBlock *MBB = getParent(); assert(MBB && "Use MachineInstrBuilder to add operands to dangling instrs"); MachineFunction *MF = MBB->getParent(); assert(MF && "Use MachineInstrBuilder to add operands to dangling instrs"); addOperand(*MF, Op); } /// Move NumOps MachineOperands from Src to Dst, with support for overlapping /// ranges. If MRI is non-null also update use-def chains. static void moveOperands(MachineOperand *Dst, MachineOperand *Src, unsigned NumOps, MachineRegisterInfo *MRI) { if (MRI) return MRI->moveOperands(Dst, Src, NumOps); // MachineOperand is a trivially copyable type so we can just use memmove. std::memmove(Dst, Src, NumOps * sizeof(MachineOperand)); } /// addOperand - Add the specified operand to the instruction. If it is an /// implicit operand, it is added to the end of the operand list. If it is /// an explicit operand it is added at the end of the explicit operand list /// (before the first implicit operand). void MachineInstr::addOperand(MachineFunction &MF, const MachineOperand &Op) { assert(MCID && "Cannot add operands before providing an instr descriptor"); // Check if we're adding one of our existing operands. if (&Op >= Operands && &Op < Operands + NumOperands) { // This is unusual: MI->addOperand(MI->getOperand(i)). // If adding Op requires reallocating or moving existing operands around, // the Op reference could go stale. Support it by copying Op. MachineOperand CopyOp(Op); return addOperand(MF, CopyOp); } // Find the insert location for the new operand. Implicit registers go at // the end, everything else goes before the implicit regs. // // FIXME: Allow mixed explicit and implicit operands on inline asm. // InstrEmitter::EmitSpecialNode() is marking inline asm clobbers as // implicit-defs, but they must not be moved around. See the FIXME in // InstrEmitter.cpp. unsigned OpNo = getNumOperands(); bool isImpReg = Op.isReg() && Op.isImplicit(); if (!isImpReg && !isInlineAsm()) { while (OpNo && Operands[OpNo-1].isReg() && Operands[OpNo-1].isImplicit()) { --OpNo; assert(!Operands[OpNo].isTied() && "Cannot move tied operands"); } } #ifndef NDEBUG - bool isMetaDataOp = Op.getType() == MachineOperand::MO_Metadata; + bool isDebugOp = Op.getType() == MachineOperand::MO_Metadata || + Op.getType() == MachineOperand::MO_MCSymbol; // OpNo now points as the desired insertion point. Unless this is a variadic // instruction, only implicit regs are allowed beyond MCID->getNumOperands(). // RegMask operands go between the explicit and implicit operands. assert((isImpReg || Op.isRegMask() || MCID->isVariadic() || - OpNo < MCID->getNumOperands() || isMetaDataOp) && + OpNo < MCID->getNumOperands() || isDebugOp) && "Trying to add an operand to a machine instr that is already done!"); #endif MachineRegisterInfo *MRI = getRegInfo(); // Determine if the Operands array needs to be reallocated. // Save the old capacity and operand array. OperandCapacity OldCap = CapOperands; MachineOperand *OldOperands = Operands; if (!OldOperands || OldCap.getSize() == getNumOperands()) { CapOperands = OldOperands ? OldCap.getNext() : OldCap.get(1); Operands = MF.allocateOperandArray(CapOperands); // Move the operands before the insertion point. if (OpNo) moveOperands(Operands, OldOperands, OpNo, MRI); } // Move the operands following the insertion point. if (OpNo != NumOperands) moveOperands(Operands + OpNo + 1, OldOperands + OpNo, NumOperands - OpNo, MRI); ++NumOperands; // Deallocate the old operand array. if (OldOperands != Operands && OldOperands) MF.deallocateOperandArray(OldCap, OldOperands); // Copy Op into place. It still needs to be inserted into the MRI use lists. MachineOperand *NewMO = new (Operands + OpNo) MachineOperand(Op); NewMO->ParentMI = this; // When adding a register operand, tell MRI about it. if (NewMO->isReg()) { // Ensure isOnRegUseList() returns false, regardless of Op's status. NewMO->Contents.Reg.Prev = nullptr; // Ignore existing ties. This is not a property that can be copied. NewMO->TiedTo = 0; // Add the new operand to MRI, but only for instructions in an MBB. if (MRI) MRI->addRegOperandToUseList(NewMO); // The MCID operand information isn't accurate until we start adding // explicit operands. The implicit operands are added first, then the // explicits are inserted before them. if (!isImpReg) { // Tie uses to defs as indicated in MCInstrDesc. if (NewMO->isUse()) { int DefIdx = MCID->getOperandConstraint(OpNo, MCOI::TIED_TO); if (DefIdx != -1) tieOperands(DefIdx, OpNo); } // If the register operand is flagged as early, mark the operand as such. if (MCID->getOperandConstraint(OpNo, MCOI::EARLY_CLOBBER) != -1) NewMO->setIsEarlyClobber(true); } } } /// RemoveOperand - Erase an operand from an instruction, leaving it with one /// fewer operand than it started with. /// void MachineInstr::RemoveOperand(unsigned OpNo) { assert(OpNo < getNumOperands() && "Invalid operand number"); untieRegOperand(OpNo); #ifndef NDEBUG // Moving tied operands would break the ties. for (unsigned i = OpNo + 1, e = getNumOperands(); i != e; ++i) if (Operands[i].isReg()) assert(!Operands[i].isTied() && "Cannot move tied operands"); #endif MachineRegisterInfo *MRI = getRegInfo(); if (MRI && Operands[OpNo].isReg()) MRI->removeRegOperandFromUseList(Operands + OpNo); // Don't call the MachineOperand destructor. A lot of this code depends on // MachineOperand having a trivial destructor anyway, and adding a call here // wouldn't make it 'destructor-correct'. if (unsigned N = NumOperands - 1 - OpNo) moveOperands(Operands + OpNo, Operands + OpNo + 1, N, MRI); --NumOperands; } void MachineInstr::dropMemRefs(MachineFunction &MF) { if (memoperands_empty()) return; // See if we can just drop all of our extra info. if (!getPreInstrSymbol() && !getPostInstrSymbol()) { Info.clear(); return; } if (!getPostInstrSymbol()) { Info.set(getPreInstrSymbol()); return; } if (!getPreInstrSymbol()) { Info.set(getPostInstrSymbol()); return; } // Otherwise allocate a fresh extra info with just these symbols. Info.set( MF.createMIExtraInfo({}, getPreInstrSymbol(), getPostInstrSymbol())); } void MachineInstr::setMemRefs(MachineFunction &MF, ArrayRef MMOs) { if (MMOs.empty()) { dropMemRefs(MF); return; } // Try to store a single MMO inline. if (MMOs.size() == 1 && !getPreInstrSymbol() && !getPostInstrSymbol()) { Info.set(MMOs[0]); return; } // Otherwise create an extra info struct with all of our info. Info.set( MF.createMIExtraInfo(MMOs, getPreInstrSymbol(), getPostInstrSymbol())); } void MachineInstr::addMemOperand(MachineFunction &MF, MachineMemOperand *MO) { SmallVector MMOs; MMOs.append(memoperands_begin(), memoperands_end()); MMOs.push_back(MO); setMemRefs(MF, MMOs); } void MachineInstr::cloneMemRefs(MachineFunction &MF, const MachineInstr &MI) { if (this == &MI) // Nothing to do for a self-clone! return; assert(&MF == MI.getMF() && "Invalid machine functions when cloning memory refrences!"); // See if we can just steal the extra info already allocated for the // instruction. We can do this whenever the pre- and post-instruction symbols // are the same (including null). if (getPreInstrSymbol() == MI.getPreInstrSymbol() && getPostInstrSymbol() == MI.getPostInstrSymbol()) { Info = MI.Info; return; } // Otherwise, fall back on a copy-based clone. setMemRefs(MF, MI.memoperands()); } /// Check to see if the MMOs pointed to by the two MemRefs arrays are /// identical. static bool hasIdenticalMMOs(ArrayRef LHS, ArrayRef RHS) { if (LHS.size() != RHS.size()) return false; auto LHSPointees = make_pointee_range(LHS); auto RHSPointees = make_pointee_range(RHS); return std::equal(LHSPointees.begin(), LHSPointees.end(), RHSPointees.begin()); } void MachineInstr::cloneMergedMemRefs(MachineFunction &MF, ArrayRef MIs) { // Try handling easy numbers of MIs with simpler mechanisms. if (MIs.empty()) { dropMemRefs(MF); return; } if (MIs.size() == 1) { cloneMemRefs(MF, *MIs[0]); return; } // Because an empty memoperands list provides *no* information and must be // handled conservatively (assuming the instruction can do anything), the only // way to merge with it is to drop all other memoperands. if (MIs[0]->memoperands_empty()) { dropMemRefs(MF); return; } // Handle the general case. SmallVector MergedMMOs; // Start with the first instruction. assert(&MF == MIs[0]->getMF() && "Invalid machine functions when cloning memory references!"); MergedMMOs.append(MIs[0]->memoperands_begin(), MIs[0]->memoperands_end()); // Now walk all the other instructions and accumulate any different MMOs. for (const MachineInstr &MI : make_pointee_range(MIs.slice(1))) { assert(&MF == MI.getMF() && "Invalid machine functions when cloning memory references!"); // Skip MIs with identical operands to the first. This is a somewhat // arbitrary hack but will catch common cases without being quadratic. // TODO: We could fully implement merge semantics here if needed. if (hasIdenticalMMOs(MIs[0]->memoperands(), MI.memoperands())) continue; // Because an empty memoperands list provides *no* information and must be // handled conservatively (assuming the instruction can do anything), the // only way to merge with it is to drop all other memoperands. if (MI.memoperands_empty()) { dropMemRefs(MF); return; } // Otherwise accumulate these into our temporary buffer of the merged state. MergedMMOs.append(MI.memoperands_begin(), MI.memoperands_end()); } setMemRefs(MF, MergedMMOs); } void MachineInstr::setPreInstrSymbol(MachineFunction &MF, MCSymbol *Symbol) { MCSymbol *OldSymbol = getPreInstrSymbol(); if (OldSymbol == Symbol) return; if (OldSymbol && !Symbol) { // We're removing a symbol rather than adding one. Try to clean up any // extra info carried around. if (Info.is()) { Info.clear(); return; } if (memoperands_empty()) { assert(getPostInstrSymbol() && "Should never have only a single symbol allocated out-of-line!"); Info.set(getPostInstrSymbol()); return; } // Otherwise fallback on the generic update. } else if (!Info || Info.is()) { // If we don't have any other extra info, we can store this inline. Info.set(Symbol); return; } // Otherwise, allocate a full new set of extra info. // FIXME: Maybe we should make the symbols in the extra info mutable? Info.set( MF.createMIExtraInfo(memoperands(), Symbol, getPostInstrSymbol())); } void MachineInstr::setPostInstrSymbol(MachineFunction &MF, MCSymbol *Symbol) { MCSymbol *OldSymbol = getPostInstrSymbol(); if (OldSymbol == Symbol) return; if (OldSymbol && !Symbol) { // We're removing a symbol rather than adding one. Try to clean up any // extra info carried around. if (Info.is()) { Info.clear(); return; } if (memoperands_empty()) { assert(getPreInstrSymbol() && "Should never have only a single symbol allocated out-of-line!"); Info.set(getPreInstrSymbol()); return; } // Otherwise fallback on the generic update. } else if (!Info || Info.is()) { // If we don't have any other extra info, we can store this inline. Info.set(Symbol); return; } // Otherwise, allocate a full new set of extra info. // FIXME: Maybe we should make the symbols in the extra info mutable? Info.set( MF.createMIExtraInfo(memoperands(), getPreInstrSymbol(), Symbol)); } uint16_t MachineInstr::mergeFlagsWith(const MachineInstr &Other) const { // For now, the just return the union of the flags. If the flags get more // complicated over time, we might need more logic here. return getFlags() | Other.getFlags(); } void MachineInstr::copyIRFlags(const Instruction &I) { // Copy the wrapping flags. if (const OverflowingBinaryOperator *OB = dyn_cast(&I)) { if (OB->hasNoSignedWrap()) setFlag(MachineInstr::MIFlag::NoSWrap); if (OB->hasNoUnsignedWrap()) setFlag(MachineInstr::MIFlag::NoUWrap); } // Copy the exact flag. if (const PossiblyExactOperator *PE = dyn_cast(&I)) if (PE->isExact()) setFlag(MachineInstr::MIFlag::IsExact); // Copy the fast-math flags. if (const FPMathOperator *FP = dyn_cast(&I)) { const FastMathFlags Flags = FP->getFastMathFlags(); if (Flags.noNaNs()) setFlag(MachineInstr::MIFlag::FmNoNans); if (Flags.noInfs()) setFlag(MachineInstr::MIFlag::FmNoInfs); if (Flags.noSignedZeros()) setFlag(MachineInstr::MIFlag::FmNsz); if (Flags.allowReciprocal()) setFlag(MachineInstr::MIFlag::FmArcp); if (Flags.allowContract()) setFlag(MachineInstr::MIFlag::FmContract); if (Flags.approxFunc()) setFlag(MachineInstr::MIFlag::FmAfn); if (Flags.allowReassoc()) setFlag(MachineInstr::MIFlag::FmReassoc); } } bool MachineInstr::hasPropertyInBundle(uint64_t Mask, QueryType Type) const { assert(!isBundledWithPred() && "Must be called on bundle header"); for (MachineBasicBlock::const_instr_iterator MII = getIterator();; ++MII) { if (MII->getDesc().getFlags() & Mask) { if (Type == AnyInBundle) return true; } else { if (Type == AllInBundle && !MII->isBundle()) return false; } // This was the last instruction in the bundle. if (!MII->isBundledWithSucc()) return Type == AllInBundle; } } bool MachineInstr::isIdenticalTo(const MachineInstr &Other, MICheckType Check) const { // If opcodes or number of operands are not the same then the two // instructions are obviously not identical. if (Other.getOpcode() != getOpcode() || Other.getNumOperands() != getNumOperands()) return false; if (isBundle()) { // We have passed the test above that both instructions have the same // opcode, so we know that both instructions are bundles here. Let's compare // MIs inside the bundle. assert(Other.isBundle() && "Expected that both instructions are bundles."); MachineBasicBlock::const_instr_iterator I1 = getIterator(); MachineBasicBlock::const_instr_iterator I2 = Other.getIterator(); // Loop until we analysed the last intruction inside at least one of the // bundles. while (I1->isBundledWithSucc() && I2->isBundledWithSucc()) { ++I1; ++I2; if (!I1->isIdenticalTo(*I2, Check)) return false; } // If we've reached the end of just one of the two bundles, but not both, // the instructions are not identical. if (I1->isBundledWithSucc() || I2->isBundledWithSucc()) return false; } // Check operands to make sure they match. for (unsigned i = 0, e = getNumOperands(); i != e; ++i) { const MachineOperand &MO = getOperand(i); const MachineOperand &OMO = Other.getOperand(i); if (!MO.isReg()) { if (!MO.isIdenticalTo(OMO)) return false; continue; } // Clients may or may not want to ignore defs when testing for equality. // For example, machine CSE pass only cares about finding common // subexpressions, so it's safe to ignore virtual register defs. if (MO.isDef()) { if (Check == IgnoreDefs) continue; else if (Check == IgnoreVRegDefs) { if (!TargetRegisterInfo::isVirtualRegister(MO.getReg()) || !TargetRegisterInfo::isVirtualRegister(OMO.getReg())) if (!MO.isIdenticalTo(OMO)) return false; } else { if (!MO.isIdenticalTo(OMO)) return false; if (Check == CheckKillDead && MO.isDead() != OMO.isDead()) return false; } } else { if (!MO.isIdenticalTo(OMO)) return false; if (Check == CheckKillDead && MO.isKill() != OMO.isKill()) return false; } } // If DebugLoc does not match then two debug instructions are not identical. if (isDebugInstr()) if (getDebugLoc() && Other.getDebugLoc() && getDebugLoc() != Other.getDebugLoc()) return false; return true; } const MachineFunction *MachineInstr::getMF() const { return getParent()->getParent(); } MachineInstr *MachineInstr::removeFromParent() { assert(getParent() && "Not embedded in a basic block!"); return getParent()->remove(this); } MachineInstr *MachineInstr::removeFromBundle() { assert(getParent() && "Not embedded in a basic block!"); return getParent()->remove_instr(this); } void MachineInstr::eraseFromParent() { assert(getParent() && "Not embedded in a basic block!"); getParent()->erase(this); } void MachineInstr::eraseFromParentAndMarkDBGValuesForRemoval() { assert(getParent() && "Not embedded in a basic block!"); MachineBasicBlock *MBB = getParent(); MachineFunction *MF = MBB->getParent(); assert(MF && "Not embedded in a function!"); MachineInstr *MI = (MachineInstr *)this; MachineRegisterInfo &MRI = MF->getRegInfo(); for (const MachineOperand &MO : MI->operands()) { if (!MO.isReg() || !MO.isDef()) continue; unsigned Reg = MO.getReg(); if (!TargetRegisterInfo::isVirtualRegister(Reg)) continue; MRI.markUsesInDebugValueAsUndef(Reg); } MI->eraseFromParent(); } void MachineInstr::eraseFromBundle() { assert(getParent() && "Not embedded in a basic block!"); getParent()->erase_instr(this); } unsigned MachineInstr::getNumExplicitOperands() const { unsigned NumOperands = MCID->getNumOperands(); if (!MCID->isVariadic()) return NumOperands; for (unsigned I = NumOperands, E = getNumOperands(); I != E; ++I) { const MachineOperand &MO = getOperand(I); // The operands must always be in the following order: // - explicit reg defs, // - other explicit operands (reg uses, immediates, etc.), // - implicit reg defs // - implicit reg uses if (MO.isReg() && MO.isImplicit()) break; ++NumOperands; } return NumOperands; } unsigned MachineInstr::getNumExplicitDefs() const { unsigned NumDefs = MCID->getNumDefs(); if (!MCID->isVariadic()) return NumDefs; for (unsigned I = NumDefs, E = getNumOperands(); I != E; ++I) { const MachineOperand &MO = getOperand(I); if (!MO.isReg() || !MO.isDef() || MO.isImplicit()) break; ++NumDefs; } return NumDefs; } void MachineInstr::bundleWithPred() { assert(!isBundledWithPred() && "MI is already bundled with its predecessor"); setFlag(BundledPred); MachineBasicBlock::instr_iterator Pred = getIterator(); --Pred; assert(!Pred->isBundledWithSucc() && "Inconsistent bundle flags"); Pred->setFlag(BundledSucc); } void MachineInstr::bundleWithSucc() { assert(!isBundledWithSucc() && "MI is already bundled with its successor"); setFlag(BundledSucc); MachineBasicBlock::instr_iterator Succ = getIterator(); ++Succ; assert(!Succ->isBundledWithPred() && "Inconsistent bundle flags"); Succ->setFlag(BundledPred); } void MachineInstr::unbundleFromPred() { assert(isBundledWithPred() && "MI isn't bundled with its predecessor"); clearFlag(BundledPred); MachineBasicBlock::instr_iterator Pred = getIterator(); --Pred; assert(Pred->isBundledWithSucc() && "Inconsistent bundle flags"); Pred->clearFlag(BundledSucc); } void MachineInstr::unbundleFromSucc() { assert(isBundledWithSucc() && "MI isn't bundled with its successor"); clearFlag(BundledSucc); MachineBasicBlock::instr_iterator Succ = getIterator(); ++Succ; assert(Succ->isBundledWithPred() && "Inconsistent bundle flags"); Succ->clearFlag(BundledPred); } bool MachineInstr::isStackAligningInlineAsm() const { if (isInlineAsm()) { unsigned ExtraInfo = getOperand(InlineAsm::MIOp_ExtraInfo).getImm(); if (ExtraInfo & InlineAsm::Extra_IsAlignStack) return true; } return false; } InlineAsm::AsmDialect MachineInstr::getInlineAsmDialect() const { assert(isInlineAsm() && "getInlineAsmDialect() only works for inline asms!"); unsigned ExtraInfo = getOperand(InlineAsm::MIOp_ExtraInfo).getImm(); return InlineAsm::AsmDialect((ExtraInfo & InlineAsm::Extra_AsmDialect) != 0); } int MachineInstr::findInlineAsmFlagIdx(unsigned OpIdx, unsigned *GroupNo) const { assert(isInlineAsm() && "Expected an inline asm instruction"); assert(OpIdx < getNumOperands() && "OpIdx out of range"); // Ignore queries about the initial operands. if (OpIdx < InlineAsm::MIOp_FirstOperand) return -1; unsigned Group = 0; unsigned NumOps; for (unsigned i = InlineAsm::MIOp_FirstOperand, e = getNumOperands(); i < e; i += NumOps) { const MachineOperand &FlagMO = getOperand(i); // If we reach the implicit register operands, stop looking. if (!FlagMO.isImm()) return -1; NumOps = 1 + InlineAsm::getNumOperandRegisters(FlagMO.getImm()); if (i + NumOps > OpIdx) { if (GroupNo) *GroupNo = Group; return i; } ++Group; } return -1; } const DILabel *MachineInstr::getDebugLabel() const { assert(isDebugLabel() && "not a DBG_LABEL"); return cast(getOperand(0).getMetadata()); } const DILocalVariable *MachineInstr::getDebugVariable() const { assert(isDebugValue() && "not a DBG_VALUE"); return cast(getOperand(2).getMetadata()); } const DIExpression *MachineInstr::getDebugExpression() const { assert(isDebugValue() && "not a DBG_VALUE"); return cast(getOperand(3).getMetadata()); } const TargetRegisterClass* MachineInstr::getRegClassConstraint(unsigned OpIdx, const TargetInstrInfo *TII, const TargetRegisterInfo *TRI) const { assert(getParent() && "Can't have an MBB reference here!"); assert(getMF() && "Can't have an MF reference here!"); const MachineFunction &MF = *getMF(); // Most opcodes have fixed constraints in their MCInstrDesc. if (!isInlineAsm()) return TII->getRegClass(getDesc(), OpIdx, TRI, MF); if (!getOperand(OpIdx).isReg()) return nullptr; // For tied uses on inline asm, get the constraint from the def. unsigned DefIdx; if (getOperand(OpIdx).isUse() && isRegTiedToDefOperand(OpIdx, &DefIdx)) OpIdx = DefIdx; // Inline asm stores register class constraints in the flag word. int FlagIdx = findInlineAsmFlagIdx(OpIdx); if (FlagIdx < 0) return nullptr; unsigned Flag = getOperand(FlagIdx).getImm(); unsigned RCID; if ((InlineAsm::getKind(Flag) == InlineAsm::Kind_RegUse || InlineAsm::getKind(Flag) == InlineAsm::Kind_RegDef || InlineAsm::getKind(Flag) == InlineAsm::Kind_RegDefEarlyClobber) && InlineAsm::hasRegClassConstraint(Flag, RCID)) return TRI->getRegClass(RCID); // Assume that all registers in a memory operand are pointers. if (InlineAsm::getKind(Flag) == InlineAsm::Kind_Mem) return TRI->getPointerRegClass(MF); return nullptr; } const TargetRegisterClass *MachineInstr::getRegClassConstraintEffectForVReg( unsigned Reg, const TargetRegisterClass *CurRC, const TargetInstrInfo *TII, const TargetRegisterInfo *TRI, bool ExploreBundle) const { // Check every operands inside the bundle if we have // been asked to. if (ExploreBundle) for (ConstMIBundleOperands OpndIt(*this); OpndIt.isValid() && CurRC; ++OpndIt) CurRC = OpndIt->getParent()->getRegClassConstraintEffectForVRegImpl( OpndIt.getOperandNo(), Reg, CurRC, TII, TRI); else // Otherwise, just check the current operands. for (unsigned i = 0, e = NumOperands; i < e && CurRC; ++i) CurRC = getRegClassConstraintEffectForVRegImpl(i, Reg, CurRC, TII, TRI); return CurRC; } const TargetRegisterClass *MachineInstr::getRegClassConstraintEffectForVRegImpl( unsigned OpIdx, unsigned Reg, const TargetRegisterClass *CurRC, const TargetInstrInfo *TII, const TargetRegisterInfo *TRI) const { assert(CurRC && "Invalid initial register class"); // Check if Reg is constrained by some of its use/def from MI. const MachineOperand &MO = getOperand(OpIdx); if (!MO.isReg() || MO.getReg() != Reg) return CurRC; // If yes, accumulate the constraints through the operand. return getRegClassConstraintEffect(OpIdx, CurRC, TII, TRI); } const TargetRegisterClass *MachineInstr::getRegClassConstraintEffect( unsigned OpIdx, const TargetRegisterClass *CurRC, const TargetInstrInfo *TII, const TargetRegisterInfo *TRI) const { const TargetRegisterClass *OpRC = getRegClassConstraint(OpIdx, TII, TRI); const MachineOperand &MO = getOperand(OpIdx); assert(MO.isReg() && "Cannot get register constraints for non-register operand"); assert(CurRC && "Invalid initial register class"); if (unsigned SubIdx = MO.getSubReg()) { if (OpRC) CurRC = TRI->getMatchingSuperRegClass(CurRC, OpRC, SubIdx); else CurRC = TRI->getSubClassWithSubReg(CurRC, SubIdx); } else if (OpRC) CurRC = TRI->getCommonSubClass(CurRC, OpRC); return CurRC; } /// Return the number of instructions inside the MI bundle, not counting the /// header instruction. unsigned MachineInstr::getBundleSize() const { MachineBasicBlock::const_instr_iterator I = getIterator(); unsigned Size = 0; while (I->isBundledWithSucc()) { ++Size; ++I; } return Size; } /// Returns true if the MachineInstr has an implicit-use operand of exactly /// the given register (not considering sub/super-registers). bool MachineInstr::hasRegisterImplicitUseOperand(unsigned Reg) const { for (unsigned i = 0, e = getNumOperands(); i != e; ++i) { const MachineOperand &MO = getOperand(i); if (MO.isReg() && MO.isUse() && MO.isImplicit() && MO.getReg() == Reg) return true; } return false; } /// findRegisterUseOperandIdx() - Returns the MachineOperand that is a use of /// the specific register or -1 if it is not found. It further tightens /// the search criteria to a use that kills the register if isKill is true. int MachineInstr::findRegisterUseOperandIdx( unsigned Reg, bool isKill, const TargetRegisterInfo *TRI) const { for (unsigned i = 0, e = getNumOperands(); i != e; ++i) { const MachineOperand &MO = getOperand(i); if (!MO.isReg() || !MO.isUse()) continue; unsigned MOReg = MO.getReg(); if (!MOReg) continue; if (MOReg == Reg || (TRI && Reg && MOReg && TRI->regsOverlap(MOReg, Reg))) if (!isKill || MO.isKill()) return i; } return -1; } /// readsWritesVirtualRegister - Return a pair of bools (reads, writes) /// indicating if this instruction reads or writes Reg. This also considers /// partial defines. std::pair MachineInstr::readsWritesVirtualRegister(unsigned Reg, SmallVectorImpl *Ops) const { bool PartDef = false; // Partial redefine. bool FullDef = false; // Full define. bool Use = false; for (unsigned i = 0, e = getNumOperands(); i != e; ++i) { const MachineOperand &MO = getOperand(i); if (!MO.isReg() || MO.getReg() != Reg) continue; if (Ops) Ops->push_back(i); if (MO.isUse()) Use |= !MO.isUndef(); else if (MO.getSubReg() && !MO.isUndef()) // A partial def undef doesn't count as reading the register. PartDef = true; else FullDef = true; } // A partial redefine uses Reg unless there is also a full define. return std::make_pair(Use || (PartDef && !FullDef), PartDef || FullDef); } /// findRegisterDefOperandIdx() - Returns the operand index that is a def of /// the specified register or -1 if it is not found. If isDead is true, defs /// that are not dead are skipped. If TargetRegisterInfo is non-null, then it /// also checks if there is a def of a super-register. int MachineInstr::findRegisterDefOperandIdx(unsigned Reg, bool isDead, bool Overlap, const TargetRegisterInfo *TRI) const { bool isPhys = TargetRegisterInfo::isPhysicalRegister(Reg); for (unsigned i = 0, e = getNumOperands(); i != e; ++i) { const MachineOperand &MO = getOperand(i); // Accept regmask operands when Overlap is set. // Ignore them when looking for a specific def operand (Overlap == false). if (isPhys && Overlap && MO.isRegMask() && MO.clobbersPhysReg(Reg)) return i; if (!MO.isReg() || !MO.isDef()) continue; unsigned MOReg = MO.getReg(); bool Found = (MOReg == Reg); if (!Found && TRI && isPhys && TargetRegisterInfo::isPhysicalRegister(MOReg)) { if (Overlap) Found = TRI->regsOverlap(MOReg, Reg); else Found = TRI->isSubRegister(MOReg, Reg); } if (Found && (!isDead || MO.isDead())) return i; } return -1; } /// findFirstPredOperandIdx() - Find the index of the first operand in the /// operand list that is used to represent the predicate. It returns -1 if /// none is found. int MachineInstr::findFirstPredOperandIdx() const { // Don't call MCID.findFirstPredOperandIdx() because this variant // is sometimes called on an instruction that's not yet complete, and // so the number of operands is less than the MCID indicates. In // particular, the PTX target does this. const MCInstrDesc &MCID = getDesc(); if (MCID.isPredicable()) { for (unsigned i = 0, e = getNumOperands(); i != e; ++i) if (MCID.OpInfo[i].isPredicate()) return i; } return -1; } // MachineOperand::TiedTo is 4 bits wide. const unsigned TiedMax = 15; /// tieOperands - Mark operands at DefIdx and UseIdx as tied to each other. /// /// Use and def operands can be tied together, indicated by a non-zero TiedTo /// field. TiedTo can have these values: /// /// 0: Operand is not tied to anything. /// 1 to TiedMax-1: Tied to getOperand(TiedTo-1). /// TiedMax: Tied to an operand >= TiedMax-1. /// /// The tied def must be one of the first TiedMax operands on a normal /// instruction. INLINEASM instructions allow more tied defs. /// void MachineInstr::tieOperands(unsigned DefIdx, unsigned UseIdx) { MachineOperand &DefMO = getOperand(DefIdx); MachineOperand &UseMO = getOperand(UseIdx); assert(DefMO.isDef() && "DefIdx must be a def operand"); assert(UseMO.isUse() && "UseIdx must be a use operand"); assert(!DefMO.isTied() && "Def is already tied to another use"); assert(!UseMO.isTied() && "Use is already tied to another def"); if (DefIdx < TiedMax) UseMO.TiedTo = DefIdx + 1; else { // Inline asm can use the group descriptors to find tied operands, but on // normal instruction, the tied def must be within the first TiedMax // operands. assert(isInlineAsm() && "DefIdx out of range"); UseMO.TiedTo = TiedMax; } // UseIdx can be out of range, we'll search for it in findTiedOperandIdx(). DefMO.TiedTo = std::min(UseIdx + 1, TiedMax); } /// Given the index of a tied register operand, find the operand it is tied to. /// Defs are tied to uses and vice versa. Returns the index of the tied operand /// which must exist. unsigned MachineInstr::findTiedOperandIdx(unsigned OpIdx) const { const MachineOperand &MO = getOperand(OpIdx); assert(MO.isTied() && "Operand isn't tied"); // Normally TiedTo is in range. if (MO.TiedTo < TiedMax) return MO.TiedTo - 1; // Uses on normal instructions can be out of range. if (!isInlineAsm()) { // Normal tied defs must be in the 0..TiedMax-1 range. if (MO.isUse()) return TiedMax - 1; // MO is a def. Search for the tied use. for (unsigned i = TiedMax - 1, e = getNumOperands(); i != e; ++i) { const MachineOperand &UseMO = getOperand(i); if (UseMO.isReg() && UseMO.isUse() && UseMO.TiedTo == OpIdx + 1) return i; } llvm_unreachable("Can't find tied use"); } // Now deal with inline asm by parsing the operand group descriptor flags. // Find the beginning of each operand group. SmallVector GroupIdx; unsigned OpIdxGroup = ~0u; unsigned NumOps; for (unsigned i = InlineAsm::MIOp_FirstOperand, e = getNumOperands(); i < e; i += NumOps) { const MachineOperand &FlagMO = getOperand(i); assert(FlagMO.isImm() && "Invalid tied operand on inline asm"); unsigned CurGroup = GroupIdx.size(); GroupIdx.push_back(i); NumOps = 1 + InlineAsm::getNumOperandRegisters(FlagMO.getImm()); // OpIdx belongs to this operand group. if (OpIdx > i && OpIdx < i + NumOps) OpIdxGroup = CurGroup; unsigned TiedGroup; if (!InlineAsm::isUseOperandTiedToDef(FlagMO.getImm(), TiedGroup)) continue; // Operands in this group are tied to operands in TiedGroup which must be // earlier. Find the number of operands between the two groups. unsigned Delta = i - GroupIdx[TiedGroup]; // OpIdx is a use tied to TiedGroup. if (OpIdxGroup == CurGroup) return OpIdx - Delta; // OpIdx is a def tied to this use group. if (OpIdxGroup == TiedGroup) return OpIdx + Delta; } llvm_unreachable("Invalid tied operand on inline asm"); } /// clearKillInfo - Clears kill flags on all operands. /// void MachineInstr::clearKillInfo() { for (MachineOperand &MO : operands()) { if (MO.isReg() && MO.isUse()) MO.setIsKill(false); } } void MachineInstr::substituteRegister(unsigned FromReg, unsigned ToReg, unsigned SubIdx, const TargetRegisterInfo &RegInfo) { if (TargetRegisterInfo::isPhysicalRegister(ToReg)) { if (SubIdx) ToReg = RegInfo.getSubReg(ToReg, SubIdx); for (MachineOperand &MO : operands()) { if (!MO.isReg() || MO.getReg() != FromReg) continue; MO.substPhysReg(ToReg, RegInfo); } } else { for (MachineOperand &MO : operands()) { if (!MO.isReg() || MO.getReg() != FromReg) continue; MO.substVirtReg(ToReg, SubIdx, RegInfo); } } } /// isSafeToMove - Return true if it is safe to move this instruction. If /// SawStore is set to true, it means that there is a store (or call) between /// the instruction's location and its intended destination. bool MachineInstr::isSafeToMove(AliasAnalysis *AA, bool &SawStore) const { // Ignore stuff that we obviously can't move. // // Treat volatile loads as stores. This is not strictly necessary for // volatiles, but it is required for atomic loads. It is not allowed to move // a load across an atomic load with Ordering > Monotonic. if (mayStore() || isCall() || isPHI() || (mayLoad() && hasOrderedMemoryRef())) { SawStore = true; return false; } if (isPosition() || isDebugInstr() || isTerminator() || hasUnmodeledSideEffects()) return false; // See if this instruction does a load. If so, we have to guarantee that the // loaded value doesn't change between the load and the its intended // destination. The check for isInvariantLoad gives the targe the chance to // classify the load as always returning a constant, e.g. a constant pool // load. if (mayLoad() && !isDereferenceableInvariantLoad(AA)) // Otherwise, this is a real load. If there is a store between the load and // end of block, we can't move it. return !SawStore; return true; } bool MachineInstr::mayAlias(AliasAnalysis *AA, MachineInstr &Other, bool UseTBAA) { const MachineFunction *MF = getMF(); const TargetInstrInfo *TII = MF->getSubtarget().getInstrInfo(); const MachineFrameInfo &MFI = MF->getFrameInfo(); // If neither instruction stores to memory, they can't alias in any // meaningful way, even if they read from the same address. if (!mayStore() && !Other.mayStore()) return false; // Let the target decide if memory accesses cannot possibly overlap. if (TII->areMemAccessesTriviallyDisjoint(*this, Other, AA)) return false; // FIXME: Need to handle multiple memory operands to support all targets. if (!hasOneMemOperand() || !Other.hasOneMemOperand()) return true; MachineMemOperand *MMOa = *memoperands_begin(); MachineMemOperand *MMOb = *Other.memoperands_begin(); // The following interface to AA is fashioned after DAGCombiner::isAlias // and operates with MachineMemOperand offset with some important // assumptions: // - LLVM fundamentally assumes flat address spaces. // - MachineOperand offset can *only* result from legalization and // cannot affect queries other than the trivial case of overlap // checking. // - These offsets never wrap and never step outside // of allocated objects. // - There should never be any negative offsets here. // // FIXME: Modify API to hide this math from "user" // Even before we go to AA we can reason locally about some // memory objects. It can save compile time, and possibly catch some // corner cases not currently covered. int64_t OffsetA = MMOa->getOffset(); int64_t OffsetB = MMOb->getOffset(); int64_t MinOffset = std::min(OffsetA, OffsetB); uint64_t WidthA = MMOa->getSize(); uint64_t WidthB = MMOb->getSize(); bool KnownWidthA = WidthA != MemoryLocation::UnknownSize; bool KnownWidthB = WidthB != MemoryLocation::UnknownSize; const Value *ValA = MMOa->getValue(); const Value *ValB = MMOb->getValue(); bool SameVal = (ValA && ValB && (ValA == ValB)); if (!SameVal) { const PseudoSourceValue *PSVa = MMOa->getPseudoValue(); const PseudoSourceValue *PSVb = MMOb->getPseudoValue(); if (PSVa && ValB && !PSVa->mayAlias(&MFI)) return false; if (PSVb && ValA && !PSVb->mayAlias(&MFI)) return false; if (PSVa && PSVb && (PSVa == PSVb)) SameVal = true; } if (SameVal) { if (!KnownWidthA || !KnownWidthB) return true; int64_t MaxOffset = std::max(OffsetA, OffsetB); int64_t LowWidth = (MinOffset == OffsetA) ? WidthA : WidthB; return (MinOffset + LowWidth > MaxOffset); } if (!AA) return true; if (!ValA || !ValB) return true; assert((OffsetA >= 0) && "Negative MachineMemOperand offset"); assert((OffsetB >= 0) && "Negative MachineMemOperand offset"); int64_t OverlapA = KnownWidthA ? WidthA + OffsetA - MinOffset : MemoryLocation::UnknownSize; int64_t OverlapB = KnownWidthB ? WidthB + OffsetB - MinOffset : MemoryLocation::UnknownSize; AliasResult AAResult = AA->alias( MemoryLocation(ValA, OverlapA, UseTBAA ? MMOa->getAAInfo() : AAMDNodes()), MemoryLocation(ValB, OverlapB, UseTBAA ? MMOb->getAAInfo() : AAMDNodes())); return (AAResult != NoAlias); } /// hasOrderedMemoryRef - Return true if this instruction may have an ordered /// or volatile memory reference, or if the information describing the memory /// reference is not available. Return false if it is known to have no ordered /// memory references. bool MachineInstr::hasOrderedMemoryRef() const { // An instruction known never to access memory won't have a volatile access. if (!mayStore() && !mayLoad() && !isCall() && !hasUnmodeledSideEffects()) return false; // Otherwise, if the instruction has no memory reference information, // conservatively assume it wasn't preserved. if (memoperands_empty()) return true; // Check if any of our memory operands are ordered. return llvm::any_of(memoperands(), [](const MachineMemOperand *MMO) { return !MMO->isUnordered(); }); } /// isDereferenceableInvariantLoad - Return true if this instruction will never /// trap and is loading from a location whose value is invariant across a run of /// this function. bool MachineInstr::isDereferenceableInvariantLoad(AliasAnalysis *AA) const { // If the instruction doesn't load at all, it isn't an invariant load. if (!mayLoad()) return false; // If the instruction has lost its memoperands, conservatively assume that // it may not be an invariant load. if (memoperands_empty()) return false; const MachineFrameInfo &MFI = getParent()->getParent()->getFrameInfo(); for (MachineMemOperand *MMO : memoperands()) { if (MMO->isVolatile()) return false; if (MMO->isStore()) return false; if (MMO->isInvariant() && MMO->isDereferenceable()) continue; // A load from a constant PseudoSourceValue is invariant. if (const PseudoSourceValue *PSV = MMO->getPseudoValue()) if (PSV->isConstant(&MFI)) continue; if (const Value *V = MMO->getValue()) { // If we have an AliasAnalysis, ask it whether the memory is constant. if (AA && AA->pointsToConstantMemory( MemoryLocation(V, MMO->getSize(), MMO->getAAInfo()))) continue; } // Otherwise assume conservatively. return false; } // Everything checks out. return true; } /// isConstantValuePHI - If the specified instruction is a PHI that always /// merges together the same virtual register, return the register, otherwise /// return 0. unsigned MachineInstr::isConstantValuePHI() const { if (!isPHI()) return 0; assert(getNumOperands() >= 3 && "It's illegal to have a PHI without source operands"); unsigned Reg = getOperand(1).getReg(); for (unsigned i = 3, e = getNumOperands(); i < e; i += 2) if (getOperand(i).getReg() != Reg) return 0; return Reg; } bool MachineInstr::hasUnmodeledSideEffects() const { if (hasProperty(MCID::UnmodeledSideEffects)) return true; if (isInlineAsm()) { unsigned ExtraInfo = getOperand(InlineAsm::MIOp_ExtraInfo).getImm(); if (ExtraInfo & InlineAsm::Extra_HasSideEffects) return true; } return false; } bool MachineInstr::isLoadFoldBarrier() const { return mayStore() || isCall() || hasUnmodeledSideEffects(); } /// allDefsAreDead - Return true if all the defs of this instruction are dead. /// bool MachineInstr::allDefsAreDead() const { for (const MachineOperand &MO : operands()) { if (!MO.isReg() || MO.isUse()) continue; if (!MO.isDead()) return false; } return true; } /// copyImplicitOps - Copy implicit register operands from specified /// instruction to this instruction. void MachineInstr::copyImplicitOps(MachineFunction &MF, const MachineInstr &MI) { for (unsigned i = MI.getDesc().getNumOperands(), e = MI.getNumOperands(); i != e; ++i) { const MachineOperand &MO = MI.getOperand(i); if ((MO.isReg() && MO.isImplicit()) || MO.isRegMask()) addOperand(MF, MO); } } bool MachineInstr::hasComplexRegisterTies() const { const MCInstrDesc &MCID = getDesc(); for (unsigned I = 0, E = getNumOperands(); I < E; ++I) { const auto &Operand = getOperand(I); if (!Operand.isReg() || Operand.isDef()) // Ignore the defined registers as MCID marks only the uses as tied. continue; int ExpectedTiedIdx = MCID.getOperandConstraint(I, MCOI::TIED_TO); int TiedIdx = Operand.isTied() ? int(findTiedOperandIdx(I)) : -1; if (ExpectedTiedIdx != TiedIdx) return true; } return false; } LLT MachineInstr::getTypeToPrint(unsigned OpIdx, SmallBitVector &PrintedTypes, const MachineRegisterInfo &MRI) const { const MachineOperand &Op = getOperand(OpIdx); if (!Op.isReg()) return LLT{}; if (isVariadic() || OpIdx >= getNumExplicitOperands()) return MRI.getType(Op.getReg()); auto &OpInfo = getDesc().OpInfo[OpIdx]; if (!OpInfo.isGenericType()) return MRI.getType(Op.getReg()); if (PrintedTypes[OpInfo.getGenericTypeIndex()]) return LLT{}; LLT TypeToPrint = MRI.getType(Op.getReg()); // Don't mark the type index printed if it wasn't actually printed: maybe // another operand with the same type index has an actual type attached: if (TypeToPrint.isValid()) PrintedTypes.set(OpInfo.getGenericTypeIndex()); return TypeToPrint; } #if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP) LLVM_DUMP_METHOD void MachineInstr::dump() const { dbgs() << " "; print(dbgs()); } #endif void MachineInstr::print(raw_ostream &OS, bool IsStandalone, bool SkipOpers, bool SkipDebugLoc, bool AddNewLine, const TargetInstrInfo *TII) const { const Module *M = nullptr; const Function *F = nullptr; if (const MachineFunction *MF = getMFIfAvailable(*this)) { F = &MF->getFunction(); M = F->getParent(); if (!TII) TII = MF->getSubtarget().getInstrInfo(); } ModuleSlotTracker MST(M); if (F) MST.incorporateFunction(*F); print(OS, MST, IsStandalone, SkipOpers, SkipDebugLoc, TII); } void MachineInstr::print(raw_ostream &OS, ModuleSlotTracker &MST, bool IsStandalone, bool SkipOpers, bool SkipDebugLoc, bool AddNewLine, const TargetInstrInfo *TII) const { // We can be a bit tidier if we know the MachineFunction. const MachineFunction *MF = nullptr; const TargetRegisterInfo *TRI = nullptr; const MachineRegisterInfo *MRI = nullptr; const TargetIntrinsicInfo *IntrinsicInfo = nullptr; tryToGetTargetInfo(*this, TRI, MRI, IntrinsicInfo, TII); if (isCFIInstruction()) assert(getNumOperands() == 1 && "Expected 1 operand in CFI instruction"); SmallBitVector PrintedTypes(8); bool ShouldPrintRegisterTies = IsStandalone || hasComplexRegisterTies(); auto getTiedOperandIdx = [&](unsigned OpIdx) { if (!ShouldPrintRegisterTies) return 0U; const MachineOperand &MO = getOperand(OpIdx); if (MO.isReg() && MO.isTied() && !MO.isDef()) return findTiedOperandIdx(OpIdx); return 0U; }; unsigned StartOp = 0; unsigned e = getNumOperands(); // Print explicitly defined operands on the left of an assignment syntax. while (StartOp < e) { const MachineOperand &MO = getOperand(StartOp); if (!MO.isReg() || !MO.isDef() || MO.isImplicit()) break; if (StartOp != 0) OS << ", "; LLT TypeToPrint = MRI ? getTypeToPrint(StartOp, PrintedTypes, *MRI) : LLT{}; unsigned TiedOperandIdx = getTiedOperandIdx(StartOp); MO.print(OS, MST, TypeToPrint, /*PrintDef=*/false, IsStandalone, ShouldPrintRegisterTies, TiedOperandIdx, TRI, IntrinsicInfo); ++StartOp; } if (StartOp != 0) OS << " = "; if (getFlag(MachineInstr::FrameSetup)) OS << "frame-setup "; if (getFlag(MachineInstr::FrameDestroy)) OS << "frame-destroy "; if (getFlag(MachineInstr::FmNoNans)) OS << "nnan "; if (getFlag(MachineInstr::FmNoInfs)) OS << "ninf "; if (getFlag(MachineInstr::FmNsz)) OS << "nsz "; if (getFlag(MachineInstr::FmArcp)) OS << "arcp "; if (getFlag(MachineInstr::FmContract)) OS << "contract "; if (getFlag(MachineInstr::FmAfn)) OS << "afn "; if (getFlag(MachineInstr::FmReassoc)) OS << "reassoc "; if (getFlag(MachineInstr::NoUWrap)) OS << "nuw "; if (getFlag(MachineInstr::NoSWrap)) OS << "nsw "; if (getFlag(MachineInstr::IsExact)) OS << "exact "; // Print the opcode name. if (TII) OS << TII->getName(getOpcode()); else OS << "UNKNOWN"; if (SkipOpers) return; // Print the rest of the operands. bool FirstOp = true; unsigned AsmDescOp = ~0u; unsigned AsmOpCount = 0; if (isInlineAsm() && e >= InlineAsm::MIOp_FirstOperand) { // Print asm string. OS << " "; const unsigned OpIdx = InlineAsm::MIOp_AsmString; LLT TypeToPrint = MRI ? getTypeToPrint(OpIdx, PrintedTypes, *MRI) : LLT{}; unsigned TiedOperandIdx = getTiedOperandIdx(OpIdx); getOperand(OpIdx).print(OS, MST, TypeToPrint, /*PrintDef=*/true, IsStandalone, ShouldPrintRegisterTies, TiedOperandIdx, TRI, IntrinsicInfo); // Print HasSideEffects, MayLoad, MayStore, IsAlignStack unsigned ExtraInfo = getOperand(InlineAsm::MIOp_ExtraInfo).getImm(); if (ExtraInfo & InlineAsm::Extra_HasSideEffects) OS << " [sideeffect]"; if (ExtraInfo & InlineAsm::Extra_MayLoad) OS << " [mayload]"; if (ExtraInfo & InlineAsm::Extra_MayStore) OS << " [maystore]"; if (ExtraInfo & InlineAsm::Extra_IsConvergent) OS << " [isconvergent]"; if (ExtraInfo & InlineAsm::Extra_IsAlignStack) OS << " [alignstack]"; if (getInlineAsmDialect() == InlineAsm::AD_ATT) OS << " [attdialect]"; if (getInlineAsmDialect() == InlineAsm::AD_Intel) OS << " [inteldialect]"; StartOp = AsmDescOp = InlineAsm::MIOp_FirstOperand; FirstOp = false; } for (unsigned i = StartOp, e = getNumOperands(); i != e; ++i) { const MachineOperand &MO = getOperand(i); if (FirstOp) FirstOp = false; else OS << ","; OS << " "; if (isDebugValue() && MO.isMetadata()) { // Pretty print DBG_VALUE instructions. auto *DIV = dyn_cast(MO.getMetadata()); if (DIV && !DIV->getName().empty()) OS << "!\"" << DIV->getName() << '\"'; else { LLT TypeToPrint = MRI ? getTypeToPrint(i, PrintedTypes, *MRI) : LLT{}; unsigned TiedOperandIdx = getTiedOperandIdx(i); MO.print(OS, MST, TypeToPrint, /*PrintDef=*/true, IsStandalone, ShouldPrintRegisterTies, TiedOperandIdx, TRI, IntrinsicInfo); } } else if (isDebugLabel() && MO.isMetadata()) { // Pretty print DBG_LABEL instructions. auto *DIL = dyn_cast(MO.getMetadata()); if (DIL && !DIL->getName().empty()) OS << "\"" << DIL->getName() << '\"'; else { LLT TypeToPrint = MRI ? getTypeToPrint(i, PrintedTypes, *MRI) : LLT{}; unsigned TiedOperandIdx = getTiedOperandIdx(i); MO.print(OS, MST, TypeToPrint, /*PrintDef=*/true, IsStandalone, ShouldPrintRegisterTies, TiedOperandIdx, TRI, IntrinsicInfo); } } else if (i == AsmDescOp && MO.isImm()) { // Pretty print the inline asm operand descriptor. OS << '$' << AsmOpCount++; unsigned Flag = MO.getImm(); switch (InlineAsm::getKind(Flag)) { case InlineAsm::Kind_RegUse: OS << ":[reguse"; break; case InlineAsm::Kind_RegDef: OS << ":[regdef"; break; case InlineAsm::Kind_RegDefEarlyClobber: OS << ":[regdef-ec"; break; case InlineAsm::Kind_Clobber: OS << ":[clobber"; break; case InlineAsm::Kind_Imm: OS << ":[imm"; break; case InlineAsm::Kind_Mem: OS << ":[mem"; break; default: OS << ":[??" << InlineAsm::getKind(Flag); break; } unsigned RCID = 0; if (!InlineAsm::isImmKind(Flag) && !InlineAsm::isMemKind(Flag) && InlineAsm::hasRegClassConstraint(Flag, RCID)) { if (TRI) { OS << ':' << TRI->getRegClassName(TRI->getRegClass(RCID)); } else OS << ":RC" << RCID; } if (InlineAsm::isMemKind(Flag)) { unsigned MCID = InlineAsm::getMemoryConstraintID(Flag); switch (MCID) { case InlineAsm::Constraint_es: OS << ":es"; break; case InlineAsm::Constraint_i: OS << ":i"; break; case InlineAsm::Constraint_m: OS << ":m"; break; case InlineAsm::Constraint_o: OS << ":o"; break; case InlineAsm::Constraint_v: OS << ":v"; break; case InlineAsm::Constraint_Q: OS << ":Q"; break; case InlineAsm::Constraint_R: OS << ":R"; break; case InlineAsm::Constraint_S: OS << ":S"; break; case InlineAsm::Constraint_T: OS << ":T"; break; case InlineAsm::Constraint_Um: OS << ":Um"; break; case InlineAsm::Constraint_Un: OS << ":Un"; break; case InlineAsm::Constraint_Uq: OS << ":Uq"; break; case InlineAsm::Constraint_Us: OS << ":Us"; break; case InlineAsm::Constraint_Ut: OS << ":Ut"; break; case InlineAsm::Constraint_Uv: OS << ":Uv"; break; case InlineAsm::Constraint_Uy: OS << ":Uy"; break; case InlineAsm::Constraint_X: OS << ":X"; break; case InlineAsm::Constraint_Z: OS << ":Z"; break; case InlineAsm::Constraint_ZC: OS << ":ZC"; break; case InlineAsm::Constraint_Zy: OS << ":Zy"; break; default: OS << ":?"; break; } } unsigned TiedTo = 0; if (InlineAsm::isUseOperandTiedToDef(Flag, TiedTo)) OS << " tiedto:$" << TiedTo; OS << ']'; // Compute the index of the next operand descriptor. AsmDescOp += 1 + InlineAsm::getNumOperandRegisters(Flag); } else { LLT TypeToPrint = MRI ? getTypeToPrint(i, PrintedTypes, *MRI) : LLT{}; unsigned TiedOperandIdx = getTiedOperandIdx(i); if (MO.isImm() && isOperandSubregIdx(i)) MachineOperand::printSubRegIdx(OS, MO.getImm(), TRI); else MO.print(OS, MST, TypeToPrint, /*PrintDef=*/true, IsStandalone, ShouldPrintRegisterTies, TiedOperandIdx, TRI, IntrinsicInfo); } } // Print any optional symbols attached to this instruction as-if they were // operands. if (MCSymbol *PreInstrSymbol = getPreInstrSymbol()) { if (!FirstOp) { FirstOp = false; OS << ','; } OS << " pre-instr-symbol "; MachineOperand::printSymbol(OS, *PreInstrSymbol); } if (MCSymbol *PostInstrSymbol = getPostInstrSymbol()) { if (!FirstOp) { FirstOp = false; OS << ','; } OS << " post-instr-symbol "; MachineOperand::printSymbol(OS, *PostInstrSymbol); } if (!SkipDebugLoc) { if (const DebugLoc &DL = getDebugLoc()) { if (!FirstOp) OS << ','; OS << " debug-location "; DL->printAsOperand(OS, MST); } } if (!memoperands_empty()) { SmallVector SSNs; const LLVMContext *Context = nullptr; std::unique_ptr CtxPtr; const MachineFrameInfo *MFI = nullptr; if (const MachineFunction *MF = getMFIfAvailable(*this)) { MFI = &MF->getFrameInfo(); Context = &MF->getFunction().getContext(); } else { CtxPtr = llvm::make_unique(); Context = CtxPtr.get(); } OS << " :: "; bool NeedComma = false; for (const MachineMemOperand *Op : memoperands()) { if (NeedComma) OS << ", "; Op->print(OS, MST, SSNs, *Context, MFI, TII); NeedComma = true; } } if (SkipDebugLoc) return; bool HaveSemi = false; // Print debug location information. if (const DebugLoc &DL = getDebugLoc()) { if (!HaveSemi) { OS << ';'; HaveSemi = true; } OS << ' '; DL.print(OS); } // Print extra comments for DEBUG_VALUE. if (isDebugValue() && getOperand(e - 2).isMetadata()) { if (!HaveSemi) { OS << ";"; HaveSemi = true; } auto *DV = cast(getOperand(e - 2).getMetadata()); OS << " line no:" << DV->getLine(); if (auto *InlinedAt = debugLoc->getInlinedAt()) { DebugLoc InlinedAtDL(InlinedAt); if (InlinedAtDL && MF) { OS << " inlined @[ "; InlinedAtDL.print(OS); OS << " ]"; } } if (isIndirectDebugValue()) OS << " indirect"; } // TODO: DBG_LABEL if (AddNewLine) OS << '\n'; } bool MachineInstr::addRegisterKilled(unsigned IncomingReg, const TargetRegisterInfo *RegInfo, bool AddIfNotFound) { bool isPhysReg = TargetRegisterInfo::isPhysicalRegister(IncomingReg); bool hasAliases = isPhysReg && MCRegAliasIterator(IncomingReg, RegInfo, false).isValid(); bool Found = false; SmallVector DeadOps; for (unsigned i = 0, e = getNumOperands(); i != e; ++i) { MachineOperand &MO = getOperand(i); if (!MO.isReg() || !MO.isUse() || MO.isUndef()) continue; // DEBUG_VALUE nodes do not contribute to code generation and should // always be ignored. Failure to do so may result in trying to modify // KILL flags on DEBUG_VALUE nodes. if (MO.isDebug()) continue; unsigned Reg = MO.getReg(); if (!Reg) continue; if (Reg == IncomingReg) { if (!Found) { if (MO.isKill()) // The register is already marked kill. return true; if (isPhysReg && isRegTiedToDefOperand(i)) // Two-address uses of physregs must not be marked kill. return true; MO.setIsKill(); Found = true; } } else if (hasAliases && MO.isKill() && TargetRegisterInfo::isPhysicalRegister(Reg)) { // A super-register kill already exists. if (RegInfo->isSuperRegister(IncomingReg, Reg)) return true; if (RegInfo->isSubRegister(IncomingReg, Reg)) DeadOps.push_back(i); } } // Trim unneeded kill operands. while (!DeadOps.empty()) { unsigned OpIdx = DeadOps.back(); if (getOperand(OpIdx).isImplicit() && (!isInlineAsm() || findInlineAsmFlagIdx(OpIdx) < 0)) RemoveOperand(OpIdx); else getOperand(OpIdx).setIsKill(false); DeadOps.pop_back(); } // If not found, this means an alias of one of the operands is killed. Add a // new implicit operand if required. if (!Found && AddIfNotFound) { addOperand(MachineOperand::CreateReg(IncomingReg, false /*IsDef*/, true /*IsImp*/, true /*IsKill*/)); return true; } return Found; } void MachineInstr::clearRegisterKills(unsigned Reg, const TargetRegisterInfo *RegInfo) { if (!TargetRegisterInfo::isPhysicalRegister(Reg)) RegInfo = nullptr; for (MachineOperand &MO : operands()) { if (!MO.isReg() || !MO.isUse() || !MO.isKill()) continue; unsigned OpReg = MO.getReg(); if ((RegInfo && RegInfo->regsOverlap(Reg, OpReg)) || Reg == OpReg) MO.setIsKill(false); } } bool MachineInstr::addRegisterDead(unsigned Reg, const TargetRegisterInfo *RegInfo, bool AddIfNotFound) { bool isPhysReg = TargetRegisterInfo::isPhysicalRegister(Reg); bool hasAliases = isPhysReg && MCRegAliasIterator(Reg, RegInfo, false).isValid(); bool Found = false; SmallVector DeadOps; for (unsigned i = 0, e = getNumOperands(); i != e; ++i) { MachineOperand &MO = getOperand(i); if (!MO.isReg() || !MO.isDef()) continue; unsigned MOReg = MO.getReg(); if (!MOReg) continue; if (MOReg == Reg) { MO.setIsDead(); Found = true; } else if (hasAliases && MO.isDead() && TargetRegisterInfo::isPhysicalRegister(MOReg)) { // There exists a super-register that's marked dead. if (RegInfo->isSuperRegister(Reg, MOReg)) return true; if (RegInfo->isSubRegister(Reg, MOReg)) DeadOps.push_back(i); } } // Trim unneeded dead operands. while (!DeadOps.empty()) { unsigned OpIdx = DeadOps.back(); if (getOperand(OpIdx).isImplicit() && (!isInlineAsm() || findInlineAsmFlagIdx(OpIdx) < 0)) RemoveOperand(OpIdx); else getOperand(OpIdx).setIsDead(false); DeadOps.pop_back(); } // If not found, this means an alias of one of the operands is dead. Add a // new implicit operand if required. if (Found || !AddIfNotFound) return Found; addOperand(MachineOperand::CreateReg(Reg, true /*IsDef*/, true /*IsImp*/, false /*IsKill*/, true /*IsDead*/)); return true; } void MachineInstr::clearRegisterDeads(unsigned Reg) { for (MachineOperand &MO : operands()) { if (!MO.isReg() || !MO.isDef() || MO.getReg() != Reg) continue; MO.setIsDead(false); } } void MachineInstr::setRegisterDefReadUndef(unsigned Reg, bool IsUndef) { for (MachineOperand &MO : operands()) { if (!MO.isReg() || !MO.isDef() || MO.getReg() != Reg || MO.getSubReg() == 0) continue; MO.setIsUndef(IsUndef); } } void MachineInstr::addRegisterDefined(unsigned Reg, const TargetRegisterInfo *RegInfo) { if (TargetRegisterInfo::isPhysicalRegister(Reg)) { MachineOperand *MO = findRegisterDefOperand(Reg, false, RegInfo); if (MO) return; } else { for (const MachineOperand &MO : operands()) { if (MO.isReg() && MO.getReg() == Reg && MO.isDef() && MO.getSubReg() == 0) return; } } addOperand(MachineOperand::CreateReg(Reg, true /*IsDef*/, true /*IsImp*/)); } void MachineInstr::setPhysRegsDeadExcept(ArrayRef UsedRegs, const TargetRegisterInfo &TRI) { bool HasRegMask = false; for (MachineOperand &MO : operands()) { if (MO.isRegMask()) { HasRegMask = true; continue; } if (!MO.isReg() || !MO.isDef()) continue; unsigned Reg = MO.getReg(); if (!TargetRegisterInfo::isPhysicalRegister(Reg)) continue; // If there are no uses, including partial uses, the def is dead. if (llvm::none_of(UsedRegs, [&](unsigned Use) { return TRI.regsOverlap(Use, Reg); })) MO.setIsDead(); } // This is a call with a register mask operand. // Mask clobbers are always dead, so add defs for the non-dead defines. if (HasRegMask) for (ArrayRef::iterator I = UsedRegs.begin(), E = UsedRegs.end(); I != E; ++I) addRegisterDefined(*I, &TRI); } unsigned MachineInstrExpressionTrait::getHashValue(const MachineInstr* const &MI) { // Build up a buffer of hash code components. SmallVector HashComponents; HashComponents.reserve(MI->getNumOperands() + 1); HashComponents.push_back(MI->getOpcode()); for (const MachineOperand &MO : MI->operands()) { if (MO.isReg() && MO.isDef() && TargetRegisterInfo::isVirtualRegister(MO.getReg())) continue; // Skip virtual register defs. HashComponents.push_back(hash_value(MO)); } return hash_combine_range(HashComponents.begin(), HashComponents.end()); } void MachineInstr::emitError(StringRef Msg) const { // Find the source location cookie. unsigned LocCookie = 0; const MDNode *LocMD = nullptr; for (unsigned i = getNumOperands(); i != 0; --i) { if (getOperand(i-1).isMetadata() && (LocMD = getOperand(i-1).getMetadata()) && LocMD->getNumOperands() != 0) { if (const ConstantInt *CI = mdconst::dyn_extract(LocMD->getOperand(0))) { LocCookie = CI->getZExtValue(); break; } } } if (const MachineBasicBlock *MBB = getParent()) if (const MachineFunction *MF = MBB->getParent()) return MF->getMMI().getModule()->getContext().emitError(LocCookie, Msg); report_fatal_error(Msg); } MachineInstrBuilder llvm::BuildMI(MachineFunction &MF, const DebugLoc &DL, const MCInstrDesc &MCID, bool IsIndirect, unsigned Reg, const MDNode *Variable, const MDNode *Expr) { assert(isa(Variable) && "not a variable"); assert(cast(Expr)->isValid() && "not an expression"); assert(cast(Variable)->isValidLocationForIntrinsic(DL) && "Expected inlined-at fields to agree"); auto MIB = BuildMI(MF, DL, MCID).addReg(Reg, RegState::Debug); if (IsIndirect) MIB.addImm(0U); else MIB.addReg(0U, RegState::Debug); return MIB.addMetadata(Variable).addMetadata(Expr); } MachineInstrBuilder llvm::BuildMI(MachineFunction &MF, const DebugLoc &DL, const MCInstrDesc &MCID, bool IsIndirect, MachineOperand &MO, const MDNode *Variable, const MDNode *Expr) { assert(isa(Variable) && "not a variable"); assert(cast(Expr)->isValid() && "not an expression"); assert(cast(Variable)->isValidLocationForIntrinsic(DL) && "Expected inlined-at fields to agree"); if (MO.isReg()) return BuildMI(MF, DL, MCID, IsIndirect, MO.getReg(), Variable, Expr); auto MIB = BuildMI(MF, DL, MCID).add(MO); if (IsIndirect) MIB.addImm(0U); else MIB.addReg(0U, RegState::Debug); return MIB.addMetadata(Variable).addMetadata(Expr); } MachineInstrBuilder llvm::BuildMI(MachineBasicBlock &BB, MachineBasicBlock::iterator I, const DebugLoc &DL, const MCInstrDesc &MCID, bool IsIndirect, unsigned Reg, const MDNode *Variable, const MDNode *Expr) { MachineFunction &MF = *BB.getParent(); MachineInstr *MI = BuildMI(MF, DL, MCID, IsIndirect, Reg, Variable, Expr); BB.insert(I, MI); return MachineInstrBuilder(MF, MI); } MachineInstrBuilder llvm::BuildMI(MachineBasicBlock &BB, MachineBasicBlock::iterator I, const DebugLoc &DL, const MCInstrDesc &MCID, bool IsIndirect, MachineOperand &MO, const MDNode *Variable, const MDNode *Expr) { MachineFunction &MF = *BB.getParent(); MachineInstr *MI = BuildMI(MF, DL, MCID, IsIndirect, MO, Variable, Expr); BB.insert(I, MI); return MachineInstrBuilder(MF, *MI); } /// Compute the new DIExpression to use with a DBG_VALUE for a spill slot. /// This prepends DW_OP_deref when spilling an indirect DBG_VALUE. static const DIExpression *computeExprForSpill(const MachineInstr &MI) { assert(MI.getOperand(0).isReg() && "can't spill non-register"); assert(MI.getDebugVariable()->isValidLocationForIntrinsic(MI.getDebugLoc()) && "Expected inlined-at fields to agree"); const DIExpression *Expr = MI.getDebugExpression(); if (MI.isIndirectDebugValue()) { assert(MI.getOperand(1).getImm() == 0 && "DBG_VALUE with nonzero offset"); Expr = DIExpression::prepend(Expr, DIExpression::WithDeref); } return Expr; } MachineInstr *llvm::buildDbgValueForSpill(MachineBasicBlock &BB, MachineBasicBlock::iterator I, const MachineInstr &Orig, int FrameIndex) { const DIExpression *Expr = computeExprForSpill(Orig); return BuildMI(BB, I, Orig.getDebugLoc(), Orig.getDesc()) .addFrameIndex(FrameIndex) .addImm(0U) .addMetadata(Orig.getDebugVariable()) .addMetadata(Expr); } void llvm::updateDbgValueForSpill(MachineInstr &Orig, int FrameIndex) { const DIExpression *Expr = computeExprForSpill(Orig); Orig.getOperand(0).ChangeToFrameIndex(FrameIndex); Orig.getOperand(1).ChangeToImmediate(0U); Orig.getOperand(3).setMetadata(Expr); } void MachineInstr::collectDebugValues( SmallVectorImpl &DbgValues) { MachineInstr &MI = *this; if (!MI.getOperand(0).isReg()) return; MachineBasicBlock::iterator DI = MI; ++DI; for (MachineBasicBlock::iterator DE = MI.getParent()->end(); DI != DE; ++DI) { if (!DI->isDebugValue()) return; if (DI->getOperand(0).isReg() && DI->getOperand(0).getReg() == MI.getOperand(0).getReg()) DbgValues.push_back(&*DI); } } void MachineInstr::changeDebugValuesDefReg(unsigned Reg) { // Collect matching debug values. SmallVector DbgValues; collectDebugValues(DbgValues); // Propagate Reg to debug value instructions. for (auto *DBI : DbgValues) DBI->getOperand(0).setReg(Reg); } Index: vendor/llvm/dist-release_80/lib/CodeGen/SelectionDAG/DAGCombiner.cpp =================================================================== --- vendor/llvm/dist-release_80/lib/CodeGen/SelectionDAG/DAGCombiner.cpp (revision 343793) +++ vendor/llvm/dist-release_80/lib/CodeGen/SelectionDAG/DAGCombiner.cpp (revision 343794) @@ -1,19395 +1,19401 @@ //===- DAGCombiner.cpp - Implement a DAG node combiner --------------------===// // // The LLVM Compiler Infrastructure // // This file is distributed under the University of Illinois Open Source // License. See LICENSE.TXT for details. // //===----------------------------------------------------------------------===// // // This pass combines dag nodes to form fewer, simpler DAG nodes. It can be run // both before and after the DAG is legalized. // // This pass is not a substitute for the LLVM IR instcombine pass. This pass is // primarily intended to handle simplification opportunities that are implicit // in the LLVM IR and exposed by the various codegen lowering phases. // //===----------------------------------------------------------------------===// #include "llvm/ADT/APFloat.h" #include "llvm/ADT/APInt.h" #include "llvm/ADT/ArrayRef.h" #include "llvm/ADT/DenseMap.h" #include "llvm/ADT/IntervalMap.h" #include "llvm/ADT/None.h" #include "llvm/ADT/Optional.h" #include "llvm/ADT/STLExtras.h" #include "llvm/ADT/SetVector.h" #include "llvm/ADT/SmallBitVector.h" #include "llvm/ADT/SmallPtrSet.h" #include "llvm/ADT/SmallSet.h" #include "llvm/ADT/SmallVector.h" #include "llvm/ADT/Statistic.h" #include "llvm/Analysis/AliasAnalysis.h" #include "llvm/Analysis/MemoryLocation.h" #include "llvm/CodeGen/DAGCombine.h" #include "llvm/CodeGen/ISDOpcodes.h" #include "llvm/CodeGen/MachineFrameInfo.h" #include "llvm/CodeGen/MachineFunction.h" #include "llvm/CodeGen/MachineMemOperand.h" #include "llvm/CodeGen/RuntimeLibcalls.h" #include "llvm/CodeGen/SelectionDAG.h" #include "llvm/CodeGen/SelectionDAGAddressAnalysis.h" #include "llvm/CodeGen/SelectionDAGNodes.h" #include "llvm/CodeGen/SelectionDAGTargetInfo.h" #include "llvm/CodeGen/TargetLowering.h" #include "llvm/CodeGen/TargetRegisterInfo.h" #include "llvm/CodeGen/TargetSubtargetInfo.h" #include "llvm/CodeGen/ValueTypes.h" #include "llvm/IR/Attributes.h" #include "llvm/IR/Constant.h" #include "llvm/IR/DataLayout.h" #include "llvm/IR/DerivedTypes.h" #include "llvm/IR/Function.h" #include "llvm/IR/LLVMContext.h" #include "llvm/IR/Metadata.h" #include "llvm/Support/Casting.h" #include "llvm/Support/CodeGen.h" #include "llvm/Support/CommandLine.h" #include "llvm/Support/Compiler.h" #include "llvm/Support/Debug.h" #include "llvm/Support/ErrorHandling.h" #include "llvm/Support/KnownBits.h" #include "llvm/Support/MachineValueType.h" #include "llvm/Support/MathExtras.h" #include "llvm/Support/raw_ostream.h" #include "llvm/Target/TargetMachine.h" #include "llvm/Target/TargetOptions.h" #include #include #include #include #include #include #include #include using namespace llvm; #define DEBUG_TYPE "dagcombine" STATISTIC(NodesCombined , "Number of dag nodes combined"); STATISTIC(PreIndexedNodes , "Number of pre-indexed nodes created"); STATISTIC(PostIndexedNodes, "Number of post-indexed nodes created"); STATISTIC(OpsNarrowed , "Number of load/op/store narrowed"); STATISTIC(LdStFP2Int , "Number of fp load/store pairs transformed to int"); STATISTIC(SlicedLoads, "Number of load sliced"); STATISTIC(NumFPLogicOpsConv, "Number of logic ops converted to fp ops"); static cl::opt CombinerGlobalAA("combiner-global-alias-analysis", cl::Hidden, cl::desc("Enable DAG combiner's use of IR alias analysis")); static cl::opt UseTBAA("combiner-use-tbaa", cl::Hidden, cl::init(true), cl::desc("Enable DAG combiner's use of TBAA")); #ifndef NDEBUG static cl::opt CombinerAAOnlyFunc("combiner-aa-only-func", cl::Hidden, cl::desc("Only use DAG-combiner alias analysis in this" " function")); #endif /// Hidden option to stress test load slicing, i.e., when this option /// is enabled, load slicing bypasses most of its profitability guards. static cl::opt StressLoadSlicing("combiner-stress-load-slicing", cl::Hidden, cl::desc("Bypass the profitability model of load slicing"), cl::init(false)); static cl::opt MaySplitLoadIndex("combiner-split-load-index", cl::Hidden, cl::init(true), cl::desc("DAG combiner may split indexing from loads")); namespace { class DAGCombiner { SelectionDAG &DAG; const TargetLowering &TLI; CombineLevel Level; CodeGenOpt::Level OptLevel; bool LegalOperations = false; bool LegalTypes = false; bool ForCodeSize; /// Worklist of all of the nodes that need to be simplified. /// /// This must behave as a stack -- new nodes to process are pushed onto the /// back and when processing we pop off of the back. /// /// The worklist will not contain duplicates but may contain null entries /// due to nodes being deleted from the underlying DAG. SmallVector Worklist; /// Mapping from an SDNode to its position on the worklist. /// /// This is used to find and remove nodes from the worklist (by nulling /// them) when they are deleted from the underlying DAG. It relies on /// stable indices of nodes within the worklist. DenseMap WorklistMap; /// Set of nodes which have been combined (at least once). /// /// This is used to allow us to reliably add any operands of a DAG node /// which have not yet been combined to the worklist. SmallPtrSet CombinedNodes; // AA - Used for DAG load/store alias analysis. AliasAnalysis *AA; /// When an instruction is simplified, add all users of the instruction to /// the work lists because they might get more simplified now. void AddUsersToWorklist(SDNode *N) { for (SDNode *Node : N->uses()) AddToWorklist(Node); } /// Call the node-specific routine that folds each particular type of node. SDValue visit(SDNode *N); public: DAGCombiner(SelectionDAG &D, AliasAnalysis *AA, CodeGenOpt::Level OL) : DAG(D), TLI(D.getTargetLoweringInfo()), Level(BeforeLegalizeTypes), OptLevel(OL), AA(AA) { ForCodeSize = DAG.getMachineFunction().getFunction().optForSize(); MaximumLegalStoreInBits = 0; for (MVT VT : MVT::all_valuetypes()) if (EVT(VT).isSimple() && VT != MVT::Other && TLI.isTypeLegal(EVT(VT)) && VT.getSizeInBits() >= MaximumLegalStoreInBits) MaximumLegalStoreInBits = VT.getSizeInBits(); } /// Add to the worklist making sure its instance is at the back (next to be /// processed.) void AddToWorklist(SDNode *N) { assert(N->getOpcode() != ISD::DELETED_NODE && "Deleted Node added to Worklist"); // Skip handle nodes as they can't usefully be combined and confuse the // zero-use deletion strategy. if (N->getOpcode() == ISD::HANDLENODE) return; if (WorklistMap.insert(std::make_pair(N, Worklist.size())).second) Worklist.push_back(N); } /// Remove all instances of N from the worklist. void removeFromWorklist(SDNode *N) { CombinedNodes.erase(N); auto It = WorklistMap.find(N); if (It == WorklistMap.end()) return; // Not in the worklist. // Null out the entry rather than erasing it to avoid a linear operation. Worklist[It->second] = nullptr; WorklistMap.erase(It); } void deleteAndRecombine(SDNode *N); bool recursivelyDeleteUnusedNodes(SDNode *N); /// Replaces all uses of the results of one DAG node with new values. SDValue CombineTo(SDNode *N, const SDValue *To, unsigned NumTo, bool AddTo = true); /// Replaces all uses of the results of one DAG node with new values. SDValue CombineTo(SDNode *N, SDValue Res, bool AddTo = true) { return CombineTo(N, &Res, 1, AddTo); } /// Replaces all uses of the results of one DAG node with new values. SDValue CombineTo(SDNode *N, SDValue Res0, SDValue Res1, bool AddTo = true) { SDValue To[] = { Res0, Res1 }; return CombineTo(N, To, 2, AddTo); } void CommitTargetLoweringOpt(const TargetLowering::TargetLoweringOpt &TLO); private: unsigned MaximumLegalStoreInBits; /// Check the specified integer node value to see if it can be simplified or /// if things it uses can be simplified by bit propagation. /// If so, return true. bool SimplifyDemandedBits(SDValue Op) { unsigned BitWidth = Op.getScalarValueSizeInBits(); APInt Demanded = APInt::getAllOnesValue(BitWidth); return SimplifyDemandedBits(Op, Demanded); } /// Check the specified vector node value to see if it can be simplified or /// if things it uses can be simplified as it only uses some of the /// elements. If so, return true. bool SimplifyDemandedVectorElts(SDValue Op) { unsigned NumElts = Op.getValueType().getVectorNumElements(); APInt Demanded = APInt::getAllOnesValue(NumElts); return SimplifyDemandedVectorElts(Op, Demanded); } bool SimplifyDemandedBits(SDValue Op, const APInt &Demanded); bool SimplifyDemandedVectorElts(SDValue Op, const APInt &Demanded, bool AssumeSingleUse = false); bool CombineToPreIndexedLoadStore(SDNode *N); bool CombineToPostIndexedLoadStore(SDNode *N); SDValue SplitIndexingFromLoad(LoadSDNode *LD); bool SliceUpLoad(SDNode *N); // Scalars have size 0 to distinguish from singleton vectors. SDValue ForwardStoreValueToDirectLoad(LoadSDNode *LD); bool getTruncatedStoreValue(StoreSDNode *ST, SDValue &Val); bool extendLoadedValueToExtension(LoadSDNode *LD, SDValue &Val); /// Replace an ISD::EXTRACT_VECTOR_ELT of a load with a narrowed /// load. /// /// \param EVE ISD::EXTRACT_VECTOR_ELT to be replaced. /// \param InVecVT type of the input vector to EVE with bitcasts resolved. /// \param EltNo index of the vector element to load. /// \param OriginalLoad load that EVE came from to be replaced. /// \returns EVE on success SDValue() on failure. SDValue scalarizeExtractedVectorLoad(SDNode *EVE, EVT InVecVT, SDValue EltNo, LoadSDNode *OriginalLoad); void ReplaceLoadWithPromotedLoad(SDNode *Load, SDNode *ExtLoad); SDValue PromoteOperand(SDValue Op, EVT PVT, bool &Replace); SDValue SExtPromoteOperand(SDValue Op, EVT PVT); SDValue ZExtPromoteOperand(SDValue Op, EVT PVT); SDValue PromoteIntBinOp(SDValue Op); SDValue PromoteIntShiftOp(SDValue Op); SDValue PromoteExtend(SDValue Op); bool PromoteLoad(SDValue Op); /// Call the node-specific routine that knows how to fold each /// particular type of node. If that doesn't do anything, try the /// target-specific DAG combines. SDValue combine(SDNode *N); // Visitation implementation - Implement dag node combining for different // node types. The semantics are as follows: // Return Value: // SDValue.getNode() == 0 - No change was made // SDValue.getNode() == N - N was replaced, is dead and has been handled. // otherwise - N should be replaced by the returned Operand. // SDValue visitTokenFactor(SDNode *N); SDValue visitMERGE_VALUES(SDNode *N); SDValue visitADD(SDNode *N); SDValue visitADDLike(SDValue N0, SDValue N1, SDNode *LocReference); SDValue visitSUB(SDNode *N); SDValue visitADDSAT(SDNode *N); SDValue visitSUBSAT(SDNode *N); SDValue visitADDC(SDNode *N); SDValue visitUADDO(SDNode *N); SDValue visitUADDOLike(SDValue N0, SDValue N1, SDNode *N); SDValue visitSUBC(SDNode *N); SDValue visitUSUBO(SDNode *N); SDValue visitADDE(SDNode *N); SDValue visitADDCARRY(SDNode *N); SDValue visitADDCARRYLike(SDValue N0, SDValue N1, SDValue CarryIn, SDNode *N); SDValue visitSUBE(SDNode *N); SDValue visitSUBCARRY(SDNode *N); SDValue visitMUL(SDNode *N); SDValue useDivRem(SDNode *N); SDValue visitSDIV(SDNode *N); SDValue visitSDIVLike(SDValue N0, SDValue N1, SDNode *N); SDValue visitUDIV(SDNode *N); SDValue visitUDIVLike(SDValue N0, SDValue N1, SDNode *N); SDValue visitREM(SDNode *N); SDValue visitMULHU(SDNode *N); SDValue visitMULHS(SDNode *N); SDValue visitSMUL_LOHI(SDNode *N); SDValue visitUMUL_LOHI(SDNode *N); SDValue visitSMULO(SDNode *N); SDValue visitUMULO(SDNode *N); SDValue visitIMINMAX(SDNode *N); SDValue visitAND(SDNode *N); SDValue visitANDLike(SDValue N0, SDValue N1, SDNode *N); SDValue visitOR(SDNode *N); SDValue visitORLike(SDValue N0, SDValue N1, SDNode *N); SDValue visitXOR(SDNode *N); SDValue SimplifyVBinOp(SDNode *N); SDValue visitSHL(SDNode *N); SDValue visitSRA(SDNode *N); SDValue visitSRL(SDNode *N); SDValue visitFunnelShift(SDNode *N); SDValue visitRotate(SDNode *N); SDValue visitABS(SDNode *N); SDValue visitBSWAP(SDNode *N); SDValue visitBITREVERSE(SDNode *N); SDValue visitCTLZ(SDNode *N); SDValue visitCTLZ_ZERO_UNDEF(SDNode *N); SDValue visitCTTZ(SDNode *N); SDValue visitCTTZ_ZERO_UNDEF(SDNode *N); SDValue visitCTPOP(SDNode *N); SDValue visitSELECT(SDNode *N); SDValue visitVSELECT(SDNode *N); SDValue visitSELECT_CC(SDNode *N); SDValue visitSETCC(SDNode *N); SDValue visitSETCCCARRY(SDNode *N); SDValue visitSIGN_EXTEND(SDNode *N); SDValue visitZERO_EXTEND(SDNode *N); SDValue visitANY_EXTEND(SDNode *N); SDValue visitAssertExt(SDNode *N); SDValue visitSIGN_EXTEND_INREG(SDNode *N); SDValue visitSIGN_EXTEND_VECTOR_INREG(SDNode *N); SDValue visitZERO_EXTEND_VECTOR_INREG(SDNode *N); SDValue visitTRUNCATE(SDNode *N); SDValue visitBITCAST(SDNode *N); SDValue visitBUILD_PAIR(SDNode *N); SDValue visitFADD(SDNode *N); SDValue visitFSUB(SDNode *N); SDValue visitFMUL(SDNode *N); SDValue visitFMA(SDNode *N); SDValue visitFDIV(SDNode *N); SDValue visitFREM(SDNode *N); SDValue visitFSQRT(SDNode *N); SDValue visitFCOPYSIGN(SDNode *N); SDValue visitFPOW(SDNode *N); SDValue visitSINT_TO_FP(SDNode *N); SDValue visitUINT_TO_FP(SDNode *N); SDValue visitFP_TO_SINT(SDNode *N); SDValue visitFP_TO_UINT(SDNode *N); SDValue visitFP_ROUND(SDNode *N); SDValue visitFP_ROUND_INREG(SDNode *N); SDValue visitFP_EXTEND(SDNode *N); SDValue visitFNEG(SDNode *N); SDValue visitFABS(SDNode *N); SDValue visitFCEIL(SDNode *N); SDValue visitFTRUNC(SDNode *N); SDValue visitFFLOOR(SDNode *N); SDValue visitFMINNUM(SDNode *N); SDValue visitFMAXNUM(SDNode *N); SDValue visitFMINIMUM(SDNode *N); SDValue visitFMAXIMUM(SDNode *N); SDValue visitBRCOND(SDNode *N); SDValue visitBR_CC(SDNode *N); SDValue visitLOAD(SDNode *N); SDValue replaceStoreChain(StoreSDNode *ST, SDValue BetterChain); SDValue replaceStoreOfFPConstant(StoreSDNode *ST); SDValue visitSTORE(SDNode *N); SDValue visitINSERT_VECTOR_ELT(SDNode *N); SDValue visitEXTRACT_VECTOR_ELT(SDNode *N); SDValue visitBUILD_VECTOR(SDNode *N); SDValue visitCONCAT_VECTORS(SDNode *N); SDValue visitEXTRACT_SUBVECTOR(SDNode *N); SDValue visitVECTOR_SHUFFLE(SDNode *N); SDValue visitSCALAR_TO_VECTOR(SDNode *N); SDValue visitINSERT_SUBVECTOR(SDNode *N); SDValue visitMLOAD(SDNode *N); SDValue visitMSTORE(SDNode *N); SDValue visitMGATHER(SDNode *N); SDValue visitMSCATTER(SDNode *N); SDValue visitFP_TO_FP16(SDNode *N); SDValue visitFP16_TO_FP(SDNode *N); SDValue visitFADDForFMACombine(SDNode *N); SDValue visitFSUBForFMACombine(SDNode *N); SDValue visitFMULForFMADistributiveCombine(SDNode *N); SDValue XformToShuffleWithZero(SDNode *N); SDValue ReassociateOps(unsigned Opc, const SDLoc &DL, SDValue N0, SDValue N1, SDNodeFlags Flags); SDValue visitShiftByConstant(SDNode *N, ConstantSDNode *Amt); SDValue foldSelectOfConstants(SDNode *N); SDValue foldVSelectOfConstants(SDNode *N); SDValue foldBinOpIntoSelect(SDNode *BO); bool SimplifySelectOps(SDNode *SELECT, SDValue LHS, SDValue RHS); SDValue hoistLogicOpWithSameOpcodeHands(SDNode *N); SDValue SimplifySelect(const SDLoc &DL, SDValue N0, SDValue N1, SDValue N2); SDValue SimplifySelectCC(const SDLoc &DL, SDValue N0, SDValue N1, SDValue N2, SDValue N3, ISD::CondCode CC, bool NotExtCompare = false); SDValue convertSelectOfFPConstantsToLoadOffset( const SDLoc &DL, SDValue N0, SDValue N1, SDValue N2, SDValue N3, ISD::CondCode CC); SDValue foldSelectCCToShiftAnd(const SDLoc &DL, SDValue N0, SDValue N1, SDValue N2, SDValue N3, ISD::CondCode CC); SDValue foldLogicOfSetCCs(bool IsAnd, SDValue N0, SDValue N1, const SDLoc &DL); SDValue unfoldMaskedMerge(SDNode *N); SDValue unfoldExtremeBitClearingToShifts(SDNode *N); SDValue SimplifySetCC(EVT VT, SDValue N0, SDValue N1, ISD::CondCode Cond, const SDLoc &DL, bool foldBooleans); SDValue rebuildSetCC(SDValue N); bool isSetCCEquivalent(SDValue N, SDValue &LHS, SDValue &RHS, SDValue &CC) const; bool isOneUseSetCC(SDValue N) const; SDValue SimplifyNodeWithTwoResults(SDNode *N, unsigned LoOp, unsigned HiOp); SDValue CombineConsecutiveLoads(SDNode *N, EVT VT); SDValue CombineExtLoad(SDNode *N); SDValue CombineZExtLogicopShiftLoad(SDNode *N); SDValue combineRepeatedFPDivisors(SDNode *N); SDValue combineInsertEltToShuffle(SDNode *N, unsigned InsIndex); SDValue ConstantFoldBITCASTofBUILD_VECTOR(SDNode *, EVT); SDValue BuildSDIV(SDNode *N); SDValue BuildSDIVPow2(SDNode *N); SDValue BuildUDIV(SDNode *N); SDValue BuildLogBase2(SDValue V, const SDLoc &DL); SDValue BuildReciprocalEstimate(SDValue Op, SDNodeFlags Flags); SDValue buildRsqrtEstimate(SDValue Op, SDNodeFlags Flags); SDValue buildSqrtEstimate(SDValue Op, SDNodeFlags Flags); SDValue buildSqrtEstimateImpl(SDValue Op, SDNodeFlags Flags, bool Recip); SDValue buildSqrtNROneConst(SDValue Arg, SDValue Est, unsigned Iterations, SDNodeFlags Flags, bool Reciprocal); SDValue buildSqrtNRTwoConst(SDValue Arg, SDValue Est, unsigned Iterations, SDNodeFlags Flags, bool Reciprocal); SDValue MatchBSwapHWordLow(SDNode *N, SDValue N0, SDValue N1, bool DemandHighBits = true); SDValue MatchBSwapHWord(SDNode *N, SDValue N0, SDValue N1); SDNode *MatchRotatePosNeg(SDValue Shifted, SDValue Pos, SDValue Neg, SDValue InnerPos, SDValue InnerNeg, unsigned PosOpcode, unsigned NegOpcode, const SDLoc &DL); SDNode *MatchRotate(SDValue LHS, SDValue RHS, const SDLoc &DL); SDValue MatchLoadCombine(SDNode *N); SDValue ReduceLoadWidth(SDNode *N); SDValue ReduceLoadOpStoreWidth(SDNode *N); SDValue splitMergedValStore(StoreSDNode *ST); SDValue TransformFPLoadStorePair(SDNode *N); SDValue convertBuildVecZextToZext(SDNode *N); SDValue reduceBuildVecExtToExtBuildVec(SDNode *N); SDValue reduceBuildVecToShuffle(SDNode *N); SDValue createBuildVecShuffle(const SDLoc &DL, SDNode *N, ArrayRef VectorMask, SDValue VecIn1, SDValue VecIn2, unsigned LeftIdx); SDValue matchVSelectOpSizesWithSetCC(SDNode *Cast); /// Walk up chain skipping non-aliasing memory nodes, /// looking for aliasing nodes and adding them to the Aliases vector. void GatherAllAliases(SDNode *N, SDValue OriginalChain, SmallVectorImpl &Aliases); /// Return true if there is any possibility that the two addresses overlap. bool isAlias(LSBaseSDNode *Op0, LSBaseSDNode *Op1) const; /// Walk up chain skipping non-aliasing memory nodes, looking for a better /// chain (aliasing node.) SDValue FindBetterChain(SDNode *N, SDValue Chain); /// Try to replace a store and any possibly adjacent stores on /// consecutive chains with better chains. Return true only if St is /// replaced. /// /// Notice that other chains may still be replaced even if the function /// returns false. bool findBetterNeighborChains(StoreSDNode *St); // Helper for findBetterNeighborChains. Walk up store chain add additional // chained stores that do not overlap and can be parallelized. bool parallelizeChainedStores(StoreSDNode *St); /// Holds a pointer to an LSBaseSDNode as well as information on where it /// is located in a sequence of memory operations connected by a chain. struct MemOpLink { // Ptr to the mem node. LSBaseSDNode *MemNode; // Offset from the base ptr. int64_t OffsetFromBase; MemOpLink(LSBaseSDNode *N, int64_t Offset) : MemNode(N), OffsetFromBase(Offset) {} }; /// This is a helper function for visitMUL to check the profitability /// of folding (mul (add x, c1), c2) -> (add (mul x, c2), c1*c2). /// MulNode is the original multiply, AddNode is (add x, c1), /// and ConstNode is c2. bool isMulAddWithConstProfitable(SDNode *MulNode, SDValue &AddNode, SDValue &ConstNode); /// This is a helper function for visitAND and visitZERO_EXTEND. Returns /// true if the (and (load x) c) pattern matches an extload. ExtVT returns /// the type of the loaded value to be extended. bool isAndLoadExtLoad(ConstantSDNode *AndC, LoadSDNode *LoadN, EVT LoadResultTy, EVT &ExtVT); /// Helper function to calculate whether the given Load/Store can have its /// width reduced to ExtVT. bool isLegalNarrowLdSt(LSBaseSDNode *LDSTN, ISD::LoadExtType ExtType, EVT &MemVT, unsigned ShAmt = 0); /// Used by BackwardsPropagateMask to find suitable loads. bool SearchForAndLoads(SDNode *N, SmallVectorImpl &Loads, SmallPtrSetImpl &NodesWithConsts, ConstantSDNode *Mask, SDNode *&NodeToMask); /// Attempt to propagate a given AND node back to load leaves so that they /// can be combined into narrow loads. bool BackwardsPropagateMask(SDNode *N, SelectionDAG &DAG); /// Helper function for MergeConsecutiveStores which merges the /// component store chains. SDValue getMergeStoreChains(SmallVectorImpl &StoreNodes, unsigned NumStores); /// This is a helper function for MergeConsecutiveStores. When the /// source elements of the consecutive stores are all constants or /// all extracted vector elements, try to merge them into one /// larger store introducing bitcasts if necessary. \return True /// if a merged store was created. bool MergeStoresOfConstantsOrVecElts(SmallVectorImpl &StoreNodes, EVT MemVT, unsigned NumStores, bool IsConstantSrc, bool UseVector, bool UseTrunc); /// This is a helper function for MergeConsecutiveStores. Stores /// that potentially may be merged with St are placed in /// StoreNodes. RootNode is a chain predecessor to all store /// candidates. void getStoreMergeCandidates(StoreSDNode *St, SmallVectorImpl &StoreNodes, SDNode *&Root); /// Helper function for MergeConsecutiveStores. Checks if /// candidate stores have indirect dependency through their /// operands. RootNode is the predecessor to all stores calculated /// by getStoreMergeCandidates and is used to prune the dependency check. /// \return True if safe to merge. bool checkMergeStoreCandidatesForDependencies( SmallVectorImpl &StoreNodes, unsigned NumStores, SDNode *RootNode); /// Merge consecutive store operations into a wide store. /// This optimization uses wide integers or vectors when possible. /// \return number of stores that were merged into a merged store (the /// affected nodes are stored as a prefix in \p StoreNodes). bool MergeConsecutiveStores(StoreSDNode *St); /// Try to transform a truncation where C is a constant: /// (trunc (and X, C)) -> (and (trunc X), (trunc C)) /// /// \p N needs to be a truncation and its first operand an AND. Other /// requirements are checked by the function (e.g. that trunc is /// single-use) and if missed an empty SDValue is returned. SDValue distributeTruncateThroughAnd(SDNode *N); /// Helper function to determine whether the target supports operation /// given by \p Opcode for type \p VT, that is, whether the operation /// is legal or custom before legalizing operations, and whether is /// legal (but not custom) after legalization. bool hasOperation(unsigned Opcode, EVT VT) { if (LegalOperations) return TLI.isOperationLegal(Opcode, VT); return TLI.isOperationLegalOrCustom(Opcode, VT); } public: /// Runs the dag combiner on all nodes in the work list void Run(CombineLevel AtLevel); SelectionDAG &getDAG() const { return DAG; } /// Returns a type large enough to hold any valid shift amount - before type /// legalization these can be huge. EVT getShiftAmountTy(EVT LHSTy) { assert(LHSTy.isInteger() && "Shift amount is not an integer type!"); return TLI.getShiftAmountTy(LHSTy, DAG.getDataLayout(), LegalTypes); } /// This method returns true if we are running before type legalization or /// if the specified VT is legal. bool isTypeLegal(const EVT &VT) { if (!LegalTypes) return true; return TLI.isTypeLegal(VT); } /// Convenience wrapper around TargetLowering::getSetCCResultType EVT getSetCCResultType(EVT VT) const { return TLI.getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), VT); } void ExtendSetCCUses(const SmallVectorImpl &SetCCs, SDValue OrigLoad, SDValue ExtLoad, ISD::NodeType ExtType); }; /// This class is a DAGUpdateListener that removes any deleted /// nodes from the worklist. class WorklistRemover : public SelectionDAG::DAGUpdateListener { DAGCombiner &DC; public: explicit WorklistRemover(DAGCombiner &dc) : SelectionDAG::DAGUpdateListener(dc.getDAG()), DC(dc) {} void NodeDeleted(SDNode *N, SDNode *E) override { DC.removeFromWorklist(N); } }; } // end anonymous namespace //===----------------------------------------------------------------------===// // TargetLowering::DAGCombinerInfo implementation //===----------------------------------------------------------------------===// void TargetLowering::DAGCombinerInfo::AddToWorklist(SDNode *N) { ((DAGCombiner*)DC)->AddToWorklist(N); } SDValue TargetLowering::DAGCombinerInfo:: CombineTo(SDNode *N, ArrayRef To, bool AddTo) { return ((DAGCombiner*)DC)->CombineTo(N, &To[0], To.size(), AddTo); } SDValue TargetLowering::DAGCombinerInfo:: CombineTo(SDNode *N, SDValue Res, bool AddTo) { return ((DAGCombiner*)DC)->CombineTo(N, Res, AddTo); } SDValue TargetLowering::DAGCombinerInfo:: CombineTo(SDNode *N, SDValue Res0, SDValue Res1, bool AddTo) { return ((DAGCombiner*)DC)->CombineTo(N, Res0, Res1, AddTo); } void TargetLowering::DAGCombinerInfo:: CommitTargetLoweringOpt(const TargetLowering::TargetLoweringOpt &TLO) { return ((DAGCombiner*)DC)->CommitTargetLoweringOpt(TLO); } //===----------------------------------------------------------------------===// // Helper Functions //===----------------------------------------------------------------------===// void DAGCombiner::deleteAndRecombine(SDNode *N) { removeFromWorklist(N); // If the operands of this node are only used by the node, they will now be // dead. Make sure to re-visit them and recursively delete dead nodes. for (const SDValue &Op : N->ops()) // For an operand generating multiple values, one of the values may // become dead allowing further simplification (e.g. split index // arithmetic from an indexed load). if (Op->hasOneUse() || Op->getNumValues() > 1) AddToWorklist(Op.getNode()); DAG.DeleteNode(N); } /// Return 1 if we can compute the negated form of the specified expression for /// the same cost as the expression itself, or 2 if we can compute the negated /// form more cheaply than the expression itself. static char isNegatibleForFree(SDValue Op, bool LegalOperations, const TargetLowering &TLI, const TargetOptions *Options, unsigned Depth = 0) { // fneg is removable even if it has multiple uses. if (Op.getOpcode() == ISD::FNEG) return 2; // Don't allow anything with multiple uses unless we know it is free. EVT VT = Op.getValueType(); const SDNodeFlags Flags = Op->getFlags(); if (!Op.hasOneUse()) if (!(Op.getOpcode() == ISD::FP_EXTEND && TLI.isFPExtFree(VT, Op.getOperand(0).getValueType()))) return 0; // Don't recurse exponentially. if (Depth > 6) return 0; switch (Op.getOpcode()) { default: return false; case ISD::ConstantFP: { if (!LegalOperations) return 1; // Don't invert constant FP values after legalization unless the target says // the negated constant is legal. return TLI.isOperationLegal(ISD::ConstantFP, VT) || TLI.isFPImmLegal(neg(cast(Op)->getValueAPF()), VT); } case ISD::FADD: if (!Options->UnsafeFPMath && !Flags.hasNoSignedZeros()) return 0; // After operation legalization, it might not be legal to create new FSUBs. if (LegalOperations && !TLI.isOperationLegalOrCustom(ISD::FSUB, VT)) return 0; // fold (fneg (fadd A, B)) -> (fsub (fneg A), B) if (char V = isNegatibleForFree(Op.getOperand(0), LegalOperations, TLI, Options, Depth + 1)) return V; // fold (fneg (fadd A, B)) -> (fsub (fneg B), A) return isNegatibleForFree(Op.getOperand(1), LegalOperations, TLI, Options, Depth + 1); case ISD::FSUB: // We can't turn -(A-B) into B-A when we honor signed zeros. if (!Options->NoSignedZerosFPMath && !Flags.hasNoSignedZeros()) return 0; // fold (fneg (fsub A, B)) -> (fsub B, A) return 1; case ISD::FMUL: case ISD::FDIV: // fold (fneg (fmul X, Y)) -> (fmul (fneg X), Y) or (fmul X, (fneg Y)) if (char V = isNegatibleForFree(Op.getOperand(0), LegalOperations, TLI, Options, Depth + 1)) return V; return isNegatibleForFree(Op.getOperand(1), LegalOperations, TLI, Options, Depth + 1); case ISD::FP_EXTEND: case ISD::FP_ROUND: case ISD::FSIN: return isNegatibleForFree(Op.getOperand(0), LegalOperations, TLI, Options, Depth + 1); } } /// If isNegatibleForFree returns true, return the newly negated expression. static SDValue GetNegatedExpression(SDValue Op, SelectionDAG &DAG, bool LegalOperations, unsigned Depth = 0) { const TargetOptions &Options = DAG.getTarget().Options; // fneg is removable even if it has multiple uses. if (Op.getOpcode() == ISD::FNEG) return Op.getOperand(0); assert(Depth <= 6 && "GetNegatedExpression doesn't match isNegatibleForFree"); const SDNodeFlags Flags = Op.getNode()->getFlags(); switch (Op.getOpcode()) { default: llvm_unreachable("Unknown code"); case ISD::ConstantFP: { APFloat V = cast(Op)->getValueAPF(); V.changeSign(); return DAG.getConstantFP(V, SDLoc(Op), Op.getValueType()); } case ISD::FADD: assert(Options.UnsafeFPMath || Flags.hasNoSignedZeros()); // fold (fneg (fadd A, B)) -> (fsub (fneg A), B) if (isNegatibleForFree(Op.getOperand(0), LegalOperations, DAG.getTargetLoweringInfo(), &Options, Depth+1)) return DAG.getNode(ISD::FSUB, SDLoc(Op), Op.getValueType(), GetNegatedExpression(Op.getOperand(0), DAG, LegalOperations, Depth+1), Op.getOperand(1), Flags); // fold (fneg (fadd A, B)) -> (fsub (fneg B), A) return DAG.getNode(ISD::FSUB, SDLoc(Op), Op.getValueType(), GetNegatedExpression(Op.getOperand(1), DAG, LegalOperations, Depth+1), Op.getOperand(0), Flags); case ISD::FSUB: // fold (fneg (fsub 0, B)) -> B if (ConstantFPSDNode *N0CFP = dyn_cast(Op.getOperand(0))) if (N0CFP->isZero()) return Op.getOperand(1); // fold (fneg (fsub A, B)) -> (fsub B, A) return DAG.getNode(ISD::FSUB, SDLoc(Op), Op.getValueType(), Op.getOperand(1), Op.getOperand(0), Flags); case ISD::FMUL: case ISD::FDIV: // fold (fneg (fmul X, Y)) -> (fmul (fneg X), Y) if (isNegatibleForFree(Op.getOperand(0), LegalOperations, DAG.getTargetLoweringInfo(), &Options, Depth+1)) return DAG.getNode(Op.getOpcode(), SDLoc(Op), Op.getValueType(), GetNegatedExpression(Op.getOperand(0), DAG, LegalOperations, Depth+1), Op.getOperand(1), Flags); // fold (fneg (fmul X, Y)) -> (fmul X, (fneg Y)) return DAG.getNode(Op.getOpcode(), SDLoc(Op), Op.getValueType(), Op.getOperand(0), GetNegatedExpression(Op.getOperand(1), DAG, LegalOperations, Depth+1), Flags); case ISD::FP_EXTEND: case ISD::FSIN: return DAG.getNode(Op.getOpcode(), SDLoc(Op), Op.getValueType(), GetNegatedExpression(Op.getOperand(0), DAG, LegalOperations, Depth+1)); case ISD::FP_ROUND: return DAG.getNode(ISD::FP_ROUND, SDLoc(Op), Op.getValueType(), GetNegatedExpression(Op.getOperand(0), DAG, LegalOperations, Depth+1), Op.getOperand(1)); } } // APInts must be the same size for most operations, this helper // function zero extends the shorter of the pair so that they match. // We provide an Offset so that we can create bitwidths that won't overflow. static void zeroExtendToMatch(APInt &LHS, APInt &RHS, unsigned Offset = 0) { unsigned Bits = Offset + std::max(LHS.getBitWidth(), RHS.getBitWidth()); LHS = LHS.zextOrSelf(Bits); RHS = RHS.zextOrSelf(Bits); } // Return true if this node is a setcc, or is a select_cc // that selects between the target values used for true and false, making it // equivalent to a setcc. Also, set the incoming LHS, RHS, and CC references to // the appropriate nodes based on the type of node we are checking. This // simplifies life a bit for the callers. bool DAGCombiner::isSetCCEquivalent(SDValue N, SDValue &LHS, SDValue &RHS, SDValue &CC) const { if (N.getOpcode() == ISD::SETCC) { LHS = N.getOperand(0); RHS = N.getOperand(1); CC = N.getOperand(2); return true; } if (N.getOpcode() != ISD::SELECT_CC || !TLI.isConstTrueVal(N.getOperand(2).getNode()) || !TLI.isConstFalseVal(N.getOperand(3).getNode())) return false; if (TLI.getBooleanContents(N.getValueType()) == TargetLowering::UndefinedBooleanContent) return false; LHS = N.getOperand(0); RHS = N.getOperand(1); CC = N.getOperand(4); return true; } /// Return true if this is a SetCC-equivalent operation with only one use. /// If this is true, it allows the users to invert the operation for free when /// it is profitable to do so. bool DAGCombiner::isOneUseSetCC(SDValue N) const { SDValue N0, N1, N2; if (isSetCCEquivalent(N, N0, N1, N2) && N.getNode()->hasOneUse()) return true; return false; } // Returns the SDNode if it is a constant float BuildVector // or constant float. static SDNode *isConstantFPBuildVectorOrConstantFP(SDValue N) { if (isa(N)) return N.getNode(); if (ISD::isBuildVectorOfConstantFPSDNodes(N.getNode())) return N.getNode(); return nullptr; } // Determines if it is a constant integer or a build vector of constant // integers (and undefs). // Do not permit build vector implicit truncation. static bool isConstantOrConstantVector(SDValue N, bool NoOpaques = false) { if (ConstantSDNode *Const = dyn_cast(N)) return !(Const->isOpaque() && NoOpaques); if (N.getOpcode() != ISD::BUILD_VECTOR) return false; unsigned BitWidth = N.getScalarValueSizeInBits(); for (const SDValue &Op : N->op_values()) { if (Op.isUndef()) continue; ConstantSDNode *Const = dyn_cast(Op); if (!Const || Const->getAPIntValue().getBitWidth() != BitWidth || (Const->isOpaque() && NoOpaques)) return false; } return true; } // Determines if a BUILD_VECTOR is composed of all-constants possibly mixed with // undef's. static bool isAnyConstantBuildVector(SDValue V, bool NoOpaques = false) { if (V.getOpcode() != ISD::BUILD_VECTOR) return false; return isConstantOrConstantVector(V, NoOpaques) || ISD::isBuildVectorOfConstantFPSDNodes(V.getNode()); } SDValue DAGCombiner::ReassociateOps(unsigned Opc, const SDLoc &DL, SDValue N0, SDValue N1, SDNodeFlags Flags) { // Don't reassociate reductions. if (Flags.hasVectorReduction()) return SDValue(); EVT VT = N0.getValueType(); if (N0.getOpcode() == Opc && !N0->getFlags().hasVectorReduction()) { if (SDNode *L = DAG.isConstantIntBuildVectorOrConstantInt(N0.getOperand(1))) { if (SDNode *R = DAG.isConstantIntBuildVectorOrConstantInt(N1)) { // reassoc. (op (op x, c1), c2) -> (op x, (op c1, c2)) if (SDValue OpNode = DAG.FoldConstantArithmetic(Opc, DL, VT, L, R)) return DAG.getNode(Opc, DL, VT, N0.getOperand(0), OpNode); return SDValue(); } if (N0.hasOneUse()) { // reassoc. (op (op x, c1), y) -> (op (op x, y), c1) iff x+c1 has one // use SDValue OpNode = DAG.getNode(Opc, SDLoc(N0), VT, N0.getOperand(0), N1); if (!OpNode.getNode()) return SDValue(); AddToWorklist(OpNode.getNode()); return DAG.getNode(Opc, DL, VT, OpNode, N0.getOperand(1)); } } } if (N1.getOpcode() == Opc && !N1->getFlags().hasVectorReduction()) { if (SDNode *R = DAG.isConstantIntBuildVectorOrConstantInt(N1.getOperand(1))) { if (SDNode *L = DAG.isConstantIntBuildVectorOrConstantInt(N0)) { // reassoc. (op c2, (op x, c1)) -> (op x, (op c1, c2)) if (SDValue OpNode = DAG.FoldConstantArithmetic(Opc, DL, VT, R, L)) return DAG.getNode(Opc, DL, VT, N1.getOperand(0), OpNode); return SDValue(); } if (N1.hasOneUse()) { // reassoc. (op x, (op y, c1)) -> (op (op x, y), c1) iff x+c1 has one // use SDValue OpNode = DAG.getNode(Opc, SDLoc(N0), VT, N0, N1.getOperand(0)); if (!OpNode.getNode()) return SDValue(); AddToWorklist(OpNode.getNode()); return DAG.getNode(Opc, DL, VT, OpNode, N1.getOperand(1)); } } } return SDValue(); } SDValue DAGCombiner::CombineTo(SDNode *N, const SDValue *To, unsigned NumTo, bool AddTo) { assert(N->getNumValues() == NumTo && "Broken CombineTo call!"); ++NodesCombined; LLVM_DEBUG(dbgs() << "\nReplacing.1 "; N->dump(&DAG); dbgs() << "\nWith: "; To[0].getNode()->dump(&DAG); dbgs() << " and " << NumTo - 1 << " other values\n"); for (unsigned i = 0, e = NumTo; i != e; ++i) assert((!To[i].getNode() || N->getValueType(i) == To[i].getValueType()) && "Cannot combine value to value of different type!"); WorklistRemover DeadNodes(*this); DAG.ReplaceAllUsesWith(N, To); if (AddTo) { // Push the new nodes and any users onto the worklist for (unsigned i = 0, e = NumTo; i != e; ++i) { if (To[i].getNode()) { AddToWorklist(To[i].getNode()); AddUsersToWorklist(To[i].getNode()); } } } // Finally, if the node is now dead, remove it from the graph. The node // may not be dead if the replacement process recursively simplified to // something else needing this node. if (N->use_empty()) deleteAndRecombine(N); return SDValue(N, 0); } void DAGCombiner:: CommitTargetLoweringOpt(const TargetLowering::TargetLoweringOpt &TLO) { // Replace all uses. If any nodes become isomorphic to other nodes and // are deleted, make sure to remove them from our worklist. WorklistRemover DeadNodes(*this); DAG.ReplaceAllUsesOfValueWith(TLO.Old, TLO.New); // Push the new node and any (possibly new) users onto the worklist. AddToWorklist(TLO.New.getNode()); AddUsersToWorklist(TLO.New.getNode()); // Finally, if the node is now dead, remove it from the graph. The node // may not be dead if the replacement process recursively simplified to // something else needing this node. if (TLO.Old.getNode()->use_empty()) deleteAndRecombine(TLO.Old.getNode()); } /// Check the specified integer node value to see if it can be simplified or if /// things it uses can be simplified by bit propagation. If so, return true. bool DAGCombiner::SimplifyDemandedBits(SDValue Op, const APInt &Demanded) { TargetLowering::TargetLoweringOpt TLO(DAG, LegalTypes, LegalOperations); KnownBits Known; if (!TLI.SimplifyDemandedBits(Op, Demanded, Known, TLO)) return false; // Revisit the node. AddToWorklist(Op.getNode()); // Replace the old value with the new one. ++NodesCombined; LLVM_DEBUG(dbgs() << "\nReplacing.2 "; TLO.Old.getNode()->dump(&DAG); dbgs() << "\nWith: "; TLO.New.getNode()->dump(&DAG); dbgs() << '\n'); CommitTargetLoweringOpt(TLO); return true; } /// Check the specified vector node value to see if it can be simplified or /// if things it uses can be simplified as it only uses some of the elements. /// If so, return true. bool DAGCombiner::SimplifyDemandedVectorElts(SDValue Op, const APInt &Demanded, bool AssumeSingleUse) { TargetLowering::TargetLoweringOpt TLO(DAG, LegalTypes, LegalOperations); APInt KnownUndef, KnownZero; if (!TLI.SimplifyDemandedVectorElts(Op, Demanded, KnownUndef, KnownZero, TLO, 0, AssumeSingleUse)) return false; // Revisit the node. AddToWorklist(Op.getNode()); // Replace the old value with the new one. ++NodesCombined; LLVM_DEBUG(dbgs() << "\nReplacing.2 "; TLO.Old.getNode()->dump(&DAG); dbgs() << "\nWith: "; TLO.New.getNode()->dump(&DAG); dbgs() << '\n'); CommitTargetLoweringOpt(TLO); return true; } void DAGCombiner::ReplaceLoadWithPromotedLoad(SDNode *Load, SDNode *ExtLoad) { SDLoc DL(Load); EVT VT = Load->getValueType(0); SDValue Trunc = DAG.getNode(ISD::TRUNCATE, DL, VT, SDValue(ExtLoad, 0)); LLVM_DEBUG(dbgs() << "\nReplacing.9 "; Load->dump(&DAG); dbgs() << "\nWith: "; Trunc.getNode()->dump(&DAG); dbgs() << '\n'); WorklistRemover DeadNodes(*this); DAG.ReplaceAllUsesOfValueWith(SDValue(Load, 0), Trunc); DAG.ReplaceAllUsesOfValueWith(SDValue(Load, 1), SDValue(ExtLoad, 1)); deleteAndRecombine(Load); AddToWorklist(Trunc.getNode()); } SDValue DAGCombiner::PromoteOperand(SDValue Op, EVT PVT, bool &Replace) { Replace = false; SDLoc DL(Op); if (ISD::isUNINDEXEDLoad(Op.getNode())) { LoadSDNode *LD = cast(Op); EVT MemVT = LD->getMemoryVT(); ISD::LoadExtType ExtType = ISD::isNON_EXTLoad(LD) ? ISD::EXTLOAD : LD->getExtensionType(); Replace = true; return DAG.getExtLoad(ExtType, DL, PVT, LD->getChain(), LD->getBasePtr(), MemVT, LD->getMemOperand()); } unsigned Opc = Op.getOpcode(); switch (Opc) { default: break; case ISD::AssertSext: if (SDValue Op0 = SExtPromoteOperand(Op.getOperand(0), PVT)) return DAG.getNode(ISD::AssertSext, DL, PVT, Op0, Op.getOperand(1)); break; case ISD::AssertZext: if (SDValue Op0 = ZExtPromoteOperand(Op.getOperand(0), PVT)) return DAG.getNode(ISD::AssertZext, DL, PVT, Op0, Op.getOperand(1)); break; case ISD::Constant: { unsigned ExtOpc = Op.getValueType().isByteSized() ? ISD::SIGN_EXTEND : ISD::ZERO_EXTEND; return DAG.getNode(ExtOpc, DL, PVT, Op); } } if (!TLI.isOperationLegal(ISD::ANY_EXTEND, PVT)) return SDValue(); return DAG.getNode(ISD::ANY_EXTEND, DL, PVT, Op); } SDValue DAGCombiner::SExtPromoteOperand(SDValue Op, EVT PVT) { if (!TLI.isOperationLegal(ISD::SIGN_EXTEND_INREG, PVT)) return SDValue(); EVT OldVT = Op.getValueType(); SDLoc DL(Op); bool Replace = false; SDValue NewOp = PromoteOperand(Op, PVT, Replace); if (!NewOp.getNode()) return SDValue(); AddToWorklist(NewOp.getNode()); if (Replace) ReplaceLoadWithPromotedLoad(Op.getNode(), NewOp.getNode()); return DAG.getNode(ISD::SIGN_EXTEND_INREG, DL, NewOp.getValueType(), NewOp, DAG.getValueType(OldVT)); } SDValue DAGCombiner::ZExtPromoteOperand(SDValue Op, EVT PVT) { EVT OldVT = Op.getValueType(); SDLoc DL(Op); bool Replace = false; SDValue NewOp = PromoteOperand(Op, PVT, Replace); if (!NewOp.getNode()) return SDValue(); AddToWorklist(NewOp.getNode()); if (Replace) ReplaceLoadWithPromotedLoad(Op.getNode(), NewOp.getNode()); return DAG.getZeroExtendInReg(NewOp, DL, OldVT); } /// Promote the specified integer binary operation if the target indicates it is /// beneficial. e.g. On x86, it's usually better to promote i16 operations to /// i32 since i16 instructions are longer. SDValue DAGCombiner::PromoteIntBinOp(SDValue Op) { if (!LegalOperations) return SDValue(); EVT VT = Op.getValueType(); if (VT.isVector() || !VT.isInteger()) return SDValue(); // If operation type is 'undesirable', e.g. i16 on x86, consider // promoting it. unsigned Opc = Op.getOpcode(); if (TLI.isTypeDesirableForOp(Opc, VT)) return SDValue(); EVT PVT = VT; // Consult target whether it is a good idea to promote this operation and // what's the right type to promote it to. if (TLI.IsDesirableToPromoteOp(Op, PVT)) { assert(PVT != VT && "Don't know what type to promote to!"); LLVM_DEBUG(dbgs() << "\nPromoting "; Op.getNode()->dump(&DAG)); bool Replace0 = false; SDValue N0 = Op.getOperand(0); SDValue NN0 = PromoteOperand(N0, PVT, Replace0); bool Replace1 = false; SDValue N1 = Op.getOperand(1); SDValue NN1 = PromoteOperand(N1, PVT, Replace1); SDLoc DL(Op); SDValue RV = DAG.getNode(ISD::TRUNCATE, DL, VT, DAG.getNode(Opc, DL, PVT, NN0, NN1)); // We are always replacing N0/N1's use in N and only need // additional replacements if there are additional uses. Replace0 &= !N0->hasOneUse(); Replace1 &= (N0 != N1) && !N1->hasOneUse(); // Combine Op here so it is preserved past replacements. CombineTo(Op.getNode(), RV); // If operands have a use ordering, make sure we deal with // predecessor first. if (Replace0 && Replace1 && N0.getNode()->isPredecessorOf(N1.getNode())) { std::swap(N0, N1); std::swap(NN0, NN1); } if (Replace0) { AddToWorklist(NN0.getNode()); ReplaceLoadWithPromotedLoad(N0.getNode(), NN0.getNode()); } if (Replace1) { AddToWorklist(NN1.getNode()); ReplaceLoadWithPromotedLoad(N1.getNode(), NN1.getNode()); } return Op; } return SDValue(); } /// Promote the specified integer shift operation if the target indicates it is /// beneficial. e.g. On x86, it's usually better to promote i16 operations to /// i32 since i16 instructions are longer. SDValue DAGCombiner::PromoteIntShiftOp(SDValue Op) { if (!LegalOperations) return SDValue(); EVT VT = Op.getValueType(); if (VT.isVector() || !VT.isInteger()) return SDValue(); // If operation type is 'undesirable', e.g. i16 on x86, consider // promoting it. unsigned Opc = Op.getOpcode(); if (TLI.isTypeDesirableForOp(Opc, VT)) return SDValue(); EVT PVT = VT; // Consult target whether it is a good idea to promote this operation and // what's the right type to promote it to. if (TLI.IsDesirableToPromoteOp(Op, PVT)) { assert(PVT != VT && "Don't know what type to promote to!"); LLVM_DEBUG(dbgs() << "\nPromoting "; Op.getNode()->dump(&DAG)); bool Replace = false; SDValue N0 = Op.getOperand(0); SDValue N1 = Op.getOperand(1); if (Opc == ISD::SRA) N0 = SExtPromoteOperand(N0, PVT); else if (Opc == ISD::SRL) N0 = ZExtPromoteOperand(N0, PVT); else N0 = PromoteOperand(N0, PVT, Replace); if (!N0.getNode()) return SDValue(); SDLoc DL(Op); SDValue RV = DAG.getNode(ISD::TRUNCATE, DL, VT, DAG.getNode(Opc, DL, PVT, N0, N1)); AddToWorklist(N0.getNode()); if (Replace) ReplaceLoadWithPromotedLoad(Op.getOperand(0).getNode(), N0.getNode()); // Deal with Op being deleted. if (Op && Op.getOpcode() != ISD::DELETED_NODE) return RV; } return SDValue(); } SDValue DAGCombiner::PromoteExtend(SDValue Op) { if (!LegalOperations) return SDValue(); EVT VT = Op.getValueType(); if (VT.isVector() || !VT.isInteger()) return SDValue(); // If operation type is 'undesirable', e.g. i16 on x86, consider // promoting it. unsigned Opc = Op.getOpcode(); if (TLI.isTypeDesirableForOp(Opc, VT)) return SDValue(); EVT PVT = VT; // Consult target whether it is a good idea to promote this operation and // what's the right type to promote it to. if (TLI.IsDesirableToPromoteOp(Op, PVT)) { assert(PVT != VT && "Don't know what type to promote to!"); // fold (aext (aext x)) -> (aext x) // fold (aext (zext x)) -> (zext x) // fold (aext (sext x)) -> (sext x) LLVM_DEBUG(dbgs() << "\nPromoting "; Op.getNode()->dump(&DAG)); return DAG.getNode(Op.getOpcode(), SDLoc(Op), VT, Op.getOperand(0)); } return SDValue(); } bool DAGCombiner::PromoteLoad(SDValue Op) { if (!LegalOperations) return false; if (!ISD::isUNINDEXEDLoad(Op.getNode())) return false; EVT VT = Op.getValueType(); if (VT.isVector() || !VT.isInteger()) return false; // If operation type is 'undesirable', e.g. i16 on x86, consider // promoting it. unsigned Opc = Op.getOpcode(); if (TLI.isTypeDesirableForOp(Opc, VT)) return false; EVT PVT = VT; // Consult target whether it is a good idea to promote this operation and // what's the right type to promote it to. if (TLI.IsDesirableToPromoteOp(Op, PVT)) { assert(PVT != VT && "Don't know what type to promote to!"); SDLoc DL(Op); SDNode *N = Op.getNode(); LoadSDNode *LD = cast(N); EVT MemVT = LD->getMemoryVT(); ISD::LoadExtType ExtType = ISD::isNON_EXTLoad(LD) ? ISD::EXTLOAD : LD->getExtensionType(); SDValue NewLD = DAG.getExtLoad(ExtType, DL, PVT, LD->getChain(), LD->getBasePtr(), MemVT, LD->getMemOperand()); SDValue Result = DAG.getNode(ISD::TRUNCATE, DL, VT, NewLD); LLVM_DEBUG(dbgs() << "\nPromoting "; N->dump(&DAG); dbgs() << "\nTo: "; Result.getNode()->dump(&DAG); dbgs() << '\n'); WorklistRemover DeadNodes(*this); DAG.ReplaceAllUsesOfValueWith(SDValue(N, 0), Result); DAG.ReplaceAllUsesOfValueWith(SDValue(N, 1), NewLD.getValue(1)); deleteAndRecombine(N); AddToWorklist(Result.getNode()); return true; } return false; } /// Recursively delete a node which has no uses and any operands for /// which it is the only use. /// /// Note that this both deletes the nodes and removes them from the worklist. /// It also adds any nodes who have had a user deleted to the worklist as they /// may now have only one use and subject to other combines. bool DAGCombiner::recursivelyDeleteUnusedNodes(SDNode *N) { if (!N->use_empty()) return false; SmallSetVector Nodes; Nodes.insert(N); do { N = Nodes.pop_back_val(); if (!N) continue; if (N->use_empty()) { for (const SDValue &ChildN : N->op_values()) Nodes.insert(ChildN.getNode()); removeFromWorklist(N); DAG.DeleteNode(N); } else { AddToWorklist(N); } } while (!Nodes.empty()); return true; } //===----------------------------------------------------------------------===// // Main DAG Combiner implementation //===----------------------------------------------------------------------===// void DAGCombiner::Run(CombineLevel AtLevel) { // set the instance variables, so that the various visit routines may use it. Level = AtLevel; LegalOperations = Level >= AfterLegalizeVectorOps; LegalTypes = Level >= AfterLegalizeTypes; // Add all the dag nodes to the worklist. for (SDNode &Node : DAG.allnodes()) AddToWorklist(&Node); // Create a dummy node (which is not added to allnodes), that adds a reference // to the root node, preventing it from being deleted, and tracking any // changes of the root. HandleSDNode Dummy(DAG.getRoot()); // While the worklist isn't empty, find a node and try to combine it. while (!WorklistMap.empty()) { SDNode *N; // The Worklist holds the SDNodes in order, but it may contain null entries. do { N = Worklist.pop_back_val(); } while (!N); bool GoodWorklistEntry = WorklistMap.erase(N); (void)GoodWorklistEntry; assert(GoodWorklistEntry && "Found a worklist entry without a corresponding map entry!"); // If N has no uses, it is dead. Make sure to revisit all N's operands once // N is deleted from the DAG, since they too may now be dead or may have a // reduced number of uses, allowing other xforms. if (recursivelyDeleteUnusedNodes(N)) continue; WorklistRemover DeadNodes(*this); // If this combine is running after legalizing the DAG, re-legalize any // nodes pulled off the worklist. if (Level == AfterLegalizeDAG) { SmallSetVector UpdatedNodes; bool NIsValid = DAG.LegalizeOp(N, UpdatedNodes); for (SDNode *LN : UpdatedNodes) { AddToWorklist(LN); AddUsersToWorklist(LN); } if (!NIsValid) continue; } LLVM_DEBUG(dbgs() << "\nCombining: "; N->dump(&DAG)); // Add any operands of the new node which have not yet been combined to the // worklist as well. Because the worklist uniques things already, this // won't repeatedly process the same operand. CombinedNodes.insert(N); for (const SDValue &ChildN : N->op_values()) if (!CombinedNodes.count(ChildN.getNode())) AddToWorklist(ChildN.getNode()); SDValue RV = combine(N); if (!RV.getNode()) continue; ++NodesCombined; // If we get back the same node we passed in, rather than a new node or // zero, we know that the node must have defined multiple values and // CombineTo was used. Since CombineTo takes care of the worklist // mechanics for us, we have no work to do in this case. if (RV.getNode() == N) continue; assert(N->getOpcode() != ISD::DELETED_NODE && RV.getOpcode() != ISD::DELETED_NODE && "Node was deleted but visit returned new node!"); LLVM_DEBUG(dbgs() << " ... into: "; RV.getNode()->dump(&DAG)); if (N->getNumValues() == RV.getNode()->getNumValues()) DAG.ReplaceAllUsesWith(N, RV.getNode()); else { assert(N->getValueType(0) == RV.getValueType() && N->getNumValues() == 1 && "Type mismatch"); DAG.ReplaceAllUsesWith(N, &RV); } // Push the new node and any users onto the worklist AddToWorklist(RV.getNode()); AddUsersToWorklist(RV.getNode()); // Finally, if the node is now dead, remove it from the graph. The node // may not be dead if the replacement process recursively simplified to // something else needing this node. This will also take care of adding any // operands which have lost a user to the worklist. recursivelyDeleteUnusedNodes(N); } // If the root changed (e.g. it was a dead load, update the root). DAG.setRoot(Dummy.getValue()); DAG.RemoveDeadNodes(); } SDValue DAGCombiner::visit(SDNode *N) { switch (N->getOpcode()) { default: break; case ISD::TokenFactor: return visitTokenFactor(N); case ISD::MERGE_VALUES: return visitMERGE_VALUES(N); case ISD::ADD: return visitADD(N); case ISD::SUB: return visitSUB(N); case ISD::SADDSAT: case ISD::UADDSAT: return visitADDSAT(N); case ISD::SSUBSAT: case ISD::USUBSAT: return visitSUBSAT(N); case ISD::ADDC: return visitADDC(N); case ISD::UADDO: return visitUADDO(N); case ISD::SUBC: return visitSUBC(N); case ISD::USUBO: return visitUSUBO(N); case ISD::ADDE: return visitADDE(N); case ISD::ADDCARRY: return visitADDCARRY(N); case ISD::SUBE: return visitSUBE(N); case ISD::SUBCARRY: return visitSUBCARRY(N); case ISD::MUL: return visitMUL(N); case ISD::SDIV: return visitSDIV(N); case ISD::UDIV: return visitUDIV(N); case ISD::SREM: case ISD::UREM: return visitREM(N); case ISD::MULHU: return visitMULHU(N); case ISD::MULHS: return visitMULHS(N); case ISD::SMUL_LOHI: return visitSMUL_LOHI(N); case ISD::UMUL_LOHI: return visitUMUL_LOHI(N); case ISD::SMULO: return visitSMULO(N); case ISD::UMULO: return visitUMULO(N); case ISD::SMIN: case ISD::SMAX: case ISD::UMIN: case ISD::UMAX: return visitIMINMAX(N); case ISD::AND: return visitAND(N); case ISD::OR: return visitOR(N); case ISD::XOR: return visitXOR(N); case ISD::SHL: return visitSHL(N); case ISD::SRA: return visitSRA(N); case ISD::SRL: return visitSRL(N); case ISD::ROTR: case ISD::ROTL: return visitRotate(N); case ISD::FSHL: case ISD::FSHR: return visitFunnelShift(N); case ISD::ABS: return visitABS(N); case ISD::BSWAP: return visitBSWAP(N); case ISD::BITREVERSE: return visitBITREVERSE(N); case ISD::CTLZ: return visitCTLZ(N); case ISD::CTLZ_ZERO_UNDEF: return visitCTLZ_ZERO_UNDEF(N); case ISD::CTTZ: return visitCTTZ(N); case ISD::CTTZ_ZERO_UNDEF: return visitCTTZ_ZERO_UNDEF(N); case ISD::CTPOP: return visitCTPOP(N); case ISD::SELECT: return visitSELECT(N); case ISD::VSELECT: return visitVSELECT(N); case ISD::SELECT_CC: return visitSELECT_CC(N); case ISD::SETCC: return visitSETCC(N); case ISD::SETCCCARRY: return visitSETCCCARRY(N); case ISD::SIGN_EXTEND: return visitSIGN_EXTEND(N); case ISD::ZERO_EXTEND: return visitZERO_EXTEND(N); case ISD::ANY_EXTEND: return visitANY_EXTEND(N); case ISD::AssertSext: case ISD::AssertZext: return visitAssertExt(N); case ISD::SIGN_EXTEND_INREG: return visitSIGN_EXTEND_INREG(N); case ISD::SIGN_EXTEND_VECTOR_INREG: return visitSIGN_EXTEND_VECTOR_INREG(N); case ISD::ZERO_EXTEND_VECTOR_INREG: return visitZERO_EXTEND_VECTOR_INREG(N); case ISD::TRUNCATE: return visitTRUNCATE(N); case ISD::BITCAST: return visitBITCAST(N); case ISD::BUILD_PAIR: return visitBUILD_PAIR(N); case ISD::FADD: return visitFADD(N); case ISD::FSUB: return visitFSUB(N); case ISD::FMUL: return visitFMUL(N); case ISD::FMA: return visitFMA(N); case ISD::FDIV: return visitFDIV(N); case ISD::FREM: return visitFREM(N); case ISD::FSQRT: return visitFSQRT(N); case ISD::FCOPYSIGN: return visitFCOPYSIGN(N); case ISD::FPOW: return visitFPOW(N); case ISD::SINT_TO_FP: return visitSINT_TO_FP(N); case ISD::UINT_TO_FP: return visitUINT_TO_FP(N); case ISD::FP_TO_SINT: return visitFP_TO_SINT(N); case ISD::FP_TO_UINT: return visitFP_TO_UINT(N); case ISD::FP_ROUND: return visitFP_ROUND(N); case ISD::FP_ROUND_INREG: return visitFP_ROUND_INREG(N); case ISD::FP_EXTEND: return visitFP_EXTEND(N); case ISD::FNEG: return visitFNEG(N); case ISD::FABS: return visitFABS(N); case ISD::FFLOOR: return visitFFLOOR(N); case ISD::FMINNUM: return visitFMINNUM(N); case ISD::FMAXNUM: return visitFMAXNUM(N); case ISD::FMINIMUM: return visitFMINIMUM(N); case ISD::FMAXIMUM: return visitFMAXIMUM(N); case ISD::FCEIL: return visitFCEIL(N); case ISD::FTRUNC: return visitFTRUNC(N); case ISD::BRCOND: return visitBRCOND(N); case ISD::BR_CC: return visitBR_CC(N); case ISD::LOAD: return visitLOAD(N); case ISD::STORE: return visitSTORE(N); case ISD::INSERT_VECTOR_ELT: return visitINSERT_VECTOR_ELT(N); case ISD::EXTRACT_VECTOR_ELT: return visitEXTRACT_VECTOR_ELT(N); case ISD::BUILD_VECTOR: return visitBUILD_VECTOR(N); case ISD::CONCAT_VECTORS: return visitCONCAT_VECTORS(N); case ISD::EXTRACT_SUBVECTOR: return visitEXTRACT_SUBVECTOR(N); case ISD::VECTOR_SHUFFLE: return visitVECTOR_SHUFFLE(N); case ISD::SCALAR_TO_VECTOR: return visitSCALAR_TO_VECTOR(N); case ISD::INSERT_SUBVECTOR: return visitINSERT_SUBVECTOR(N); case ISD::MGATHER: return visitMGATHER(N); case ISD::MLOAD: return visitMLOAD(N); case ISD::MSCATTER: return visitMSCATTER(N); case ISD::MSTORE: return visitMSTORE(N); case ISD::FP_TO_FP16: return visitFP_TO_FP16(N); case ISD::FP16_TO_FP: return visitFP16_TO_FP(N); } return SDValue(); } SDValue DAGCombiner::combine(SDNode *N) { SDValue RV = visit(N); // If nothing happened, try a target-specific DAG combine. if (!RV.getNode()) { assert(N->getOpcode() != ISD::DELETED_NODE && "Node was deleted but visit returned NULL!"); if (N->getOpcode() >= ISD::BUILTIN_OP_END || TLI.hasTargetDAGCombine((ISD::NodeType)N->getOpcode())) { // Expose the DAG combiner to the target combiner impls. TargetLowering::DAGCombinerInfo DagCombineInfo(DAG, Level, false, this); RV = TLI.PerformDAGCombine(N, DagCombineInfo); } } // If nothing happened still, try promoting the operation. if (!RV.getNode()) { switch (N->getOpcode()) { default: break; case ISD::ADD: case ISD::SUB: case ISD::MUL: case ISD::AND: case ISD::OR: case ISD::XOR: RV = PromoteIntBinOp(SDValue(N, 0)); break; case ISD::SHL: case ISD::SRA: case ISD::SRL: RV = PromoteIntShiftOp(SDValue(N, 0)); break; case ISD::SIGN_EXTEND: case ISD::ZERO_EXTEND: case ISD::ANY_EXTEND: RV = PromoteExtend(SDValue(N, 0)); break; case ISD::LOAD: if (PromoteLoad(SDValue(N, 0))) RV = SDValue(N, 0); break; } } // If N is a commutative binary node, try eliminate it if the commuted // version is already present in the DAG. if (!RV.getNode() && TLI.isCommutativeBinOp(N->getOpcode()) && N->getNumValues() == 1) { SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); // Constant operands are canonicalized to RHS. if (N0 != N1 && (isa(N0) || !isa(N1))) { SDValue Ops[] = {N1, N0}; SDNode *CSENode = DAG.getNodeIfExists(N->getOpcode(), N->getVTList(), Ops, N->getFlags()); if (CSENode) return SDValue(CSENode, 0); } } return RV; } /// Given a node, return its input chain if it has one, otherwise return a null /// sd operand. static SDValue getInputChainForNode(SDNode *N) { if (unsigned NumOps = N->getNumOperands()) { if (N->getOperand(0).getValueType() == MVT::Other) return N->getOperand(0); if (N->getOperand(NumOps-1).getValueType() == MVT::Other) return N->getOperand(NumOps-1); for (unsigned i = 1; i < NumOps-1; ++i) if (N->getOperand(i).getValueType() == MVT::Other) return N->getOperand(i); } return SDValue(); } SDValue DAGCombiner::visitTokenFactor(SDNode *N) { // If N has two operands, where one has an input chain equal to the other, // the 'other' chain is redundant. if (N->getNumOperands() == 2) { if (getInputChainForNode(N->getOperand(0).getNode()) == N->getOperand(1)) return N->getOperand(0); if (getInputChainForNode(N->getOperand(1).getNode()) == N->getOperand(0)) return N->getOperand(1); } // Don't simplify token factors if optnone. if (OptLevel == CodeGenOpt::None) return SDValue(); SmallVector TFs; // List of token factors to visit. SmallVector Ops; // Ops for replacing token factor. SmallPtrSet SeenOps; bool Changed = false; // If we should replace this token factor. // Start out with this token factor. TFs.push_back(N); // Iterate through token factors. The TFs grows when new token factors are // encountered. for (unsigned i = 0; i < TFs.size(); ++i) { SDNode *TF = TFs[i]; // Check each of the operands. for (const SDValue &Op : TF->op_values()) { switch (Op.getOpcode()) { case ISD::EntryToken: // Entry tokens don't need to be added to the list. They are // redundant. Changed = true; break; case ISD::TokenFactor: if (Op.hasOneUse() && !is_contained(TFs, Op.getNode())) { // Queue up for processing. TFs.push_back(Op.getNode()); // Clean up in case the token factor is removed. AddToWorklist(Op.getNode()); Changed = true; break; } LLVM_FALLTHROUGH; default: // Only add if it isn't already in the list. if (SeenOps.insert(Op.getNode()).second) Ops.push_back(Op); else Changed = true; break; } } } // Remove Nodes that are chained to another node in the list. Do so // by walking up chains breath-first stopping when we've seen // another operand. In general we must climb to the EntryNode, but we can exit // early if we find all remaining work is associated with just one operand as // no further pruning is possible. // List of nodes to search through and original Ops from which they originate. SmallVector, 8> Worklist; SmallVector OpWorkCount; // Count of work for each Op. SmallPtrSet SeenChains; bool DidPruneOps = false; unsigned NumLeftToConsider = 0; for (const SDValue &Op : Ops) { Worklist.push_back(std::make_pair(Op.getNode(), NumLeftToConsider++)); OpWorkCount.push_back(1); } auto AddToWorklist = [&](unsigned CurIdx, SDNode *Op, unsigned OpNumber) { // If this is an Op, we can remove the op from the list. Remark any // search associated with it as from the current OpNumber. if (SeenOps.count(Op) != 0) { Changed = true; DidPruneOps = true; unsigned OrigOpNumber = 0; while (OrigOpNumber < Ops.size() && Ops[OrigOpNumber].getNode() != Op) OrigOpNumber++; assert((OrigOpNumber != Ops.size()) && "expected to find TokenFactor Operand"); // Re-mark worklist from OrigOpNumber to OpNumber for (unsigned i = CurIdx + 1; i < Worklist.size(); ++i) { if (Worklist[i].second == OrigOpNumber) { Worklist[i].second = OpNumber; } } OpWorkCount[OpNumber] += OpWorkCount[OrigOpNumber]; OpWorkCount[OrigOpNumber] = 0; NumLeftToConsider--; } // Add if it's a new chain if (SeenChains.insert(Op).second) { OpWorkCount[OpNumber]++; Worklist.push_back(std::make_pair(Op, OpNumber)); } }; for (unsigned i = 0; i < Worklist.size() && i < 1024; ++i) { // We need at least be consider at least 2 Ops to prune. if (NumLeftToConsider <= 1) break; auto CurNode = Worklist[i].first; auto CurOpNumber = Worklist[i].second; assert((OpWorkCount[CurOpNumber] > 0) && "Node should not appear in worklist"); switch (CurNode->getOpcode()) { case ISD::EntryToken: // Hitting EntryToken is the only way for the search to terminate without // hitting // another operand's search. Prevent us from marking this operand // considered. NumLeftToConsider++; break; case ISD::TokenFactor: for (const SDValue &Op : CurNode->op_values()) AddToWorklist(i, Op.getNode(), CurOpNumber); break; case ISD::CopyFromReg: case ISD::CopyToReg: AddToWorklist(i, CurNode->getOperand(0).getNode(), CurOpNumber); break; default: if (auto *MemNode = dyn_cast(CurNode)) AddToWorklist(i, MemNode->getChain().getNode(), CurOpNumber); break; } OpWorkCount[CurOpNumber]--; if (OpWorkCount[CurOpNumber] == 0) NumLeftToConsider--; } // If we've changed things around then replace token factor. if (Changed) { SDValue Result; if (Ops.empty()) { // The entry token is the only possible outcome. Result = DAG.getEntryNode(); } else { if (DidPruneOps) { SmallVector PrunedOps; // for (const SDValue &Op : Ops) { if (SeenChains.count(Op.getNode()) == 0) PrunedOps.push_back(Op); } Result = DAG.getNode(ISD::TokenFactor, SDLoc(N), MVT::Other, PrunedOps); } else { Result = DAG.getNode(ISD::TokenFactor, SDLoc(N), MVT::Other, Ops); } } return Result; } return SDValue(); } /// MERGE_VALUES can always be eliminated. SDValue DAGCombiner::visitMERGE_VALUES(SDNode *N) { WorklistRemover DeadNodes(*this); // Replacing results may cause a different MERGE_VALUES to suddenly // be CSE'd with N, and carry its uses with it. Iterate until no // uses remain, to ensure that the node can be safely deleted. // First add the users of this node to the work list so that they // can be tried again once they have new operands. AddUsersToWorklist(N); do { // Do as a single replacement to avoid rewalking use lists. SmallVector Ops; for (unsigned i = 0, e = N->getNumOperands(); i != e; ++i) Ops.push_back(N->getOperand(i)); DAG.ReplaceAllUsesWith(N, Ops.data()); } while (!N->use_empty()); deleteAndRecombine(N); return SDValue(N, 0); // Return N so it doesn't get rechecked! } /// If \p N is a ConstantSDNode with isOpaque() == false return it casted to a /// ConstantSDNode pointer else nullptr. static ConstantSDNode *getAsNonOpaqueConstant(SDValue N) { ConstantSDNode *Const = dyn_cast(N); return Const != nullptr && !Const->isOpaque() ? Const : nullptr; } SDValue DAGCombiner::foldBinOpIntoSelect(SDNode *BO) { assert(ISD::isBinaryOp(BO) && "Unexpected binary operator"); // Don't do this unless the old select is going away. We want to eliminate the // binary operator, not replace a binop with a select. // TODO: Handle ISD::SELECT_CC. unsigned SelOpNo = 0; SDValue Sel = BO->getOperand(0); if (Sel.getOpcode() != ISD::SELECT || !Sel.hasOneUse()) { SelOpNo = 1; Sel = BO->getOperand(1); } if (Sel.getOpcode() != ISD::SELECT || !Sel.hasOneUse()) return SDValue(); SDValue CT = Sel.getOperand(1); if (!isConstantOrConstantVector(CT, true) && !isConstantFPBuildVectorOrConstantFP(CT)) return SDValue(); SDValue CF = Sel.getOperand(2); if (!isConstantOrConstantVector(CF, true) && !isConstantFPBuildVectorOrConstantFP(CF)) return SDValue(); // Bail out if any constants are opaque because we can't constant fold those. // The exception is "and" and "or" with either 0 or -1 in which case we can // propagate non constant operands into select. I.e.: // and (select Cond, 0, -1), X --> select Cond, 0, X // or X, (select Cond, -1, 0) --> select Cond, -1, X auto BinOpcode = BO->getOpcode(); bool CanFoldNonConst = (BinOpcode == ISD::AND || BinOpcode == ISD::OR) && (isNullOrNullSplat(CT) || isAllOnesOrAllOnesSplat(CT)) && (isNullOrNullSplat(CF) || isAllOnesOrAllOnesSplat(CF)); SDValue CBO = BO->getOperand(SelOpNo ^ 1); if (!CanFoldNonConst && !isConstantOrConstantVector(CBO, true) && !isConstantFPBuildVectorOrConstantFP(CBO)) return SDValue(); EVT VT = Sel.getValueType(); // In case of shift value and shift amount may have different VT. For instance // on x86 shift amount is i8 regardles of LHS type. Bail out if we have // swapped operands and value types do not match. NB: x86 is fine if operands // are not swapped with shift amount VT being not bigger than shifted value. // TODO: that is possible to check for a shift operation, correct VTs and // still perform optimization on x86 if needed. if (SelOpNo && VT != CBO.getValueType()) return SDValue(); // We have a select-of-constants followed by a binary operator with a // constant. Eliminate the binop by pulling the constant math into the select. // Example: add (select Cond, CT, CF), CBO --> select Cond, CT + CBO, CF + CBO SDLoc DL(Sel); SDValue NewCT = SelOpNo ? DAG.getNode(BinOpcode, DL, VT, CBO, CT) : DAG.getNode(BinOpcode, DL, VT, CT, CBO); if (!CanFoldNonConst && !NewCT.isUndef() && !isConstantOrConstantVector(NewCT, true) && !isConstantFPBuildVectorOrConstantFP(NewCT)) return SDValue(); SDValue NewCF = SelOpNo ? DAG.getNode(BinOpcode, DL, VT, CBO, CF) : DAG.getNode(BinOpcode, DL, VT, CF, CBO); if (!CanFoldNonConst && !NewCF.isUndef() && !isConstantOrConstantVector(NewCF, true) && !isConstantFPBuildVectorOrConstantFP(NewCF)) return SDValue(); return DAG.getSelect(DL, VT, Sel.getOperand(0), NewCT, NewCF); } static SDValue foldAddSubBoolOfMaskedVal(SDNode *N, SelectionDAG &DAG) { assert((N->getOpcode() == ISD::ADD || N->getOpcode() == ISD::SUB) && "Expecting add or sub"); // Match a constant operand and a zext operand for the math instruction: // add Z, C // sub C, Z bool IsAdd = N->getOpcode() == ISD::ADD; SDValue C = IsAdd ? N->getOperand(1) : N->getOperand(0); SDValue Z = IsAdd ? N->getOperand(0) : N->getOperand(1); auto *CN = dyn_cast(C); if (!CN || Z.getOpcode() != ISD::ZERO_EXTEND) return SDValue(); // Match the zext operand as a setcc of a boolean. if (Z.getOperand(0).getOpcode() != ISD::SETCC || Z.getOperand(0).getValueType() != MVT::i1) return SDValue(); // Match the compare as: setcc (X & 1), 0, eq. SDValue SetCC = Z.getOperand(0); ISD::CondCode CC = cast(SetCC->getOperand(2))->get(); if (CC != ISD::SETEQ || !isNullConstant(SetCC.getOperand(1)) || SetCC.getOperand(0).getOpcode() != ISD::AND || !isOneConstant(SetCC.getOperand(0).getOperand(1))) return SDValue(); // We are adding/subtracting a constant and an inverted low bit. Turn that // into a subtract/add of the low bit with incremented/decremented constant: // add (zext i1 (seteq (X & 1), 0)), C --> sub C+1, (zext (X & 1)) // sub C, (zext i1 (seteq (X & 1), 0)) --> add C-1, (zext (X & 1)) EVT VT = C.getValueType(); SDLoc DL(N); SDValue LowBit = DAG.getZExtOrTrunc(SetCC.getOperand(0), DL, VT); SDValue C1 = IsAdd ? DAG.getConstant(CN->getAPIntValue() + 1, DL, VT) : DAG.getConstant(CN->getAPIntValue() - 1, DL, VT); return DAG.getNode(IsAdd ? ISD::SUB : ISD::ADD, DL, VT, C1, LowBit); } /// Try to fold a 'not' shifted sign-bit with add/sub with constant operand into /// a shift and add with a different constant. static SDValue foldAddSubOfSignBit(SDNode *N, SelectionDAG &DAG) { assert((N->getOpcode() == ISD::ADD || N->getOpcode() == ISD::SUB) && "Expecting add or sub"); // We need a constant operand for the add/sub, and the other operand is a // logical shift right: add (srl), C or sub C, (srl). bool IsAdd = N->getOpcode() == ISD::ADD; SDValue ConstantOp = IsAdd ? N->getOperand(1) : N->getOperand(0); SDValue ShiftOp = IsAdd ? N->getOperand(0) : N->getOperand(1); ConstantSDNode *C = isConstOrConstSplat(ConstantOp); if (!C || ShiftOp.getOpcode() != ISD::SRL) return SDValue(); // The shift must be of a 'not' value. SDValue Not = ShiftOp.getOperand(0); if (!Not.hasOneUse() || !isBitwiseNot(Not)) return SDValue(); // The shift must be moving the sign bit to the least-significant-bit. EVT VT = ShiftOp.getValueType(); SDValue ShAmt = ShiftOp.getOperand(1); ConstantSDNode *ShAmtC = isConstOrConstSplat(ShAmt); if (!ShAmtC || ShAmtC->getZExtValue() != VT.getScalarSizeInBits() - 1) return SDValue(); // Eliminate the 'not' by adjusting the shift and add/sub constant: // add (srl (not X), 31), C --> add (sra X, 31), (C + 1) // sub C, (srl (not X), 31) --> add (srl X, 31), (C - 1) SDLoc DL(N); auto ShOpcode = IsAdd ? ISD::SRA : ISD::SRL; SDValue NewShift = DAG.getNode(ShOpcode, DL, VT, Not.getOperand(0), ShAmt); APInt NewC = IsAdd ? C->getAPIntValue() + 1 : C->getAPIntValue() - 1; return DAG.getNode(ISD::ADD, DL, VT, NewShift, DAG.getConstant(NewC, DL, VT)); } SDValue DAGCombiner::visitADD(SDNode *N) { SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); EVT VT = N0.getValueType(); SDLoc DL(N); // fold vector ops if (VT.isVector()) { if (SDValue FoldedVOp = SimplifyVBinOp(N)) return FoldedVOp; // fold (add x, 0) -> x, vector edition if (ISD::isBuildVectorAllZeros(N1.getNode())) return N0; if (ISD::isBuildVectorAllZeros(N0.getNode())) return N1; } // fold (add x, undef) -> undef if (N0.isUndef()) return N0; if (N1.isUndef()) return N1; if (DAG.isConstantIntBuildVectorOrConstantInt(N0)) { // canonicalize constant to RHS if (!DAG.isConstantIntBuildVectorOrConstantInt(N1)) return DAG.getNode(ISD::ADD, DL, VT, N1, N0); // fold (add c1, c2) -> c1+c2 return DAG.FoldConstantArithmetic(ISD::ADD, DL, VT, N0.getNode(), N1.getNode()); } // fold (add x, 0) -> x if (isNullConstant(N1)) return N0; if (isConstantOrConstantVector(N1, /* NoOpaque */ true)) { // fold ((c1-A)+c2) -> (c1+c2)-A if (N0.getOpcode() == ISD::SUB && isConstantOrConstantVector(N0.getOperand(0), /* NoOpaque */ true)) { // FIXME: Adding 2 constants should be handled by FoldConstantArithmetic. return DAG.getNode(ISD::SUB, DL, VT, DAG.getNode(ISD::ADD, DL, VT, N1, N0.getOperand(0)), N0.getOperand(1)); } // add (sext i1 X), 1 -> zext (not i1 X) // We don't transform this pattern: // add (zext i1 X), -1 -> sext (not i1 X) // because most (?) targets generate better code for the zext form. if (N0.getOpcode() == ISD::SIGN_EXTEND && N0.hasOneUse() && isOneOrOneSplat(N1)) { SDValue X = N0.getOperand(0); if ((!LegalOperations || (TLI.isOperationLegal(ISD::XOR, X.getValueType()) && TLI.isOperationLegal(ISD::ZERO_EXTEND, VT))) && X.getScalarValueSizeInBits() == 1) { SDValue Not = DAG.getNOT(DL, X, X.getValueType()); return DAG.getNode(ISD::ZERO_EXTEND, DL, VT, Not); } } // Undo the add -> or combine to merge constant offsets from a frame index. if (N0.getOpcode() == ISD::OR && isa(N0.getOperand(0)) && isa(N0.getOperand(1)) && DAG.haveNoCommonBitsSet(N0.getOperand(0), N0.getOperand(1))) { SDValue Add0 = DAG.getNode(ISD::ADD, DL, VT, N1, N0.getOperand(1)); return DAG.getNode(ISD::ADD, DL, VT, N0.getOperand(0), Add0); } } if (SDValue NewSel = foldBinOpIntoSelect(N)) return NewSel; // reassociate add if (SDValue RADD = ReassociateOps(ISD::ADD, DL, N0, N1, N->getFlags())) return RADD; // fold ((0-A) + B) -> B-A if (N0.getOpcode() == ISD::SUB && isNullOrNullSplat(N0.getOperand(0))) return DAG.getNode(ISD::SUB, DL, VT, N1, N0.getOperand(1)); // fold (A + (0-B)) -> A-B if (N1.getOpcode() == ISD::SUB && isNullOrNullSplat(N1.getOperand(0))) return DAG.getNode(ISD::SUB, DL, VT, N0, N1.getOperand(1)); // fold (A+(B-A)) -> B if (N1.getOpcode() == ISD::SUB && N0 == N1.getOperand(1)) return N1.getOperand(0); // fold ((B-A)+A) -> B if (N0.getOpcode() == ISD::SUB && N1 == N0.getOperand(1)) return N0.getOperand(0); // fold (A+(B-(A+C))) to (B-C) if (N1.getOpcode() == ISD::SUB && N1.getOperand(1).getOpcode() == ISD::ADD && N0 == N1.getOperand(1).getOperand(0)) return DAG.getNode(ISD::SUB, DL, VT, N1.getOperand(0), N1.getOperand(1).getOperand(1)); // fold (A+(B-(C+A))) to (B-C) if (N1.getOpcode() == ISD::SUB && N1.getOperand(1).getOpcode() == ISD::ADD && N0 == N1.getOperand(1).getOperand(1)) return DAG.getNode(ISD::SUB, DL, VT, N1.getOperand(0), N1.getOperand(1).getOperand(0)); // fold (A+((B-A)+or-C)) to (B+or-C) if ((N1.getOpcode() == ISD::SUB || N1.getOpcode() == ISD::ADD) && N1.getOperand(0).getOpcode() == ISD::SUB && N0 == N1.getOperand(0).getOperand(1)) return DAG.getNode(N1.getOpcode(), DL, VT, N1.getOperand(0).getOperand(0), N1.getOperand(1)); // fold (A-B)+(C-D) to (A+C)-(B+D) when A or C is constant if (N0.getOpcode() == ISD::SUB && N1.getOpcode() == ISD::SUB) { SDValue N00 = N0.getOperand(0); SDValue N01 = N0.getOperand(1); SDValue N10 = N1.getOperand(0); SDValue N11 = N1.getOperand(1); if (isConstantOrConstantVector(N00) || isConstantOrConstantVector(N10)) return DAG.getNode(ISD::SUB, DL, VT, DAG.getNode(ISD::ADD, SDLoc(N0), VT, N00, N10), DAG.getNode(ISD::ADD, SDLoc(N1), VT, N01, N11)); } if (SDValue V = foldAddSubBoolOfMaskedVal(N, DAG)) return V; if (SDValue V = foldAddSubOfSignBit(N, DAG)) return V; if (SimplifyDemandedBits(SDValue(N, 0))) return SDValue(N, 0); // fold (a+b) -> (a|b) iff a and b share no bits. if ((!LegalOperations || TLI.isOperationLegal(ISD::OR, VT)) && DAG.haveNoCommonBitsSet(N0, N1)) return DAG.getNode(ISD::OR, DL, VT, N0, N1); // fold (add (xor a, -1), 1) -> (sub 0, a) if (isBitwiseNot(N0) && isOneOrOneSplat(N1)) return DAG.getNode(ISD::SUB, DL, VT, DAG.getConstant(0, DL, VT), N0.getOperand(0)); if (SDValue Combined = visitADDLike(N0, N1, N)) return Combined; if (SDValue Combined = visitADDLike(N1, N0, N)) return Combined; return SDValue(); } SDValue DAGCombiner::visitADDSAT(SDNode *N) { unsigned Opcode = N->getOpcode(); SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); EVT VT = N0.getValueType(); SDLoc DL(N); // fold vector ops if (VT.isVector()) { // TODO SimplifyVBinOp // fold (add_sat x, 0) -> x, vector edition if (ISD::isBuildVectorAllZeros(N1.getNode())) return N0; if (ISD::isBuildVectorAllZeros(N0.getNode())) return N1; } // fold (add_sat x, undef) -> -1 if (N0.isUndef() || N1.isUndef()) return DAG.getAllOnesConstant(DL, VT); if (DAG.isConstantIntBuildVectorOrConstantInt(N0)) { // canonicalize constant to RHS if (!DAG.isConstantIntBuildVectorOrConstantInt(N1)) return DAG.getNode(Opcode, DL, VT, N1, N0); // fold (add_sat c1, c2) -> c3 return DAG.FoldConstantArithmetic(Opcode, DL, VT, N0.getNode(), N1.getNode()); } // fold (add_sat x, 0) -> x if (isNullConstant(N1)) return N0; // If it cannot overflow, transform into an add. if (Opcode == ISD::UADDSAT) if (DAG.computeOverflowKind(N0, N1) == SelectionDAG::OFK_Never) return DAG.getNode(ISD::ADD, DL, VT, N0, N1); return SDValue(); } static SDValue getAsCarry(const TargetLowering &TLI, SDValue V) { bool Masked = false; // First, peel away TRUNCATE/ZERO_EXTEND/AND nodes due to legalization. while (true) { if (V.getOpcode() == ISD::TRUNCATE || V.getOpcode() == ISD::ZERO_EXTEND) { V = V.getOperand(0); continue; } if (V.getOpcode() == ISD::AND && isOneConstant(V.getOperand(1))) { Masked = true; V = V.getOperand(0); continue; } break; } // If this is not a carry, return. if (V.getResNo() != 1) return SDValue(); if (V.getOpcode() != ISD::ADDCARRY && V.getOpcode() != ISD::SUBCARRY && V.getOpcode() != ISD::UADDO && V.getOpcode() != ISD::USUBO) return SDValue(); // If the result is masked, then no matter what kind of bool it is we can // return. If it isn't, then we need to make sure the bool type is either 0 or // 1 and not other values. if (Masked || TLI.getBooleanContents(V.getValueType()) == TargetLoweringBase::ZeroOrOneBooleanContent) return V; return SDValue(); } SDValue DAGCombiner::visitADDLike(SDValue N0, SDValue N1, SDNode *LocReference) { EVT VT = N0.getValueType(); SDLoc DL(LocReference); // fold (add x, shl(0 - y, n)) -> sub(x, shl(y, n)) if (N1.getOpcode() == ISD::SHL && N1.getOperand(0).getOpcode() == ISD::SUB && isNullOrNullSplat(N1.getOperand(0).getOperand(0))) return DAG.getNode(ISD::SUB, DL, VT, N0, DAG.getNode(ISD::SHL, DL, VT, N1.getOperand(0).getOperand(1), N1.getOperand(1))); if (N1.getOpcode() == ISD::AND) { SDValue AndOp0 = N1.getOperand(0); unsigned NumSignBits = DAG.ComputeNumSignBits(AndOp0); unsigned DestBits = VT.getScalarSizeInBits(); // (add z, (and (sbbl x, x), 1)) -> (sub z, (sbbl x, x)) // and similar xforms where the inner op is either ~0 or 0. if (NumSignBits == DestBits && isOneOrOneSplat(N1->getOperand(1))) return DAG.getNode(ISD::SUB, DL, VT, N0, AndOp0); } // add (sext i1), X -> sub X, (zext i1) if (N0.getOpcode() == ISD::SIGN_EXTEND && N0.getOperand(0).getValueType() == MVT::i1 && !TLI.isOperationLegal(ISD::SIGN_EXTEND, MVT::i1)) { SDValue ZExt = DAG.getNode(ISD::ZERO_EXTEND, DL, VT, N0.getOperand(0)); return DAG.getNode(ISD::SUB, DL, VT, N1, ZExt); } // add X, (sextinreg Y i1) -> sub X, (and Y 1) if (N1.getOpcode() == ISD::SIGN_EXTEND_INREG) { VTSDNode *TN = cast(N1.getOperand(1)); if (TN->getVT() == MVT::i1) { SDValue ZExt = DAG.getNode(ISD::AND, DL, VT, N1.getOperand(0), DAG.getConstant(1, DL, VT)); return DAG.getNode(ISD::SUB, DL, VT, N0, ZExt); } } // (add X, (addcarry Y, 0, Carry)) -> (addcarry X, Y, Carry) if (N1.getOpcode() == ISD::ADDCARRY && isNullConstant(N1.getOperand(1)) && N1.getResNo() == 0) return DAG.getNode(ISD::ADDCARRY, DL, N1->getVTList(), N0, N1.getOperand(0), N1.getOperand(2)); // (add X, Carry) -> (addcarry X, 0, Carry) if (TLI.isOperationLegalOrCustom(ISD::ADDCARRY, VT)) if (SDValue Carry = getAsCarry(TLI, N1)) return DAG.getNode(ISD::ADDCARRY, DL, DAG.getVTList(VT, Carry.getValueType()), N0, DAG.getConstant(0, DL, VT), Carry); return SDValue(); } SDValue DAGCombiner::visitADDC(SDNode *N) { SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); EVT VT = N0.getValueType(); SDLoc DL(N); // If the flag result is dead, turn this into an ADD. if (!N->hasAnyUseOfValue(1)) return CombineTo(N, DAG.getNode(ISD::ADD, DL, VT, N0, N1), DAG.getNode(ISD::CARRY_FALSE, DL, MVT::Glue)); // canonicalize constant to RHS. ConstantSDNode *N0C = dyn_cast(N0); ConstantSDNode *N1C = dyn_cast(N1); if (N0C && !N1C) return DAG.getNode(ISD::ADDC, DL, N->getVTList(), N1, N0); // fold (addc x, 0) -> x + no carry out if (isNullConstant(N1)) return CombineTo(N, N0, DAG.getNode(ISD::CARRY_FALSE, DL, MVT::Glue)); // If it cannot overflow, transform into an add. if (DAG.computeOverflowKind(N0, N1) == SelectionDAG::OFK_Never) return CombineTo(N, DAG.getNode(ISD::ADD, DL, VT, N0, N1), DAG.getNode(ISD::CARRY_FALSE, DL, MVT::Glue)); return SDValue(); } static SDValue flipBoolean(SDValue V, const SDLoc &DL, EVT VT, SelectionDAG &DAG, const TargetLowering &TLI) { SDValue Cst; switch (TLI.getBooleanContents(VT)) { case TargetLowering::ZeroOrOneBooleanContent: case TargetLowering::UndefinedBooleanContent: Cst = DAG.getConstant(1, DL, VT); break; case TargetLowering::ZeroOrNegativeOneBooleanContent: Cst = DAG.getConstant(-1, DL, VT); break; } return DAG.getNode(ISD::XOR, DL, VT, V, Cst); } static bool isBooleanFlip(SDValue V, EVT VT, const TargetLowering &TLI) { if (V.getOpcode() != ISD::XOR) return false; ConstantSDNode *Const = dyn_cast(V.getOperand(1)); if (!Const) return false; switch(TLI.getBooleanContents(VT)) { case TargetLowering::ZeroOrOneBooleanContent: return Const->isOne(); case TargetLowering::ZeroOrNegativeOneBooleanContent: return Const->isAllOnesValue(); case TargetLowering::UndefinedBooleanContent: return (Const->getAPIntValue() & 0x01) == 1; } llvm_unreachable("Unsupported boolean content"); } SDValue DAGCombiner::visitUADDO(SDNode *N) { SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); EVT VT = N0.getValueType(); if (VT.isVector()) return SDValue(); EVT CarryVT = N->getValueType(1); SDLoc DL(N); // If the flag result is dead, turn this into an ADD. if (!N->hasAnyUseOfValue(1)) return CombineTo(N, DAG.getNode(ISD::ADD, DL, VT, N0, N1), DAG.getUNDEF(CarryVT)); // canonicalize constant to RHS. ConstantSDNode *N0C = dyn_cast(N0); ConstantSDNode *N1C = dyn_cast(N1); if (N0C && !N1C) return DAG.getNode(ISD::UADDO, DL, N->getVTList(), N1, N0); // fold (uaddo x, 0) -> x + no carry out if (isNullConstant(N1)) return CombineTo(N, N0, DAG.getConstant(0, DL, CarryVT)); // If it cannot overflow, transform into an add. if (DAG.computeOverflowKind(N0, N1) == SelectionDAG::OFK_Never) return CombineTo(N, DAG.getNode(ISD::ADD, DL, VT, N0, N1), DAG.getConstant(0, DL, CarryVT)); // fold (uaddo (xor a, -1), 1) -> (usub 0, a) and flip carry. if (isBitwiseNot(N0) && isOneOrOneSplat(N1)) { SDValue Sub = DAG.getNode(ISD::USUBO, DL, N->getVTList(), DAG.getConstant(0, DL, VT), N0.getOperand(0)); return CombineTo(N, Sub, flipBoolean(Sub.getValue(1), DL, CarryVT, DAG, TLI)); } if (SDValue Combined = visitUADDOLike(N0, N1, N)) return Combined; if (SDValue Combined = visitUADDOLike(N1, N0, N)) return Combined; return SDValue(); } SDValue DAGCombiner::visitUADDOLike(SDValue N0, SDValue N1, SDNode *N) { auto VT = N0.getValueType(); // (uaddo X, (addcarry Y, 0, Carry)) -> (addcarry X, Y, Carry) // If Y + 1 cannot overflow. if (N1.getOpcode() == ISD::ADDCARRY && isNullConstant(N1.getOperand(1))) { SDValue Y = N1.getOperand(0); SDValue One = DAG.getConstant(1, SDLoc(N), Y.getValueType()); if (DAG.computeOverflowKind(Y, One) == SelectionDAG::OFK_Never) return DAG.getNode(ISD::ADDCARRY, SDLoc(N), N->getVTList(), N0, Y, N1.getOperand(2)); } // (uaddo X, Carry) -> (addcarry X, 0, Carry) if (TLI.isOperationLegalOrCustom(ISD::ADDCARRY, VT)) if (SDValue Carry = getAsCarry(TLI, N1)) return DAG.getNode(ISD::ADDCARRY, SDLoc(N), N->getVTList(), N0, DAG.getConstant(0, SDLoc(N), VT), Carry); return SDValue(); } SDValue DAGCombiner::visitADDE(SDNode *N) { SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); SDValue CarryIn = N->getOperand(2); // canonicalize constant to RHS ConstantSDNode *N0C = dyn_cast(N0); ConstantSDNode *N1C = dyn_cast(N1); if (N0C && !N1C) return DAG.getNode(ISD::ADDE, SDLoc(N), N->getVTList(), N1, N0, CarryIn); // fold (adde x, y, false) -> (addc x, y) if (CarryIn.getOpcode() == ISD::CARRY_FALSE) return DAG.getNode(ISD::ADDC, SDLoc(N), N->getVTList(), N0, N1); return SDValue(); } SDValue DAGCombiner::visitADDCARRY(SDNode *N) { SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); SDValue CarryIn = N->getOperand(2); SDLoc DL(N); // canonicalize constant to RHS ConstantSDNode *N0C = dyn_cast(N0); ConstantSDNode *N1C = dyn_cast(N1); if (N0C && !N1C) return DAG.getNode(ISD::ADDCARRY, DL, N->getVTList(), N1, N0, CarryIn); // fold (addcarry x, y, false) -> (uaddo x, y) if (isNullConstant(CarryIn)) { if (!LegalOperations || TLI.isOperationLegalOrCustom(ISD::UADDO, N->getValueType(0))) return DAG.getNode(ISD::UADDO, DL, N->getVTList(), N0, N1); } EVT CarryVT = CarryIn.getValueType(); // fold (addcarry 0, 0, X) -> (and (ext/trunc X), 1) and no carry. if (isNullConstant(N0) && isNullConstant(N1)) { EVT VT = N0.getValueType(); SDValue CarryExt = DAG.getBoolExtOrTrunc(CarryIn, DL, VT, CarryVT); AddToWorklist(CarryExt.getNode()); return CombineTo(N, DAG.getNode(ISD::AND, DL, VT, CarryExt, DAG.getConstant(1, DL, VT)), DAG.getConstant(0, DL, CarryVT)); } // fold (addcarry (xor a, -1), 0, !b) -> (subcarry 0, a, b) and flip carry. if (isBitwiseNot(N0) && isNullConstant(N1) && isBooleanFlip(CarryIn, CarryVT, TLI)) { SDValue Sub = DAG.getNode(ISD::SUBCARRY, DL, N->getVTList(), DAG.getConstant(0, DL, N0.getValueType()), N0.getOperand(0), CarryIn.getOperand(0)); return CombineTo(N, Sub, flipBoolean(Sub.getValue(1), DL, CarryVT, DAG, TLI)); } if (SDValue Combined = visitADDCARRYLike(N0, N1, CarryIn, N)) return Combined; if (SDValue Combined = visitADDCARRYLike(N1, N0, CarryIn, N)) return Combined; return SDValue(); } SDValue DAGCombiner::visitADDCARRYLike(SDValue N0, SDValue N1, SDValue CarryIn, SDNode *N) { // Iff the flag result is dead: // (addcarry (add|uaddo X, Y), 0, Carry) -> (addcarry X, Y, Carry) if ((N0.getOpcode() == ISD::ADD || (N0.getOpcode() == ISD::UADDO && N0.getResNo() == 0)) && isNullConstant(N1) && !N->hasAnyUseOfValue(1)) return DAG.getNode(ISD::ADDCARRY, SDLoc(N), N->getVTList(), N0.getOperand(0), N0.getOperand(1), CarryIn); /** * When one of the addcarry argument is itself a carry, we may be facing * a diamond carry propagation. In which case we try to transform the DAG * to ensure linear carry propagation if that is possible. * * We are trying to get: * (addcarry X, 0, (addcarry A, B, Z):Carry) */ if (auto Y = getAsCarry(TLI, N1)) { /** * (uaddo A, B) * / \ * Carry Sum * | \ * | (addcarry *, 0, Z) * | / * \ Carry * | / * (addcarry X, *, *) */ if (Y.getOpcode() == ISD::UADDO && CarryIn.getResNo() == 1 && CarryIn.getOpcode() == ISD::ADDCARRY && isNullConstant(CarryIn.getOperand(1)) && CarryIn.getOperand(0) == Y.getValue(0)) { auto NewY = DAG.getNode(ISD::ADDCARRY, SDLoc(N), Y->getVTList(), Y.getOperand(0), Y.getOperand(1), CarryIn.getOperand(2)); AddToWorklist(NewY.getNode()); return DAG.getNode(ISD::ADDCARRY, SDLoc(N), N->getVTList(), N0, DAG.getConstant(0, SDLoc(N), N0.getValueType()), NewY.getValue(1)); } } return SDValue(); } // Since it may not be valid to emit a fold to zero for vector initializers // check if we can before folding. static SDValue tryFoldToZero(const SDLoc &DL, const TargetLowering &TLI, EVT VT, SelectionDAG &DAG, bool LegalOperations) { if (!VT.isVector()) return DAG.getConstant(0, DL, VT); if (!LegalOperations || TLI.isOperationLegal(ISD::BUILD_VECTOR, VT)) return DAG.getConstant(0, DL, VT); return SDValue(); } SDValue DAGCombiner::visitSUB(SDNode *N) { SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); EVT VT = N0.getValueType(); SDLoc DL(N); // fold vector ops if (VT.isVector()) { if (SDValue FoldedVOp = SimplifyVBinOp(N)) return FoldedVOp; // fold (sub x, 0) -> x, vector edition if (ISD::isBuildVectorAllZeros(N1.getNode())) return N0; } // fold (sub x, x) -> 0 // FIXME: Refactor this and xor and other similar operations together. if (N0 == N1) return tryFoldToZero(DL, TLI, VT, DAG, LegalOperations); if (DAG.isConstantIntBuildVectorOrConstantInt(N0) && DAG.isConstantIntBuildVectorOrConstantInt(N1)) { // fold (sub c1, c2) -> c1-c2 return DAG.FoldConstantArithmetic(ISD::SUB, DL, VT, N0.getNode(), N1.getNode()); } if (SDValue NewSel = foldBinOpIntoSelect(N)) return NewSel; ConstantSDNode *N1C = getAsNonOpaqueConstant(N1); // fold (sub x, c) -> (add x, -c) if (N1C) { return DAG.getNode(ISD::ADD, DL, VT, N0, DAG.getConstant(-N1C->getAPIntValue(), DL, VT)); } if (isNullOrNullSplat(N0)) { unsigned BitWidth = VT.getScalarSizeInBits(); // Right-shifting everything out but the sign bit followed by negation is // the same as flipping arithmetic/logical shift type without the negation: // -(X >>u 31) -> (X >>s 31) // -(X >>s 31) -> (X >>u 31) if (N1->getOpcode() == ISD::SRA || N1->getOpcode() == ISD::SRL) { ConstantSDNode *ShiftAmt = isConstOrConstSplat(N1.getOperand(1)); if (ShiftAmt && ShiftAmt->getZExtValue() == BitWidth - 1) { auto NewSh = N1->getOpcode() == ISD::SRA ? ISD::SRL : ISD::SRA; if (!LegalOperations || TLI.isOperationLegal(NewSh, VT)) return DAG.getNode(NewSh, DL, VT, N1.getOperand(0), N1.getOperand(1)); } } // 0 - X --> 0 if the sub is NUW. if (N->getFlags().hasNoUnsignedWrap()) return N0; if (DAG.MaskedValueIsZero(N1, ~APInt::getSignMask(BitWidth))) { // N1 is either 0 or the minimum signed value. If the sub is NSW, then // N1 must be 0 because negating the minimum signed value is undefined. if (N->getFlags().hasNoSignedWrap()) return N0; // 0 - X --> X if X is 0 or the minimum signed value. return N1; } } // Canonicalize (sub -1, x) -> ~x, i.e. (xor x, -1) if (isAllOnesOrAllOnesSplat(N0)) return DAG.getNode(ISD::XOR, DL, VT, N1, N0); // fold (A - (0-B)) -> A+B if (N1.getOpcode() == ISD::SUB && isNullOrNullSplat(N1.getOperand(0))) return DAG.getNode(ISD::ADD, DL, VT, N0, N1.getOperand(1)); // fold A-(A-B) -> B if (N1.getOpcode() == ISD::SUB && N0 == N1.getOperand(0)) return N1.getOperand(1); // fold (A+B)-A -> B if (N0.getOpcode() == ISD::ADD && N0.getOperand(0) == N1) return N0.getOperand(1); // fold (A+B)-B -> A if (N0.getOpcode() == ISD::ADD && N0.getOperand(1) == N1) return N0.getOperand(0); // fold C2-(A+C1) -> (C2-C1)-A if (N1.getOpcode() == ISD::ADD) { SDValue N11 = N1.getOperand(1); if (isConstantOrConstantVector(N0, /* NoOpaques */ true) && isConstantOrConstantVector(N11, /* NoOpaques */ true)) { SDValue NewC = DAG.getNode(ISD::SUB, DL, VT, N0, N11); return DAG.getNode(ISD::SUB, DL, VT, NewC, N1.getOperand(0)); } } // fold ((A+(B+or-C))-B) -> A+or-C if (N0.getOpcode() == ISD::ADD && (N0.getOperand(1).getOpcode() == ISD::SUB || N0.getOperand(1).getOpcode() == ISD::ADD) && N0.getOperand(1).getOperand(0) == N1) return DAG.getNode(N0.getOperand(1).getOpcode(), DL, VT, N0.getOperand(0), N0.getOperand(1).getOperand(1)); // fold ((A+(C+B))-B) -> A+C if (N0.getOpcode() == ISD::ADD && N0.getOperand(1).getOpcode() == ISD::ADD && N0.getOperand(1).getOperand(1) == N1) return DAG.getNode(ISD::ADD, DL, VT, N0.getOperand(0), N0.getOperand(1).getOperand(0)); // fold ((A-(B-C))-C) -> A-B if (N0.getOpcode() == ISD::SUB && N0.getOperand(1).getOpcode() == ISD::SUB && N0.getOperand(1).getOperand(1) == N1) return DAG.getNode(ISD::SUB, DL, VT, N0.getOperand(0), N0.getOperand(1).getOperand(0)); // fold (A-(B-C)) -> A+(C-B) if (N1.getOpcode() == ISD::SUB && N1.hasOneUse()) return DAG.getNode(ISD::ADD, DL, VT, N0, DAG.getNode(ISD::SUB, DL, VT, N1.getOperand(1), N1.getOperand(0))); // fold (X - (-Y * Z)) -> (X + (Y * Z)) if (N1.getOpcode() == ISD::MUL && N1.hasOneUse()) { if (N1.getOperand(0).getOpcode() == ISD::SUB && isNullOrNullSplat(N1.getOperand(0).getOperand(0))) { SDValue Mul = DAG.getNode(ISD::MUL, DL, VT, N1.getOperand(0).getOperand(1), N1.getOperand(1)); return DAG.getNode(ISD::ADD, DL, VT, N0, Mul); } if (N1.getOperand(1).getOpcode() == ISD::SUB && isNullOrNullSplat(N1.getOperand(1).getOperand(0))) { SDValue Mul = DAG.getNode(ISD::MUL, DL, VT, N1.getOperand(0), N1.getOperand(1).getOperand(1)); return DAG.getNode(ISD::ADD, DL, VT, N0, Mul); } } // If either operand of a sub is undef, the result is undef if (N0.isUndef()) return N0; if (N1.isUndef()) return N1; if (SDValue V = foldAddSubBoolOfMaskedVal(N, DAG)) return V; if (SDValue V = foldAddSubOfSignBit(N, DAG)) return V; // fold Y = sra (X, size(X)-1); sub (xor (X, Y), Y) -> (abs X) if (TLI.isOperationLegalOrCustom(ISD::ABS, VT)) { if (N0.getOpcode() == ISD::XOR && N1.getOpcode() == ISD::SRA) { SDValue X0 = N0.getOperand(0), X1 = N0.getOperand(1); SDValue S0 = N1.getOperand(0); if ((X0 == S0 && X1 == N1) || (X0 == N1 && X1 == S0)) { unsigned OpSizeInBits = VT.getScalarSizeInBits(); if (ConstantSDNode *C = isConstOrConstSplat(N1.getOperand(1))) if (C->getAPIntValue() == (OpSizeInBits - 1)) return DAG.getNode(ISD::ABS, SDLoc(N), VT, S0); } } } // If the relocation model supports it, consider symbol offsets. if (GlobalAddressSDNode *GA = dyn_cast(N0)) if (!LegalOperations && TLI.isOffsetFoldingLegal(GA)) { // fold (sub Sym, c) -> Sym-c if (N1C && GA->getOpcode() == ISD::GlobalAddress) return DAG.getGlobalAddress(GA->getGlobal(), SDLoc(N1C), VT, GA->getOffset() - (uint64_t)N1C->getSExtValue()); // fold (sub Sym+c1, Sym+c2) -> c1-c2 if (GlobalAddressSDNode *GB = dyn_cast(N1)) if (GA->getGlobal() == GB->getGlobal()) return DAG.getConstant((uint64_t)GA->getOffset() - GB->getOffset(), DL, VT); } // sub X, (sextinreg Y i1) -> add X, (and Y 1) if (N1.getOpcode() == ISD::SIGN_EXTEND_INREG) { VTSDNode *TN = cast(N1.getOperand(1)); if (TN->getVT() == MVT::i1) { SDValue ZExt = DAG.getNode(ISD::AND, DL, VT, N1.getOperand(0), DAG.getConstant(1, DL, VT)); return DAG.getNode(ISD::ADD, DL, VT, N0, ZExt); } } // Prefer an add for more folding potential and possibly better codegen: // sub N0, (lshr N10, width-1) --> add N0, (ashr N10, width-1) if (!LegalOperations && N1.getOpcode() == ISD::SRL && N1.hasOneUse()) { SDValue ShAmt = N1.getOperand(1); ConstantSDNode *ShAmtC = isConstOrConstSplat(ShAmt); if (ShAmtC && ShAmtC->getZExtValue() == N1.getScalarValueSizeInBits() - 1) { SDValue SRA = DAG.getNode(ISD::SRA, DL, VT, N1.getOperand(0), ShAmt); return DAG.getNode(ISD::ADD, DL, VT, N0, SRA); } } return SDValue(); } SDValue DAGCombiner::visitSUBSAT(SDNode *N) { SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); EVT VT = N0.getValueType(); SDLoc DL(N); // fold vector ops if (VT.isVector()) { // TODO SimplifyVBinOp // fold (sub_sat x, 0) -> x, vector edition if (ISD::isBuildVectorAllZeros(N1.getNode())) return N0; } // fold (sub_sat x, undef) -> 0 if (N0.isUndef() || N1.isUndef()) return DAG.getConstant(0, DL, VT); // fold (sub_sat x, x) -> 0 if (N0 == N1) return DAG.getConstant(0, DL, VT); if (DAG.isConstantIntBuildVectorOrConstantInt(N0) && DAG.isConstantIntBuildVectorOrConstantInt(N1)) { // fold (sub_sat c1, c2) -> c3 return DAG.FoldConstantArithmetic(N->getOpcode(), DL, VT, N0.getNode(), N1.getNode()); } // fold (sub_sat x, 0) -> x if (isNullConstant(N1)) return N0; return SDValue(); } SDValue DAGCombiner::visitSUBC(SDNode *N) { SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); EVT VT = N0.getValueType(); SDLoc DL(N); // If the flag result is dead, turn this into an SUB. if (!N->hasAnyUseOfValue(1)) return CombineTo(N, DAG.getNode(ISD::SUB, DL, VT, N0, N1), DAG.getNode(ISD::CARRY_FALSE, DL, MVT::Glue)); // fold (subc x, x) -> 0 + no borrow if (N0 == N1) return CombineTo(N, DAG.getConstant(0, DL, VT), DAG.getNode(ISD::CARRY_FALSE, DL, MVT::Glue)); // fold (subc x, 0) -> x + no borrow if (isNullConstant(N1)) return CombineTo(N, N0, DAG.getNode(ISD::CARRY_FALSE, DL, MVT::Glue)); // Canonicalize (sub -1, x) -> ~x, i.e. (xor x, -1) + no borrow if (isAllOnesConstant(N0)) return CombineTo(N, DAG.getNode(ISD::XOR, DL, VT, N1, N0), DAG.getNode(ISD::CARRY_FALSE, DL, MVT::Glue)); return SDValue(); } SDValue DAGCombiner::visitUSUBO(SDNode *N) { SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); EVT VT = N0.getValueType(); if (VT.isVector()) return SDValue(); EVT CarryVT = N->getValueType(1); SDLoc DL(N); // If the flag result is dead, turn this into an SUB. if (!N->hasAnyUseOfValue(1)) return CombineTo(N, DAG.getNode(ISD::SUB, DL, VT, N0, N1), DAG.getUNDEF(CarryVT)); // fold (usubo x, x) -> 0 + no borrow if (N0 == N1) return CombineTo(N, DAG.getConstant(0, DL, VT), DAG.getConstant(0, DL, CarryVT)); // fold (usubo x, 0) -> x + no borrow if (isNullConstant(N1)) return CombineTo(N, N0, DAG.getConstant(0, DL, CarryVT)); // Canonicalize (usubo -1, x) -> ~x, i.e. (xor x, -1) + no borrow if (isAllOnesConstant(N0)) return CombineTo(N, DAG.getNode(ISD::XOR, DL, VT, N1, N0), DAG.getConstant(0, DL, CarryVT)); return SDValue(); } SDValue DAGCombiner::visitSUBE(SDNode *N) { SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); SDValue CarryIn = N->getOperand(2); // fold (sube x, y, false) -> (subc x, y) if (CarryIn.getOpcode() == ISD::CARRY_FALSE) return DAG.getNode(ISD::SUBC, SDLoc(N), N->getVTList(), N0, N1); return SDValue(); } SDValue DAGCombiner::visitSUBCARRY(SDNode *N) { SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); SDValue CarryIn = N->getOperand(2); // fold (subcarry x, y, false) -> (usubo x, y) if (isNullConstant(CarryIn)) { if (!LegalOperations || TLI.isOperationLegalOrCustom(ISD::USUBO, N->getValueType(0))) return DAG.getNode(ISD::USUBO, SDLoc(N), N->getVTList(), N0, N1); } return SDValue(); } SDValue DAGCombiner::visitMUL(SDNode *N) { SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); EVT VT = N0.getValueType(); // fold (mul x, undef) -> 0 if (N0.isUndef() || N1.isUndef()) return DAG.getConstant(0, SDLoc(N), VT); bool N0IsConst = false; bool N1IsConst = false; bool N1IsOpaqueConst = false; bool N0IsOpaqueConst = false; APInt ConstValue0, ConstValue1; // fold vector ops if (VT.isVector()) { if (SDValue FoldedVOp = SimplifyVBinOp(N)) return FoldedVOp; N0IsConst = ISD::isConstantSplatVector(N0.getNode(), ConstValue0); N1IsConst = ISD::isConstantSplatVector(N1.getNode(), ConstValue1); assert((!N0IsConst || ConstValue0.getBitWidth() == VT.getScalarSizeInBits()) && "Splat APInt should be element width"); assert((!N1IsConst || ConstValue1.getBitWidth() == VT.getScalarSizeInBits()) && "Splat APInt should be element width"); } else { N0IsConst = isa(N0); if (N0IsConst) { ConstValue0 = cast(N0)->getAPIntValue(); N0IsOpaqueConst = cast(N0)->isOpaque(); } N1IsConst = isa(N1); if (N1IsConst) { ConstValue1 = cast(N1)->getAPIntValue(); N1IsOpaqueConst = cast(N1)->isOpaque(); } } // fold (mul c1, c2) -> c1*c2 if (N0IsConst && N1IsConst && !N0IsOpaqueConst && !N1IsOpaqueConst) return DAG.FoldConstantArithmetic(ISD::MUL, SDLoc(N), VT, N0.getNode(), N1.getNode()); // canonicalize constant to RHS (vector doesn't have to splat) if (DAG.isConstantIntBuildVectorOrConstantInt(N0) && !DAG.isConstantIntBuildVectorOrConstantInt(N1)) return DAG.getNode(ISD::MUL, SDLoc(N), VT, N1, N0); // fold (mul x, 0) -> 0 if (N1IsConst && ConstValue1.isNullValue()) return N1; // fold (mul x, 1) -> x if (N1IsConst && ConstValue1.isOneValue()) return N0; if (SDValue NewSel = foldBinOpIntoSelect(N)) return NewSel; // fold (mul x, -1) -> 0-x if (N1IsConst && ConstValue1.isAllOnesValue()) { SDLoc DL(N); return DAG.getNode(ISD::SUB, DL, VT, DAG.getConstant(0, DL, VT), N0); } // fold (mul x, (1 << c)) -> x << c if (isConstantOrConstantVector(N1, /*NoOpaques*/ true) && DAG.isKnownToBeAPowerOfTwo(N1) && (!VT.isVector() || Level <= AfterLegalizeVectorOps)) { SDLoc DL(N); SDValue LogBase2 = BuildLogBase2(N1, DL); EVT ShiftVT = getShiftAmountTy(N0.getValueType()); SDValue Trunc = DAG.getZExtOrTrunc(LogBase2, DL, ShiftVT); return DAG.getNode(ISD::SHL, DL, VT, N0, Trunc); } // fold (mul x, -(1 << c)) -> -(x << c) or (-x) << c if (N1IsConst && !N1IsOpaqueConst && (-ConstValue1).isPowerOf2()) { unsigned Log2Val = (-ConstValue1).logBase2(); SDLoc DL(N); // FIXME: If the input is something that is easily negated (e.g. a // single-use add), we should put the negate there. return DAG.getNode(ISD::SUB, DL, VT, DAG.getConstant(0, DL, VT), DAG.getNode(ISD::SHL, DL, VT, N0, DAG.getConstant(Log2Val, DL, getShiftAmountTy(N0.getValueType())))); } // Try to transform multiply-by-(power-of-2 +/- 1) into shift and add/sub. // mul x, (2^N + 1) --> add (shl x, N), x // mul x, (2^N - 1) --> sub (shl x, N), x // Examples: x * 33 --> (x << 5) + x // x * 15 --> (x << 4) - x // x * -33 --> -((x << 5) + x) // x * -15 --> -((x << 4) - x) ; this reduces --> x - (x << 4) if (N1IsConst && TLI.decomposeMulByConstant(VT, N1)) { // TODO: We could handle more general decomposition of any constant by // having the target set a limit on number of ops and making a // callback to determine that sequence (similar to sqrt expansion). unsigned MathOp = ISD::DELETED_NODE; APInt MulC = ConstValue1.abs(); if ((MulC - 1).isPowerOf2()) MathOp = ISD::ADD; else if ((MulC + 1).isPowerOf2()) MathOp = ISD::SUB; if (MathOp != ISD::DELETED_NODE) { unsigned ShAmt = MathOp == ISD::ADD ? (MulC - 1).logBase2() : (MulC + 1).logBase2(); assert(ShAmt > 0 && ShAmt < VT.getScalarSizeInBits() && "Not expecting multiply-by-constant that could have simplified"); SDLoc DL(N); SDValue Shl = DAG.getNode(ISD::SHL, DL, VT, N0, DAG.getConstant(ShAmt, DL, VT)); SDValue R = DAG.getNode(MathOp, DL, VT, Shl, N0); if (ConstValue1.isNegative()) R = DAG.getNode(ISD::SUB, DL, VT, DAG.getConstant(0, DL, VT), R); return R; } } // (mul (shl X, c1), c2) -> (mul X, c2 << c1) if (N0.getOpcode() == ISD::SHL && isConstantOrConstantVector(N1, /* NoOpaques */ true) && isConstantOrConstantVector(N0.getOperand(1), /* NoOpaques */ true)) { SDValue C3 = DAG.getNode(ISD::SHL, SDLoc(N), VT, N1, N0.getOperand(1)); if (isConstantOrConstantVector(C3)) return DAG.getNode(ISD::MUL, SDLoc(N), VT, N0.getOperand(0), C3); } // Change (mul (shl X, C), Y) -> (shl (mul X, Y), C) when the shift has one // use. { SDValue Sh(nullptr, 0), Y(nullptr, 0); // Check for both (mul (shl X, C), Y) and (mul Y, (shl X, C)). if (N0.getOpcode() == ISD::SHL && isConstantOrConstantVector(N0.getOperand(1)) && N0.getNode()->hasOneUse()) { Sh = N0; Y = N1; } else if (N1.getOpcode() == ISD::SHL && isConstantOrConstantVector(N1.getOperand(1)) && N1.getNode()->hasOneUse()) { Sh = N1; Y = N0; } if (Sh.getNode()) { SDValue Mul = DAG.getNode(ISD::MUL, SDLoc(N), VT, Sh.getOperand(0), Y); return DAG.getNode(ISD::SHL, SDLoc(N), VT, Mul, Sh.getOperand(1)); } } // fold (mul (add x, c1), c2) -> (add (mul x, c2), c1*c2) if (DAG.isConstantIntBuildVectorOrConstantInt(N1) && N0.getOpcode() == ISD::ADD && DAG.isConstantIntBuildVectorOrConstantInt(N0.getOperand(1)) && isMulAddWithConstProfitable(N, N0, N1)) return DAG.getNode(ISD::ADD, SDLoc(N), VT, DAG.getNode(ISD::MUL, SDLoc(N0), VT, N0.getOperand(0), N1), DAG.getNode(ISD::MUL, SDLoc(N1), VT, N0.getOperand(1), N1)); // reassociate mul if (SDValue RMUL = ReassociateOps(ISD::MUL, SDLoc(N), N0, N1, N->getFlags())) return RMUL; return SDValue(); } /// Return true if divmod libcall is available. static bool isDivRemLibcallAvailable(SDNode *Node, bool isSigned, const TargetLowering &TLI) { RTLIB::Libcall LC; EVT NodeType = Node->getValueType(0); if (!NodeType.isSimple()) return false; switch (NodeType.getSimpleVT().SimpleTy) { default: return false; // No libcall for vector types. case MVT::i8: LC= isSigned ? RTLIB::SDIVREM_I8 : RTLIB::UDIVREM_I8; break; case MVT::i16: LC= isSigned ? RTLIB::SDIVREM_I16 : RTLIB::UDIVREM_I16; break; case MVT::i32: LC= isSigned ? RTLIB::SDIVREM_I32 : RTLIB::UDIVREM_I32; break; case MVT::i64: LC= isSigned ? RTLIB::SDIVREM_I64 : RTLIB::UDIVREM_I64; break; case MVT::i128: LC= isSigned ? RTLIB::SDIVREM_I128:RTLIB::UDIVREM_I128; break; } return TLI.getLibcallName(LC) != nullptr; } /// Issue divrem if both quotient and remainder are needed. SDValue DAGCombiner::useDivRem(SDNode *Node) { if (Node->use_empty()) return SDValue(); // This is a dead node, leave it alone. unsigned Opcode = Node->getOpcode(); bool isSigned = (Opcode == ISD::SDIV) || (Opcode == ISD::SREM); unsigned DivRemOpc = isSigned ? ISD::SDIVREM : ISD::UDIVREM; // DivMod lib calls can still work on non-legal types if using lib-calls. EVT VT = Node->getValueType(0); if (VT.isVector() || !VT.isInteger()) return SDValue(); if (!TLI.isTypeLegal(VT) && !TLI.isOperationCustom(DivRemOpc, VT)) return SDValue(); // If DIVREM is going to get expanded into a libcall, // but there is no libcall available, then don't combine. if (!TLI.isOperationLegalOrCustom(DivRemOpc, VT) && !isDivRemLibcallAvailable(Node, isSigned, TLI)) return SDValue(); // If div is legal, it's better to do the normal expansion unsigned OtherOpcode = 0; if ((Opcode == ISD::SDIV) || (Opcode == ISD::UDIV)) { OtherOpcode = isSigned ? ISD::SREM : ISD::UREM; if (TLI.isOperationLegalOrCustom(Opcode, VT)) return SDValue(); } else { OtherOpcode = isSigned ? ISD::SDIV : ISD::UDIV; if (TLI.isOperationLegalOrCustom(OtherOpcode, VT)) return SDValue(); } SDValue Op0 = Node->getOperand(0); SDValue Op1 = Node->getOperand(1); SDValue combined; for (SDNode::use_iterator UI = Op0.getNode()->use_begin(), UE = Op0.getNode()->use_end(); UI != UE; ++UI) { SDNode *User = *UI; if (User == Node || User->getOpcode() == ISD::DELETED_NODE || User->use_empty()) continue; // Convert the other matching node(s), too; // otherwise, the DIVREM may get target-legalized into something // target-specific that we won't be able to recognize. unsigned UserOpc = User->getOpcode(); if ((UserOpc == Opcode || UserOpc == OtherOpcode || UserOpc == DivRemOpc) && User->getOperand(0) == Op0 && User->getOperand(1) == Op1) { if (!combined) { if (UserOpc == OtherOpcode) { SDVTList VTs = DAG.getVTList(VT, VT); combined = DAG.getNode(DivRemOpc, SDLoc(Node), VTs, Op0, Op1); } else if (UserOpc == DivRemOpc) { combined = SDValue(User, 0); } else { assert(UserOpc == Opcode); continue; } } if (UserOpc == ISD::SDIV || UserOpc == ISD::UDIV) CombineTo(User, combined); else if (UserOpc == ISD::SREM || UserOpc == ISD::UREM) CombineTo(User, combined.getValue(1)); } } return combined; } static SDValue simplifyDivRem(SDNode *N, SelectionDAG &DAG) { SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); EVT VT = N->getValueType(0); SDLoc DL(N); unsigned Opc = N->getOpcode(); bool IsDiv = (ISD::SDIV == Opc) || (ISD::UDIV == Opc); ConstantSDNode *N1C = isConstOrConstSplat(N1); // X / undef -> undef // X % undef -> undef // X / 0 -> undef // X % 0 -> undef // NOTE: This includes vectors where any divisor element is zero/undef. if (DAG.isUndef(Opc, {N0, N1})) return DAG.getUNDEF(VT); // undef / X -> 0 // undef % X -> 0 if (N0.isUndef()) return DAG.getConstant(0, DL, VT); // 0 / X -> 0 // 0 % X -> 0 ConstantSDNode *N0C = isConstOrConstSplat(N0); if (N0C && N0C->isNullValue()) return N0; // X / X -> 1 // X % X -> 0 if (N0 == N1) return DAG.getConstant(IsDiv ? 1 : 0, DL, VT); // X / 1 -> X // X % 1 -> 0 // If this is a boolean op (single-bit element type), we can't have // division-by-zero or remainder-by-zero, so assume the divisor is 1. // TODO: Similarly, if we're zero-extending a boolean divisor, then assume // it's a 1. if ((N1C && N1C->isOne()) || (VT.getScalarType() == MVT::i1)) return IsDiv ? N0 : DAG.getConstant(0, DL, VT); return SDValue(); } SDValue DAGCombiner::visitSDIV(SDNode *N) { SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); EVT VT = N->getValueType(0); EVT CCVT = getSetCCResultType(VT); // fold vector ops if (VT.isVector()) if (SDValue FoldedVOp = SimplifyVBinOp(N)) return FoldedVOp; SDLoc DL(N); // fold (sdiv c1, c2) -> c1/c2 ConstantSDNode *N0C = isConstOrConstSplat(N0); ConstantSDNode *N1C = isConstOrConstSplat(N1); if (N0C && N1C && !N0C->isOpaque() && !N1C->isOpaque()) return DAG.FoldConstantArithmetic(ISD::SDIV, DL, VT, N0C, N1C); // fold (sdiv X, -1) -> 0-X if (N1C && N1C->isAllOnesValue()) return DAG.getNode(ISD::SUB, DL, VT, DAG.getConstant(0, DL, VT), N0); // fold (sdiv X, MIN_SIGNED) -> select(X == MIN_SIGNED, 1, 0) if (N1C && N1C->getAPIntValue().isMinSignedValue()) return DAG.getSelect(DL, VT, DAG.getSetCC(DL, CCVT, N0, N1, ISD::SETEQ), DAG.getConstant(1, DL, VT), DAG.getConstant(0, DL, VT)); if (SDValue V = simplifyDivRem(N, DAG)) return V; if (SDValue NewSel = foldBinOpIntoSelect(N)) return NewSel; // If we know the sign bits of both operands are zero, strength reduce to a // udiv instead. Handles (X&15) /s 4 -> X&15 >> 2 if (DAG.SignBitIsZero(N1) && DAG.SignBitIsZero(N0)) return DAG.getNode(ISD::UDIV, DL, N1.getValueType(), N0, N1); if (SDValue V = visitSDIVLike(N0, N1, N)) { // If the corresponding remainder node exists, update its users with // (Dividend - (Quotient * Divisor). if (SDNode *RemNode = DAG.getNodeIfExists(ISD::SREM, N->getVTList(), { N0, N1 })) { SDValue Mul = DAG.getNode(ISD::MUL, DL, VT, V, N1); SDValue Sub = DAG.getNode(ISD::SUB, DL, VT, N0, Mul); AddToWorklist(Mul.getNode()); AddToWorklist(Sub.getNode()); CombineTo(RemNode, Sub); } return V; } // sdiv, srem -> sdivrem // If the divisor is constant, then return DIVREM only if isIntDivCheap() is // true. Otherwise, we break the simplification logic in visitREM(). AttributeList Attr = DAG.getMachineFunction().getFunction().getAttributes(); if (!N1C || TLI.isIntDivCheap(N->getValueType(0), Attr)) if (SDValue DivRem = useDivRem(N)) return DivRem; return SDValue(); } SDValue DAGCombiner::visitSDIVLike(SDValue N0, SDValue N1, SDNode *N) { SDLoc DL(N); EVT VT = N->getValueType(0); EVT CCVT = getSetCCResultType(VT); unsigned BitWidth = VT.getScalarSizeInBits(); // Helper for determining whether a value is a power-2 constant scalar or a // vector of such elements. auto IsPowerOfTwo = [](ConstantSDNode *C) { if (C->isNullValue() || C->isOpaque()) return false; if (C->getAPIntValue().isPowerOf2()) return true; if ((-C->getAPIntValue()).isPowerOf2()) return true; return false; }; // fold (sdiv X, pow2) -> simple ops after legalize // FIXME: We check for the exact bit here because the generic lowering gives // better results in that case. The target-specific lowering should learn how // to handle exact sdivs efficiently. if (!N->getFlags().hasExact() && ISD::matchUnaryPredicate(N1, IsPowerOfTwo)) { // Target-specific implementation of sdiv x, pow2. if (SDValue Res = BuildSDIVPow2(N)) return Res; // Create constants that are functions of the shift amount value. EVT ShiftAmtTy = getShiftAmountTy(N0.getValueType()); SDValue Bits = DAG.getConstant(BitWidth, DL, ShiftAmtTy); SDValue C1 = DAG.getNode(ISD::CTTZ, DL, VT, N1); C1 = DAG.getZExtOrTrunc(C1, DL, ShiftAmtTy); SDValue Inexact = DAG.getNode(ISD::SUB, DL, ShiftAmtTy, Bits, C1); if (!isConstantOrConstantVector(Inexact)) return SDValue(); // Splat the sign bit into the register SDValue Sign = DAG.getNode(ISD::SRA, DL, VT, N0, DAG.getConstant(BitWidth - 1, DL, ShiftAmtTy)); AddToWorklist(Sign.getNode()); // Add (N0 < 0) ? abs2 - 1 : 0; SDValue Srl = DAG.getNode(ISD::SRL, DL, VT, Sign, Inexact); AddToWorklist(Srl.getNode()); SDValue Add = DAG.getNode(ISD::ADD, DL, VT, N0, Srl); AddToWorklist(Add.getNode()); SDValue Sra = DAG.getNode(ISD::SRA, DL, VT, Add, C1); AddToWorklist(Sra.getNode()); // Special case: (sdiv X, 1) -> X // Special Case: (sdiv X, -1) -> 0-X SDValue One = DAG.getConstant(1, DL, VT); SDValue AllOnes = DAG.getAllOnesConstant(DL, VT); SDValue IsOne = DAG.getSetCC(DL, CCVT, N1, One, ISD::SETEQ); SDValue IsAllOnes = DAG.getSetCC(DL, CCVT, N1, AllOnes, ISD::SETEQ); SDValue IsOneOrAllOnes = DAG.getNode(ISD::OR, DL, CCVT, IsOne, IsAllOnes); Sra = DAG.getSelect(DL, VT, IsOneOrAllOnes, N0, Sra); // If dividing by a positive value, we're done. Otherwise, the result must // be negated. SDValue Zero = DAG.getConstant(0, DL, VT); SDValue Sub = DAG.getNode(ISD::SUB, DL, VT, Zero, Sra); // FIXME: Use SELECT_CC once we improve SELECT_CC constant-folding. SDValue IsNeg = DAG.getSetCC(DL, CCVT, N1, Zero, ISD::SETLT); SDValue Res = DAG.getSelect(DL, VT, IsNeg, Sub, Sra); return Res; } // If integer divide is expensive and we satisfy the requirements, emit an // alternate sequence. Targets may check function attributes for size/speed // trade-offs. AttributeList Attr = DAG.getMachineFunction().getFunction().getAttributes(); if (isConstantOrConstantVector(N1) && !TLI.isIntDivCheap(N->getValueType(0), Attr)) if (SDValue Op = BuildSDIV(N)) return Op; return SDValue(); } SDValue DAGCombiner::visitUDIV(SDNode *N) { SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); EVT VT = N->getValueType(0); EVT CCVT = getSetCCResultType(VT); // fold vector ops if (VT.isVector()) if (SDValue FoldedVOp = SimplifyVBinOp(N)) return FoldedVOp; SDLoc DL(N); // fold (udiv c1, c2) -> c1/c2 ConstantSDNode *N0C = isConstOrConstSplat(N0); ConstantSDNode *N1C = isConstOrConstSplat(N1); if (N0C && N1C) if (SDValue Folded = DAG.FoldConstantArithmetic(ISD::UDIV, DL, VT, N0C, N1C)) return Folded; // fold (udiv X, -1) -> select(X == -1, 1, 0) if (N1C && N1C->getAPIntValue().isAllOnesValue()) return DAG.getSelect(DL, VT, DAG.getSetCC(DL, CCVT, N0, N1, ISD::SETEQ), DAG.getConstant(1, DL, VT), DAG.getConstant(0, DL, VT)); if (SDValue V = simplifyDivRem(N, DAG)) return V; if (SDValue NewSel = foldBinOpIntoSelect(N)) return NewSel; if (SDValue V = visitUDIVLike(N0, N1, N)) { // If the corresponding remainder node exists, update its users with // (Dividend - (Quotient * Divisor). if (SDNode *RemNode = DAG.getNodeIfExists(ISD::UREM, N->getVTList(), { N0, N1 })) { SDValue Mul = DAG.getNode(ISD::MUL, DL, VT, V, N1); SDValue Sub = DAG.getNode(ISD::SUB, DL, VT, N0, Mul); AddToWorklist(Mul.getNode()); AddToWorklist(Sub.getNode()); CombineTo(RemNode, Sub); } return V; } // sdiv, srem -> sdivrem // If the divisor is constant, then return DIVREM only if isIntDivCheap() is // true. Otherwise, we break the simplification logic in visitREM(). AttributeList Attr = DAG.getMachineFunction().getFunction().getAttributes(); if (!N1C || TLI.isIntDivCheap(N->getValueType(0), Attr)) if (SDValue DivRem = useDivRem(N)) return DivRem; return SDValue(); } SDValue DAGCombiner::visitUDIVLike(SDValue N0, SDValue N1, SDNode *N) { SDLoc DL(N); EVT VT = N->getValueType(0); // fold (udiv x, (1 << c)) -> x >>u c if (isConstantOrConstantVector(N1, /*NoOpaques*/ true) && DAG.isKnownToBeAPowerOfTwo(N1)) { SDValue LogBase2 = BuildLogBase2(N1, DL); AddToWorklist(LogBase2.getNode()); EVT ShiftVT = getShiftAmountTy(N0.getValueType()); SDValue Trunc = DAG.getZExtOrTrunc(LogBase2, DL, ShiftVT); AddToWorklist(Trunc.getNode()); return DAG.getNode(ISD::SRL, DL, VT, N0, Trunc); } // fold (udiv x, (shl c, y)) -> x >>u (log2(c)+y) iff c is power of 2 if (N1.getOpcode() == ISD::SHL) { SDValue N10 = N1.getOperand(0); if (isConstantOrConstantVector(N10, /*NoOpaques*/ true) && DAG.isKnownToBeAPowerOfTwo(N10)) { SDValue LogBase2 = BuildLogBase2(N10, DL); AddToWorklist(LogBase2.getNode()); EVT ADDVT = N1.getOperand(1).getValueType(); SDValue Trunc = DAG.getZExtOrTrunc(LogBase2, DL, ADDVT); AddToWorklist(Trunc.getNode()); SDValue Add = DAG.getNode(ISD::ADD, DL, ADDVT, N1.getOperand(1), Trunc); AddToWorklist(Add.getNode()); return DAG.getNode(ISD::SRL, DL, VT, N0, Add); } } // fold (udiv x, c) -> alternate AttributeList Attr = DAG.getMachineFunction().getFunction().getAttributes(); if (isConstantOrConstantVector(N1) && !TLI.isIntDivCheap(N->getValueType(0), Attr)) if (SDValue Op = BuildUDIV(N)) return Op; return SDValue(); } // handles ISD::SREM and ISD::UREM SDValue DAGCombiner::visitREM(SDNode *N) { unsigned Opcode = N->getOpcode(); SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); EVT VT = N->getValueType(0); EVT CCVT = getSetCCResultType(VT); bool isSigned = (Opcode == ISD::SREM); SDLoc DL(N); // fold (rem c1, c2) -> c1%c2 ConstantSDNode *N0C = isConstOrConstSplat(N0); ConstantSDNode *N1C = isConstOrConstSplat(N1); if (N0C && N1C) if (SDValue Folded = DAG.FoldConstantArithmetic(Opcode, DL, VT, N0C, N1C)) return Folded; // fold (urem X, -1) -> select(X == -1, 0, x) if (!isSigned && N1C && N1C->getAPIntValue().isAllOnesValue()) return DAG.getSelect(DL, VT, DAG.getSetCC(DL, CCVT, N0, N1, ISD::SETEQ), DAG.getConstant(0, DL, VT), N0); if (SDValue V = simplifyDivRem(N, DAG)) return V; if (SDValue NewSel = foldBinOpIntoSelect(N)) return NewSel; if (isSigned) { // If we know the sign bits of both operands are zero, strength reduce to a // urem instead. Handles (X & 0x0FFFFFFF) %s 16 -> X&15 if (DAG.SignBitIsZero(N1) && DAG.SignBitIsZero(N0)) return DAG.getNode(ISD::UREM, DL, VT, N0, N1); } else { SDValue NegOne = DAG.getAllOnesConstant(DL, VT); if (DAG.isKnownToBeAPowerOfTwo(N1)) { // fold (urem x, pow2) -> (and x, pow2-1) SDValue Add = DAG.getNode(ISD::ADD, DL, VT, N1, NegOne); AddToWorklist(Add.getNode()); return DAG.getNode(ISD::AND, DL, VT, N0, Add); } if (N1.getOpcode() == ISD::SHL && DAG.isKnownToBeAPowerOfTwo(N1.getOperand(0))) { // fold (urem x, (shl pow2, y)) -> (and x, (add (shl pow2, y), -1)) SDValue Add = DAG.getNode(ISD::ADD, DL, VT, N1, NegOne); AddToWorklist(Add.getNode()); return DAG.getNode(ISD::AND, DL, VT, N0, Add); } } AttributeList Attr = DAG.getMachineFunction().getFunction().getAttributes(); // If X/C can be simplified by the division-by-constant logic, lower // X%C to the equivalent of X-X/C*C. // Reuse the SDIVLike/UDIVLike combines - to avoid mangling nodes, the // speculative DIV must not cause a DIVREM conversion. We guard against this // by skipping the simplification if isIntDivCheap(). When div is not cheap, // combine will not return a DIVREM. Regardless, checking cheapness here // makes sense since the simplification results in fatter code. if (DAG.isKnownNeverZero(N1) && !TLI.isIntDivCheap(VT, Attr)) { SDValue OptimizedDiv = isSigned ? visitSDIVLike(N0, N1, N) : visitUDIVLike(N0, N1, N); if (OptimizedDiv.getNode()) { // If the equivalent Div node also exists, update its users. unsigned DivOpcode = isSigned ? ISD::SDIV : ISD::UDIV; if (SDNode *DivNode = DAG.getNodeIfExists(DivOpcode, N->getVTList(), { N0, N1 })) CombineTo(DivNode, OptimizedDiv); SDValue Mul = DAG.getNode(ISD::MUL, DL, VT, OptimizedDiv, N1); SDValue Sub = DAG.getNode(ISD::SUB, DL, VT, N0, Mul); AddToWorklist(OptimizedDiv.getNode()); AddToWorklist(Mul.getNode()); return Sub; } } // sdiv, srem -> sdivrem if (SDValue DivRem = useDivRem(N)) return DivRem.getValue(1); return SDValue(); } SDValue DAGCombiner::visitMULHS(SDNode *N) { SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); EVT VT = N->getValueType(0); SDLoc DL(N); if (VT.isVector()) { // fold (mulhs x, 0) -> 0 if (ISD::isBuildVectorAllZeros(N1.getNode())) return N1; if (ISD::isBuildVectorAllZeros(N0.getNode())) return N0; } // fold (mulhs x, 0) -> 0 if (isNullConstant(N1)) return N1; // fold (mulhs x, 1) -> (sra x, size(x)-1) if (isOneConstant(N1)) return DAG.getNode(ISD::SRA, DL, N0.getValueType(), N0, DAG.getConstant(N0.getValueSizeInBits() - 1, DL, getShiftAmountTy(N0.getValueType()))); // fold (mulhs x, undef) -> 0 if (N0.isUndef() || N1.isUndef()) return DAG.getConstant(0, DL, VT); // If the type twice as wide is legal, transform the mulhs to a wider multiply // plus a shift. if (VT.isSimple() && !VT.isVector()) { MVT Simple = VT.getSimpleVT(); unsigned SimpleSize = Simple.getSizeInBits(); EVT NewVT = EVT::getIntegerVT(*DAG.getContext(), SimpleSize*2); if (TLI.isOperationLegal(ISD::MUL, NewVT)) { N0 = DAG.getNode(ISD::SIGN_EXTEND, DL, NewVT, N0); N1 = DAG.getNode(ISD::SIGN_EXTEND, DL, NewVT, N1); N1 = DAG.getNode(ISD::MUL, DL, NewVT, N0, N1); N1 = DAG.getNode(ISD::SRL, DL, NewVT, N1, DAG.getConstant(SimpleSize, DL, getShiftAmountTy(N1.getValueType()))); return DAG.getNode(ISD::TRUNCATE, DL, VT, N1); } } return SDValue(); } SDValue DAGCombiner::visitMULHU(SDNode *N) { SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); EVT VT = N->getValueType(0); SDLoc DL(N); if (VT.isVector()) { // fold (mulhu x, 0) -> 0 if (ISD::isBuildVectorAllZeros(N1.getNode())) return N1; if (ISD::isBuildVectorAllZeros(N0.getNode())) return N0; } // fold (mulhu x, 0) -> 0 if (isNullConstant(N1)) return N1; // fold (mulhu x, 1) -> 0 if (isOneConstant(N1)) return DAG.getConstant(0, DL, N0.getValueType()); // fold (mulhu x, undef) -> 0 if (N0.isUndef() || N1.isUndef()) return DAG.getConstant(0, DL, VT); // fold (mulhu x, (1 << c)) -> x >> (bitwidth - c) if (isConstantOrConstantVector(N1, /*NoOpaques*/ true) && DAG.isKnownToBeAPowerOfTwo(N1) && hasOperation(ISD::SRL, VT)) { SDLoc DL(N); unsigned NumEltBits = VT.getScalarSizeInBits(); SDValue LogBase2 = BuildLogBase2(N1, DL); SDValue SRLAmt = DAG.getNode( ISD::SUB, DL, VT, DAG.getConstant(NumEltBits, DL, VT), LogBase2); EVT ShiftVT = getShiftAmountTy(N0.getValueType()); SDValue Trunc = DAG.getZExtOrTrunc(SRLAmt, DL, ShiftVT); return DAG.getNode(ISD::SRL, DL, VT, N0, Trunc); } // If the type twice as wide is legal, transform the mulhu to a wider multiply // plus a shift. if (VT.isSimple() && !VT.isVector()) { MVT Simple = VT.getSimpleVT(); unsigned SimpleSize = Simple.getSizeInBits(); EVT NewVT = EVT::getIntegerVT(*DAG.getContext(), SimpleSize*2); if (TLI.isOperationLegal(ISD::MUL, NewVT)) { N0 = DAG.getNode(ISD::ZERO_EXTEND, DL, NewVT, N0); N1 = DAG.getNode(ISD::ZERO_EXTEND, DL, NewVT, N1); N1 = DAG.getNode(ISD::MUL, DL, NewVT, N0, N1); N1 = DAG.getNode(ISD::SRL, DL, NewVT, N1, DAG.getConstant(SimpleSize, DL, getShiftAmountTy(N1.getValueType()))); return DAG.getNode(ISD::TRUNCATE, DL, VT, N1); } } return SDValue(); } /// Perform optimizations common to nodes that compute two values. LoOp and HiOp /// give the opcodes for the two computations that are being performed. Return /// true if a simplification was made. SDValue DAGCombiner::SimplifyNodeWithTwoResults(SDNode *N, unsigned LoOp, unsigned HiOp) { // If the high half is not needed, just compute the low half. bool HiExists = N->hasAnyUseOfValue(1); if (!HiExists && (!LegalOperations || TLI.isOperationLegalOrCustom(LoOp, N->getValueType(0)))) { SDValue Res = DAG.getNode(LoOp, SDLoc(N), N->getValueType(0), N->ops()); return CombineTo(N, Res, Res); } // If the low half is not needed, just compute the high half. bool LoExists = N->hasAnyUseOfValue(0); if (!LoExists && (!LegalOperations || TLI.isOperationLegalOrCustom(HiOp, N->getValueType(1)))) { SDValue Res = DAG.getNode(HiOp, SDLoc(N), N->getValueType(1), N->ops()); return CombineTo(N, Res, Res); } // If both halves are used, return as it is. if (LoExists && HiExists) return SDValue(); // If the two computed results can be simplified separately, separate them. if (LoExists) { SDValue Lo = DAG.getNode(LoOp, SDLoc(N), N->getValueType(0), N->ops()); AddToWorklist(Lo.getNode()); SDValue LoOpt = combine(Lo.getNode()); if (LoOpt.getNode() && LoOpt.getNode() != Lo.getNode() && (!LegalOperations || TLI.isOperationLegalOrCustom(LoOpt.getOpcode(), LoOpt.getValueType()))) return CombineTo(N, LoOpt, LoOpt); } if (HiExists) { SDValue Hi = DAG.getNode(HiOp, SDLoc(N), N->getValueType(1), N->ops()); AddToWorklist(Hi.getNode()); SDValue HiOpt = combine(Hi.getNode()); if (HiOpt.getNode() && HiOpt != Hi && (!LegalOperations || TLI.isOperationLegalOrCustom(HiOpt.getOpcode(), HiOpt.getValueType()))) return CombineTo(N, HiOpt, HiOpt); } return SDValue(); } SDValue DAGCombiner::visitSMUL_LOHI(SDNode *N) { if (SDValue Res = SimplifyNodeWithTwoResults(N, ISD::MUL, ISD::MULHS)) return Res; EVT VT = N->getValueType(0); SDLoc DL(N); // If the type is twice as wide is legal, transform the mulhu to a wider // multiply plus a shift. if (VT.isSimple() && !VT.isVector()) { MVT Simple = VT.getSimpleVT(); unsigned SimpleSize = Simple.getSizeInBits(); EVT NewVT = EVT::getIntegerVT(*DAG.getContext(), SimpleSize*2); if (TLI.isOperationLegal(ISD::MUL, NewVT)) { SDValue Lo = DAG.getNode(ISD::SIGN_EXTEND, DL, NewVT, N->getOperand(0)); SDValue Hi = DAG.getNode(ISD::SIGN_EXTEND, DL, NewVT, N->getOperand(1)); Lo = DAG.getNode(ISD::MUL, DL, NewVT, Lo, Hi); // Compute the high part as N1. Hi = DAG.getNode(ISD::SRL, DL, NewVT, Lo, DAG.getConstant(SimpleSize, DL, getShiftAmountTy(Lo.getValueType()))); Hi = DAG.getNode(ISD::TRUNCATE, DL, VT, Hi); // Compute the low part as N0. Lo = DAG.getNode(ISD::TRUNCATE, DL, VT, Lo); return CombineTo(N, Lo, Hi); } } return SDValue(); } SDValue DAGCombiner::visitUMUL_LOHI(SDNode *N) { if (SDValue Res = SimplifyNodeWithTwoResults(N, ISD::MUL, ISD::MULHU)) return Res; EVT VT = N->getValueType(0); SDLoc DL(N); // If the type is twice as wide is legal, transform the mulhu to a wider // multiply plus a shift. if (VT.isSimple() && !VT.isVector()) { MVT Simple = VT.getSimpleVT(); unsigned SimpleSize = Simple.getSizeInBits(); EVT NewVT = EVT::getIntegerVT(*DAG.getContext(), SimpleSize*2); if (TLI.isOperationLegal(ISD::MUL, NewVT)) { SDValue Lo = DAG.getNode(ISD::ZERO_EXTEND, DL, NewVT, N->getOperand(0)); SDValue Hi = DAG.getNode(ISD::ZERO_EXTEND, DL, NewVT, N->getOperand(1)); Lo = DAG.getNode(ISD::MUL, DL, NewVT, Lo, Hi); // Compute the high part as N1. Hi = DAG.getNode(ISD::SRL, DL, NewVT, Lo, DAG.getConstant(SimpleSize, DL, getShiftAmountTy(Lo.getValueType()))); Hi = DAG.getNode(ISD::TRUNCATE, DL, VT, Hi); // Compute the low part as N0. Lo = DAG.getNode(ISD::TRUNCATE, DL, VT, Lo); return CombineTo(N, Lo, Hi); } } return SDValue(); } SDValue DAGCombiner::visitSMULO(SDNode *N) { // (smulo x, 2) -> (saddo x, x) if (ConstantSDNode *C2 = dyn_cast(N->getOperand(1))) if (C2->getAPIntValue() == 2) return DAG.getNode(ISD::SADDO, SDLoc(N), N->getVTList(), N->getOperand(0), N->getOperand(0)); return SDValue(); } SDValue DAGCombiner::visitUMULO(SDNode *N) { // (umulo x, 2) -> (uaddo x, x) if (ConstantSDNode *C2 = dyn_cast(N->getOperand(1))) if (C2->getAPIntValue() == 2) return DAG.getNode(ISD::UADDO, SDLoc(N), N->getVTList(), N->getOperand(0), N->getOperand(0)); return SDValue(); } SDValue DAGCombiner::visitIMINMAX(SDNode *N) { SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); EVT VT = N0.getValueType(); // fold vector ops if (VT.isVector()) if (SDValue FoldedVOp = SimplifyVBinOp(N)) return FoldedVOp; // fold operation with constant operands. ConstantSDNode *N0C = getAsNonOpaqueConstant(N0); ConstantSDNode *N1C = getAsNonOpaqueConstant(N1); if (N0C && N1C) return DAG.FoldConstantArithmetic(N->getOpcode(), SDLoc(N), VT, N0C, N1C); // canonicalize constant to RHS if (DAG.isConstantIntBuildVectorOrConstantInt(N0) && !DAG.isConstantIntBuildVectorOrConstantInt(N1)) return DAG.getNode(N->getOpcode(), SDLoc(N), VT, N1, N0); // Is sign bits are zero, flip between UMIN/UMAX and SMIN/SMAX. // Only do this if the current op isn't legal and the flipped is. unsigned Opcode = N->getOpcode(); const TargetLowering &TLI = DAG.getTargetLoweringInfo(); if (!TLI.isOperationLegal(Opcode, VT) && (N0.isUndef() || DAG.SignBitIsZero(N0)) && (N1.isUndef() || DAG.SignBitIsZero(N1))) { unsigned AltOpcode; switch (Opcode) { case ISD::SMIN: AltOpcode = ISD::UMIN; break; case ISD::SMAX: AltOpcode = ISD::UMAX; break; case ISD::UMIN: AltOpcode = ISD::SMIN; break; case ISD::UMAX: AltOpcode = ISD::SMAX; break; default: llvm_unreachable("Unknown MINMAX opcode"); } if (TLI.isOperationLegal(AltOpcode, VT)) return DAG.getNode(AltOpcode, SDLoc(N), VT, N0, N1); } return SDValue(); } /// If this is a bitwise logic instruction and both operands have the same /// opcode, try to sink the other opcode after the logic instruction. SDValue DAGCombiner::hoistLogicOpWithSameOpcodeHands(SDNode *N) { SDValue N0 = N->getOperand(0), N1 = N->getOperand(1); EVT VT = N0.getValueType(); unsigned LogicOpcode = N->getOpcode(); unsigned HandOpcode = N0.getOpcode(); assert((LogicOpcode == ISD::AND || LogicOpcode == ISD::OR || LogicOpcode == ISD::XOR) && "Expected logic opcode"); assert(HandOpcode == N1.getOpcode() && "Bad input!"); // Bail early if none of these transforms apply. if (N0.getNumOperands() == 0) return SDValue(); // FIXME: We should check number of uses of the operands to not increase // the instruction count for all transforms. // Handle size-changing casts. SDValue X = N0.getOperand(0); SDValue Y = N1.getOperand(0); EVT XVT = X.getValueType(); SDLoc DL(N); if (HandOpcode == ISD::ANY_EXTEND || HandOpcode == ISD::ZERO_EXTEND || HandOpcode == ISD::SIGN_EXTEND) { // If both operands have other uses, this transform would create extra // instructions without eliminating anything. if (!N0.hasOneUse() && !N1.hasOneUse()) return SDValue(); // We need matching integer source types. if (XVT != Y.getValueType()) return SDValue(); // Don't create an illegal op during or after legalization. Don't ever // create an unsupported vector op. if ((VT.isVector() || LegalOperations) && !TLI.isOperationLegalOrCustom(LogicOpcode, XVT)) return SDValue(); // Avoid infinite looping with PromoteIntBinOp. // TODO: Should we apply desirable/legal constraints to all opcodes? if (HandOpcode == ISD::ANY_EXTEND && LegalTypes && !TLI.isTypeDesirableForOp(LogicOpcode, XVT)) return SDValue(); // logic_op (hand_op X), (hand_op Y) --> hand_op (logic_op X, Y) SDValue Logic = DAG.getNode(LogicOpcode, DL, XVT, X, Y); return DAG.getNode(HandOpcode, DL, VT, Logic); } // logic_op (truncate x), (truncate y) --> truncate (logic_op x, y) if (HandOpcode == ISD::TRUNCATE) { // If both operands have other uses, this transform would create extra // instructions without eliminating anything. if (!N0.hasOneUse() && !N1.hasOneUse()) return SDValue(); // We need matching source types. if (XVT != Y.getValueType()) return SDValue(); // Don't create an illegal op during or after legalization. if (LegalOperations && !TLI.isOperationLegal(LogicOpcode, XVT)) return SDValue(); // Be extra careful sinking truncate. If it's free, there's no benefit in // widening a binop. Also, don't create a logic op on an illegal type. if (TLI.isZExtFree(VT, XVT) && TLI.isTruncateFree(XVT, VT)) return SDValue(); if (!TLI.isTypeLegal(XVT)) return SDValue(); SDValue Logic = DAG.getNode(LogicOpcode, DL, XVT, X, Y); return DAG.getNode(HandOpcode, DL, VT, Logic); } // For binops SHL/SRL/SRA/AND: // logic_op (OP x, z), (OP y, z) --> OP (logic_op x, y), z if ((HandOpcode == ISD::SHL || HandOpcode == ISD::SRL || HandOpcode == ISD::SRA || HandOpcode == ISD::AND) && N0.getOperand(1) == N1.getOperand(1)) { // If either operand has other uses, this transform is not an improvement. if (!N0.hasOneUse() || !N1.hasOneUse()) return SDValue(); SDValue Logic = DAG.getNode(LogicOpcode, DL, XVT, X, Y); return DAG.getNode(HandOpcode, DL, VT, Logic, N0.getOperand(1)); } // Unary ops: logic_op (bswap x), (bswap y) --> bswap (logic_op x, y) if (HandOpcode == ISD::BSWAP) { // If either operand has other uses, this transform is not an improvement. if (!N0.hasOneUse() || !N1.hasOneUse()) return SDValue(); SDValue Logic = DAG.getNode(LogicOpcode, DL, XVT, X, Y); return DAG.getNode(HandOpcode, DL, VT, Logic); } // Simplify xor/and/or (bitcast(A), bitcast(B)) -> bitcast(op (A,B)) // Only perform this optimization up until type legalization, before // LegalizeVectorOprs. LegalizeVectorOprs promotes vector operations by // adding bitcasts. For example (xor v4i32) is promoted to (v2i64), and // we don't want to undo this promotion. // We also handle SCALAR_TO_VECTOR because xor/or/and operations are cheaper // on scalars. if ((HandOpcode == ISD::BITCAST || HandOpcode == ISD::SCALAR_TO_VECTOR) && Level <= AfterLegalizeTypes) { // Input types must be integer and the same. if (XVT.isInteger() && XVT == Y.getValueType()) { SDValue Logic = DAG.getNode(LogicOpcode, DL, XVT, X, Y); return DAG.getNode(HandOpcode, DL, VT, Logic); } } // Xor/and/or are indifferent to the swizzle operation (shuffle of one value). // Simplify xor/and/or (shuff(A), shuff(B)) -> shuff(op (A,B)) // If both shuffles use the same mask, and both shuffle within a single // vector, then it is worthwhile to move the swizzle after the operation. // The type-legalizer generates this pattern when loading illegal // vector types from memory. In many cases this allows additional shuffle // optimizations. // There are other cases where moving the shuffle after the xor/and/or // is profitable even if shuffles don't perform a swizzle. // If both shuffles use the same mask, and both shuffles have the same first // or second operand, then it might still be profitable to move the shuffle // after the xor/and/or operation. if (HandOpcode == ISD::VECTOR_SHUFFLE && Level < AfterLegalizeDAG) { auto *SVN0 = cast(N0); auto *SVN1 = cast(N1); assert(X.getValueType() == Y.getValueType() && "Inputs to shuffles are not the same type"); // Check that both shuffles use the same mask. The masks are known to be of // the same length because the result vector type is the same. // Check also that shuffles have only one use to avoid introducing extra // instructions. if (!SVN0->hasOneUse() || !SVN1->hasOneUse() || !SVN0->getMask().equals(SVN1->getMask())) return SDValue(); // Don't try to fold this node if it requires introducing a // build vector of all zeros that might be illegal at this stage. SDValue ShOp = N0.getOperand(1); if (LogicOpcode == ISD::XOR && !ShOp.isUndef()) ShOp = tryFoldToZero(DL, TLI, VT, DAG, LegalOperations); // (logic_op (shuf (A, C), shuf (B, C))) --> shuf (logic_op (A, B), C) if (N0.getOperand(1) == N1.getOperand(1) && ShOp.getNode()) { SDValue Logic = DAG.getNode(LogicOpcode, DL, VT, N0.getOperand(0), N1.getOperand(0)); return DAG.getVectorShuffle(VT, DL, Logic, ShOp, SVN0->getMask()); } // Don't try to fold this node if it requires introducing a // build vector of all zeros that might be illegal at this stage. ShOp = N0.getOperand(0); if (LogicOpcode == ISD::XOR && !ShOp.isUndef()) ShOp = tryFoldToZero(DL, TLI, VT, DAG, LegalOperations); // (logic_op (shuf (C, A), shuf (C, B))) --> shuf (C, logic_op (A, B)) if (N0.getOperand(0) == N1.getOperand(0) && ShOp.getNode()) { SDValue Logic = DAG.getNode(LogicOpcode, DL, VT, N0.getOperand(1), N1.getOperand(1)); return DAG.getVectorShuffle(VT, DL, ShOp, Logic, SVN0->getMask()); } } return SDValue(); } /// Try to make (and/or setcc (LL, LR), setcc (RL, RR)) more efficient. SDValue DAGCombiner::foldLogicOfSetCCs(bool IsAnd, SDValue N0, SDValue N1, const SDLoc &DL) { SDValue LL, LR, RL, RR, N0CC, N1CC; if (!isSetCCEquivalent(N0, LL, LR, N0CC) || !isSetCCEquivalent(N1, RL, RR, N1CC)) return SDValue(); assert(N0.getValueType() == N1.getValueType() && "Unexpected operand types for bitwise logic op"); assert(LL.getValueType() == LR.getValueType() && RL.getValueType() == RR.getValueType() && "Unexpected operand types for setcc"); // If we're here post-legalization or the logic op type is not i1, the logic // op type must match a setcc result type. Also, all folds require new // operations on the left and right operands, so those types must match. EVT VT = N0.getValueType(); EVT OpVT = LL.getValueType(); if (LegalOperations || VT.getScalarType() != MVT::i1) if (VT != getSetCCResultType(OpVT)) return SDValue(); if (OpVT != RL.getValueType()) return SDValue(); ISD::CondCode CC0 = cast(N0CC)->get(); ISD::CondCode CC1 = cast(N1CC)->get(); bool IsInteger = OpVT.isInteger(); if (LR == RR && CC0 == CC1 && IsInteger) { bool IsZero = isNullOrNullSplat(LR); bool IsNeg1 = isAllOnesOrAllOnesSplat(LR); // All bits clear? bool AndEqZero = IsAnd && CC1 == ISD::SETEQ && IsZero; // All sign bits clear? bool AndGtNeg1 = IsAnd && CC1 == ISD::SETGT && IsNeg1; // Any bits set? bool OrNeZero = !IsAnd && CC1 == ISD::SETNE && IsZero; // Any sign bits set? bool OrLtZero = !IsAnd && CC1 == ISD::SETLT && IsZero; // (and (seteq X, 0), (seteq Y, 0)) --> (seteq (or X, Y), 0) // (and (setgt X, -1), (setgt Y, -1)) --> (setgt (or X, Y), -1) // (or (setne X, 0), (setne Y, 0)) --> (setne (or X, Y), 0) // (or (setlt X, 0), (setlt Y, 0)) --> (setlt (or X, Y), 0) if (AndEqZero || AndGtNeg1 || OrNeZero || OrLtZero) { SDValue Or = DAG.getNode(ISD::OR, SDLoc(N0), OpVT, LL, RL); AddToWorklist(Or.getNode()); return DAG.getSetCC(DL, VT, Or, LR, CC1); } // All bits set? bool AndEqNeg1 = IsAnd && CC1 == ISD::SETEQ && IsNeg1; // All sign bits set? bool AndLtZero = IsAnd && CC1 == ISD::SETLT && IsZero; // Any bits clear? bool OrNeNeg1 = !IsAnd && CC1 == ISD::SETNE && IsNeg1; // Any sign bits clear? bool OrGtNeg1 = !IsAnd && CC1 == ISD::SETGT && IsNeg1; // (and (seteq X, -1), (seteq Y, -1)) --> (seteq (and X, Y), -1) // (and (setlt X, 0), (setlt Y, 0)) --> (setlt (and X, Y), 0) // (or (setne X, -1), (setne Y, -1)) --> (setne (and X, Y), -1) // (or (setgt X, -1), (setgt Y -1)) --> (setgt (and X, Y), -1) if (AndEqNeg1 || AndLtZero || OrNeNeg1 || OrGtNeg1) { SDValue And = DAG.getNode(ISD::AND, SDLoc(N0), OpVT, LL, RL); AddToWorklist(And.getNode()); return DAG.getSetCC(DL, VT, And, LR, CC1); } } // TODO: What is the 'or' equivalent of this fold? // (and (setne X, 0), (setne X, -1)) --> (setuge (add X, 1), 2) if (IsAnd && LL == RL && CC0 == CC1 && OpVT.getScalarSizeInBits() > 1 && IsInteger && CC0 == ISD::SETNE && ((isNullConstant(LR) && isAllOnesConstant(RR)) || (isAllOnesConstant(LR) && isNullConstant(RR)))) { SDValue One = DAG.getConstant(1, DL, OpVT); SDValue Two = DAG.getConstant(2, DL, OpVT); SDValue Add = DAG.getNode(ISD::ADD, SDLoc(N0), OpVT, LL, One); AddToWorklist(Add.getNode()); return DAG.getSetCC(DL, VT, Add, Two, ISD::SETUGE); } // Try more general transforms if the predicates match and the only user of // the compares is the 'and' or 'or'. if (IsInteger && TLI.convertSetCCLogicToBitwiseLogic(OpVT) && CC0 == CC1 && N0.hasOneUse() && N1.hasOneUse()) { // and (seteq A, B), (seteq C, D) --> seteq (or (xor A, B), (xor C, D)), 0 // or (setne A, B), (setne C, D) --> setne (or (xor A, B), (xor C, D)), 0 if ((IsAnd && CC1 == ISD::SETEQ) || (!IsAnd && CC1 == ISD::SETNE)) { SDValue XorL = DAG.getNode(ISD::XOR, SDLoc(N0), OpVT, LL, LR); SDValue XorR = DAG.getNode(ISD::XOR, SDLoc(N1), OpVT, RL, RR); SDValue Or = DAG.getNode(ISD::OR, DL, OpVT, XorL, XorR); SDValue Zero = DAG.getConstant(0, DL, OpVT); return DAG.getSetCC(DL, VT, Or, Zero, CC1); } } // Canonicalize equivalent operands to LL == RL. if (LL == RR && LR == RL) { CC1 = ISD::getSetCCSwappedOperands(CC1); std::swap(RL, RR); } // (and (setcc X, Y, CC0), (setcc X, Y, CC1)) --> (setcc X, Y, NewCC) // (or (setcc X, Y, CC0), (setcc X, Y, CC1)) --> (setcc X, Y, NewCC) if (LL == RL && LR == RR) { ISD::CondCode NewCC = IsAnd ? ISD::getSetCCAndOperation(CC0, CC1, IsInteger) : ISD::getSetCCOrOperation(CC0, CC1, IsInteger); if (NewCC != ISD::SETCC_INVALID && (!LegalOperations || (TLI.isCondCodeLegal(NewCC, LL.getSimpleValueType()) && TLI.isOperationLegal(ISD::SETCC, OpVT)))) return DAG.getSetCC(DL, VT, LL, LR, NewCC); } return SDValue(); } /// This contains all DAGCombine rules which reduce two values combined by /// an And operation to a single value. This makes them reusable in the context /// of visitSELECT(). Rules involving constants are not included as /// visitSELECT() already handles those cases. SDValue DAGCombiner::visitANDLike(SDValue N0, SDValue N1, SDNode *N) { EVT VT = N1.getValueType(); SDLoc DL(N); // fold (and x, undef) -> 0 if (N0.isUndef() || N1.isUndef()) return DAG.getConstant(0, DL, VT); if (SDValue V = foldLogicOfSetCCs(true, N0, N1, DL)) return V; if (N0.getOpcode() == ISD::ADD && N1.getOpcode() == ISD::SRL && VT.getSizeInBits() <= 64) { if (ConstantSDNode *ADDI = dyn_cast(N0.getOperand(1))) { if (ConstantSDNode *SRLI = dyn_cast(N1.getOperand(1))) { // Look for (and (add x, c1), (lshr y, c2)). If C1 wasn't a legal // immediate for an add, but it is legal if its top c2 bits are set, // transform the ADD so the immediate doesn't need to be materialized // in a register. APInt ADDC = ADDI->getAPIntValue(); APInt SRLC = SRLI->getAPIntValue(); if (ADDC.getMinSignedBits() <= 64 && SRLC.ult(VT.getSizeInBits()) && !TLI.isLegalAddImmediate(ADDC.getSExtValue())) { APInt Mask = APInt::getHighBitsSet(VT.getSizeInBits(), SRLC.getZExtValue()); if (DAG.MaskedValueIsZero(N0.getOperand(1), Mask)) { ADDC |= Mask; if (TLI.isLegalAddImmediate(ADDC.getSExtValue())) { SDLoc DL0(N0); SDValue NewAdd = DAG.getNode(ISD::ADD, DL0, VT, N0.getOperand(0), DAG.getConstant(ADDC, DL, VT)); CombineTo(N0.getNode(), NewAdd); // Return N so it doesn't get rechecked! return SDValue(N, 0); } } } } } } // Reduce bit extract of low half of an integer to the narrower type. // (and (srl i64:x, K), KMask) -> // (i64 zero_extend (and (srl (i32 (trunc i64:x)), K)), KMask) if (N0.getOpcode() == ISD::SRL && N0.hasOneUse()) { if (ConstantSDNode *CAnd = dyn_cast(N1)) { if (ConstantSDNode *CShift = dyn_cast(N0.getOperand(1))) { unsigned Size = VT.getSizeInBits(); const APInt &AndMask = CAnd->getAPIntValue(); unsigned ShiftBits = CShift->getZExtValue(); // Bail out, this node will probably disappear anyway. if (ShiftBits == 0) return SDValue(); unsigned MaskBits = AndMask.countTrailingOnes(); EVT HalfVT = EVT::getIntegerVT(*DAG.getContext(), Size / 2); if (AndMask.isMask() && // Required bits must not span the two halves of the integer and // must fit in the half size type. (ShiftBits + MaskBits <= Size / 2) && TLI.isNarrowingProfitable(VT, HalfVT) && TLI.isTypeDesirableForOp(ISD::AND, HalfVT) && TLI.isTypeDesirableForOp(ISD::SRL, HalfVT) && TLI.isTruncateFree(VT, HalfVT) && TLI.isZExtFree(HalfVT, VT)) { // The isNarrowingProfitable is to avoid regressions on PPC and // AArch64 which match a few 64-bit bit insert / bit extract patterns // on downstream users of this. Those patterns could probably be // extended to handle extensions mixed in. SDValue SL(N0); assert(MaskBits <= Size); // Extracting the highest bit of the low half. EVT ShiftVT = TLI.getShiftAmountTy(HalfVT, DAG.getDataLayout()); SDValue Trunc = DAG.getNode(ISD::TRUNCATE, SL, HalfVT, N0.getOperand(0)); SDValue NewMask = DAG.getConstant(AndMask.trunc(Size / 2), SL, HalfVT); SDValue ShiftK = DAG.getConstant(ShiftBits, SL, ShiftVT); SDValue Shift = DAG.getNode(ISD::SRL, SL, HalfVT, Trunc, ShiftK); SDValue And = DAG.getNode(ISD::AND, SL, HalfVT, Shift, NewMask); return DAG.getNode(ISD::ZERO_EXTEND, SL, VT, And); } } } } return SDValue(); } bool DAGCombiner::isAndLoadExtLoad(ConstantSDNode *AndC, LoadSDNode *LoadN, EVT LoadResultTy, EVT &ExtVT) { if (!AndC->getAPIntValue().isMask()) return false; unsigned ActiveBits = AndC->getAPIntValue().countTrailingOnes(); ExtVT = EVT::getIntegerVT(*DAG.getContext(), ActiveBits); EVT LoadedVT = LoadN->getMemoryVT(); if (ExtVT == LoadedVT && (!LegalOperations || TLI.isLoadExtLegal(ISD::ZEXTLOAD, LoadResultTy, ExtVT))) { // ZEXTLOAD will match without needing to change the size of the value being // loaded. return true; } // Do not change the width of a volatile load. if (LoadN->isVolatile()) return false; // Do not generate loads of non-round integer types since these can // be expensive (and would be wrong if the type is not byte sized). if (!LoadedVT.bitsGT(ExtVT) || !ExtVT.isRound()) return false; if (LegalOperations && !TLI.isLoadExtLegal(ISD::ZEXTLOAD, LoadResultTy, ExtVT)) return false; if (!TLI.shouldReduceLoadWidth(LoadN, ISD::ZEXTLOAD, ExtVT)) return false; return true; } bool DAGCombiner::isLegalNarrowLdSt(LSBaseSDNode *LDST, ISD::LoadExtType ExtType, EVT &MemVT, unsigned ShAmt) { if (!LDST) return false; // Only allow byte offsets. if (ShAmt % 8) return false; // Do not generate loads of non-round integer types since these can // be expensive (and would be wrong if the type is not byte sized). if (!MemVT.isRound()) return false; // Don't change the width of a volatile load. if (LDST->isVolatile()) return false; // Verify that we are actually reducing a load width here. if (LDST->getMemoryVT().getSizeInBits() < MemVT.getSizeInBits()) return false; // Ensure that this isn't going to produce an unsupported unaligned access. if (ShAmt && !TLI.allowsMemoryAccess(*DAG.getContext(), DAG.getDataLayout(), MemVT, LDST->getAddressSpace(), ShAmt / 8)) return false; // It's not possible to generate a constant of extended or untyped type. EVT PtrType = LDST->getBasePtr().getValueType(); if (PtrType == MVT::Untyped || PtrType.isExtended()) return false; if (isa(LDST)) { LoadSDNode *Load = cast(LDST); // Don't transform one with multiple uses, this would require adding a new // load. if (!SDValue(Load, 0).hasOneUse()) return false; if (LegalOperations && !TLI.isLoadExtLegal(ExtType, Load->getValueType(0), MemVT)) return false; // For the transform to be legal, the load must produce only two values // (the value loaded and the chain). Don't transform a pre-increment // load, for example, which produces an extra value. Otherwise the // transformation is not equivalent, and the downstream logic to replace // uses gets things wrong. if (Load->getNumValues() > 2) return false; // If the load that we're shrinking is an extload and we're not just // discarding the extension we can't simply shrink the load. Bail. // TODO: It would be possible to merge the extensions in some cases. if (Load->getExtensionType() != ISD::NON_EXTLOAD && Load->getMemoryVT().getSizeInBits() < MemVT.getSizeInBits() + ShAmt) return false; if (!TLI.shouldReduceLoadWidth(Load, ExtType, MemVT)) return false; } else { assert(isa(LDST) && "It is not a Load nor a Store SDNode"); StoreSDNode *Store = cast(LDST); // Can't write outside the original store if (Store->getMemoryVT().getSizeInBits() < MemVT.getSizeInBits() + ShAmt) return false; if (LegalOperations && !TLI.isTruncStoreLegal(Store->getValue().getValueType(), MemVT)) return false; } return true; } bool DAGCombiner::SearchForAndLoads(SDNode *N, SmallVectorImpl &Loads, SmallPtrSetImpl &NodesWithConsts, ConstantSDNode *Mask, SDNode *&NodeToMask) { // Recursively search for the operands, looking for loads which can be // narrowed. for (unsigned i = 0, e = N->getNumOperands(); i < e; ++i) { SDValue Op = N->getOperand(i); if (Op.getValueType().isVector()) return false; // Some constants may need fixing up later if they are too large. if (auto *C = dyn_cast(Op)) { if ((N->getOpcode() == ISD::OR || N->getOpcode() == ISD::XOR) && (Mask->getAPIntValue() & C->getAPIntValue()) != C->getAPIntValue()) NodesWithConsts.insert(N); continue; } if (!Op.hasOneUse()) return false; switch(Op.getOpcode()) { case ISD::LOAD: { auto *Load = cast(Op); EVT ExtVT; if (isAndLoadExtLoad(Mask, Load, Load->getValueType(0), ExtVT) && isLegalNarrowLdSt(Load, ISD::ZEXTLOAD, ExtVT)) { // ZEXTLOAD is already small enough. if (Load->getExtensionType() == ISD::ZEXTLOAD && ExtVT.bitsGE(Load->getMemoryVT())) continue; // Use LE to convert equal sized loads to zext. if (ExtVT.bitsLE(Load->getMemoryVT())) Loads.push_back(Load); continue; } return false; } case ISD::ZERO_EXTEND: case ISD::AssertZext: { unsigned ActiveBits = Mask->getAPIntValue().countTrailingOnes(); EVT ExtVT = EVT::getIntegerVT(*DAG.getContext(), ActiveBits); EVT VT = Op.getOpcode() == ISD::AssertZext ? cast(Op.getOperand(1))->getVT() : Op.getOperand(0).getValueType(); // We can accept extending nodes if the mask is wider or an equal // width to the original type. if (ExtVT.bitsGE(VT)) continue; break; } case ISD::OR: case ISD::XOR: case ISD::AND: if (!SearchForAndLoads(Op.getNode(), Loads, NodesWithConsts, Mask, NodeToMask)) return false; continue; } // Allow one node which will masked along with any loads found. if (NodeToMask) return false; // Also ensure that the node to be masked only produces one data result. NodeToMask = Op.getNode(); if (NodeToMask->getNumValues() > 1) { bool HasValue = false; for (unsigned i = 0, e = NodeToMask->getNumValues(); i < e; ++i) { MVT VT = SDValue(NodeToMask, i).getSimpleValueType(); if (VT != MVT::Glue && VT != MVT::Other) { if (HasValue) { NodeToMask = nullptr; return false; } HasValue = true; } } assert(HasValue && "Node to be masked has no data result?"); } } return true; } bool DAGCombiner::BackwardsPropagateMask(SDNode *N, SelectionDAG &DAG) { auto *Mask = dyn_cast(N->getOperand(1)); if (!Mask) return false; if (!Mask->getAPIntValue().isMask()) return false; // No need to do anything if the and directly uses a load. if (isa(N->getOperand(0))) return false; SmallVector Loads; SmallPtrSet NodesWithConsts; SDNode *FixupNode = nullptr; if (SearchForAndLoads(N, Loads, NodesWithConsts, Mask, FixupNode)) { if (Loads.size() == 0) return false; LLVM_DEBUG(dbgs() << "Backwards propagate AND: "; N->dump()); SDValue MaskOp = N->getOperand(1); // If it exists, fixup the single node we allow in the tree that needs // masking. if (FixupNode) { LLVM_DEBUG(dbgs() << "First, need to fix up: "; FixupNode->dump()); SDValue And = DAG.getNode(ISD::AND, SDLoc(FixupNode), FixupNode->getValueType(0), SDValue(FixupNode, 0), MaskOp); DAG.ReplaceAllUsesOfValueWith(SDValue(FixupNode, 0), And); if (And.getOpcode() == ISD ::AND) DAG.UpdateNodeOperands(And.getNode(), SDValue(FixupNode, 0), MaskOp); } // Narrow any constants that need it. for (auto *LogicN : NodesWithConsts) { SDValue Op0 = LogicN->getOperand(0); SDValue Op1 = LogicN->getOperand(1); if (isa(Op0)) std::swap(Op0, Op1); SDValue And = DAG.getNode(ISD::AND, SDLoc(Op1), Op1.getValueType(), Op1, MaskOp); DAG.UpdateNodeOperands(LogicN, Op0, And); } // Create narrow loads. for (auto *Load : Loads) { LLVM_DEBUG(dbgs() << "Propagate AND back to: "; Load->dump()); SDValue And = DAG.getNode(ISD::AND, SDLoc(Load), Load->getValueType(0), SDValue(Load, 0), MaskOp); DAG.ReplaceAllUsesOfValueWith(SDValue(Load, 0), And); if (And.getOpcode() == ISD ::AND) And = SDValue( DAG.UpdateNodeOperands(And.getNode(), SDValue(Load, 0), MaskOp), 0); SDValue NewLoad = ReduceLoadWidth(And.getNode()); assert(NewLoad && "Shouldn't be masking the load if it can't be narrowed"); CombineTo(Load, NewLoad, NewLoad.getValue(1)); } DAG.ReplaceAllUsesWith(N, N->getOperand(0).getNode()); return true; } return false; } // Unfold // x & (-1 'logical shift' y) // To // (x 'opposite logical shift' y) 'logical shift' y // if it is better for performance. SDValue DAGCombiner::unfoldExtremeBitClearingToShifts(SDNode *N) { assert(N->getOpcode() == ISD::AND); SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); // Do we actually prefer shifts over mask? if (!TLI.preferShiftsToClearExtremeBits(N0)) return SDValue(); // Try to match (-1 '[outer] logical shift' y) unsigned OuterShift; unsigned InnerShift; // The opposite direction to the OuterShift. SDValue Y; // Shift amount. auto matchMask = [&OuterShift, &InnerShift, &Y](SDValue M) -> bool { if (!M.hasOneUse()) return false; OuterShift = M->getOpcode(); if (OuterShift == ISD::SHL) InnerShift = ISD::SRL; else if (OuterShift == ISD::SRL) InnerShift = ISD::SHL; else return false; if (!isAllOnesConstant(M->getOperand(0))) return false; Y = M->getOperand(1); return true; }; SDValue X; if (matchMask(N1)) X = N0; else if (matchMask(N0)) X = N1; else return SDValue(); SDLoc DL(N); EVT VT = N->getValueType(0); // tmp = x 'opposite logical shift' y SDValue T0 = DAG.getNode(InnerShift, DL, VT, X, Y); // ret = tmp 'logical shift' y SDValue T1 = DAG.getNode(OuterShift, DL, VT, T0, Y); return T1; } SDValue DAGCombiner::visitAND(SDNode *N) { SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); EVT VT = N1.getValueType(); // x & x --> x if (N0 == N1) return N0; // fold vector ops if (VT.isVector()) { if (SDValue FoldedVOp = SimplifyVBinOp(N)) return FoldedVOp; // fold (and x, 0) -> 0, vector edition if (ISD::isBuildVectorAllZeros(N0.getNode())) // do not return N0, because undef node may exist in N0 return DAG.getConstant(APInt::getNullValue(N0.getScalarValueSizeInBits()), SDLoc(N), N0.getValueType()); if (ISD::isBuildVectorAllZeros(N1.getNode())) // do not return N1, because undef node may exist in N1 return DAG.getConstant(APInt::getNullValue(N1.getScalarValueSizeInBits()), SDLoc(N), N1.getValueType()); // fold (and x, -1) -> x, vector edition if (ISD::isBuildVectorAllOnes(N0.getNode())) return N1; if (ISD::isBuildVectorAllOnes(N1.getNode())) return N0; } // fold (and c1, c2) -> c1&c2 ConstantSDNode *N0C = getAsNonOpaqueConstant(N0); ConstantSDNode *N1C = isConstOrConstSplat(N1); if (N0C && N1C && !N1C->isOpaque()) return DAG.FoldConstantArithmetic(ISD::AND, SDLoc(N), VT, N0C, N1C); // canonicalize constant to RHS if (DAG.isConstantIntBuildVectorOrConstantInt(N0) && !DAG.isConstantIntBuildVectorOrConstantInt(N1)) return DAG.getNode(ISD::AND, SDLoc(N), VT, N1, N0); // fold (and x, -1) -> x if (isAllOnesConstant(N1)) return N0; // if (and x, c) is known to be zero, return 0 unsigned BitWidth = VT.getScalarSizeInBits(); if (N1C && DAG.MaskedValueIsZero(SDValue(N, 0), APInt::getAllOnesValue(BitWidth))) return DAG.getConstant(0, SDLoc(N), VT); if (SDValue NewSel = foldBinOpIntoSelect(N)) return NewSel; // reassociate and if (SDValue RAND = ReassociateOps(ISD::AND, SDLoc(N), N0, N1, N->getFlags())) return RAND; // Try to convert a constant mask AND into a shuffle clear mask. if (VT.isVector()) if (SDValue Shuffle = XformToShuffleWithZero(N)) return Shuffle; // fold (and (or x, C), D) -> D if (C & D) == D auto MatchSubset = [](ConstantSDNode *LHS, ConstantSDNode *RHS) { return RHS->getAPIntValue().isSubsetOf(LHS->getAPIntValue()); }; if (N0.getOpcode() == ISD::OR && ISD::matchBinaryPredicate(N0.getOperand(1), N1, MatchSubset)) return N1; // fold (and (any_ext V), c) -> (zero_ext V) if 'and' only clears top bits. if (N1C && N0.getOpcode() == ISD::ANY_EXTEND) { SDValue N0Op0 = N0.getOperand(0); APInt Mask = ~N1C->getAPIntValue(); Mask = Mask.trunc(N0Op0.getScalarValueSizeInBits()); if (DAG.MaskedValueIsZero(N0Op0, Mask)) { SDValue Zext = DAG.getNode(ISD::ZERO_EXTEND, SDLoc(N), N0.getValueType(), N0Op0); // Replace uses of the AND with uses of the Zero extend node. CombineTo(N, Zext); // We actually want to replace all uses of the any_extend with the // zero_extend, to avoid duplicating things. This will later cause this // AND to be folded. CombineTo(N0.getNode(), Zext); return SDValue(N, 0); // Return N so it doesn't get rechecked! } } // similarly fold (and (X (load ([non_ext|any_ext|zero_ext] V))), c) -> // (X (load ([non_ext|zero_ext] V))) if 'and' only clears top bits which must // already be zero by virtue of the width of the base type of the load. // // the 'X' node here can either be nothing or an extract_vector_elt to catch // more cases. if ((N0.getOpcode() == ISD::EXTRACT_VECTOR_ELT && N0.getValueSizeInBits() == N0.getOperand(0).getScalarValueSizeInBits() && N0.getOperand(0).getOpcode() == ISD::LOAD && N0.getOperand(0).getResNo() == 0) || (N0.getOpcode() == ISD::LOAD && N0.getResNo() == 0)) { LoadSDNode *Load = cast( (N0.getOpcode() == ISD::LOAD) ? N0 : N0.getOperand(0) ); // Get the constant (if applicable) the zero'th operand is being ANDed with. // This can be a pure constant or a vector splat, in which case we treat the // vector as a scalar and use the splat value. APInt Constant = APInt::getNullValue(1); if (const ConstantSDNode *C = dyn_cast(N1)) { Constant = C->getAPIntValue(); } else if (BuildVectorSDNode *Vector = dyn_cast(N1)) { APInt SplatValue, SplatUndef; unsigned SplatBitSize; bool HasAnyUndefs; bool IsSplat = Vector->isConstantSplat(SplatValue, SplatUndef, SplatBitSize, HasAnyUndefs); if (IsSplat) { // Undef bits can contribute to a possible optimisation if set, so // set them. SplatValue |= SplatUndef; // The splat value may be something like "0x00FFFFFF", which means 0 for // the first vector value and FF for the rest, repeating. We need a mask // that will apply equally to all members of the vector, so AND all the // lanes of the constant together. EVT VT = Vector->getValueType(0); unsigned BitWidth = VT.getScalarSizeInBits(); // If the splat value has been compressed to a bitlength lower // than the size of the vector lane, we need to re-expand it to // the lane size. if (BitWidth > SplatBitSize) for (SplatValue = SplatValue.zextOrTrunc(BitWidth); SplatBitSize < BitWidth; SplatBitSize = SplatBitSize * 2) SplatValue |= SplatValue.shl(SplatBitSize); // Make sure that variable 'Constant' is only set if 'SplatBitSize' is a // multiple of 'BitWidth'. Otherwise, we could propagate a wrong value. if (SplatBitSize % BitWidth == 0) { Constant = APInt::getAllOnesValue(BitWidth); for (unsigned i = 0, n = SplatBitSize/BitWidth; i < n; ++i) Constant &= SplatValue.lshr(i*BitWidth).zextOrTrunc(BitWidth); } } } // If we want to change an EXTLOAD to a ZEXTLOAD, ensure a ZEXTLOAD is // actually legal and isn't going to get expanded, else this is a false // optimisation. bool CanZextLoadProfitably = TLI.isLoadExtLegal(ISD::ZEXTLOAD, Load->getValueType(0), Load->getMemoryVT()); // Resize the constant to the same size as the original memory access before // extension. If it is still the AllOnesValue then this AND is completely // unneeded. Constant = Constant.zextOrTrunc(Load->getMemoryVT().getScalarSizeInBits()); bool B; switch (Load->getExtensionType()) { default: B = false; break; case ISD::EXTLOAD: B = CanZextLoadProfitably; break; case ISD::ZEXTLOAD: case ISD::NON_EXTLOAD: B = true; break; } if (B && Constant.isAllOnesValue()) { // If the load type was an EXTLOAD, convert to ZEXTLOAD in order to // preserve semantics once we get rid of the AND. SDValue NewLoad(Load, 0); // Fold the AND away. NewLoad may get replaced immediately. CombineTo(N, (N0.getNode() == Load) ? NewLoad : N0); if (Load->getExtensionType() == ISD::EXTLOAD) { NewLoad = DAG.getLoad(Load->getAddressingMode(), ISD::ZEXTLOAD, Load->getValueType(0), SDLoc(Load), Load->getChain(), Load->getBasePtr(), Load->getOffset(), Load->getMemoryVT(), Load->getMemOperand()); // Replace uses of the EXTLOAD with the new ZEXTLOAD. if (Load->getNumValues() == 3) { // PRE/POST_INC loads have 3 values. SDValue To[] = { NewLoad.getValue(0), NewLoad.getValue(1), NewLoad.getValue(2) }; CombineTo(Load, To, 3, true); } else { CombineTo(Load, NewLoad.getValue(0), NewLoad.getValue(1)); } } return SDValue(N, 0); // Return N so it doesn't get rechecked! } } // fold (and (load x), 255) -> (zextload x, i8) // fold (and (extload x, i16), 255) -> (zextload x, i8) // fold (and (any_ext (extload x, i16)), 255) -> (zextload x, i8) if (!VT.isVector() && N1C && (N0.getOpcode() == ISD::LOAD || (N0.getOpcode() == ISD::ANY_EXTEND && N0.getOperand(0).getOpcode() == ISD::LOAD))) { if (SDValue Res = ReduceLoadWidth(N)) { LoadSDNode *LN0 = N0->getOpcode() == ISD::ANY_EXTEND ? cast(N0.getOperand(0)) : cast(N0); AddToWorklist(N); DAG.ReplaceAllUsesOfValueWith(SDValue(LN0, 0), Res); return SDValue(N, 0); } } if (Level >= AfterLegalizeTypes) { // Attempt to propagate the AND back up to the leaves which, if they're // loads, can be combined to narrow loads and the AND node can be removed. // Perform after legalization so that extend nodes will already be // combined into the loads. if (BackwardsPropagateMask(N, DAG)) { return SDValue(N, 0); } } if (SDValue Combined = visitANDLike(N0, N1, N)) return Combined; // Simplify: (and (op x...), (op y...)) -> (op (and x, y)) if (N0.getOpcode() == N1.getOpcode()) if (SDValue V = hoistLogicOpWithSameOpcodeHands(N)) return V; // Masking the negated extension of a boolean is just the zero-extended // boolean: // and (sub 0, zext(bool X)), 1 --> zext(bool X) // and (sub 0, sext(bool X)), 1 --> zext(bool X) // // Note: the SimplifyDemandedBits fold below can make an information-losing // transform, and then we have no way to find this better fold. if (N1C && N1C->isOne() && N0.getOpcode() == ISD::SUB) { if (isNullOrNullSplat(N0.getOperand(0))) { SDValue SubRHS = N0.getOperand(1); if (SubRHS.getOpcode() == ISD::ZERO_EXTEND && SubRHS.getOperand(0).getScalarValueSizeInBits() == 1) return SubRHS; if (SubRHS.getOpcode() == ISD::SIGN_EXTEND && SubRHS.getOperand(0).getScalarValueSizeInBits() == 1) return DAG.getNode(ISD::ZERO_EXTEND, SDLoc(N), VT, SubRHS.getOperand(0)); } } // fold (and (sign_extend_inreg x, i16 to i32), 1) -> (and x, 1) // fold (and (sra)) -> (and (srl)) when possible. if (SimplifyDemandedBits(SDValue(N, 0))) return SDValue(N, 0); // fold (zext_inreg (extload x)) -> (zextload x) if (ISD::isEXTLoad(N0.getNode()) && ISD::isUNINDEXEDLoad(N0.getNode())) { LoadSDNode *LN0 = cast(N0); EVT MemVT = LN0->getMemoryVT(); // If we zero all the possible extended bits, then we can turn this into // a zextload if we are running before legalize or the operation is legal. unsigned BitWidth = N1.getScalarValueSizeInBits(); if (DAG.MaskedValueIsZero(N1, APInt::getHighBitsSet(BitWidth, BitWidth - MemVT.getScalarSizeInBits())) && ((!LegalOperations && !LN0->isVolatile()) || TLI.isLoadExtLegal(ISD::ZEXTLOAD, VT, MemVT))) { SDValue ExtLoad = DAG.getExtLoad(ISD::ZEXTLOAD, SDLoc(N0), VT, LN0->getChain(), LN0->getBasePtr(), MemVT, LN0->getMemOperand()); AddToWorklist(N); CombineTo(N0.getNode(), ExtLoad, ExtLoad.getValue(1)); return SDValue(N, 0); // Return N so it doesn't get rechecked! } } // fold (zext_inreg (sextload x)) -> (zextload x) iff load has one use if (ISD::isSEXTLoad(N0.getNode()) && ISD::isUNINDEXEDLoad(N0.getNode()) && N0.hasOneUse()) { LoadSDNode *LN0 = cast(N0); EVT MemVT = LN0->getMemoryVT(); // If we zero all the possible extended bits, then we can turn this into // a zextload if we are running before legalize or the operation is legal. unsigned BitWidth = N1.getScalarValueSizeInBits(); if (DAG.MaskedValueIsZero(N1, APInt::getHighBitsSet(BitWidth, BitWidth - MemVT.getScalarSizeInBits())) && ((!LegalOperations && !LN0->isVolatile()) || TLI.isLoadExtLegal(ISD::ZEXTLOAD, VT, MemVT))) { SDValue ExtLoad = DAG.getExtLoad(ISD::ZEXTLOAD, SDLoc(N0), VT, LN0->getChain(), LN0->getBasePtr(), MemVT, LN0->getMemOperand()); AddToWorklist(N); CombineTo(N0.getNode(), ExtLoad, ExtLoad.getValue(1)); return SDValue(N, 0); // Return N so it doesn't get rechecked! } } // fold (and (or (srl N, 8), (shl N, 8)), 0xffff) -> (srl (bswap N), const) if (N1C && N1C->getAPIntValue() == 0xffff && N0.getOpcode() == ISD::OR) { if (SDValue BSwap = MatchBSwapHWordLow(N0.getNode(), N0.getOperand(0), N0.getOperand(1), false)) return BSwap; } if (SDValue Shifts = unfoldExtremeBitClearingToShifts(N)) return Shifts; return SDValue(); } /// Match (a >> 8) | (a << 8) as (bswap a) >> 16. SDValue DAGCombiner::MatchBSwapHWordLow(SDNode *N, SDValue N0, SDValue N1, bool DemandHighBits) { if (!LegalOperations) return SDValue(); EVT VT = N->getValueType(0); if (VT != MVT::i64 && VT != MVT::i32 && VT != MVT::i16) return SDValue(); if (!TLI.isOperationLegalOrCustom(ISD::BSWAP, VT)) return SDValue(); // Recognize (and (shl a, 8), 0xff00), (and (srl a, 8), 0xff) bool LookPassAnd0 = false; bool LookPassAnd1 = false; if (N0.getOpcode() == ISD::AND && N0.getOperand(0).getOpcode() == ISD::SRL) std::swap(N0, N1); if (N1.getOpcode() == ISD::AND && N1.getOperand(0).getOpcode() == ISD::SHL) std::swap(N0, N1); if (N0.getOpcode() == ISD::AND) { if (!N0.getNode()->hasOneUse()) return SDValue(); ConstantSDNode *N01C = dyn_cast(N0.getOperand(1)); // Also handle 0xffff since the LHS is guaranteed to have zeros there. // This is needed for X86. if (!N01C || (N01C->getZExtValue() != 0xFF00 && N01C->getZExtValue() != 0xFFFF)) return SDValue(); N0 = N0.getOperand(0); LookPassAnd0 = true; } if (N1.getOpcode() == ISD::AND) { if (!N1.getNode()->hasOneUse()) return SDValue(); ConstantSDNode *N11C = dyn_cast(N1.getOperand(1)); if (!N11C || N11C->getZExtValue() != 0xFF) return SDValue(); N1 = N1.getOperand(0); LookPassAnd1 = true; } if (N0.getOpcode() == ISD::SRL && N1.getOpcode() == ISD::SHL) std::swap(N0, N1); if (N0.getOpcode() != ISD::SHL || N1.getOpcode() != ISD::SRL) return SDValue(); if (!N0.getNode()->hasOneUse() || !N1.getNode()->hasOneUse()) return SDValue(); ConstantSDNode *N01C = dyn_cast(N0.getOperand(1)); ConstantSDNode *N11C = dyn_cast(N1.getOperand(1)); if (!N01C || !N11C) return SDValue(); if (N01C->getZExtValue() != 8 || N11C->getZExtValue() != 8) return SDValue(); // Look for (shl (and a, 0xff), 8), (srl (and a, 0xff00), 8) SDValue N00 = N0->getOperand(0); if (!LookPassAnd0 && N00.getOpcode() == ISD::AND) { if (!N00.getNode()->hasOneUse()) return SDValue(); ConstantSDNode *N001C = dyn_cast(N00.getOperand(1)); if (!N001C || N001C->getZExtValue() != 0xFF) return SDValue(); N00 = N00.getOperand(0); LookPassAnd0 = true; } SDValue N10 = N1->getOperand(0); if (!LookPassAnd1 && N10.getOpcode() == ISD::AND) { if (!N10.getNode()->hasOneUse()) return SDValue(); ConstantSDNode *N101C = dyn_cast(N10.getOperand(1)); // Also allow 0xFFFF since the bits will be shifted out. This is needed // for X86. if (!N101C || (N101C->getZExtValue() != 0xFF00 && N101C->getZExtValue() != 0xFFFF)) return SDValue(); N10 = N10.getOperand(0); LookPassAnd1 = true; } if (N00 != N10) return SDValue(); // Make sure everything beyond the low halfword gets set to zero since the SRL // 16 will clear the top bits. unsigned OpSizeInBits = VT.getSizeInBits(); if (DemandHighBits && OpSizeInBits > 16) { // If the left-shift isn't masked out then the only way this is a bswap is // if all bits beyond the low 8 are 0. In that case the entire pattern // reduces to a left shift anyway: leave it for other parts of the combiner. if (!LookPassAnd0) return SDValue(); // However, if the right shift isn't masked out then it might be because // it's not needed. See if we can spot that too. if (!LookPassAnd1 && !DAG.MaskedValueIsZero( N10, APInt::getHighBitsSet(OpSizeInBits, OpSizeInBits - 16))) return SDValue(); } SDValue Res = DAG.getNode(ISD::BSWAP, SDLoc(N), VT, N00); if (OpSizeInBits > 16) { SDLoc DL(N); Res = DAG.getNode(ISD::SRL, DL, VT, Res, DAG.getConstant(OpSizeInBits - 16, DL, getShiftAmountTy(VT))); } return Res; } /// Return true if the specified node is an element that makes up a 32-bit /// packed halfword byteswap. /// ((x & 0x000000ff) << 8) | /// ((x & 0x0000ff00) >> 8) | /// ((x & 0x00ff0000) << 8) | /// ((x & 0xff000000) >> 8) static bool isBSwapHWordElement(SDValue N, MutableArrayRef Parts) { if (!N.getNode()->hasOneUse()) return false; unsigned Opc = N.getOpcode(); if (Opc != ISD::AND && Opc != ISD::SHL && Opc != ISD::SRL) return false; SDValue N0 = N.getOperand(0); unsigned Opc0 = N0.getOpcode(); if (Opc0 != ISD::AND && Opc0 != ISD::SHL && Opc0 != ISD::SRL) return false; ConstantSDNode *N1C = nullptr; // SHL or SRL: look upstream for AND mask operand if (Opc == ISD::AND) N1C = dyn_cast(N.getOperand(1)); else if (Opc0 == ISD::AND) N1C = dyn_cast(N0.getOperand(1)); if (!N1C) return false; unsigned MaskByteOffset; switch (N1C->getZExtValue()) { default: return false; case 0xFF: MaskByteOffset = 0; break; case 0xFF00: MaskByteOffset = 1; break; case 0xFFFF: // In case demanded bits didn't clear the bits that will be shifted out. // This is needed for X86. if (Opc == ISD::SRL || (Opc == ISD::AND && Opc0 == ISD::SHL)) { MaskByteOffset = 1; break; } return false; case 0xFF0000: MaskByteOffset = 2; break; case 0xFF000000: MaskByteOffset = 3; break; } // Look for (x & 0xff) << 8 as well as ((x << 8) & 0xff00). if (Opc == ISD::AND) { if (MaskByteOffset == 0 || MaskByteOffset == 2) { // (x >> 8) & 0xff // (x >> 8) & 0xff0000 if (Opc0 != ISD::SRL) return false; ConstantSDNode *C = dyn_cast(N0.getOperand(1)); if (!C || C->getZExtValue() != 8) return false; } else { // (x << 8) & 0xff00 // (x << 8) & 0xff000000 if (Opc0 != ISD::SHL) return false; ConstantSDNode *C = dyn_cast(N0.getOperand(1)); if (!C || C->getZExtValue() != 8) return false; } } else if (Opc == ISD::SHL) { // (x & 0xff) << 8 // (x & 0xff0000) << 8 if (MaskByteOffset != 0 && MaskByteOffset != 2) return false; ConstantSDNode *C = dyn_cast(N.getOperand(1)); if (!C || C->getZExtValue() != 8) return false; } else { // Opc == ISD::SRL // (x & 0xff00) >> 8 // (x & 0xff000000) >> 8 if (MaskByteOffset != 1 && MaskByteOffset != 3) return false; ConstantSDNode *C = dyn_cast(N.getOperand(1)); if (!C || C->getZExtValue() != 8) return false; } if (Parts[MaskByteOffset]) return false; Parts[MaskByteOffset] = N0.getOperand(0).getNode(); return true; } /// Match a 32-bit packed halfword bswap. That is /// ((x & 0x000000ff) << 8) | /// ((x & 0x0000ff00) >> 8) | /// ((x & 0x00ff0000) << 8) | /// ((x & 0xff000000) >> 8) /// => (rotl (bswap x), 16) SDValue DAGCombiner::MatchBSwapHWord(SDNode *N, SDValue N0, SDValue N1) { if (!LegalOperations) return SDValue(); EVT VT = N->getValueType(0); if (VT != MVT::i32) return SDValue(); if (!TLI.isOperationLegalOrCustom(ISD::BSWAP, VT)) return SDValue(); // Look for either // (or (or (and), (and)), (or (and), (and))) // (or (or (or (and), (and)), (and)), (and)) if (N0.getOpcode() != ISD::OR) return SDValue(); SDValue N00 = N0.getOperand(0); SDValue N01 = N0.getOperand(1); SDNode *Parts[4] = {}; if (N1.getOpcode() == ISD::OR && N00.getNumOperands() == 2 && N01.getNumOperands() == 2) { // (or (or (and), (and)), (or (and), (and))) if (!isBSwapHWordElement(N00, Parts)) return SDValue(); if (!isBSwapHWordElement(N01, Parts)) return SDValue(); SDValue N10 = N1.getOperand(0); if (!isBSwapHWordElement(N10, Parts)) return SDValue(); SDValue N11 = N1.getOperand(1); if (!isBSwapHWordElement(N11, Parts)) return SDValue(); } else { // (or (or (or (and), (and)), (and)), (and)) if (!isBSwapHWordElement(N1, Parts)) return SDValue(); if (!isBSwapHWordElement(N01, Parts)) return SDValue(); if (N00.getOpcode() != ISD::OR) return SDValue(); SDValue N000 = N00.getOperand(0); if (!isBSwapHWordElement(N000, Parts)) return SDValue(); SDValue N001 = N00.getOperand(1); if (!isBSwapHWordElement(N001, Parts)) return SDValue(); } // Make sure the parts are all coming from the same node. if (Parts[0] != Parts[1] || Parts[0] != Parts[2] || Parts[0] != Parts[3]) return SDValue(); SDLoc DL(N); SDValue BSwap = DAG.getNode(ISD::BSWAP, DL, VT, SDValue(Parts[0], 0)); // Result of the bswap should be rotated by 16. If it's not legal, then // do (x << 16) | (x >> 16). SDValue ShAmt = DAG.getConstant(16, DL, getShiftAmountTy(VT)); if (TLI.isOperationLegalOrCustom(ISD::ROTL, VT)) return DAG.getNode(ISD::ROTL, DL, VT, BSwap, ShAmt); if (TLI.isOperationLegalOrCustom(ISD::ROTR, VT)) return DAG.getNode(ISD::ROTR, DL, VT, BSwap, ShAmt); return DAG.getNode(ISD::OR, DL, VT, DAG.getNode(ISD::SHL, DL, VT, BSwap, ShAmt), DAG.getNode(ISD::SRL, DL, VT, BSwap, ShAmt)); } /// This contains all DAGCombine rules which reduce two values combined by /// an Or operation to a single value \see visitANDLike(). SDValue DAGCombiner::visitORLike(SDValue N0, SDValue N1, SDNode *N) { EVT VT = N1.getValueType(); SDLoc DL(N); // fold (or x, undef) -> -1 if (!LegalOperations && (N0.isUndef() || N1.isUndef())) return DAG.getAllOnesConstant(DL, VT); if (SDValue V = foldLogicOfSetCCs(false, N0, N1, DL)) return V; // (or (and X, C1), (and Y, C2)) -> (and (or X, Y), C3) if possible. if (N0.getOpcode() == ISD::AND && N1.getOpcode() == ISD::AND && // Don't increase # computations. (N0.getNode()->hasOneUse() || N1.getNode()->hasOneUse())) { // We can only do this xform if we know that bits from X that are set in C2 // but not in C1 are already zero. Likewise for Y. if (const ConstantSDNode *N0O1C = getAsNonOpaqueConstant(N0.getOperand(1))) { if (const ConstantSDNode *N1O1C = getAsNonOpaqueConstant(N1.getOperand(1))) { // We can only do this xform if we know that bits from X that are set in // C2 but not in C1 are already zero. Likewise for Y. const APInt &LHSMask = N0O1C->getAPIntValue(); const APInt &RHSMask = N1O1C->getAPIntValue(); if (DAG.MaskedValueIsZero(N0.getOperand(0), RHSMask&~LHSMask) && DAG.MaskedValueIsZero(N1.getOperand(0), LHSMask&~RHSMask)) { SDValue X = DAG.getNode(ISD::OR, SDLoc(N0), VT, N0.getOperand(0), N1.getOperand(0)); return DAG.getNode(ISD::AND, DL, VT, X, DAG.getConstant(LHSMask | RHSMask, DL, VT)); } } } } // (or (and X, M), (and X, N)) -> (and X, (or M, N)) if (N0.getOpcode() == ISD::AND && N1.getOpcode() == ISD::AND && N0.getOperand(0) == N1.getOperand(0) && // Don't increase # computations. (N0.getNode()->hasOneUse() || N1.getNode()->hasOneUse())) { SDValue X = DAG.getNode(ISD::OR, SDLoc(N0), VT, N0.getOperand(1), N1.getOperand(1)); return DAG.getNode(ISD::AND, DL, VT, N0.getOperand(0), X); } return SDValue(); } SDValue DAGCombiner::visitOR(SDNode *N) { SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); EVT VT = N1.getValueType(); // x | x --> x if (N0 == N1) return N0; // fold vector ops if (VT.isVector()) { if (SDValue FoldedVOp = SimplifyVBinOp(N)) return FoldedVOp; // fold (or x, 0) -> x, vector edition if (ISD::isBuildVectorAllZeros(N0.getNode())) return N1; if (ISD::isBuildVectorAllZeros(N1.getNode())) return N0; // fold (or x, -1) -> -1, vector edition if (ISD::isBuildVectorAllOnes(N0.getNode())) // do not return N0, because undef node may exist in N0 return DAG.getAllOnesConstant(SDLoc(N), N0.getValueType()); if (ISD::isBuildVectorAllOnes(N1.getNode())) // do not return N1, because undef node may exist in N1 return DAG.getAllOnesConstant(SDLoc(N), N1.getValueType()); // fold (or (shuf A, V_0, MA), (shuf B, V_0, MB)) -> (shuf A, B, Mask) // Do this only if the resulting shuffle is legal. if (isa(N0) && isa(N1) && // Avoid folding a node with illegal type. TLI.isTypeLegal(VT)) { bool ZeroN00 = ISD::isBuildVectorAllZeros(N0.getOperand(0).getNode()); bool ZeroN01 = ISD::isBuildVectorAllZeros(N0.getOperand(1).getNode()); bool ZeroN10 = ISD::isBuildVectorAllZeros(N1.getOperand(0).getNode()); bool ZeroN11 = ISD::isBuildVectorAllZeros(N1.getOperand(1).getNode()); // Ensure both shuffles have a zero input. if ((ZeroN00 != ZeroN01) && (ZeroN10 != ZeroN11)) { assert((!ZeroN00 || !ZeroN01) && "Both inputs zero!"); assert((!ZeroN10 || !ZeroN11) && "Both inputs zero!"); const ShuffleVectorSDNode *SV0 = cast(N0); const ShuffleVectorSDNode *SV1 = cast(N1); bool CanFold = true; int NumElts = VT.getVectorNumElements(); SmallVector Mask(NumElts); for (int i = 0; i != NumElts; ++i) { int M0 = SV0->getMaskElt(i); int M1 = SV1->getMaskElt(i); // Determine if either index is pointing to a zero vector. bool M0Zero = M0 < 0 || (ZeroN00 == (M0 < NumElts)); bool M1Zero = M1 < 0 || (ZeroN10 == (M1 < NumElts)); // If one element is zero and the otherside is undef, keep undef. // This also handles the case that both are undef. if ((M0Zero && M1 < 0) || (M1Zero && M0 < 0)) { Mask[i] = -1; continue; } // Make sure only one of the elements is zero. if (M0Zero == M1Zero) { CanFold = false; break; } assert((M0 >= 0 || M1 >= 0) && "Undef index!"); // We have a zero and non-zero element. If the non-zero came from // SV0 make the index a LHS index. If it came from SV1, make it // a RHS index. We need to mod by NumElts because we don't care // which operand it came from in the original shuffles. Mask[i] = M1Zero ? M0 % NumElts : (M1 % NumElts) + NumElts; } if (CanFold) { SDValue NewLHS = ZeroN00 ? N0.getOperand(1) : N0.getOperand(0); SDValue NewRHS = ZeroN10 ? N1.getOperand(1) : N1.getOperand(0); bool LegalMask = TLI.isShuffleMaskLegal(Mask, VT); if (!LegalMask) { std::swap(NewLHS, NewRHS); ShuffleVectorSDNode::commuteMask(Mask); LegalMask = TLI.isShuffleMaskLegal(Mask, VT); } if (LegalMask) return DAG.getVectorShuffle(VT, SDLoc(N), NewLHS, NewRHS, Mask); } } } } // fold (or c1, c2) -> c1|c2 ConstantSDNode *N0C = getAsNonOpaqueConstant(N0); ConstantSDNode *N1C = dyn_cast(N1); if (N0C && N1C && !N1C->isOpaque()) return DAG.FoldConstantArithmetic(ISD::OR, SDLoc(N), VT, N0C, N1C); // canonicalize constant to RHS if (DAG.isConstantIntBuildVectorOrConstantInt(N0) && !DAG.isConstantIntBuildVectorOrConstantInt(N1)) return DAG.getNode(ISD::OR, SDLoc(N), VT, N1, N0); // fold (or x, 0) -> x if (isNullConstant(N1)) return N0; // fold (or x, -1) -> -1 if (isAllOnesConstant(N1)) return N1; if (SDValue NewSel = foldBinOpIntoSelect(N)) return NewSel; // fold (or x, c) -> c iff (x & ~c) == 0 if (N1C && DAG.MaskedValueIsZero(N0, ~N1C->getAPIntValue())) return N1; if (SDValue Combined = visitORLike(N0, N1, N)) return Combined; // Recognize halfword bswaps as (bswap + rotl 16) or (bswap + shl 16) if (SDValue BSwap = MatchBSwapHWord(N, N0, N1)) return BSwap; if (SDValue BSwap = MatchBSwapHWordLow(N, N0, N1)) return BSwap; // reassociate or if (SDValue ROR = ReassociateOps(ISD::OR, SDLoc(N), N0, N1, N->getFlags())) return ROR; // Canonicalize (or (and X, c1), c2) -> (and (or X, c2), c1|c2) // iff (c1 & c2) != 0 or c1/c2 are undef. auto MatchIntersect = [](ConstantSDNode *C1, ConstantSDNode *C2) { return !C1 || !C2 || C1->getAPIntValue().intersects(C2->getAPIntValue()); }; if (N0.getOpcode() == ISD::AND && N0.getNode()->hasOneUse() && ISD::matchBinaryPredicate(N0.getOperand(1), N1, MatchIntersect, true)) { if (SDValue COR = DAG.FoldConstantArithmetic( ISD::OR, SDLoc(N1), VT, N1.getNode(), N0.getOperand(1).getNode())) { SDValue IOR = DAG.getNode(ISD::OR, SDLoc(N0), VT, N0.getOperand(0), N1); AddToWorklist(IOR.getNode()); return DAG.getNode(ISD::AND, SDLoc(N), VT, COR, IOR); } } // Simplify: (or (op x...), (op y...)) -> (op (or x, y)) if (N0.getOpcode() == N1.getOpcode()) if (SDValue V = hoistLogicOpWithSameOpcodeHands(N)) return V; // See if this is some rotate idiom. if (SDNode *Rot = MatchRotate(N0, N1, SDLoc(N))) return SDValue(Rot, 0); if (SDValue Load = MatchLoadCombine(N)) return Load; // Simplify the operands using demanded-bits information. if (SimplifyDemandedBits(SDValue(N, 0))) return SDValue(N, 0); return SDValue(); } static SDValue stripConstantMask(SelectionDAG &DAG, SDValue Op, SDValue &Mask) { if (Op.getOpcode() == ISD::AND && DAG.isConstantIntBuildVectorOrConstantInt(Op.getOperand(1))) { Mask = Op.getOperand(1); return Op.getOperand(0); } return Op; } /// Match "(X shl/srl V1) & V2" where V2 may not be present. static bool matchRotateHalf(SelectionDAG &DAG, SDValue Op, SDValue &Shift, SDValue &Mask) { Op = stripConstantMask(DAG, Op, Mask); if (Op.getOpcode() == ISD::SRL || Op.getOpcode() == ISD::SHL) { Shift = Op; return true; } return false; } /// Helper function for visitOR to extract the needed side of a rotate idiom /// from a shl/srl/mul/udiv. This is meant to handle cases where /// InstCombine merged some outside op with one of the shifts from /// the rotate pattern. /// \returns An empty \c SDValue if the needed shift couldn't be extracted. /// Otherwise, returns an expansion of \p ExtractFrom based on the following /// patterns: /// /// (or (mul v c0) (shrl (mul v c1) c2)): /// expands (mul v c0) -> (shl (mul v c1) c3) /// /// (or (udiv v c0) (shl (udiv v c1) c2)): /// expands (udiv v c0) -> (shrl (udiv v c1) c3) /// /// (or (shl v c0) (shrl (shl v c1) c2)): /// expands (shl v c0) -> (shl (shl v c1) c3) /// /// (or (shrl v c0) (shl (shrl v c1) c2)): /// expands (shrl v c0) -> (shrl (shrl v c1) c3) /// /// Such that in all cases, c3+c2==bitwidth(op v c1). static SDValue extractShiftForRotate(SelectionDAG &DAG, SDValue OppShift, SDValue ExtractFrom, SDValue &Mask, const SDLoc &DL) { assert(OppShift && ExtractFrom && "Empty SDValue"); assert( (OppShift.getOpcode() == ISD::SHL || OppShift.getOpcode() == ISD::SRL) && "Existing shift must be valid as a rotate half"); ExtractFrom = stripConstantMask(DAG, ExtractFrom, Mask); // Preconditions: // (or (op0 v c0) (shiftl/r (op0 v c1) c2)) // // Find opcode of the needed shift to be extracted from (op0 v c0). unsigned Opcode = ISD::DELETED_NODE; bool IsMulOrDiv = false; // Set Opcode and IsMulOrDiv if the extract opcode matches the needed shift // opcode or its arithmetic (mul or udiv) variant. auto SelectOpcode = [&](unsigned NeededShift, unsigned MulOrDivVariant) { IsMulOrDiv = ExtractFrom.getOpcode() == MulOrDivVariant; if (!IsMulOrDiv && ExtractFrom.getOpcode() != NeededShift) return false; Opcode = NeededShift; return true; }; // op0 must be either the needed shift opcode or the mul/udiv equivalent // that the needed shift can be extracted from. if ((OppShift.getOpcode() != ISD::SRL || !SelectOpcode(ISD::SHL, ISD::MUL)) && (OppShift.getOpcode() != ISD::SHL || !SelectOpcode(ISD::SRL, ISD::UDIV))) return SDValue(); // op0 must be the same opcode on both sides, have the same LHS argument, // and produce the same value type. SDValue OppShiftLHS = OppShift.getOperand(0); EVT ShiftedVT = OppShiftLHS.getValueType(); if (OppShiftLHS.getOpcode() != ExtractFrom.getOpcode() || OppShiftLHS.getOperand(0) != ExtractFrom.getOperand(0) || ShiftedVT != ExtractFrom.getValueType()) return SDValue(); // Amount of the existing shift. ConstantSDNode *OppShiftCst = isConstOrConstSplat(OppShift.getOperand(1)); // Constant mul/udiv/shift amount from the RHS of the shift's LHS op. ConstantSDNode *OppLHSCst = isConstOrConstSplat(OppShiftLHS.getOperand(1)); // Constant mul/udiv/shift amount from the RHS of the ExtractFrom op. ConstantSDNode *ExtractFromCst = isConstOrConstSplat(ExtractFrom.getOperand(1)); // TODO: We should be able to handle non-uniform constant vectors for these values // Check that we have constant values. if (!OppShiftCst || !OppShiftCst->getAPIntValue() || !OppLHSCst || !OppLHSCst->getAPIntValue() || !ExtractFromCst || !ExtractFromCst->getAPIntValue()) return SDValue(); // Compute the shift amount we need to extract to complete the rotate. const unsigned VTWidth = ShiftedVT.getScalarSizeInBits(); if (OppShiftCst->getAPIntValue().ugt(VTWidth)) return SDValue(); APInt NeededShiftAmt = VTWidth - OppShiftCst->getAPIntValue(); // Normalize the bitwidth of the two mul/udiv/shift constant operands. APInt ExtractFromAmt = ExtractFromCst->getAPIntValue(); APInt OppLHSAmt = OppLHSCst->getAPIntValue(); zeroExtendToMatch(ExtractFromAmt, OppLHSAmt); // Now try extract the needed shift from the ExtractFrom op and see if the // result matches up with the existing shift's LHS op. if (IsMulOrDiv) { // Op to extract from is a mul or udiv by a constant. // Check: // c2 / (1 << (bitwidth(op0 v c0) - c1)) == c0 // c2 % (1 << (bitwidth(op0 v c0) - c1)) == 0 const APInt ExtractDiv = APInt::getOneBitSet(ExtractFromAmt.getBitWidth(), NeededShiftAmt.getZExtValue()); APInt ResultAmt; APInt Rem; APInt::udivrem(ExtractFromAmt, ExtractDiv, ResultAmt, Rem); if (Rem != 0 || ResultAmt != OppLHSAmt) return SDValue(); } else { // Op to extract from is a shift by a constant. // Check: // c2 - (bitwidth(op0 v c0) - c1) == c0 if (OppLHSAmt != ExtractFromAmt - NeededShiftAmt.zextOrTrunc( ExtractFromAmt.getBitWidth())) return SDValue(); } // Return the expanded shift op that should allow a rotate to be formed. EVT ShiftVT = OppShift.getOperand(1).getValueType(); EVT ResVT = ExtractFrom.getValueType(); SDValue NewShiftNode = DAG.getConstant(NeededShiftAmt, DL, ShiftVT); return DAG.getNode(Opcode, DL, ResVT, OppShiftLHS, NewShiftNode); } // Return true if we can prove that, whenever Neg and Pos are both in the // range [0, EltSize), Neg == (Pos == 0 ? 0 : EltSize - Pos). This means that // for two opposing shifts shift1 and shift2 and a value X with OpBits bits: // // (or (shift1 X, Neg), (shift2 X, Pos)) // // reduces to a rotate in direction shift2 by Pos or (equivalently) a rotate // in direction shift1 by Neg. The range [0, EltSize) means that we only need // to consider shift amounts with defined behavior. static bool matchRotateSub(SDValue Pos, SDValue Neg, unsigned EltSize, SelectionDAG &DAG) { // If EltSize is a power of 2 then: // // (a) (Pos == 0 ? 0 : EltSize - Pos) == (EltSize - Pos) & (EltSize - 1) // (b) Neg == Neg & (EltSize - 1) whenever Neg is in [0, EltSize). // // So if EltSize is a power of 2 and Neg is (and Neg', EltSize-1), we check // for the stronger condition: // // Neg & (EltSize - 1) == (EltSize - Pos) & (EltSize - 1) [A] // // for all Neg and Pos. Since Neg & (EltSize - 1) == Neg' & (EltSize - 1) // we can just replace Neg with Neg' for the rest of the function. // // In other cases we check for the even stronger condition: // // Neg == EltSize - Pos [B] // // for all Neg and Pos. Note that the (or ...) then invokes undefined // behavior if Pos == 0 (and consequently Neg == EltSize). // // We could actually use [A] whenever EltSize is a power of 2, but the // only extra cases that it would match are those uninteresting ones // where Neg and Pos are never in range at the same time. E.g. for // EltSize == 32, using [A] would allow a Neg of the form (sub 64, Pos) // as well as (sub 32, Pos), but: // // (or (shift1 X, (sub 64, Pos)), (shift2 X, Pos)) // // always invokes undefined behavior for 32-bit X. // // Below, Mask == EltSize - 1 when using [A] and is all-ones otherwise. unsigned MaskLoBits = 0; if (Neg.getOpcode() == ISD::AND && isPowerOf2_64(EltSize)) { if (ConstantSDNode *NegC = isConstOrConstSplat(Neg.getOperand(1))) { KnownBits Known = DAG.computeKnownBits(Neg.getOperand(0)); unsigned Bits = Log2_64(EltSize); if (NegC->getAPIntValue().getActiveBits() <= Bits && ((NegC->getAPIntValue() | Known.Zero).countTrailingOnes() >= Bits)) { Neg = Neg.getOperand(0); MaskLoBits = Bits; } } } // Check whether Neg has the form (sub NegC, NegOp1) for some NegC and NegOp1. if (Neg.getOpcode() != ISD::SUB) return false; ConstantSDNode *NegC = isConstOrConstSplat(Neg.getOperand(0)); if (!NegC) return false; SDValue NegOp1 = Neg.getOperand(1); // On the RHS of [A], if Pos is Pos' & (EltSize - 1), just replace Pos with // Pos'. The truncation is redundant for the purpose of the equality. if (MaskLoBits && Pos.getOpcode() == ISD::AND) { if (ConstantSDNode *PosC = isConstOrConstSplat(Pos.getOperand(1))) { KnownBits Known = DAG.computeKnownBits(Pos.getOperand(0)); if (PosC->getAPIntValue().getActiveBits() <= MaskLoBits && ((PosC->getAPIntValue() | Known.Zero).countTrailingOnes() >= MaskLoBits)) Pos = Pos.getOperand(0); } } // The condition we need is now: // // (NegC - NegOp1) & Mask == (EltSize - Pos) & Mask // // If NegOp1 == Pos then we need: // // EltSize & Mask == NegC & Mask // // (because "x & Mask" is a truncation and distributes through subtraction). APInt Width; if (Pos == NegOp1) Width = NegC->getAPIntValue(); // Check for cases where Pos has the form (add NegOp1, PosC) for some PosC. // Then the condition we want to prove becomes: // // (NegC - NegOp1) & Mask == (EltSize - (NegOp1 + PosC)) & Mask // // which, again because "x & Mask" is a truncation, becomes: // // NegC & Mask == (EltSize - PosC) & Mask // EltSize & Mask == (NegC + PosC) & Mask else if (Pos.getOpcode() == ISD::ADD && Pos.getOperand(0) == NegOp1) { if (ConstantSDNode *PosC = isConstOrConstSplat(Pos.getOperand(1))) Width = PosC->getAPIntValue() + NegC->getAPIntValue(); else return false; } else return false; // Now we just need to check that EltSize & Mask == Width & Mask. if (MaskLoBits) // EltSize & Mask is 0 since Mask is EltSize - 1. return Width.getLoBits(MaskLoBits) == 0; return Width == EltSize; } // A subroutine of MatchRotate used once we have found an OR of two opposite // shifts of Shifted. If Neg == - Pos then the OR reduces // to both (PosOpcode Shifted, Pos) and (NegOpcode Shifted, Neg), with the // former being preferred if supported. InnerPos and InnerNeg are Pos and // Neg with outer conversions stripped away. SDNode *DAGCombiner::MatchRotatePosNeg(SDValue Shifted, SDValue Pos, SDValue Neg, SDValue InnerPos, SDValue InnerNeg, unsigned PosOpcode, unsigned NegOpcode, const SDLoc &DL) { // fold (or (shl x, (*ext y)), // (srl x, (*ext (sub 32, y)))) -> // (rotl x, y) or (rotr x, (sub 32, y)) // // fold (or (shl x, (*ext (sub 32, y))), // (srl x, (*ext y))) -> // (rotr x, y) or (rotl x, (sub 32, y)) EVT VT = Shifted.getValueType(); if (matchRotateSub(InnerPos, InnerNeg, VT.getScalarSizeInBits(), DAG)) { bool HasPos = TLI.isOperationLegalOrCustom(PosOpcode, VT); return DAG.getNode(HasPos ? PosOpcode : NegOpcode, DL, VT, Shifted, HasPos ? Pos : Neg).getNode(); } return nullptr; } // MatchRotate - Handle an 'or' of two operands. If this is one of the many // idioms for rotate, and if the target supports rotation instructions, generate // a rot[lr]. SDNode *DAGCombiner::MatchRotate(SDValue LHS, SDValue RHS, const SDLoc &DL) { // Must be a legal type. Expanded 'n promoted things won't work with rotates. EVT VT = LHS.getValueType(); if (!TLI.isTypeLegal(VT)) return nullptr; // The target must have at least one rotate flavor. bool HasROTL = hasOperation(ISD::ROTL, VT); bool HasROTR = hasOperation(ISD::ROTR, VT); if (!HasROTL && !HasROTR) return nullptr; // Check for truncated rotate. if (LHS.getOpcode() == ISD::TRUNCATE && RHS.getOpcode() == ISD::TRUNCATE && LHS.getOperand(0).getValueType() == RHS.getOperand(0).getValueType()) { assert(LHS.getValueType() == RHS.getValueType()); if (SDNode *Rot = MatchRotate(LHS.getOperand(0), RHS.getOperand(0), DL)) { return DAG.getNode(ISD::TRUNCATE, SDLoc(LHS), LHS.getValueType(), SDValue(Rot, 0)).getNode(); } } // Match "(X shl/srl V1) & V2" where V2 may not be present. SDValue LHSShift; // The shift. SDValue LHSMask; // AND value if any. matchRotateHalf(DAG, LHS, LHSShift, LHSMask); SDValue RHSShift; // The shift. SDValue RHSMask; // AND value if any. matchRotateHalf(DAG, RHS, RHSShift, RHSMask); // If neither side matched a rotate half, bail if (!LHSShift && !RHSShift) return nullptr; // InstCombine may have combined a constant shl, srl, mul, or udiv with one // side of the rotate, so try to handle that here. In all cases we need to // pass the matched shift from the opposite side to compute the opcode and // needed shift amount to extract. We still want to do this if both sides // matched a rotate half because one half may be a potential overshift that // can be broken down (ie if InstCombine merged two shl or srl ops into a // single one). // Have LHS side of the rotate, try to extract the needed shift from the RHS. if (LHSShift) if (SDValue NewRHSShift = extractShiftForRotate(DAG, LHSShift, RHS, RHSMask, DL)) RHSShift = NewRHSShift; // Have RHS side of the rotate, try to extract the needed shift from the LHS. if (RHSShift) if (SDValue NewLHSShift = extractShiftForRotate(DAG, RHSShift, LHS, LHSMask, DL)) LHSShift = NewLHSShift; // If a side is still missing, nothing else we can do. if (!RHSShift || !LHSShift) return nullptr; // At this point we've matched or extracted a shift op on each side. if (LHSShift.getOperand(0) != RHSShift.getOperand(0)) return nullptr; // Not shifting the same value. if (LHSShift.getOpcode() == RHSShift.getOpcode()) return nullptr; // Shifts must disagree. // Canonicalize shl to left side in a shl/srl pair. if (RHSShift.getOpcode() == ISD::SHL) { std::swap(LHS, RHS); std::swap(LHSShift, RHSShift); std::swap(LHSMask, RHSMask); } unsigned EltSizeInBits = VT.getScalarSizeInBits(); SDValue LHSShiftArg = LHSShift.getOperand(0); SDValue LHSShiftAmt = LHSShift.getOperand(1); SDValue RHSShiftArg = RHSShift.getOperand(0); SDValue RHSShiftAmt = RHSShift.getOperand(1); // fold (or (shl x, C1), (srl x, C2)) -> (rotl x, C1) // fold (or (shl x, C1), (srl x, C2)) -> (rotr x, C2) auto MatchRotateSum = [EltSizeInBits](ConstantSDNode *LHS, ConstantSDNode *RHS) { return (LHS->getAPIntValue() + RHS->getAPIntValue()) == EltSizeInBits; }; if (ISD::matchBinaryPredicate(LHSShiftAmt, RHSShiftAmt, MatchRotateSum)) { SDValue Rot = DAG.getNode(HasROTL ? ISD::ROTL : ISD::ROTR, DL, VT, LHSShiftArg, HasROTL ? LHSShiftAmt : RHSShiftAmt); // If there is an AND of either shifted operand, apply it to the result. if (LHSMask.getNode() || RHSMask.getNode()) { SDValue AllOnes = DAG.getAllOnesConstant(DL, VT); SDValue Mask = AllOnes; if (LHSMask.getNode()) { SDValue RHSBits = DAG.getNode(ISD::SRL, DL, VT, AllOnes, RHSShiftAmt); Mask = DAG.getNode(ISD::AND, DL, VT, Mask, DAG.getNode(ISD::OR, DL, VT, LHSMask, RHSBits)); } if (RHSMask.getNode()) { SDValue LHSBits = DAG.getNode(ISD::SHL, DL, VT, AllOnes, LHSShiftAmt); Mask = DAG.getNode(ISD::AND, DL, VT, Mask, DAG.getNode(ISD::OR, DL, VT, RHSMask, LHSBits)); } Rot = DAG.getNode(ISD::AND, DL, VT, Rot, Mask); } return Rot.getNode(); } // If there is a mask here, and we have a variable shift, we can't be sure // that we're masking out the right stuff. if (LHSMask.getNode() || RHSMask.getNode()) return nullptr; // If the shift amount is sign/zext/any-extended just peel it off. SDValue LExtOp0 = LHSShiftAmt; SDValue RExtOp0 = RHSShiftAmt; if ((LHSShiftAmt.getOpcode() == ISD::SIGN_EXTEND || LHSShiftAmt.getOpcode() == ISD::ZERO_EXTEND || LHSShiftAmt.getOpcode() == ISD::ANY_EXTEND || LHSShiftAmt.getOpcode() == ISD::TRUNCATE) && (RHSShiftAmt.getOpcode() == ISD::SIGN_EXTEND || RHSShiftAmt.getOpcode() == ISD::ZERO_EXTEND || RHSShiftAmt.getOpcode() == ISD::ANY_EXTEND || RHSShiftAmt.getOpcode() == ISD::TRUNCATE)) { LExtOp0 = LHSShiftAmt.getOperand(0); RExtOp0 = RHSShiftAmt.getOperand(0); } SDNode *TryL = MatchRotatePosNeg(LHSShiftArg, LHSShiftAmt, RHSShiftAmt, LExtOp0, RExtOp0, ISD::ROTL, ISD::ROTR, DL); if (TryL) return TryL; SDNode *TryR = MatchRotatePosNeg(RHSShiftArg, RHSShiftAmt, LHSShiftAmt, RExtOp0, LExtOp0, ISD::ROTR, ISD::ROTL, DL); if (TryR) return TryR; return nullptr; } namespace { /// Represents known origin of an individual byte in load combine pattern. The /// value of the byte is either constant zero or comes from memory. struct ByteProvider { // For constant zero providers Load is set to nullptr. For memory providers // Load represents the node which loads the byte from memory. // ByteOffset is the offset of the byte in the value produced by the load. LoadSDNode *Load = nullptr; unsigned ByteOffset = 0; ByteProvider() = default; static ByteProvider getMemory(LoadSDNode *Load, unsigned ByteOffset) { return ByteProvider(Load, ByteOffset); } static ByteProvider getConstantZero() { return ByteProvider(nullptr, 0); } bool isConstantZero() const { return !Load; } bool isMemory() const { return Load; } bool operator==(const ByteProvider &Other) const { return Other.Load == Load && Other.ByteOffset == ByteOffset; } private: ByteProvider(LoadSDNode *Load, unsigned ByteOffset) : Load(Load), ByteOffset(ByteOffset) {} }; } // end anonymous namespace /// Recursively traverses the expression calculating the origin of the requested /// byte of the given value. Returns None if the provider can't be calculated. /// /// For all the values except the root of the expression verifies that the value /// has exactly one use and if it's not true return None. This way if the origin /// of the byte is returned it's guaranteed that the values which contribute to /// the byte are not used outside of this expression. /// /// Because the parts of the expression are not allowed to have more than one /// use this function iterates over trees, not DAGs. So it never visits the same /// node more than once. static const Optional calculateByteProvider(SDValue Op, unsigned Index, unsigned Depth, bool Root = false) { // Typical i64 by i8 pattern requires recursion up to 8 calls depth if (Depth == 10) return None; if (!Root && !Op.hasOneUse()) return None; assert(Op.getValueType().isScalarInteger() && "can't handle other types"); unsigned BitWidth = Op.getValueSizeInBits(); if (BitWidth % 8 != 0) return None; unsigned ByteWidth = BitWidth / 8; assert(Index < ByteWidth && "invalid index requested"); (void) ByteWidth; switch (Op.getOpcode()) { case ISD::OR: { auto LHS = calculateByteProvider(Op->getOperand(0), Index, Depth + 1); if (!LHS) return None; auto RHS = calculateByteProvider(Op->getOperand(1), Index, Depth + 1); if (!RHS) return None; if (LHS->isConstantZero()) return RHS; if (RHS->isConstantZero()) return LHS; return None; } case ISD::SHL: { auto ShiftOp = dyn_cast(Op->getOperand(1)); if (!ShiftOp) return None; uint64_t BitShift = ShiftOp->getZExtValue(); if (BitShift % 8 != 0) return None; uint64_t ByteShift = BitShift / 8; return Index < ByteShift ? ByteProvider::getConstantZero() : calculateByteProvider(Op->getOperand(0), Index - ByteShift, Depth + 1); } case ISD::ANY_EXTEND: case ISD::SIGN_EXTEND: case ISD::ZERO_EXTEND: { SDValue NarrowOp = Op->getOperand(0); unsigned NarrowBitWidth = NarrowOp.getScalarValueSizeInBits(); if (NarrowBitWidth % 8 != 0) return None; uint64_t NarrowByteWidth = NarrowBitWidth / 8; if (Index >= NarrowByteWidth) return Op.getOpcode() == ISD::ZERO_EXTEND ? Optional(ByteProvider::getConstantZero()) : None; return calculateByteProvider(NarrowOp, Index, Depth + 1); } case ISD::BSWAP: return calculateByteProvider(Op->getOperand(0), ByteWidth - Index - 1, Depth + 1); case ISD::LOAD: { auto L = cast(Op.getNode()); if (L->isVolatile() || L->isIndexed()) return None; unsigned NarrowBitWidth = L->getMemoryVT().getSizeInBits(); if (NarrowBitWidth % 8 != 0) return None; uint64_t NarrowByteWidth = NarrowBitWidth / 8; if (Index >= NarrowByteWidth) return L->getExtensionType() == ISD::ZEXTLOAD ? Optional(ByteProvider::getConstantZero()) : None; return ByteProvider::getMemory(L, Index); } } return None; } /// Match a pattern where a wide type scalar value is loaded by several narrow /// loads and combined by shifts and ors. Fold it into a single load or a load /// and a BSWAP if the targets supports it. /// /// Assuming little endian target: /// i8 *a = ... /// i32 val = a[0] | (a[1] << 8) | (a[2] << 16) | (a[3] << 24) /// => /// i32 val = *((i32)a) /// /// i8 *a = ... /// i32 val = (a[0] << 24) | (a[1] << 16) | (a[2] << 8) | a[3] /// => /// i32 val = BSWAP(*((i32)a)) /// /// TODO: This rule matches complex patterns with OR node roots and doesn't /// interact well with the worklist mechanism. When a part of the pattern is /// updated (e.g. one of the loads) its direct users are put into the worklist, /// but the root node of the pattern which triggers the load combine is not /// necessarily a direct user of the changed node. For example, once the address /// of t28 load is reassociated load combine won't be triggered: /// t25: i32 = add t4, Constant:i32<2> /// t26: i64 = sign_extend t25 /// t27: i64 = add t2, t26 /// t28: i8,ch = load t0, t27, undef:i64 /// t29: i32 = zero_extend t28 /// t32: i32 = shl t29, Constant:i8<8> /// t33: i32 = or t23, t32 /// As a possible fix visitLoad can check if the load can be a part of a load /// combine pattern and add corresponding OR roots to the worklist. SDValue DAGCombiner::MatchLoadCombine(SDNode *N) { assert(N->getOpcode() == ISD::OR && "Can only match load combining against OR nodes"); // Handles simple types only EVT VT = N->getValueType(0); if (VT != MVT::i16 && VT != MVT::i32 && VT != MVT::i64) return SDValue(); unsigned ByteWidth = VT.getSizeInBits() / 8; const TargetLowering &TLI = DAG.getTargetLoweringInfo(); // Before legalize we can introduce too wide illegal loads which will be later // split into legal sized loads. This enables us to combine i64 load by i8 // patterns to a couple of i32 loads on 32 bit targets. if (LegalOperations && !TLI.isOperationLegal(ISD::LOAD, VT)) return SDValue(); std::function LittleEndianByteAt = []( unsigned BW, unsigned i) { return i; }; std::function BigEndianByteAt = []( unsigned BW, unsigned i) { return BW - i - 1; }; bool IsBigEndianTarget = DAG.getDataLayout().isBigEndian(); auto MemoryByteOffset = [&] (ByteProvider P) { assert(P.isMemory() && "Must be a memory byte provider"); unsigned LoadBitWidth = P.Load->getMemoryVT().getSizeInBits(); assert(LoadBitWidth % 8 == 0 && "can only analyze providers for individual bytes not bit"); unsigned LoadByteWidth = LoadBitWidth / 8; return IsBigEndianTarget ? BigEndianByteAt(LoadByteWidth, P.ByteOffset) : LittleEndianByteAt(LoadByteWidth, P.ByteOffset); }; Optional Base; SDValue Chain; SmallPtrSet Loads; Optional FirstByteProvider; int64_t FirstOffset = INT64_MAX; // Check if all the bytes of the OR we are looking at are loaded from the same // base address. Collect bytes offsets from Base address in ByteOffsets. SmallVector ByteOffsets(ByteWidth); for (unsigned i = 0; i < ByteWidth; i++) { auto P = calculateByteProvider(SDValue(N, 0), i, 0, /*Root=*/true); if (!P || !P->isMemory()) // All the bytes must be loaded from memory return SDValue(); LoadSDNode *L = P->Load; assert(L->hasNUsesOfValue(1, 0) && !L->isVolatile() && !L->isIndexed() && "Must be enforced by calculateByteProvider"); assert(L->getOffset().isUndef() && "Unindexed load must have undef offset"); // All loads must share the same chain SDValue LChain = L->getChain(); if (!Chain) Chain = LChain; else if (Chain != LChain) return SDValue(); // Loads must share the same base address BaseIndexOffset Ptr = BaseIndexOffset::match(L, DAG); int64_t ByteOffsetFromBase = 0; if (!Base) Base = Ptr; else if (!Base->equalBaseIndex(Ptr, DAG, ByteOffsetFromBase)) return SDValue(); // Calculate the offset of the current byte from the base address ByteOffsetFromBase += MemoryByteOffset(*P); ByteOffsets[i] = ByteOffsetFromBase; // Remember the first byte load if (ByteOffsetFromBase < FirstOffset) { FirstByteProvider = P; FirstOffset = ByteOffsetFromBase; } Loads.insert(L); } assert(!Loads.empty() && "All the bytes of the value must be loaded from " "memory, so there must be at least one load which produces the value"); assert(Base && "Base address of the accessed memory location must be set"); assert(FirstOffset != INT64_MAX && "First byte offset must be set"); // Check if the bytes of the OR we are looking at match with either big or // little endian value load bool BigEndian = true, LittleEndian = true; for (unsigned i = 0; i < ByteWidth; i++) { int64_t CurrentByteOffset = ByteOffsets[i] - FirstOffset; LittleEndian &= CurrentByteOffset == LittleEndianByteAt(ByteWidth, i); BigEndian &= CurrentByteOffset == BigEndianByteAt(ByteWidth, i); if (!BigEndian && !LittleEndian) return SDValue(); } assert((BigEndian != LittleEndian) && "should be either or"); assert(FirstByteProvider && "must be set"); // Ensure that the first byte is loaded from zero offset of the first load. // So the combined value can be loaded from the first load address. if (MemoryByteOffset(*FirstByteProvider) != 0) return SDValue(); LoadSDNode *FirstLoad = FirstByteProvider->Load; // The node we are looking at matches with the pattern, check if we can // replace it with a single load and bswap if needed. // If the load needs byte swap check if the target supports it bool NeedsBswap = IsBigEndianTarget != BigEndian; // Before legalize we can introduce illegal bswaps which will be later // converted to an explicit bswap sequence. This way we end up with a single // load and byte shuffling instead of several loads and byte shuffling. if (NeedsBswap && LegalOperations && !TLI.isOperationLegal(ISD::BSWAP, VT)) return SDValue(); // Check that a load of the wide type is both allowed and fast on the target bool Fast = false; bool Allowed = TLI.allowsMemoryAccess(*DAG.getContext(), DAG.getDataLayout(), VT, FirstLoad->getAddressSpace(), FirstLoad->getAlignment(), &Fast); if (!Allowed || !Fast) return SDValue(); SDValue NewLoad = DAG.getLoad(VT, SDLoc(N), Chain, FirstLoad->getBasePtr(), FirstLoad->getPointerInfo(), FirstLoad->getAlignment()); // Transfer chain users from old loads to the new load. for (LoadSDNode *L : Loads) DAG.ReplaceAllUsesOfValueWith(SDValue(L, 1), SDValue(NewLoad.getNode(), 1)); return NeedsBswap ? DAG.getNode(ISD::BSWAP, SDLoc(N), VT, NewLoad) : NewLoad; } // If the target has andn, bsl, or a similar bit-select instruction, // we want to unfold masked merge, with canonical pattern of: // | A | |B| // ((x ^ y) & m) ^ y // | D | // Into: // (x & m) | (y & ~m) // If y is a constant, and the 'andn' does not work with immediates, // we unfold into a different pattern: // ~(~x & m) & (m | y) // NOTE: we don't unfold the pattern if 'xor' is actually a 'not', because at // the very least that breaks andnpd / andnps patterns, and because those // patterns are simplified in IR and shouldn't be created in the DAG SDValue DAGCombiner::unfoldMaskedMerge(SDNode *N) { assert(N->getOpcode() == ISD::XOR); // Don't touch 'not' (i.e. where y = -1). if (isAllOnesOrAllOnesSplat(N->getOperand(1))) return SDValue(); EVT VT = N->getValueType(0); // There are 3 commutable operators in the pattern, // so we have to deal with 8 possible variants of the basic pattern. SDValue X, Y, M; auto matchAndXor = [&X, &Y, &M](SDValue And, unsigned XorIdx, SDValue Other) { if (And.getOpcode() != ISD::AND || !And.hasOneUse()) return false; SDValue Xor = And.getOperand(XorIdx); if (Xor.getOpcode() != ISD::XOR || !Xor.hasOneUse()) return false; SDValue Xor0 = Xor.getOperand(0); SDValue Xor1 = Xor.getOperand(1); // Don't touch 'not' (i.e. where y = -1). if (isAllOnesOrAllOnesSplat(Xor1)) return false; if (Other == Xor0) std::swap(Xor0, Xor1); if (Other != Xor1) return false; X = Xor0; Y = Xor1; M = And.getOperand(XorIdx ? 0 : 1); return true; }; SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); if (!matchAndXor(N0, 0, N1) && !matchAndXor(N0, 1, N1) && !matchAndXor(N1, 0, N0) && !matchAndXor(N1, 1, N0)) return SDValue(); // Don't do anything if the mask is constant. This should not be reachable. // InstCombine should have already unfolded this pattern, and DAGCombiner // probably shouldn't produce it, too. if (isa(M.getNode())) return SDValue(); // We can transform if the target has AndNot if (!TLI.hasAndNot(M)) return SDValue(); SDLoc DL(N); // If Y is a constant, check that 'andn' works with immediates. if (!TLI.hasAndNot(Y)) { assert(TLI.hasAndNot(X) && "Only mask is a variable? Unreachable."); // If not, we need to do a bit more work to make sure andn is still used. SDValue NotX = DAG.getNOT(DL, X, VT); SDValue LHS = DAG.getNode(ISD::AND, DL, VT, NotX, M); SDValue NotLHS = DAG.getNOT(DL, LHS, VT); SDValue RHS = DAG.getNode(ISD::OR, DL, VT, M, Y); return DAG.getNode(ISD::AND, DL, VT, NotLHS, RHS); } SDValue LHS = DAG.getNode(ISD::AND, DL, VT, X, M); SDValue NotM = DAG.getNOT(DL, M, VT); SDValue RHS = DAG.getNode(ISD::AND, DL, VT, Y, NotM); return DAG.getNode(ISD::OR, DL, VT, LHS, RHS); } SDValue DAGCombiner::visitXOR(SDNode *N) { SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); EVT VT = N0.getValueType(); // fold vector ops if (VT.isVector()) { if (SDValue FoldedVOp = SimplifyVBinOp(N)) return FoldedVOp; // fold (xor x, 0) -> x, vector edition if (ISD::isBuildVectorAllZeros(N0.getNode())) return N1; if (ISD::isBuildVectorAllZeros(N1.getNode())) return N0; } // fold (xor undef, undef) -> 0. This is a common idiom (misuse). SDLoc DL(N); if (N0.isUndef() && N1.isUndef()) return DAG.getConstant(0, DL, VT); // fold (xor x, undef) -> undef if (N0.isUndef()) return N0; if (N1.isUndef()) return N1; // fold (xor c1, c2) -> c1^c2 ConstantSDNode *N0C = getAsNonOpaqueConstant(N0); ConstantSDNode *N1C = getAsNonOpaqueConstant(N1); if (N0C && N1C) return DAG.FoldConstantArithmetic(ISD::XOR, DL, VT, N0C, N1C); // canonicalize constant to RHS if (DAG.isConstantIntBuildVectorOrConstantInt(N0) && !DAG.isConstantIntBuildVectorOrConstantInt(N1)) return DAG.getNode(ISD::XOR, DL, VT, N1, N0); // fold (xor x, 0) -> x if (isNullConstant(N1)) return N0; if (SDValue NewSel = foldBinOpIntoSelect(N)) return NewSel; // reassociate xor if (SDValue RXOR = ReassociateOps(ISD::XOR, DL, N0, N1, N->getFlags())) return RXOR; // fold !(x cc y) -> (x !cc y) unsigned N0Opcode = N0.getOpcode(); SDValue LHS, RHS, CC; if (TLI.isConstTrueVal(N1.getNode()) && isSetCCEquivalent(N0, LHS, RHS, CC)) { ISD::CondCode NotCC = ISD::getSetCCInverse(cast(CC)->get(), LHS.getValueType().isInteger()); if (!LegalOperations || TLI.isCondCodeLegal(NotCC, LHS.getSimpleValueType())) { switch (N0Opcode) { default: llvm_unreachable("Unhandled SetCC Equivalent!"); case ISD::SETCC: return DAG.getSetCC(SDLoc(N0), VT, LHS, RHS, NotCC); case ISD::SELECT_CC: return DAG.getSelectCC(SDLoc(N0), LHS, RHS, N0.getOperand(2), N0.getOperand(3), NotCC); } } } // fold (not (zext (setcc x, y))) -> (zext (not (setcc x, y))) if (isOneConstant(N1) && N0Opcode == ISD::ZERO_EXTEND && N0.hasOneUse() && isSetCCEquivalent(N0.getOperand(0), LHS, RHS, CC)){ SDValue V = N0.getOperand(0); SDLoc DL0(N0); V = DAG.getNode(ISD::XOR, DL0, V.getValueType(), V, DAG.getConstant(1, DL0, V.getValueType())); AddToWorklist(V.getNode()); return DAG.getNode(ISD::ZERO_EXTEND, DL, VT, V); } // fold (not (or x, y)) -> (and (not x), (not y)) iff x or y are setcc if (isOneConstant(N1) && VT == MVT::i1 && N0.hasOneUse() && (N0Opcode == ISD::OR || N0Opcode == ISD::AND)) { SDValue LHS = N0.getOperand(0), RHS = N0.getOperand(1); if (isOneUseSetCC(RHS) || isOneUseSetCC(LHS)) { unsigned NewOpcode = N0Opcode == ISD::AND ? ISD::OR : ISD::AND; LHS = DAG.getNode(ISD::XOR, SDLoc(LHS), VT, LHS, N1); // LHS = ~LHS RHS = DAG.getNode(ISD::XOR, SDLoc(RHS), VT, RHS, N1); // RHS = ~RHS AddToWorklist(LHS.getNode()); AddToWorklist(RHS.getNode()); return DAG.getNode(NewOpcode, DL, VT, LHS, RHS); } } // fold (not (or x, y)) -> (and (not x), (not y)) iff x or y are constants if (isAllOnesConstant(N1) && N0.hasOneUse() && (N0Opcode == ISD::OR || N0Opcode == ISD::AND)) { SDValue LHS = N0.getOperand(0), RHS = N0.getOperand(1); if (isa(RHS) || isa(LHS)) { unsigned NewOpcode = N0Opcode == ISD::AND ? ISD::OR : ISD::AND; LHS = DAG.getNode(ISD::XOR, SDLoc(LHS), VT, LHS, N1); // LHS = ~LHS RHS = DAG.getNode(ISD::XOR, SDLoc(RHS), VT, RHS, N1); // RHS = ~RHS AddToWorklist(LHS.getNode()); AddToWorklist(RHS.getNode()); return DAG.getNode(NewOpcode, DL, VT, LHS, RHS); } } // fold (xor (and x, y), y) -> (and (not x), y) if (N0Opcode == ISD::AND && N0.hasOneUse() && N0->getOperand(1) == N1) { SDValue X = N0.getOperand(0); SDValue NotX = DAG.getNOT(SDLoc(X), X, VT); AddToWorklist(NotX.getNode()); return DAG.getNode(ISD::AND, DL, VT, NotX, N1); } if ((N0Opcode == ISD::SRL || N0Opcode == ISD::SHL) && N0.hasOneUse()) { ConstantSDNode *XorC = isConstOrConstSplat(N1); ConstantSDNode *ShiftC = isConstOrConstSplat(N0.getOperand(1)); unsigned BitWidth = VT.getScalarSizeInBits(); if (XorC && ShiftC) { // Don't crash on an oversized shift. We can not guarantee that a bogus // shift has been simplified to undef. uint64_t ShiftAmt = ShiftC->getLimitedValue(); if (ShiftAmt < BitWidth) { APInt Ones = APInt::getAllOnesValue(BitWidth); Ones = N0Opcode == ISD::SHL ? Ones.shl(ShiftAmt) : Ones.lshr(ShiftAmt); if (XorC->getAPIntValue() == Ones) { // If the xor constant is a shifted -1, do a 'not' before the shift: // xor (X << ShiftC), XorC --> (not X) << ShiftC // xor (X >> ShiftC), XorC --> (not X) >> ShiftC SDValue Not = DAG.getNOT(DL, N0.getOperand(0), VT); return DAG.getNode(N0Opcode, DL, VT, Not, N0.getOperand(1)); } } } } // fold Y = sra (X, size(X)-1); xor (add (X, Y), Y) -> (abs X) if (TLI.isOperationLegalOrCustom(ISD::ABS, VT)) { SDValue A = N0Opcode == ISD::ADD ? N0 : N1; SDValue S = N0Opcode == ISD::SRA ? N0 : N1; if (A.getOpcode() == ISD::ADD && S.getOpcode() == ISD::SRA) { SDValue A0 = A.getOperand(0), A1 = A.getOperand(1); SDValue S0 = S.getOperand(0); if ((A0 == S && A1 == S0) || (A1 == S && A0 == S0)) { unsigned OpSizeInBits = VT.getScalarSizeInBits(); if (ConstantSDNode *C = isConstOrConstSplat(S.getOperand(1))) if (C->getAPIntValue() == (OpSizeInBits - 1)) return DAG.getNode(ISD::ABS, DL, VT, S0); } } } // fold (xor x, x) -> 0 if (N0 == N1) return tryFoldToZero(DL, TLI, VT, DAG, LegalOperations); // fold (xor (shl 1, x), -1) -> (rotl ~1, x) // Here is a concrete example of this equivalence: // i16 x == 14 // i16 shl == 1 << 14 == 16384 == 0b0100000000000000 // i16 xor == ~(1 << 14) == 49151 == 0b1011111111111111 // // => // // i16 ~1 == 0b1111111111111110 // i16 rol(~1, 14) == 0b1011111111111111 // // Some additional tips to help conceptualize this transform: // - Try to see the operation as placing a single zero in a value of all ones. // - There exists no value for x which would allow the result to contain zero. // - Values of x larger than the bitwidth are undefined and do not require a // consistent result. // - Pushing the zero left requires shifting one bits in from the right. // A rotate left of ~1 is a nice way of achieving the desired result. if (TLI.isOperationLegalOrCustom(ISD::ROTL, VT) && N0Opcode == ISD::SHL && isAllOnesConstant(N1) && isOneConstant(N0.getOperand(0))) { return DAG.getNode(ISD::ROTL, DL, VT, DAG.getConstant(~1, DL, VT), N0.getOperand(1)); } // Simplify: xor (op x...), (op y...) -> (op (xor x, y)) if (N0Opcode == N1.getOpcode()) if (SDValue V = hoistLogicOpWithSameOpcodeHands(N)) return V; // Unfold ((x ^ y) & m) ^ y into (x & m) | (y & ~m) if profitable if (SDValue MM = unfoldMaskedMerge(N)) return MM; // Simplify the expression using non-local knowledge. if (SimplifyDemandedBits(SDValue(N, 0))) return SDValue(N, 0); return SDValue(); } /// Handle transforms common to the three shifts, when the shift amount is a /// constant. SDValue DAGCombiner::visitShiftByConstant(SDNode *N, ConstantSDNode *Amt) { // Do not turn a 'not' into a regular xor. if (isBitwiseNot(N->getOperand(0))) return SDValue(); SDNode *LHS = N->getOperand(0).getNode(); if (!LHS->hasOneUse()) return SDValue(); // We want to pull some binops through shifts, so that we have (and (shift)) // instead of (shift (and)), likewise for add, or, xor, etc. This sort of // thing happens with address calculations, so it's important to canonicalize // it. bool HighBitSet = false; // Can we transform this if the high bit is set? switch (LHS->getOpcode()) { default: return SDValue(); case ISD::OR: case ISD::XOR: HighBitSet = false; // We can only transform sra if the high bit is clear. break; case ISD::AND: HighBitSet = true; // We can only transform sra if the high bit is set. break; case ISD::ADD: if (N->getOpcode() != ISD::SHL) return SDValue(); // only shl(add) not sr[al](add). HighBitSet = false; // We can only transform sra if the high bit is clear. break; } // We require the RHS of the binop to be a constant and not opaque as well. ConstantSDNode *BinOpCst = getAsNonOpaqueConstant(LHS->getOperand(1)); if (!BinOpCst) return SDValue(); // FIXME: disable this unless the input to the binop is a shift by a constant // or is copy/select.Enable this in other cases when figure out it's exactly profitable. SDNode *BinOpLHSVal = LHS->getOperand(0).getNode(); bool isShift = BinOpLHSVal->getOpcode() == ISD::SHL || BinOpLHSVal->getOpcode() == ISD::SRA || BinOpLHSVal->getOpcode() == ISD::SRL; bool isCopyOrSelect = BinOpLHSVal->getOpcode() == ISD::CopyFromReg || BinOpLHSVal->getOpcode() == ISD::SELECT; if ((!isShift || !isa(BinOpLHSVal->getOperand(1))) && !isCopyOrSelect) return SDValue(); if (isCopyOrSelect && N->hasOneUse()) return SDValue(); EVT VT = N->getValueType(0); // If this is a signed shift right, and the high bit is modified by the // logical operation, do not perform the transformation. The highBitSet // boolean indicates the value of the high bit of the constant which would // cause it to be modified for this operation. if (N->getOpcode() == ISD::SRA) { bool BinOpRHSSignSet = BinOpCst->getAPIntValue().isNegative(); if (BinOpRHSSignSet != HighBitSet) return SDValue(); } if (!TLI.isDesirableToCommuteWithShift(N, Level)) return SDValue(); // Fold the constants, shifting the binop RHS by the shift amount. SDValue NewRHS = DAG.getNode(N->getOpcode(), SDLoc(LHS->getOperand(1)), N->getValueType(0), LHS->getOperand(1), N->getOperand(1)); assert(isa(NewRHS) && "Folding was not successful!"); // Create the new shift. SDValue NewShift = DAG.getNode(N->getOpcode(), SDLoc(LHS->getOperand(0)), VT, LHS->getOperand(0), N->getOperand(1)); // Create the new binop. return DAG.getNode(LHS->getOpcode(), SDLoc(N), VT, NewShift, NewRHS); } SDValue DAGCombiner::distributeTruncateThroughAnd(SDNode *N) { assert(N->getOpcode() == ISD::TRUNCATE); assert(N->getOperand(0).getOpcode() == ISD::AND); // (truncate:TruncVT (and N00, N01C)) -> (and (truncate:TruncVT N00), TruncC) if (N->hasOneUse() && N->getOperand(0).hasOneUse()) { SDValue N01 = N->getOperand(0).getOperand(1); if (isConstantOrConstantVector(N01, /* NoOpaques */ true)) { SDLoc DL(N); EVT TruncVT = N->getValueType(0); SDValue N00 = N->getOperand(0).getOperand(0); SDValue Trunc00 = DAG.getNode(ISD::TRUNCATE, DL, TruncVT, N00); SDValue Trunc01 = DAG.getNode(ISD::TRUNCATE, DL, TruncVT, N01); AddToWorklist(Trunc00.getNode()); AddToWorklist(Trunc01.getNode()); return DAG.getNode(ISD::AND, DL, TruncVT, Trunc00, Trunc01); } } return SDValue(); } SDValue DAGCombiner::visitRotate(SDNode *N) { SDLoc dl(N); SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); EVT VT = N->getValueType(0); unsigned Bitsize = VT.getScalarSizeInBits(); // fold (rot x, 0) -> x if (isNullOrNullSplat(N1)) return N0; // fold (rot x, c) -> x iff (c % BitSize) == 0 if (isPowerOf2_32(Bitsize) && Bitsize > 1) { APInt ModuloMask(N1.getScalarValueSizeInBits(), Bitsize - 1); if (DAG.MaskedValueIsZero(N1, ModuloMask)) return N0; } // fold (rot x, c) -> (rot x, c % BitSize) if (ConstantSDNode *Cst = isConstOrConstSplat(N1)) { if (Cst->getAPIntValue().uge(Bitsize)) { uint64_t RotAmt = Cst->getAPIntValue().urem(Bitsize); return DAG.getNode(N->getOpcode(), dl, VT, N0, DAG.getConstant(RotAmt, dl, N1.getValueType())); } } // fold (rot* x, (trunc (and y, c))) -> (rot* x, (and (trunc y), (trunc c))). if (N1.getOpcode() == ISD::TRUNCATE && N1.getOperand(0).getOpcode() == ISD::AND) { if (SDValue NewOp1 = distributeTruncateThroughAnd(N1.getNode())) return DAG.getNode(N->getOpcode(), dl, VT, N0, NewOp1); } unsigned NextOp = N0.getOpcode(); // fold (rot* (rot* x, c2), c1) -> (rot* x, c1 +- c2 % bitsize) if (NextOp == ISD::ROTL || NextOp == ISD::ROTR) { SDNode *C1 = DAG.isConstantIntBuildVectorOrConstantInt(N1); SDNode *C2 = DAG.isConstantIntBuildVectorOrConstantInt(N0.getOperand(1)); if (C1 && C2 && C1->getValueType(0) == C2->getValueType(0)) { EVT ShiftVT = C1->getValueType(0); bool SameSide = (N->getOpcode() == NextOp); unsigned CombineOp = SameSide ? ISD::ADD : ISD::SUB; if (SDValue CombinedShift = DAG.FoldConstantArithmetic(CombineOp, dl, ShiftVT, C1, C2)) { SDValue BitsizeC = DAG.getConstant(Bitsize, dl, ShiftVT); SDValue CombinedShiftNorm = DAG.FoldConstantArithmetic( ISD::SREM, dl, ShiftVT, CombinedShift.getNode(), BitsizeC.getNode()); return DAG.getNode(N->getOpcode(), dl, VT, N0->getOperand(0), CombinedShiftNorm); } } } return SDValue(); } SDValue DAGCombiner::visitSHL(SDNode *N) { SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); if (SDValue V = DAG.simplifyShift(N0, N1)) return V; EVT VT = N0.getValueType(); unsigned OpSizeInBits = VT.getScalarSizeInBits(); // fold vector ops if (VT.isVector()) { if (SDValue FoldedVOp = SimplifyVBinOp(N)) return FoldedVOp; BuildVectorSDNode *N1CV = dyn_cast(N1); // If setcc produces all-one true value then: // (shl (and (setcc) N01CV) N1CV) -> (and (setcc) N01CV<isConstant()) { if (N0.getOpcode() == ISD::AND) { SDValue N00 = N0->getOperand(0); SDValue N01 = N0->getOperand(1); BuildVectorSDNode *N01CV = dyn_cast(N01); if (N01CV && N01CV->isConstant() && N00.getOpcode() == ISD::SETCC && TLI.getBooleanContents(N00.getOperand(0).getValueType()) == TargetLowering::ZeroOrNegativeOneBooleanContent) { if (SDValue C = DAG.FoldConstantArithmetic(ISD::SHL, SDLoc(N), VT, N01CV, N1CV)) return DAG.getNode(ISD::AND, SDLoc(N), VT, N00, C); } } } } ConstantSDNode *N1C = isConstOrConstSplat(N1); // fold (shl c1, c2) -> c1<isOpaque()) return DAG.FoldConstantArithmetic(ISD::SHL, SDLoc(N), VT, N0C, N1C); if (SDValue NewSel = foldBinOpIntoSelect(N)) return NewSel; // if (shl x, c) is known to be zero, return 0 if (DAG.MaskedValueIsZero(SDValue(N, 0), APInt::getAllOnesValue(OpSizeInBits))) return DAG.getConstant(0, SDLoc(N), VT); // fold (shl x, (trunc (and y, c))) -> (shl x, (and (trunc y), (trunc c))). if (N1.getOpcode() == ISD::TRUNCATE && N1.getOperand(0).getOpcode() == ISD::AND) { if (SDValue NewOp1 = distributeTruncateThroughAnd(N1.getNode())) return DAG.getNode(ISD::SHL, SDLoc(N), VT, N0, NewOp1); } if (N1C && SimplifyDemandedBits(SDValue(N, 0))) return SDValue(N, 0); // fold (shl (shl x, c1), c2) -> 0 or (shl x, (add c1, c2)) if (N0.getOpcode() == ISD::SHL) { auto MatchOutOfRange = [OpSizeInBits](ConstantSDNode *LHS, ConstantSDNode *RHS) { APInt c1 = LHS->getAPIntValue(); APInt c2 = RHS->getAPIntValue(); zeroExtendToMatch(c1, c2, 1 /* Overflow Bit */); return (c1 + c2).uge(OpSizeInBits); }; if (ISD::matchBinaryPredicate(N1, N0.getOperand(1), MatchOutOfRange)) return DAG.getConstant(0, SDLoc(N), VT); auto MatchInRange = [OpSizeInBits](ConstantSDNode *LHS, ConstantSDNode *RHS) { APInt c1 = LHS->getAPIntValue(); APInt c2 = RHS->getAPIntValue(); zeroExtendToMatch(c1, c2, 1 /* Overflow Bit */); return (c1 + c2).ult(OpSizeInBits); }; if (ISD::matchBinaryPredicate(N1, N0.getOperand(1), MatchInRange)) { SDLoc DL(N); EVT ShiftVT = N1.getValueType(); SDValue Sum = DAG.getNode(ISD::ADD, DL, ShiftVT, N1, N0.getOperand(1)); return DAG.getNode(ISD::SHL, DL, VT, N0.getOperand(0), Sum); } } // fold (shl (ext (shl x, c1)), c2) -> (ext (shl x, (add c1, c2))) // For this to be valid, the second form must not preserve any of the bits // that are shifted out by the inner shift in the first form. This means // the outer shift size must be >= the number of bits added by the ext. // As a corollary, we don't care what kind of ext it is. if (N1C && (N0.getOpcode() == ISD::ZERO_EXTEND || N0.getOpcode() == ISD::ANY_EXTEND || N0.getOpcode() == ISD::SIGN_EXTEND) && N0.getOperand(0).getOpcode() == ISD::SHL) { SDValue N0Op0 = N0.getOperand(0); if (ConstantSDNode *N0Op0C1 = isConstOrConstSplat(N0Op0.getOperand(1))) { APInt c1 = N0Op0C1->getAPIntValue(); APInt c2 = N1C->getAPIntValue(); zeroExtendToMatch(c1, c2, 1 /* Overflow Bit */); EVT InnerShiftVT = N0Op0.getValueType(); uint64_t InnerShiftSize = InnerShiftVT.getScalarSizeInBits(); if (c2.uge(OpSizeInBits - InnerShiftSize)) { SDLoc DL(N0); APInt Sum = c1 + c2; if (Sum.uge(OpSizeInBits)) return DAG.getConstant(0, DL, VT); return DAG.getNode( ISD::SHL, DL, VT, DAG.getNode(N0.getOpcode(), DL, VT, N0Op0->getOperand(0)), DAG.getConstant(Sum.getZExtValue(), DL, N1.getValueType())); } } } // fold (shl (zext (srl x, C)), C) -> (zext (shl (srl x, C), C)) // Only fold this if the inner zext has no other uses to avoid increasing // the total number of instructions. if (N1C && N0.getOpcode() == ISD::ZERO_EXTEND && N0.hasOneUse() && N0.getOperand(0).getOpcode() == ISD::SRL) { SDValue N0Op0 = N0.getOperand(0); if (ConstantSDNode *N0Op0C1 = isConstOrConstSplat(N0Op0.getOperand(1))) { if (N0Op0C1->getAPIntValue().ult(VT.getScalarSizeInBits())) { uint64_t c1 = N0Op0C1->getZExtValue(); uint64_t c2 = N1C->getZExtValue(); if (c1 == c2) { SDValue NewOp0 = N0.getOperand(0); EVT CountVT = NewOp0.getOperand(1).getValueType(); SDLoc DL(N); SDValue NewSHL = DAG.getNode(ISD::SHL, DL, NewOp0.getValueType(), NewOp0, DAG.getConstant(c2, DL, CountVT)); AddToWorklist(NewSHL.getNode()); return DAG.getNode(ISD::ZERO_EXTEND, SDLoc(N0), VT, NewSHL); } } } } // fold (shl (sr[la] exact X, C1), C2) -> (shl X, (C2-C1)) if C1 <= C2 // fold (shl (sr[la] exact X, C1), C2) -> (sr[la] X, (C2-C1)) if C1 > C2 if (N1C && (N0.getOpcode() == ISD::SRL || N0.getOpcode() == ISD::SRA) && N0->getFlags().hasExact()) { if (ConstantSDNode *N0C1 = isConstOrConstSplat(N0.getOperand(1))) { uint64_t C1 = N0C1->getZExtValue(); uint64_t C2 = N1C->getZExtValue(); SDLoc DL(N); if (C1 <= C2) return DAG.getNode(ISD::SHL, DL, VT, N0.getOperand(0), DAG.getConstant(C2 - C1, DL, N1.getValueType())); return DAG.getNode(N0.getOpcode(), DL, VT, N0.getOperand(0), DAG.getConstant(C1 - C2, DL, N1.getValueType())); } } // fold (shl (srl x, c1), c2) -> (and (shl x, (sub c2, c1), MASK) or // (and (srl x, (sub c1, c2), MASK) // Only fold this if the inner shift has no other uses -- if it does, folding // this will increase the total number of instructions. if (N1C && N0.getOpcode() == ISD::SRL && N0.hasOneUse() && TLI.shouldFoldShiftPairToMask(N, Level)) { if (ConstantSDNode *N0C1 = isConstOrConstSplat(N0.getOperand(1))) { uint64_t c1 = N0C1->getZExtValue(); if (c1 < OpSizeInBits) { uint64_t c2 = N1C->getZExtValue(); APInt Mask = APInt::getHighBitsSet(OpSizeInBits, OpSizeInBits - c1); SDValue Shift; if (c2 > c1) { Mask <<= c2 - c1; SDLoc DL(N); Shift = DAG.getNode(ISD::SHL, DL, VT, N0.getOperand(0), DAG.getConstant(c2 - c1, DL, N1.getValueType())); } else { Mask.lshrInPlace(c1 - c2); SDLoc DL(N); Shift = DAG.getNode(ISD::SRL, DL, VT, N0.getOperand(0), DAG.getConstant(c1 - c2, DL, N1.getValueType())); } SDLoc DL(N0); return DAG.getNode(ISD::AND, DL, VT, Shift, DAG.getConstant(Mask, DL, VT)); } } } // fold (shl (sra x, c1), c1) -> (and x, (shl -1, c1)) if (N0.getOpcode() == ISD::SRA && N1 == N0.getOperand(1) && isConstantOrConstantVector(N1, /* No Opaques */ true)) { SDLoc DL(N); SDValue AllBits = DAG.getAllOnesConstant(DL, VT); SDValue HiBitsMask = DAG.getNode(ISD::SHL, DL, VT, AllBits, N1); return DAG.getNode(ISD::AND, DL, VT, N0.getOperand(0), HiBitsMask); } // fold (shl (add x, c1), c2) -> (add (shl x, c2), c1 << c2) // fold (shl (or x, c1), c2) -> (or (shl x, c2), c1 << c2) // Variant of version done on multiply, except mul by a power of 2 is turned // into a shift. if ((N0.getOpcode() == ISD::ADD || N0.getOpcode() == ISD::OR) && N0.getNode()->hasOneUse() && isConstantOrConstantVector(N1, /* No Opaques */ true) && isConstantOrConstantVector(N0.getOperand(1), /* No Opaques */ true) && TLI.isDesirableToCommuteWithShift(N, Level)) { SDValue Shl0 = DAG.getNode(ISD::SHL, SDLoc(N0), VT, N0.getOperand(0), N1); SDValue Shl1 = DAG.getNode(ISD::SHL, SDLoc(N1), VT, N0.getOperand(1), N1); AddToWorklist(Shl0.getNode()); AddToWorklist(Shl1.getNode()); return DAG.getNode(N0.getOpcode(), SDLoc(N), VT, Shl0, Shl1); } // fold (shl (mul x, c1), c2) -> (mul x, c1 << c2) if (N0.getOpcode() == ISD::MUL && N0.getNode()->hasOneUse() && isConstantOrConstantVector(N1, /* No Opaques */ true) && isConstantOrConstantVector(N0.getOperand(1), /* No Opaques */ true)) { SDValue Shl = DAG.getNode(ISD::SHL, SDLoc(N1), VT, N0.getOperand(1), N1); if (isConstantOrConstantVector(Shl)) return DAG.getNode(ISD::MUL, SDLoc(N), VT, N0.getOperand(0), Shl); } if (N1C && !N1C->isOpaque()) if (SDValue NewSHL = visitShiftByConstant(N, N1C)) return NewSHL; return SDValue(); } SDValue DAGCombiner::visitSRA(SDNode *N) { SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); if (SDValue V = DAG.simplifyShift(N0, N1)) return V; EVT VT = N0.getValueType(); unsigned OpSizeInBits = VT.getScalarSizeInBits(); // Arithmetic shifting an all-sign-bit value is a no-op. // fold (sra 0, x) -> 0 // fold (sra -1, x) -> -1 if (DAG.ComputeNumSignBits(N0) == OpSizeInBits) return N0; // fold vector ops if (VT.isVector()) if (SDValue FoldedVOp = SimplifyVBinOp(N)) return FoldedVOp; ConstantSDNode *N1C = isConstOrConstSplat(N1); // fold (sra c1, c2) -> (sra c1, c2) ConstantSDNode *N0C = getAsNonOpaqueConstant(N0); if (N0C && N1C && !N1C->isOpaque()) return DAG.FoldConstantArithmetic(ISD::SRA, SDLoc(N), VT, N0C, N1C); if (SDValue NewSel = foldBinOpIntoSelect(N)) return NewSel; // fold (sra (shl x, c1), c1) -> sext_inreg for some c1 and target supports // sext_inreg. if (N1C && N0.getOpcode() == ISD::SHL && N1 == N0.getOperand(1)) { unsigned LowBits = OpSizeInBits - (unsigned)N1C->getZExtValue(); EVT ExtVT = EVT::getIntegerVT(*DAG.getContext(), LowBits); if (VT.isVector()) ExtVT = EVT::getVectorVT(*DAG.getContext(), ExtVT, VT.getVectorNumElements()); if ((!LegalOperations || TLI.isOperationLegal(ISD::SIGN_EXTEND_INREG, ExtVT))) return DAG.getNode(ISD::SIGN_EXTEND_INREG, SDLoc(N), VT, N0.getOperand(0), DAG.getValueType(ExtVT)); } // fold (sra (sra x, c1), c2) -> (sra x, (add c1, c2)) // clamp (add c1, c2) to max shift. if (N0.getOpcode() == ISD::SRA) { SDLoc DL(N); EVT ShiftVT = N1.getValueType(); EVT ShiftSVT = ShiftVT.getScalarType(); SmallVector ShiftValues; auto SumOfShifts = [&](ConstantSDNode *LHS, ConstantSDNode *RHS) { APInt c1 = LHS->getAPIntValue(); APInt c2 = RHS->getAPIntValue(); zeroExtendToMatch(c1, c2, 1 /* Overflow Bit */); APInt Sum = c1 + c2; unsigned ShiftSum = Sum.uge(OpSizeInBits) ? (OpSizeInBits - 1) : Sum.getZExtValue(); ShiftValues.push_back(DAG.getConstant(ShiftSum, DL, ShiftSVT)); return true; }; if (ISD::matchBinaryPredicate(N1, N0.getOperand(1), SumOfShifts)) { SDValue ShiftValue; if (VT.isVector()) ShiftValue = DAG.getBuildVector(ShiftVT, DL, ShiftValues); else ShiftValue = ShiftValues[0]; return DAG.getNode(ISD::SRA, DL, VT, N0.getOperand(0), ShiftValue); } } // fold (sra (shl X, m), (sub result_size, n)) // -> (sign_extend (trunc (shl X, (sub (sub result_size, n), m)))) for // result_size - n != m. // If truncate is free for the target sext(shl) is likely to result in better // code. if (N0.getOpcode() == ISD::SHL && N1C) { // Get the two constanst of the shifts, CN0 = m, CN = n. const ConstantSDNode *N01C = isConstOrConstSplat(N0.getOperand(1)); if (N01C) { LLVMContext &Ctx = *DAG.getContext(); // Determine what the truncate's result bitsize and type would be. EVT TruncVT = EVT::getIntegerVT(Ctx, OpSizeInBits - N1C->getZExtValue()); if (VT.isVector()) TruncVT = EVT::getVectorVT(Ctx, TruncVT, VT.getVectorNumElements()); // Determine the residual right-shift amount. int ShiftAmt = N1C->getZExtValue() - N01C->getZExtValue(); // If the shift is not a no-op (in which case this should be just a sign // extend already), the truncated to type is legal, sign_extend is legal // on that type, and the truncate to that type is both legal and free, // perform the transform. if ((ShiftAmt > 0) && TLI.isOperationLegalOrCustom(ISD::SIGN_EXTEND, TruncVT) && TLI.isOperationLegalOrCustom(ISD::TRUNCATE, VT) && TLI.isTruncateFree(VT, TruncVT)) { SDLoc DL(N); SDValue Amt = DAG.getConstant(ShiftAmt, DL, getShiftAmountTy(N0.getOperand(0).getValueType())); SDValue Shift = DAG.getNode(ISD::SRL, DL, VT, N0.getOperand(0), Amt); SDValue Trunc = DAG.getNode(ISD::TRUNCATE, DL, TruncVT, Shift); return DAG.getNode(ISD::SIGN_EXTEND, DL, N->getValueType(0), Trunc); } } } // fold (sra x, (trunc (and y, c))) -> (sra x, (and (trunc y), (trunc c))). if (N1.getOpcode() == ISD::TRUNCATE && N1.getOperand(0).getOpcode() == ISD::AND) { if (SDValue NewOp1 = distributeTruncateThroughAnd(N1.getNode())) return DAG.getNode(ISD::SRA, SDLoc(N), VT, N0, NewOp1); } // fold (sra (trunc (srl x, c1)), c2) -> (trunc (sra x, c1 + c2)) // if c1 is equal to the number of bits the trunc removes if (N0.getOpcode() == ISD::TRUNCATE && (N0.getOperand(0).getOpcode() == ISD::SRL || N0.getOperand(0).getOpcode() == ISD::SRA) && N0.getOperand(0).hasOneUse() && N0.getOperand(0).getOperand(1).hasOneUse() && N1C) { SDValue N0Op0 = N0.getOperand(0); if (ConstantSDNode *LargeShift = isConstOrConstSplat(N0Op0.getOperand(1))) { unsigned LargeShiftVal = LargeShift->getZExtValue(); EVT LargeVT = N0Op0.getValueType(); if (LargeVT.getScalarSizeInBits() - OpSizeInBits == LargeShiftVal) { SDLoc DL(N); SDValue Amt = DAG.getConstant(LargeShiftVal + N1C->getZExtValue(), DL, getShiftAmountTy(N0Op0.getOperand(0).getValueType())); SDValue SRA = DAG.getNode(ISD::SRA, DL, LargeVT, N0Op0.getOperand(0), Amt); return DAG.getNode(ISD::TRUNCATE, DL, VT, SRA); } } } // Simplify, based on bits shifted out of the LHS. if (N1C && SimplifyDemandedBits(SDValue(N, 0))) return SDValue(N, 0); // If the sign bit is known to be zero, switch this to a SRL. if (DAG.SignBitIsZero(N0)) return DAG.getNode(ISD::SRL, SDLoc(N), VT, N0, N1); if (N1C && !N1C->isOpaque()) if (SDValue NewSRA = visitShiftByConstant(N, N1C)) return NewSRA; return SDValue(); } SDValue DAGCombiner::visitSRL(SDNode *N) { SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); if (SDValue V = DAG.simplifyShift(N0, N1)) return V; EVT VT = N0.getValueType(); unsigned OpSizeInBits = VT.getScalarSizeInBits(); // fold vector ops if (VT.isVector()) if (SDValue FoldedVOp = SimplifyVBinOp(N)) return FoldedVOp; ConstantSDNode *N1C = isConstOrConstSplat(N1); // fold (srl c1, c2) -> c1 >>u c2 ConstantSDNode *N0C = getAsNonOpaqueConstant(N0); if (N0C && N1C && !N1C->isOpaque()) return DAG.FoldConstantArithmetic(ISD::SRL, SDLoc(N), VT, N0C, N1C); if (SDValue NewSel = foldBinOpIntoSelect(N)) return NewSel; // if (srl x, c) is known to be zero, return 0 if (N1C && DAG.MaskedValueIsZero(SDValue(N, 0), APInt::getAllOnesValue(OpSizeInBits))) return DAG.getConstant(0, SDLoc(N), VT); // fold (srl (srl x, c1), c2) -> 0 or (srl x, (add c1, c2)) if (N0.getOpcode() == ISD::SRL) { auto MatchOutOfRange = [OpSizeInBits](ConstantSDNode *LHS, ConstantSDNode *RHS) { APInt c1 = LHS->getAPIntValue(); APInt c2 = RHS->getAPIntValue(); zeroExtendToMatch(c1, c2, 1 /* Overflow Bit */); return (c1 + c2).uge(OpSizeInBits); }; if (ISD::matchBinaryPredicate(N1, N0.getOperand(1), MatchOutOfRange)) return DAG.getConstant(0, SDLoc(N), VT); auto MatchInRange = [OpSizeInBits](ConstantSDNode *LHS, ConstantSDNode *RHS) { APInt c1 = LHS->getAPIntValue(); APInt c2 = RHS->getAPIntValue(); zeroExtendToMatch(c1, c2, 1 /* Overflow Bit */); return (c1 + c2).ult(OpSizeInBits); }; if (ISD::matchBinaryPredicate(N1, N0.getOperand(1), MatchInRange)) { SDLoc DL(N); EVT ShiftVT = N1.getValueType(); SDValue Sum = DAG.getNode(ISD::ADD, DL, ShiftVT, N1, N0.getOperand(1)); return DAG.getNode(ISD::SRL, DL, VT, N0.getOperand(0), Sum); } } // fold (srl (trunc (srl x, c1)), c2) -> 0 or (trunc (srl x, (add c1, c2))) if (N1C && N0.getOpcode() == ISD::TRUNCATE && N0.getOperand(0).getOpcode() == ISD::SRL) { if (auto N001C = isConstOrConstSplat(N0.getOperand(0).getOperand(1))) { uint64_t c1 = N001C->getZExtValue(); uint64_t c2 = N1C->getZExtValue(); EVT InnerShiftVT = N0.getOperand(0).getValueType(); EVT ShiftCountVT = N0.getOperand(0).getOperand(1).getValueType(); uint64_t InnerShiftSize = InnerShiftVT.getScalarSizeInBits(); // This is only valid if the OpSizeInBits + c1 = size of inner shift. if (c1 + OpSizeInBits == InnerShiftSize) { SDLoc DL(N0); if (c1 + c2 >= InnerShiftSize) return DAG.getConstant(0, DL, VT); return DAG.getNode(ISD::TRUNCATE, DL, VT, DAG.getNode(ISD::SRL, DL, InnerShiftVT, N0.getOperand(0).getOperand(0), DAG.getConstant(c1 + c2, DL, ShiftCountVT))); } } } // fold (srl (shl x, c), c) -> (and x, cst2) if (N0.getOpcode() == ISD::SHL && N0.getOperand(1) == N1 && isConstantOrConstantVector(N1, /* NoOpaques */ true)) { SDLoc DL(N); SDValue Mask = DAG.getNode(ISD::SRL, DL, VT, DAG.getAllOnesConstant(DL, VT), N1); AddToWorklist(Mask.getNode()); return DAG.getNode(ISD::AND, DL, VT, N0.getOperand(0), Mask); } // fold (srl (anyextend x), c) -> (and (anyextend (srl x, c)), mask) if (N1C && N0.getOpcode() == ISD::ANY_EXTEND) { // Shifting in all undef bits? EVT SmallVT = N0.getOperand(0).getValueType(); unsigned BitSize = SmallVT.getScalarSizeInBits(); if (N1C->getZExtValue() >= BitSize) return DAG.getUNDEF(VT); if (!LegalTypes || TLI.isTypeDesirableForOp(ISD::SRL, SmallVT)) { uint64_t ShiftAmt = N1C->getZExtValue(); SDLoc DL0(N0); SDValue SmallShift = DAG.getNode(ISD::SRL, DL0, SmallVT, N0.getOperand(0), DAG.getConstant(ShiftAmt, DL0, getShiftAmountTy(SmallVT))); AddToWorklist(SmallShift.getNode()); APInt Mask = APInt::getLowBitsSet(OpSizeInBits, OpSizeInBits - ShiftAmt); SDLoc DL(N); return DAG.getNode(ISD::AND, DL, VT, DAG.getNode(ISD::ANY_EXTEND, DL, VT, SmallShift), DAG.getConstant(Mask, DL, VT)); } } // fold (srl (sra X, Y), 31) -> (srl X, 31). This srl only looks at the sign // bit, which is unmodified by sra. if (N1C && N1C->getZExtValue() + 1 == OpSizeInBits) { if (N0.getOpcode() == ISD::SRA) return DAG.getNode(ISD::SRL, SDLoc(N), VT, N0.getOperand(0), N1); } // fold (srl (ctlz x), "5") -> x iff x has one bit set (the low bit). if (N1C && N0.getOpcode() == ISD::CTLZ && N1C->getAPIntValue() == Log2_32(OpSizeInBits)) { KnownBits Known = DAG.computeKnownBits(N0.getOperand(0)); // If any of the input bits are KnownOne, then the input couldn't be all // zeros, thus the result of the srl will always be zero. if (Known.One.getBoolValue()) return DAG.getConstant(0, SDLoc(N0), VT); // If all of the bits input the to ctlz node are known to be zero, then // the result of the ctlz is "32" and the result of the shift is one. APInt UnknownBits = ~Known.Zero; if (UnknownBits == 0) return DAG.getConstant(1, SDLoc(N0), VT); // Otherwise, check to see if there is exactly one bit input to the ctlz. if (UnknownBits.isPowerOf2()) { // Okay, we know that only that the single bit specified by UnknownBits // could be set on input to the CTLZ node. If this bit is set, the SRL // will return 0, if it is clear, it returns 1. Change the CTLZ/SRL pair // to an SRL/XOR pair, which is likely to simplify more. unsigned ShAmt = UnknownBits.countTrailingZeros(); SDValue Op = N0.getOperand(0); if (ShAmt) { SDLoc DL(N0); Op = DAG.getNode(ISD::SRL, DL, VT, Op, DAG.getConstant(ShAmt, DL, getShiftAmountTy(Op.getValueType()))); AddToWorklist(Op.getNode()); } SDLoc DL(N); return DAG.getNode(ISD::XOR, DL, VT, Op, DAG.getConstant(1, DL, VT)); } } // fold (srl x, (trunc (and y, c))) -> (srl x, (and (trunc y), (trunc c))). if (N1.getOpcode() == ISD::TRUNCATE && N1.getOperand(0).getOpcode() == ISD::AND) { if (SDValue NewOp1 = distributeTruncateThroughAnd(N1.getNode())) return DAG.getNode(ISD::SRL, SDLoc(N), VT, N0, NewOp1); } // fold operands of srl based on knowledge that the low bits are not // demanded. if (N1C && SimplifyDemandedBits(SDValue(N, 0))) return SDValue(N, 0); if (N1C && !N1C->isOpaque()) if (SDValue NewSRL = visitShiftByConstant(N, N1C)) return NewSRL; // Attempt to convert a srl of a load into a narrower zero-extending load. if (SDValue NarrowLoad = ReduceLoadWidth(N)) return NarrowLoad; // Here is a common situation. We want to optimize: // // %a = ... // %b = and i32 %a, 2 // %c = srl i32 %b, 1 // brcond i32 %c ... // // into // // %a = ... // %b = and %a, 2 // %c = setcc eq %b, 0 // brcond %c ... // // However when after the source operand of SRL is optimized into AND, the SRL // itself may not be optimized further. Look for it and add the BRCOND into // the worklist. if (N->hasOneUse()) { SDNode *Use = *N->use_begin(); if (Use->getOpcode() == ISD::BRCOND) AddToWorklist(Use); else if (Use->getOpcode() == ISD::TRUNCATE && Use->hasOneUse()) { // Also look pass the truncate. Use = *Use->use_begin(); if (Use->getOpcode() == ISD::BRCOND) AddToWorklist(Use); } } return SDValue(); } SDValue DAGCombiner::visitFunnelShift(SDNode *N) { EVT VT = N->getValueType(0); SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); SDValue N2 = N->getOperand(2); bool IsFSHL = N->getOpcode() == ISD::FSHL; unsigned BitWidth = VT.getScalarSizeInBits(); // fold (fshl N0, N1, 0) -> N0 // fold (fshr N0, N1, 0) -> N1 if (isPowerOf2_32(BitWidth)) if (DAG.MaskedValueIsZero( N2, APInt(N2.getScalarValueSizeInBits(), BitWidth - 1))) return IsFSHL ? N0 : N1; // fold (fsh* N0, N1, c) -> (fsh* N0, N1, c % BitWidth) if (ConstantSDNode *Cst = isConstOrConstSplat(N2)) { if (Cst->getAPIntValue().uge(BitWidth)) { uint64_t RotAmt = Cst->getAPIntValue().urem(BitWidth); return DAG.getNode(N->getOpcode(), SDLoc(N), VT, N0, N1, DAG.getConstant(RotAmt, SDLoc(N), N2.getValueType())); } } // fold (fshl N0, N0, N2) -> (rotl N0, N2) // fold (fshr N0, N0, N2) -> (rotr N0, N2) // TODO: Investigate flipping this rotate if only one is legal, if funnel shift // is legal as well we might be better off avoiding non-constant (BW - N2). unsigned RotOpc = IsFSHL ? ISD::ROTL : ISD::ROTR; if (N0 == N1 && hasOperation(RotOpc, VT)) return DAG.getNode(RotOpc, SDLoc(N), VT, N0, N2); return SDValue(); } SDValue DAGCombiner::visitABS(SDNode *N) { SDValue N0 = N->getOperand(0); EVT VT = N->getValueType(0); // fold (abs c1) -> c2 if (DAG.isConstantIntBuildVectorOrConstantInt(N0)) return DAG.getNode(ISD::ABS, SDLoc(N), VT, N0); // fold (abs (abs x)) -> (abs x) if (N0.getOpcode() == ISD::ABS) return N0; // fold (abs x) -> x iff not-negative if (DAG.SignBitIsZero(N0)) return N0; return SDValue(); } SDValue DAGCombiner::visitBSWAP(SDNode *N) { SDValue N0 = N->getOperand(0); EVT VT = N->getValueType(0); // fold (bswap c1) -> c2 if (DAG.isConstantIntBuildVectorOrConstantInt(N0)) return DAG.getNode(ISD::BSWAP, SDLoc(N), VT, N0); // fold (bswap (bswap x)) -> x if (N0.getOpcode() == ISD::BSWAP) return N0->getOperand(0); return SDValue(); } SDValue DAGCombiner::visitBITREVERSE(SDNode *N) { SDValue N0 = N->getOperand(0); EVT VT = N->getValueType(0); // fold (bitreverse c1) -> c2 if (DAG.isConstantIntBuildVectorOrConstantInt(N0)) return DAG.getNode(ISD::BITREVERSE, SDLoc(N), VT, N0); // fold (bitreverse (bitreverse x)) -> x if (N0.getOpcode() == ISD::BITREVERSE) return N0.getOperand(0); return SDValue(); } SDValue DAGCombiner::visitCTLZ(SDNode *N) { SDValue N0 = N->getOperand(0); EVT VT = N->getValueType(0); // fold (ctlz c1) -> c2 if (DAG.isConstantIntBuildVectorOrConstantInt(N0)) return DAG.getNode(ISD::CTLZ, SDLoc(N), VT, N0); // If the value is known never to be zero, switch to the undef version. if (!LegalOperations || TLI.isOperationLegal(ISD::CTLZ_ZERO_UNDEF, VT)) { if (DAG.isKnownNeverZero(N0)) return DAG.getNode(ISD::CTLZ_ZERO_UNDEF, SDLoc(N), VT, N0); } return SDValue(); } SDValue DAGCombiner::visitCTLZ_ZERO_UNDEF(SDNode *N) { SDValue N0 = N->getOperand(0); EVT VT = N->getValueType(0); // fold (ctlz_zero_undef c1) -> c2 if (DAG.isConstantIntBuildVectorOrConstantInt(N0)) return DAG.getNode(ISD::CTLZ_ZERO_UNDEF, SDLoc(N), VT, N0); return SDValue(); } SDValue DAGCombiner::visitCTTZ(SDNode *N) { SDValue N0 = N->getOperand(0); EVT VT = N->getValueType(0); // fold (cttz c1) -> c2 if (DAG.isConstantIntBuildVectorOrConstantInt(N0)) return DAG.getNode(ISD::CTTZ, SDLoc(N), VT, N0); // If the value is known never to be zero, switch to the undef version. if (!LegalOperations || TLI.isOperationLegal(ISD::CTTZ_ZERO_UNDEF, VT)) { if (DAG.isKnownNeverZero(N0)) return DAG.getNode(ISD::CTTZ_ZERO_UNDEF, SDLoc(N), VT, N0); } return SDValue(); } SDValue DAGCombiner::visitCTTZ_ZERO_UNDEF(SDNode *N) { SDValue N0 = N->getOperand(0); EVT VT = N->getValueType(0); // fold (cttz_zero_undef c1) -> c2 if (DAG.isConstantIntBuildVectorOrConstantInt(N0)) return DAG.getNode(ISD::CTTZ_ZERO_UNDEF, SDLoc(N), VT, N0); return SDValue(); } SDValue DAGCombiner::visitCTPOP(SDNode *N) { SDValue N0 = N->getOperand(0); EVT VT = N->getValueType(0); // fold (ctpop c1) -> c2 if (DAG.isConstantIntBuildVectorOrConstantInt(N0)) return DAG.getNode(ISD::CTPOP, SDLoc(N), VT, N0); return SDValue(); } // FIXME: This should be checking for no signed zeros on individual operands, as // well as no nans. static bool isLegalToCombineMinNumMaxNum(SelectionDAG &DAG, SDValue LHS, SDValue RHS) { const TargetOptions &Options = DAG.getTarget().Options; EVT VT = LHS.getValueType(); return Options.NoSignedZerosFPMath && VT.isFloatingPoint() && DAG.isKnownNeverNaN(LHS) && DAG.isKnownNeverNaN(RHS); } /// Generate Min/Max node static SDValue combineMinNumMaxNum(const SDLoc &DL, EVT VT, SDValue LHS, SDValue RHS, SDValue True, SDValue False, ISD::CondCode CC, const TargetLowering &TLI, SelectionDAG &DAG) { if (!(LHS == True && RHS == False) && !(LHS == False && RHS == True)) return SDValue(); EVT TransformVT = TLI.getTypeToTransformTo(*DAG.getContext(), VT); switch (CC) { case ISD::SETOLT: case ISD::SETOLE: case ISD::SETLT: case ISD::SETLE: case ISD::SETULT: case ISD::SETULE: { // Since it's known never nan to get here already, either fminnum or // fminnum_ieee are OK. Try the ieee version first, since it's fminnum is // expanded in terms of it. unsigned IEEEOpcode = (LHS == True) ? ISD::FMINNUM_IEEE : ISD::FMAXNUM_IEEE; if (TLI.isOperationLegalOrCustom(IEEEOpcode, VT)) return DAG.getNode(IEEEOpcode, DL, VT, LHS, RHS); unsigned Opcode = (LHS == True) ? ISD::FMINNUM : ISD::FMAXNUM; if (TLI.isOperationLegalOrCustom(Opcode, TransformVT)) return DAG.getNode(Opcode, DL, VT, LHS, RHS); return SDValue(); } case ISD::SETOGT: case ISD::SETOGE: case ISD::SETGT: case ISD::SETGE: case ISD::SETUGT: case ISD::SETUGE: { unsigned IEEEOpcode = (LHS == True) ? ISD::FMAXNUM_IEEE : ISD::FMINNUM_IEEE; if (TLI.isOperationLegalOrCustom(IEEEOpcode, VT)) return DAG.getNode(IEEEOpcode, DL, VT, LHS, RHS); unsigned Opcode = (LHS == True) ? ISD::FMAXNUM : ISD::FMINNUM; if (TLI.isOperationLegalOrCustom(Opcode, TransformVT)) return DAG.getNode(Opcode, DL, VT, LHS, RHS); return SDValue(); } default: return SDValue(); } } SDValue DAGCombiner::foldSelectOfConstants(SDNode *N) { SDValue Cond = N->getOperand(0); SDValue N1 = N->getOperand(1); SDValue N2 = N->getOperand(2); EVT VT = N->getValueType(0); EVT CondVT = Cond.getValueType(); SDLoc DL(N); if (!VT.isInteger()) return SDValue(); auto *C1 = dyn_cast(N1); auto *C2 = dyn_cast(N2); if (!C1 || !C2) return SDValue(); // Only do this before legalization to avoid conflicting with target-specific // transforms in the other direction (create a select from a zext/sext). There // is also a target-independent combine here in DAGCombiner in the other // direction for (select Cond, -1, 0) when the condition is not i1. if (CondVT == MVT::i1 && !LegalOperations) { if (C1->isNullValue() && C2->isOne()) { // select Cond, 0, 1 --> zext (!Cond) SDValue NotCond = DAG.getNOT(DL, Cond, MVT::i1); if (VT != MVT::i1) NotCond = DAG.getNode(ISD::ZERO_EXTEND, DL, VT, NotCond); return NotCond; } if (C1->isNullValue() && C2->isAllOnesValue()) { // select Cond, 0, -1 --> sext (!Cond) SDValue NotCond = DAG.getNOT(DL, Cond, MVT::i1); if (VT != MVT::i1) NotCond = DAG.getNode(ISD::SIGN_EXTEND, DL, VT, NotCond); return NotCond; } if (C1->isOne() && C2->isNullValue()) { // select Cond, 1, 0 --> zext (Cond) if (VT != MVT::i1) Cond = DAG.getNode(ISD::ZERO_EXTEND, DL, VT, Cond); return Cond; } if (C1->isAllOnesValue() && C2->isNullValue()) { // select Cond, -1, 0 --> sext (Cond) if (VT != MVT::i1) Cond = DAG.getNode(ISD::SIGN_EXTEND, DL, VT, Cond); return Cond; } // For any constants that differ by 1, we can transform the select into an // extend and add. Use a target hook because some targets may prefer to // transform in the other direction. if (TLI.convertSelectOfConstantsToMath(VT)) { if (C1->getAPIntValue() - 1 == C2->getAPIntValue()) { // select Cond, C1, C1-1 --> add (zext Cond), C1-1 if (VT != MVT::i1) Cond = DAG.getNode(ISD::ZERO_EXTEND, DL, VT, Cond); return DAG.getNode(ISD::ADD, DL, VT, Cond, N2); } if (C1->getAPIntValue() + 1 == C2->getAPIntValue()) { // select Cond, C1, C1+1 --> add (sext Cond), C1+1 if (VT != MVT::i1) Cond = DAG.getNode(ISD::SIGN_EXTEND, DL, VT, Cond); return DAG.getNode(ISD::ADD, DL, VT, Cond, N2); } } return SDValue(); } // fold (select Cond, 0, 1) -> (xor Cond, 1) // We can't do this reliably if integer based booleans have different contents // to floating point based booleans. This is because we can't tell whether we // have an integer-based boolean or a floating-point-based boolean unless we // can find the SETCC that produced it and inspect its operands. This is // fairly easy if C is the SETCC node, but it can potentially be // undiscoverable (or not reasonably discoverable). For example, it could be // in another basic block or it could require searching a complicated // expression. if (CondVT.isInteger() && TLI.getBooleanContents(/*isVec*/false, /*isFloat*/true) == TargetLowering::ZeroOrOneBooleanContent && TLI.getBooleanContents(/*isVec*/false, /*isFloat*/false) == TargetLowering::ZeroOrOneBooleanContent && C1->isNullValue() && C2->isOne()) { SDValue NotCond = DAG.getNode(ISD::XOR, DL, CondVT, Cond, DAG.getConstant(1, DL, CondVT)); if (VT.bitsEq(CondVT)) return NotCond; return DAG.getZExtOrTrunc(NotCond, DL, VT); } return SDValue(); } SDValue DAGCombiner::visitSELECT(SDNode *N) { SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); SDValue N2 = N->getOperand(2); EVT VT = N->getValueType(0); EVT VT0 = N0.getValueType(); SDLoc DL(N); if (SDValue V = DAG.simplifySelect(N0, N1, N2)) return V; // fold (select X, X, Y) -> (or X, Y) // fold (select X, 1, Y) -> (or C, Y) if (VT == VT0 && VT == MVT::i1 && (N0 == N1 || isOneConstant(N1))) return DAG.getNode(ISD::OR, DL, VT, N0, N2); if (SDValue V = foldSelectOfConstants(N)) return V; // fold (select C, 0, X) -> (and (not C), X) if (VT == VT0 && VT == MVT::i1 && isNullConstant(N1)) { SDValue NOTNode = DAG.getNOT(SDLoc(N0), N0, VT); AddToWorklist(NOTNode.getNode()); return DAG.getNode(ISD::AND, DL, VT, NOTNode, N2); } // fold (select C, X, 1) -> (or (not C), X) if (VT == VT0 && VT == MVT::i1 && isOneConstant(N2)) { SDValue NOTNode = DAG.getNOT(SDLoc(N0), N0, VT); AddToWorklist(NOTNode.getNode()); return DAG.getNode(ISD::OR, DL, VT, NOTNode, N1); } // fold (select X, Y, X) -> (and X, Y) // fold (select X, Y, 0) -> (and X, Y) if (VT == VT0 && VT == MVT::i1 && (N0 == N2 || isNullConstant(N2))) return DAG.getNode(ISD::AND, DL, VT, N0, N1); // If we can fold this based on the true/false value, do so. if (SimplifySelectOps(N, N1, N2)) return SDValue(N, 0); // Don't revisit N. if (VT0 == MVT::i1) { // The code in this block deals with the following 2 equivalences: // select(C0|C1, x, y) <=> select(C0, x, select(C1, x, y)) // select(C0&C1, x, y) <=> select(C0, select(C1, x, y), y) // The target can specify its preferred form with the // shouldNormalizeToSelectSequence() callback. However we always transform // to the right anyway if we find the inner select exists in the DAG anyway // and we always transform to the left side if we know that we can further // optimize the combination of the conditions. bool normalizeToSequence = TLI.shouldNormalizeToSelectSequence(*DAG.getContext(), VT); // select (and Cond0, Cond1), X, Y // -> select Cond0, (select Cond1, X, Y), Y if (N0->getOpcode() == ISD::AND && N0->hasOneUse()) { SDValue Cond0 = N0->getOperand(0); SDValue Cond1 = N0->getOperand(1); SDValue InnerSelect = DAG.getNode(ISD::SELECT, DL, N1.getValueType(), Cond1, N1, N2); if (normalizeToSequence || !InnerSelect.use_empty()) return DAG.getNode(ISD::SELECT, DL, N1.getValueType(), Cond0, InnerSelect, N2); } // select (or Cond0, Cond1), X, Y -> select Cond0, X, (select Cond1, X, Y) if (N0->getOpcode() == ISD::OR && N0->hasOneUse()) { SDValue Cond0 = N0->getOperand(0); SDValue Cond1 = N0->getOperand(1); SDValue InnerSelect = DAG.getNode(ISD::SELECT, DL, N1.getValueType(), Cond1, N1, N2); if (normalizeToSequence || !InnerSelect.use_empty()) return DAG.getNode(ISD::SELECT, DL, N1.getValueType(), Cond0, N1, InnerSelect); } // select Cond0, (select Cond1, X, Y), Y -> select (and Cond0, Cond1), X, Y if (N1->getOpcode() == ISD::SELECT && N1->hasOneUse()) { SDValue N1_0 = N1->getOperand(0); SDValue N1_1 = N1->getOperand(1); SDValue N1_2 = N1->getOperand(2); if (N1_2 == N2 && N0.getValueType() == N1_0.getValueType()) { // Create the actual and node if we can generate good code for it. if (!normalizeToSequence) { SDValue And = DAG.getNode(ISD::AND, DL, N0.getValueType(), N0, N1_0); return DAG.getNode(ISD::SELECT, DL, N1.getValueType(), And, N1_1, N2); } // Otherwise see if we can optimize the "and" to a better pattern. if (SDValue Combined = visitANDLike(N0, N1_0, N)) return DAG.getNode(ISD::SELECT, DL, N1.getValueType(), Combined, N1_1, N2); } } // select Cond0, X, (select Cond1, X, Y) -> select (or Cond0, Cond1), X, Y if (N2->getOpcode() == ISD::SELECT && N2->hasOneUse()) { SDValue N2_0 = N2->getOperand(0); SDValue N2_1 = N2->getOperand(1); SDValue N2_2 = N2->getOperand(2); if (N2_1 == N1 && N0.getValueType() == N2_0.getValueType()) { // Create the actual or node if we can generate good code for it. if (!normalizeToSequence) { SDValue Or = DAG.getNode(ISD::OR, DL, N0.getValueType(), N0, N2_0); return DAG.getNode(ISD::SELECT, DL, N1.getValueType(), Or, N1, N2_2); } // Otherwise see if we can optimize to a better pattern. if (SDValue Combined = visitORLike(N0, N2_0, N)) return DAG.getNode(ISD::SELECT, DL, N1.getValueType(), Combined, N1, N2_2); } } } if (VT0 == MVT::i1) { // select (not Cond), N1, N2 -> select Cond, N2, N1 if (isBitwiseNot(N0)) return DAG.getNode(ISD::SELECT, DL, VT, N0->getOperand(0), N2, N1); } // Fold selects based on a setcc into other things, such as min/max/abs. if (N0.getOpcode() == ISD::SETCC) { SDValue Cond0 = N0.getOperand(0), Cond1 = N0.getOperand(1); ISD::CondCode CC = cast(N0.getOperand(2))->get(); // select (fcmp lt x, y), x, y -> fminnum x, y // select (fcmp gt x, y), x, y -> fmaxnum x, y // // This is OK if we don't care what happens if either operand is a NaN. if (N0.hasOneUse() && isLegalToCombineMinNumMaxNum(DAG, N1, N2)) if (SDValue FMinMax = combineMinNumMaxNum(DL, VT, Cond0, Cond1, N1, N2, CC, TLI, DAG)) return FMinMax; // Use 'unsigned add with overflow' to optimize an unsigned saturating add. // This is conservatively limited to pre-legal-operations to give targets // a chance to reverse the transform if they want to do that. Also, it is // unlikely that the pattern would be formed late, so it's probably not // worth going through the other checks. if (!LegalOperations && TLI.isOperationLegalOrCustom(ISD::UADDO, VT) && CC == ISD::SETUGT && N0.hasOneUse() && isAllOnesConstant(N1) && N2.getOpcode() == ISD::ADD && Cond0 == N2.getOperand(0)) { auto *C = dyn_cast(N2.getOperand(1)); auto *NotC = dyn_cast(Cond1); if (C && NotC && C->getAPIntValue() == ~NotC->getAPIntValue()) { // select (setcc Cond0, ~C, ugt), -1, (add Cond0, C) --> // uaddo Cond0, C; select uaddo.1, -1, uaddo.0 // // The IR equivalent of this transform would have this form: // %a = add %x, C // %c = icmp ugt %x, ~C // %r = select %c, -1, %a // => // %u = call {iN,i1} llvm.uadd.with.overflow(%x, C) // %u0 = extractvalue %u, 0 // %u1 = extractvalue %u, 1 // %r = select %u1, -1, %u0 SDVTList VTs = DAG.getVTList(VT, VT0); SDValue UAO = DAG.getNode(ISD::UADDO, DL, VTs, Cond0, N2.getOperand(1)); return DAG.getSelect(DL, VT, UAO.getValue(1), N1, UAO.getValue(0)); } } if (TLI.isOperationLegal(ISD::SELECT_CC, VT) || (!LegalOperations && TLI.isOperationLegalOrCustom(ISD::SELECT_CC, VT))) return DAG.getNode(ISD::SELECT_CC, DL, VT, Cond0, Cond1, N1, N2, N0.getOperand(2)); return SimplifySelect(DL, N0, N1, N2); } return SDValue(); } static std::pair SplitVSETCC(const SDNode *N, SelectionDAG &DAG) { SDLoc DL(N); EVT LoVT, HiVT; std::tie(LoVT, HiVT) = DAG.GetSplitDestVTs(N->getValueType(0)); // Split the inputs. SDValue Lo, Hi, LL, LH, RL, RH; std::tie(LL, LH) = DAG.SplitVectorOperand(N, 0); std::tie(RL, RH) = DAG.SplitVectorOperand(N, 1); Lo = DAG.getNode(N->getOpcode(), DL, LoVT, LL, RL, N->getOperand(2)); Hi = DAG.getNode(N->getOpcode(), DL, HiVT, LH, RH, N->getOperand(2)); return std::make_pair(Lo, Hi); } // This function assumes all the vselect's arguments are CONCAT_VECTOR // nodes and that the condition is a BV of ConstantSDNodes (or undefs). static SDValue ConvertSelectToConcatVector(SDNode *N, SelectionDAG &DAG) { SDLoc DL(N); SDValue Cond = N->getOperand(0); SDValue LHS = N->getOperand(1); SDValue RHS = N->getOperand(2); EVT VT = N->getValueType(0); int NumElems = VT.getVectorNumElements(); assert(LHS.getOpcode() == ISD::CONCAT_VECTORS && RHS.getOpcode() == ISD::CONCAT_VECTORS && Cond.getOpcode() == ISD::BUILD_VECTOR); // CONCAT_VECTOR can take an arbitrary number of arguments. We only care about // binary ones here. if (LHS->getNumOperands() != 2 || RHS->getNumOperands() != 2) return SDValue(); // We're sure we have an even number of elements due to the // concat_vectors we have as arguments to vselect. // Skip BV elements until we find one that's not an UNDEF // After we find an UNDEF element, keep looping until we get to half the // length of the BV and see if all the non-undef nodes are the same. ConstantSDNode *BottomHalf = nullptr; for (int i = 0; i < NumElems / 2; ++i) { if (Cond->getOperand(i)->isUndef()) continue; if (BottomHalf == nullptr) BottomHalf = cast(Cond.getOperand(i)); else if (Cond->getOperand(i).getNode() != BottomHalf) return SDValue(); } // Do the same for the second half of the BuildVector ConstantSDNode *TopHalf = nullptr; for (int i = NumElems / 2; i < NumElems; ++i) { if (Cond->getOperand(i)->isUndef()) continue; if (TopHalf == nullptr) TopHalf = cast(Cond.getOperand(i)); else if (Cond->getOperand(i).getNode() != TopHalf) return SDValue(); } assert(TopHalf && BottomHalf && "One half of the selector was all UNDEFs and the other was all the " "same value. This should have been addressed before this function."); return DAG.getNode( ISD::CONCAT_VECTORS, DL, VT, BottomHalf->isNullValue() ? RHS->getOperand(0) : LHS->getOperand(0), TopHalf->isNullValue() ? RHS->getOperand(1) : LHS->getOperand(1)); } SDValue DAGCombiner::visitMSCATTER(SDNode *N) { if (Level >= AfterLegalizeTypes) return SDValue(); MaskedScatterSDNode *MSC = cast(N); SDValue Mask = MSC->getMask(); SDValue Data = MSC->getValue(); SDLoc DL(N); // If the MSCATTER data type requires splitting and the mask is provided by a // SETCC, then split both nodes and its operands before legalization. This // prevents the type legalizer from unrolling SETCC into scalar comparisons // and enables future optimizations (e.g. min/max pattern matching on X86). if (Mask.getOpcode() != ISD::SETCC) return SDValue(); // Check if any splitting is required. if (TLI.getTypeAction(*DAG.getContext(), Data.getValueType()) != TargetLowering::TypeSplitVector) return SDValue(); SDValue MaskLo, MaskHi; std::tie(MaskLo, MaskHi) = SplitVSETCC(Mask.getNode(), DAG); EVT LoVT, HiVT; std::tie(LoVT, HiVT) = DAG.GetSplitDestVTs(MSC->getValueType(0)); SDValue Chain = MSC->getChain(); EVT MemoryVT = MSC->getMemoryVT(); unsigned Alignment = MSC->getOriginalAlignment(); EVT LoMemVT, HiMemVT; std::tie(LoMemVT, HiMemVT) = DAG.GetSplitDestVTs(MemoryVT); SDValue DataLo, DataHi; std::tie(DataLo, DataHi) = DAG.SplitVector(Data, DL); SDValue Scale = MSC->getScale(); SDValue BasePtr = MSC->getBasePtr(); SDValue IndexLo, IndexHi; std::tie(IndexLo, IndexHi) = DAG.SplitVector(MSC->getIndex(), DL); MachineMemOperand *MMO = DAG.getMachineFunction(). getMachineMemOperand(MSC->getPointerInfo(), MachineMemOperand::MOStore, LoMemVT.getStoreSize(), Alignment, MSC->getAAInfo(), MSC->getRanges()); SDValue OpsLo[] = { Chain, DataLo, MaskLo, BasePtr, IndexLo, Scale }; SDValue Lo = DAG.getMaskedScatter(DAG.getVTList(MVT::Other), DataLo.getValueType(), DL, OpsLo, MMO); // The order of the Scatter operation after split is well defined. The "Hi" // part comes after the "Lo". So these two operations should be chained one // after another. SDValue OpsHi[] = { Lo, DataHi, MaskHi, BasePtr, IndexHi, Scale }; return DAG.getMaskedScatter(DAG.getVTList(MVT::Other), DataHi.getValueType(), DL, OpsHi, MMO); } SDValue DAGCombiner::visitMSTORE(SDNode *N) { if (Level >= AfterLegalizeTypes) return SDValue(); MaskedStoreSDNode *MST = dyn_cast(N); SDValue Mask = MST->getMask(); SDValue Data = MST->getValue(); EVT VT = Data.getValueType(); SDLoc DL(N); // If the MSTORE data type requires splitting and the mask is provided by a // SETCC, then split both nodes and its operands before legalization. This // prevents the type legalizer from unrolling SETCC into scalar comparisons // and enables future optimizations (e.g. min/max pattern matching on X86). if (Mask.getOpcode() == ISD::SETCC) { // Check if any splitting is required. if (TLI.getTypeAction(*DAG.getContext(), VT) != TargetLowering::TypeSplitVector) return SDValue(); SDValue MaskLo, MaskHi, Lo, Hi; std::tie(MaskLo, MaskHi) = SplitVSETCC(Mask.getNode(), DAG); SDValue Chain = MST->getChain(); SDValue Ptr = MST->getBasePtr(); EVT MemoryVT = MST->getMemoryVT(); unsigned Alignment = MST->getOriginalAlignment(); // if Alignment is equal to the vector size, // take the half of it for the second part unsigned SecondHalfAlignment = (Alignment == VT.getSizeInBits() / 8) ? Alignment / 2 : Alignment; EVT LoMemVT, HiMemVT; std::tie(LoMemVT, HiMemVT) = DAG.GetSplitDestVTs(MemoryVT); SDValue DataLo, DataHi; std::tie(DataLo, DataHi) = DAG.SplitVector(Data, DL); MachineMemOperand *MMO = DAG.getMachineFunction(). getMachineMemOperand(MST->getPointerInfo(), MachineMemOperand::MOStore, LoMemVT.getStoreSize(), Alignment, MST->getAAInfo(), MST->getRanges()); Lo = DAG.getMaskedStore(Chain, DL, DataLo, Ptr, MaskLo, LoMemVT, MMO, MST->isTruncatingStore(), MST->isCompressingStore()); Ptr = TLI.IncrementMemoryAddress(Ptr, MaskLo, DL, LoMemVT, DAG, MST->isCompressingStore()); unsigned HiOffset = LoMemVT.getStoreSize(); MMO = DAG.getMachineFunction().getMachineMemOperand( MST->getPointerInfo().getWithOffset(HiOffset), MachineMemOperand::MOStore, HiMemVT.getStoreSize(), SecondHalfAlignment, MST->getAAInfo(), MST->getRanges()); Hi = DAG.getMaskedStore(Chain, DL, DataHi, Ptr, MaskHi, HiMemVT, MMO, MST->isTruncatingStore(), MST->isCompressingStore()); AddToWorklist(Lo.getNode()); AddToWorklist(Hi.getNode()); return DAG.getNode(ISD::TokenFactor, DL, MVT::Other, Lo, Hi); } return SDValue(); } SDValue DAGCombiner::visitMGATHER(SDNode *N) { if (Level >= AfterLegalizeTypes) return SDValue(); MaskedGatherSDNode *MGT = cast(N); SDValue Mask = MGT->getMask(); SDLoc DL(N); // If the MGATHER result requires splitting and the mask is provided by a // SETCC, then split both nodes and its operands before legalization. This // prevents the type legalizer from unrolling SETCC into scalar comparisons // and enables future optimizations (e.g. min/max pattern matching on X86). if (Mask.getOpcode() != ISD::SETCC) return SDValue(); EVT VT = N->getValueType(0); // Check if any splitting is required. if (TLI.getTypeAction(*DAG.getContext(), VT) != TargetLowering::TypeSplitVector) return SDValue(); SDValue MaskLo, MaskHi, Lo, Hi; std::tie(MaskLo, MaskHi) = SplitVSETCC(Mask.getNode(), DAG); SDValue PassThru = MGT->getPassThru(); SDValue PassThruLo, PassThruHi; std::tie(PassThruLo, PassThruHi) = DAG.SplitVector(PassThru, DL); EVT LoVT, HiVT; std::tie(LoVT, HiVT) = DAG.GetSplitDestVTs(VT); SDValue Chain = MGT->getChain(); EVT MemoryVT = MGT->getMemoryVT(); unsigned Alignment = MGT->getOriginalAlignment(); EVT LoMemVT, HiMemVT; std::tie(LoMemVT, HiMemVT) = DAG.GetSplitDestVTs(MemoryVT); SDValue Scale = MGT->getScale(); SDValue BasePtr = MGT->getBasePtr(); SDValue Index = MGT->getIndex(); SDValue IndexLo, IndexHi; std::tie(IndexLo, IndexHi) = DAG.SplitVector(Index, DL); MachineMemOperand *MMO = DAG.getMachineFunction(). getMachineMemOperand(MGT->getPointerInfo(), MachineMemOperand::MOLoad, LoMemVT.getStoreSize(), Alignment, MGT->getAAInfo(), MGT->getRanges()); SDValue OpsLo[] = { Chain, PassThruLo, MaskLo, BasePtr, IndexLo, Scale }; Lo = DAG.getMaskedGather(DAG.getVTList(LoVT, MVT::Other), LoVT, DL, OpsLo, MMO); SDValue OpsHi[] = { Chain, PassThruHi, MaskHi, BasePtr, IndexHi, Scale }; Hi = DAG.getMaskedGather(DAG.getVTList(HiVT, MVT::Other), HiVT, DL, OpsHi, MMO); AddToWorklist(Lo.getNode()); AddToWorklist(Hi.getNode()); // Build a factor node to remember that this load is independent of the // other one. Chain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, Lo.getValue(1), Hi.getValue(1)); // Legalized the chain result - switch anything that used the old chain to // use the new one. DAG.ReplaceAllUsesOfValueWith(SDValue(MGT, 1), Chain); SDValue GatherRes = DAG.getNode(ISD::CONCAT_VECTORS, DL, VT, Lo, Hi); SDValue RetOps[] = { GatherRes, Chain }; return DAG.getMergeValues(RetOps, DL); } SDValue DAGCombiner::visitMLOAD(SDNode *N) { if (Level >= AfterLegalizeTypes) return SDValue(); MaskedLoadSDNode *MLD = dyn_cast(N); SDValue Mask = MLD->getMask(); SDLoc DL(N); // If the MLOAD result requires splitting and the mask is provided by a // SETCC, then split both nodes and its operands before legalization. This // prevents the type legalizer from unrolling SETCC into scalar comparisons // and enables future optimizations (e.g. min/max pattern matching on X86). if (Mask.getOpcode() == ISD::SETCC) { EVT VT = N->getValueType(0); // Check if any splitting is required. if (TLI.getTypeAction(*DAG.getContext(), VT) != TargetLowering::TypeSplitVector) return SDValue(); SDValue MaskLo, MaskHi, Lo, Hi; std::tie(MaskLo, MaskHi) = SplitVSETCC(Mask.getNode(), DAG); SDValue PassThru = MLD->getPassThru(); SDValue PassThruLo, PassThruHi; std::tie(PassThruLo, PassThruHi) = DAG.SplitVector(PassThru, DL); EVT LoVT, HiVT; std::tie(LoVT, HiVT) = DAG.GetSplitDestVTs(MLD->getValueType(0)); SDValue Chain = MLD->getChain(); SDValue Ptr = MLD->getBasePtr(); EVT MemoryVT = MLD->getMemoryVT(); unsigned Alignment = MLD->getOriginalAlignment(); // if Alignment is equal to the vector size, // take the half of it for the second part unsigned SecondHalfAlignment = (Alignment == MLD->getValueType(0).getSizeInBits()/8) ? Alignment/2 : Alignment; EVT LoMemVT, HiMemVT; std::tie(LoMemVT, HiMemVT) = DAG.GetSplitDestVTs(MemoryVT); MachineMemOperand *MMO = DAG.getMachineFunction(). getMachineMemOperand(MLD->getPointerInfo(), MachineMemOperand::MOLoad, LoMemVT.getStoreSize(), Alignment, MLD->getAAInfo(), MLD->getRanges()); Lo = DAG.getMaskedLoad(LoVT, DL, Chain, Ptr, MaskLo, PassThruLo, LoMemVT, MMO, ISD::NON_EXTLOAD, MLD->isExpandingLoad()); Ptr = TLI.IncrementMemoryAddress(Ptr, MaskLo, DL, LoMemVT, DAG, MLD->isExpandingLoad()); unsigned HiOffset = LoMemVT.getStoreSize(); MMO = DAG.getMachineFunction().getMachineMemOperand( MLD->getPointerInfo().getWithOffset(HiOffset), MachineMemOperand::MOLoad, HiMemVT.getStoreSize(), SecondHalfAlignment, MLD->getAAInfo(), MLD->getRanges()); Hi = DAG.getMaskedLoad(HiVT, DL, Chain, Ptr, MaskHi, PassThruHi, HiMemVT, MMO, ISD::NON_EXTLOAD, MLD->isExpandingLoad()); AddToWorklist(Lo.getNode()); AddToWorklist(Hi.getNode()); // Build a factor node to remember that this load is independent of the // other one. Chain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, Lo.getValue(1), Hi.getValue(1)); // Legalized the chain result - switch anything that used the old chain to // use the new one. DAG.ReplaceAllUsesOfValueWith(SDValue(MLD, 1), Chain); SDValue LoadRes = DAG.getNode(ISD::CONCAT_VECTORS, DL, VT, Lo, Hi); SDValue RetOps[] = { LoadRes, Chain }; return DAG.getMergeValues(RetOps, DL); } return SDValue(); } /// A vector select of 2 constant vectors can be simplified to math/logic to /// avoid a variable select instruction and possibly avoid constant loads. SDValue DAGCombiner::foldVSelectOfConstants(SDNode *N) { SDValue Cond = N->getOperand(0); SDValue N1 = N->getOperand(1); SDValue N2 = N->getOperand(2); EVT VT = N->getValueType(0); if (!Cond.hasOneUse() || Cond.getScalarValueSizeInBits() != 1 || !TLI.convertSelectOfConstantsToMath(VT) || !ISD::isBuildVectorOfConstantSDNodes(N1.getNode()) || !ISD::isBuildVectorOfConstantSDNodes(N2.getNode())) return SDValue(); // Check if we can use the condition value to increment/decrement a single // constant value. This simplifies a select to an add and removes a constant // load/materialization from the general case. bool AllAddOne = true; bool AllSubOne = true; unsigned Elts = VT.getVectorNumElements(); for (unsigned i = 0; i != Elts; ++i) { SDValue N1Elt = N1.getOperand(i); SDValue N2Elt = N2.getOperand(i); if (N1Elt.isUndef() || N2Elt.isUndef()) continue; const APInt &C1 = cast(N1Elt)->getAPIntValue(); const APInt &C2 = cast(N2Elt)->getAPIntValue(); if (C1 != C2 + 1) AllAddOne = false; if (C1 != C2 - 1) AllSubOne = false; } // Further simplifications for the extra-special cases where the constants are // all 0 or all -1 should be implemented as folds of these patterns. SDLoc DL(N); if (AllAddOne || AllSubOne) { // vselect Cond, C+1, C --> add (zext Cond), C // vselect Cond, C-1, C --> add (sext Cond), C auto ExtendOpcode = AllAddOne ? ISD::ZERO_EXTEND : ISD::SIGN_EXTEND; SDValue ExtendedCond = DAG.getNode(ExtendOpcode, DL, VT, Cond); return DAG.getNode(ISD::ADD, DL, VT, ExtendedCond, N2); } // The general case for select-of-constants: // vselect Cond, C1, C2 --> xor (and (sext Cond), (C1^C2)), C2 // ...but that only makes sense if a vselect is slower than 2 logic ops, so // leave that to a machine-specific pass. return SDValue(); } SDValue DAGCombiner::visitVSELECT(SDNode *N) { SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); SDValue N2 = N->getOperand(2); SDLoc DL(N); if (SDValue V = DAG.simplifySelect(N0, N1, N2)) return V; // Canonicalize integer abs. // vselect (setg[te] X, 0), X, -X -> // vselect (setgt X, -1), X, -X -> // vselect (setl[te] X, 0), -X, X -> // Y = sra (X, size(X)-1); xor (add (X, Y), Y) if (N0.getOpcode() == ISD::SETCC) { SDValue LHS = N0.getOperand(0), RHS = N0.getOperand(1); ISD::CondCode CC = cast(N0.getOperand(2))->get(); bool isAbs = false; bool RHSIsAllZeros = ISD::isBuildVectorAllZeros(RHS.getNode()); if (((RHSIsAllZeros && (CC == ISD::SETGT || CC == ISD::SETGE)) || (ISD::isBuildVectorAllOnes(RHS.getNode()) && CC == ISD::SETGT)) && N1 == LHS && N2.getOpcode() == ISD::SUB && N1 == N2.getOperand(1)) isAbs = ISD::isBuildVectorAllZeros(N2.getOperand(0).getNode()); else if ((RHSIsAllZeros && (CC == ISD::SETLT || CC == ISD::SETLE)) && N2 == LHS && N1.getOpcode() == ISD::SUB && N2 == N1.getOperand(1)) isAbs = ISD::isBuildVectorAllZeros(N1.getOperand(0).getNode()); if (isAbs) { EVT VT = LHS.getValueType(); if (TLI.isOperationLegalOrCustom(ISD::ABS, VT)) return DAG.getNode(ISD::ABS, DL, VT, LHS); SDValue Shift = DAG.getNode( ISD::SRA, DL, VT, LHS, DAG.getConstant(VT.getScalarSizeInBits() - 1, DL, VT)); SDValue Add = DAG.getNode(ISD::ADD, DL, VT, LHS, Shift); AddToWorklist(Shift.getNode()); AddToWorklist(Add.getNode()); return DAG.getNode(ISD::XOR, DL, VT, Add, Shift); } // vselect x, y (fcmp lt x, y) -> fminnum x, y // vselect x, y (fcmp gt x, y) -> fmaxnum x, y // // This is OK if we don't care about what happens if either operand is a // NaN. // EVT VT = N->getValueType(0); if (N0.hasOneUse() && isLegalToCombineMinNumMaxNum(DAG, N0.getOperand(0), N0.getOperand(1))) { ISD::CondCode CC = cast(N0.getOperand(2))->get(); if (SDValue FMinMax = combineMinNumMaxNum( DL, VT, N0.getOperand(0), N0.getOperand(1), N1, N2, CC, TLI, DAG)) return FMinMax; } // If this select has a condition (setcc) with narrower operands than the // select, try to widen the compare to match the select width. // TODO: This should be extended to handle any constant. // TODO: This could be extended to handle non-loading patterns, but that // requires thorough testing to avoid regressions. if (isNullOrNullSplat(RHS)) { EVT NarrowVT = LHS.getValueType(); EVT WideVT = N1.getValueType().changeVectorElementTypeToInteger(); EVT SetCCVT = getSetCCResultType(LHS.getValueType()); unsigned SetCCWidth = SetCCVT.getScalarSizeInBits(); unsigned WideWidth = WideVT.getScalarSizeInBits(); bool IsSigned = isSignedIntSetCC(CC); auto LoadExtOpcode = IsSigned ? ISD::SEXTLOAD : ISD::ZEXTLOAD; if (LHS.getOpcode() == ISD::LOAD && LHS.hasOneUse() && SetCCWidth != 1 && SetCCWidth < WideWidth && TLI.isLoadExtLegalOrCustom(LoadExtOpcode, WideVT, NarrowVT) && TLI.isOperationLegalOrCustom(ISD::SETCC, WideVT)) { // Both compare operands can be widened for free. The LHS can use an // extended load, and the RHS is a constant: // vselect (ext (setcc load(X), C)), N1, N2 --> // vselect (setcc extload(X), C'), N1, N2 auto ExtOpcode = IsSigned ? ISD::SIGN_EXTEND : ISD::ZERO_EXTEND; SDValue WideLHS = DAG.getNode(ExtOpcode, DL, WideVT, LHS); SDValue WideRHS = DAG.getNode(ExtOpcode, DL, WideVT, RHS); EVT WideSetCCVT = getSetCCResultType(WideVT); SDValue WideSetCC = DAG.getSetCC(DL, WideSetCCVT, WideLHS, WideRHS, CC); return DAG.getSelect(DL, N1.getValueType(), WideSetCC, N1, N2); } } } if (SimplifySelectOps(N, N1, N2)) return SDValue(N, 0); // Don't revisit N. // Fold (vselect (build_vector all_ones), N1, N2) -> N1 if (ISD::isBuildVectorAllOnes(N0.getNode())) return N1; // Fold (vselect (build_vector all_zeros), N1, N2) -> N2 if (ISD::isBuildVectorAllZeros(N0.getNode())) return N2; // The ConvertSelectToConcatVector function is assuming both the above // checks for (vselect (build_vector all{ones,zeros) ...) have been made // and addressed. if (N1.getOpcode() == ISD::CONCAT_VECTORS && N2.getOpcode() == ISD::CONCAT_VECTORS && ISD::isBuildVectorOfConstantSDNodes(N0.getNode())) { if (SDValue CV = ConvertSelectToConcatVector(N, DAG)) return CV; } if (SDValue V = foldVSelectOfConstants(N)) return V; return SDValue(); } SDValue DAGCombiner::visitSELECT_CC(SDNode *N) { SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); SDValue N2 = N->getOperand(2); SDValue N3 = N->getOperand(3); SDValue N4 = N->getOperand(4); ISD::CondCode CC = cast(N4)->get(); // fold select_cc lhs, rhs, x, x, cc -> x if (N2 == N3) return N2; // Determine if the condition we're dealing with is constant if (SDValue SCC = SimplifySetCC(getSetCCResultType(N0.getValueType()), N0, N1, CC, SDLoc(N), false)) { AddToWorklist(SCC.getNode()); if (ConstantSDNode *SCCC = dyn_cast(SCC.getNode())) { if (!SCCC->isNullValue()) return N2; // cond always true -> true val else return N3; // cond always false -> false val } else if (SCC->isUndef()) { // When the condition is UNDEF, just return the first operand. This is // coherent the DAG creation, no setcc node is created in this case return N2; } else if (SCC.getOpcode() == ISD::SETCC) { // Fold to a simpler select_cc return DAG.getNode(ISD::SELECT_CC, SDLoc(N), N2.getValueType(), SCC.getOperand(0), SCC.getOperand(1), N2, N3, SCC.getOperand(2)); } } // If we can fold this based on the true/false value, do so. if (SimplifySelectOps(N, N2, N3)) return SDValue(N, 0); // Don't revisit N. // fold select_cc into other things, such as min/max/abs return SimplifySelectCC(SDLoc(N), N0, N1, N2, N3, CC); } SDValue DAGCombiner::visitSETCC(SDNode *N) { // setcc is very commonly used as an argument to brcond. This pattern // also lend itself to numerous combines and, as a result, it is desired // we keep the argument to a brcond as a setcc as much as possible. bool PreferSetCC = N->hasOneUse() && N->use_begin()->getOpcode() == ISD::BRCOND; SDValue Combined = SimplifySetCC( N->getValueType(0), N->getOperand(0), N->getOperand(1), cast(N->getOperand(2))->get(), SDLoc(N), !PreferSetCC); if (!Combined) return SDValue(); // If we prefer to have a setcc, and we don't, we'll try our best to // recreate one using rebuildSetCC. if (PreferSetCC && Combined.getOpcode() != ISD::SETCC) { SDValue NewSetCC = rebuildSetCC(Combined); // We don't have anything interesting to combine to. if (NewSetCC.getNode() == N) return SDValue(); if (NewSetCC) return NewSetCC; } return Combined; } SDValue DAGCombiner::visitSETCCCARRY(SDNode *N) { SDValue LHS = N->getOperand(0); SDValue RHS = N->getOperand(1); SDValue Carry = N->getOperand(2); SDValue Cond = N->getOperand(3); // If Carry is false, fold to a regular SETCC. if (isNullConstant(Carry)) return DAG.getNode(ISD::SETCC, SDLoc(N), N->getVTList(), LHS, RHS, Cond); return SDValue(); } /// Try to fold a sext/zext/aext dag node into a ConstantSDNode or /// a build_vector of constants. /// This function is called by the DAGCombiner when visiting sext/zext/aext /// dag nodes (see for example method DAGCombiner::visitSIGN_EXTEND). /// Vector extends are not folded if operations are legal; this is to /// avoid introducing illegal build_vector dag nodes. static SDValue tryToFoldExtendOfConstant(SDNode *N, const TargetLowering &TLI, SelectionDAG &DAG, bool LegalTypes) { unsigned Opcode = N->getOpcode(); SDValue N0 = N->getOperand(0); EVT VT = N->getValueType(0); assert((Opcode == ISD::SIGN_EXTEND || Opcode == ISD::ZERO_EXTEND || Opcode == ISD::ANY_EXTEND || Opcode == ISD::SIGN_EXTEND_VECTOR_INREG || Opcode == ISD::ZERO_EXTEND_VECTOR_INREG) && "Expected EXTEND dag node in input!"); // fold (sext c1) -> c1 // fold (zext c1) -> c1 // fold (aext c1) -> c1 if (isa(N0)) return DAG.getNode(Opcode, SDLoc(N), VT, N0); // fold (sext (build_vector AllConstants) -> (build_vector AllConstants) // fold (zext (build_vector AllConstants) -> (build_vector AllConstants) // fold (aext (build_vector AllConstants) -> (build_vector AllConstants) EVT SVT = VT.getScalarType(); if (!(VT.isVector() && (!LegalTypes || TLI.isTypeLegal(SVT)) && ISD::isBuildVectorOfConstantSDNodes(N0.getNode()))) return SDValue(); // We can fold this node into a build_vector. unsigned VTBits = SVT.getSizeInBits(); unsigned EVTBits = N0->getValueType(0).getScalarSizeInBits(); SmallVector Elts; unsigned NumElts = VT.getVectorNumElements(); SDLoc DL(N); // For zero-extensions, UNDEF elements still guarantee to have the upper // bits set to zero. bool IsZext = Opcode == ISD::ZERO_EXTEND || Opcode == ISD::ZERO_EXTEND_VECTOR_INREG; for (unsigned i = 0; i != NumElts; ++i) { SDValue Op = N0.getOperand(i); if (Op.isUndef()) { Elts.push_back(IsZext ? DAG.getConstant(0, DL, SVT) : DAG.getUNDEF(SVT)); continue; } SDLoc DL(Op); // Get the constant value and if needed trunc it to the size of the type. // Nodes like build_vector might have constants wider than the scalar type. APInt C = cast(Op)->getAPIntValue().zextOrTrunc(EVTBits); if (Opcode == ISD::SIGN_EXTEND || Opcode == ISD::SIGN_EXTEND_VECTOR_INREG) Elts.push_back(DAG.getConstant(C.sext(VTBits), DL, SVT)); else Elts.push_back(DAG.getConstant(C.zext(VTBits), DL, SVT)); } return DAG.getBuildVector(VT, DL, Elts); } // ExtendUsesToFormExtLoad - Trying to extend uses of a load to enable this: // "fold ({s|z|a}ext (load x)) -> ({s|z|a}ext (truncate ({s|z|a}extload x)))" // transformation. Returns true if extension are possible and the above // mentioned transformation is profitable. static bool ExtendUsesToFormExtLoad(EVT VT, SDNode *N, SDValue N0, unsigned ExtOpc, SmallVectorImpl &ExtendNodes, const TargetLowering &TLI) { bool HasCopyToRegUses = false; bool isTruncFree = TLI.isTruncateFree(VT, N0.getValueType()); for (SDNode::use_iterator UI = N0.getNode()->use_begin(), UE = N0.getNode()->use_end(); UI != UE; ++UI) { SDNode *User = *UI; if (User == N) continue; if (UI.getUse().getResNo() != N0.getResNo()) continue; // FIXME: Only extend SETCC N, N and SETCC N, c for now. if (ExtOpc != ISD::ANY_EXTEND && User->getOpcode() == ISD::SETCC) { ISD::CondCode CC = cast(User->getOperand(2))->get(); if (ExtOpc == ISD::ZERO_EXTEND && ISD::isSignedIntSetCC(CC)) // Sign bits will be lost after a zext. return false; bool Add = false; for (unsigned i = 0; i != 2; ++i) { SDValue UseOp = User->getOperand(i); if (UseOp == N0) continue; if (!isa(UseOp)) return false; Add = true; } if (Add) ExtendNodes.push_back(User); continue; } // If truncates aren't free and there are users we can't // extend, it isn't worthwhile. if (!isTruncFree) return false; // Remember if this value is live-out. if (User->getOpcode() == ISD::CopyToReg) HasCopyToRegUses = true; } if (HasCopyToRegUses) { bool BothLiveOut = false; for (SDNode::use_iterator UI = N->use_begin(), UE = N->use_end(); UI != UE; ++UI) { SDUse &Use = UI.getUse(); if (Use.getResNo() == 0 && Use.getUser()->getOpcode() == ISD::CopyToReg) { BothLiveOut = true; break; } } if (BothLiveOut) // Both unextended and extended values are live out. There had better be // a good reason for the transformation. return ExtendNodes.size(); } return true; } void DAGCombiner::ExtendSetCCUses(const SmallVectorImpl &SetCCs, SDValue OrigLoad, SDValue ExtLoad, ISD::NodeType ExtType) { // Extend SetCC uses if necessary. SDLoc DL(ExtLoad); for (SDNode *SetCC : SetCCs) { SmallVector Ops; for (unsigned j = 0; j != 2; ++j) { SDValue SOp = SetCC->getOperand(j); if (SOp == OrigLoad) Ops.push_back(ExtLoad); else Ops.push_back(DAG.getNode(ExtType, DL, ExtLoad->getValueType(0), SOp)); } Ops.push_back(SetCC->getOperand(2)); CombineTo(SetCC, DAG.getNode(ISD::SETCC, DL, SetCC->getValueType(0), Ops)); } } // FIXME: Bring more similar combines here, common to sext/zext (maybe aext?). SDValue DAGCombiner::CombineExtLoad(SDNode *N) { SDValue N0 = N->getOperand(0); EVT DstVT = N->getValueType(0); EVT SrcVT = N0.getValueType(); assert((N->getOpcode() == ISD::SIGN_EXTEND || N->getOpcode() == ISD::ZERO_EXTEND) && "Unexpected node type (not an extend)!"); // fold (sext (load x)) to multiple smaller sextloads; same for zext. // For example, on a target with legal v4i32, but illegal v8i32, turn: // (v8i32 (sext (v8i16 (load x)))) // into: // (v8i32 (concat_vectors (v4i32 (sextload x)), // (v4i32 (sextload (x + 16))))) // Where uses of the original load, i.e.: // (v8i16 (load x)) // are replaced with: // (v8i16 (truncate // (v8i32 (concat_vectors (v4i32 (sextload x)), // (v4i32 (sextload (x + 16))))))) // // This combine is only applicable to illegal, but splittable, vectors. // All legal types, and illegal non-vector types, are handled elsewhere. // This combine is controlled by TargetLowering::isVectorLoadExtDesirable. // if (N0->getOpcode() != ISD::LOAD) return SDValue(); LoadSDNode *LN0 = cast(N0); if (!ISD::isNON_EXTLoad(LN0) || !ISD::isUNINDEXEDLoad(LN0) || !N0.hasOneUse() || LN0->isVolatile() || !DstVT.isVector() || !DstVT.isPow2VectorType() || !TLI.isVectorLoadExtDesirable(SDValue(N, 0))) return SDValue(); SmallVector SetCCs; if (!ExtendUsesToFormExtLoad(DstVT, N, N0, N->getOpcode(), SetCCs, TLI)) return SDValue(); ISD::LoadExtType ExtType = N->getOpcode() == ISD::SIGN_EXTEND ? ISD::SEXTLOAD : ISD::ZEXTLOAD; // Try to split the vector types to get down to legal types. EVT SplitSrcVT = SrcVT; EVT SplitDstVT = DstVT; while (!TLI.isLoadExtLegalOrCustom(ExtType, SplitDstVT, SplitSrcVT) && SplitSrcVT.getVectorNumElements() > 1) { SplitDstVT = DAG.GetSplitDestVTs(SplitDstVT).first; SplitSrcVT = DAG.GetSplitDestVTs(SplitSrcVT).first; } if (!TLI.isLoadExtLegalOrCustom(ExtType, SplitDstVT, SplitSrcVT)) return SDValue(); SDLoc DL(N); const unsigned NumSplits = DstVT.getVectorNumElements() / SplitDstVT.getVectorNumElements(); const unsigned Stride = SplitSrcVT.getStoreSize(); SmallVector Loads; SmallVector Chains; SDValue BasePtr = LN0->getBasePtr(); for (unsigned Idx = 0; Idx < NumSplits; Idx++) { const unsigned Offset = Idx * Stride; const unsigned Align = MinAlign(LN0->getAlignment(), Offset); SDValue SplitLoad = DAG.getExtLoad( ExtType, SDLoc(LN0), SplitDstVT, LN0->getChain(), BasePtr, LN0->getPointerInfo().getWithOffset(Offset), SplitSrcVT, Align, LN0->getMemOperand()->getFlags(), LN0->getAAInfo()); BasePtr = DAG.getNode(ISD::ADD, DL, BasePtr.getValueType(), BasePtr, DAG.getConstant(Stride, DL, BasePtr.getValueType())); Loads.push_back(SplitLoad.getValue(0)); Chains.push_back(SplitLoad.getValue(1)); } SDValue NewChain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, Chains); SDValue NewValue = DAG.getNode(ISD::CONCAT_VECTORS, DL, DstVT, Loads); // Simplify TF. AddToWorklist(NewChain.getNode()); CombineTo(N, NewValue); // Replace uses of the original load (before extension) // with a truncate of the concatenated sextloaded vectors. SDValue Trunc = DAG.getNode(ISD::TRUNCATE, SDLoc(N0), N0.getValueType(), NewValue); ExtendSetCCUses(SetCCs, N0, NewValue, (ISD::NodeType)N->getOpcode()); CombineTo(N0.getNode(), Trunc, NewChain); return SDValue(N, 0); // Return N so it doesn't get rechecked! } // fold (zext (and/or/xor (shl/shr (load x), cst), cst)) -> // (and/or/xor (shl/shr (zextload x), (zext cst)), (zext cst)) SDValue DAGCombiner::CombineZExtLogicopShiftLoad(SDNode *N) { assert(N->getOpcode() == ISD::ZERO_EXTEND); EVT VT = N->getValueType(0); // and/or/xor SDValue N0 = N->getOperand(0); if (!(N0.getOpcode() == ISD::AND || N0.getOpcode() == ISD::OR || N0.getOpcode() == ISD::XOR) || N0.getOperand(1).getOpcode() != ISD::Constant || (LegalOperations && !TLI.isOperationLegal(N0.getOpcode(), VT))) return SDValue(); // shl/shr SDValue N1 = N0->getOperand(0); if (!(N1.getOpcode() == ISD::SHL || N1.getOpcode() == ISD::SRL) || N1.getOperand(1).getOpcode() != ISD::Constant || (LegalOperations && !TLI.isOperationLegal(N1.getOpcode(), VT))) return SDValue(); // load if (!isa(N1.getOperand(0))) return SDValue(); LoadSDNode *Load = cast(N1.getOperand(0)); EVT MemVT = Load->getMemoryVT(); if (!TLI.isLoadExtLegal(ISD::ZEXTLOAD, VT, MemVT) || Load->getExtensionType() == ISD::SEXTLOAD || Load->isIndexed()) return SDValue(); // If the shift op is SHL, the logic op must be AND, otherwise the result // will be wrong. if (N1.getOpcode() == ISD::SHL && N0.getOpcode() != ISD::AND) return SDValue(); if (!N0.hasOneUse() || !N1.hasOneUse()) return SDValue(); SmallVector SetCCs; if (!ExtendUsesToFormExtLoad(VT, N1.getNode(), N1.getOperand(0), ISD::ZERO_EXTEND, SetCCs, TLI)) return SDValue(); // Actually do the transformation. SDValue ExtLoad = DAG.getExtLoad(ISD::ZEXTLOAD, SDLoc(Load), VT, Load->getChain(), Load->getBasePtr(), Load->getMemoryVT(), Load->getMemOperand()); SDLoc DL1(N1); SDValue Shift = DAG.getNode(N1.getOpcode(), DL1, VT, ExtLoad, N1.getOperand(1)); APInt Mask = cast(N0.getOperand(1))->getAPIntValue(); Mask = Mask.zext(VT.getSizeInBits()); SDLoc DL0(N0); SDValue And = DAG.getNode(N0.getOpcode(), DL0, VT, Shift, DAG.getConstant(Mask, DL0, VT)); ExtendSetCCUses(SetCCs, N1.getOperand(0), ExtLoad, ISD::ZERO_EXTEND); CombineTo(N, And); if (SDValue(Load, 0).hasOneUse()) { DAG.ReplaceAllUsesOfValueWith(SDValue(Load, 1), ExtLoad.getValue(1)); } else { SDValue Trunc = DAG.getNode(ISD::TRUNCATE, SDLoc(Load), Load->getValueType(0), ExtLoad); CombineTo(Load, Trunc, ExtLoad.getValue(1)); } return SDValue(N,0); // Return N so it doesn't get rechecked! } /// If we're narrowing or widening the result of a vector select and the final /// size is the same size as a setcc (compare) feeding the select, then try to /// apply the cast operation to the select's operands because matching vector /// sizes for a select condition and other operands should be more efficient. SDValue DAGCombiner::matchVSelectOpSizesWithSetCC(SDNode *Cast) { unsigned CastOpcode = Cast->getOpcode(); assert((CastOpcode == ISD::SIGN_EXTEND || CastOpcode == ISD::ZERO_EXTEND || CastOpcode == ISD::TRUNCATE || CastOpcode == ISD::FP_EXTEND || CastOpcode == ISD::FP_ROUND) && "Unexpected opcode for vector select narrowing/widening"); // We only do this transform before legal ops because the pattern may be // obfuscated by target-specific operations after legalization. Do not create // an illegal select op, however, because that may be difficult to lower. EVT VT = Cast->getValueType(0); if (LegalOperations || !TLI.isOperationLegalOrCustom(ISD::VSELECT, VT)) return SDValue(); SDValue VSel = Cast->getOperand(0); if (VSel.getOpcode() != ISD::VSELECT || !VSel.hasOneUse() || VSel.getOperand(0).getOpcode() != ISD::SETCC) return SDValue(); // Does the setcc have the same vector size as the casted select? SDValue SetCC = VSel.getOperand(0); EVT SetCCVT = getSetCCResultType(SetCC.getOperand(0).getValueType()); if (SetCCVT.getSizeInBits() != VT.getSizeInBits()) return SDValue(); // cast (vsel (setcc X), A, B) --> vsel (setcc X), (cast A), (cast B) SDValue A = VSel.getOperand(1); SDValue B = VSel.getOperand(2); SDValue CastA, CastB; SDLoc DL(Cast); if (CastOpcode == ISD::FP_ROUND) { // FP_ROUND (fptrunc) has an extra flag operand to pass along. CastA = DAG.getNode(CastOpcode, DL, VT, A, Cast->getOperand(1)); CastB = DAG.getNode(CastOpcode, DL, VT, B, Cast->getOperand(1)); } else { CastA = DAG.getNode(CastOpcode, DL, VT, A); CastB = DAG.getNode(CastOpcode, DL, VT, B); } return DAG.getNode(ISD::VSELECT, DL, VT, SetCC, CastA, CastB); } // fold ([s|z]ext ([s|z]extload x)) -> ([s|z]ext (truncate ([s|z]extload x))) // fold ([s|z]ext ( extload x)) -> ([s|z]ext (truncate ([s|z]extload x))) static SDValue tryToFoldExtOfExtload(SelectionDAG &DAG, DAGCombiner &Combiner, const TargetLowering &TLI, EVT VT, bool LegalOperations, SDNode *N, SDValue N0, ISD::LoadExtType ExtLoadType) { SDNode *N0Node = N0.getNode(); bool isAExtLoad = (ExtLoadType == ISD::SEXTLOAD) ? ISD::isSEXTLoad(N0Node) : ISD::isZEXTLoad(N0Node); if ((!isAExtLoad && !ISD::isEXTLoad(N0Node)) || !ISD::isUNINDEXEDLoad(N0Node) || !N0.hasOneUse()) return {}; LoadSDNode *LN0 = cast(N0); EVT MemVT = LN0->getMemoryVT(); if ((LegalOperations || LN0->isVolatile() || VT.isVector()) && !TLI.isLoadExtLegal(ExtLoadType, VT, MemVT)) return {}; SDValue ExtLoad = DAG.getExtLoad(ExtLoadType, SDLoc(LN0), VT, LN0->getChain(), LN0->getBasePtr(), MemVT, LN0->getMemOperand()); Combiner.CombineTo(N, ExtLoad); DAG.ReplaceAllUsesOfValueWith(SDValue(LN0, 1), ExtLoad.getValue(1)); return SDValue(N, 0); // Return N so it doesn't get rechecked! } // fold ([s|z]ext (load x)) -> ([s|z]ext (truncate ([s|z]extload x))) // Only generate vector extloads when 1) they're legal, and 2) they are // deemed desirable by the target. static SDValue tryToFoldExtOfLoad(SelectionDAG &DAG, DAGCombiner &Combiner, const TargetLowering &TLI, EVT VT, bool LegalOperations, SDNode *N, SDValue N0, ISD::LoadExtType ExtLoadType, ISD::NodeType ExtOpc) { if (!ISD::isNON_EXTLoad(N0.getNode()) || !ISD::isUNINDEXEDLoad(N0.getNode()) || ((LegalOperations || VT.isVector() || cast(N0)->isVolatile()) && !TLI.isLoadExtLegal(ExtLoadType, VT, N0.getValueType()))) return {}; bool DoXform = true; SmallVector SetCCs; if (!N0.hasOneUse()) DoXform = ExtendUsesToFormExtLoad(VT, N, N0, ExtOpc, SetCCs, TLI); if (VT.isVector()) DoXform &= TLI.isVectorLoadExtDesirable(SDValue(N, 0)); if (!DoXform) return {}; LoadSDNode *LN0 = cast(N0); SDValue ExtLoad = DAG.getExtLoad(ExtLoadType, SDLoc(LN0), VT, LN0->getChain(), LN0->getBasePtr(), N0.getValueType(), LN0->getMemOperand()); Combiner.ExtendSetCCUses(SetCCs, N0, ExtLoad, ExtOpc); // If the load value is used only by N, replace it via CombineTo N. bool NoReplaceTrunc = SDValue(LN0, 0).hasOneUse(); Combiner.CombineTo(N, ExtLoad); if (NoReplaceTrunc) { DAG.ReplaceAllUsesOfValueWith(SDValue(LN0, 1), ExtLoad.getValue(1)); } else { SDValue Trunc = DAG.getNode(ISD::TRUNCATE, SDLoc(N0), N0.getValueType(), ExtLoad); Combiner.CombineTo(LN0, Trunc, ExtLoad.getValue(1)); } return SDValue(N, 0); // Return N so it doesn't get rechecked! } static SDValue foldExtendedSignBitTest(SDNode *N, SelectionDAG &DAG, bool LegalOperations) { assert((N->getOpcode() == ISD::SIGN_EXTEND || N->getOpcode() == ISD::ZERO_EXTEND) && "Expected sext or zext"); SDValue SetCC = N->getOperand(0); if (LegalOperations || SetCC.getOpcode() != ISD::SETCC || !SetCC.hasOneUse() || SetCC.getValueType() != MVT::i1) return SDValue(); SDValue X = SetCC.getOperand(0); SDValue Ones = SetCC.getOperand(1); ISD::CondCode CC = cast(SetCC.getOperand(2))->get(); EVT VT = N->getValueType(0); EVT XVT = X.getValueType(); // setge X, C is canonicalized to setgt, so we do not need to match that // pattern. The setlt sibling is folded in SimplifySelectCC() because it does // not require the 'not' op. if (CC == ISD::SETGT && isAllOnesConstant(Ones) && VT == XVT) { // Invert and smear/shift the sign bit: // sext i1 (setgt iN X, -1) --> sra (not X), (N - 1) // zext i1 (setgt iN X, -1) --> srl (not X), (N - 1) SDLoc DL(N); SDValue NotX = DAG.getNOT(DL, X, VT); SDValue ShiftAmount = DAG.getConstant(VT.getSizeInBits() - 1, DL, VT); auto ShiftOpcode = N->getOpcode() == ISD::SIGN_EXTEND ? ISD::SRA : ISD::SRL; return DAG.getNode(ShiftOpcode, DL, VT, NotX, ShiftAmount); } return SDValue(); } SDValue DAGCombiner::visitSIGN_EXTEND(SDNode *N) { SDValue N0 = N->getOperand(0); EVT VT = N->getValueType(0); SDLoc DL(N); if (SDValue Res = tryToFoldExtendOfConstant(N, TLI, DAG, LegalTypes)) return Res; // fold (sext (sext x)) -> (sext x) // fold (sext (aext x)) -> (sext x) if (N0.getOpcode() == ISD::SIGN_EXTEND || N0.getOpcode() == ISD::ANY_EXTEND) return DAG.getNode(ISD::SIGN_EXTEND, DL, VT, N0.getOperand(0)); if (N0.getOpcode() == ISD::TRUNCATE) { // fold (sext (truncate (load x))) -> (sext (smaller load x)) // fold (sext (truncate (srl (load x), c))) -> (sext (smaller load (x+c/n))) if (SDValue NarrowLoad = ReduceLoadWidth(N0.getNode())) { SDNode *oye = N0.getOperand(0).getNode(); if (NarrowLoad.getNode() != N0.getNode()) { CombineTo(N0.getNode(), NarrowLoad); // CombineTo deleted the truncate, if needed, but not what's under it. AddToWorklist(oye); } return SDValue(N, 0); // Return N so it doesn't get rechecked! } // See if the value being truncated is already sign extended. If so, just // eliminate the trunc/sext pair. SDValue Op = N0.getOperand(0); unsigned OpBits = Op.getScalarValueSizeInBits(); unsigned MidBits = N0.getScalarValueSizeInBits(); unsigned DestBits = VT.getScalarSizeInBits(); unsigned NumSignBits = DAG.ComputeNumSignBits(Op); if (OpBits == DestBits) { // Op is i32, Mid is i8, and Dest is i32. If Op has more than 24 sign // bits, it is already ready. if (NumSignBits > DestBits-MidBits) return Op; } else if (OpBits < DestBits) { // Op is i32, Mid is i8, and Dest is i64. If Op has more than 24 sign // bits, just sext from i32. if (NumSignBits > OpBits-MidBits) return DAG.getNode(ISD::SIGN_EXTEND, DL, VT, Op); } else { // Op is i64, Mid is i8, and Dest is i32. If Op has more than 56 sign // bits, just truncate to i32. if (NumSignBits > OpBits-MidBits) return DAG.getNode(ISD::TRUNCATE, DL, VT, Op); } // fold (sext (truncate x)) -> (sextinreg x). if (!LegalOperations || TLI.isOperationLegal(ISD::SIGN_EXTEND_INREG, N0.getValueType())) { if (OpBits < DestBits) Op = DAG.getNode(ISD::ANY_EXTEND, SDLoc(N0), VT, Op); else if (OpBits > DestBits) Op = DAG.getNode(ISD::TRUNCATE, SDLoc(N0), VT, Op); return DAG.getNode(ISD::SIGN_EXTEND_INREG, DL, VT, Op, DAG.getValueType(N0.getValueType())); } } // Try to simplify (sext (load x)). if (SDValue foldedExt = tryToFoldExtOfLoad(DAG, *this, TLI, VT, LegalOperations, N, N0, ISD::SEXTLOAD, ISD::SIGN_EXTEND)) return foldedExt; // fold (sext (load x)) to multiple smaller sextloads. // Only on illegal but splittable vectors. if (SDValue ExtLoad = CombineExtLoad(N)) return ExtLoad; // Try to simplify (sext (sextload x)). if (SDValue foldedExt = tryToFoldExtOfExtload( DAG, *this, TLI, VT, LegalOperations, N, N0, ISD::SEXTLOAD)) return foldedExt; // fold (sext (and/or/xor (load x), cst)) -> // (and/or/xor (sextload x), (sext cst)) if ((N0.getOpcode() == ISD::AND || N0.getOpcode() == ISD::OR || N0.getOpcode() == ISD::XOR) && isa(N0.getOperand(0)) && N0.getOperand(1).getOpcode() == ISD::Constant && (!LegalOperations && TLI.isOperationLegal(N0.getOpcode(), VT))) { LoadSDNode *LN00 = cast(N0.getOperand(0)); EVT MemVT = LN00->getMemoryVT(); if (TLI.isLoadExtLegal(ISD::SEXTLOAD, VT, MemVT) && LN00->getExtensionType() != ISD::ZEXTLOAD && LN00->isUnindexed()) { SmallVector SetCCs; bool DoXform = ExtendUsesToFormExtLoad(VT, N0.getNode(), N0.getOperand(0), ISD::SIGN_EXTEND, SetCCs, TLI); if (DoXform) { SDValue ExtLoad = DAG.getExtLoad(ISD::SEXTLOAD, SDLoc(LN00), VT, LN00->getChain(), LN00->getBasePtr(), LN00->getMemoryVT(), LN00->getMemOperand()); APInt Mask = cast(N0.getOperand(1))->getAPIntValue(); Mask = Mask.sext(VT.getSizeInBits()); SDValue And = DAG.getNode(N0.getOpcode(), DL, VT, ExtLoad, DAG.getConstant(Mask, DL, VT)); ExtendSetCCUses(SetCCs, N0.getOperand(0), ExtLoad, ISD::SIGN_EXTEND); bool NoReplaceTruncAnd = !N0.hasOneUse(); bool NoReplaceTrunc = SDValue(LN00, 0).hasOneUse(); CombineTo(N, And); // If N0 has multiple uses, change other uses as well. if (NoReplaceTruncAnd) { SDValue TruncAnd = DAG.getNode(ISD::TRUNCATE, DL, N0.getValueType(), And); CombineTo(N0.getNode(), TruncAnd); } if (NoReplaceTrunc) { DAG.ReplaceAllUsesOfValueWith(SDValue(LN00, 1), ExtLoad.getValue(1)); } else { SDValue Trunc = DAG.getNode(ISD::TRUNCATE, SDLoc(LN00), LN00->getValueType(0), ExtLoad); CombineTo(LN00, Trunc, ExtLoad.getValue(1)); } return SDValue(N,0); // Return N so it doesn't get rechecked! } } } if (SDValue V = foldExtendedSignBitTest(N, DAG, LegalOperations)) return V; if (N0.getOpcode() == ISD::SETCC) { SDValue N00 = N0.getOperand(0); SDValue N01 = N0.getOperand(1); ISD::CondCode CC = cast(N0.getOperand(2))->get(); EVT N00VT = N0.getOperand(0).getValueType(); // sext(setcc) -> sext_in_reg(vsetcc) for vectors. // Only do this before legalize for now. if (VT.isVector() && !LegalOperations && TLI.getBooleanContents(N00VT) == TargetLowering::ZeroOrNegativeOneBooleanContent) { // On some architectures (such as SSE/NEON/etc) the SETCC result type is // of the same size as the compared operands. Only optimize sext(setcc()) // if this is the case. EVT SVT = getSetCCResultType(N00VT); // If we already have the desired type, don't change it. if (SVT != N0.getValueType()) { // We know that the # elements of the results is the same as the // # elements of the compare (and the # elements of the compare result // for that matter). Check to see that they are the same size. If so, // we know that the element size of the sext'd result matches the // element size of the compare operands. if (VT.getSizeInBits() == SVT.getSizeInBits()) return DAG.getSetCC(DL, VT, N00, N01, CC); // If the desired elements are smaller or larger than the source // elements, we can use a matching integer vector type and then // truncate/sign extend. EVT MatchingVecType = N00VT.changeVectorElementTypeToInteger(); if (SVT == MatchingVecType) { SDValue VsetCC = DAG.getSetCC(DL, MatchingVecType, N00, N01, CC); return DAG.getSExtOrTrunc(VsetCC, DL, VT); } } } // sext(setcc x, y, cc) -> (select (setcc x, y, cc), T, 0) // Here, T can be 1 or -1, depending on the type of the setcc and // getBooleanContents(). unsigned SetCCWidth = N0.getScalarValueSizeInBits(); // To determine the "true" side of the select, we need to know the high bit // of the value returned by the setcc if it evaluates to true. // If the type of the setcc is i1, then the true case of the select is just // sext(i1 1), that is, -1. // If the type of the setcc is larger (say, i8) then the value of the high // bit depends on getBooleanContents(), so ask TLI for a real "true" value // of the appropriate width. SDValue ExtTrueVal = (SetCCWidth == 1) ? DAG.getAllOnesConstant(DL, VT) : DAG.getBoolConstant(true, DL, VT, N00VT); SDValue Zero = DAG.getConstant(0, DL, VT); if (SDValue SCC = SimplifySelectCC(DL, N00, N01, ExtTrueVal, Zero, CC, true)) return SCC; if (!VT.isVector() && !TLI.convertSelectOfConstantsToMath(VT)) { EVT SetCCVT = getSetCCResultType(N00VT); // Don't do this transform for i1 because there's a select transform // that would reverse it. // TODO: We should not do this transform at all without a target hook // because a sext is likely cheaper than a select? if (SetCCVT.getScalarSizeInBits() != 1 && (!LegalOperations || TLI.isOperationLegal(ISD::SETCC, N00VT))) { SDValue SetCC = DAG.getSetCC(DL, SetCCVT, N00, N01, CC); return DAG.getSelect(DL, VT, SetCC, ExtTrueVal, Zero); } } } // fold (sext x) -> (zext x) if the sign bit is known zero. if ((!LegalOperations || TLI.isOperationLegal(ISD::ZERO_EXTEND, VT)) && DAG.SignBitIsZero(N0)) return DAG.getNode(ISD::ZERO_EXTEND, DL, VT, N0); if (SDValue NewVSel = matchVSelectOpSizesWithSetCC(N)) return NewVSel; return SDValue(); } // isTruncateOf - If N is a truncate of some other value, return true, record // the value being truncated in Op and which of Op's bits are zero/one in Known. // This function computes KnownBits to avoid a duplicated call to // computeKnownBits in the caller. static bool isTruncateOf(SelectionDAG &DAG, SDValue N, SDValue &Op, KnownBits &Known) { if (N->getOpcode() == ISD::TRUNCATE) { Op = N->getOperand(0); Known = DAG.computeKnownBits(Op); return true; } if (N.getOpcode() != ISD::SETCC || N.getValueType().getScalarType() != MVT::i1 || cast(N.getOperand(2))->get() != ISD::SETNE) return false; SDValue Op0 = N->getOperand(0); SDValue Op1 = N->getOperand(1); assert(Op0.getValueType() == Op1.getValueType()); if (isNullOrNullSplat(Op0)) Op = Op1; else if (isNullOrNullSplat(Op1)) Op = Op0; else return false; Known = DAG.computeKnownBits(Op); return (Known.Zero | 1).isAllOnesValue(); } SDValue DAGCombiner::visitZERO_EXTEND(SDNode *N) { SDValue N0 = N->getOperand(0); EVT VT = N->getValueType(0); if (SDValue Res = tryToFoldExtendOfConstant(N, TLI, DAG, LegalTypes)) return Res; // fold (zext (zext x)) -> (zext x) // fold (zext (aext x)) -> (zext x) if (N0.getOpcode() == ISD::ZERO_EXTEND || N0.getOpcode() == ISD::ANY_EXTEND) return DAG.getNode(ISD::ZERO_EXTEND, SDLoc(N), VT, N0.getOperand(0)); // fold (zext (truncate x)) -> (zext x) or // (zext (truncate x)) -> (truncate x) // This is valid when the truncated bits of x are already zero. SDValue Op; KnownBits Known; if (isTruncateOf(DAG, N0, Op, Known)) { APInt TruncatedBits = (Op.getScalarValueSizeInBits() == N0.getScalarValueSizeInBits()) ? APInt(Op.getScalarValueSizeInBits(), 0) : APInt::getBitsSet(Op.getScalarValueSizeInBits(), N0.getScalarValueSizeInBits(), std::min(Op.getScalarValueSizeInBits(), VT.getScalarSizeInBits())); if (TruncatedBits.isSubsetOf(Known.Zero)) return DAG.getZExtOrTrunc(Op, SDLoc(N), VT); } // fold (zext (truncate x)) -> (and x, mask) if (N0.getOpcode() == ISD::TRUNCATE) { // fold (zext (truncate (load x))) -> (zext (smaller load x)) // fold (zext (truncate (srl (load x), c))) -> (zext (smaller load (x+c/n))) if (SDValue NarrowLoad = ReduceLoadWidth(N0.getNode())) { SDNode *oye = N0.getOperand(0).getNode(); if (NarrowLoad.getNode() != N0.getNode()) { CombineTo(N0.getNode(), NarrowLoad); // CombineTo deleted the truncate, if needed, but not what's under it. AddToWorklist(oye); } return SDValue(N, 0); // Return N so it doesn't get rechecked! } EVT SrcVT = N0.getOperand(0).getValueType(); EVT MinVT = N0.getValueType(); // Try to mask before the extension to avoid having to generate a larger mask, // possibly over several sub-vectors. if (SrcVT.bitsLT(VT) && VT.isVector()) { if (!LegalOperations || (TLI.isOperationLegal(ISD::AND, SrcVT) && TLI.isOperationLegal(ISD::ZERO_EXTEND, VT))) { SDValue Op = N0.getOperand(0); Op = DAG.getZeroExtendInReg(Op, SDLoc(N), MinVT.getScalarType()); AddToWorklist(Op.getNode()); SDValue ZExtOrTrunc = DAG.getZExtOrTrunc(Op, SDLoc(N), VT); // Transfer the debug info; the new node is equivalent to N0. DAG.transferDbgValues(N0, ZExtOrTrunc); return ZExtOrTrunc; } } if (!LegalOperations || TLI.isOperationLegal(ISD::AND, VT)) { SDValue Op = DAG.getAnyExtOrTrunc(N0.getOperand(0), SDLoc(N), VT); AddToWorklist(Op.getNode()); SDValue And = DAG.getZeroExtendInReg(Op, SDLoc(N), MinVT.getScalarType()); // We may safely transfer the debug info describing the truncate node over // to the equivalent and operation. DAG.transferDbgValues(N0, And); return And; } } // Fold (zext (and (trunc x), cst)) -> (and x, cst), // if either of the casts is not free. if (N0.getOpcode() == ISD::AND && N0.getOperand(0).getOpcode() == ISD::TRUNCATE && N0.getOperand(1).getOpcode() == ISD::Constant && (!TLI.isTruncateFree(N0.getOperand(0).getOperand(0).getValueType(), N0.getValueType()) || !TLI.isZExtFree(N0.getValueType(), VT))) { SDValue X = N0.getOperand(0).getOperand(0); X = DAG.getAnyExtOrTrunc(X, SDLoc(X), VT); APInt Mask = cast(N0.getOperand(1))->getAPIntValue(); Mask = Mask.zext(VT.getSizeInBits()); SDLoc DL(N); return DAG.getNode(ISD::AND, DL, VT, X, DAG.getConstant(Mask, DL, VT)); } // Try to simplify (zext (load x)). if (SDValue foldedExt = tryToFoldExtOfLoad(DAG, *this, TLI, VT, LegalOperations, N, N0, ISD::ZEXTLOAD, ISD::ZERO_EXTEND)) return foldedExt; // fold (zext (load x)) to multiple smaller zextloads. // Only on illegal but splittable vectors. if (SDValue ExtLoad = CombineExtLoad(N)) return ExtLoad; // fold (zext (and/or/xor (load x), cst)) -> // (and/or/xor (zextload x), (zext cst)) // Unless (and (load x) cst) will match as a zextload already and has // additional users. if ((N0.getOpcode() == ISD::AND || N0.getOpcode() == ISD::OR || N0.getOpcode() == ISD::XOR) && isa(N0.getOperand(0)) && N0.getOperand(1).getOpcode() == ISD::Constant && (!LegalOperations && TLI.isOperationLegal(N0.getOpcode(), VT))) { LoadSDNode *LN00 = cast(N0.getOperand(0)); EVT MemVT = LN00->getMemoryVT(); if (TLI.isLoadExtLegal(ISD::ZEXTLOAD, VT, MemVT) && LN00->getExtensionType() != ISD::SEXTLOAD && LN00->isUnindexed()) { bool DoXform = true; SmallVector SetCCs; if (!N0.hasOneUse()) { if (N0.getOpcode() == ISD::AND) { auto *AndC = cast(N0.getOperand(1)); EVT LoadResultTy = AndC->getValueType(0); EVT ExtVT; if (isAndLoadExtLoad(AndC, LN00, LoadResultTy, ExtVT)) DoXform = false; } } if (DoXform) DoXform = ExtendUsesToFormExtLoad(VT, N0.getNode(), N0.getOperand(0), ISD::ZERO_EXTEND, SetCCs, TLI); if (DoXform) { SDValue ExtLoad = DAG.getExtLoad(ISD::ZEXTLOAD, SDLoc(LN00), VT, LN00->getChain(), LN00->getBasePtr(), LN00->getMemoryVT(), LN00->getMemOperand()); APInt Mask = cast(N0.getOperand(1))->getAPIntValue(); Mask = Mask.zext(VT.getSizeInBits()); SDLoc DL(N); SDValue And = DAG.getNode(N0.getOpcode(), DL, VT, ExtLoad, DAG.getConstant(Mask, DL, VT)); ExtendSetCCUses(SetCCs, N0.getOperand(0), ExtLoad, ISD::ZERO_EXTEND); bool NoReplaceTruncAnd = !N0.hasOneUse(); bool NoReplaceTrunc = SDValue(LN00, 0).hasOneUse(); CombineTo(N, And); // If N0 has multiple uses, change other uses as well. if (NoReplaceTruncAnd) { SDValue TruncAnd = DAG.getNode(ISD::TRUNCATE, DL, N0.getValueType(), And); CombineTo(N0.getNode(), TruncAnd); } if (NoReplaceTrunc) { DAG.ReplaceAllUsesOfValueWith(SDValue(LN00, 1), ExtLoad.getValue(1)); } else { SDValue Trunc = DAG.getNode(ISD::TRUNCATE, SDLoc(LN00), LN00->getValueType(0), ExtLoad); CombineTo(LN00, Trunc, ExtLoad.getValue(1)); } return SDValue(N,0); // Return N so it doesn't get rechecked! } } } // fold (zext (and/or/xor (shl/shr (load x), cst), cst)) -> // (and/or/xor (shl/shr (zextload x), (zext cst)), (zext cst)) if (SDValue ZExtLoad = CombineZExtLogicopShiftLoad(N)) return ZExtLoad; // Try to simplify (zext (zextload x)). if (SDValue foldedExt = tryToFoldExtOfExtload( DAG, *this, TLI, VT, LegalOperations, N, N0, ISD::ZEXTLOAD)) return foldedExt; if (SDValue V = foldExtendedSignBitTest(N, DAG, LegalOperations)) return V; if (N0.getOpcode() == ISD::SETCC) { // Only do this before legalize for now. if (!LegalOperations && VT.isVector() && N0.getValueType().getVectorElementType() == MVT::i1) { EVT N00VT = N0.getOperand(0).getValueType(); if (getSetCCResultType(N00VT) == N0.getValueType()) return SDValue(); // We know that the # elements of the results is the same as the # // elements of the compare (and the # elements of the compare result for // that matter). Check to see that they are the same size. If so, we know // that the element size of the sext'd result matches the element size of // the compare operands. SDLoc DL(N); SDValue VecOnes = DAG.getConstant(1, DL, VT); if (VT.getSizeInBits() == N00VT.getSizeInBits()) { // zext(setcc) -> (and (vsetcc), (1, 1, ...) for vectors. SDValue VSetCC = DAG.getNode(ISD::SETCC, DL, VT, N0.getOperand(0), N0.getOperand(1), N0.getOperand(2)); return DAG.getNode(ISD::AND, DL, VT, VSetCC, VecOnes); } // If the desired elements are smaller or larger than the source // elements we can use a matching integer vector type and then // truncate/sign extend. EVT MatchingVectorType = N00VT.changeVectorElementTypeToInteger(); SDValue VsetCC = DAG.getNode(ISD::SETCC, DL, MatchingVectorType, N0.getOperand(0), N0.getOperand(1), N0.getOperand(2)); return DAG.getNode(ISD::AND, DL, VT, DAG.getSExtOrTrunc(VsetCC, DL, VT), VecOnes); } // zext(setcc x,y,cc) -> select_cc x, y, 1, 0, cc SDLoc DL(N); if (SDValue SCC = SimplifySelectCC( DL, N0.getOperand(0), N0.getOperand(1), DAG.getConstant(1, DL, VT), DAG.getConstant(0, DL, VT), cast(N0.getOperand(2))->get(), true)) return SCC; } // (zext (shl (zext x), cst)) -> (shl (zext x), cst) if ((N0.getOpcode() == ISD::SHL || N0.getOpcode() == ISD::SRL) && isa(N0.getOperand(1)) && N0.getOperand(0).getOpcode() == ISD::ZERO_EXTEND && N0.hasOneUse()) { SDValue ShAmt = N0.getOperand(1); unsigned ShAmtVal = cast(ShAmt)->getZExtValue(); if (N0.getOpcode() == ISD::SHL) { SDValue InnerZExt = N0.getOperand(0); // If the original shl may be shifting out bits, do not perform this // transformation. unsigned KnownZeroBits = InnerZExt.getValueSizeInBits() - InnerZExt.getOperand(0).getValueSizeInBits(); if (ShAmtVal > KnownZeroBits) return SDValue(); } SDLoc DL(N); // Ensure that the shift amount is wide enough for the shifted value. if (VT.getSizeInBits() >= 256) ShAmt = DAG.getNode(ISD::ZERO_EXTEND, DL, MVT::i32, ShAmt); return DAG.getNode(N0.getOpcode(), DL, VT, DAG.getNode(ISD::ZERO_EXTEND, DL, VT, N0.getOperand(0)), ShAmt); } if (SDValue NewVSel = matchVSelectOpSizesWithSetCC(N)) return NewVSel; return SDValue(); } SDValue DAGCombiner::visitANY_EXTEND(SDNode *N) { SDValue N0 = N->getOperand(0); EVT VT = N->getValueType(0); if (SDValue Res = tryToFoldExtendOfConstant(N, TLI, DAG, LegalTypes)) return Res; // fold (aext (aext x)) -> (aext x) // fold (aext (zext x)) -> (zext x) // fold (aext (sext x)) -> (sext x) if (N0.getOpcode() == ISD::ANY_EXTEND || N0.getOpcode() == ISD::ZERO_EXTEND || N0.getOpcode() == ISD::SIGN_EXTEND) return DAG.getNode(N0.getOpcode(), SDLoc(N), VT, N0.getOperand(0)); // fold (aext (truncate (load x))) -> (aext (smaller load x)) // fold (aext (truncate (srl (load x), c))) -> (aext (small load (x+c/n))) if (N0.getOpcode() == ISD::TRUNCATE) { if (SDValue NarrowLoad = ReduceLoadWidth(N0.getNode())) { SDNode *oye = N0.getOperand(0).getNode(); if (NarrowLoad.getNode() != N0.getNode()) { CombineTo(N0.getNode(), NarrowLoad); // CombineTo deleted the truncate, if needed, but not what's under it. AddToWorklist(oye); } return SDValue(N, 0); // Return N so it doesn't get rechecked! } } // fold (aext (truncate x)) if (N0.getOpcode() == ISD::TRUNCATE) return DAG.getAnyExtOrTrunc(N0.getOperand(0), SDLoc(N), VT); // Fold (aext (and (trunc x), cst)) -> (and x, cst) // if the trunc is not free. if (N0.getOpcode() == ISD::AND && N0.getOperand(0).getOpcode() == ISD::TRUNCATE && N0.getOperand(1).getOpcode() == ISD::Constant && !TLI.isTruncateFree(N0.getOperand(0).getOperand(0).getValueType(), N0.getValueType())) { SDLoc DL(N); SDValue X = N0.getOperand(0).getOperand(0); X = DAG.getAnyExtOrTrunc(X, DL, VT); APInt Mask = cast(N0.getOperand(1))->getAPIntValue(); Mask = Mask.zext(VT.getSizeInBits()); return DAG.getNode(ISD::AND, DL, VT, X, DAG.getConstant(Mask, DL, VT)); } // fold (aext (load x)) -> (aext (truncate (extload x))) // None of the supported targets knows how to perform load and any_ext // on vectors in one instruction. We only perform this transformation on // scalars. if (ISD::isNON_EXTLoad(N0.getNode()) && !VT.isVector() && ISD::isUNINDEXEDLoad(N0.getNode()) && TLI.isLoadExtLegal(ISD::EXTLOAD, VT, N0.getValueType())) { bool DoXform = true; SmallVector SetCCs; if (!N0.hasOneUse()) DoXform = ExtendUsesToFormExtLoad(VT, N, N0, ISD::ANY_EXTEND, SetCCs, TLI); if (DoXform) { LoadSDNode *LN0 = cast(N0); SDValue ExtLoad = DAG.getExtLoad(ISD::EXTLOAD, SDLoc(N), VT, LN0->getChain(), LN0->getBasePtr(), N0.getValueType(), LN0->getMemOperand()); ExtendSetCCUses(SetCCs, N0, ExtLoad, ISD::ANY_EXTEND); // If the load value is used only by N, replace it via CombineTo N. bool NoReplaceTrunc = N0.hasOneUse(); CombineTo(N, ExtLoad); if (NoReplaceTrunc) { DAG.ReplaceAllUsesOfValueWith(SDValue(LN0, 1), ExtLoad.getValue(1)); } else { SDValue Trunc = DAG.getNode(ISD::TRUNCATE, SDLoc(N0), N0.getValueType(), ExtLoad); CombineTo(LN0, Trunc, ExtLoad.getValue(1)); } return SDValue(N, 0); // Return N so it doesn't get rechecked! } } // fold (aext (zextload x)) -> (aext (truncate (zextload x))) // fold (aext (sextload x)) -> (aext (truncate (sextload x))) // fold (aext ( extload x)) -> (aext (truncate (extload x))) if (N0.getOpcode() == ISD::LOAD && !ISD::isNON_EXTLoad(N0.getNode()) && ISD::isUNINDEXEDLoad(N0.getNode()) && N0.hasOneUse()) { LoadSDNode *LN0 = cast(N0); ISD::LoadExtType ExtType = LN0->getExtensionType(); EVT MemVT = LN0->getMemoryVT(); if (!LegalOperations || TLI.isLoadExtLegal(ExtType, VT, MemVT)) { SDValue ExtLoad = DAG.getExtLoad(ExtType, SDLoc(N), VT, LN0->getChain(), LN0->getBasePtr(), MemVT, LN0->getMemOperand()); CombineTo(N, ExtLoad); DAG.ReplaceAllUsesOfValueWith(SDValue(LN0, 1), ExtLoad.getValue(1)); return SDValue(N, 0); // Return N so it doesn't get rechecked! } } if (N0.getOpcode() == ISD::SETCC) { // For vectors: // aext(setcc) -> vsetcc // aext(setcc) -> truncate(vsetcc) // aext(setcc) -> aext(vsetcc) // Only do this before legalize for now. if (VT.isVector() && !LegalOperations) { EVT N00VT = N0.getOperand(0).getValueType(); if (getSetCCResultType(N00VT) == N0.getValueType()) return SDValue(); // We know that the # elements of the results is the same as the // # elements of the compare (and the # elements of the compare result // for that matter). Check to see that they are the same size. If so, // we know that the element size of the sext'd result matches the // element size of the compare operands. if (VT.getSizeInBits() == N00VT.getSizeInBits()) return DAG.getSetCC(SDLoc(N), VT, N0.getOperand(0), N0.getOperand(1), cast(N0.getOperand(2))->get()); // If the desired elements are smaller or larger than the source // elements we can use a matching integer vector type and then // truncate/any extend EVT MatchingVectorType = N00VT.changeVectorElementTypeToInteger(); SDValue VsetCC = DAG.getSetCC(SDLoc(N), MatchingVectorType, N0.getOperand(0), N0.getOperand(1), cast(N0.getOperand(2))->get()); return DAG.getAnyExtOrTrunc(VsetCC, SDLoc(N), VT); } // aext(setcc x,y,cc) -> select_cc x, y, 1, 0, cc SDLoc DL(N); if (SDValue SCC = SimplifySelectCC( DL, N0.getOperand(0), N0.getOperand(1), DAG.getConstant(1, DL, VT), DAG.getConstant(0, DL, VT), cast(N0.getOperand(2))->get(), true)) return SCC; } return SDValue(); } SDValue DAGCombiner::visitAssertExt(SDNode *N) { unsigned Opcode = N->getOpcode(); SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); EVT AssertVT = cast(N1)->getVT(); // fold (assert?ext (assert?ext x, vt), vt) -> (assert?ext x, vt) if (N0.getOpcode() == Opcode && AssertVT == cast(N0.getOperand(1))->getVT()) return N0; if (N0.getOpcode() == ISD::TRUNCATE && N0.hasOneUse() && N0.getOperand(0).getOpcode() == Opcode) { // We have an assert, truncate, assert sandwich. Make one stronger assert // by asserting on the smallest asserted type to the larger source type. // This eliminates the later assert: // assert (trunc (assert X, i8) to iN), i1 --> trunc (assert X, i1) to iN // assert (trunc (assert X, i1) to iN), i8 --> trunc (assert X, i1) to iN SDValue BigA = N0.getOperand(0); EVT BigA_AssertVT = cast(BigA.getOperand(1))->getVT(); assert(BigA_AssertVT.bitsLE(N0.getValueType()) && "Asserting zero/sign-extended bits to a type larger than the " "truncated destination does not provide information"); SDLoc DL(N); EVT MinAssertVT = AssertVT.bitsLT(BigA_AssertVT) ? AssertVT : BigA_AssertVT; SDValue MinAssertVTVal = DAG.getValueType(MinAssertVT); SDValue NewAssert = DAG.getNode(Opcode, DL, BigA.getValueType(), BigA.getOperand(0), MinAssertVTVal); return DAG.getNode(ISD::TRUNCATE, DL, N->getValueType(0), NewAssert); } // If we have (AssertZext (truncate (AssertSext X, iX)), iY) and Y is smaller // than X. Just move the AssertZext in front of the truncate and drop the // AssertSExt. if (N0.getOpcode() == ISD::TRUNCATE && N0.hasOneUse() && N0.getOperand(0).getOpcode() == ISD::AssertSext && Opcode == ISD::AssertZext) { SDValue BigA = N0.getOperand(0); EVT BigA_AssertVT = cast(BigA.getOperand(1))->getVT(); assert(BigA_AssertVT.bitsLE(N0.getValueType()) && "Asserting zero/sign-extended bits to a type larger than the " "truncated destination does not provide information"); if (AssertVT.bitsLT(BigA_AssertVT)) { SDLoc DL(N); SDValue NewAssert = DAG.getNode(Opcode, DL, BigA.getValueType(), BigA.getOperand(0), N1); return DAG.getNode(ISD::TRUNCATE, DL, N->getValueType(0), NewAssert); } } return SDValue(); } /// If the result of a wider load is shifted to right of N bits and then /// truncated to a narrower type and where N is a multiple of number of bits of /// the narrower type, transform it to a narrower load from address + N / num of /// bits of new type. Also narrow the load if the result is masked with an AND /// to effectively produce a smaller type. If the result is to be extended, also /// fold the extension to form a extending load. SDValue DAGCombiner::ReduceLoadWidth(SDNode *N) { unsigned Opc = N->getOpcode(); ISD::LoadExtType ExtType = ISD::NON_EXTLOAD; SDValue N0 = N->getOperand(0); EVT VT = N->getValueType(0); EVT ExtVT = VT; // This transformation isn't valid for vector loads. if (VT.isVector()) return SDValue(); unsigned ShAmt = 0; bool HasShiftedOffset = false; // Special case: SIGN_EXTEND_INREG is basically truncating to ExtVT then // extended to VT. if (Opc == ISD::SIGN_EXTEND_INREG) { ExtType = ISD::SEXTLOAD; ExtVT = cast(N->getOperand(1))->getVT(); } else if (Opc == ISD::SRL) { // Another special-case: SRL is basically zero-extending a narrower value, // or it maybe shifting a higher subword, half or byte into the lowest // bits. ExtType = ISD::ZEXTLOAD; N0 = SDValue(N, 0); auto *LN0 = dyn_cast(N0.getOperand(0)); auto *N01 = dyn_cast(N0.getOperand(1)); if (!N01 || !LN0) return SDValue(); uint64_t ShiftAmt = N01->getZExtValue(); uint64_t MemoryWidth = LN0->getMemoryVT().getSizeInBits(); if (LN0->getExtensionType() != ISD::SEXTLOAD && MemoryWidth > ShiftAmt) ExtVT = EVT::getIntegerVT(*DAG.getContext(), MemoryWidth - ShiftAmt); else ExtVT = EVT::getIntegerVT(*DAG.getContext(), VT.getSizeInBits() - ShiftAmt); } else if (Opc == ISD::AND) { // An AND with a constant mask is the same as a truncate + zero-extend. auto AndC = dyn_cast(N->getOperand(1)); if (!AndC) return SDValue(); const APInt &Mask = AndC->getAPIntValue(); unsigned ActiveBits = 0; if (Mask.isMask()) { ActiveBits = Mask.countTrailingOnes(); } else if (Mask.isShiftedMask()) { ShAmt = Mask.countTrailingZeros(); APInt ShiftedMask = Mask.lshr(ShAmt); ActiveBits = ShiftedMask.countTrailingOnes(); HasShiftedOffset = true; } else return SDValue(); ExtType = ISD::ZEXTLOAD; ExtVT = EVT::getIntegerVT(*DAG.getContext(), ActiveBits); } if (N0.getOpcode() == ISD::SRL && N0.hasOneUse()) { SDValue SRL = N0; if (auto *ConstShift = dyn_cast(SRL.getOperand(1))) { ShAmt = ConstShift->getZExtValue(); unsigned EVTBits = ExtVT.getSizeInBits(); // Is the shift amount a multiple of size of VT? if ((ShAmt & (EVTBits-1)) == 0) { N0 = N0.getOperand(0); // Is the load width a multiple of size of VT? if ((N0.getValueSizeInBits() & (EVTBits-1)) != 0) return SDValue(); } // At this point, we must have a load or else we can't do the transform. if (!isa(N0)) return SDValue(); auto *LN0 = cast(N0); // Because a SRL must be assumed to *need* to zero-extend the high bits // (as opposed to anyext the high bits), we can't combine the zextload // lowering of SRL and an sextload. if (LN0->getExtensionType() == ISD::SEXTLOAD) return SDValue(); // If the shift amount is larger than the input type then we're not // accessing any of the loaded bytes. If the load was a zextload/extload // then the result of the shift+trunc is zero/undef (handled elsewhere). if (ShAmt >= LN0->getMemoryVT().getSizeInBits()) return SDValue(); // If the SRL is only used by a masking AND, we may be able to adjust // the ExtVT to make the AND redundant. SDNode *Mask = *(SRL->use_begin()); if (Mask->getOpcode() == ISD::AND && isa(Mask->getOperand(1))) { const APInt &ShiftMask = cast(Mask->getOperand(1))->getAPIntValue(); if (ShiftMask.isMask()) { EVT MaskedVT = EVT::getIntegerVT(*DAG.getContext(), ShiftMask.countTrailingOnes()); // If the mask is smaller, recompute the type. if ((ExtVT.getSizeInBits() > MaskedVT.getSizeInBits()) && TLI.isLoadExtLegal(ExtType, N0.getValueType(), MaskedVT)) ExtVT = MaskedVT; } } } } // If the load is shifted left (and the result isn't shifted back right), // we can fold the truncate through the shift. unsigned ShLeftAmt = 0; if (ShAmt == 0 && N0.getOpcode() == ISD::SHL && N0.hasOneUse() && ExtVT == VT && TLI.isNarrowingProfitable(N0.getValueType(), VT)) { if (ConstantSDNode *N01 = dyn_cast(N0.getOperand(1))) { ShLeftAmt = N01->getZExtValue(); N0 = N0.getOperand(0); } } // If we haven't found a load, we can't narrow it. if (!isa(N0)) return SDValue(); LoadSDNode *LN0 = cast(N0); if (!isLegalNarrowLdSt(LN0, ExtType, ExtVT, ShAmt)) return SDValue(); auto AdjustBigEndianShift = [&](unsigned ShAmt) { unsigned LVTStoreBits = LN0->getMemoryVT().getStoreSizeInBits(); unsigned EVTStoreBits = ExtVT.getStoreSizeInBits(); return LVTStoreBits - EVTStoreBits - ShAmt; }; // For big endian targets, we need to adjust the offset to the pointer to // load the correct bytes. if (DAG.getDataLayout().isBigEndian()) ShAmt = AdjustBigEndianShift(ShAmt); EVT PtrType = N0.getOperand(1).getValueType(); uint64_t PtrOff = ShAmt / 8; unsigned NewAlign = MinAlign(LN0->getAlignment(), PtrOff); SDLoc DL(LN0); // The original load itself didn't wrap, so an offset within it doesn't. SDNodeFlags Flags; Flags.setNoUnsignedWrap(true); SDValue NewPtr = DAG.getNode(ISD::ADD, DL, PtrType, LN0->getBasePtr(), DAG.getConstant(PtrOff, DL, PtrType), Flags); AddToWorklist(NewPtr.getNode()); SDValue Load; if (ExtType == ISD::NON_EXTLOAD) Load = DAG.getLoad(VT, SDLoc(N0), LN0->getChain(), NewPtr, LN0->getPointerInfo().getWithOffset(PtrOff), NewAlign, LN0->getMemOperand()->getFlags(), LN0->getAAInfo()); else Load = DAG.getExtLoad(ExtType, SDLoc(N0), VT, LN0->getChain(), NewPtr, LN0->getPointerInfo().getWithOffset(PtrOff), ExtVT, NewAlign, LN0->getMemOperand()->getFlags(), LN0->getAAInfo()); // Replace the old load's chain with the new load's chain. WorklistRemover DeadNodes(*this); DAG.ReplaceAllUsesOfValueWith(N0.getValue(1), Load.getValue(1)); // Shift the result left, if we've swallowed a left shift. SDValue Result = Load; if (ShLeftAmt != 0) { EVT ShImmTy = getShiftAmountTy(Result.getValueType()); if (!isUIntN(ShImmTy.getSizeInBits(), ShLeftAmt)) ShImmTy = VT; // If the shift amount is as large as the result size (but, presumably, // no larger than the source) then the useful bits of the result are // zero; we can't simply return the shortened shift, because the result // of that operation is undefined. SDLoc DL(N0); if (ShLeftAmt >= VT.getSizeInBits()) Result = DAG.getConstant(0, DL, VT); else Result = DAG.getNode(ISD::SHL, DL, VT, Result, DAG.getConstant(ShLeftAmt, DL, ShImmTy)); } if (HasShiftedOffset) { // Recalculate the shift amount after it has been altered to calculate // the offset. if (DAG.getDataLayout().isBigEndian()) ShAmt = AdjustBigEndianShift(ShAmt); // We're using a shifted mask, so the load now has an offset. This means // that data has been loaded into the lower bytes than it would have been // before, so we need to shl the loaded data into the correct position in the // register. SDValue ShiftC = DAG.getConstant(ShAmt, DL, VT); Result = DAG.getNode(ISD::SHL, DL, VT, Result, ShiftC); DAG.ReplaceAllUsesOfValueWith(SDValue(N, 0), Result); } // Return the new loaded value. return Result; } SDValue DAGCombiner::visitSIGN_EXTEND_INREG(SDNode *N) { SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); EVT VT = N->getValueType(0); EVT EVT = cast(N1)->getVT(); unsigned VTBits = VT.getScalarSizeInBits(); unsigned EVTBits = EVT.getScalarSizeInBits(); if (N0.isUndef()) return DAG.getUNDEF(VT); // fold (sext_in_reg c1) -> c1 if (DAG.isConstantIntBuildVectorOrConstantInt(N0)) return DAG.getNode(ISD::SIGN_EXTEND_INREG, SDLoc(N), VT, N0, N1); // If the input is already sign extended, just drop the extension. if (DAG.ComputeNumSignBits(N0) >= VTBits-EVTBits+1) return N0; // fold (sext_in_reg (sext_in_reg x, VT2), VT1) -> (sext_in_reg x, minVT) pt2 if (N0.getOpcode() == ISD::SIGN_EXTEND_INREG && EVT.bitsLT(cast(N0.getOperand(1))->getVT())) return DAG.getNode(ISD::SIGN_EXTEND_INREG, SDLoc(N), VT, N0.getOperand(0), N1); // fold (sext_in_reg (sext x)) -> (sext x) // fold (sext_in_reg (aext x)) -> (sext x) // if x is small enough or if we know that x has more than 1 sign bit and the // sign_extend_inreg is extending from one of them. if (N0.getOpcode() == ISD::SIGN_EXTEND || N0.getOpcode() == ISD::ANY_EXTEND) { SDValue N00 = N0.getOperand(0); unsigned N00Bits = N00.getScalarValueSizeInBits(); if ((N00Bits <= EVTBits || (N00Bits - DAG.ComputeNumSignBits(N00)) < EVTBits) && (!LegalOperations || TLI.isOperationLegal(ISD::SIGN_EXTEND, VT))) return DAG.getNode(ISD::SIGN_EXTEND, SDLoc(N), VT, N00); } // fold (sext_in_reg (*_extend_vector_inreg x)) -> (sext_vector_inreg x) if ((N0.getOpcode() == ISD::ANY_EXTEND_VECTOR_INREG || N0.getOpcode() == ISD::SIGN_EXTEND_VECTOR_INREG || N0.getOpcode() == ISD::ZERO_EXTEND_VECTOR_INREG) && N0.getOperand(0).getScalarValueSizeInBits() == EVTBits) { if (!LegalOperations || TLI.isOperationLegal(ISD::SIGN_EXTEND_VECTOR_INREG, VT)) return DAG.getNode(ISD::SIGN_EXTEND_VECTOR_INREG, SDLoc(N), VT, N0.getOperand(0)); } // fold (sext_in_reg (zext x)) -> (sext x) // iff we are extending the source sign bit. if (N0.getOpcode() == ISD::ZERO_EXTEND) { SDValue N00 = N0.getOperand(0); if (N00.getScalarValueSizeInBits() == EVTBits && (!LegalOperations || TLI.isOperationLegal(ISD::SIGN_EXTEND, VT))) return DAG.getNode(ISD::SIGN_EXTEND, SDLoc(N), VT, N00, N1); } // fold (sext_in_reg x) -> (zext_in_reg x) if the sign bit is known zero. if (DAG.MaskedValueIsZero(N0, APInt::getOneBitSet(VTBits, EVTBits - 1))) return DAG.getZeroExtendInReg(N0, SDLoc(N), EVT.getScalarType()); // fold operands of sext_in_reg based on knowledge that the top bits are not // demanded. if (SimplifyDemandedBits(SDValue(N, 0))) return SDValue(N, 0); // fold (sext_in_reg (load x)) -> (smaller sextload x) // fold (sext_in_reg (srl (load x), c)) -> (smaller sextload (x+c/evtbits)) if (SDValue NarrowLoad = ReduceLoadWidth(N)) return NarrowLoad; // fold (sext_in_reg (srl X, 24), i8) -> (sra X, 24) // fold (sext_in_reg (srl X, 23), i8) -> (sra X, 23) iff possible. // We already fold "(sext_in_reg (srl X, 25), i8) -> srl X, 25" above. if (N0.getOpcode() == ISD::SRL) { if (ConstantSDNode *ShAmt = dyn_cast(N0.getOperand(1))) if (ShAmt->getZExtValue()+EVTBits <= VTBits) { // We can turn this into an SRA iff the input to the SRL is already sign // extended enough. unsigned InSignBits = DAG.ComputeNumSignBits(N0.getOperand(0)); if (VTBits-(ShAmt->getZExtValue()+EVTBits) < InSignBits) return DAG.getNode(ISD::SRA, SDLoc(N), VT, N0.getOperand(0), N0.getOperand(1)); } } // fold (sext_inreg (extload x)) -> (sextload x) // If sextload is not supported by target, we can only do the combine when // load has one use. Doing otherwise can block folding the extload with other // extends that the target does support. if (ISD::isEXTLoad(N0.getNode()) && ISD::isUNINDEXEDLoad(N0.getNode()) && EVT == cast(N0)->getMemoryVT() && ((!LegalOperations && !cast(N0)->isVolatile() && N0.hasOneUse()) || TLI.isLoadExtLegal(ISD::SEXTLOAD, VT, EVT))) { LoadSDNode *LN0 = cast(N0); SDValue ExtLoad = DAG.getExtLoad(ISD::SEXTLOAD, SDLoc(N), VT, LN0->getChain(), LN0->getBasePtr(), EVT, LN0->getMemOperand()); CombineTo(N, ExtLoad); CombineTo(N0.getNode(), ExtLoad, ExtLoad.getValue(1)); AddToWorklist(ExtLoad.getNode()); return SDValue(N, 0); // Return N so it doesn't get rechecked! } // fold (sext_inreg (zextload x)) -> (sextload x) iff load has one use if (ISD::isZEXTLoad(N0.getNode()) && ISD::isUNINDEXEDLoad(N0.getNode()) && N0.hasOneUse() && EVT == cast(N0)->getMemoryVT() && ((!LegalOperations && !cast(N0)->isVolatile()) || TLI.isLoadExtLegal(ISD::SEXTLOAD, VT, EVT))) { LoadSDNode *LN0 = cast(N0); SDValue ExtLoad = DAG.getExtLoad(ISD::SEXTLOAD, SDLoc(N), VT, LN0->getChain(), LN0->getBasePtr(), EVT, LN0->getMemOperand()); CombineTo(N, ExtLoad); CombineTo(N0.getNode(), ExtLoad, ExtLoad.getValue(1)); return SDValue(N, 0); // Return N so it doesn't get rechecked! } // Form (sext_inreg (bswap >> 16)) or (sext_inreg (rotl (bswap) 16)) if (EVTBits <= 16 && N0.getOpcode() == ISD::OR) { if (SDValue BSwap = MatchBSwapHWordLow(N0.getNode(), N0.getOperand(0), N0.getOperand(1), false)) return DAG.getNode(ISD::SIGN_EXTEND_INREG, SDLoc(N), VT, BSwap, N1); } return SDValue(); } SDValue DAGCombiner::visitSIGN_EXTEND_VECTOR_INREG(SDNode *N) { SDValue N0 = N->getOperand(0); EVT VT = N->getValueType(0); if (N0.isUndef()) return DAG.getUNDEF(VT); if (SDValue Res = tryToFoldExtendOfConstant(N, TLI, DAG, LegalTypes)) return Res; if (SimplifyDemandedVectorElts(SDValue(N, 0))) return SDValue(N, 0); return SDValue(); } SDValue DAGCombiner::visitZERO_EXTEND_VECTOR_INREG(SDNode *N) { SDValue N0 = N->getOperand(0); EVT VT = N->getValueType(0); if (N0.isUndef()) return DAG.getUNDEF(VT); if (SDValue Res = tryToFoldExtendOfConstant(N, TLI, DAG, LegalTypes)) return Res; if (SimplifyDemandedVectorElts(SDValue(N, 0))) return SDValue(N, 0); return SDValue(); } SDValue DAGCombiner::visitTRUNCATE(SDNode *N) { SDValue N0 = N->getOperand(0); EVT VT = N->getValueType(0); bool isLE = DAG.getDataLayout().isLittleEndian(); // noop truncate if (N0.getValueType() == N->getValueType(0)) return N0; // fold (truncate (truncate x)) -> (truncate x) if (N0.getOpcode() == ISD::TRUNCATE) return DAG.getNode(ISD::TRUNCATE, SDLoc(N), VT, N0.getOperand(0)); // fold (truncate c1) -> c1 if (DAG.isConstantIntBuildVectorOrConstantInt(N0)) { SDValue C = DAG.getNode(ISD::TRUNCATE, SDLoc(N), VT, N0); if (C.getNode() != N) return C; } // fold (truncate (ext x)) -> (ext x) or (truncate x) or x if (N0.getOpcode() == ISD::ZERO_EXTEND || N0.getOpcode() == ISD::SIGN_EXTEND || N0.getOpcode() == ISD::ANY_EXTEND) { // if the source is smaller than the dest, we still need an extend. if (N0.getOperand(0).getValueType().bitsLT(VT)) return DAG.getNode(N0.getOpcode(), SDLoc(N), VT, N0.getOperand(0)); // if the source is larger than the dest, than we just need the truncate. if (N0.getOperand(0).getValueType().bitsGT(VT)) return DAG.getNode(ISD::TRUNCATE, SDLoc(N), VT, N0.getOperand(0)); // if the source and dest are the same type, we can drop both the extend // and the truncate. return N0.getOperand(0); } // If this is anyext(trunc), don't fold it, allow ourselves to be folded. if (N->hasOneUse() && (N->use_begin()->getOpcode() == ISD::ANY_EXTEND)) return SDValue(); // Fold extract-and-trunc into a narrow extract. For example: // i64 x = EXTRACT_VECTOR_ELT(v2i64 val, i32 1) // i32 y = TRUNCATE(i64 x) // -- becomes -- // v16i8 b = BITCAST (v2i64 val) // i8 x = EXTRACT_VECTOR_ELT(v16i8 b, i32 8) // // Note: We only run this optimization after type legalization (which often // creates this pattern) and before operation legalization after which // we need to be more careful about the vector instructions that we generate. if (N0.getOpcode() == ISD::EXTRACT_VECTOR_ELT && LegalTypes && !LegalOperations && N0->hasOneUse() && VT != MVT::i1) { EVT VecTy = N0.getOperand(0).getValueType(); EVT ExTy = N0.getValueType(); EVT TrTy = N->getValueType(0); unsigned NumElem = VecTy.getVectorNumElements(); unsigned SizeRatio = ExTy.getSizeInBits()/TrTy.getSizeInBits(); EVT NVT = EVT::getVectorVT(*DAG.getContext(), TrTy, SizeRatio * NumElem); assert(NVT.getSizeInBits() == VecTy.getSizeInBits() && "Invalid Size"); SDValue EltNo = N0->getOperand(1); if (isa(EltNo) && isTypeLegal(NVT)) { int Elt = cast(EltNo)->getZExtValue(); EVT IndexTy = TLI.getVectorIdxTy(DAG.getDataLayout()); int Index = isLE ? (Elt*SizeRatio) : (Elt*SizeRatio + (SizeRatio-1)); SDLoc DL(N); return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, TrTy, DAG.getBitcast(NVT, N0.getOperand(0)), DAG.getConstant(Index, DL, IndexTy)); } } // trunc (select c, a, b) -> select c, (trunc a), (trunc b) if (N0.getOpcode() == ISD::SELECT && N0.hasOneUse()) { EVT SrcVT = N0.getValueType(); if ((!LegalOperations || TLI.isOperationLegal(ISD::SELECT, SrcVT)) && TLI.isTruncateFree(SrcVT, VT)) { SDLoc SL(N0); SDValue Cond = N0.getOperand(0); SDValue TruncOp0 = DAG.getNode(ISD::TRUNCATE, SL, VT, N0.getOperand(1)); SDValue TruncOp1 = DAG.getNode(ISD::TRUNCATE, SL, VT, N0.getOperand(2)); return DAG.getNode(ISD::SELECT, SDLoc(N), VT, Cond, TruncOp0, TruncOp1); } } // trunc (shl x, K) -> shl (trunc x), K => K < VT.getScalarSizeInBits() if (N0.getOpcode() == ISD::SHL && N0.hasOneUse() && (!LegalOperations || TLI.isOperationLegalOrCustom(ISD::SHL, VT)) && TLI.isTypeDesirableForOp(ISD::SHL, VT)) { SDValue Amt = N0.getOperand(1); KnownBits Known = DAG.computeKnownBits(Amt); unsigned Size = VT.getScalarSizeInBits(); if (Known.getBitWidth() - Known.countMinLeadingZeros() <= Log2_32(Size)) { SDLoc SL(N); EVT AmtVT = TLI.getShiftAmountTy(VT, DAG.getDataLayout()); SDValue Trunc = DAG.getNode(ISD::TRUNCATE, SL, VT, N0.getOperand(0)); if (AmtVT != Amt.getValueType()) { Amt = DAG.getZExtOrTrunc(Amt, SL, AmtVT); AddToWorklist(Amt.getNode()); } return DAG.getNode(ISD::SHL, SL, VT, Trunc, Amt); } } // Fold a series of buildvector, bitcast, and truncate if possible. // For example fold // (2xi32 trunc (bitcast ((4xi32)buildvector x, x, y, y) 2xi64)) to // (2xi32 (buildvector x, y)). if (Level == AfterLegalizeVectorOps && VT.isVector() && N0.getOpcode() == ISD::BITCAST && N0.hasOneUse() && N0.getOperand(0).getOpcode() == ISD::BUILD_VECTOR && N0.getOperand(0).hasOneUse()) { SDValue BuildVect = N0.getOperand(0); EVT BuildVectEltTy = BuildVect.getValueType().getVectorElementType(); EVT TruncVecEltTy = VT.getVectorElementType(); // Check that the element types match. if (BuildVectEltTy == TruncVecEltTy) { // Now we only need to compute the offset of the truncated elements. unsigned BuildVecNumElts = BuildVect.getNumOperands(); unsigned TruncVecNumElts = VT.getVectorNumElements(); unsigned TruncEltOffset = BuildVecNumElts / TruncVecNumElts; assert((BuildVecNumElts % TruncVecNumElts) == 0 && "Invalid number of elements"); SmallVector Opnds; for (unsigned i = 0, e = BuildVecNumElts; i != e; i += TruncEltOffset) Opnds.push_back(BuildVect.getOperand(i)); return DAG.getBuildVector(VT, SDLoc(N), Opnds); } } // See if we can simplify the input to this truncate through knowledge that // only the low bits are being used. // For example "trunc (or (shl x, 8), y)" // -> trunc y // Currently we only perform this optimization on scalars because vectors // may have different active low bits. if (!VT.isVector()) { APInt Mask = APInt::getLowBitsSet(N0.getValueSizeInBits(), VT.getSizeInBits()); if (SDValue Shorter = DAG.GetDemandedBits(N0, Mask)) return DAG.getNode(ISD::TRUNCATE, SDLoc(N), VT, Shorter); } // fold (truncate (load x)) -> (smaller load x) // fold (truncate (srl (load x), c)) -> (smaller load (x+c/evtbits)) if (!LegalTypes || TLI.isTypeDesirableForOp(N0.getOpcode(), VT)) { if (SDValue Reduced = ReduceLoadWidth(N)) return Reduced; // Handle the case where the load remains an extending load even // after truncation. if (N0.hasOneUse() && ISD::isUNINDEXEDLoad(N0.getNode())) { LoadSDNode *LN0 = cast(N0); if (!LN0->isVolatile() && LN0->getMemoryVT().getStoreSizeInBits() < VT.getSizeInBits()) { SDValue NewLoad = DAG.getExtLoad(LN0->getExtensionType(), SDLoc(LN0), VT, LN0->getChain(), LN0->getBasePtr(), LN0->getMemoryVT(), LN0->getMemOperand()); DAG.ReplaceAllUsesOfValueWith(N0.getValue(1), NewLoad.getValue(1)); return NewLoad; } } } // fold (trunc (concat ... x ...)) -> (concat ..., (trunc x), ...)), // where ... are all 'undef'. if (N0.getOpcode() == ISD::CONCAT_VECTORS && !LegalTypes) { SmallVector VTs; SDValue V; unsigned Idx = 0; unsigned NumDefs = 0; for (unsigned i = 0, e = N0.getNumOperands(); i != e; ++i) { SDValue X = N0.getOperand(i); if (!X.isUndef()) { V = X; Idx = i; NumDefs++; } // Stop if more than one members are non-undef. if (NumDefs > 1) break; VTs.push_back(EVT::getVectorVT(*DAG.getContext(), VT.getVectorElementType(), X.getValueType().getVectorNumElements())); } if (NumDefs == 0) return DAG.getUNDEF(VT); if (NumDefs == 1) { assert(V.getNode() && "The single defined operand is empty!"); SmallVector Opnds; for (unsigned i = 0, e = VTs.size(); i != e; ++i) { if (i != Idx) { Opnds.push_back(DAG.getUNDEF(VTs[i])); continue; } SDValue NV = DAG.getNode(ISD::TRUNCATE, SDLoc(V), VTs[i], V); AddToWorklist(NV.getNode()); Opnds.push_back(NV); } return DAG.getNode(ISD::CONCAT_VECTORS, SDLoc(N), VT, Opnds); } } // Fold truncate of a bitcast of a vector to an extract of the low vector // element. // // e.g. trunc (i64 (bitcast v2i32:x)) -> extract_vector_elt v2i32:x, idx if (N0.getOpcode() == ISD::BITCAST && !VT.isVector()) { SDValue VecSrc = N0.getOperand(0); EVT SrcVT = VecSrc.getValueType(); if (SrcVT.isVector() && SrcVT.getScalarType() == VT && (!LegalOperations || TLI.isOperationLegal(ISD::EXTRACT_VECTOR_ELT, SrcVT))) { SDLoc SL(N); EVT IdxVT = TLI.getVectorIdxTy(DAG.getDataLayout()); unsigned Idx = isLE ? 0 : SrcVT.getVectorNumElements() - 1; return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, VT, VecSrc, DAG.getConstant(Idx, SL, IdxVT)); } } // Simplify the operands using demanded-bits information. if (!VT.isVector() && SimplifyDemandedBits(SDValue(N, 0))) return SDValue(N, 0); // (trunc adde(X, Y, Carry)) -> (adde trunc(X), trunc(Y), Carry) // (trunc addcarry(X, Y, Carry)) -> (addcarry trunc(X), trunc(Y), Carry) // When the adde's carry is not used. if ((N0.getOpcode() == ISD::ADDE || N0.getOpcode() == ISD::ADDCARRY) && N0.hasOneUse() && !N0.getNode()->hasAnyUseOfValue(1) && (!LegalOperations || TLI.isOperationLegal(N0.getOpcode(), VT))) { SDLoc SL(N); auto X = DAG.getNode(ISD::TRUNCATE, SL, VT, N0.getOperand(0)); auto Y = DAG.getNode(ISD::TRUNCATE, SL, VT, N0.getOperand(1)); auto VTs = DAG.getVTList(VT, N0->getValueType(1)); return DAG.getNode(N0.getOpcode(), SL, VTs, X, Y, N0.getOperand(2)); } // fold (truncate (extract_subvector(ext x))) -> // (extract_subvector x) // TODO: This can be generalized to cover cases where the truncate and extract // do not fully cancel each other out. if (!LegalTypes && N0.getOpcode() == ISD::EXTRACT_SUBVECTOR) { SDValue N00 = N0.getOperand(0); if (N00.getOpcode() == ISD::SIGN_EXTEND || N00.getOpcode() == ISD::ZERO_EXTEND || N00.getOpcode() == ISD::ANY_EXTEND) { if (N00.getOperand(0)->getValueType(0).getVectorElementType() == VT.getVectorElementType()) return DAG.getNode(ISD::EXTRACT_SUBVECTOR, SDLoc(N0->getOperand(0)), VT, N00.getOperand(0), N0.getOperand(1)); } } if (SDValue NewVSel = matchVSelectOpSizesWithSetCC(N)) return NewVSel; // Narrow a suitable binary operation with a non-opaque constant operand by // moving it ahead of the truncate. This is limited to pre-legalization // because targets may prefer a wider type during later combines and invert // this transform. switch (N0.getOpcode()) { case ISD::ADD: case ISD::SUB: case ISD::MUL: case ISD::AND: case ISD::OR: case ISD::XOR: if (!LegalOperations && N0.hasOneUse() && (isConstantOrConstantVector(N0.getOperand(0), true) || isConstantOrConstantVector(N0.getOperand(1), true))) { // TODO: We already restricted this to pre-legalization, but for vectors // we are extra cautious to not create an unsupported operation. // Target-specific changes are likely needed to avoid regressions here. if (VT.isScalarInteger() || TLI.isOperationLegal(N0.getOpcode(), VT)) { SDLoc DL(N); SDValue NarrowL = DAG.getNode(ISD::TRUNCATE, DL, VT, N0.getOperand(0)); SDValue NarrowR = DAG.getNode(ISD::TRUNCATE, DL, VT, N0.getOperand(1)); return DAG.getNode(N0.getOpcode(), DL, VT, NarrowL, NarrowR); } } } return SDValue(); } static SDNode *getBuildPairElt(SDNode *N, unsigned i) { SDValue Elt = N->getOperand(i); if (Elt.getOpcode() != ISD::MERGE_VALUES) return Elt.getNode(); return Elt.getOperand(Elt.getResNo()).getNode(); } /// build_pair (load, load) -> load /// if load locations are consecutive. SDValue DAGCombiner::CombineConsecutiveLoads(SDNode *N, EVT VT) { assert(N->getOpcode() == ISD::BUILD_PAIR); LoadSDNode *LD1 = dyn_cast(getBuildPairElt(N, 0)); LoadSDNode *LD2 = dyn_cast(getBuildPairElt(N, 1)); // A BUILD_PAIR is always having the least significant part in elt 0 and the // most significant part in elt 1. So when combining into one large load, we // need to consider the endianness. if (DAG.getDataLayout().isBigEndian()) std::swap(LD1, LD2); if (!LD1 || !LD2 || !ISD::isNON_EXTLoad(LD1) || !LD1->hasOneUse() || LD1->getAddressSpace() != LD2->getAddressSpace()) return SDValue(); EVT LD1VT = LD1->getValueType(0); unsigned LD1Bytes = LD1VT.getStoreSize(); if (ISD::isNON_EXTLoad(LD2) && LD2->hasOneUse() && DAG.areNonVolatileConsecutiveLoads(LD2, LD1, LD1Bytes, 1)) { unsigned Align = LD1->getAlignment(); unsigned NewAlign = DAG.getDataLayout().getABITypeAlignment( VT.getTypeForEVT(*DAG.getContext())); if (NewAlign <= Align && (!LegalOperations || TLI.isOperationLegal(ISD::LOAD, VT))) return DAG.getLoad(VT, SDLoc(N), LD1->getChain(), LD1->getBasePtr(), LD1->getPointerInfo(), Align); } return SDValue(); } static unsigned getPPCf128HiElementSelector(const SelectionDAG &DAG) { // On little-endian machines, bitcasting from ppcf128 to i128 does swap the Hi // and Lo parts; on big-endian machines it doesn't. return DAG.getDataLayout().isBigEndian() ? 1 : 0; } static SDValue foldBitcastedFPLogic(SDNode *N, SelectionDAG &DAG, const TargetLowering &TLI) { // If this is not a bitcast to an FP type or if the target doesn't have // IEEE754-compliant FP logic, we're done. EVT VT = N->getValueType(0); if (!VT.isFloatingPoint() || !TLI.hasBitPreservingFPLogic(VT)) return SDValue(); // TODO: Handle cases where the integer constant is a different scalar // bitwidth to the FP. SDValue N0 = N->getOperand(0); EVT SourceVT = N0.getValueType(); if (VT.getScalarSizeInBits() != SourceVT.getScalarSizeInBits()) return SDValue(); unsigned FPOpcode; APInt SignMask; switch (N0.getOpcode()) { case ISD::AND: FPOpcode = ISD::FABS; SignMask = ~APInt::getSignMask(SourceVT.getScalarSizeInBits()); break; case ISD::XOR: FPOpcode = ISD::FNEG; SignMask = APInt::getSignMask(SourceVT.getScalarSizeInBits()); break; case ISD::OR: FPOpcode = ISD::FABS; SignMask = APInt::getSignMask(SourceVT.getScalarSizeInBits()); break; default: return SDValue(); } // Fold (bitcast int (and (bitcast fp X to int), 0x7fff...) to fp) -> fabs X // Fold (bitcast int (xor (bitcast fp X to int), 0x8000...) to fp) -> fneg X // Fold (bitcast int (or (bitcast fp X to int), 0x8000...) to fp) -> // fneg (fabs X) SDValue LogicOp0 = N0.getOperand(0); ConstantSDNode *LogicOp1 = isConstOrConstSplat(N0.getOperand(1), true); if (LogicOp1 && LogicOp1->getAPIntValue() == SignMask && LogicOp0.getOpcode() == ISD::BITCAST && LogicOp0.getOperand(0).getValueType() == VT) { SDValue FPOp = DAG.getNode(FPOpcode, SDLoc(N), VT, LogicOp0.getOperand(0)); NumFPLogicOpsConv++; if (N0.getOpcode() == ISD::OR) return DAG.getNode(ISD::FNEG, SDLoc(N), VT, FPOp); return FPOp; } return SDValue(); } SDValue DAGCombiner::visitBITCAST(SDNode *N) { SDValue N0 = N->getOperand(0); EVT VT = N->getValueType(0); if (N0.isUndef()) return DAG.getUNDEF(VT); // If the input is a BUILD_VECTOR with all constant elements, fold this now. // Only do this before legalize types, since we might create an illegal // scalar type. Even if we knew we wouldn't create an illegal scalar type // we can only do this before legalize ops, since the target maybe // depending on the bitcast. // First check to see if this is all constant. if (!LegalTypes && N0.getOpcode() == ISD::BUILD_VECTOR && N0.getNode()->hasOneUse() && VT.isVector() && cast(N0)->isConstant()) return ConstantFoldBITCASTofBUILD_VECTOR(N0.getNode(), VT.getVectorElementType()); // If the input is a constant, let getNode fold it. if (isa(N0) || isa(N0)) { // If we can't allow illegal operations, we need to check that this is just // a fp -> int or int -> conversion and that the resulting operation will // be legal. if (!LegalOperations || (isa(N0) && VT.isFloatingPoint() && !VT.isVector() && TLI.isOperationLegal(ISD::ConstantFP, VT)) || (isa(N0) && VT.isInteger() && !VT.isVector() && TLI.isOperationLegal(ISD::Constant, VT))) { SDValue C = DAG.getBitcast(VT, N0); if (C.getNode() != N) return C; } } // (conv (conv x, t1), t2) -> (conv x, t2) if (N0.getOpcode() == ISD::BITCAST) return DAG.getBitcast(VT, N0.getOperand(0)); // fold (conv (load x)) -> (load (conv*)x) // If the resultant load doesn't need a higher alignment than the original! if (ISD::isNormalLoad(N0.getNode()) && N0.hasOneUse() && // Do not remove the cast if the types differ in endian layout. TLI.hasBigEndianPartOrdering(N0.getValueType(), DAG.getDataLayout()) == TLI.hasBigEndianPartOrdering(VT, DAG.getDataLayout()) && // If the load is volatile, we only want to change the load type if the // resulting load is legal. Otherwise we might increase the number of // memory accesses. We don't care if the original type was legal or not // as we assume software couldn't rely on the number of accesses of an // illegal type. ((!LegalOperations && !cast(N0)->isVolatile()) || TLI.isOperationLegal(ISD::LOAD, VT)) && TLI.isLoadBitCastBeneficial(N0.getValueType(), VT)) { LoadSDNode *LN0 = cast(N0); unsigned OrigAlign = LN0->getAlignment(); bool Fast = false; if (TLI.allowsMemoryAccess(*DAG.getContext(), DAG.getDataLayout(), VT, LN0->getAddressSpace(), OrigAlign, &Fast) && Fast) { SDValue Load = DAG.getLoad(VT, SDLoc(N), LN0->getChain(), LN0->getBasePtr(), LN0->getPointerInfo(), OrigAlign, LN0->getMemOperand()->getFlags(), LN0->getAAInfo()); DAG.ReplaceAllUsesOfValueWith(N0.getValue(1), Load.getValue(1)); return Load; } } if (SDValue V = foldBitcastedFPLogic(N, DAG, TLI)) return V; // fold (bitconvert (fneg x)) -> (xor (bitconvert x), signbit) // fold (bitconvert (fabs x)) -> (and (bitconvert x), (not signbit)) // // For ppc_fp128: // fold (bitcast (fneg x)) -> // flipbit = signbit // (xor (bitcast x) (build_pair flipbit, flipbit)) // // fold (bitcast (fabs x)) -> // flipbit = (and (extract_element (bitcast x), 0), signbit) // (xor (bitcast x) (build_pair flipbit, flipbit)) // This often reduces constant pool loads. if (((N0.getOpcode() == ISD::FNEG && !TLI.isFNegFree(N0.getValueType())) || (N0.getOpcode() == ISD::FABS && !TLI.isFAbsFree(N0.getValueType()))) && N0.getNode()->hasOneUse() && VT.isInteger() && !VT.isVector() && !N0.getValueType().isVector()) { SDValue NewConv = DAG.getBitcast(VT, N0.getOperand(0)); AddToWorklist(NewConv.getNode()); SDLoc DL(N); if (N0.getValueType() == MVT::ppcf128 && !LegalTypes) { assert(VT.getSizeInBits() == 128); SDValue SignBit = DAG.getConstant( APInt::getSignMask(VT.getSizeInBits() / 2), SDLoc(N0), MVT::i64); SDValue FlipBit; if (N0.getOpcode() == ISD::FNEG) { FlipBit = SignBit; AddToWorklist(FlipBit.getNode()); } else { assert(N0.getOpcode() == ISD::FABS); SDValue Hi = DAG.getNode(ISD::EXTRACT_ELEMENT, SDLoc(NewConv), MVT::i64, NewConv, DAG.getIntPtrConstant(getPPCf128HiElementSelector(DAG), SDLoc(NewConv))); AddToWorklist(Hi.getNode()); FlipBit = DAG.getNode(ISD::AND, SDLoc(N0), MVT::i64, Hi, SignBit); AddToWorklist(FlipBit.getNode()); } SDValue FlipBits = DAG.getNode(ISD::BUILD_PAIR, SDLoc(N0), VT, FlipBit, FlipBit); AddToWorklist(FlipBits.getNode()); return DAG.getNode(ISD::XOR, DL, VT, NewConv, FlipBits); } APInt SignBit = APInt::getSignMask(VT.getSizeInBits()); if (N0.getOpcode() == ISD::FNEG) return DAG.getNode(ISD::XOR, DL, VT, NewConv, DAG.getConstant(SignBit, DL, VT)); assert(N0.getOpcode() == ISD::FABS); return DAG.getNode(ISD::AND, DL, VT, NewConv, DAG.getConstant(~SignBit, DL, VT)); } // fold (bitconvert (fcopysign cst, x)) -> // (or (and (bitconvert x), sign), (and cst, (not sign))) // Note that we don't handle (copysign x, cst) because this can always be // folded to an fneg or fabs. // // For ppc_fp128: // fold (bitcast (fcopysign cst, x)) -> // flipbit = (and (extract_element // (xor (bitcast cst), (bitcast x)), 0), // signbit) // (xor (bitcast cst) (build_pair flipbit, flipbit)) if (N0.getOpcode() == ISD::FCOPYSIGN && N0.getNode()->hasOneUse() && isa(N0.getOperand(0)) && VT.isInteger() && !VT.isVector()) { unsigned OrigXWidth = N0.getOperand(1).getValueSizeInBits(); EVT IntXVT = EVT::getIntegerVT(*DAG.getContext(), OrigXWidth); if (isTypeLegal(IntXVT)) { SDValue X = DAG.getBitcast(IntXVT, N0.getOperand(1)); AddToWorklist(X.getNode()); // If X has a different width than the result/lhs, sext it or truncate it. unsigned VTWidth = VT.getSizeInBits(); if (OrigXWidth < VTWidth) { X = DAG.getNode(ISD::SIGN_EXTEND, SDLoc(N), VT, X); AddToWorklist(X.getNode()); } else if (OrigXWidth > VTWidth) { // To get the sign bit in the right place, we have to shift it right // before truncating. SDLoc DL(X); X = DAG.getNode(ISD::SRL, DL, X.getValueType(), X, DAG.getConstant(OrigXWidth-VTWidth, DL, X.getValueType())); AddToWorklist(X.getNode()); X = DAG.getNode(ISD::TRUNCATE, SDLoc(X), VT, X); AddToWorklist(X.getNode()); } if (N0.getValueType() == MVT::ppcf128 && !LegalTypes) { APInt SignBit = APInt::getSignMask(VT.getSizeInBits() / 2); SDValue Cst = DAG.getBitcast(VT, N0.getOperand(0)); AddToWorklist(Cst.getNode()); SDValue X = DAG.getBitcast(VT, N0.getOperand(1)); AddToWorklist(X.getNode()); SDValue XorResult = DAG.getNode(ISD::XOR, SDLoc(N0), VT, Cst, X); AddToWorklist(XorResult.getNode()); SDValue XorResult64 = DAG.getNode( ISD::EXTRACT_ELEMENT, SDLoc(XorResult), MVT::i64, XorResult, DAG.getIntPtrConstant(getPPCf128HiElementSelector(DAG), SDLoc(XorResult))); AddToWorklist(XorResult64.getNode()); SDValue FlipBit = DAG.getNode(ISD::AND, SDLoc(XorResult64), MVT::i64, XorResult64, DAG.getConstant(SignBit, SDLoc(XorResult64), MVT::i64)); AddToWorklist(FlipBit.getNode()); SDValue FlipBits = DAG.getNode(ISD::BUILD_PAIR, SDLoc(N0), VT, FlipBit, FlipBit); AddToWorklist(FlipBits.getNode()); return DAG.getNode(ISD::XOR, SDLoc(N), VT, Cst, FlipBits); } APInt SignBit = APInt::getSignMask(VT.getSizeInBits()); X = DAG.getNode(ISD::AND, SDLoc(X), VT, X, DAG.getConstant(SignBit, SDLoc(X), VT)); AddToWorklist(X.getNode()); SDValue Cst = DAG.getBitcast(VT, N0.getOperand(0)); Cst = DAG.getNode(ISD::AND, SDLoc(Cst), VT, Cst, DAG.getConstant(~SignBit, SDLoc(Cst), VT)); AddToWorklist(Cst.getNode()); return DAG.getNode(ISD::OR, SDLoc(N), VT, X, Cst); } } // bitconvert(build_pair(ld, ld)) -> ld iff load locations are consecutive. if (N0.getOpcode() == ISD::BUILD_PAIR) if (SDValue CombineLD = CombineConsecutiveLoads(N0.getNode(), VT)) return CombineLD; // Remove double bitcasts from shuffles - this is often a legacy of // XformToShuffleWithZero being used to combine bitmaskings (of // float vectors bitcast to integer vectors) into shuffles. // bitcast(shuffle(bitcast(s0),bitcast(s1))) -> shuffle(s0,s1) if (Level < AfterLegalizeDAG && TLI.isTypeLegal(VT) && VT.isVector() && N0->getOpcode() == ISD::VECTOR_SHUFFLE && N0.hasOneUse() && VT.getVectorNumElements() >= N0.getValueType().getVectorNumElements() && !(VT.getVectorNumElements() % N0.getValueType().getVectorNumElements())) { ShuffleVectorSDNode *SVN = cast(N0); // If operands are a bitcast, peek through if it casts the original VT. // If operands are a constant, just bitcast back to original VT. auto PeekThroughBitcast = [&](SDValue Op) { if (Op.getOpcode() == ISD::BITCAST && Op.getOperand(0).getValueType() == VT) return SDValue(Op.getOperand(0)); if (Op.isUndef() || ISD::isBuildVectorOfConstantSDNodes(Op.getNode()) || ISD::isBuildVectorOfConstantFPSDNodes(Op.getNode())) return DAG.getBitcast(VT, Op); return SDValue(); }; // FIXME: If either input vector is bitcast, try to convert the shuffle to // the result type of this bitcast. This would eliminate at least one // bitcast. See the transform in InstCombine. SDValue SV0 = PeekThroughBitcast(N0->getOperand(0)); SDValue SV1 = PeekThroughBitcast(N0->getOperand(1)); if (!(SV0 && SV1)) return SDValue(); int MaskScale = VT.getVectorNumElements() / N0.getValueType().getVectorNumElements(); SmallVector NewMask; for (int M : SVN->getMask()) for (int i = 0; i != MaskScale; ++i) NewMask.push_back(M < 0 ? -1 : M * MaskScale + i); bool LegalMask = TLI.isShuffleMaskLegal(NewMask, VT); if (!LegalMask) { std::swap(SV0, SV1); ShuffleVectorSDNode::commuteMask(NewMask); LegalMask = TLI.isShuffleMaskLegal(NewMask, VT); } if (LegalMask) return DAG.getVectorShuffle(VT, SDLoc(N), SV0, SV1, NewMask); } return SDValue(); } SDValue DAGCombiner::visitBUILD_PAIR(SDNode *N) { EVT VT = N->getValueType(0); return CombineConsecutiveLoads(N, VT); } /// We know that BV is a build_vector node with Constant, ConstantFP or Undef /// operands. DstEltVT indicates the destination element value type. SDValue DAGCombiner:: ConstantFoldBITCASTofBUILD_VECTOR(SDNode *BV, EVT DstEltVT) { EVT SrcEltVT = BV->getValueType(0).getVectorElementType(); // If this is already the right type, we're done. if (SrcEltVT == DstEltVT) return SDValue(BV, 0); unsigned SrcBitSize = SrcEltVT.getSizeInBits(); unsigned DstBitSize = DstEltVT.getSizeInBits(); // If this is a conversion of N elements of one type to N elements of another // type, convert each element. This handles FP<->INT cases. if (SrcBitSize == DstBitSize) { SmallVector Ops; for (SDValue Op : BV->op_values()) { // If the vector element type is not legal, the BUILD_VECTOR operands // are promoted and implicitly truncated. Make that explicit here. if (Op.getValueType() != SrcEltVT) Op = DAG.getNode(ISD::TRUNCATE, SDLoc(BV), SrcEltVT, Op); Ops.push_back(DAG.getBitcast(DstEltVT, Op)); AddToWorklist(Ops.back().getNode()); } EVT VT = EVT::getVectorVT(*DAG.getContext(), DstEltVT, BV->getValueType(0).getVectorNumElements()); return DAG.getBuildVector(VT, SDLoc(BV), Ops); } // Otherwise, we're growing or shrinking the elements. To avoid having to // handle annoying details of growing/shrinking FP values, we convert them to // int first. if (SrcEltVT.isFloatingPoint()) { // Convert the input float vector to a int vector where the elements are the // same sizes. EVT IntVT = EVT::getIntegerVT(*DAG.getContext(), SrcEltVT.getSizeInBits()); BV = ConstantFoldBITCASTofBUILD_VECTOR(BV, IntVT).getNode(); SrcEltVT = IntVT; } // Now we know the input is an integer vector. If the output is a FP type, // convert to integer first, then to FP of the right size. if (DstEltVT.isFloatingPoint()) { EVT TmpVT = EVT::getIntegerVT(*DAG.getContext(), DstEltVT.getSizeInBits()); SDNode *Tmp = ConstantFoldBITCASTofBUILD_VECTOR(BV, TmpVT).getNode(); // Next, convert to FP elements of the same size. return ConstantFoldBITCASTofBUILD_VECTOR(Tmp, DstEltVT); } SDLoc DL(BV); // Okay, we know the src/dst types are both integers of differing types. // Handling growing first. assert(SrcEltVT.isInteger() && DstEltVT.isInteger()); if (SrcBitSize < DstBitSize) { unsigned NumInputsPerOutput = DstBitSize/SrcBitSize; SmallVector Ops; for (unsigned i = 0, e = BV->getNumOperands(); i != e; i += NumInputsPerOutput) { bool isLE = DAG.getDataLayout().isLittleEndian(); APInt NewBits = APInt(DstBitSize, 0); bool EltIsUndef = true; for (unsigned j = 0; j != NumInputsPerOutput; ++j) { // Shift the previously computed bits over. NewBits <<= SrcBitSize; SDValue Op = BV->getOperand(i+ (isLE ? (NumInputsPerOutput-j-1) : j)); if (Op.isUndef()) continue; EltIsUndef = false; NewBits |= cast(Op)->getAPIntValue(). zextOrTrunc(SrcBitSize).zext(DstBitSize); } if (EltIsUndef) Ops.push_back(DAG.getUNDEF(DstEltVT)); else Ops.push_back(DAG.getConstant(NewBits, DL, DstEltVT)); } EVT VT = EVT::getVectorVT(*DAG.getContext(), DstEltVT, Ops.size()); return DAG.getBuildVector(VT, DL, Ops); } // Finally, this must be the case where we are shrinking elements: each input // turns into multiple outputs. unsigned NumOutputsPerInput = SrcBitSize/DstBitSize; EVT VT = EVT::getVectorVT(*DAG.getContext(), DstEltVT, NumOutputsPerInput*BV->getNumOperands()); SmallVector Ops; for (const SDValue &Op : BV->op_values()) { if (Op.isUndef()) { Ops.append(NumOutputsPerInput, DAG.getUNDEF(DstEltVT)); continue; } APInt OpVal = cast(Op)-> getAPIntValue().zextOrTrunc(SrcBitSize); for (unsigned j = 0; j != NumOutputsPerInput; ++j) { APInt ThisVal = OpVal.trunc(DstBitSize); Ops.push_back(DAG.getConstant(ThisVal, DL, DstEltVT)); OpVal.lshrInPlace(DstBitSize); } // For big endian targets, swap the order of the pieces of each element. if (DAG.getDataLayout().isBigEndian()) std::reverse(Ops.end()-NumOutputsPerInput, Ops.end()); } return DAG.getBuildVector(VT, DL, Ops); } static bool isContractable(SDNode *N) { SDNodeFlags F = N->getFlags(); return F.hasAllowContract() || F.hasAllowReassociation(); } /// Try to perform FMA combining on a given FADD node. SDValue DAGCombiner::visitFADDForFMACombine(SDNode *N) { SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); EVT VT = N->getValueType(0); SDLoc SL(N); const TargetOptions &Options = DAG.getTarget().Options; // Floating-point multiply-add with intermediate rounding. bool HasFMAD = (LegalOperations && TLI.isOperationLegal(ISD::FMAD, VT)); // Floating-point multiply-add without intermediate rounding. bool HasFMA = TLI.isFMAFasterThanFMulAndFAdd(VT) && (!LegalOperations || TLI.isOperationLegalOrCustom(ISD::FMA, VT)); // No valid opcode, do not combine. if (!HasFMAD && !HasFMA) return SDValue(); SDNodeFlags Flags = N->getFlags(); bool CanFuse = Options.UnsafeFPMath || isContractable(N); bool AllowFusionGlobally = (Options.AllowFPOpFusion == FPOpFusion::Fast || CanFuse || HasFMAD); // If the addition is not contractable, do not combine. if (!AllowFusionGlobally && !isContractable(N)) return SDValue(); const SelectionDAGTargetInfo *STI = DAG.getSubtarget().getSelectionDAGInfo(); if (STI && STI->generateFMAsInMachineCombiner(OptLevel)) return SDValue(); // Always prefer FMAD to FMA for precision. unsigned PreferredFusedOpcode = HasFMAD ? ISD::FMAD : ISD::FMA; bool Aggressive = TLI.enableAggressiveFMAFusion(VT); // Is the node an FMUL and contractable either due to global flags or // SDNodeFlags. auto isContractableFMUL = [AllowFusionGlobally](SDValue N) { if (N.getOpcode() != ISD::FMUL) return false; return AllowFusionGlobally || isContractable(N.getNode()); }; // If we have two choices trying to fold (fadd (fmul u, v), (fmul x, y)), // prefer to fold the multiply with fewer uses. if (Aggressive && isContractableFMUL(N0) && isContractableFMUL(N1)) { if (N0.getNode()->use_size() > N1.getNode()->use_size()) std::swap(N0, N1); } // fold (fadd (fmul x, y), z) -> (fma x, y, z) if (isContractableFMUL(N0) && (Aggressive || N0->hasOneUse())) { return DAG.getNode(PreferredFusedOpcode, SL, VT, N0.getOperand(0), N0.getOperand(1), N1, Flags); } // fold (fadd x, (fmul y, z)) -> (fma y, z, x) // Note: Commutes FADD operands. if (isContractableFMUL(N1) && (Aggressive || N1->hasOneUse())) { return DAG.getNode(PreferredFusedOpcode, SL, VT, N1.getOperand(0), N1.getOperand(1), N0, Flags); } // Look through FP_EXTEND nodes to do more combining. // fold (fadd (fpext (fmul x, y)), z) -> (fma (fpext x), (fpext y), z) if (N0.getOpcode() == ISD::FP_EXTEND) { SDValue N00 = N0.getOperand(0); if (isContractableFMUL(N00) && TLI.isFPExtFoldable(PreferredFusedOpcode, VT, N00.getValueType())) { return DAG.getNode(PreferredFusedOpcode, SL, VT, DAG.getNode(ISD::FP_EXTEND, SL, VT, N00.getOperand(0)), DAG.getNode(ISD::FP_EXTEND, SL, VT, N00.getOperand(1)), N1, Flags); } } // fold (fadd x, (fpext (fmul y, z))) -> (fma (fpext y), (fpext z), x) // Note: Commutes FADD operands. if (N1.getOpcode() == ISD::FP_EXTEND) { SDValue N10 = N1.getOperand(0); if (isContractableFMUL(N10) && TLI.isFPExtFoldable(PreferredFusedOpcode, VT, N10.getValueType())) { return DAG.getNode(PreferredFusedOpcode, SL, VT, DAG.getNode(ISD::FP_EXTEND, SL, VT, N10.getOperand(0)), DAG.getNode(ISD::FP_EXTEND, SL, VT, N10.getOperand(1)), N0, Flags); } } // More folding opportunities when target permits. if (Aggressive) { // fold (fadd (fma x, y, (fmul u, v)), z) -> (fma x, y (fma u, v, z)) if (CanFuse && N0.getOpcode() == PreferredFusedOpcode && N0.getOperand(2).getOpcode() == ISD::FMUL && N0->hasOneUse() && N0.getOperand(2)->hasOneUse()) { return DAG.getNode(PreferredFusedOpcode, SL, VT, N0.getOperand(0), N0.getOperand(1), DAG.getNode(PreferredFusedOpcode, SL, VT, N0.getOperand(2).getOperand(0), N0.getOperand(2).getOperand(1), N1, Flags), Flags); } // fold (fadd x, (fma y, z, (fmul u, v)) -> (fma y, z (fma u, v, x)) if (CanFuse && N1->getOpcode() == PreferredFusedOpcode && N1.getOperand(2).getOpcode() == ISD::FMUL && N1->hasOneUse() && N1.getOperand(2)->hasOneUse()) { return DAG.getNode(PreferredFusedOpcode, SL, VT, N1.getOperand(0), N1.getOperand(1), DAG.getNode(PreferredFusedOpcode, SL, VT, N1.getOperand(2).getOperand(0), N1.getOperand(2).getOperand(1), N0, Flags), Flags); } // fold (fadd (fma x, y, (fpext (fmul u, v))), z) // -> (fma x, y, (fma (fpext u), (fpext v), z)) auto FoldFAddFMAFPExtFMul = [&] ( SDValue X, SDValue Y, SDValue U, SDValue V, SDValue Z, SDNodeFlags Flags) { return DAG.getNode(PreferredFusedOpcode, SL, VT, X, Y, DAG.getNode(PreferredFusedOpcode, SL, VT, DAG.getNode(ISD::FP_EXTEND, SL, VT, U), DAG.getNode(ISD::FP_EXTEND, SL, VT, V), Z, Flags), Flags); }; if (N0.getOpcode() == PreferredFusedOpcode) { SDValue N02 = N0.getOperand(2); if (N02.getOpcode() == ISD::FP_EXTEND) { SDValue N020 = N02.getOperand(0); if (isContractableFMUL(N020) && TLI.isFPExtFoldable(PreferredFusedOpcode, VT, N020.getValueType())) { return FoldFAddFMAFPExtFMul(N0.getOperand(0), N0.getOperand(1), N020.getOperand(0), N020.getOperand(1), N1, Flags); } } } // fold (fadd (fpext (fma x, y, (fmul u, v))), z) // -> (fma (fpext x), (fpext y), (fma (fpext u), (fpext v), z)) // FIXME: This turns two single-precision and one double-precision // operation into two double-precision operations, which might not be // interesting for all targets, especially GPUs. auto FoldFAddFPExtFMAFMul = [&] ( SDValue X, SDValue Y, SDValue U, SDValue V, SDValue Z, SDNodeFlags Flags) { return DAG.getNode(PreferredFusedOpcode, SL, VT, DAG.getNode(ISD::FP_EXTEND, SL, VT, X), DAG.getNode(ISD::FP_EXTEND, SL, VT, Y), DAG.getNode(PreferredFusedOpcode, SL, VT, DAG.getNode(ISD::FP_EXTEND, SL, VT, U), DAG.getNode(ISD::FP_EXTEND, SL, VT, V), Z, Flags), Flags); }; if (N0.getOpcode() == ISD::FP_EXTEND) { SDValue N00 = N0.getOperand(0); if (N00.getOpcode() == PreferredFusedOpcode) { SDValue N002 = N00.getOperand(2); if (isContractableFMUL(N002) && TLI.isFPExtFoldable(PreferredFusedOpcode, VT, N00.getValueType())) { return FoldFAddFPExtFMAFMul(N00.getOperand(0), N00.getOperand(1), N002.getOperand(0), N002.getOperand(1), N1, Flags); } } } // fold (fadd x, (fma y, z, (fpext (fmul u, v))) // -> (fma y, z, (fma (fpext u), (fpext v), x)) if (N1.getOpcode() == PreferredFusedOpcode) { SDValue N12 = N1.getOperand(2); if (N12.getOpcode() == ISD::FP_EXTEND) { SDValue N120 = N12.getOperand(0); if (isContractableFMUL(N120) && TLI.isFPExtFoldable(PreferredFusedOpcode, VT, N120.getValueType())) { return FoldFAddFMAFPExtFMul(N1.getOperand(0), N1.getOperand(1), N120.getOperand(0), N120.getOperand(1), N0, Flags); } } } // fold (fadd x, (fpext (fma y, z, (fmul u, v))) // -> (fma (fpext y), (fpext z), (fma (fpext u), (fpext v), x)) // FIXME: This turns two single-precision and one double-precision // operation into two double-precision operations, which might not be // interesting for all targets, especially GPUs. if (N1.getOpcode() == ISD::FP_EXTEND) { SDValue N10 = N1.getOperand(0); if (N10.getOpcode() == PreferredFusedOpcode) { SDValue N102 = N10.getOperand(2); if (isContractableFMUL(N102) && TLI.isFPExtFoldable(PreferredFusedOpcode, VT, N10.getValueType())) { return FoldFAddFPExtFMAFMul(N10.getOperand(0), N10.getOperand(1), N102.getOperand(0), N102.getOperand(1), N0, Flags); } } } } return SDValue(); } /// Try to perform FMA combining on a given FSUB node. SDValue DAGCombiner::visitFSUBForFMACombine(SDNode *N) { SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); EVT VT = N->getValueType(0); SDLoc SL(N); const TargetOptions &Options = DAG.getTarget().Options; // Floating-point multiply-add with intermediate rounding. bool HasFMAD = (LegalOperations && TLI.isOperationLegal(ISD::FMAD, VT)); // Floating-point multiply-add without intermediate rounding. bool HasFMA = TLI.isFMAFasterThanFMulAndFAdd(VT) && (!LegalOperations || TLI.isOperationLegalOrCustom(ISD::FMA, VT)); // No valid opcode, do not combine. if (!HasFMAD && !HasFMA) return SDValue(); const SDNodeFlags Flags = N->getFlags(); bool CanFuse = Options.UnsafeFPMath || isContractable(N); bool AllowFusionGlobally = (Options.AllowFPOpFusion == FPOpFusion::Fast || CanFuse || HasFMAD); // If the subtraction is not contractable, do not combine. if (!AllowFusionGlobally && !isContractable(N)) return SDValue(); const SelectionDAGTargetInfo *STI = DAG.getSubtarget().getSelectionDAGInfo(); if (STI && STI->generateFMAsInMachineCombiner(OptLevel)) return SDValue(); // Always prefer FMAD to FMA for precision. unsigned PreferredFusedOpcode = HasFMAD ? ISD::FMAD : ISD::FMA; bool Aggressive = TLI.enableAggressiveFMAFusion(VT); // Is the node an FMUL and contractable either due to global flags or // SDNodeFlags. auto isContractableFMUL = [AllowFusionGlobally](SDValue N) { if (N.getOpcode() != ISD::FMUL) return false; return AllowFusionGlobally || isContractable(N.getNode()); }; // fold (fsub (fmul x, y), z) -> (fma x, y, (fneg z)) if (isContractableFMUL(N0) && (Aggressive || N0->hasOneUse())) { return DAG.getNode(PreferredFusedOpcode, SL, VT, N0.getOperand(0), N0.getOperand(1), DAG.getNode(ISD::FNEG, SL, VT, N1), Flags); } // fold (fsub x, (fmul y, z)) -> (fma (fneg y), z, x) // Note: Commutes FSUB operands. if (isContractableFMUL(N1) && (Aggressive || N1->hasOneUse())) { return DAG.getNode(PreferredFusedOpcode, SL, VT, DAG.getNode(ISD::FNEG, SL, VT, N1.getOperand(0)), N1.getOperand(1), N0, Flags); } // fold (fsub (fneg (fmul, x, y)), z) -> (fma (fneg x), y, (fneg z)) if (N0.getOpcode() == ISD::FNEG && isContractableFMUL(N0.getOperand(0)) && (Aggressive || (N0->hasOneUse() && N0.getOperand(0).hasOneUse()))) { SDValue N00 = N0.getOperand(0).getOperand(0); SDValue N01 = N0.getOperand(0).getOperand(1); return DAG.getNode(PreferredFusedOpcode, SL, VT, DAG.getNode(ISD::FNEG, SL, VT, N00), N01, DAG.getNode(ISD::FNEG, SL, VT, N1), Flags); } // Look through FP_EXTEND nodes to do more combining. // fold (fsub (fpext (fmul x, y)), z) // -> (fma (fpext x), (fpext y), (fneg z)) if (N0.getOpcode() == ISD::FP_EXTEND) { SDValue N00 = N0.getOperand(0); if (isContractableFMUL(N00) && TLI.isFPExtFoldable(PreferredFusedOpcode, VT, N00.getValueType())) { return DAG.getNode(PreferredFusedOpcode, SL, VT, DAG.getNode(ISD::FP_EXTEND, SL, VT, N00.getOperand(0)), DAG.getNode(ISD::FP_EXTEND, SL, VT, N00.getOperand(1)), DAG.getNode(ISD::FNEG, SL, VT, N1), Flags); } } // fold (fsub x, (fpext (fmul y, z))) // -> (fma (fneg (fpext y)), (fpext z), x) // Note: Commutes FSUB operands. if (N1.getOpcode() == ISD::FP_EXTEND) { SDValue N10 = N1.getOperand(0); if (isContractableFMUL(N10) && TLI.isFPExtFoldable(PreferredFusedOpcode, VT, N10.getValueType())) { return DAG.getNode(PreferredFusedOpcode, SL, VT, DAG.getNode(ISD::FNEG, SL, VT, DAG.getNode(ISD::FP_EXTEND, SL, VT, N10.getOperand(0))), DAG.getNode(ISD::FP_EXTEND, SL, VT, N10.getOperand(1)), N0, Flags); } } // fold (fsub (fpext (fneg (fmul, x, y))), z) // -> (fneg (fma (fpext x), (fpext y), z)) // Note: This could be removed with appropriate canonicalization of the // input expression into (fneg (fadd (fpext (fmul, x, y)), z). However, the // orthogonal flags -fp-contract=fast and -enable-unsafe-fp-math prevent // from implementing the canonicalization in visitFSUB. if (N0.getOpcode() == ISD::FP_EXTEND) { SDValue N00 = N0.getOperand(0); if (N00.getOpcode() == ISD::FNEG) { SDValue N000 = N00.getOperand(0); if (isContractableFMUL(N000) && TLI.isFPExtFoldable(PreferredFusedOpcode, VT, N00.getValueType())) { return DAG.getNode(ISD::FNEG, SL, VT, DAG.getNode(PreferredFusedOpcode, SL, VT, DAG.getNode(ISD::FP_EXTEND, SL, VT, N000.getOperand(0)), DAG.getNode(ISD::FP_EXTEND, SL, VT, N000.getOperand(1)), N1, Flags)); } } } // fold (fsub (fneg (fpext (fmul, x, y))), z) // -> (fneg (fma (fpext x)), (fpext y), z) // Note: This could be removed with appropriate canonicalization of the // input expression into (fneg (fadd (fpext (fmul, x, y)), z). However, the // orthogonal flags -fp-contract=fast and -enable-unsafe-fp-math prevent // from implementing the canonicalization in visitFSUB. if (N0.getOpcode() == ISD::FNEG) { SDValue N00 = N0.getOperand(0); if (N00.getOpcode() == ISD::FP_EXTEND) { SDValue N000 = N00.getOperand(0); if (isContractableFMUL(N000) && TLI.isFPExtFoldable(PreferredFusedOpcode, VT, N000.getValueType())) { return DAG.getNode(ISD::FNEG, SL, VT, DAG.getNode(PreferredFusedOpcode, SL, VT, DAG.getNode(ISD::FP_EXTEND, SL, VT, N000.getOperand(0)), DAG.getNode(ISD::FP_EXTEND, SL, VT, N000.getOperand(1)), N1, Flags)); } } } // More folding opportunities when target permits. if (Aggressive) { // fold (fsub (fma x, y, (fmul u, v)), z) // -> (fma x, y (fma u, v, (fneg z))) if (CanFuse && N0.getOpcode() == PreferredFusedOpcode && isContractableFMUL(N0.getOperand(2)) && N0->hasOneUse() && N0.getOperand(2)->hasOneUse()) { return DAG.getNode(PreferredFusedOpcode, SL, VT, N0.getOperand(0), N0.getOperand(1), DAG.getNode(PreferredFusedOpcode, SL, VT, N0.getOperand(2).getOperand(0), N0.getOperand(2).getOperand(1), DAG.getNode(ISD::FNEG, SL, VT, N1), Flags), Flags); } // fold (fsub x, (fma y, z, (fmul u, v))) // -> (fma (fneg y), z, (fma (fneg u), v, x)) if (CanFuse && N1.getOpcode() == PreferredFusedOpcode && isContractableFMUL(N1.getOperand(2))) { SDValue N20 = N1.getOperand(2).getOperand(0); SDValue N21 = N1.getOperand(2).getOperand(1); return DAG.getNode(PreferredFusedOpcode, SL, VT, DAG.getNode(ISD::FNEG, SL, VT, N1.getOperand(0)), N1.getOperand(1), DAG.getNode(PreferredFusedOpcode, SL, VT, DAG.getNode(ISD::FNEG, SL, VT, N20), N21, N0, Flags), Flags); } // fold (fsub (fma x, y, (fpext (fmul u, v))), z) // -> (fma x, y (fma (fpext u), (fpext v), (fneg z))) if (N0.getOpcode() == PreferredFusedOpcode) { SDValue N02 = N0.getOperand(2); if (N02.getOpcode() == ISD::FP_EXTEND) { SDValue N020 = N02.getOperand(0); if (isContractableFMUL(N020) && TLI.isFPExtFoldable(PreferredFusedOpcode, VT, N020.getValueType())) { return DAG.getNode(PreferredFusedOpcode, SL, VT, N0.getOperand(0), N0.getOperand(1), DAG.getNode(PreferredFusedOpcode, SL, VT, DAG.getNode(ISD::FP_EXTEND, SL, VT, N020.getOperand(0)), DAG.getNode(ISD::FP_EXTEND, SL, VT, N020.getOperand(1)), DAG.getNode(ISD::FNEG, SL, VT, N1), Flags), Flags); } } } // fold (fsub (fpext (fma x, y, (fmul u, v))), z) // -> (fma (fpext x), (fpext y), // (fma (fpext u), (fpext v), (fneg z))) // FIXME: This turns two single-precision and one double-precision // operation into two double-precision operations, which might not be // interesting for all targets, especially GPUs. if (N0.getOpcode() == ISD::FP_EXTEND) { SDValue N00 = N0.getOperand(0); if (N00.getOpcode() == PreferredFusedOpcode) { SDValue N002 = N00.getOperand(2); if (isContractableFMUL(N002) && TLI.isFPExtFoldable(PreferredFusedOpcode, VT, N00.getValueType())) { return DAG.getNode(PreferredFusedOpcode, SL, VT, DAG.getNode(ISD::FP_EXTEND, SL, VT, N00.getOperand(0)), DAG.getNode(ISD::FP_EXTEND, SL, VT, N00.getOperand(1)), DAG.getNode(PreferredFusedOpcode, SL, VT, DAG.getNode(ISD::FP_EXTEND, SL, VT, N002.getOperand(0)), DAG.getNode(ISD::FP_EXTEND, SL, VT, N002.getOperand(1)), DAG.getNode(ISD::FNEG, SL, VT, N1), Flags), Flags); } } } // fold (fsub x, (fma y, z, (fpext (fmul u, v)))) // -> (fma (fneg y), z, (fma (fneg (fpext u)), (fpext v), x)) if (N1.getOpcode() == PreferredFusedOpcode && N1.getOperand(2).getOpcode() == ISD::FP_EXTEND) { SDValue N120 = N1.getOperand(2).getOperand(0); if (isContractableFMUL(N120) && TLI.isFPExtFoldable(PreferredFusedOpcode, VT, N120.getValueType())) { SDValue N1200 = N120.getOperand(0); SDValue N1201 = N120.getOperand(1); return DAG.getNode(PreferredFusedOpcode, SL, VT, DAG.getNode(ISD::FNEG, SL, VT, N1.getOperand(0)), N1.getOperand(1), DAG.getNode(PreferredFusedOpcode, SL, VT, DAG.getNode(ISD::FNEG, SL, VT, DAG.getNode(ISD::FP_EXTEND, SL, VT, N1200)), DAG.getNode(ISD::FP_EXTEND, SL, VT, N1201), N0, Flags), Flags); } } // fold (fsub x, (fpext (fma y, z, (fmul u, v)))) // -> (fma (fneg (fpext y)), (fpext z), // (fma (fneg (fpext u)), (fpext v), x)) // FIXME: This turns two single-precision and one double-precision // operation into two double-precision operations, which might not be // interesting for all targets, especially GPUs. if (N1.getOpcode() == ISD::FP_EXTEND && N1.getOperand(0).getOpcode() == PreferredFusedOpcode) { SDValue CvtSrc = N1.getOperand(0); SDValue N100 = CvtSrc.getOperand(0); SDValue N101 = CvtSrc.getOperand(1); SDValue N102 = CvtSrc.getOperand(2); if (isContractableFMUL(N102) && TLI.isFPExtFoldable(PreferredFusedOpcode, VT, CvtSrc.getValueType())) { SDValue N1020 = N102.getOperand(0); SDValue N1021 = N102.getOperand(1); return DAG.getNode(PreferredFusedOpcode, SL, VT, DAG.getNode(ISD::FNEG, SL, VT, DAG.getNode(ISD::FP_EXTEND, SL, VT, N100)), DAG.getNode(ISD::FP_EXTEND, SL, VT, N101), DAG.getNode(PreferredFusedOpcode, SL, VT, DAG.getNode(ISD::FNEG, SL, VT, DAG.getNode(ISD::FP_EXTEND, SL, VT, N1020)), DAG.getNode(ISD::FP_EXTEND, SL, VT, N1021), N0, Flags), Flags); } } } return SDValue(); } /// Try to perform FMA combining on a given FMUL node based on the distributive /// law x * (y + 1) = x * y + x and variants thereof (commuted versions, /// subtraction instead of addition). SDValue DAGCombiner::visitFMULForFMADistributiveCombine(SDNode *N) { SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); EVT VT = N->getValueType(0); SDLoc SL(N); const SDNodeFlags Flags = N->getFlags(); assert(N->getOpcode() == ISD::FMUL && "Expected FMUL Operation"); const TargetOptions &Options = DAG.getTarget().Options; // The transforms below are incorrect when x == 0 and y == inf, because the // intermediate multiplication produces a nan. if (!Options.NoInfsFPMath) return SDValue(); // Floating-point multiply-add without intermediate rounding. bool HasFMA = (Options.AllowFPOpFusion == FPOpFusion::Fast || Options.UnsafeFPMath) && TLI.isFMAFasterThanFMulAndFAdd(VT) && (!LegalOperations || TLI.isOperationLegalOrCustom(ISD::FMA, VT)); // Floating-point multiply-add with intermediate rounding. This can result // in a less precise result due to the changed rounding order. bool HasFMAD = Options.UnsafeFPMath && (LegalOperations && TLI.isOperationLegal(ISD::FMAD, VT)); // No valid opcode, do not combine. if (!HasFMAD && !HasFMA) return SDValue(); // Always prefer FMAD to FMA for precision. unsigned PreferredFusedOpcode = HasFMAD ? ISD::FMAD : ISD::FMA; bool Aggressive = TLI.enableAggressiveFMAFusion(VT); // fold (fmul (fadd x0, +1.0), y) -> (fma x0, y, y) // fold (fmul (fadd x0, -1.0), y) -> (fma x0, y, (fneg y)) auto FuseFADD = [&](SDValue X, SDValue Y, const SDNodeFlags Flags) { if (X.getOpcode() == ISD::FADD && (Aggressive || X->hasOneUse())) { if (auto *C = isConstOrConstSplatFP(X.getOperand(1), true)) { if (C->isExactlyValue(+1.0)) return DAG.getNode(PreferredFusedOpcode, SL, VT, X.getOperand(0), Y, Y, Flags); if (C->isExactlyValue(-1.0)) return DAG.getNode(PreferredFusedOpcode, SL, VT, X.getOperand(0), Y, DAG.getNode(ISD::FNEG, SL, VT, Y), Flags); } } return SDValue(); }; if (SDValue FMA = FuseFADD(N0, N1, Flags)) return FMA; if (SDValue FMA = FuseFADD(N1, N0, Flags)) return FMA; // fold (fmul (fsub +1.0, x1), y) -> (fma (fneg x1), y, y) // fold (fmul (fsub -1.0, x1), y) -> (fma (fneg x1), y, (fneg y)) // fold (fmul (fsub x0, +1.0), y) -> (fma x0, y, (fneg y)) // fold (fmul (fsub x0, -1.0), y) -> (fma x0, y, y) auto FuseFSUB = [&](SDValue X, SDValue Y, const SDNodeFlags Flags) { if (X.getOpcode() == ISD::FSUB && (Aggressive || X->hasOneUse())) { if (auto *C0 = isConstOrConstSplatFP(X.getOperand(0), true)) { if (C0->isExactlyValue(+1.0)) return DAG.getNode(PreferredFusedOpcode, SL, VT, DAG.getNode(ISD::FNEG, SL, VT, X.getOperand(1)), Y, Y, Flags); if (C0->isExactlyValue(-1.0)) return DAG.getNode(PreferredFusedOpcode, SL, VT, DAG.getNode(ISD::FNEG, SL, VT, X.getOperand(1)), Y, DAG.getNode(ISD::FNEG, SL, VT, Y), Flags); } if (auto *C1 = isConstOrConstSplatFP(X.getOperand(1), true)) { if (C1->isExactlyValue(+1.0)) return DAG.getNode(PreferredFusedOpcode, SL, VT, X.getOperand(0), Y, DAG.getNode(ISD::FNEG, SL, VT, Y), Flags); if (C1->isExactlyValue(-1.0)) return DAG.getNode(PreferredFusedOpcode, SL, VT, X.getOperand(0), Y, Y, Flags); } } return SDValue(); }; if (SDValue FMA = FuseFSUB(N0, N1, Flags)) return FMA; if (SDValue FMA = FuseFSUB(N1, N0, Flags)) return FMA; return SDValue(); } SDValue DAGCombiner::visitFADD(SDNode *N) { SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); bool N0CFP = isConstantFPBuildVectorOrConstantFP(N0); bool N1CFP = isConstantFPBuildVectorOrConstantFP(N1); EVT VT = N->getValueType(0); SDLoc DL(N); const TargetOptions &Options = DAG.getTarget().Options; const SDNodeFlags Flags = N->getFlags(); // fold vector ops if (VT.isVector()) if (SDValue FoldedVOp = SimplifyVBinOp(N)) return FoldedVOp; // fold (fadd c1, c2) -> c1 + c2 if (N0CFP && N1CFP) return DAG.getNode(ISD::FADD, DL, VT, N0, N1, Flags); // canonicalize constant to RHS if (N0CFP && !N1CFP) return DAG.getNode(ISD::FADD, DL, VT, N1, N0, Flags); // N0 + -0.0 --> N0 (also allowed with +0.0 and fast-math) ConstantFPSDNode *N1C = isConstOrConstSplatFP(N1, true); if (N1C && N1C->isZero()) if (N1C->isNegative() || Options.UnsafeFPMath || Flags.hasNoSignedZeros()) return N0; if (SDValue NewSel = foldBinOpIntoSelect(N)) return NewSel; // fold (fadd A, (fneg B)) -> (fsub A, B) if ((!LegalOperations || TLI.isOperationLegalOrCustom(ISD::FSUB, VT)) && isNegatibleForFree(N1, LegalOperations, TLI, &Options) == 2) return DAG.getNode(ISD::FSUB, DL, VT, N0, GetNegatedExpression(N1, DAG, LegalOperations), Flags); // fold (fadd (fneg A), B) -> (fsub B, A) if ((!LegalOperations || TLI.isOperationLegalOrCustom(ISD::FSUB, VT)) && isNegatibleForFree(N0, LegalOperations, TLI, &Options) == 2) return DAG.getNode(ISD::FSUB, DL, VT, N1, GetNegatedExpression(N0, DAG, LegalOperations), Flags); auto isFMulNegTwo = [](SDValue FMul) { if (!FMul.hasOneUse() || FMul.getOpcode() != ISD::FMUL) return false; auto *C = isConstOrConstSplatFP(FMul.getOperand(1), true); return C && C->isExactlyValue(-2.0); }; // fadd (fmul B, -2.0), A --> fsub A, (fadd B, B) if (isFMulNegTwo(N0)) { SDValue B = N0.getOperand(0); SDValue Add = DAG.getNode(ISD::FADD, DL, VT, B, B, Flags); return DAG.getNode(ISD::FSUB, DL, VT, N1, Add, Flags); } // fadd A, (fmul B, -2.0) --> fsub A, (fadd B, B) if (isFMulNegTwo(N1)) { SDValue B = N1.getOperand(0); SDValue Add = DAG.getNode(ISD::FADD, DL, VT, B, B, Flags); return DAG.getNode(ISD::FSUB, DL, VT, N0, Add, Flags); } // No FP constant should be created after legalization as Instruction // Selection pass has a hard time dealing with FP constants. bool AllowNewConst = (Level < AfterLegalizeDAG); // If 'unsafe math' or nnan is enabled, fold lots of things. if ((Options.UnsafeFPMath || Flags.hasNoNaNs()) && AllowNewConst) { // If allowed, fold (fadd (fneg x), x) -> 0.0 if (N0.getOpcode() == ISD::FNEG && N0.getOperand(0) == N1) return DAG.getConstantFP(0.0, DL, VT); // If allowed, fold (fadd x, (fneg x)) -> 0.0 if (N1.getOpcode() == ISD::FNEG && N1.getOperand(0) == N0) return DAG.getConstantFP(0.0, DL, VT); } // If 'unsafe math' or reassoc and nsz, fold lots of things. // TODO: break out portions of the transformations below for which Unsafe is // considered and which do not require both nsz and reassoc if ((Options.UnsafeFPMath || (Flags.hasAllowReassociation() && Flags.hasNoSignedZeros())) && AllowNewConst) { // fadd (fadd x, c1), c2 -> fadd x, c1 + c2 if (N1CFP && N0.getOpcode() == ISD::FADD && isConstantFPBuildVectorOrConstantFP(N0.getOperand(1))) { SDValue NewC = DAG.getNode(ISD::FADD, DL, VT, N0.getOperand(1), N1, Flags); return DAG.getNode(ISD::FADD, DL, VT, N0.getOperand(0), NewC, Flags); } // We can fold chains of FADD's of the same value into multiplications. // This transform is not safe in general because we are reducing the number // of rounding steps. if (TLI.isOperationLegalOrCustom(ISD::FMUL, VT) && !N0CFP && !N1CFP) { if (N0.getOpcode() == ISD::FMUL) { bool CFP00 = isConstantFPBuildVectorOrConstantFP(N0.getOperand(0)); bool CFP01 = isConstantFPBuildVectorOrConstantFP(N0.getOperand(1)); // (fadd (fmul x, c), x) -> (fmul x, c+1) if (CFP01 && !CFP00 && N0.getOperand(0) == N1) { SDValue NewCFP = DAG.getNode(ISD::FADD, DL, VT, N0.getOperand(1), DAG.getConstantFP(1.0, DL, VT), Flags); return DAG.getNode(ISD::FMUL, DL, VT, N1, NewCFP, Flags); } // (fadd (fmul x, c), (fadd x, x)) -> (fmul x, c+2) if (CFP01 && !CFP00 && N1.getOpcode() == ISD::FADD && N1.getOperand(0) == N1.getOperand(1) && N0.getOperand(0) == N1.getOperand(0)) { SDValue NewCFP = DAG.getNode(ISD::FADD, DL, VT, N0.getOperand(1), DAG.getConstantFP(2.0, DL, VT), Flags); return DAG.getNode(ISD::FMUL, DL, VT, N0.getOperand(0), NewCFP, Flags); } } if (N1.getOpcode() == ISD::FMUL) { bool CFP10 = isConstantFPBuildVectorOrConstantFP(N1.getOperand(0)); bool CFP11 = isConstantFPBuildVectorOrConstantFP(N1.getOperand(1)); // (fadd x, (fmul x, c)) -> (fmul x, c+1) if (CFP11 && !CFP10 && N1.getOperand(0) == N0) { SDValue NewCFP = DAG.getNode(ISD::FADD, DL, VT, N1.getOperand(1), DAG.getConstantFP(1.0, DL, VT), Flags); return DAG.getNode(ISD::FMUL, DL, VT, N0, NewCFP, Flags); } // (fadd (fadd x, x), (fmul x, c)) -> (fmul x, c+2) if (CFP11 && !CFP10 && N0.getOpcode() == ISD::FADD && N0.getOperand(0) == N0.getOperand(1) && N1.getOperand(0) == N0.getOperand(0)) { SDValue NewCFP = DAG.getNode(ISD::FADD, DL, VT, N1.getOperand(1), DAG.getConstantFP(2.0, DL, VT), Flags); return DAG.getNode(ISD::FMUL, DL, VT, N1.getOperand(0), NewCFP, Flags); } } if (N0.getOpcode() == ISD::FADD) { bool CFP00 = isConstantFPBuildVectorOrConstantFP(N0.getOperand(0)); // (fadd (fadd x, x), x) -> (fmul x, 3.0) if (!CFP00 && N0.getOperand(0) == N0.getOperand(1) && (N0.getOperand(0) == N1)) { return DAG.getNode(ISD::FMUL, DL, VT, N1, DAG.getConstantFP(3.0, DL, VT), Flags); } } if (N1.getOpcode() == ISD::FADD) { bool CFP10 = isConstantFPBuildVectorOrConstantFP(N1.getOperand(0)); // (fadd x, (fadd x, x)) -> (fmul x, 3.0) if (!CFP10 && N1.getOperand(0) == N1.getOperand(1) && N1.getOperand(0) == N0) { return DAG.getNode(ISD::FMUL, DL, VT, N0, DAG.getConstantFP(3.0, DL, VT), Flags); } } // (fadd (fadd x, x), (fadd x, x)) -> (fmul x, 4.0) if (N0.getOpcode() == ISD::FADD && N1.getOpcode() == ISD::FADD && N0.getOperand(0) == N0.getOperand(1) && N1.getOperand(0) == N1.getOperand(1) && N0.getOperand(0) == N1.getOperand(0)) { return DAG.getNode(ISD::FMUL, DL, VT, N0.getOperand(0), DAG.getConstantFP(4.0, DL, VT), Flags); } } } // enable-unsafe-fp-math // FADD -> FMA combines: if (SDValue Fused = visitFADDForFMACombine(N)) { AddToWorklist(Fused.getNode()); return Fused; } return SDValue(); } SDValue DAGCombiner::visitFSUB(SDNode *N) { SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); ConstantFPSDNode *N0CFP = isConstOrConstSplatFP(N0, true); ConstantFPSDNode *N1CFP = isConstOrConstSplatFP(N1, true); EVT VT = N->getValueType(0); SDLoc DL(N); const TargetOptions &Options = DAG.getTarget().Options; const SDNodeFlags Flags = N->getFlags(); // fold vector ops if (VT.isVector()) if (SDValue FoldedVOp = SimplifyVBinOp(N)) return FoldedVOp; // fold (fsub c1, c2) -> c1-c2 if (N0CFP && N1CFP) return DAG.getNode(ISD::FSUB, DL, VT, N0, N1, Flags); if (SDValue NewSel = foldBinOpIntoSelect(N)) return NewSel; // (fsub A, 0) -> A if (N1CFP && N1CFP->isZero()) { if (!N1CFP->isNegative() || Options.UnsafeFPMath || Flags.hasNoSignedZeros()) { return N0; } } if (N0 == N1) { // (fsub x, x) -> 0.0 if (Options.UnsafeFPMath || Flags.hasNoNaNs()) return DAG.getConstantFP(0.0f, DL, VT); } // (fsub -0.0, N1) -> -N1 if (N0CFP && N0CFP->isZero()) { if (N0CFP->isNegative() || (Options.NoSignedZerosFPMath || Flags.hasNoSignedZeros())) { if (isNegatibleForFree(N1, LegalOperations, TLI, &Options)) return GetNegatedExpression(N1, DAG, LegalOperations); if (!LegalOperations || TLI.isOperationLegal(ISD::FNEG, VT)) return DAG.getNode(ISD::FNEG, DL, VT, N1, Flags); } } if ((Options.UnsafeFPMath || (Flags.hasAllowReassociation() && Flags.hasNoSignedZeros())) && N1.getOpcode() == ISD::FADD) { // X - (X + Y) -> -Y if (N0 == N1->getOperand(0)) return DAG.getNode(ISD::FNEG, DL, VT, N1->getOperand(1), Flags); // X - (Y + X) -> -Y if (N0 == N1->getOperand(1)) return DAG.getNode(ISD::FNEG, DL, VT, N1->getOperand(0), Flags); } // fold (fsub A, (fneg B)) -> (fadd A, B) if (isNegatibleForFree(N1, LegalOperations, TLI, &Options)) return DAG.getNode(ISD::FADD, DL, VT, N0, GetNegatedExpression(N1, DAG, LegalOperations), Flags); // FSUB -> FMA combines: if (SDValue Fused = visitFSUBForFMACombine(N)) { AddToWorklist(Fused.getNode()); return Fused; } return SDValue(); } SDValue DAGCombiner::visitFMUL(SDNode *N) { SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); ConstantFPSDNode *N0CFP = isConstOrConstSplatFP(N0, true); ConstantFPSDNode *N1CFP = isConstOrConstSplatFP(N1, true); EVT VT = N->getValueType(0); SDLoc DL(N); const TargetOptions &Options = DAG.getTarget().Options; const SDNodeFlags Flags = N->getFlags(); // fold vector ops if (VT.isVector()) { // This just handles C1 * C2 for vectors. Other vector folds are below. if (SDValue FoldedVOp = SimplifyVBinOp(N)) return FoldedVOp; } // fold (fmul c1, c2) -> c1*c2 if (N0CFP && N1CFP) return DAG.getNode(ISD::FMUL, DL, VT, N0, N1, Flags); // canonicalize constant to RHS if (isConstantFPBuildVectorOrConstantFP(N0) && !isConstantFPBuildVectorOrConstantFP(N1)) return DAG.getNode(ISD::FMUL, DL, VT, N1, N0, Flags); // fold (fmul A, 1.0) -> A if (N1CFP && N1CFP->isExactlyValue(1.0)) return N0; if (SDValue NewSel = foldBinOpIntoSelect(N)) return NewSel; if (Options.UnsafeFPMath || (Flags.hasNoNaNs() && Flags.hasNoSignedZeros())) { // fold (fmul A, 0) -> 0 if (N1CFP && N1CFP->isZero()) return N1; } if (Options.UnsafeFPMath || Flags.hasAllowReassociation()) { // fmul (fmul X, C1), C2 -> fmul X, C1 * C2 if (isConstantFPBuildVectorOrConstantFP(N1) && N0.getOpcode() == ISD::FMUL) { SDValue N00 = N0.getOperand(0); SDValue N01 = N0.getOperand(1); // Avoid an infinite loop by making sure that N00 is not a constant // (the inner multiply has not been constant folded yet). if (isConstantFPBuildVectorOrConstantFP(N01) && !isConstantFPBuildVectorOrConstantFP(N00)) { SDValue MulConsts = DAG.getNode(ISD::FMUL, DL, VT, N01, N1, Flags); return DAG.getNode(ISD::FMUL, DL, VT, N00, MulConsts, Flags); } } // Match a special-case: we convert X * 2.0 into fadd. // fmul (fadd X, X), C -> fmul X, 2.0 * C if (N0.getOpcode() == ISD::FADD && N0.hasOneUse() && N0.getOperand(0) == N0.getOperand(1)) { const SDValue Two = DAG.getConstantFP(2.0, DL, VT); SDValue MulConsts = DAG.getNode(ISD::FMUL, DL, VT, Two, N1, Flags); return DAG.getNode(ISD::FMUL, DL, VT, N0.getOperand(0), MulConsts, Flags); } } // fold (fmul X, 2.0) -> (fadd X, X) if (N1CFP && N1CFP->isExactlyValue(+2.0)) return DAG.getNode(ISD::FADD, DL, VT, N0, N0, Flags); // fold (fmul X, -1.0) -> (fneg X) if (N1CFP && N1CFP->isExactlyValue(-1.0)) if (!LegalOperations || TLI.isOperationLegal(ISD::FNEG, VT)) return DAG.getNode(ISD::FNEG, DL, VT, N0); // fold (fmul (fneg X), (fneg Y)) -> (fmul X, Y) if (char LHSNeg = isNegatibleForFree(N0, LegalOperations, TLI, &Options)) { if (char RHSNeg = isNegatibleForFree(N1, LegalOperations, TLI, &Options)) { // Both can be negated for free, check to see if at least one is cheaper // negated. if (LHSNeg == 2 || RHSNeg == 2) return DAG.getNode(ISD::FMUL, DL, VT, GetNegatedExpression(N0, DAG, LegalOperations), GetNegatedExpression(N1, DAG, LegalOperations), Flags); } } // fold (fmul X, (select (fcmp X > 0.0), -1.0, 1.0)) -> (fneg (fabs X)) // fold (fmul X, (select (fcmp X > 0.0), 1.0, -1.0)) -> (fabs X) if (Flags.hasNoNaNs() && Flags.hasNoSignedZeros() && (N0.getOpcode() == ISD::SELECT || N1.getOpcode() == ISD::SELECT) && TLI.isOperationLegal(ISD::FABS, VT)) { SDValue Select = N0, X = N1; if (Select.getOpcode() != ISD::SELECT) std::swap(Select, X); SDValue Cond = Select.getOperand(0); auto TrueOpnd = dyn_cast(Select.getOperand(1)); auto FalseOpnd = dyn_cast(Select.getOperand(2)); if (TrueOpnd && FalseOpnd && Cond.getOpcode() == ISD::SETCC && Cond.getOperand(0) == X && isa(Cond.getOperand(1)) && cast(Cond.getOperand(1))->isExactlyValue(0.0)) { ISD::CondCode CC = cast(Cond.getOperand(2))->get(); switch (CC) { default: break; case ISD::SETOLT: case ISD::SETULT: case ISD::SETOLE: case ISD::SETULE: case ISD::SETLT: case ISD::SETLE: std::swap(TrueOpnd, FalseOpnd); LLVM_FALLTHROUGH; case ISD::SETOGT: case ISD::SETUGT: case ISD::SETOGE: case ISD::SETUGE: case ISD::SETGT: case ISD::SETGE: if (TrueOpnd->isExactlyValue(-1.0) && FalseOpnd->isExactlyValue(1.0) && TLI.isOperationLegal(ISD::FNEG, VT)) return DAG.getNode(ISD::FNEG, DL, VT, DAG.getNode(ISD::FABS, DL, VT, X)); if (TrueOpnd->isExactlyValue(1.0) && FalseOpnd->isExactlyValue(-1.0)) return DAG.getNode(ISD::FABS, DL, VT, X); break; } } } // FMUL -> FMA combines: if (SDValue Fused = visitFMULForFMADistributiveCombine(N)) { AddToWorklist(Fused.getNode()); return Fused; } return SDValue(); } SDValue DAGCombiner::visitFMA(SDNode *N) { SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); SDValue N2 = N->getOperand(2); ConstantFPSDNode *N0CFP = dyn_cast(N0); ConstantFPSDNode *N1CFP = dyn_cast(N1); EVT VT = N->getValueType(0); SDLoc DL(N); const TargetOptions &Options = DAG.getTarget().Options; // FMA nodes have flags that propagate to the created nodes. const SDNodeFlags Flags = N->getFlags(); bool UnsafeFPMath = Options.UnsafeFPMath || isContractable(N); // Constant fold FMA. if (isa(N0) && isa(N1) && isa(N2)) { return DAG.getNode(ISD::FMA, DL, VT, N0, N1, N2); } if (UnsafeFPMath) { if (N0CFP && N0CFP->isZero()) return N2; if (N1CFP && N1CFP->isZero()) return N2; } // TODO: The FMA node should have flags that propagate to these nodes. if (N0CFP && N0CFP->isExactlyValue(1.0)) return DAG.getNode(ISD::FADD, SDLoc(N), VT, N1, N2); if (N1CFP && N1CFP->isExactlyValue(1.0)) return DAG.getNode(ISD::FADD, SDLoc(N), VT, N0, N2); // Canonicalize (fma c, x, y) -> (fma x, c, y) if (isConstantFPBuildVectorOrConstantFP(N0) && !isConstantFPBuildVectorOrConstantFP(N1)) return DAG.getNode(ISD::FMA, SDLoc(N), VT, N1, N0, N2); if (UnsafeFPMath) { // (fma x, c1, (fmul x, c2)) -> (fmul x, c1+c2) if (N2.getOpcode() == ISD::FMUL && N0 == N2.getOperand(0) && isConstantFPBuildVectorOrConstantFP(N1) && isConstantFPBuildVectorOrConstantFP(N2.getOperand(1))) { return DAG.getNode(ISD::FMUL, DL, VT, N0, DAG.getNode(ISD::FADD, DL, VT, N1, N2.getOperand(1), Flags), Flags); } // (fma (fmul x, c1), c2, y) -> (fma x, c1*c2, y) if (N0.getOpcode() == ISD::FMUL && isConstantFPBuildVectorOrConstantFP(N1) && isConstantFPBuildVectorOrConstantFP(N0.getOperand(1))) { return DAG.getNode(ISD::FMA, DL, VT, N0.getOperand(0), DAG.getNode(ISD::FMUL, DL, VT, N1, N0.getOperand(1), Flags), N2); } } // (fma x, 1, y) -> (fadd x, y) // (fma x, -1, y) -> (fadd (fneg x), y) if (N1CFP) { if (N1CFP->isExactlyValue(1.0)) // TODO: The FMA node should have flags that propagate to this node. return DAG.getNode(ISD::FADD, DL, VT, N0, N2); if (N1CFP->isExactlyValue(-1.0) && (!LegalOperations || TLI.isOperationLegal(ISD::FNEG, VT))) { SDValue RHSNeg = DAG.getNode(ISD::FNEG, DL, VT, N0); AddToWorklist(RHSNeg.getNode()); // TODO: The FMA node should have flags that propagate to this node. return DAG.getNode(ISD::FADD, DL, VT, N2, RHSNeg); } // fma (fneg x), K, y -> fma x -K, y if (N0.getOpcode() == ISD::FNEG && (TLI.isOperationLegal(ISD::ConstantFP, VT) || (N1.hasOneUse() && !TLI.isFPImmLegal(N1CFP->getValueAPF(), VT)))) { return DAG.getNode(ISD::FMA, DL, VT, N0.getOperand(0), DAG.getNode(ISD::FNEG, DL, VT, N1, Flags), N2); } } if (UnsafeFPMath) { // (fma x, c, x) -> (fmul x, (c+1)) if (N1CFP && N0 == N2) { return DAG.getNode(ISD::FMUL, DL, VT, N0, DAG.getNode(ISD::FADD, DL, VT, N1, DAG.getConstantFP(1.0, DL, VT), Flags), Flags); } // (fma x, c, (fneg x)) -> (fmul x, (c-1)) if (N1CFP && N2.getOpcode() == ISD::FNEG && N2.getOperand(0) == N0) { return DAG.getNode(ISD::FMUL, DL, VT, N0, DAG.getNode(ISD::FADD, DL, VT, N1, DAG.getConstantFP(-1.0, DL, VT), Flags), Flags); } } return SDValue(); } // Combine multiple FDIVs with the same divisor into multiple FMULs by the // reciprocal. // E.g., (a / D; b / D;) -> (recip = 1.0 / D; a * recip; b * recip) // Notice that this is not always beneficial. One reason is different targets // may have different costs for FDIV and FMUL, so sometimes the cost of two // FDIVs may be lower than the cost of one FDIV and two FMULs. Another reason // is the critical path is increased from "one FDIV" to "one FDIV + one FMUL". SDValue DAGCombiner::combineRepeatedFPDivisors(SDNode *N) { bool UnsafeMath = DAG.getTarget().Options.UnsafeFPMath; const SDNodeFlags Flags = N->getFlags(); if (!UnsafeMath && !Flags.hasAllowReciprocal()) return SDValue(); // Skip if current node is a reciprocal. SDValue N0 = N->getOperand(0); ConstantFPSDNode *N0CFP = dyn_cast(N0); if (N0CFP && N0CFP->isExactlyValue(1.0)) return SDValue(); // Exit early if the target does not want this transform or if there can't // possibly be enough uses of the divisor to make the transform worthwhile. SDValue N1 = N->getOperand(1); unsigned MinUses = TLI.combineRepeatedFPDivisors(); if (!MinUses || N1->use_size() < MinUses) return SDValue(); // Find all FDIV users of the same divisor. // Use a set because duplicates may be present in the user list. SetVector Users; for (auto *U : N1->uses()) { if (U->getOpcode() == ISD::FDIV && U->getOperand(1) == N1) { // This division is eligible for optimization only if global unsafe math // is enabled or if this division allows reciprocal formation. if (UnsafeMath || U->getFlags().hasAllowReciprocal()) Users.insert(U); } } // Now that we have the actual number of divisor uses, make sure it meets // the minimum threshold specified by the target. if (Users.size() < MinUses) return SDValue(); EVT VT = N->getValueType(0); SDLoc DL(N); SDValue FPOne = DAG.getConstantFP(1.0, DL, VT); SDValue Reciprocal = DAG.getNode(ISD::FDIV, DL, VT, FPOne, N1, Flags); // Dividend / Divisor -> Dividend * Reciprocal for (auto *U : Users) { SDValue Dividend = U->getOperand(0); if (Dividend != FPOne) { SDValue NewNode = DAG.getNode(ISD::FMUL, SDLoc(U), VT, Dividend, Reciprocal, Flags); CombineTo(U, NewNode); } else if (U != Reciprocal.getNode()) { // In the absence of fast-math-flags, this user node is always the // same node as Reciprocal, but with FMF they may be different nodes. CombineTo(U, Reciprocal); } } return SDValue(N, 0); // N was replaced. } SDValue DAGCombiner::visitFDIV(SDNode *N) { SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); ConstantFPSDNode *N0CFP = dyn_cast(N0); ConstantFPSDNode *N1CFP = dyn_cast(N1); EVT VT = N->getValueType(0); SDLoc DL(N); const TargetOptions &Options = DAG.getTarget().Options; SDNodeFlags Flags = N->getFlags(); // fold vector ops if (VT.isVector()) if (SDValue FoldedVOp = SimplifyVBinOp(N)) return FoldedVOp; // fold (fdiv c1, c2) -> c1/c2 if (N0CFP && N1CFP) return DAG.getNode(ISD::FDIV, SDLoc(N), VT, N0, N1, Flags); if (SDValue NewSel = foldBinOpIntoSelect(N)) return NewSel; if (Options.UnsafeFPMath || Flags.hasAllowReciprocal()) { // fold (fdiv X, c2) -> fmul X, 1/c2 if losing precision is acceptable. if (N1CFP) { // Compute the reciprocal 1.0 / c2. const APFloat &N1APF = N1CFP->getValueAPF(); APFloat Recip(N1APF.getSemantics(), 1); // 1.0 APFloat::opStatus st = Recip.divide(N1APF, APFloat::rmNearestTiesToEven); // Only do the transform if the reciprocal is a legal fp immediate that // isn't too nasty (eg NaN, denormal, ...). if ((st == APFloat::opOK || st == APFloat::opInexact) && // Not too nasty (!LegalOperations || // FIXME: custom lowering of ConstantFP might fail (see e.g. ARM // backend)... we should handle this gracefully after Legalize. // TLI.isOperationLegalOrCustom(ISD::ConstantFP, VT) || TLI.isOperationLegal(ISD::ConstantFP, VT) || TLI.isFPImmLegal(Recip, VT))) return DAG.getNode(ISD::FMUL, DL, VT, N0, DAG.getConstantFP(Recip, DL, VT), Flags); } // If this FDIV is part of a reciprocal square root, it may be folded // into a target-specific square root estimate instruction. if (N1.getOpcode() == ISD::FSQRT) { if (SDValue RV = buildRsqrtEstimate(N1.getOperand(0), Flags)) { return DAG.getNode(ISD::FMUL, DL, VT, N0, RV, Flags); } } else if (N1.getOpcode() == ISD::FP_EXTEND && N1.getOperand(0).getOpcode() == ISD::FSQRT) { if (SDValue RV = buildRsqrtEstimate(N1.getOperand(0).getOperand(0), Flags)) { RV = DAG.getNode(ISD::FP_EXTEND, SDLoc(N1), VT, RV); AddToWorklist(RV.getNode()); return DAG.getNode(ISD::FMUL, DL, VT, N0, RV, Flags); } } else if (N1.getOpcode() == ISD::FP_ROUND && N1.getOperand(0).getOpcode() == ISD::FSQRT) { if (SDValue RV = buildRsqrtEstimate(N1.getOperand(0).getOperand(0), Flags)) { RV = DAG.getNode(ISD::FP_ROUND, SDLoc(N1), VT, RV, N1.getOperand(1)); AddToWorklist(RV.getNode()); return DAG.getNode(ISD::FMUL, DL, VT, N0, RV, Flags); } } else if (N1.getOpcode() == ISD::FMUL) { // Look through an FMUL. Even though this won't remove the FDIV directly, // it's still worthwhile to get rid of the FSQRT if possible. SDValue SqrtOp; SDValue OtherOp; if (N1.getOperand(0).getOpcode() == ISD::FSQRT) { SqrtOp = N1.getOperand(0); OtherOp = N1.getOperand(1); } else if (N1.getOperand(1).getOpcode() == ISD::FSQRT) { SqrtOp = N1.getOperand(1); OtherOp = N1.getOperand(0); } if (SqrtOp.getNode()) { // We found a FSQRT, so try to make this fold: // x / (y * sqrt(z)) -> x * (rsqrt(z) / y) if (SDValue RV = buildRsqrtEstimate(SqrtOp.getOperand(0), Flags)) { RV = DAG.getNode(ISD::FDIV, SDLoc(N1), VT, RV, OtherOp, Flags); AddToWorklist(RV.getNode()); return DAG.getNode(ISD::FMUL, DL, VT, N0, RV, Flags); } } } // Fold into a reciprocal estimate and multiply instead of a real divide. if (SDValue RV = BuildReciprocalEstimate(N1, Flags)) { AddToWorklist(RV.getNode()); return DAG.getNode(ISD::FMUL, DL, VT, N0, RV, Flags); } } // (fdiv (fneg X), (fneg Y)) -> (fdiv X, Y) if (char LHSNeg = isNegatibleForFree(N0, LegalOperations, TLI, &Options)) { if (char RHSNeg = isNegatibleForFree(N1, LegalOperations, TLI, &Options)) { // Both can be negated for free, check to see if at least one is cheaper // negated. if (LHSNeg == 2 || RHSNeg == 2) return DAG.getNode(ISD::FDIV, SDLoc(N), VT, GetNegatedExpression(N0, DAG, LegalOperations), GetNegatedExpression(N1, DAG, LegalOperations), Flags); } } if (SDValue CombineRepeatedDivisors = combineRepeatedFPDivisors(N)) return CombineRepeatedDivisors; return SDValue(); } SDValue DAGCombiner::visitFREM(SDNode *N) { SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); ConstantFPSDNode *N0CFP = dyn_cast(N0); ConstantFPSDNode *N1CFP = dyn_cast(N1); EVT VT = N->getValueType(0); // fold (frem c1, c2) -> fmod(c1,c2) if (N0CFP && N1CFP) return DAG.getNode(ISD::FREM, SDLoc(N), VT, N0, N1, N->getFlags()); if (SDValue NewSel = foldBinOpIntoSelect(N)) return NewSel; return SDValue(); } SDValue DAGCombiner::visitFSQRT(SDNode *N) { SDNodeFlags Flags = N->getFlags(); if (!DAG.getTarget().Options.UnsafeFPMath && !Flags.hasApproximateFuncs()) return SDValue(); SDValue N0 = N->getOperand(0); if (TLI.isFsqrtCheap(N0, DAG)) return SDValue(); // FSQRT nodes have flags that propagate to the created nodes. return buildSqrtEstimate(N0, Flags); } /// copysign(x, fp_extend(y)) -> copysign(x, y) /// copysign(x, fp_round(y)) -> copysign(x, y) static inline bool CanCombineFCOPYSIGN_EXTEND_ROUND(SDNode *N) { SDValue N1 = N->getOperand(1); if ((N1.getOpcode() == ISD::FP_EXTEND || N1.getOpcode() == ISD::FP_ROUND)) { // Do not optimize out type conversion of f128 type yet. // For some targets like x86_64, configuration is changed to keep one f128 // value in one SSE register, but instruction selection cannot handle // FCOPYSIGN on SSE registers yet. EVT N1VT = N1->getValueType(0); EVT N1Op0VT = N1->getOperand(0).getValueType(); return (N1VT == N1Op0VT || N1Op0VT != MVT::f128); } return false; } SDValue DAGCombiner::visitFCOPYSIGN(SDNode *N) { SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); bool N0CFP = isConstantFPBuildVectorOrConstantFP(N0); bool N1CFP = isConstantFPBuildVectorOrConstantFP(N1); EVT VT = N->getValueType(0); if (N0CFP && N1CFP) // Constant fold return DAG.getNode(ISD::FCOPYSIGN, SDLoc(N), VT, N0, N1); if (ConstantFPSDNode *N1C = isConstOrConstSplatFP(N->getOperand(1))) { const APFloat &V = N1C->getValueAPF(); // copysign(x, c1) -> fabs(x) iff ispos(c1) // copysign(x, c1) -> fneg(fabs(x)) iff isneg(c1) if (!V.isNegative()) { if (!LegalOperations || TLI.isOperationLegal(ISD::FABS, VT)) return DAG.getNode(ISD::FABS, SDLoc(N), VT, N0); } else { if (!LegalOperations || TLI.isOperationLegal(ISD::FNEG, VT)) return DAG.getNode(ISD::FNEG, SDLoc(N), VT, DAG.getNode(ISD::FABS, SDLoc(N0), VT, N0)); } } // copysign(fabs(x), y) -> copysign(x, y) // copysign(fneg(x), y) -> copysign(x, y) // copysign(copysign(x,z), y) -> copysign(x, y) if (N0.getOpcode() == ISD::FABS || N0.getOpcode() == ISD::FNEG || N0.getOpcode() == ISD::FCOPYSIGN) return DAG.getNode(ISD::FCOPYSIGN, SDLoc(N), VT, N0.getOperand(0), N1); // copysign(x, abs(y)) -> abs(x) if (N1.getOpcode() == ISD::FABS) return DAG.getNode(ISD::FABS, SDLoc(N), VT, N0); // copysign(x, copysign(y,z)) -> copysign(x, z) if (N1.getOpcode() == ISD::FCOPYSIGN) return DAG.getNode(ISD::FCOPYSIGN, SDLoc(N), VT, N0, N1.getOperand(1)); // copysign(x, fp_extend(y)) -> copysign(x, y) // copysign(x, fp_round(y)) -> copysign(x, y) if (CanCombineFCOPYSIGN_EXTEND_ROUND(N)) return DAG.getNode(ISD::FCOPYSIGN, SDLoc(N), VT, N0, N1.getOperand(0)); return SDValue(); } SDValue DAGCombiner::visitFPOW(SDNode *N) { ConstantFPSDNode *ExponentC = isConstOrConstSplatFP(N->getOperand(1)); if (!ExponentC) return SDValue(); // Try to convert x ** (1/3) into cube root. // TODO: Handle the various flavors of long double. // TODO: Since we're approximating, we don't need an exact 1/3 exponent. // Some range near 1/3 should be fine. EVT VT = N->getValueType(0); if ((VT == MVT::f32 && ExponentC->getValueAPF().isExactlyValue(1.0f/3.0f)) || (VT == MVT::f64 && ExponentC->getValueAPF().isExactlyValue(1.0/3.0))) { // pow(-0.0, 1/3) = +0.0; cbrt(-0.0) = -0.0. // pow(-inf, 1/3) = +inf; cbrt(-inf) = -inf. // pow(-val, 1/3) = nan; cbrt(-val) = -num. // For regular numbers, rounding may cause the results to differ. // Therefore, we require { nsz ninf nnan afn } for this transform. // TODO: We could select out the special cases if we don't have nsz/ninf. SDNodeFlags Flags = N->getFlags(); if (!Flags.hasNoSignedZeros() || !Flags.hasNoInfs() || !Flags.hasNoNaNs() || !Flags.hasApproximateFuncs()) return SDValue(); // Do not create a cbrt() libcall if the target does not have it, and do not // turn a pow that has lowering support into a cbrt() libcall. if (!DAG.getLibInfo().has(LibFunc_cbrt) || (!DAG.getTargetLoweringInfo().isOperationExpand(ISD::FPOW, VT) && DAG.getTargetLoweringInfo().isOperationExpand(ISD::FCBRT, VT))) return SDValue(); return DAG.getNode(ISD::FCBRT, SDLoc(N), VT, N->getOperand(0), Flags); } // Try to convert x ** (1/4) into square roots. // x ** (1/2) is canonicalized to sqrt, so we do not bother with that case. // TODO: This could be extended (using a target hook) to handle smaller // power-of-2 fractional exponents. if (ExponentC->getValueAPF().isExactlyValue(0.25)) { // pow(-0.0, 0.25) = +0.0; sqrt(sqrt(-0.0)) = -0.0. // pow(-inf, 0.25) = +inf; sqrt(sqrt(-inf)) = NaN. // For regular numbers, rounding may cause the results to differ. // Therefore, we require { nsz ninf afn } for this transform. // TODO: We could select out the special cases if we don't have nsz/ninf. SDNodeFlags Flags = N->getFlags(); if (!Flags.hasNoSignedZeros() || !Flags.hasNoInfs() || !Flags.hasApproximateFuncs()) return SDValue(); // Don't double the number of libcalls. We are trying to inline fast code. if (!DAG.getTargetLoweringInfo().isOperationLegalOrCustom(ISD::FSQRT, VT)) return SDValue(); // Assume that libcalls are the smallest code. // TODO: This restriction should probably be lifted for vectors. if (DAG.getMachineFunction().getFunction().optForSize()) return SDValue(); // pow(X, 0.25) --> sqrt(sqrt(X)) SDLoc DL(N); SDValue Sqrt = DAG.getNode(ISD::FSQRT, DL, VT, N->getOperand(0), Flags); return DAG.getNode(ISD::FSQRT, DL, VT, Sqrt, Flags); } return SDValue(); } static SDValue foldFPToIntToFP(SDNode *N, SelectionDAG &DAG, const TargetLowering &TLI) { // This optimization is guarded by a function attribute because it may produce // unexpected results. Ie, programs may be relying on the platform-specific // undefined behavior when the float-to-int conversion overflows. const Function &F = DAG.getMachineFunction().getFunction(); Attribute StrictOverflow = F.getFnAttribute("strict-float-cast-overflow"); if (StrictOverflow.getValueAsString().equals("false")) return SDValue(); // We only do this if the target has legal ftrunc. Otherwise, we'd likely be // replacing casts with a libcall. We also must be allowed to ignore -0.0 // because FTRUNC will return -0.0 for (-1.0, -0.0), but using integer // conversions would return +0.0. // FIXME: We should be able to use node-level FMF here. // TODO: If strict math, should we use FABS (+ range check for signed cast)? EVT VT = N->getValueType(0); if (!TLI.isOperationLegal(ISD::FTRUNC, VT) || !DAG.getTarget().Options.NoSignedZerosFPMath) return SDValue(); // fptosi/fptoui round towards zero, so converting from FP to integer and // back is the same as an 'ftrunc': [us]itofp (fpto[us]i X) --> ftrunc X SDValue N0 = N->getOperand(0); if (N->getOpcode() == ISD::SINT_TO_FP && N0.getOpcode() == ISD::FP_TO_SINT && N0.getOperand(0).getValueType() == VT) return DAG.getNode(ISD::FTRUNC, SDLoc(N), VT, N0.getOperand(0)); if (N->getOpcode() == ISD::UINT_TO_FP && N0.getOpcode() == ISD::FP_TO_UINT && N0.getOperand(0).getValueType() == VT) return DAG.getNode(ISD::FTRUNC, SDLoc(N), VT, N0.getOperand(0)); return SDValue(); } SDValue DAGCombiner::visitSINT_TO_FP(SDNode *N) { SDValue N0 = N->getOperand(0); EVT VT = N->getValueType(0); EVT OpVT = N0.getValueType(); // fold (sint_to_fp c1) -> c1fp if (DAG.isConstantIntBuildVectorOrConstantInt(N0) && // ...but only if the target supports immediate floating-point values (!LegalOperations || TLI.isOperationLegalOrCustom(ISD::ConstantFP, VT))) return DAG.getNode(ISD::SINT_TO_FP, SDLoc(N), VT, N0); // If the input is a legal type, and SINT_TO_FP is not legal on this target, // but UINT_TO_FP is legal on this target, try to convert. if (!hasOperation(ISD::SINT_TO_FP, OpVT) && hasOperation(ISD::UINT_TO_FP, OpVT)) { // If the sign bit is known to be zero, we can change this to UINT_TO_FP. if (DAG.SignBitIsZero(N0)) return DAG.getNode(ISD::UINT_TO_FP, SDLoc(N), VT, N0); } // The next optimizations are desirable only if SELECT_CC can be lowered. if (TLI.isOperationLegalOrCustom(ISD::SELECT_CC, VT) || !LegalOperations) { // fold (sint_to_fp (setcc x, y, cc)) -> (select_cc x, y, -1.0, 0.0,, cc) if (N0.getOpcode() == ISD::SETCC && N0.getValueType() == MVT::i1 && !VT.isVector() && (!LegalOperations || TLI.isOperationLegalOrCustom(ISD::ConstantFP, VT))) { SDLoc DL(N); SDValue Ops[] = { N0.getOperand(0), N0.getOperand(1), DAG.getConstantFP(-1.0, DL, VT), DAG.getConstantFP(0.0, DL, VT), N0.getOperand(2) }; return DAG.getNode(ISD::SELECT_CC, DL, VT, Ops); } // fold (sint_to_fp (zext (setcc x, y, cc))) -> // (select_cc x, y, 1.0, 0.0,, cc) if (N0.getOpcode() == ISD::ZERO_EXTEND && N0.getOperand(0).getOpcode() == ISD::SETCC &&!VT.isVector() && (!LegalOperations || TLI.isOperationLegalOrCustom(ISD::ConstantFP, VT))) { SDLoc DL(N); SDValue Ops[] = { N0.getOperand(0).getOperand(0), N0.getOperand(0).getOperand(1), DAG.getConstantFP(1.0, DL, VT), DAG.getConstantFP(0.0, DL, VT), N0.getOperand(0).getOperand(2) }; return DAG.getNode(ISD::SELECT_CC, DL, VT, Ops); } } if (SDValue FTrunc = foldFPToIntToFP(N, DAG, TLI)) return FTrunc; return SDValue(); } SDValue DAGCombiner::visitUINT_TO_FP(SDNode *N) { SDValue N0 = N->getOperand(0); EVT VT = N->getValueType(0); EVT OpVT = N0.getValueType(); // fold (uint_to_fp c1) -> c1fp if (DAG.isConstantIntBuildVectorOrConstantInt(N0) && // ...but only if the target supports immediate floating-point values (!LegalOperations || TLI.isOperationLegalOrCustom(ISD::ConstantFP, VT))) return DAG.getNode(ISD::UINT_TO_FP, SDLoc(N), VT, N0); // If the input is a legal type, and UINT_TO_FP is not legal on this target, // but SINT_TO_FP is legal on this target, try to convert. if (!hasOperation(ISD::UINT_TO_FP, OpVT) && hasOperation(ISD::SINT_TO_FP, OpVT)) { // If the sign bit is known to be zero, we can change this to SINT_TO_FP. if (DAG.SignBitIsZero(N0)) return DAG.getNode(ISD::SINT_TO_FP, SDLoc(N), VT, N0); } // The next optimizations are desirable only if SELECT_CC can be lowered. if (TLI.isOperationLegalOrCustom(ISD::SELECT_CC, VT) || !LegalOperations) { // fold (uint_to_fp (setcc x, y, cc)) -> (select_cc x, y, -1.0, 0.0,, cc) if (N0.getOpcode() == ISD::SETCC && !VT.isVector() && (!LegalOperations || TLI.isOperationLegalOrCustom(ISD::ConstantFP, VT))) { SDLoc DL(N); SDValue Ops[] = { N0.getOperand(0), N0.getOperand(1), DAG.getConstantFP(1.0, DL, VT), DAG.getConstantFP(0.0, DL, VT), N0.getOperand(2) }; return DAG.getNode(ISD::SELECT_CC, DL, VT, Ops); } } if (SDValue FTrunc = foldFPToIntToFP(N, DAG, TLI)) return FTrunc; return SDValue(); } // Fold (fp_to_{s/u}int ({s/u}int_to_fpx)) -> zext x, sext x, trunc x, or x static SDValue FoldIntToFPToInt(SDNode *N, SelectionDAG &DAG) { SDValue N0 = N->getOperand(0); EVT VT = N->getValueType(0); if (N0.getOpcode() != ISD::UINT_TO_FP && N0.getOpcode() != ISD::SINT_TO_FP) return SDValue(); SDValue Src = N0.getOperand(0); EVT SrcVT = Src.getValueType(); bool IsInputSigned = N0.getOpcode() == ISD::SINT_TO_FP; bool IsOutputSigned = N->getOpcode() == ISD::FP_TO_SINT; // We can safely assume the conversion won't overflow the output range, // because (for example) (uint8_t)18293.f is undefined behavior. // Since we can assume the conversion won't overflow, our decision as to // whether the input will fit in the float should depend on the minimum // of the input range and output range. // This means this is also safe for a signed input and unsigned output, since // a negative input would lead to undefined behavior. unsigned InputSize = (int)SrcVT.getScalarSizeInBits() - IsInputSigned; unsigned OutputSize = (int)VT.getScalarSizeInBits() - IsOutputSigned; unsigned ActualSize = std::min(InputSize, OutputSize); const fltSemantics &sem = DAG.EVTToAPFloatSemantics(N0.getValueType()); // We can only fold away the float conversion if the input range can be // represented exactly in the float range. if (APFloat::semanticsPrecision(sem) >= ActualSize) { if (VT.getScalarSizeInBits() > SrcVT.getScalarSizeInBits()) { unsigned ExtOp = IsInputSigned && IsOutputSigned ? ISD::SIGN_EXTEND : ISD::ZERO_EXTEND; return DAG.getNode(ExtOp, SDLoc(N), VT, Src); } if (VT.getScalarSizeInBits() < SrcVT.getScalarSizeInBits()) return DAG.getNode(ISD::TRUNCATE, SDLoc(N), VT, Src); return DAG.getBitcast(VT, Src); } return SDValue(); } SDValue DAGCombiner::visitFP_TO_SINT(SDNode *N) { SDValue N0 = N->getOperand(0); EVT VT = N->getValueType(0); // fold (fp_to_sint c1fp) -> c1 if (isConstantFPBuildVectorOrConstantFP(N0)) return DAG.getNode(ISD::FP_TO_SINT, SDLoc(N), VT, N0); return FoldIntToFPToInt(N, DAG); } SDValue DAGCombiner::visitFP_TO_UINT(SDNode *N) { SDValue N0 = N->getOperand(0); EVT VT = N->getValueType(0); // fold (fp_to_uint c1fp) -> c1 if (isConstantFPBuildVectorOrConstantFP(N0)) return DAG.getNode(ISD::FP_TO_UINT, SDLoc(N), VT, N0); return FoldIntToFPToInt(N, DAG); } SDValue DAGCombiner::visitFP_ROUND(SDNode *N) { SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); ConstantFPSDNode *N0CFP = dyn_cast(N0); EVT VT = N->getValueType(0); // fold (fp_round c1fp) -> c1fp if (N0CFP) return DAG.getNode(ISD::FP_ROUND, SDLoc(N), VT, N0, N1); // fold (fp_round (fp_extend x)) -> x if (N0.getOpcode() == ISD::FP_EXTEND && VT == N0.getOperand(0).getValueType()) return N0.getOperand(0); // fold (fp_round (fp_round x)) -> (fp_round x) if (N0.getOpcode() == ISD::FP_ROUND) { const bool NIsTrunc = N->getConstantOperandVal(1) == 1; const bool N0IsTrunc = N0.getConstantOperandVal(1) == 1; // Skip this folding if it results in an fp_round from f80 to f16. // // f80 to f16 always generates an expensive (and as yet, unimplemented) // libcall to __truncxfhf2 instead of selecting native f16 conversion // instructions from f32 or f64. Moreover, the first (value-preserving) // fp_round from f80 to either f32 or f64 may become a NOP in platforms like // x86. if (N0.getOperand(0).getValueType() == MVT::f80 && VT == MVT::f16) return SDValue(); // If the first fp_round isn't a value preserving truncation, it might // introduce a tie in the second fp_round, that wouldn't occur in the // single-step fp_round we want to fold to. // In other words, double rounding isn't the same as rounding. // Also, this is a value preserving truncation iff both fp_round's are. if (DAG.getTarget().Options.UnsafeFPMath || N0IsTrunc) { SDLoc DL(N); return DAG.getNode(ISD::FP_ROUND, DL, VT, N0.getOperand(0), DAG.getIntPtrConstant(NIsTrunc && N0IsTrunc, DL)); } } // fold (fp_round (copysign X, Y)) -> (copysign (fp_round X), Y) if (N0.getOpcode() == ISD::FCOPYSIGN && N0.getNode()->hasOneUse()) { SDValue Tmp = DAG.getNode(ISD::FP_ROUND, SDLoc(N0), VT, N0.getOperand(0), N1); AddToWorklist(Tmp.getNode()); return DAG.getNode(ISD::FCOPYSIGN, SDLoc(N), VT, Tmp, N0.getOperand(1)); } if (SDValue NewVSel = matchVSelectOpSizesWithSetCC(N)) return NewVSel; return SDValue(); } SDValue DAGCombiner::visitFP_ROUND_INREG(SDNode *N) { SDValue N0 = N->getOperand(0); EVT VT = N->getValueType(0); EVT EVT = cast(N->getOperand(1))->getVT(); ConstantFPSDNode *N0CFP = dyn_cast(N0); // fold (fp_round_inreg c1fp) -> c1fp if (N0CFP && isTypeLegal(EVT)) { SDLoc DL(N); SDValue Round = DAG.getConstantFP(*N0CFP->getConstantFPValue(), DL, EVT); return DAG.getNode(ISD::FP_EXTEND, DL, VT, Round); } return SDValue(); } SDValue DAGCombiner::visitFP_EXTEND(SDNode *N) { SDValue N0 = N->getOperand(0); EVT VT = N->getValueType(0); // If this is fp_round(fpextend), don't fold it, allow ourselves to be folded. if (N->hasOneUse() && N->use_begin()->getOpcode() == ISD::FP_ROUND) return SDValue(); // fold (fp_extend c1fp) -> c1fp if (isConstantFPBuildVectorOrConstantFP(N0)) return DAG.getNode(ISD::FP_EXTEND, SDLoc(N), VT, N0); // fold (fp_extend (fp16_to_fp op)) -> (fp16_to_fp op) if (N0.getOpcode() == ISD::FP16_TO_FP && TLI.getOperationAction(ISD::FP16_TO_FP, VT) == TargetLowering::Legal) return DAG.getNode(ISD::FP16_TO_FP, SDLoc(N), VT, N0.getOperand(0)); // Turn fp_extend(fp_round(X, 1)) -> x since the fp_round doesn't affect the // value of X. if (N0.getOpcode() == ISD::FP_ROUND && N0.getConstantOperandVal(1) == 1) { SDValue In = N0.getOperand(0); if (In.getValueType() == VT) return In; if (VT.bitsLT(In.getValueType())) return DAG.getNode(ISD::FP_ROUND, SDLoc(N), VT, In, N0.getOperand(1)); return DAG.getNode(ISD::FP_EXTEND, SDLoc(N), VT, In); } // fold (fpext (load x)) -> (fpext (fptrunc (extload x))) if (ISD::isNormalLoad(N0.getNode()) && N0.hasOneUse() && TLI.isLoadExtLegal(ISD::EXTLOAD, VT, N0.getValueType())) { LoadSDNode *LN0 = cast(N0); SDValue ExtLoad = DAG.getExtLoad(ISD::EXTLOAD, SDLoc(N), VT, LN0->getChain(), LN0->getBasePtr(), N0.getValueType(), LN0->getMemOperand()); CombineTo(N, ExtLoad); CombineTo(N0.getNode(), DAG.getNode(ISD::FP_ROUND, SDLoc(N0), N0.getValueType(), ExtLoad, DAG.getIntPtrConstant(1, SDLoc(N0))), ExtLoad.getValue(1)); return SDValue(N, 0); // Return N so it doesn't get rechecked! } if (SDValue NewVSel = matchVSelectOpSizesWithSetCC(N)) return NewVSel; return SDValue(); } SDValue DAGCombiner::visitFCEIL(SDNode *N) { SDValue N0 = N->getOperand(0); EVT VT = N->getValueType(0); // fold (fceil c1) -> fceil(c1) if (isConstantFPBuildVectorOrConstantFP(N0)) return DAG.getNode(ISD::FCEIL, SDLoc(N), VT, N0); return SDValue(); } SDValue DAGCombiner::visitFTRUNC(SDNode *N) { SDValue N0 = N->getOperand(0); EVT VT = N->getValueType(0); // fold (ftrunc c1) -> ftrunc(c1) if (isConstantFPBuildVectorOrConstantFP(N0)) return DAG.getNode(ISD::FTRUNC, SDLoc(N), VT, N0); // fold ftrunc (known rounded int x) -> x // ftrunc is a part of fptosi/fptoui expansion on some targets, so this is // likely to be generated to extract integer from a rounded floating value. switch (N0.getOpcode()) { default: break; case ISD::FRINT: case ISD::FTRUNC: case ISD::FNEARBYINT: case ISD::FFLOOR: case ISD::FCEIL: return N0; } return SDValue(); } SDValue DAGCombiner::visitFFLOOR(SDNode *N) { SDValue N0 = N->getOperand(0); EVT VT = N->getValueType(0); // fold (ffloor c1) -> ffloor(c1) if (isConstantFPBuildVectorOrConstantFP(N0)) return DAG.getNode(ISD::FFLOOR, SDLoc(N), VT, N0); return SDValue(); } // FIXME: FNEG and FABS have a lot in common; refactor. SDValue DAGCombiner::visitFNEG(SDNode *N) { SDValue N0 = N->getOperand(0); EVT VT = N->getValueType(0); // Constant fold FNEG. if (isConstantFPBuildVectorOrConstantFP(N0)) return DAG.getNode(ISD::FNEG, SDLoc(N), VT, N0); if (isNegatibleForFree(N0, LegalOperations, DAG.getTargetLoweringInfo(), &DAG.getTarget().Options)) return GetNegatedExpression(N0, DAG, LegalOperations); // Transform fneg(bitconvert(x)) -> bitconvert(x ^ sign) to avoid loading // constant pool values. if (!TLI.isFNegFree(VT) && N0.getOpcode() == ISD::BITCAST && N0.getNode()->hasOneUse()) { SDValue Int = N0.getOperand(0); EVT IntVT = Int.getValueType(); if (IntVT.isInteger() && !IntVT.isVector()) { APInt SignMask; if (N0.getValueType().isVector()) { // For a vector, get a mask such as 0x80... per scalar element // and splat it. SignMask = APInt::getSignMask(N0.getScalarValueSizeInBits()); SignMask = APInt::getSplat(IntVT.getSizeInBits(), SignMask); } else { // For a scalar, just generate 0x80... SignMask = APInt::getSignMask(IntVT.getSizeInBits()); } SDLoc DL0(N0); Int = DAG.getNode(ISD::XOR, DL0, IntVT, Int, DAG.getConstant(SignMask, DL0, IntVT)); AddToWorklist(Int.getNode()); return DAG.getBitcast(VT, Int); } } // (fneg (fmul c, x)) -> (fmul -c, x) if (N0.getOpcode() == ISD::FMUL && (N0.getNode()->hasOneUse() || !TLI.isFNegFree(VT))) { ConstantFPSDNode *CFP1 = dyn_cast(N0.getOperand(1)); if (CFP1) { APFloat CVal = CFP1->getValueAPF(); CVal.changeSign(); if (Level >= AfterLegalizeDAG && (TLI.isFPImmLegal(CVal, VT) || TLI.isOperationLegal(ISD::ConstantFP, VT))) return DAG.getNode( ISD::FMUL, SDLoc(N), VT, N0.getOperand(0), DAG.getNode(ISD::FNEG, SDLoc(N), VT, N0.getOperand(1)), N0->getFlags()); } } return SDValue(); } static SDValue visitFMinMax(SelectionDAG &DAG, SDNode *N, APFloat (*Op)(const APFloat &, const APFloat &)) { SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); EVT VT = N->getValueType(0); const ConstantFPSDNode *N0CFP = isConstOrConstSplatFP(N0); const ConstantFPSDNode *N1CFP = isConstOrConstSplatFP(N1); if (N0CFP && N1CFP) { const APFloat &C0 = N0CFP->getValueAPF(); const APFloat &C1 = N1CFP->getValueAPF(); return DAG.getConstantFP(Op(C0, C1), SDLoc(N), VT); } // Canonicalize to constant on RHS. if (isConstantFPBuildVectorOrConstantFP(N0) && !isConstantFPBuildVectorOrConstantFP(N1)) return DAG.getNode(N->getOpcode(), SDLoc(N), VT, N1, N0); return SDValue(); } SDValue DAGCombiner::visitFMINNUM(SDNode *N) { return visitFMinMax(DAG, N, minnum); } SDValue DAGCombiner::visitFMAXNUM(SDNode *N) { return visitFMinMax(DAG, N, maxnum); } SDValue DAGCombiner::visitFMINIMUM(SDNode *N) { return visitFMinMax(DAG, N, minimum); } SDValue DAGCombiner::visitFMAXIMUM(SDNode *N) { return visitFMinMax(DAG, N, maximum); } SDValue DAGCombiner::visitFABS(SDNode *N) { SDValue N0 = N->getOperand(0); EVT VT = N->getValueType(0); // fold (fabs c1) -> fabs(c1) if (isConstantFPBuildVectorOrConstantFP(N0)) return DAG.getNode(ISD::FABS, SDLoc(N), VT, N0); // fold (fabs (fabs x)) -> (fabs x) if (N0.getOpcode() == ISD::FABS) return N->getOperand(0); // fold (fabs (fneg x)) -> (fabs x) // fold (fabs (fcopysign x, y)) -> (fabs x) if (N0.getOpcode() == ISD::FNEG || N0.getOpcode() == ISD::FCOPYSIGN) return DAG.getNode(ISD::FABS, SDLoc(N), VT, N0.getOperand(0)); // fabs(bitcast(x)) -> bitcast(x & ~sign) to avoid constant pool loads. if (!TLI.isFAbsFree(VT) && N0.getOpcode() == ISD::BITCAST && N0.hasOneUse()) { SDValue Int = N0.getOperand(0); EVT IntVT = Int.getValueType(); if (IntVT.isInteger() && !IntVT.isVector()) { APInt SignMask; if (N0.getValueType().isVector()) { // For a vector, get a mask such as 0x7f... per scalar element // and splat it. SignMask = ~APInt::getSignMask(N0.getScalarValueSizeInBits()); SignMask = APInt::getSplat(IntVT.getSizeInBits(), SignMask); } else { // For a scalar, just generate 0x7f... SignMask = ~APInt::getSignMask(IntVT.getSizeInBits()); } SDLoc DL(N0); Int = DAG.getNode(ISD::AND, DL, IntVT, Int, DAG.getConstant(SignMask, DL, IntVT)); AddToWorklist(Int.getNode()); return DAG.getBitcast(N->getValueType(0), Int); } } return SDValue(); } SDValue DAGCombiner::visitBRCOND(SDNode *N) { SDValue Chain = N->getOperand(0); SDValue N1 = N->getOperand(1); SDValue N2 = N->getOperand(2); // If N is a constant we could fold this into a fallthrough or unconditional // branch. However that doesn't happen very often in normal code, because // Instcombine/SimplifyCFG should have handled the available opportunities. // If we did this folding here, it would be necessary to update the // MachineBasicBlock CFG, which is awkward. // fold a brcond with a setcc condition into a BR_CC node if BR_CC is legal // on the target. if (N1.getOpcode() == ISD::SETCC && TLI.isOperationLegalOrCustom(ISD::BR_CC, N1.getOperand(0).getValueType())) { return DAG.getNode(ISD::BR_CC, SDLoc(N), MVT::Other, Chain, N1.getOperand(2), N1.getOperand(0), N1.getOperand(1), N2); } if (N1.hasOneUse()) { if (SDValue NewN1 = rebuildSetCC(N1)) return DAG.getNode(ISD::BRCOND, SDLoc(N), MVT::Other, Chain, NewN1, N2); } return SDValue(); } SDValue DAGCombiner::rebuildSetCC(SDValue N) { if (N.getOpcode() == ISD::SRL || (N.getOpcode() == ISD::TRUNCATE && (N.getOperand(0).hasOneUse() && N.getOperand(0).getOpcode() == ISD::SRL))) { // Look pass the truncate. if (N.getOpcode() == ISD::TRUNCATE) N = N.getOperand(0); // Match this pattern so that we can generate simpler code: // // %a = ... // %b = and i32 %a, 2 // %c = srl i32 %b, 1 // brcond i32 %c ... // // into // // %a = ... // %b = and i32 %a, 2 // %c = setcc eq %b, 0 // brcond %c ... // // This applies only when the AND constant value has one bit set and the // SRL constant is equal to the log2 of the AND constant. The back-end is // smart enough to convert the result into a TEST/JMP sequence. SDValue Op0 = N.getOperand(0); SDValue Op1 = N.getOperand(1); if (Op0.getOpcode() == ISD::AND && Op1.getOpcode() == ISD::Constant) { SDValue AndOp1 = Op0.getOperand(1); if (AndOp1.getOpcode() == ISD::Constant) { const APInt &AndConst = cast(AndOp1)->getAPIntValue(); if (AndConst.isPowerOf2() && cast(Op1)->getAPIntValue() == AndConst.logBase2()) { SDLoc DL(N); return DAG.getSetCC(DL, getSetCCResultType(Op0.getValueType()), Op0, DAG.getConstant(0, DL, Op0.getValueType()), ISD::SETNE); } } } } // Transform br(xor(x, y)) -> br(x != y) // Transform br(xor(xor(x,y), 1)) -> br (x == y) if (N.getOpcode() == ISD::XOR) { // Because we may call this on a speculatively constructed // SimplifiedSetCC Node, we need to simplify this node first. // Ideally this should be folded into SimplifySetCC and not // here. For now, grab a handle to N so we don't lose it from // replacements interal to the visit. HandleSDNode XORHandle(N); while (N.getOpcode() == ISD::XOR) { SDValue Tmp = visitXOR(N.getNode()); // No simplification done. if (!Tmp.getNode()) break; // Returning N is form in-visit replacement that may invalidated // N. Grab value from Handle. if (Tmp.getNode() == N.getNode()) N = XORHandle.getValue(); else // Node simplified. Try simplifying again. N = Tmp; } if (N.getOpcode() != ISD::XOR) return N; SDNode *TheXor = N.getNode(); SDValue Op0 = TheXor->getOperand(0); SDValue Op1 = TheXor->getOperand(1); if (Op0.getOpcode() != ISD::SETCC && Op1.getOpcode() != ISD::SETCC) { bool Equal = false; if (isOneConstant(Op0) && Op0.hasOneUse() && Op0.getOpcode() == ISD::XOR) { TheXor = Op0.getNode(); Equal = true; } EVT SetCCVT = N.getValueType(); if (LegalTypes) SetCCVT = getSetCCResultType(SetCCVT); // Replace the uses of XOR with SETCC return DAG.getSetCC(SDLoc(TheXor), SetCCVT, Op0, Op1, Equal ? ISD::SETEQ : ISD::SETNE); } } return SDValue(); } // Operand List for BR_CC: Chain, CondCC, CondLHS, CondRHS, DestBB. // SDValue DAGCombiner::visitBR_CC(SDNode *N) { CondCodeSDNode *CC = cast(N->getOperand(1)); SDValue CondLHS = N->getOperand(2), CondRHS = N->getOperand(3); // If N is a constant we could fold this into a fallthrough or unconditional // branch. However that doesn't happen very often in normal code, because // Instcombine/SimplifyCFG should have handled the available opportunities. // If we did this folding here, it would be necessary to update the // MachineBasicBlock CFG, which is awkward. // Use SimplifySetCC to simplify SETCC's. SDValue Simp = SimplifySetCC(getSetCCResultType(CondLHS.getValueType()), CondLHS, CondRHS, CC->get(), SDLoc(N), false); if (Simp.getNode()) AddToWorklist(Simp.getNode()); // fold to a simpler setcc if (Simp.getNode() && Simp.getOpcode() == ISD::SETCC) return DAG.getNode(ISD::BR_CC, SDLoc(N), MVT::Other, N->getOperand(0), Simp.getOperand(2), Simp.getOperand(0), Simp.getOperand(1), N->getOperand(4)); return SDValue(); } /// Return true if 'Use' is a load or a store that uses N as its base pointer /// and that N may be folded in the load / store addressing mode. static bool canFoldInAddressingMode(SDNode *N, SDNode *Use, SelectionDAG &DAG, const TargetLowering &TLI) { EVT VT; unsigned AS; if (LoadSDNode *LD = dyn_cast(Use)) { if (LD->isIndexed() || LD->getBasePtr().getNode() != N) return false; VT = LD->getMemoryVT(); AS = LD->getAddressSpace(); } else if (StoreSDNode *ST = dyn_cast(Use)) { if (ST->isIndexed() || ST->getBasePtr().getNode() != N) return false; VT = ST->getMemoryVT(); AS = ST->getAddressSpace(); } else return false; TargetLowering::AddrMode AM; if (N->getOpcode() == ISD::ADD) { ConstantSDNode *Offset = dyn_cast(N->getOperand(1)); if (Offset) // [reg +/- imm] AM.BaseOffs = Offset->getSExtValue(); else // [reg +/- reg] AM.Scale = 1; } else if (N->getOpcode() == ISD::SUB) { ConstantSDNode *Offset = dyn_cast(N->getOperand(1)); if (Offset) // [reg +/- imm] AM.BaseOffs = -Offset->getSExtValue(); else // [reg +/- reg] AM.Scale = 1; } else return false; return TLI.isLegalAddressingMode(DAG.getDataLayout(), AM, VT.getTypeForEVT(*DAG.getContext()), AS); } /// Try turning a load/store into a pre-indexed load/store when the base /// pointer is an add or subtract and it has other uses besides the load/store. /// After the transformation, the new indexed load/store has effectively folded /// the add/subtract in and all of its other uses are redirected to the /// new load/store. bool DAGCombiner::CombineToPreIndexedLoadStore(SDNode *N) { if (Level < AfterLegalizeDAG) return false; bool isLoad = true; SDValue Ptr; EVT VT; if (LoadSDNode *LD = dyn_cast(N)) { if (LD->isIndexed()) return false; VT = LD->getMemoryVT(); if (!TLI.isIndexedLoadLegal(ISD::PRE_INC, VT) && !TLI.isIndexedLoadLegal(ISD::PRE_DEC, VT)) return false; Ptr = LD->getBasePtr(); } else if (StoreSDNode *ST = dyn_cast(N)) { if (ST->isIndexed()) return false; VT = ST->getMemoryVT(); if (!TLI.isIndexedStoreLegal(ISD::PRE_INC, VT) && !TLI.isIndexedStoreLegal(ISD::PRE_DEC, VT)) return false; Ptr = ST->getBasePtr(); isLoad = false; } else { return false; } // If the pointer is not an add/sub, or if it doesn't have multiple uses, bail // out. There is no reason to make this a preinc/predec. if ((Ptr.getOpcode() != ISD::ADD && Ptr.getOpcode() != ISD::SUB) || Ptr.getNode()->hasOneUse()) return false; // Ask the target to do addressing mode selection. SDValue BasePtr; SDValue Offset; ISD::MemIndexedMode AM = ISD::UNINDEXED; if (!TLI.getPreIndexedAddressParts(N, BasePtr, Offset, AM, DAG)) return false; // Backends without true r+i pre-indexed forms may need to pass a // constant base with a variable offset so that constant coercion // will work with the patterns in canonical form. bool Swapped = false; if (isa(BasePtr)) { std::swap(BasePtr, Offset); Swapped = true; } // Don't create a indexed load / store with zero offset. if (isNullConstant(Offset)) return false; // Try turning it into a pre-indexed load / store except when: // 1) The new base ptr is a frame index. // 2) If N is a store and the new base ptr is either the same as or is a // predecessor of the value being stored. // 3) Another use of old base ptr is a predecessor of N. If ptr is folded // that would create a cycle. // 4) All uses are load / store ops that use it as old base ptr. // Check #1. Preinc'ing a frame index would require copying the stack pointer // (plus the implicit offset) to a register to preinc anyway. if (isa(BasePtr) || isa(BasePtr)) return false; // Check #2. if (!isLoad) { SDValue Val = cast(N)->getValue(); if (Val == BasePtr || BasePtr.getNode()->isPredecessorOf(Val.getNode())) return false; } // Caches for hasPredecessorHelper. SmallPtrSet Visited; SmallVector Worklist; Worklist.push_back(N); // If the offset is a constant, there may be other adds of constants that // can be folded with this one. We should do this to avoid having to keep // a copy of the original base pointer. SmallVector OtherUses; if (isa(Offset)) for (SDNode::use_iterator UI = BasePtr.getNode()->use_begin(), UE = BasePtr.getNode()->use_end(); UI != UE; ++UI) { SDUse &Use = UI.getUse(); // Skip the use that is Ptr and uses of other results from BasePtr's // node (important for nodes that return multiple results). if (Use.getUser() == Ptr.getNode() || Use != BasePtr) continue; if (SDNode::hasPredecessorHelper(Use.getUser(), Visited, Worklist)) continue; if (Use.getUser()->getOpcode() != ISD::ADD && Use.getUser()->getOpcode() != ISD::SUB) { OtherUses.clear(); break; } SDValue Op1 = Use.getUser()->getOperand((UI.getOperandNo() + 1) & 1); if (!isa(Op1)) { OtherUses.clear(); break; } // FIXME: In some cases, we can be smarter about this. if (Op1.getValueType() != Offset.getValueType()) { OtherUses.clear(); break; } OtherUses.push_back(Use.getUser()); } if (Swapped) std::swap(BasePtr, Offset); // Now check for #3 and #4. bool RealUse = false; for (SDNode *Use : Ptr.getNode()->uses()) { if (Use == N) continue; if (SDNode::hasPredecessorHelper(Use, Visited, Worklist)) return false; // If Ptr may be folded in addressing mode of other use, then it's // not profitable to do this transformation. if (!canFoldInAddressingMode(Ptr.getNode(), Use, DAG, TLI)) RealUse = true; } if (!RealUse) return false; SDValue Result; if (isLoad) Result = DAG.getIndexedLoad(SDValue(N,0), SDLoc(N), BasePtr, Offset, AM); else Result = DAG.getIndexedStore(SDValue(N,0), SDLoc(N), BasePtr, Offset, AM); ++PreIndexedNodes; ++NodesCombined; LLVM_DEBUG(dbgs() << "\nReplacing.4 "; N->dump(&DAG); dbgs() << "\nWith: "; Result.getNode()->dump(&DAG); dbgs() << '\n'); WorklistRemover DeadNodes(*this); if (isLoad) { DAG.ReplaceAllUsesOfValueWith(SDValue(N, 0), Result.getValue(0)); DAG.ReplaceAllUsesOfValueWith(SDValue(N, 1), Result.getValue(2)); } else { DAG.ReplaceAllUsesOfValueWith(SDValue(N, 0), Result.getValue(1)); } // Finally, since the node is now dead, remove it from the graph. deleteAndRecombine(N); if (Swapped) std::swap(BasePtr, Offset); // Replace other uses of BasePtr that can be updated to use Ptr for (unsigned i = 0, e = OtherUses.size(); i != e; ++i) { unsigned OffsetIdx = 1; if (OtherUses[i]->getOperand(OffsetIdx).getNode() == BasePtr.getNode()) OffsetIdx = 0; assert(OtherUses[i]->getOperand(!OffsetIdx).getNode() == BasePtr.getNode() && "Expected BasePtr operand"); // We need to replace ptr0 in the following expression: // x0 * offset0 + y0 * ptr0 = t0 // knowing that // x1 * offset1 + y1 * ptr0 = t1 (the indexed load/store) // // where x0, x1, y0 and y1 in {-1, 1} are given by the types of the // indexed load/store and the expression that needs to be re-written. // // Therefore, we have: // t0 = (x0 * offset0 - x1 * y0 * y1 *offset1) + (y0 * y1) * t1 ConstantSDNode *CN = cast(OtherUses[i]->getOperand(OffsetIdx)); int X0, X1, Y0, Y1; const APInt &Offset0 = CN->getAPIntValue(); APInt Offset1 = cast(Offset)->getAPIntValue(); X0 = (OtherUses[i]->getOpcode() == ISD::SUB && OffsetIdx == 1) ? -1 : 1; Y0 = (OtherUses[i]->getOpcode() == ISD::SUB && OffsetIdx == 0) ? -1 : 1; X1 = (AM == ISD::PRE_DEC && !Swapped) ? -1 : 1; Y1 = (AM == ISD::PRE_DEC && Swapped) ? -1 : 1; unsigned Opcode = (Y0 * Y1 < 0) ? ISD::SUB : ISD::ADD; APInt CNV = Offset0; if (X0 < 0) CNV = -CNV; if (X1 * Y0 * Y1 < 0) CNV = CNV + Offset1; else CNV = CNV - Offset1; SDLoc DL(OtherUses[i]); // We can now generate the new expression. SDValue NewOp1 = DAG.getConstant(CNV, DL, CN->getValueType(0)); SDValue NewOp2 = Result.getValue(isLoad ? 1 : 0); SDValue NewUse = DAG.getNode(Opcode, DL, OtherUses[i]->getValueType(0), NewOp1, NewOp2); DAG.ReplaceAllUsesOfValueWith(SDValue(OtherUses[i], 0), NewUse); deleteAndRecombine(OtherUses[i]); } // Replace the uses of Ptr with uses of the updated base value. DAG.ReplaceAllUsesOfValueWith(Ptr, Result.getValue(isLoad ? 1 : 0)); deleteAndRecombine(Ptr.getNode()); AddToWorklist(Result.getNode()); return true; } /// Try to combine a load/store with a add/sub of the base pointer node into a /// post-indexed load/store. The transformation folded the add/subtract into the /// new indexed load/store effectively and all of its uses are redirected to the /// new load/store. bool DAGCombiner::CombineToPostIndexedLoadStore(SDNode *N) { if (Level < AfterLegalizeDAG) return false; bool isLoad = true; SDValue Ptr; EVT VT; if (LoadSDNode *LD = dyn_cast(N)) { if (LD->isIndexed()) return false; VT = LD->getMemoryVT(); if (!TLI.isIndexedLoadLegal(ISD::POST_INC, VT) && !TLI.isIndexedLoadLegal(ISD::POST_DEC, VT)) return false; Ptr = LD->getBasePtr(); } else if (StoreSDNode *ST = dyn_cast(N)) { if (ST->isIndexed()) return false; VT = ST->getMemoryVT(); if (!TLI.isIndexedStoreLegal(ISD::POST_INC, VT) && !TLI.isIndexedStoreLegal(ISD::POST_DEC, VT)) return false; Ptr = ST->getBasePtr(); isLoad = false; } else { return false; } if (Ptr.getNode()->hasOneUse()) return false; for (SDNode *Op : Ptr.getNode()->uses()) { if (Op == N || (Op->getOpcode() != ISD::ADD && Op->getOpcode() != ISD::SUB)) continue; SDValue BasePtr; SDValue Offset; ISD::MemIndexedMode AM = ISD::UNINDEXED; if (TLI.getPostIndexedAddressParts(N, Op, BasePtr, Offset, AM, DAG)) { // Don't create a indexed load / store with zero offset. if (isNullConstant(Offset)) continue; // Try turning it into a post-indexed load / store except when // 1) All uses are load / store ops that use it as base ptr (and // it may be folded as addressing mmode). // 2) Op must be independent of N, i.e. Op is neither a predecessor // nor a successor of N. Otherwise, if Op is folded that would // create a cycle. if (isa(BasePtr) || isa(BasePtr)) continue; // Check for #1. bool TryNext = false; for (SDNode *Use : BasePtr.getNode()->uses()) { if (Use == Ptr.getNode()) continue; // If all the uses are load / store addresses, then don't do the // transformation. if (Use->getOpcode() == ISD::ADD || Use->getOpcode() == ISD::SUB){ bool RealUse = false; for (SDNode *UseUse : Use->uses()) { if (!canFoldInAddressingMode(Use, UseUse, DAG, TLI)) RealUse = true; } if (!RealUse) { TryNext = true; break; } } } if (TryNext) continue; // Check for #2. SmallPtrSet Visited; SmallVector Worklist; // Ptr is predecessor to both N and Op. Visited.insert(Ptr.getNode()); Worklist.push_back(N); Worklist.push_back(Op); if (!SDNode::hasPredecessorHelper(N, Visited, Worklist) && !SDNode::hasPredecessorHelper(Op, Visited, Worklist)) { SDValue Result = isLoad ? DAG.getIndexedLoad(SDValue(N,0), SDLoc(N), BasePtr, Offset, AM) : DAG.getIndexedStore(SDValue(N,0), SDLoc(N), BasePtr, Offset, AM); ++PostIndexedNodes; ++NodesCombined; LLVM_DEBUG(dbgs() << "\nReplacing.5 "; N->dump(&DAG); dbgs() << "\nWith: "; Result.getNode()->dump(&DAG); dbgs() << '\n'); WorklistRemover DeadNodes(*this); if (isLoad) { DAG.ReplaceAllUsesOfValueWith(SDValue(N, 0), Result.getValue(0)); DAG.ReplaceAllUsesOfValueWith(SDValue(N, 1), Result.getValue(2)); } else { DAG.ReplaceAllUsesOfValueWith(SDValue(N, 0), Result.getValue(1)); } // Finally, since the node is now dead, remove it from the graph. deleteAndRecombine(N); // Replace the uses of Use with uses of the updated base value. DAG.ReplaceAllUsesOfValueWith(SDValue(Op, 0), Result.getValue(isLoad ? 1 : 0)); deleteAndRecombine(Op); return true; } } } return false; } /// Return the base-pointer arithmetic from an indexed \p LD. SDValue DAGCombiner::SplitIndexingFromLoad(LoadSDNode *LD) { ISD::MemIndexedMode AM = LD->getAddressingMode(); assert(AM != ISD::UNINDEXED); SDValue BP = LD->getOperand(1); SDValue Inc = LD->getOperand(2); // Some backends use TargetConstants for load offsets, but don't expect // TargetConstants in general ADD nodes. We can convert these constants into // regular Constants (if the constant is not opaque). assert((Inc.getOpcode() != ISD::TargetConstant || !cast(Inc)->isOpaque()) && "Cannot split out indexing using opaque target constants"); if (Inc.getOpcode() == ISD::TargetConstant) { ConstantSDNode *ConstInc = cast(Inc); Inc = DAG.getConstant(*ConstInc->getConstantIntValue(), SDLoc(Inc), ConstInc->getValueType(0)); } unsigned Opc = (AM == ISD::PRE_INC || AM == ISD::POST_INC ? ISD::ADD : ISD::SUB); return DAG.getNode(Opc, SDLoc(LD), BP.getSimpleValueType(), BP, Inc); } static inline int numVectorEltsOrZero(EVT T) { return T.isVector() ? T.getVectorNumElements() : 0; } bool DAGCombiner::getTruncatedStoreValue(StoreSDNode *ST, SDValue &Val) { Val = ST->getValue(); EVT STType = Val.getValueType(); EVT STMemType = ST->getMemoryVT(); if (STType == STMemType) return true; if (isTypeLegal(STMemType)) return false; // fail. if (STType.isFloatingPoint() && STMemType.isFloatingPoint() && TLI.isOperationLegal(ISD::FTRUNC, STMemType)) { Val = DAG.getNode(ISD::FTRUNC, SDLoc(ST), STMemType, Val); return true; } if (numVectorEltsOrZero(STType) == numVectorEltsOrZero(STMemType) && STType.isInteger() && STMemType.isInteger()) { Val = DAG.getNode(ISD::TRUNCATE, SDLoc(ST), STMemType, Val); return true; } if (STType.getSizeInBits() == STMemType.getSizeInBits()) { Val = DAG.getBitcast(STMemType, Val); return true; } return false; // fail. } bool DAGCombiner::extendLoadedValueToExtension(LoadSDNode *LD, SDValue &Val) { EVT LDMemType = LD->getMemoryVT(); EVT LDType = LD->getValueType(0); assert(Val.getValueType() == LDMemType && "Attempting to extend value of non-matching type"); if (LDType == LDMemType) return true; if (LDMemType.isInteger() && LDType.isInteger()) { switch (LD->getExtensionType()) { case ISD::NON_EXTLOAD: Val = DAG.getBitcast(LDType, Val); return true; case ISD::EXTLOAD: Val = DAG.getNode(ISD::ANY_EXTEND, SDLoc(LD), LDType, Val); return true; case ISD::SEXTLOAD: Val = DAG.getNode(ISD::SIGN_EXTEND, SDLoc(LD), LDType, Val); return true; case ISD::ZEXTLOAD: Val = DAG.getNode(ISD::ZERO_EXTEND, SDLoc(LD), LDType, Val); return true; } } return false; } SDValue DAGCombiner::ForwardStoreValueToDirectLoad(LoadSDNode *LD) { if (OptLevel == CodeGenOpt::None || LD->isVolatile()) return SDValue(); SDValue Chain = LD->getOperand(0); StoreSDNode *ST = dyn_cast(Chain.getNode()); if (!ST || ST->isVolatile()) return SDValue(); EVT LDType = LD->getValueType(0); EVT LDMemType = LD->getMemoryVT(); EVT STMemType = ST->getMemoryVT(); EVT STType = ST->getValue().getValueType(); BaseIndexOffset BasePtrLD = BaseIndexOffset::match(LD, DAG); BaseIndexOffset BasePtrST = BaseIndexOffset::match(ST, DAG); int64_t Offset; if (!BasePtrST.equalBaseIndex(BasePtrLD, DAG, Offset)) return SDValue(); // Normalize for Endianness. After this Offset=0 will denote that the least // significant bit in the loaded value maps to the least significant bit in // the stored value). With Offset=n (for n > 0) the loaded value starts at the // n:th least significant byte of the stored value. if (DAG.getDataLayout().isBigEndian()) Offset = (STMemType.getStoreSizeInBits() - LDMemType.getStoreSizeInBits()) / 8 - Offset; // Check that the stored value cover all bits that are loaded. bool STCoversLD = (Offset >= 0) && (Offset * 8 + LDMemType.getSizeInBits() <= STMemType.getSizeInBits()); auto ReplaceLd = [&](LoadSDNode *LD, SDValue Val, SDValue Chain) -> SDValue { if (LD->isIndexed()) { bool IsSub = (LD->getAddressingMode() == ISD::PRE_DEC || LD->getAddressingMode() == ISD::POST_DEC); unsigned Opc = IsSub ? ISD::SUB : ISD::ADD; SDValue Idx = DAG.getNode(Opc, SDLoc(LD), LD->getOperand(1).getValueType(), LD->getOperand(1), LD->getOperand(2)); SDValue Ops[] = {Val, Idx, Chain}; return CombineTo(LD, Ops, 3); } return CombineTo(LD, Val, Chain); }; if (!STCoversLD) return SDValue(); // Memory as copy space (potentially masked). if (Offset == 0 && LDType == STType && STMemType == LDMemType) { // Simple case: Direct non-truncating forwarding if (LDType.getSizeInBits() == LDMemType.getSizeInBits()) return ReplaceLd(LD, ST->getValue(), Chain); // Can we model the truncate and extension with an and mask? if (STType.isInteger() && LDMemType.isInteger() && !STType.isVector() && !LDMemType.isVector() && LD->getExtensionType() != ISD::SEXTLOAD) { // Mask to size of LDMemType auto Mask = DAG.getConstant(APInt::getLowBitsSet(STType.getSizeInBits(), STMemType.getSizeInBits()), SDLoc(ST), STType); auto Val = DAG.getNode(ISD::AND, SDLoc(LD), LDType, ST->getValue(), Mask); return ReplaceLd(LD, Val, Chain); } } // TODO: Deal with nonzero offset. if (LD->getBasePtr().isUndef() || Offset != 0) return SDValue(); // Model necessary truncations / extenstions. SDValue Val; // Truncate Value To Stored Memory Size. do { if (!getTruncatedStoreValue(ST, Val)) continue; if (!isTypeLegal(LDMemType)) continue; if (STMemType != LDMemType) { // TODO: Support vectors? This requires extract_subvector/bitcast. if (!STMemType.isVector() && !LDMemType.isVector() && STMemType.isInteger() && LDMemType.isInteger()) Val = DAG.getNode(ISD::TRUNCATE, SDLoc(LD), LDMemType, Val); else continue; } if (!extendLoadedValueToExtension(LD, Val)) continue; return ReplaceLd(LD, Val, Chain); } while (false); // On failure, cleanup dead nodes we may have created. if (Val->use_empty()) deleteAndRecombine(Val.getNode()); return SDValue(); } SDValue DAGCombiner::visitLOAD(SDNode *N) { LoadSDNode *LD = cast(N); SDValue Chain = LD->getChain(); SDValue Ptr = LD->getBasePtr(); // If load is not volatile and there are no uses of the loaded value (and // the updated indexed value in case of indexed loads), change uses of the // chain value into uses of the chain input (i.e. delete the dead load). if (!LD->isVolatile()) { if (N->getValueType(1) == MVT::Other) { // Unindexed loads. if (!N->hasAnyUseOfValue(0)) { // It's not safe to use the two value CombineTo variant here. e.g. // v1, chain2 = load chain1, loc // v2, chain3 = load chain2, loc // v3 = add v2, c // Now we replace use of chain2 with chain1. This makes the second load // isomorphic to the one we are deleting, and thus makes this load live. LLVM_DEBUG(dbgs() << "\nReplacing.6 "; N->dump(&DAG); dbgs() << "\nWith chain: "; Chain.getNode()->dump(&DAG); dbgs() << "\n"); WorklistRemover DeadNodes(*this); DAG.ReplaceAllUsesOfValueWith(SDValue(N, 1), Chain); AddUsersToWorklist(Chain.getNode()); if (N->use_empty()) deleteAndRecombine(N); return SDValue(N, 0); // Return N so it doesn't get rechecked! } } else { // Indexed loads. assert(N->getValueType(2) == MVT::Other && "Malformed indexed loads?"); // If this load has an opaque TargetConstant offset, then we cannot split // the indexing into an add/sub directly (that TargetConstant may not be // valid for a different type of node, and we cannot convert an opaque // target constant into a regular constant). bool HasOTCInc = LD->getOperand(2).getOpcode() == ISD::TargetConstant && cast(LD->getOperand(2))->isOpaque(); if (!N->hasAnyUseOfValue(0) && ((MaySplitLoadIndex && !HasOTCInc) || !N->hasAnyUseOfValue(1))) { SDValue Undef = DAG.getUNDEF(N->getValueType(0)); SDValue Index; if (N->hasAnyUseOfValue(1) && MaySplitLoadIndex && !HasOTCInc) { Index = SplitIndexingFromLoad(LD); // Try to fold the base pointer arithmetic into subsequent loads and // stores. AddUsersToWorklist(N); } else Index = DAG.getUNDEF(N->getValueType(1)); LLVM_DEBUG(dbgs() << "\nReplacing.7 "; N->dump(&DAG); dbgs() << "\nWith: "; Undef.getNode()->dump(&DAG); dbgs() << " and 2 other values\n"); WorklistRemover DeadNodes(*this); DAG.ReplaceAllUsesOfValueWith(SDValue(N, 0), Undef); DAG.ReplaceAllUsesOfValueWith(SDValue(N, 1), Index); DAG.ReplaceAllUsesOfValueWith(SDValue(N, 2), Chain); deleteAndRecombine(N); return SDValue(N, 0); // Return N so it doesn't get rechecked! } } } // If this load is directly stored, replace the load value with the stored // value. if (auto V = ForwardStoreValueToDirectLoad(LD)) return V; // Try to infer better alignment information than the load already has. if (OptLevel != CodeGenOpt::None && LD->isUnindexed()) { if (unsigned Align = DAG.InferPtrAlignment(Ptr)) { if (Align > LD->getAlignment() && LD->getSrcValueOffset() % Align == 0) { SDValue NewLoad = DAG.getExtLoad( LD->getExtensionType(), SDLoc(N), LD->getValueType(0), Chain, Ptr, LD->getPointerInfo(), LD->getMemoryVT(), Align, LD->getMemOperand()->getFlags(), LD->getAAInfo()); // NewLoad will always be N as we are only refining the alignment assert(NewLoad.getNode() == N); (void)NewLoad; } } } if (LD->isUnindexed()) { // Walk up chain skipping non-aliasing memory nodes. SDValue BetterChain = FindBetterChain(N, Chain); // If there is a better chain. if (Chain != BetterChain) { SDValue ReplLoad; // Replace the chain to void dependency. if (LD->getExtensionType() == ISD::NON_EXTLOAD) { ReplLoad = DAG.getLoad(N->getValueType(0), SDLoc(LD), BetterChain, Ptr, LD->getMemOperand()); } else { ReplLoad = DAG.getExtLoad(LD->getExtensionType(), SDLoc(LD), LD->getValueType(0), BetterChain, Ptr, LD->getMemoryVT(), LD->getMemOperand()); } // Create token factor to keep old chain connected. SDValue Token = DAG.getNode(ISD::TokenFactor, SDLoc(N), MVT::Other, Chain, ReplLoad.getValue(1)); // Replace uses with load result and token factor return CombineTo(N, ReplLoad.getValue(0), Token); } } // Try transforming N to an indexed load. if (CombineToPreIndexedLoadStore(N) || CombineToPostIndexedLoadStore(N)) return SDValue(N, 0); // Try to slice up N to more direct loads if the slices are mapped to // different register banks or pairing can take place. if (SliceUpLoad(N)) return SDValue(N, 0); return SDValue(); } namespace { /// Helper structure used to slice a load in smaller loads. /// Basically a slice is obtained from the following sequence: /// Origin = load Ty1, Base /// Shift = srl Ty1 Origin, CstTy Amount /// Inst = trunc Shift to Ty2 /// /// Then, it will be rewritten into: /// Slice = load SliceTy, Base + SliceOffset /// [Inst = zext Slice to Ty2], only if SliceTy <> Ty2 /// /// SliceTy is deduced from the number of bits that are actually used to /// build Inst. struct LoadedSlice { /// Helper structure used to compute the cost of a slice. struct Cost { /// Are we optimizing for code size. bool ForCodeSize; /// Various cost. unsigned Loads = 0; unsigned Truncates = 0; unsigned CrossRegisterBanksCopies = 0; unsigned ZExts = 0; unsigned Shift = 0; Cost(bool ForCodeSize = false) : ForCodeSize(ForCodeSize) {} /// Get the cost of one isolated slice. Cost(const LoadedSlice &LS, bool ForCodeSize = false) : ForCodeSize(ForCodeSize), Loads(1) { EVT TruncType = LS.Inst->getValueType(0); EVT LoadedType = LS.getLoadedType(); if (TruncType != LoadedType && !LS.DAG->getTargetLoweringInfo().isZExtFree(LoadedType, TruncType)) ZExts = 1; } /// Account for slicing gain in the current cost. /// Slicing provide a few gains like removing a shift or a /// truncate. This method allows to grow the cost of the original /// load with the gain from this slice. void addSliceGain(const LoadedSlice &LS) { // Each slice saves a truncate. const TargetLowering &TLI = LS.DAG->getTargetLoweringInfo(); if (!TLI.isTruncateFree(LS.Inst->getOperand(0).getValueType(), LS.Inst->getValueType(0))) ++Truncates; // If there is a shift amount, this slice gets rid of it. if (LS.Shift) ++Shift; // If this slice can merge a cross register bank copy, account for it. if (LS.canMergeExpensiveCrossRegisterBankCopy()) ++CrossRegisterBanksCopies; } Cost &operator+=(const Cost &RHS) { Loads += RHS.Loads; Truncates += RHS.Truncates; CrossRegisterBanksCopies += RHS.CrossRegisterBanksCopies; ZExts += RHS.ZExts; Shift += RHS.Shift; return *this; } bool operator==(const Cost &RHS) const { return Loads == RHS.Loads && Truncates == RHS.Truncates && CrossRegisterBanksCopies == RHS.CrossRegisterBanksCopies && ZExts == RHS.ZExts && Shift == RHS.Shift; } bool operator!=(const Cost &RHS) const { return !(*this == RHS); } bool operator<(const Cost &RHS) const { // Assume cross register banks copies are as expensive as loads. // FIXME: Do we want some more target hooks? unsigned ExpensiveOpsLHS = Loads + CrossRegisterBanksCopies; unsigned ExpensiveOpsRHS = RHS.Loads + RHS.CrossRegisterBanksCopies; // Unless we are optimizing for code size, consider the // expensive operation first. if (!ForCodeSize && ExpensiveOpsLHS != ExpensiveOpsRHS) return ExpensiveOpsLHS < ExpensiveOpsRHS; return (Truncates + ZExts + Shift + ExpensiveOpsLHS) < (RHS.Truncates + RHS.ZExts + RHS.Shift + ExpensiveOpsRHS); } bool operator>(const Cost &RHS) const { return RHS < *this; } bool operator<=(const Cost &RHS) const { return !(RHS < *this); } bool operator>=(const Cost &RHS) const { return !(*this < RHS); } }; // The last instruction that represent the slice. This should be a // truncate instruction. SDNode *Inst; // The original load instruction. LoadSDNode *Origin; // The right shift amount in bits from the original load. unsigned Shift; // The DAG from which Origin came from. // This is used to get some contextual information about legal types, etc. SelectionDAG *DAG; LoadedSlice(SDNode *Inst = nullptr, LoadSDNode *Origin = nullptr, unsigned Shift = 0, SelectionDAG *DAG = nullptr) : Inst(Inst), Origin(Origin), Shift(Shift), DAG(DAG) {} /// Get the bits used in a chunk of bits \p BitWidth large. /// \return Result is \p BitWidth and has used bits set to 1 and /// not used bits set to 0. APInt getUsedBits() const { // Reproduce the trunc(lshr) sequence: // - Start from the truncated value. // - Zero extend to the desired bit width. // - Shift left. assert(Origin && "No original load to compare against."); unsigned BitWidth = Origin->getValueSizeInBits(0); assert(Inst && "This slice is not bound to an instruction"); assert(Inst->getValueSizeInBits(0) <= BitWidth && "Extracted slice is bigger than the whole type!"); APInt UsedBits(Inst->getValueSizeInBits(0), 0); UsedBits.setAllBits(); UsedBits = UsedBits.zext(BitWidth); UsedBits <<= Shift; return UsedBits; } /// Get the size of the slice to be loaded in bytes. unsigned getLoadedSize() const { unsigned SliceSize = getUsedBits().countPopulation(); assert(!(SliceSize & 0x7) && "Size is not a multiple of a byte."); return SliceSize / 8; } /// Get the type that will be loaded for this slice. /// Note: This may not be the final type for the slice. EVT getLoadedType() const { assert(DAG && "Missing context"); LLVMContext &Ctxt = *DAG->getContext(); return EVT::getIntegerVT(Ctxt, getLoadedSize() * 8); } /// Get the alignment of the load used for this slice. unsigned getAlignment() const { unsigned Alignment = Origin->getAlignment(); unsigned Offset = getOffsetFromBase(); if (Offset != 0) Alignment = MinAlign(Alignment, Alignment + Offset); return Alignment; } /// Check if this slice can be rewritten with legal operations. bool isLegal() const { // An invalid slice is not legal. if (!Origin || !Inst || !DAG) return false; // Offsets are for indexed load only, we do not handle that. if (!Origin->getOffset().isUndef()) return false; const TargetLowering &TLI = DAG->getTargetLoweringInfo(); // Check that the type is legal. EVT SliceType = getLoadedType(); if (!TLI.isTypeLegal(SliceType)) return false; // Check that the load is legal for this type. if (!TLI.isOperationLegal(ISD::LOAD, SliceType)) return false; // Check that the offset can be computed. // 1. Check its type. EVT PtrType = Origin->getBasePtr().getValueType(); if (PtrType == MVT::Untyped || PtrType.isExtended()) return false; // 2. Check that it fits in the immediate. if (!TLI.isLegalAddImmediate(getOffsetFromBase())) return false; // 3. Check that the computation is legal. if (!TLI.isOperationLegal(ISD::ADD, PtrType)) return false; // Check that the zext is legal if it needs one. EVT TruncateType = Inst->getValueType(0); if (TruncateType != SliceType && !TLI.isOperationLegal(ISD::ZERO_EXTEND, TruncateType)) return false; return true; } /// Get the offset in bytes of this slice in the original chunk of /// bits. /// \pre DAG != nullptr. uint64_t getOffsetFromBase() const { assert(DAG && "Missing context."); bool IsBigEndian = DAG->getDataLayout().isBigEndian(); assert(!(Shift & 0x7) && "Shifts not aligned on Bytes are not supported."); uint64_t Offset = Shift / 8; unsigned TySizeInBytes = Origin->getValueSizeInBits(0) / 8; assert(!(Origin->getValueSizeInBits(0) & 0x7) && "The size of the original loaded type is not a multiple of a" " byte."); // If Offset is bigger than TySizeInBytes, it means we are loading all // zeros. This should have been optimized before in the process. assert(TySizeInBytes > Offset && "Invalid shift amount for given loaded size"); if (IsBigEndian) Offset = TySizeInBytes - Offset - getLoadedSize(); return Offset; } /// Generate the sequence of instructions to load the slice /// represented by this object and redirect the uses of this slice to /// this new sequence of instructions. /// \pre this->Inst && this->Origin are valid Instructions and this /// object passed the legal check: LoadedSlice::isLegal returned true. /// \return The last instruction of the sequence used to load the slice. SDValue loadSlice() const { assert(Inst && Origin && "Unable to replace a non-existing slice."); const SDValue &OldBaseAddr = Origin->getBasePtr(); SDValue BaseAddr = OldBaseAddr; // Get the offset in that chunk of bytes w.r.t. the endianness. int64_t Offset = static_cast(getOffsetFromBase()); assert(Offset >= 0 && "Offset too big to fit in int64_t!"); if (Offset) { // BaseAddr = BaseAddr + Offset. EVT ArithType = BaseAddr.getValueType(); SDLoc DL(Origin); BaseAddr = DAG->getNode(ISD::ADD, DL, ArithType, BaseAddr, DAG->getConstant(Offset, DL, ArithType)); } // Create the type of the loaded slice according to its size. EVT SliceType = getLoadedType(); // Create the load for the slice. SDValue LastInst = DAG->getLoad(SliceType, SDLoc(Origin), Origin->getChain(), BaseAddr, Origin->getPointerInfo().getWithOffset(Offset), getAlignment(), Origin->getMemOperand()->getFlags()); // If the final type is not the same as the loaded type, this means that // we have to pad with zero. Create a zero extend for that. EVT FinalType = Inst->getValueType(0); if (SliceType != FinalType) LastInst = DAG->getNode(ISD::ZERO_EXTEND, SDLoc(LastInst), FinalType, LastInst); return LastInst; } /// Check if this slice can be merged with an expensive cross register /// bank copy. E.g., /// i = load i32 /// f = bitcast i32 i to float bool canMergeExpensiveCrossRegisterBankCopy() const { if (!Inst || !Inst->hasOneUse()) return false; SDNode *Use = *Inst->use_begin(); if (Use->getOpcode() != ISD::BITCAST) return false; assert(DAG && "Missing context"); const TargetLowering &TLI = DAG->getTargetLoweringInfo(); EVT ResVT = Use->getValueType(0); const TargetRegisterClass *ResRC = TLI.getRegClassFor(ResVT.getSimpleVT()); const TargetRegisterClass *ArgRC = TLI.getRegClassFor(Use->getOperand(0).getValueType().getSimpleVT()); if (ArgRC == ResRC || !TLI.isOperationLegal(ISD::LOAD, ResVT)) return false; // At this point, we know that we perform a cross-register-bank copy. // Check if it is expensive. const TargetRegisterInfo *TRI = DAG->getSubtarget().getRegisterInfo(); // Assume bitcasts are cheap, unless both register classes do not // explicitly share a common sub class. if (!TRI || TRI->getCommonSubClass(ArgRC, ResRC)) return false; // Check if it will be merged with the load. // 1. Check the alignment constraint. unsigned RequiredAlignment = DAG->getDataLayout().getABITypeAlignment( ResVT.getTypeForEVT(*DAG->getContext())); if (RequiredAlignment > getAlignment()) return false; // 2. Check that the load is a legal operation for that type. if (!TLI.isOperationLegal(ISD::LOAD, ResVT)) return false; // 3. Check that we do not have a zext in the way. if (Inst->getValueType(0) != getLoadedType()) return false; return true; } }; } // end anonymous namespace /// Check that all bits set in \p UsedBits form a dense region, i.e., /// \p UsedBits looks like 0..0 1..1 0..0. static bool areUsedBitsDense(const APInt &UsedBits) { // If all the bits are one, this is dense! if (UsedBits.isAllOnesValue()) return true; // Get rid of the unused bits on the right. APInt NarrowedUsedBits = UsedBits.lshr(UsedBits.countTrailingZeros()); // Get rid of the unused bits on the left. if (NarrowedUsedBits.countLeadingZeros()) NarrowedUsedBits = NarrowedUsedBits.trunc(NarrowedUsedBits.getActiveBits()); // Check that the chunk of bits is completely used. return NarrowedUsedBits.isAllOnesValue(); } /// Check whether or not \p First and \p Second are next to each other /// in memory. This means that there is no hole between the bits loaded /// by \p First and the bits loaded by \p Second. static bool areSlicesNextToEachOther(const LoadedSlice &First, const LoadedSlice &Second) { assert(First.Origin == Second.Origin && First.Origin && "Unable to match different memory origins."); APInt UsedBits = First.getUsedBits(); assert((UsedBits & Second.getUsedBits()) == 0 && "Slices are not supposed to overlap."); UsedBits |= Second.getUsedBits(); return areUsedBitsDense(UsedBits); } /// Adjust the \p GlobalLSCost according to the target /// paring capabilities and the layout of the slices. /// \pre \p GlobalLSCost should account for at least as many loads as /// there is in the slices in \p LoadedSlices. static void adjustCostForPairing(SmallVectorImpl &LoadedSlices, LoadedSlice::Cost &GlobalLSCost) { unsigned NumberOfSlices = LoadedSlices.size(); // If there is less than 2 elements, no pairing is possible. if (NumberOfSlices < 2) return; // Sort the slices so that elements that are likely to be next to each // other in memory are next to each other in the list. llvm::sort(LoadedSlices, [](const LoadedSlice &LHS, const LoadedSlice &RHS) { assert(LHS.Origin == RHS.Origin && "Different bases not implemented."); return LHS.getOffsetFromBase() < RHS.getOffsetFromBase(); }); const TargetLowering &TLI = LoadedSlices[0].DAG->getTargetLoweringInfo(); // First (resp. Second) is the first (resp. Second) potentially candidate // to be placed in a paired load. const LoadedSlice *First = nullptr; const LoadedSlice *Second = nullptr; for (unsigned CurrSlice = 0; CurrSlice < NumberOfSlices; ++CurrSlice, // Set the beginning of the pair. First = Second) { Second = &LoadedSlices[CurrSlice]; // If First is NULL, it means we start a new pair. // Get to the next slice. if (!First) continue; EVT LoadedType = First->getLoadedType(); // If the types of the slices are different, we cannot pair them. if (LoadedType != Second->getLoadedType()) continue; // Check if the target supplies paired loads for this type. unsigned RequiredAlignment = 0; if (!TLI.hasPairedLoad(LoadedType, RequiredAlignment)) { // move to the next pair, this type is hopeless. Second = nullptr; continue; } // Check if we meet the alignment requirement. if (RequiredAlignment > First->getAlignment()) continue; // Check that both loads are next to each other in memory. if (!areSlicesNextToEachOther(*First, *Second)) continue; assert(GlobalLSCost.Loads > 0 && "We save more loads than we created!"); --GlobalLSCost.Loads; // Move to the next pair. Second = nullptr; } } /// Check the profitability of all involved LoadedSlice. /// Currently, it is considered profitable if there is exactly two /// involved slices (1) which are (2) next to each other in memory, and /// whose cost (\see LoadedSlice::Cost) is smaller than the original load (3). /// /// Note: The order of the elements in \p LoadedSlices may be modified, but not /// the elements themselves. /// /// FIXME: When the cost model will be mature enough, we can relax /// constraints (1) and (2). static bool isSlicingProfitable(SmallVectorImpl &LoadedSlices, const APInt &UsedBits, bool ForCodeSize) { unsigned NumberOfSlices = LoadedSlices.size(); if (StressLoadSlicing) return NumberOfSlices > 1; // Check (1). if (NumberOfSlices != 2) return false; // Check (2). if (!areUsedBitsDense(UsedBits)) return false; // Check (3). LoadedSlice::Cost OrigCost(ForCodeSize), GlobalSlicingCost(ForCodeSize); // The original code has one big load. OrigCost.Loads = 1; for (unsigned CurrSlice = 0; CurrSlice < NumberOfSlices; ++CurrSlice) { const LoadedSlice &LS = LoadedSlices[CurrSlice]; // Accumulate the cost of all the slices. LoadedSlice::Cost SliceCost(LS, ForCodeSize); GlobalSlicingCost += SliceCost; // Account as cost in the original configuration the gain obtained // with the current slices. OrigCost.addSliceGain(LS); } // If the target supports paired load, adjust the cost accordingly. adjustCostForPairing(LoadedSlices, GlobalSlicingCost); return OrigCost > GlobalSlicingCost; } /// If the given load, \p LI, is used only by trunc or trunc(lshr) /// operations, split it in the various pieces being extracted. /// /// This sort of thing is introduced by SROA. /// This slicing takes care not to insert overlapping loads. /// \pre LI is a simple load (i.e., not an atomic or volatile load). bool DAGCombiner::SliceUpLoad(SDNode *N) { if (Level < AfterLegalizeDAG) return false; LoadSDNode *LD = cast(N); if (LD->isVolatile() || !ISD::isNormalLoad(LD) || !LD->getValueType(0).isInteger()) return false; // Keep track of already used bits to detect overlapping values. // In that case, we will just abort the transformation. APInt UsedBits(LD->getValueSizeInBits(0), 0); SmallVector LoadedSlices; // Check if this load is used as several smaller chunks of bits. // Basically, look for uses in trunc or trunc(lshr) and record a new chain // of computation for each trunc. for (SDNode::use_iterator UI = LD->use_begin(), UIEnd = LD->use_end(); UI != UIEnd; ++UI) { // Skip the uses of the chain. if (UI.getUse().getResNo() != 0) continue; SDNode *User = *UI; unsigned Shift = 0; // Check if this is a trunc(lshr). if (User->getOpcode() == ISD::SRL && User->hasOneUse() && isa(User->getOperand(1))) { Shift = User->getConstantOperandVal(1); User = *User->use_begin(); } // At this point, User is a Truncate, iff we encountered, trunc or // trunc(lshr). if (User->getOpcode() != ISD::TRUNCATE) return false; // The width of the type must be a power of 2 and greater than 8-bits. // Otherwise the load cannot be represented in LLVM IR. // Moreover, if we shifted with a non-8-bits multiple, the slice // will be across several bytes. We do not support that. unsigned Width = User->getValueSizeInBits(0); if (Width < 8 || !isPowerOf2_32(Width) || (Shift & 0x7)) return false; // Build the slice for this chain of computations. LoadedSlice LS(User, LD, Shift, &DAG); APInt CurrentUsedBits = LS.getUsedBits(); // Check if this slice overlaps with another. if ((CurrentUsedBits & UsedBits) != 0) return false; // Update the bits used globally. UsedBits |= CurrentUsedBits; // Check if the new slice would be legal. if (!LS.isLegal()) return false; // Record the slice. LoadedSlices.push_back(LS); } // Abort slicing if it does not seem to be profitable. if (!isSlicingProfitable(LoadedSlices, UsedBits, ForCodeSize)) return false; ++SlicedLoads; // Rewrite each chain to use an independent load. // By construction, each chain can be represented by a unique load. // Prepare the argument for the new token factor for all the slices. SmallVector ArgChains; for (SmallVectorImpl::const_iterator LSIt = LoadedSlices.begin(), LSItEnd = LoadedSlices.end(); LSIt != LSItEnd; ++LSIt) { SDValue SliceInst = LSIt->loadSlice(); CombineTo(LSIt->Inst, SliceInst, true); if (SliceInst.getOpcode() != ISD::LOAD) SliceInst = SliceInst.getOperand(0); assert(SliceInst->getOpcode() == ISD::LOAD && "It takes more than a zext to get to the loaded slice!!"); ArgChains.push_back(SliceInst.getValue(1)); } SDValue Chain = DAG.getNode(ISD::TokenFactor, SDLoc(LD), MVT::Other, ArgChains); DAG.ReplaceAllUsesOfValueWith(SDValue(N, 1), Chain); AddToWorklist(Chain.getNode()); return true; } /// Check to see if V is (and load (ptr), imm), where the load is having /// specific bytes cleared out. If so, return the byte size being masked out /// and the shift amount. static std::pair CheckForMaskedLoad(SDValue V, SDValue Ptr, SDValue Chain) { std::pair Result(0, 0); // Check for the structure we're looking for. if (V->getOpcode() != ISD::AND || !isa(V->getOperand(1)) || !ISD::isNormalLoad(V->getOperand(0).getNode())) return Result; // Check the chain and pointer. LoadSDNode *LD = cast(V->getOperand(0)); if (LD->getBasePtr() != Ptr) return Result; // Not from same pointer. // This only handles simple types. if (V.getValueType() != MVT::i16 && V.getValueType() != MVT::i32 && V.getValueType() != MVT::i64) return Result; // Check the constant mask. Invert it so that the bits being masked out are // 0 and the bits being kept are 1. Use getSExtValue so that leading bits // follow the sign bit for uniformity. uint64_t NotMask = ~cast(V->getOperand(1))->getSExtValue(); unsigned NotMaskLZ = countLeadingZeros(NotMask); if (NotMaskLZ & 7) return Result; // Must be multiple of a byte. unsigned NotMaskTZ = countTrailingZeros(NotMask); if (NotMaskTZ & 7) return Result; // Must be multiple of a byte. if (NotMaskLZ == 64) return Result; // All zero mask. // See if we have a continuous run of bits. If so, we have 0*1+0* if (countTrailingOnes(NotMask >> NotMaskTZ) + NotMaskTZ + NotMaskLZ != 64) return Result; // Adjust NotMaskLZ down to be from the actual size of the int instead of i64. if (V.getValueType() != MVT::i64 && NotMaskLZ) NotMaskLZ -= 64-V.getValueSizeInBits(); unsigned MaskedBytes = (V.getValueSizeInBits()-NotMaskLZ-NotMaskTZ)/8; switch (MaskedBytes) { case 1: case 2: case 4: break; default: return Result; // All one mask, or 5-byte mask. } // Verify that the first bit starts at a multiple of mask so that the access // is aligned the same as the access width. if (NotMaskTZ && NotMaskTZ/8 % MaskedBytes) return Result; // For narrowing to be valid, it must be the case that the load the // immediately preceeding memory operation before the store. if (LD == Chain.getNode()) ; // ok. else if (Chain->getOpcode() == ISD::TokenFactor && SDValue(LD, 1).hasOneUse()) { // LD has only 1 chain use so they are no indirect dependencies. bool isOk = false; for (const SDValue &ChainOp : Chain->op_values()) if (ChainOp.getNode() == LD) { isOk = true; break; } if (!isOk) return Result; } else return Result; // Fail. Result.first = MaskedBytes; Result.second = NotMaskTZ/8; return Result; } /// Check to see if IVal is something that provides a value as specified by /// MaskInfo. If so, replace the specified store with a narrower store of /// truncated IVal. static SDNode * ShrinkLoadReplaceStoreWithStore(const std::pair &MaskInfo, SDValue IVal, StoreSDNode *St, DAGCombiner *DC) { unsigned NumBytes = MaskInfo.first; unsigned ByteShift = MaskInfo.second; SelectionDAG &DAG = DC->getDAG(); // Check to see if IVal is all zeros in the part being masked in by the 'or' // that uses this. If not, this is not a replacement. APInt Mask = ~APInt::getBitsSet(IVal.getValueSizeInBits(), ByteShift*8, (ByteShift+NumBytes)*8); if (!DAG.MaskedValueIsZero(IVal, Mask)) return nullptr; // Check that it is legal on the target to do this. It is legal if the new // VT we're shrinking to (i8/i16/i32) is legal or we're still before type // legalization. MVT VT = MVT::getIntegerVT(NumBytes*8); if (!DC->isTypeLegal(VT)) return nullptr; // Okay, we can do this! Replace the 'St' store with a store of IVal that is // shifted by ByteShift and truncated down to NumBytes. if (ByteShift) { SDLoc DL(IVal); IVal = DAG.getNode(ISD::SRL, DL, IVal.getValueType(), IVal, DAG.getConstant(ByteShift*8, DL, DC->getShiftAmountTy(IVal.getValueType()))); } // Figure out the offset for the store and the alignment of the access. unsigned StOffset; unsigned NewAlign = St->getAlignment(); if (DAG.getDataLayout().isLittleEndian()) StOffset = ByteShift; else StOffset = IVal.getValueType().getStoreSize() - ByteShift - NumBytes; SDValue Ptr = St->getBasePtr(); if (StOffset) { SDLoc DL(IVal); Ptr = DAG.getNode(ISD::ADD, DL, Ptr.getValueType(), Ptr, DAG.getConstant(StOffset, DL, Ptr.getValueType())); NewAlign = MinAlign(NewAlign, StOffset); } // Truncate down to the new size. IVal = DAG.getNode(ISD::TRUNCATE, SDLoc(IVal), VT, IVal); ++OpsNarrowed; return DAG .getStore(St->getChain(), SDLoc(St), IVal, Ptr, St->getPointerInfo().getWithOffset(StOffset), NewAlign) .getNode(); } /// Look for sequence of load / op / store where op is one of 'or', 'xor', and /// 'and' of immediates. If 'op' is only touching some of the loaded bits, try /// narrowing the load and store if it would end up being a win for performance /// or code size. SDValue DAGCombiner::ReduceLoadOpStoreWidth(SDNode *N) { StoreSDNode *ST = cast(N); if (ST->isVolatile()) return SDValue(); SDValue Chain = ST->getChain(); SDValue Value = ST->getValue(); SDValue Ptr = ST->getBasePtr(); EVT VT = Value.getValueType(); if (ST->isTruncatingStore() || VT.isVector() || !Value.hasOneUse()) return SDValue(); unsigned Opc = Value.getOpcode(); // If this is "store (or X, Y), P" and X is "(and (load P), cst)", where cst // is a byte mask indicating a consecutive number of bytes, check to see if // Y is known to provide just those bytes. If so, we try to replace the // load + replace + store sequence with a single (narrower) store, which makes // the load dead. if (Opc == ISD::OR) { std::pair MaskedLoad; MaskedLoad = CheckForMaskedLoad(Value.getOperand(0), Ptr, Chain); if (MaskedLoad.first) if (SDNode *NewST = ShrinkLoadReplaceStoreWithStore(MaskedLoad, Value.getOperand(1), ST,this)) return SDValue(NewST, 0); // Or is commutative, so try swapping X and Y. MaskedLoad = CheckForMaskedLoad(Value.getOperand(1), Ptr, Chain); if (MaskedLoad.first) if (SDNode *NewST = ShrinkLoadReplaceStoreWithStore(MaskedLoad, Value.getOperand(0), ST,this)) return SDValue(NewST, 0); } if ((Opc != ISD::OR && Opc != ISD::XOR && Opc != ISD::AND) || Value.getOperand(1).getOpcode() != ISD::Constant) return SDValue(); SDValue N0 = Value.getOperand(0); if (ISD::isNormalLoad(N0.getNode()) && N0.hasOneUse() && Chain == SDValue(N0.getNode(), 1)) { LoadSDNode *LD = cast(N0); if (LD->getBasePtr() != Ptr || LD->getPointerInfo().getAddrSpace() != ST->getPointerInfo().getAddrSpace()) return SDValue(); // Find the type to narrow it the load / op / store to. SDValue N1 = Value.getOperand(1); unsigned BitWidth = N1.getValueSizeInBits(); APInt Imm = cast(N1)->getAPIntValue(); if (Opc == ISD::AND) Imm ^= APInt::getAllOnesValue(BitWidth); if (Imm == 0 || Imm.isAllOnesValue()) return SDValue(); unsigned ShAmt = Imm.countTrailingZeros(); unsigned MSB = BitWidth - Imm.countLeadingZeros() - 1; unsigned NewBW = NextPowerOf2(MSB - ShAmt); EVT NewVT = EVT::getIntegerVT(*DAG.getContext(), NewBW); // The narrowing should be profitable, the load/store operation should be // legal (or custom) and the store size should be equal to the NewVT width. while (NewBW < BitWidth && (NewVT.getStoreSizeInBits() != NewBW || !TLI.isOperationLegalOrCustom(Opc, NewVT) || !TLI.isNarrowingProfitable(VT, NewVT))) { NewBW = NextPowerOf2(NewBW); NewVT = EVT::getIntegerVT(*DAG.getContext(), NewBW); } if (NewBW >= BitWidth) return SDValue(); // If the lsb changed does not start at the type bitwidth boundary, // start at the previous one. if (ShAmt % NewBW) ShAmt = (((ShAmt + NewBW - 1) / NewBW) * NewBW) - NewBW; APInt Mask = APInt::getBitsSet(BitWidth, ShAmt, std::min(BitWidth, ShAmt + NewBW)); if ((Imm & Mask) == Imm) { APInt NewImm = (Imm & Mask).lshr(ShAmt).trunc(NewBW); if (Opc == ISD::AND) NewImm ^= APInt::getAllOnesValue(NewBW); uint64_t PtrOff = ShAmt / 8; // For big endian targets, we need to adjust the offset to the pointer to // load the correct bytes. if (DAG.getDataLayout().isBigEndian()) PtrOff = (BitWidth + 7 - NewBW) / 8 - PtrOff; unsigned NewAlign = MinAlign(LD->getAlignment(), PtrOff); Type *NewVTTy = NewVT.getTypeForEVT(*DAG.getContext()); if (NewAlign < DAG.getDataLayout().getABITypeAlignment(NewVTTy)) return SDValue(); SDValue NewPtr = DAG.getNode(ISD::ADD, SDLoc(LD), Ptr.getValueType(), Ptr, DAG.getConstant(PtrOff, SDLoc(LD), Ptr.getValueType())); SDValue NewLD = DAG.getLoad(NewVT, SDLoc(N0), LD->getChain(), NewPtr, LD->getPointerInfo().getWithOffset(PtrOff), NewAlign, LD->getMemOperand()->getFlags(), LD->getAAInfo()); SDValue NewVal = DAG.getNode(Opc, SDLoc(Value), NewVT, NewLD, DAG.getConstant(NewImm, SDLoc(Value), NewVT)); SDValue NewST = DAG.getStore(Chain, SDLoc(N), NewVal, NewPtr, ST->getPointerInfo().getWithOffset(PtrOff), NewAlign); AddToWorklist(NewPtr.getNode()); AddToWorklist(NewLD.getNode()); AddToWorklist(NewVal.getNode()); WorklistRemover DeadNodes(*this); DAG.ReplaceAllUsesOfValueWith(N0.getValue(1), NewLD.getValue(1)); ++OpsNarrowed; return NewST; } } return SDValue(); } /// For a given floating point load / store pair, if the load value isn't used /// by any other operations, then consider transforming the pair to integer /// load / store operations if the target deems the transformation profitable. SDValue DAGCombiner::TransformFPLoadStorePair(SDNode *N) { StoreSDNode *ST = cast(N); SDValue Chain = ST->getChain(); SDValue Value = ST->getValue(); if (ISD::isNormalStore(ST) && ISD::isNormalLoad(Value.getNode()) && Value.hasOneUse() && Chain == SDValue(Value.getNode(), 1)) { LoadSDNode *LD = cast(Value); EVT VT = LD->getMemoryVT(); if (!VT.isFloatingPoint() || VT != ST->getMemoryVT() || LD->isNonTemporal() || ST->isNonTemporal() || LD->getPointerInfo().getAddrSpace() != 0 || ST->getPointerInfo().getAddrSpace() != 0) return SDValue(); EVT IntVT = EVT::getIntegerVT(*DAG.getContext(), VT.getSizeInBits()); if (!TLI.isOperationLegal(ISD::LOAD, IntVT) || !TLI.isOperationLegal(ISD::STORE, IntVT) || !TLI.isDesirableToTransformToIntegerOp(ISD::LOAD, VT) || !TLI.isDesirableToTransformToIntegerOp(ISD::STORE, VT)) return SDValue(); unsigned LDAlign = LD->getAlignment(); unsigned STAlign = ST->getAlignment(); Type *IntVTTy = IntVT.getTypeForEVT(*DAG.getContext()); unsigned ABIAlign = DAG.getDataLayout().getABITypeAlignment(IntVTTy); if (LDAlign < ABIAlign || STAlign < ABIAlign) return SDValue(); SDValue NewLD = DAG.getLoad(IntVT, SDLoc(Value), LD->getChain(), LD->getBasePtr(), LD->getPointerInfo(), LDAlign); SDValue NewST = DAG.getStore(NewLD.getValue(1), SDLoc(N), NewLD, ST->getBasePtr(), ST->getPointerInfo(), STAlign); AddToWorklist(NewLD.getNode()); AddToWorklist(NewST.getNode()); WorklistRemover DeadNodes(*this); DAG.ReplaceAllUsesOfValueWith(Value.getValue(1), NewLD.getValue(1)); ++LdStFP2Int; return NewST; } return SDValue(); } // This is a helper function for visitMUL to check the profitability // of folding (mul (add x, c1), c2) -> (add (mul x, c2), c1*c2). // MulNode is the original multiply, AddNode is (add x, c1), // and ConstNode is c2. // // If the (add x, c1) has multiple uses, we could increase // the number of adds if we make this transformation. // It would only be worth doing this if we can remove a // multiply in the process. Check for that here. // To illustrate: // (A + c1) * c3 // (A + c2) * c3 // We're checking for cases where we have common "c3 * A" expressions. bool DAGCombiner::isMulAddWithConstProfitable(SDNode *MulNode, SDValue &AddNode, SDValue &ConstNode) { APInt Val; // If the add only has one use, this would be OK to do. if (AddNode.getNode()->hasOneUse()) return true; // Walk all the users of the constant with which we're multiplying. for (SDNode *Use : ConstNode->uses()) { if (Use == MulNode) // This use is the one we're on right now. Skip it. continue; if (Use->getOpcode() == ISD::MUL) { // We have another multiply use. SDNode *OtherOp; SDNode *MulVar = AddNode.getOperand(0).getNode(); // OtherOp is what we're multiplying against the constant. if (Use->getOperand(0) == ConstNode) OtherOp = Use->getOperand(1).getNode(); else OtherOp = Use->getOperand(0).getNode(); // Check to see if multiply is with the same operand of our "add". // // ConstNode = CONST // Use = ConstNode * A <-- visiting Use. OtherOp is A. // ... // AddNode = (A + c1) <-- MulVar is A. // = AddNode * ConstNode <-- current visiting instruction. // // If we make this transformation, we will have a common // multiply (ConstNode * A) that we can save. if (OtherOp == MulVar) return true; // Now check to see if a future expansion will give us a common // multiply. // // ConstNode = CONST // AddNode = (A + c1) // ... = AddNode * ConstNode <-- current visiting instruction. // ... // OtherOp = (A + c2) // Use = OtherOp * ConstNode <-- visiting Use. // // If we make this transformation, we will have a common // multiply (CONST * A) after we also do the same transformation // to the "t2" instruction. if (OtherOp->getOpcode() == ISD::ADD && DAG.isConstantIntBuildVectorOrConstantInt(OtherOp->getOperand(1)) && OtherOp->getOperand(0).getNode() == MulVar) return true; } } // Didn't find a case where this would be profitable. return false; } SDValue DAGCombiner::getMergeStoreChains(SmallVectorImpl &StoreNodes, unsigned NumStores) { SmallVector Chains; SmallPtrSet Visited; SDLoc StoreDL(StoreNodes[0].MemNode); for (unsigned i = 0; i < NumStores; ++i) { Visited.insert(StoreNodes[i].MemNode); } // don't include nodes that are children for (unsigned i = 0; i < NumStores; ++i) { if (Visited.count(StoreNodes[i].MemNode->getChain().getNode()) == 0) Chains.push_back(StoreNodes[i].MemNode->getChain()); } assert(Chains.size() > 0 && "Chain should have generated a chain"); return DAG.getNode(ISD::TokenFactor, StoreDL, MVT::Other, Chains); } bool DAGCombiner::MergeStoresOfConstantsOrVecElts( SmallVectorImpl &StoreNodes, EVT MemVT, unsigned NumStores, bool IsConstantSrc, bool UseVector, bool UseTrunc) { // Make sure we have something to merge. if (NumStores < 2) return false; // The latest Node in the DAG. SDLoc DL(StoreNodes[0].MemNode); int64_t ElementSizeBits = MemVT.getStoreSizeInBits(); unsigned SizeInBits = NumStores * ElementSizeBits; unsigned NumMemElts = MemVT.isVector() ? MemVT.getVectorNumElements() : 1; EVT StoreTy; if (UseVector) { unsigned Elts = NumStores * NumMemElts; // Get the type for the merged vector store. StoreTy = EVT::getVectorVT(*DAG.getContext(), MemVT.getScalarType(), Elts); } else StoreTy = EVT::getIntegerVT(*DAG.getContext(), SizeInBits); SDValue StoredVal; if (UseVector) { if (IsConstantSrc) { SmallVector BuildVector; for (unsigned I = 0; I != NumStores; ++I) { StoreSDNode *St = cast(StoreNodes[I].MemNode); SDValue Val = St->getValue(); // If constant is of the wrong type, convert it now. if (MemVT != Val.getValueType()) { Val = peekThroughBitcasts(Val); // Deal with constants of wrong size. if (ElementSizeBits != Val.getValueSizeInBits()) { EVT IntMemVT = EVT::getIntegerVT(*DAG.getContext(), MemVT.getSizeInBits()); if (isa(Val)) { // Not clear how to truncate FP values. return false; } else if (auto *C = dyn_cast(Val)) Val = DAG.getConstant(C->getAPIntValue() .zextOrTrunc(Val.getValueSizeInBits()) .zextOrTrunc(ElementSizeBits), SDLoc(C), IntMemVT); } // Make sure correctly size type is the correct type. Val = DAG.getBitcast(MemVT, Val); } BuildVector.push_back(Val); } StoredVal = DAG.getNode(MemVT.isVector() ? ISD::CONCAT_VECTORS : ISD::BUILD_VECTOR, DL, StoreTy, BuildVector); } else { SmallVector Ops; for (unsigned i = 0; i < NumStores; ++i) { StoreSDNode *St = cast(StoreNodes[i].MemNode); SDValue Val = peekThroughBitcasts(St->getValue()); // All operands of BUILD_VECTOR / CONCAT_VECTOR must be of // type MemVT. If the underlying value is not the correct // type, but it is an extraction of an appropriate vector we // can recast Val to be of the correct type. This may require // converting between EXTRACT_VECTOR_ELT and // EXTRACT_SUBVECTOR. if ((MemVT != Val.getValueType()) && (Val.getOpcode() == ISD::EXTRACT_VECTOR_ELT || Val.getOpcode() == ISD::EXTRACT_SUBVECTOR)) { EVT MemVTScalarTy = MemVT.getScalarType(); // We may need to add a bitcast here to get types to line up. if (MemVTScalarTy != Val.getValueType().getScalarType()) { Val = DAG.getBitcast(MemVT, Val); } else { unsigned OpC = MemVT.isVector() ? ISD::EXTRACT_SUBVECTOR : ISD::EXTRACT_VECTOR_ELT; SDValue Vec = Val.getOperand(0); SDValue Idx = Val.getOperand(1); Val = DAG.getNode(OpC, SDLoc(Val), MemVT, Vec, Idx); } } Ops.push_back(Val); } // Build the extracted vector elements back into a vector. StoredVal = DAG.getNode(MemVT.isVector() ? ISD::CONCAT_VECTORS : ISD::BUILD_VECTOR, DL, StoreTy, Ops); } } else { // We should always use a vector store when merging extracted vector // elements, so this path implies a store of constants. assert(IsConstantSrc && "Merged vector elements should use vector store"); APInt StoreInt(SizeInBits, 0); // Construct a single integer constant which is made of the smaller // constant inputs. bool IsLE = DAG.getDataLayout().isLittleEndian(); for (unsigned i = 0; i < NumStores; ++i) { unsigned Idx = IsLE ? (NumStores - 1 - i) : i; StoreSDNode *St = cast(StoreNodes[Idx].MemNode); SDValue Val = St->getValue(); Val = peekThroughBitcasts(Val); StoreInt <<= ElementSizeBits; if (ConstantSDNode *C = dyn_cast(Val)) { StoreInt |= C->getAPIntValue() .zextOrTrunc(ElementSizeBits) .zextOrTrunc(SizeInBits); } else if (ConstantFPSDNode *C = dyn_cast(Val)) { StoreInt |= C->getValueAPF() .bitcastToAPInt() .zextOrTrunc(ElementSizeBits) .zextOrTrunc(SizeInBits); // If fp truncation is necessary give up for now. if (MemVT.getSizeInBits() != ElementSizeBits) return false; } else { llvm_unreachable("Invalid constant element type"); } } // Create the new Load and Store operations. StoredVal = DAG.getConstant(StoreInt, DL, StoreTy); } LSBaseSDNode *FirstInChain = StoreNodes[0].MemNode; SDValue NewChain = getMergeStoreChains(StoreNodes, NumStores); // make sure we use trunc store if it's necessary to be legal. SDValue NewStore; if (!UseTrunc) { NewStore = DAG.getStore(NewChain, DL, StoredVal, FirstInChain->getBasePtr(), FirstInChain->getPointerInfo(), FirstInChain->getAlignment()); } else { // Must be realized as a trunc store EVT LegalizedStoredValTy = TLI.getTypeToTransformTo(*DAG.getContext(), StoredVal.getValueType()); unsigned LegalizedStoreSize = LegalizedStoredValTy.getSizeInBits(); ConstantSDNode *C = cast(StoredVal); SDValue ExtendedStoreVal = DAG.getConstant(C->getAPIntValue().zextOrTrunc(LegalizedStoreSize), DL, LegalizedStoredValTy); NewStore = DAG.getTruncStore( NewChain, DL, ExtendedStoreVal, FirstInChain->getBasePtr(), FirstInChain->getPointerInfo(), StoredVal.getValueType() /*TVT*/, FirstInChain->getAlignment(), FirstInChain->getMemOperand()->getFlags()); } // Replace all merged stores with the new store. for (unsigned i = 0; i < NumStores; ++i) CombineTo(StoreNodes[i].MemNode, NewStore); AddToWorklist(NewChain.getNode()); return true; } void DAGCombiner::getStoreMergeCandidates( StoreSDNode *St, SmallVectorImpl &StoreNodes, SDNode *&RootNode) { // This holds the base pointer, index, and the offset in bytes from the base // pointer. BaseIndexOffset BasePtr = BaseIndexOffset::match(St, DAG); EVT MemVT = St->getMemoryVT(); SDValue Val = peekThroughBitcasts(St->getValue()); // We must have a base and an offset. if (!BasePtr.getBase().getNode()) return; // Do not handle stores to undef base pointers. if (BasePtr.getBase().isUndef()) return; bool IsConstantSrc = isa(Val) || isa(Val); bool IsExtractVecSrc = (Val.getOpcode() == ISD::EXTRACT_VECTOR_ELT || Val.getOpcode() == ISD::EXTRACT_SUBVECTOR); bool IsLoadSrc = isa(Val); BaseIndexOffset LBasePtr; // Match on loadbaseptr if relevant. EVT LoadVT; if (IsLoadSrc) { auto *Ld = cast(Val); LBasePtr = BaseIndexOffset::match(Ld, DAG); LoadVT = Ld->getMemoryVT(); // Load and store should be the same type. if (MemVT != LoadVT) return; // Loads must only have one use. if (!Ld->hasNUsesOfValue(1, 0)) return; // The memory operands must not be volatile. if (Ld->isVolatile() || Ld->isIndexed()) return; } auto CandidateMatch = [&](StoreSDNode *Other, BaseIndexOffset &Ptr, int64_t &Offset) -> bool { if (Other->isVolatile() || Other->isIndexed()) return false; SDValue Val = peekThroughBitcasts(Other->getValue()); // Allow merging constants of different types as integers. bool NoTypeMatch = (MemVT.isInteger()) ? !MemVT.bitsEq(Other->getMemoryVT()) : Other->getMemoryVT() != MemVT; if (IsLoadSrc) { if (NoTypeMatch) return false; // The Load's Base Ptr must also match if (LoadSDNode *OtherLd = dyn_cast(Val)) { auto LPtr = BaseIndexOffset::match(OtherLd, DAG); if (LoadVT != OtherLd->getMemoryVT()) return false; // Loads must only have one use. if (!OtherLd->hasNUsesOfValue(1, 0)) return false; // The memory operands must not be volatile. if (OtherLd->isVolatile() || OtherLd->isIndexed()) return false; if (!(LBasePtr.equalBaseIndex(LPtr, DAG))) return false; } else return false; } if (IsConstantSrc) { if (NoTypeMatch) return false; if (!(isa(Val) || isa(Val))) return false; } if (IsExtractVecSrc) { // Do not merge truncated stores here. if (Other->isTruncatingStore()) return false; if (!MemVT.bitsEq(Val.getValueType())) return false; if (Val.getOpcode() != ISD::EXTRACT_VECTOR_ELT && Val.getOpcode() != ISD::EXTRACT_SUBVECTOR) return false; } Ptr = BaseIndexOffset::match(Other, DAG); return (BasePtr.equalBaseIndex(Ptr, DAG, Offset)); }; // We looking for a root node which is an ancestor to all mergable // stores. We search up through a load, to our root and then down // through all children. For instance we will find Store{1,2,3} if // St is Store1, Store2. or Store3 where the root is not a load // which always true for nonvolatile ops. TODO: Expand // the search to find all valid candidates through multiple layers of loads. // // Root // |-------|-------| // Load Load Store3 // | | // Store1 Store2 // // FIXME: We should be able to climb and // descend TokenFactors to find candidates as well. RootNode = St->getChain().getNode(); if (LoadSDNode *Ldn = dyn_cast(RootNode)) { RootNode = Ldn->getChain().getNode(); for (auto I = RootNode->use_begin(), E = RootNode->use_end(); I != E; ++I) if (I.getOperandNo() == 0 && isa(*I)) // walk down chain for (auto I2 = (*I)->use_begin(), E2 = (*I)->use_end(); I2 != E2; ++I2) if (I2.getOperandNo() == 0) if (StoreSDNode *OtherST = dyn_cast(*I2)) { BaseIndexOffset Ptr; int64_t PtrDiff; if (CandidateMatch(OtherST, Ptr, PtrDiff)) StoreNodes.push_back(MemOpLink(OtherST, PtrDiff)); } } else for (auto I = RootNode->use_begin(), E = RootNode->use_end(); I != E; ++I) if (I.getOperandNo() == 0) if (StoreSDNode *OtherST = dyn_cast(*I)) { BaseIndexOffset Ptr; int64_t PtrDiff; if (CandidateMatch(OtherST, Ptr, PtrDiff)) StoreNodes.push_back(MemOpLink(OtherST, PtrDiff)); } } // We need to check that merging these stores does not cause a loop in // the DAG. Any store candidate may depend on another candidate // indirectly through its operand (we already consider dependencies // through the chain). Check in parallel by searching up from // non-chain operands of candidates. bool DAGCombiner::checkMergeStoreCandidatesForDependencies( SmallVectorImpl &StoreNodes, unsigned NumStores, SDNode *RootNode) { // FIXME: We should be able to truncate a full search of // predecessors by doing a BFS and keeping tabs the originating // stores from which worklist nodes come from in a similar way to // TokenFactor simplfication. SmallPtrSet Visited; SmallVector Worklist; // RootNode is a predecessor to all candidates so we need not search // past it. Add RootNode (peeking through TokenFactors). Do not count // these towards size check. Worklist.push_back(RootNode); while (!Worklist.empty()) { auto N = Worklist.pop_back_val(); if (!Visited.insert(N).second) continue; // Already present in Visited. if (N->getOpcode() == ISD::TokenFactor) { for (SDValue Op : N->ops()) Worklist.push_back(Op.getNode()); } } // Don't count pruning nodes towards max. unsigned int Max = 1024 + Visited.size(); // Search Ops of store candidates. for (unsigned i = 0; i < NumStores; ++i) { SDNode *N = StoreNodes[i].MemNode; // Of the 4 Store Operands: // * Chain (Op 0) -> We have already considered these // in candidate selection and can be // safely ignored // * Value (Op 1) -> Cycles may happen (e.g. through load chains) // * Address (Op 2) -> Merged addresses may only vary by a fixed constant, // but aren't necessarily fromt the same base node, so // cycles possible (e.g. via indexed store). // * (Op 3) -> Represents the pre or post-indexing offset (or undef for // non-indexed stores). Not constant on all targets (e.g. ARM) // and so can participate in a cycle. for (unsigned j = 1; j < N->getNumOperands(); ++j) Worklist.push_back(N->getOperand(j).getNode()); } // Search through DAG. We can stop early if we find a store node. for (unsigned i = 0; i < NumStores; ++i) if (SDNode::hasPredecessorHelper(StoreNodes[i].MemNode, Visited, Worklist, Max)) return false; return true; } bool DAGCombiner::MergeConsecutiveStores(StoreSDNode *St) { if (OptLevel == CodeGenOpt::None) return false; EVT MemVT = St->getMemoryVT(); int64_t ElementSizeBytes = MemVT.getStoreSize(); unsigned NumMemElts = MemVT.isVector() ? MemVT.getVectorNumElements() : 1; if (MemVT.getSizeInBits() * 2 > MaximumLegalStoreInBits) return false; bool NoVectors = DAG.getMachineFunction().getFunction().hasFnAttribute( Attribute::NoImplicitFloat); // This function cannot currently deal with non-byte-sized memory sizes. if (ElementSizeBytes * 8 != MemVT.getSizeInBits()) return false; if (!MemVT.isSimple()) return false; // Perform an early exit check. Do not bother looking at stored values that // are not constants, loads, or extracted vector elements. SDValue StoredVal = peekThroughBitcasts(St->getValue()); bool IsLoadSrc = isa(StoredVal); bool IsConstantSrc = isa(StoredVal) || isa(StoredVal); bool IsExtractVecSrc = (StoredVal.getOpcode() == ISD::EXTRACT_VECTOR_ELT || StoredVal.getOpcode() == ISD::EXTRACT_SUBVECTOR); if (!IsConstantSrc && !IsLoadSrc && !IsExtractVecSrc) return false; SmallVector StoreNodes; SDNode *RootNode; // Find potential store merge candidates by searching through chain sub-DAG getStoreMergeCandidates(St, StoreNodes, RootNode); // Check if there is anything to merge. if (StoreNodes.size() < 2) return false; // Sort the memory operands according to their distance from the // base pointer. llvm::sort(StoreNodes, [](MemOpLink LHS, MemOpLink RHS) { return LHS.OffsetFromBase < RHS.OffsetFromBase; }); // Store Merge attempts to merge the lowest stores. This generally // works out as if successful, as the remaining stores are checked // after the first collection of stores is merged. However, in the // case that a non-mergeable store is found first, e.g., {p[-2], // p[0], p[1], p[2], p[3]}, we would fail and miss the subsequent // mergeable cases. To prevent this, we prune such stores from the // front of StoreNodes here. bool RV = false; while (StoreNodes.size() > 1) { unsigned StartIdx = 0; while ((StartIdx + 1 < StoreNodes.size()) && StoreNodes[StartIdx].OffsetFromBase + ElementSizeBytes != StoreNodes[StartIdx + 1].OffsetFromBase) ++StartIdx; // Bail if we don't have enough candidates to merge. if (StartIdx + 1 >= StoreNodes.size()) return RV; if (StartIdx) StoreNodes.erase(StoreNodes.begin(), StoreNodes.begin() + StartIdx); // Scan the memory operations on the chain and find the first // non-consecutive store memory address. unsigned NumConsecutiveStores = 1; int64_t StartAddress = StoreNodes[0].OffsetFromBase; // Check that the addresses are consecutive starting from the second // element in the list of stores. for (unsigned i = 1, e = StoreNodes.size(); i < e; ++i) { int64_t CurrAddress = StoreNodes[i].OffsetFromBase; if (CurrAddress - StartAddress != (ElementSizeBytes * i)) break; NumConsecutiveStores = i + 1; } if (NumConsecutiveStores < 2) { StoreNodes.erase(StoreNodes.begin(), StoreNodes.begin() + NumConsecutiveStores); continue; } // The node with the lowest store address. LLVMContext &Context = *DAG.getContext(); const DataLayout &DL = DAG.getDataLayout(); // Store the constants into memory as one consecutive store. if (IsConstantSrc) { while (NumConsecutiveStores >= 2) { LSBaseSDNode *FirstInChain = StoreNodes[0].MemNode; unsigned FirstStoreAS = FirstInChain->getAddressSpace(); unsigned FirstStoreAlign = FirstInChain->getAlignment(); unsigned LastLegalType = 1; unsigned LastLegalVectorType = 1; bool LastIntegerTrunc = false; bool NonZero = false; unsigned FirstZeroAfterNonZero = NumConsecutiveStores; for (unsigned i = 0; i < NumConsecutiveStores; ++i) { StoreSDNode *ST = cast(StoreNodes[i].MemNode); SDValue StoredVal = ST->getValue(); bool IsElementZero = false; if (ConstantSDNode *C = dyn_cast(StoredVal)) IsElementZero = C->isNullValue(); else if (ConstantFPSDNode *C = dyn_cast(StoredVal)) IsElementZero = C->getConstantFPValue()->isNullValue(); if (IsElementZero) { if (NonZero && FirstZeroAfterNonZero == NumConsecutiveStores) FirstZeroAfterNonZero = i; } NonZero |= !IsElementZero; // Find a legal type for the constant store. unsigned SizeInBits = (i + 1) * ElementSizeBytes * 8; EVT StoreTy = EVT::getIntegerVT(Context, SizeInBits); bool IsFast = false; // Break early when size is too large to be legal. if (StoreTy.getSizeInBits() > MaximumLegalStoreInBits) break; if (TLI.isTypeLegal(StoreTy) && TLI.canMergeStoresTo(FirstStoreAS, StoreTy, DAG) && TLI.allowsMemoryAccess(Context, DL, StoreTy, FirstStoreAS, FirstStoreAlign, &IsFast) && IsFast) { LastIntegerTrunc = false; LastLegalType = i + 1; // Or check whether a truncstore is legal. } else if (TLI.getTypeAction(Context, StoreTy) == TargetLowering::TypePromoteInteger) { EVT LegalizedStoredValTy = TLI.getTypeToTransformTo(Context, StoredVal.getValueType()); if (TLI.isTruncStoreLegal(LegalizedStoredValTy, StoreTy) && TLI.canMergeStoresTo(FirstStoreAS, LegalizedStoredValTy, DAG) && TLI.allowsMemoryAccess(Context, DL, StoreTy, FirstStoreAS, FirstStoreAlign, &IsFast) && IsFast) { LastIntegerTrunc = true; LastLegalType = i + 1; } } // We only use vectors if the constant is known to be zero or the // target allows it and the function is not marked with the // noimplicitfloat attribute. if ((!NonZero || TLI.storeOfVectorConstantIsCheap(MemVT, i + 1, FirstStoreAS)) && !NoVectors) { // Find a legal type for the vector store. unsigned Elts = (i + 1) * NumMemElts; EVT Ty = EVT::getVectorVT(Context, MemVT.getScalarType(), Elts); if (TLI.isTypeLegal(Ty) && TLI.isTypeLegal(MemVT) && TLI.canMergeStoresTo(FirstStoreAS, Ty, DAG) && TLI.allowsMemoryAccess(Context, DL, Ty, FirstStoreAS, FirstStoreAlign, &IsFast) && IsFast) LastLegalVectorType = i + 1; } } bool UseVector = (LastLegalVectorType > LastLegalType) && !NoVectors; unsigned NumElem = (UseVector) ? LastLegalVectorType : LastLegalType; // Check if we found a legal integer type that creates a meaningful // merge. if (NumElem < 2) { // We know that candidate stores are in order and of correct // shape. While there is no mergeable sequence from the // beginning one may start later in the sequence. The only // reason a merge of size N could have failed where another of // the same size would not have, is if the alignment has // improved or we've dropped a non-zero value. Drop as many // candidates as we can here. unsigned NumSkip = 1; while ( (NumSkip < NumConsecutiveStores) && (NumSkip < FirstZeroAfterNonZero) && (StoreNodes[NumSkip].MemNode->getAlignment() <= FirstStoreAlign)) NumSkip++; StoreNodes.erase(StoreNodes.begin(), StoreNodes.begin() + NumSkip); NumConsecutiveStores -= NumSkip; continue; } // Check that we can merge these candidates without causing a cycle. if (!checkMergeStoreCandidatesForDependencies(StoreNodes, NumElem, RootNode)) { StoreNodes.erase(StoreNodes.begin(), StoreNodes.begin() + NumElem); NumConsecutiveStores -= NumElem; continue; } RV |= MergeStoresOfConstantsOrVecElts(StoreNodes, MemVT, NumElem, true, UseVector, LastIntegerTrunc); // Remove merged stores for next iteration. StoreNodes.erase(StoreNodes.begin(), StoreNodes.begin() + NumElem); NumConsecutiveStores -= NumElem; } continue; } // When extracting multiple vector elements, try to store them // in one vector store rather than a sequence of scalar stores. if (IsExtractVecSrc) { // Loop on Consecutive Stores on success. while (NumConsecutiveStores >= 2) { LSBaseSDNode *FirstInChain = StoreNodes[0].MemNode; unsigned FirstStoreAS = FirstInChain->getAddressSpace(); unsigned FirstStoreAlign = FirstInChain->getAlignment(); unsigned NumStoresToMerge = 1; for (unsigned i = 0; i < NumConsecutiveStores; ++i) { // Find a legal type for the vector store. unsigned Elts = (i + 1) * NumMemElts; EVT Ty = EVT::getVectorVT(*DAG.getContext(), MemVT.getScalarType(), Elts); bool IsFast; // Break early when size is too large to be legal. if (Ty.getSizeInBits() > MaximumLegalStoreInBits) break; if (TLI.isTypeLegal(Ty) && TLI.canMergeStoresTo(FirstStoreAS, Ty, DAG) && TLI.allowsMemoryAccess(Context, DL, Ty, FirstStoreAS, FirstStoreAlign, &IsFast) && IsFast) NumStoresToMerge = i + 1; } // Check if we found a legal integer type creating a meaningful // merge. if (NumStoresToMerge < 2) { // We know that candidate stores are in order and of correct // shape. While there is no mergeable sequence from the // beginning one may start later in the sequence. The only // reason a merge of size N could have failed where another of // the same size would not have, is if the alignment has // improved. Drop as many candidates as we can here. unsigned NumSkip = 1; while ( (NumSkip < NumConsecutiveStores) && (StoreNodes[NumSkip].MemNode->getAlignment() <= FirstStoreAlign)) NumSkip++; StoreNodes.erase(StoreNodes.begin(), StoreNodes.begin() + NumSkip); NumConsecutiveStores -= NumSkip; continue; } // Check that we can merge these candidates without causing a cycle. if (!checkMergeStoreCandidatesForDependencies( StoreNodes, NumStoresToMerge, RootNode)) { StoreNodes.erase(StoreNodes.begin(), StoreNodes.begin() + NumStoresToMerge); NumConsecutiveStores -= NumStoresToMerge; continue; } RV |= MergeStoresOfConstantsOrVecElts( StoreNodes, MemVT, NumStoresToMerge, false, true, false); StoreNodes.erase(StoreNodes.begin(), StoreNodes.begin() + NumStoresToMerge); NumConsecutiveStores -= NumStoresToMerge; } continue; } // Below we handle the case of multiple consecutive stores that // come from multiple consecutive loads. We merge them into a single // wide load and a single wide store. // Look for load nodes which are used by the stored values. SmallVector LoadNodes; // Find acceptable loads. Loads need to have the same chain (token factor), // must not be zext, volatile, indexed, and they must be consecutive. BaseIndexOffset LdBasePtr; for (unsigned i = 0; i < NumConsecutiveStores; ++i) { StoreSDNode *St = cast(StoreNodes[i].MemNode); SDValue Val = peekThroughBitcasts(St->getValue()); LoadSDNode *Ld = cast(Val); BaseIndexOffset LdPtr = BaseIndexOffset::match(Ld, DAG); // If this is not the first ptr that we check. int64_t LdOffset = 0; if (LdBasePtr.getBase().getNode()) { // The base ptr must be the same. if (!LdBasePtr.equalBaseIndex(LdPtr, DAG, LdOffset)) break; } else { // Check that all other base pointers are the same as this one. LdBasePtr = LdPtr; } // We found a potential memory operand to merge. LoadNodes.push_back(MemOpLink(Ld, LdOffset)); } while (NumConsecutiveStores >= 2 && LoadNodes.size() >= 2) { // If we have load/store pair instructions and we only have two values, // don't bother merging. unsigned RequiredAlignment; if (LoadNodes.size() == 2 && TLI.hasPairedLoad(MemVT, RequiredAlignment) && StoreNodes[0].MemNode->getAlignment() >= RequiredAlignment) { StoreNodes.erase(StoreNodes.begin(), StoreNodes.begin() + 2); LoadNodes.erase(LoadNodes.begin(), LoadNodes.begin() + 2); break; } LSBaseSDNode *FirstInChain = StoreNodes[0].MemNode; unsigned FirstStoreAS = FirstInChain->getAddressSpace(); unsigned FirstStoreAlign = FirstInChain->getAlignment(); LoadSDNode *FirstLoad = cast(LoadNodes[0].MemNode); unsigned FirstLoadAS = FirstLoad->getAddressSpace(); unsigned FirstLoadAlign = FirstLoad->getAlignment(); // Scan the memory operations on the chain and find the first // non-consecutive load memory address. These variables hold the index in // the store node array. unsigned LastConsecutiveLoad = 1; // This variable refers to the size and not index in the array. unsigned LastLegalVectorType = 1; unsigned LastLegalIntegerType = 1; bool isDereferenceable = true; bool DoIntegerTruncate = false; StartAddress = LoadNodes[0].OffsetFromBase; SDValue FirstChain = FirstLoad->getChain(); for (unsigned i = 1; i < LoadNodes.size(); ++i) { // All loads must share the same chain. if (LoadNodes[i].MemNode->getChain() != FirstChain) break; int64_t CurrAddress = LoadNodes[i].OffsetFromBase; if (CurrAddress - StartAddress != (ElementSizeBytes * i)) break; LastConsecutiveLoad = i; if (isDereferenceable && !LoadNodes[i].MemNode->isDereferenceable()) isDereferenceable = false; // Find a legal type for the vector store. unsigned Elts = (i + 1) * NumMemElts; EVT StoreTy = EVT::getVectorVT(Context, MemVT.getScalarType(), Elts); // Break early when size is too large to be legal. if (StoreTy.getSizeInBits() > MaximumLegalStoreInBits) break; bool IsFastSt, IsFastLd; if (TLI.isTypeLegal(StoreTy) && TLI.canMergeStoresTo(FirstStoreAS, StoreTy, DAG) && TLI.allowsMemoryAccess(Context, DL, StoreTy, FirstStoreAS, FirstStoreAlign, &IsFastSt) && IsFastSt && TLI.allowsMemoryAccess(Context, DL, StoreTy, FirstLoadAS, FirstLoadAlign, &IsFastLd) && IsFastLd) { LastLegalVectorType = i + 1; } // Find a legal type for the integer store. unsigned SizeInBits = (i + 1) * ElementSizeBytes * 8; StoreTy = EVT::getIntegerVT(Context, SizeInBits); if (TLI.isTypeLegal(StoreTy) && TLI.canMergeStoresTo(FirstStoreAS, StoreTy, DAG) && TLI.allowsMemoryAccess(Context, DL, StoreTy, FirstStoreAS, FirstStoreAlign, &IsFastSt) && IsFastSt && TLI.allowsMemoryAccess(Context, DL, StoreTy, FirstLoadAS, FirstLoadAlign, &IsFastLd) && IsFastLd) { LastLegalIntegerType = i + 1; DoIntegerTruncate = false; // Or check whether a truncstore and extload is legal. } else if (TLI.getTypeAction(Context, StoreTy) == TargetLowering::TypePromoteInteger) { EVT LegalizedStoredValTy = TLI.getTypeToTransformTo(Context, StoreTy); if (TLI.isTruncStoreLegal(LegalizedStoredValTy, StoreTy) && TLI.canMergeStoresTo(FirstStoreAS, LegalizedStoredValTy, DAG) && TLI.isLoadExtLegal(ISD::ZEXTLOAD, LegalizedStoredValTy, StoreTy) && TLI.isLoadExtLegal(ISD::SEXTLOAD, LegalizedStoredValTy, StoreTy) && TLI.isLoadExtLegal(ISD::EXTLOAD, LegalizedStoredValTy, StoreTy) && TLI.allowsMemoryAccess(Context, DL, StoreTy, FirstStoreAS, FirstStoreAlign, &IsFastSt) && IsFastSt && TLI.allowsMemoryAccess(Context, DL, StoreTy, FirstLoadAS, FirstLoadAlign, &IsFastLd) && IsFastLd) { LastLegalIntegerType = i + 1; DoIntegerTruncate = true; } } } // Only use vector types if the vector type is larger than the integer // type. If they are the same, use integers. bool UseVectorTy = LastLegalVectorType > LastLegalIntegerType && !NoVectors; unsigned LastLegalType = std::max(LastLegalVectorType, LastLegalIntegerType); // We add +1 here because the LastXXX variables refer to location while // the NumElem refers to array/index size. unsigned NumElem = std::min(NumConsecutiveStores, LastConsecutiveLoad + 1); NumElem = std::min(LastLegalType, NumElem); if (NumElem < 2) { // We know that candidate stores are in order and of correct // shape. While there is no mergeable sequence from the // beginning one may start later in the sequence. The only // reason a merge of size N could have failed where another of // the same size would not have is if the alignment or either // the load or store has improved. Drop as many candidates as we // can here. unsigned NumSkip = 1; while ((NumSkip < LoadNodes.size()) && (LoadNodes[NumSkip].MemNode->getAlignment() <= FirstLoadAlign) && (StoreNodes[NumSkip].MemNode->getAlignment() <= FirstStoreAlign)) NumSkip++; StoreNodes.erase(StoreNodes.begin(), StoreNodes.begin() + NumSkip); LoadNodes.erase(LoadNodes.begin(), LoadNodes.begin() + NumSkip); NumConsecutiveStores -= NumSkip; continue; } // Check that we can merge these candidates without causing a cycle. if (!checkMergeStoreCandidatesForDependencies(StoreNodes, NumElem, RootNode)) { StoreNodes.erase(StoreNodes.begin(), StoreNodes.begin() + NumElem); LoadNodes.erase(LoadNodes.begin(), LoadNodes.begin() + NumElem); NumConsecutiveStores -= NumElem; continue; } // Find if it is better to use vectors or integers to load and store // to memory. EVT JointMemOpVT; if (UseVectorTy) { // Find a legal type for the vector store. unsigned Elts = NumElem * NumMemElts; JointMemOpVT = EVT::getVectorVT(Context, MemVT.getScalarType(), Elts); } else { unsigned SizeInBits = NumElem * ElementSizeBytes * 8; JointMemOpVT = EVT::getIntegerVT(Context, SizeInBits); } SDLoc LoadDL(LoadNodes[0].MemNode); SDLoc StoreDL(StoreNodes[0].MemNode); // The merged loads are required to have the same incoming chain, so // using the first's chain is acceptable. SDValue NewStoreChain = getMergeStoreChains(StoreNodes, NumElem); AddToWorklist(NewStoreChain.getNode()); MachineMemOperand::Flags MMOFlags = isDereferenceable ? MachineMemOperand::MODereferenceable : MachineMemOperand::MONone; SDValue NewLoad, NewStore; if (UseVectorTy || !DoIntegerTruncate) { NewLoad = DAG.getLoad(JointMemOpVT, LoadDL, FirstLoad->getChain(), FirstLoad->getBasePtr(), FirstLoad->getPointerInfo(), FirstLoadAlign, MMOFlags); NewStore = DAG.getStore( NewStoreChain, StoreDL, NewLoad, FirstInChain->getBasePtr(), FirstInChain->getPointerInfo(), FirstStoreAlign); } else { // This must be the truncstore/extload case EVT ExtendedTy = TLI.getTypeToTransformTo(*DAG.getContext(), JointMemOpVT); NewLoad = DAG.getExtLoad(ISD::EXTLOAD, LoadDL, ExtendedTy, FirstLoad->getChain(), FirstLoad->getBasePtr(), FirstLoad->getPointerInfo(), JointMemOpVT, FirstLoadAlign, MMOFlags); NewStore = DAG.getTruncStore(NewStoreChain, StoreDL, NewLoad, FirstInChain->getBasePtr(), FirstInChain->getPointerInfo(), JointMemOpVT, FirstInChain->getAlignment(), FirstInChain->getMemOperand()->getFlags()); } // Transfer chain users from old loads to the new load. for (unsigned i = 0; i < NumElem; ++i) { LoadSDNode *Ld = cast(LoadNodes[i].MemNode); DAG.ReplaceAllUsesOfValueWith(SDValue(Ld, 1), SDValue(NewLoad.getNode(), 1)); } // Replace the all stores with the new store. Recursively remove // corresponding value if its no longer used. for (unsigned i = 0; i < NumElem; ++i) { SDValue Val = StoreNodes[i].MemNode->getOperand(1); CombineTo(StoreNodes[i].MemNode, NewStore); if (Val.getNode()->use_empty()) recursivelyDeleteUnusedNodes(Val.getNode()); } RV = true; StoreNodes.erase(StoreNodes.begin(), StoreNodes.begin() + NumElem); LoadNodes.erase(LoadNodes.begin(), LoadNodes.begin() + NumElem); NumConsecutiveStores -= NumElem; } } return RV; } SDValue DAGCombiner::replaceStoreChain(StoreSDNode *ST, SDValue BetterChain) { SDLoc SL(ST); SDValue ReplStore; // Replace the chain to avoid dependency. if (ST->isTruncatingStore()) { ReplStore = DAG.getTruncStore(BetterChain, SL, ST->getValue(), ST->getBasePtr(), ST->getMemoryVT(), ST->getMemOperand()); } else { ReplStore = DAG.getStore(BetterChain, SL, ST->getValue(), ST->getBasePtr(), ST->getMemOperand()); } // Create token to keep both nodes around. SDValue Token = DAG.getNode(ISD::TokenFactor, SL, MVT::Other, ST->getChain(), ReplStore); // Make sure the new and old chains are cleaned up. AddToWorklist(Token.getNode()); // Don't add users to work list. return CombineTo(ST, Token, false); } SDValue DAGCombiner::replaceStoreOfFPConstant(StoreSDNode *ST) { SDValue Value = ST->getValue(); if (Value.getOpcode() == ISD::TargetConstantFP) return SDValue(); SDLoc DL(ST); SDValue Chain = ST->getChain(); SDValue Ptr = ST->getBasePtr(); const ConstantFPSDNode *CFP = cast(Value); // NOTE: If the original store is volatile, this transform must not increase // the number of stores. For example, on x86-32 an f64 can be stored in one // processor operation but an i64 (which is not legal) requires two. So the // transform should not be done in this case. SDValue Tmp; switch (CFP->getSimpleValueType(0).SimpleTy) { default: llvm_unreachable("Unknown FP type"); case MVT::f16: // We don't do this for these yet. case MVT::f80: case MVT::f128: case MVT::ppcf128: return SDValue(); case MVT::f32: if ((isTypeLegal(MVT::i32) && !LegalOperations && !ST->isVolatile()) || TLI.isOperationLegalOrCustom(ISD::STORE, MVT::i32)) { ; Tmp = DAG.getConstant((uint32_t)CFP->getValueAPF(). bitcastToAPInt().getZExtValue(), SDLoc(CFP), MVT::i32); return DAG.getStore(Chain, DL, Tmp, Ptr, ST->getMemOperand()); } return SDValue(); case MVT::f64: if ((TLI.isTypeLegal(MVT::i64) && !LegalOperations && !ST->isVolatile()) || TLI.isOperationLegalOrCustom(ISD::STORE, MVT::i64)) { ; Tmp = DAG.getConstant(CFP->getValueAPF().bitcastToAPInt(). getZExtValue(), SDLoc(CFP), MVT::i64); return DAG.getStore(Chain, DL, Tmp, Ptr, ST->getMemOperand()); } if (!ST->isVolatile() && TLI.isOperationLegalOrCustom(ISD::STORE, MVT::i32)) { // Many FP stores are not made apparent until after legalize, e.g. for // argument passing. Since this is so common, custom legalize the // 64-bit integer store into two 32-bit stores. uint64_t Val = CFP->getValueAPF().bitcastToAPInt().getZExtValue(); SDValue Lo = DAG.getConstant(Val & 0xFFFFFFFF, SDLoc(CFP), MVT::i32); SDValue Hi = DAG.getConstant(Val >> 32, SDLoc(CFP), MVT::i32); if (DAG.getDataLayout().isBigEndian()) std::swap(Lo, Hi); unsigned Alignment = ST->getAlignment(); MachineMemOperand::Flags MMOFlags = ST->getMemOperand()->getFlags(); AAMDNodes AAInfo = ST->getAAInfo(); SDValue St0 = DAG.getStore(Chain, DL, Lo, Ptr, ST->getPointerInfo(), ST->getAlignment(), MMOFlags, AAInfo); Ptr = DAG.getNode(ISD::ADD, DL, Ptr.getValueType(), Ptr, DAG.getConstant(4, DL, Ptr.getValueType())); Alignment = MinAlign(Alignment, 4U); SDValue St1 = DAG.getStore(Chain, DL, Hi, Ptr, ST->getPointerInfo().getWithOffset(4), Alignment, MMOFlags, AAInfo); return DAG.getNode(ISD::TokenFactor, DL, MVT::Other, St0, St1); } return SDValue(); } } SDValue DAGCombiner::visitSTORE(SDNode *N) { StoreSDNode *ST = cast(N); SDValue Chain = ST->getChain(); SDValue Value = ST->getValue(); SDValue Ptr = ST->getBasePtr(); // If this is a store of a bit convert, store the input value if the // resultant store does not need a higher alignment than the original. if (Value.getOpcode() == ISD::BITCAST && !ST->isTruncatingStore() && ST->isUnindexed()) { EVT SVT = Value.getOperand(0).getValueType(); // If the store is volatile, we only want to change the store type if the // resulting store is legal. Otherwise we might increase the number of // memory accesses. We don't care if the original type was legal or not // as we assume software couldn't rely on the number of accesses of an // illegal type. if (((!LegalOperations && !ST->isVolatile()) || TLI.isOperationLegal(ISD::STORE, SVT)) && TLI.isStoreBitCastBeneficial(Value.getValueType(), SVT)) { unsigned OrigAlign = ST->getAlignment(); bool Fast = false; if (TLI.allowsMemoryAccess(*DAG.getContext(), DAG.getDataLayout(), SVT, ST->getAddressSpace(), OrigAlign, &Fast) && Fast) { return DAG.getStore(Chain, SDLoc(N), Value.getOperand(0), Ptr, ST->getPointerInfo(), OrigAlign, ST->getMemOperand()->getFlags(), ST->getAAInfo()); } } } // Turn 'store undef, Ptr' -> nothing. if (Value.isUndef() && ST->isUnindexed()) return Chain; // Try to infer better alignment information than the store already has. if (OptLevel != CodeGenOpt::None && ST->isUnindexed()) { if (unsigned Align = DAG.InferPtrAlignment(Ptr)) { if (Align > ST->getAlignment() && ST->getSrcValueOffset() % Align == 0) { SDValue NewStore = DAG.getTruncStore(Chain, SDLoc(N), Value, Ptr, ST->getPointerInfo(), ST->getMemoryVT(), Align, ST->getMemOperand()->getFlags(), ST->getAAInfo()); // NewStore will always be N as we are only refining the alignment assert(NewStore.getNode() == N); (void)NewStore; } } } // Try transforming a pair floating point load / store ops to integer // load / store ops. if (SDValue NewST = TransformFPLoadStorePair(N)) return NewST; if (ST->isUnindexed()) { // Walk up chain skipping non-aliasing memory nodes, on this store and any // adjacent stores. if (findBetterNeighborChains(ST)) { // replaceStoreChain uses CombineTo, which handled all of the worklist // manipulation. Return the original node to not do anything else. return SDValue(ST, 0); } Chain = ST->getChain(); } // FIXME: is there such a thing as a truncating indexed store? if (ST->isTruncatingStore() && ST->isUnindexed() && Value.getValueType().isInteger() && (!isa(Value) || !cast(Value)->isOpaque())) { // See if we can simplify the input to this truncstore with knowledge that // only the low bits are being used. For example: // "truncstore (or (shl x, 8), y), i8" -> "truncstore y, i8" SDValue Shorter = DAG.GetDemandedBits( Value, APInt::getLowBitsSet(Value.getScalarValueSizeInBits(), ST->getMemoryVT().getScalarSizeInBits())); AddToWorklist(Value.getNode()); if (Shorter.getNode()) return DAG.getTruncStore(Chain, SDLoc(N), Shorter, Ptr, ST->getMemoryVT(), ST->getMemOperand()); // Otherwise, see if we can simplify the operation with // SimplifyDemandedBits, which only works if the value has a single use. if (SimplifyDemandedBits( Value, APInt::getLowBitsSet(Value.getScalarValueSizeInBits(), ST->getMemoryVT().getScalarSizeInBits()))) { // Re-visit the store if anything changed and the store hasn't been merged // with another node (N is deleted) SimplifyDemandedBits will add Value's // node back to the worklist if necessary, but we also need to re-visit // the Store node itself. if (N->getOpcode() != ISD::DELETED_NODE) AddToWorklist(N); return SDValue(N, 0); } } // If this is a load followed by a store to the same location, then the store // is dead/noop. if (LoadSDNode *Ld = dyn_cast(Value)) { if (Ld->getBasePtr() == Ptr && ST->getMemoryVT() == Ld->getMemoryVT() && ST->isUnindexed() && !ST->isVolatile() && // There can't be any side effects between the load and store, such as // a call or store. Chain.reachesChainWithoutSideEffects(SDValue(Ld, 1))) { // The store is dead, remove it. return Chain; } } if (StoreSDNode *ST1 = dyn_cast(Chain)) { if (ST->isUnindexed() && !ST->isVolatile() && ST1->isUnindexed() && !ST1->isVolatile() && ST1->getBasePtr() == Ptr && ST->getMemoryVT() == ST1->getMemoryVT()) { // If this is a store followed by a store with the same value to the same // location, then the store is dead/noop. if (ST1->getValue() == Value) { // The store is dead, remove it. return Chain; } // If this is a store who's preceeding store to the same location // and no one other node is chained to that store we can effectively // drop the store. Do not remove stores to undef as they may be used as // data sinks. if (OptLevel != CodeGenOpt::None && ST1->hasOneUse() && !ST1->getBasePtr().isUndef()) { // ST1 is fully overwritten and can be elided. Combine with it's chain // value. CombineTo(ST1, ST1->getChain()); return SDValue(); } } } // If this is an FP_ROUND or TRUNC followed by a store, fold this into a // truncating store. We can do this even if this is already a truncstore. if ((Value.getOpcode() == ISD::FP_ROUND || Value.getOpcode() == ISD::TRUNCATE) && Value.getNode()->hasOneUse() && ST->isUnindexed() && TLI.isTruncStoreLegal(Value.getOperand(0).getValueType(), ST->getMemoryVT())) { return DAG.getTruncStore(Chain, SDLoc(N), Value.getOperand(0), Ptr, ST->getMemoryVT(), ST->getMemOperand()); } // Always perform this optimization before types are legal. If the target // prefers, also try this after legalization to catch stores that were created // by intrinsics or other nodes. if (!LegalTypes || (TLI.mergeStoresAfterLegalization())) { while (true) { // There can be multiple store sequences on the same chain. // Keep trying to merge store sequences until we are unable to do so // or until we merge the last store on the chain. bool Changed = MergeConsecutiveStores(ST); if (!Changed) break; // Return N as merge only uses CombineTo and no worklist clean // up is necessary. if (N->getOpcode() == ISD::DELETED_NODE || !isa(N)) return SDValue(N, 0); } } // Try transforming N to an indexed store. if (CombineToPreIndexedLoadStore(N) || CombineToPostIndexedLoadStore(N)) return SDValue(N, 0); // Turn 'store float 1.0, Ptr' -> 'store int 0x12345678, Ptr' // // Make sure to do this only after attempting to merge stores in order to // avoid changing the types of some subset of stores due to visit order, // preventing their merging. if (isa(ST->getValue())) { if (SDValue NewSt = replaceStoreOfFPConstant(ST)) return NewSt; } if (SDValue NewSt = splitMergedValStore(ST)) return NewSt; return ReduceLoadOpStoreWidth(N); } /// For the instruction sequence of store below, F and I values /// are bundled together as an i64 value before being stored into memory. /// Sometimes it is more efficent to generate separate stores for F and I, /// which can remove the bitwise instructions or sink them to colder places. /// /// (store (or (zext (bitcast F to i32) to i64), /// (shl (zext I to i64), 32)), addr) --> /// (store F, addr) and (store I, addr+4) /// /// Similarly, splitting for other merged store can also be beneficial, like: /// For pair of {i32, i32}, i64 store --> two i32 stores. /// For pair of {i32, i16}, i64 store --> two i32 stores. /// For pair of {i16, i16}, i32 store --> two i16 stores. /// For pair of {i16, i8}, i32 store --> two i16 stores. /// For pair of {i8, i8}, i16 store --> two i8 stores. /// /// We allow each target to determine specifically which kind of splitting is /// supported. /// /// The store patterns are commonly seen from the simple code snippet below /// if only std::make_pair(...) is sroa transformed before inlined into hoo. /// void goo(const std::pair &); /// hoo() { /// ... /// goo(std::make_pair(tmp, ftmp)); /// ... /// } /// SDValue DAGCombiner::splitMergedValStore(StoreSDNode *ST) { if (OptLevel == CodeGenOpt::None) return SDValue(); SDValue Val = ST->getValue(); SDLoc DL(ST); // Match OR operand. if (!Val.getValueType().isScalarInteger() || Val.getOpcode() != ISD::OR) return SDValue(); // Match SHL operand and get Lower and Higher parts of Val. SDValue Op1 = Val.getOperand(0); SDValue Op2 = Val.getOperand(1); SDValue Lo, Hi; if (Op1.getOpcode() != ISD::SHL) { std::swap(Op1, Op2); if (Op1.getOpcode() != ISD::SHL) return SDValue(); } Lo = Op2; Hi = Op1.getOperand(0); if (!Op1.hasOneUse()) return SDValue(); // Match shift amount to HalfValBitSize. unsigned HalfValBitSize = Val.getValueSizeInBits() / 2; ConstantSDNode *ShAmt = dyn_cast(Op1.getOperand(1)); if (!ShAmt || ShAmt->getAPIntValue() != HalfValBitSize) return SDValue(); // Lo and Hi are zero-extended from int with size less equal than 32 // to i64. if (Lo.getOpcode() != ISD::ZERO_EXTEND || !Lo.hasOneUse() || !Lo.getOperand(0).getValueType().isScalarInteger() || Lo.getOperand(0).getValueSizeInBits() > HalfValBitSize || Hi.getOpcode() != ISD::ZERO_EXTEND || !Hi.hasOneUse() || !Hi.getOperand(0).getValueType().isScalarInteger() || Hi.getOperand(0).getValueSizeInBits() > HalfValBitSize) return SDValue(); // Use the EVT of low and high parts before bitcast as the input // of target query. EVT LowTy = (Lo.getOperand(0).getOpcode() == ISD::BITCAST) ? Lo.getOperand(0).getValueType() : Lo.getValueType(); EVT HighTy = (Hi.getOperand(0).getOpcode() == ISD::BITCAST) ? Hi.getOperand(0).getValueType() : Hi.getValueType(); if (!TLI.isMultiStoresCheaperThanBitsMerge(LowTy, HighTy)) return SDValue(); // Start to split store. unsigned Alignment = ST->getAlignment(); MachineMemOperand::Flags MMOFlags = ST->getMemOperand()->getFlags(); AAMDNodes AAInfo = ST->getAAInfo(); // Change the sizes of Lo and Hi's value types to HalfValBitSize. EVT VT = EVT::getIntegerVT(*DAG.getContext(), HalfValBitSize); Lo = DAG.getNode(ISD::ZERO_EXTEND, DL, VT, Lo.getOperand(0)); Hi = DAG.getNode(ISD::ZERO_EXTEND, DL, VT, Hi.getOperand(0)); SDValue Chain = ST->getChain(); SDValue Ptr = ST->getBasePtr(); // Lower value store. SDValue St0 = DAG.getStore(Chain, DL, Lo, Ptr, ST->getPointerInfo(), ST->getAlignment(), MMOFlags, AAInfo); Ptr = DAG.getNode(ISD::ADD, DL, Ptr.getValueType(), Ptr, DAG.getConstant(HalfValBitSize / 8, DL, Ptr.getValueType())); // Higher value store. SDValue St1 = DAG.getStore(St0, DL, Hi, Ptr, ST->getPointerInfo().getWithOffset(HalfValBitSize / 8), Alignment / 2, MMOFlags, AAInfo); return St1; } /// Convert a disguised subvector insertion into a shuffle: /// insert_vector_elt V, (bitcast X from vector type), IdxC --> /// bitcast(shuffle (bitcast V), (extended X), Mask) /// Note: We do not use an insert_subvector node because that requires a legal /// subvector type. SDValue DAGCombiner::combineInsertEltToShuffle(SDNode *N, unsigned InsIndex) { SDValue InsertVal = N->getOperand(1); if (InsertVal.getOpcode() != ISD::BITCAST || !InsertVal.hasOneUse() || !InsertVal.getOperand(0).getValueType().isVector()) return SDValue(); SDValue SubVec = InsertVal.getOperand(0); SDValue DestVec = N->getOperand(0); EVT SubVecVT = SubVec.getValueType(); EVT VT = DestVec.getValueType(); unsigned NumSrcElts = SubVecVT.getVectorNumElements(); unsigned ExtendRatio = VT.getSizeInBits() / SubVecVT.getSizeInBits(); unsigned NumMaskVals = ExtendRatio * NumSrcElts; // Step 1: Create a shuffle mask that implements this insert operation. The // vector that we are inserting into will be operand 0 of the shuffle, so // those elements are just 'i'. The inserted subvector is in the first // positions of operand 1 of the shuffle. Example: // insert v4i32 V, (v2i16 X), 2 --> shuffle v8i16 V', X', {0,1,2,3,8,9,6,7} SmallVector Mask(NumMaskVals); for (unsigned i = 0; i != NumMaskVals; ++i) { if (i / NumSrcElts == InsIndex) Mask[i] = (i % NumSrcElts) + NumMaskVals; else Mask[i] = i; } // Bail out if the target can not handle the shuffle we want to create. EVT SubVecEltVT = SubVecVT.getVectorElementType(); EVT ShufVT = EVT::getVectorVT(*DAG.getContext(), SubVecEltVT, NumMaskVals); if (!TLI.isShuffleMaskLegal(Mask, ShufVT)) return SDValue(); // Step 2: Create a wide vector from the inserted source vector by appending // undefined elements. This is the same size as our destination vector. SDLoc DL(N); SmallVector ConcatOps(ExtendRatio, DAG.getUNDEF(SubVecVT)); ConcatOps[0] = SubVec; SDValue PaddedSubV = DAG.getNode(ISD::CONCAT_VECTORS, DL, ShufVT, ConcatOps); // Step 3: Shuffle in the padded subvector. SDValue DestVecBC = DAG.getBitcast(ShufVT, DestVec); SDValue Shuf = DAG.getVectorShuffle(ShufVT, DL, DestVecBC, PaddedSubV, Mask); AddToWorklist(PaddedSubV.getNode()); AddToWorklist(DestVecBC.getNode()); AddToWorklist(Shuf.getNode()); return DAG.getBitcast(VT, Shuf); } SDValue DAGCombiner::visitINSERT_VECTOR_ELT(SDNode *N) { SDValue InVec = N->getOperand(0); SDValue InVal = N->getOperand(1); SDValue EltNo = N->getOperand(2); SDLoc DL(N); // If the inserted element is an UNDEF, just use the input vector. if (InVal.isUndef()) return InVec; EVT VT = InVec.getValueType(); unsigned NumElts = VT.getVectorNumElements(); // Remove redundant insertions: // (insert_vector_elt x (extract_vector_elt x idx) idx) -> x if (InVal.getOpcode() == ISD::EXTRACT_VECTOR_ELT && InVec == InVal.getOperand(0) && EltNo == InVal.getOperand(1)) return InVec; auto *IndexC = dyn_cast(EltNo); if (!IndexC) { // If this is variable insert to undef vector, it might be better to splat: // inselt undef, InVal, EltNo --> build_vector < InVal, InVal, ... > if (InVec.isUndef() && TLI.shouldSplatInsEltVarIndex(VT)) { SmallVector Ops(NumElts, InVal); return DAG.getBuildVector(VT, DL, Ops); } return SDValue(); } // We must know which element is being inserted for folds below here. unsigned Elt = IndexC->getZExtValue(); if (SDValue Shuf = combineInsertEltToShuffle(N, Elt)) return Shuf; // Canonicalize insert_vector_elt dag nodes. // Example: // (insert_vector_elt (insert_vector_elt A, Idx0), Idx1) // -> (insert_vector_elt (insert_vector_elt A, Idx1), Idx0) // // Do this only if the child insert_vector node has one use; also // do this only if indices are both constants and Idx1 < Idx0. if (InVec.getOpcode() == ISD::INSERT_VECTOR_ELT && InVec.hasOneUse() && isa(InVec.getOperand(2))) { unsigned OtherElt = InVec.getConstantOperandVal(2); if (Elt < OtherElt) { // Swap nodes. SDValue NewOp = DAG.getNode(ISD::INSERT_VECTOR_ELT, DL, VT, InVec.getOperand(0), InVal, EltNo); AddToWorklist(NewOp.getNode()); return DAG.getNode(ISD::INSERT_VECTOR_ELT, SDLoc(InVec.getNode()), VT, NewOp, InVec.getOperand(1), InVec.getOperand(2)); } } // If we can't generate a legal BUILD_VECTOR, exit if (LegalOperations && !TLI.isOperationLegal(ISD::BUILD_VECTOR, VT)) return SDValue(); // Check that the operand is a BUILD_VECTOR (or UNDEF, which can essentially // be converted to a BUILD_VECTOR). Fill in the Ops vector with the // vector elements. SmallVector Ops; // Do not combine these two vectors if the output vector will not replace // the input vector. if (InVec.getOpcode() == ISD::BUILD_VECTOR && InVec.hasOneUse()) { Ops.append(InVec.getNode()->op_begin(), InVec.getNode()->op_end()); } else if (InVec.isUndef()) { Ops.append(NumElts, DAG.getUNDEF(InVal.getValueType())); } else { return SDValue(); } assert(Ops.size() == NumElts && "Unexpected vector size"); // Insert the element if (Elt < Ops.size()) { // All the operands of BUILD_VECTOR must have the same type; // we enforce that here. EVT OpVT = Ops[0].getValueType(); Ops[Elt] = OpVT.isInteger() ? DAG.getAnyExtOrTrunc(InVal, DL, OpVT) : InVal; } // Return the new vector return DAG.getBuildVector(VT, DL, Ops); } SDValue DAGCombiner::scalarizeExtractedVectorLoad(SDNode *EVE, EVT InVecVT, SDValue EltNo, LoadSDNode *OriginalLoad) { assert(!OriginalLoad->isVolatile()); EVT ResultVT = EVE->getValueType(0); EVT VecEltVT = InVecVT.getVectorElementType(); unsigned Align = OriginalLoad->getAlignment(); unsigned NewAlign = DAG.getDataLayout().getABITypeAlignment( VecEltVT.getTypeForEVT(*DAG.getContext())); if (NewAlign > Align || !TLI.isOperationLegalOrCustom(ISD::LOAD, VecEltVT)) return SDValue(); ISD::LoadExtType ExtTy = ResultVT.bitsGT(VecEltVT) ? ISD::NON_EXTLOAD : ISD::EXTLOAD; if (!TLI.shouldReduceLoadWidth(OriginalLoad, ExtTy, VecEltVT)) return SDValue(); Align = NewAlign; SDValue NewPtr = OriginalLoad->getBasePtr(); SDValue Offset; EVT PtrType = NewPtr.getValueType(); MachinePointerInfo MPI; SDLoc DL(EVE); if (auto *ConstEltNo = dyn_cast(EltNo)) { int Elt = ConstEltNo->getZExtValue(); unsigned PtrOff = VecEltVT.getSizeInBits() * Elt / 8; Offset = DAG.getConstant(PtrOff, DL, PtrType); MPI = OriginalLoad->getPointerInfo().getWithOffset(PtrOff); } else { Offset = DAG.getZExtOrTrunc(EltNo, DL, PtrType); Offset = DAG.getNode( ISD::MUL, DL, PtrType, Offset, DAG.getConstant(VecEltVT.getStoreSize(), DL, PtrType)); MPI = OriginalLoad->getPointerInfo(); } NewPtr = DAG.getNode(ISD::ADD, DL, PtrType, NewPtr, Offset); // The replacement we need to do here is a little tricky: we need to // replace an extractelement of a load with a load. // Use ReplaceAllUsesOfValuesWith to do the replacement. // Note that this replacement assumes that the extractvalue is the only // use of the load; that's okay because we don't want to perform this // transformation in other cases anyway. SDValue Load; SDValue Chain; if (ResultVT.bitsGT(VecEltVT)) { // If the result type of vextract is wider than the load, then issue an // extending load instead. ISD::LoadExtType ExtType = TLI.isLoadExtLegal(ISD::ZEXTLOAD, ResultVT, VecEltVT) ? ISD::ZEXTLOAD : ISD::EXTLOAD; Load = DAG.getExtLoad(ExtType, SDLoc(EVE), ResultVT, OriginalLoad->getChain(), NewPtr, MPI, VecEltVT, Align, OriginalLoad->getMemOperand()->getFlags(), OriginalLoad->getAAInfo()); Chain = Load.getValue(1); } else { Load = DAG.getLoad(VecEltVT, SDLoc(EVE), OriginalLoad->getChain(), NewPtr, MPI, Align, OriginalLoad->getMemOperand()->getFlags(), OriginalLoad->getAAInfo()); Chain = Load.getValue(1); if (ResultVT.bitsLT(VecEltVT)) Load = DAG.getNode(ISD::TRUNCATE, SDLoc(EVE), ResultVT, Load); else Load = DAG.getBitcast(ResultVT, Load); } WorklistRemover DeadNodes(*this); SDValue From[] = { SDValue(EVE, 0), SDValue(OriginalLoad, 1) }; SDValue To[] = { Load, Chain }; DAG.ReplaceAllUsesOfValuesWith(From, To, 2); // Since we're explicitly calling ReplaceAllUses, add the new node to the // worklist explicitly as well. AddToWorklist(Load.getNode()); AddUsersToWorklist(Load.getNode()); // Add users too // Make sure to revisit this node to clean it up; it will usually be dead. AddToWorklist(EVE); ++OpsNarrowed; return SDValue(EVE, 0); } /// Transform a vector binary operation into a scalar binary operation by moving /// the math/logic after an extract element of a vector. static SDValue scalarizeExtractedBinop(SDNode *ExtElt, SelectionDAG &DAG, bool LegalOperations) { SDValue Vec = ExtElt->getOperand(0); SDValue Index = ExtElt->getOperand(1); auto *IndexC = dyn_cast(Index); if (!IndexC || !ISD::isBinaryOp(Vec.getNode()) || !Vec.hasOneUse()) return SDValue(); // Targets may want to avoid this to prevent an expensive register transfer. const TargetLowering &TLI = DAG.getTargetLoweringInfo(); if (!TLI.shouldScalarizeBinop(Vec)) return SDValue(); // Extracting an element of a vector constant is constant-folded, so this // transform is just replacing a vector op with a scalar op while moving the // extract. SDValue Op0 = Vec.getOperand(0); SDValue Op1 = Vec.getOperand(1); if (isAnyConstantBuildVector(Op0, true) || isAnyConstantBuildVector(Op1, true)) { // extractelt (binop X, C), IndexC --> binop (extractelt X, IndexC), C' // extractelt (binop C, X), IndexC --> binop C', (extractelt X, IndexC) SDLoc DL(ExtElt); EVT VT = ExtElt->getValueType(0); SDValue Ext0 = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, VT, Op0, Index); SDValue Ext1 = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, VT, Op1, Index); return DAG.getNode(Vec.getOpcode(), DL, VT, Ext0, Ext1); } return SDValue(); } SDValue DAGCombiner::visitEXTRACT_VECTOR_ELT(SDNode *N) { SDValue VecOp = N->getOperand(0); SDValue Index = N->getOperand(1); EVT ScalarVT = N->getValueType(0); EVT VecVT = VecOp.getValueType(); if (VecOp.isUndef()) return DAG.getUNDEF(ScalarVT); // extract_vector_elt (insert_vector_elt vec, val, idx), idx) -> val // // This only really matters if the index is non-constant since other combines // on the constant elements already work. SDLoc DL(N); if (VecOp.getOpcode() == ISD::INSERT_VECTOR_ELT && Index == VecOp.getOperand(2)) { SDValue Elt = VecOp.getOperand(1); return VecVT.isInteger() ? DAG.getAnyExtOrTrunc(Elt, DL, ScalarVT) : Elt; } // (vextract (scalar_to_vector val, 0) -> val if (VecOp.getOpcode() == ISD::SCALAR_TO_VECTOR) { // Check if the result type doesn't match the inserted element type. A // SCALAR_TO_VECTOR may truncate the inserted element and the // EXTRACT_VECTOR_ELT may widen the extracted vector. SDValue InOp = VecOp.getOperand(0); if (InOp.getValueType() != ScalarVT) { assert(InOp.getValueType().isInteger() && ScalarVT.isInteger()); return DAG.getSExtOrTrunc(InOp, DL, ScalarVT); } return InOp; } // extract_vector_elt of out-of-bounds element -> UNDEF auto *IndexC = dyn_cast(Index); unsigned NumElts = VecVT.getVectorNumElements(); if (IndexC && IndexC->getAPIntValue().uge(NumElts)) return DAG.getUNDEF(ScalarVT); // extract_vector_elt (build_vector x, y), 1 -> y if (IndexC && VecOp.getOpcode() == ISD::BUILD_VECTOR && TLI.isTypeLegal(VecVT) && (VecOp.hasOneUse() || TLI.aggressivelyPreferBuildVectorSources(VecVT))) { SDValue Elt = VecOp.getOperand(IndexC->getZExtValue()); EVT InEltVT = Elt.getValueType(); // Sometimes build_vector's scalar input types do not match result type. if (ScalarVT == InEltVT) return Elt; // TODO: It may be useful to truncate if free if the build_vector implicitly // converts. } // TODO: These transforms should not require the 'hasOneUse' restriction, but // there are regressions on multiple targets without it. We can end up with a // mess of scalar and vector code if we reduce only part of the DAG to scalar. if (IndexC && VecOp.getOpcode() == ISD::BITCAST && VecVT.isInteger() && VecOp.hasOneUse()) { // The vector index of the LSBs of the source depend on the endian-ness. bool IsLE = DAG.getDataLayout().isLittleEndian(); unsigned ExtractIndex = IndexC->getZExtValue(); // extract_elt (v2i32 (bitcast i64:x)), BCTruncElt -> i32 (trunc i64:x) unsigned BCTruncElt = IsLE ? 0 : NumElts - 1; SDValue BCSrc = VecOp.getOperand(0); if (ExtractIndex == BCTruncElt && BCSrc.getValueType().isScalarInteger()) return DAG.getNode(ISD::TRUNCATE, DL, ScalarVT, BCSrc); if (LegalTypes && BCSrc.getValueType().isInteger() && BCSrc.getOpcode() == ISD::SCALAR_TO_VECTOR) { // ext_elt (bitcast (scalar_to_vec i64 X to v2i64) to v4i32), TruncElt --> // trunc i64 X to i32 SDValue X = BCSrc.getOperand(0); assert(X.getValueType().isScalarInteger() && ScalarVT.isScalarInteger() && "Extract element and scalar to vector can't change element type " "from FP to integer."); unsigned XBitWidth = X.getValueSizeInBits(); unsigned VecEltBitWidth = VecVT.getScalarSizeInBits(); BCTruncElt = IsLE ? 0 : XBitWidth / VecEltBitWidth - 1; // An extract element return value type can be wider than its vector // operand element type. In that case, the high bits are undefined, so // it's possible that we may need to extend rather than truncate. if (ExtractIndex == BCTruncElt && XBitWidth > VecEltBitWidth) { assert(XBitWidth % VecEltBitWidth == 0 && "Scalar bitwidth must be a multiple of vector element bitwidth"); return DAG.getAnyExtOrTrunc(X, DL, ScalarVT); } } } if (SDValue BO = scalarizeExtractedBinop(N, DAG, LegalOperations)) return BO; // Transform: (EXTRACT_VECTOR_ELT( VECTOR_SHUFFLE )) -> EXTRACT_VECTOR_ELT. // We only perform this optimization before the op legalization phase because // we may introduce new vector instructions which are not backed by TD // patterns. For example on AVX, extracting elements from a wide vector // without using extract_subvector. However, if we can find an underlying // scalar value, then we can always use that. if (IndexC && VecOp.getOpcode() == ISD::VECTOR_SHUFFLE) { auto *Shuf = cast(VecOp); // Find the new index to extract from. int OrigElt = Shuf->getMaskElt(IndexC->getZExtValue()); // Extracting an undef index is undef. if (OrigElt == -1) return DAG.getUNDEF(ScalarVT); // Select the right vector half to extract from. SDValue SVInVec; if (OrigElt < (int)NumElts) { SVInVec = VecOp.getOperand(0); } else { SVInVec = VecOp.getOperand(1); OrigElt -= NumElts; } if (SVInVec.getOpcode() == ISD::BUILD_VECTOR) { SDValue InOp = SVInVec.getOperand(OrigElt); if (InOp.getValueType() != ScalarVT) { assert(InOp.getValueType().isInteger() && ScalarVT.isInteger()); InOp = DAG.getSExtOrTrunc(InOp, DL, ScalarVT); } return InOp; } // FIXME: We should handle recursing on other vector shuffles and // scalar_to_vector here as well. if (!LegalOperations || // FIXME: Should really be just isOperationLegalOrCustom. TLI.isOperationLegal(ISD::EXTRACT_VECTOR_ELT, VecVT) || TLI.isOperationExpand(ISD::VECTOR_SHUFFLE, VecVT)) { EVT IndexTy = TLI.getVectorIdxTy(DAG.getDataLayout()); return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, ScalarVT, SVInVec, DAG.getConstant(OrigElt, DL, IndexTy)); } } // If only EXTRACT_VECTOR_ELT nodes use the source vector we can // simplify it based on the (valid) extraction indices. if (llvm::all_of(VecOp->uses(), [&](SDNode *Use) { return Use->getOpcode() == ISD::EXTRACT_VECTOR_ELT && Use->getOperand(0) == VecOp && isa(Use->getOperand(1)); })) { APInt DemandedElts = APInt::getNullValue(NumElts); for (SDNode *Use : VecOp->uses()) { auto *CstElt = cast(Use->getOperand(1)); if (CstElt->getAPIntValue().ult(NumElts)) DemandedElts.setBit(CstElt->getZExtValue()); } if (SimplifyDemandedVectorElts(VecOp, DemandedElts, true)) { // We simplified the vector operand of this extract element. If this // extract is not dead, visit it again so it is folded properly. if (N->getOpcode() != ISD::DELETED_NODE) AddToWorklist(N); return SDValue(N, 0); } } // Everything under here is trying to match an extract of a loaded value. // If the result of load has to be truncated, then it's not necessarily // profitable. bool BCNumEltsChanged = false; EVT ExtVT = VecVT.getVectorElementType(); EVT LVT = ExtVT; if (ScalarVT.bitsLT(LVT) && !TLI.isTruncateFree(LVT, ScalarVT)) return SDValue(); if (VecOp.getOpcode() == ISD::BITCAST) { // Don't duplicate a load with other uses. if (!VecOp.hasOneUse()) return SDValue(); EVT BCVT = VecOp.getOperand(0).getValueType(); if (!BCVT.isVector() || ExtVT.bitsGT(BCVT.getVectorElementType())) return SDValue(); if (NumElts != BCVT.getVectorNumElements()) BCNumEltsChanged = true; VecOp = VecOp.getOperand(0); ExtVT = BCVT.getVectorElementType(); } // extract (vector load $addr), i --> load $addr + i * size if (!LegalOperations && !IndexC && VecOp.hasOneUse() && ISD::isNormalLoad(VecOp.getNode()) && !Index->hasPredecessor(VecOp.getNode())) { auto *VecLoad = dyn_cast(VecOp); if (VecLoad && !VecLoad->isVolatile()) return scalarizeExtractedVectorLoad(N, VecVT, Index, VecLoad); } // Perform only after legalization to ensure build_vector / vector_shuffle // optimizations have already been done. if (!LegalOperations || !IndexC) return SDValue(); // (vextract (v4f32 load $addr), c) -> (f32 load $addr+c*size) // (vextract (v4f32 s2v (f32 load $addr)), c) -> (f32 load $addr+c*size) // (vextract (v4f32 shuffle (load $addr), <1,u,u,u>), 0) -> (f32 load $addr) int Elt = IndexC->getZExtValue(); LoadSDNode *LN0 = nullptr; if (ISD::isNormalLoad(VecOp.getNode())) { LN0 = cast(VecOp); } else if (VecOp.getOpcode() == ISD::SCALAR_TO_VECTOR && VecOp.getOperand(0).getValueType() == ExtVT && ISD::isNormalLoad(VecOp.getOperand(0).getNode())) { // Don't duplicate a load with other uses. if (!VecOp.hasOneUse()) return SDValue(); LN0 = cast(VecOp.getOperand(0)); } if (auto *Shuf = dyn_cast(VecOp)) { // (vextract (vector_shuffle (load $addr), v2, <1, u, u, u>), 1) // => // (load $addr+1*size) // Don't duplicate a load with other uses. if (!VecOp.hasOneUse()) return SDValue(); // If the bit convert changed the number of elements, it is unsafe // to examine the mask. if (BCNumEltsChanged) return SDValue(); // Select the input vector, guarding against out of range extract vector. int Idx = (Elt > (int)NumElts) ? -1 : Shuf->getMaskElt(Elt); VecOp = (Idx < (int)NumElts) ? VecOp.getOperand(0) : VecOp.getOperand(1); if (VecOp.getOpcode() == ISD::BITCAST) { // Don't duplicate a load with other uses. if (!VecOp.hasOneUse()) return SDValue(); VecOp = VecOp.getOperand(0); } if (ISD::isNormalLoad(VecOp.getNode())) { LN0 = cast(VecOp); Elt = (Idx < (int)NumElts) ? Idx : Idx - (int)NumElts; Index = DAG.getConstant(Elt, DL, Index.getValueType()); } } // Make sure we found a non-volatile load and the extractelement is // the only use. if (!LN0 || !LN0->hasNUsesOfValue(1,0) || LN0->isVolatile()) return SDValue(); // If Idx was -1 above, Elt is going to be -1, so just return undef. if (Elt == -1) return DAG.getUNDEF(LVT); return scalarizeExtractedVectorLoad(N, VecVT, Index, LN0); } // Simplify (build_vec (ext )) to (bitcast (build_vec )) SDValue DAGCombiner::reduceBuildVecExtToExtBuildVec(SDNode *N) { // We perform this optimization post type-legalization because // the type-legalizer often scalarizes integer-promoted vectors. // Performing this optimization before may create bit-casts which // will be type-legalized to complex code sequences. // We perform this optimization only before the operation legalizer because we // may introduce illegal operations. if (Level != AfterLegalizeVectorOps && Level != AfterLegalizeTypes) return SDValue(); unsigned NumInScalars = N->getNumOperands(); SDLoc DL(N); EVT VT = N->getValueType(0); // Check to see if this is a BUILD_VECTOR of a bunch of values // which come from any_extend or zero_extend nodes. If so, we can create // a new BUILD_VECTOR using bit-casts which may enable other BUILD_VECTOR // optimizations. We do not handle sign-extend because we can't fill the sign // using shuffles. EVT SourceType = MVT::Other; bool AllAnyExt = true; for (unsigned i = 0; i != NumInScalars; ++i) { SDValue In = N->getOperand(i); // Ignore undef inputs. if (In.isUndef()) continue; bool AnyExt = In.getOpcode() == ISD::ANY_EXTEND; bool ZeroExt = In.getOpcode() == ISD::ZERO_EXTEND; // Abort if the element is not an extension. if (!ZeroExt && !AnyExt) { SourceType = MVT::Other; break; } // The input is a ZeroExt or AnyExt. Check the original type. EVT InTy = In.getOperand(0).getValueType(); // Check that all of the widened source types are the same. if (SourceType == MVT::Other) // First time. SourceType = InTy; else if (InTy != SourceType) { // Multiple income types. Abort. SourceType = MVT::Other; break; } // Check if all of the extends are ANY_EXTENDs. AllAnyExt &= AnyExt; } // In order to have valid types, all of the inputs must be extended from the // same source type and all of the inputs must be any or zero extend. // Scalar sizes must be a power of two. EVT OutScalarTy = VT.getScalarType(); bool ValidTypes = SourceType != MVT::Other && isPowerOf2_32(OutScalarTy.getSizeInBits()) && isPowerOf2_32(SourceType.getSizeInBits()); // Create a new simpler BUILD_VECTOR sequence which other optimizations can // turn into a single shuffle instruction. if (!ValidTypes) return SDValue(); bool isLE = DAG.getDataLayout().isLittleEndian(); unsigned ElemRatio = OutScalarTy.getSizeInBits()/SourceType.getSizeInBits(); assert(ElemRatio > 1 && "Invalid element size ratio"); SDValue Filler = AllAnyExt ? DAG.getUNDEF(SourceType): DAG.getConstant(0, DL, SourceType); unsigned NewBVElems = ElemRatio * VT.getVectorNumElements(); SmallVector Ops(NewBVElems, Filler); // Populate the new build_vector for (unsigned i = 0, e = N->getNumOperands(); i != e; ++i) { SDValue Cast = N->getOperand(i); assert((Cast.getOpcode() == ISD::ANY_EXTEND || Cast.getOpcode() == ISD::ZERO_EXTEND || Cast.isUndef()) && "Invalid cast opcode"); SDValue In; if (Cast.isUndef()) In = DAG.getUNDEF(SourceType); else In = Cast->getOperand(0); unsigned Index = isLE ? (i * ElemRatio) : (i * ElemRatio + (ElemRatio - 1)); assert(Index < Ops.size() && "Invalid index"); Ops[Index] = In; } // The type of the new BUILD_VECTOR node. EVT VecVT = EVT::getVectorVT(*DAG.getContext(), SourceType, NewBVElems); assert(VecVT.getSizeInBits() == VT.getSizeInBits() && "Invalid vector size"); // Check if the new vector type is legal. if (!isTypeLegal(VecVT) || (!TLI.isOperationLegal(ISD::BUILD_VECTOR, VecVT) && TLI.isOperationLegal(ISD::BUILD_VECTOR, VT))) return SDValue(); // Make the new BUILD_VECTOR. SDValue BV = DAG.getBuildVector(VecVT, DL, Ops); // The new BUILD_VECTOR node has the potential to be further optimized. AddToWorklist(BV.getNode()); // Bitcast to the desired type. return DAG.getBitcast(VT, BV); } SDValue DAGCombiner::createBuildVecShuffle(const SDLoc &DL, SDNode *N, ArrayRef VectorMask, SDValue VecIn1, SDValue VecIn2, unsigned LeftIdx) { MVT IdxTy = TLI.getVectorIdxTy(DAG.getDataLayout()); SDValue ZeroIdx = DAG.getConstant(0, DL, IdxTy); EVT VT = N->getValueType(0); EVT InVT1 = VecIn1.getValueType(); EVT InVT2 = VecIn2.getNode() ? VecIn2.getValueType() : InVT1; unsigned Vec2Offset = 0; unsigned NumElems = VT.getVectorNumElements(); unsigned ShuffleNumElems = NumElems; // In case both the input vectors are extracted from same base // vector we do not need extra addend (Vec2Offset) while // computing shuffle mask. if (!VecIn2 || !(VecIn1.getOpcode() == ISD::EXTRACT_SUBVECTOR) || !(VecIn2.getOpcode() == ISD::EXTRACT_SUBVECTOR) || !(VecIn1.getOperand(0) == VecIn2.getOperand(0))) Vec2Offset = InVT1.getVectorNumElements(); // We can't generate a shuffle node with mismatched input and output types. // Try to make the types match the type of the output. if (InVT1 != VT || InVT2 != VT) { if ((VT.getSizeInBits() % InVT1.getSizeInBits() == 0) && InVT1 == InVT2) { // If the output vector length is a multiple of both input lengths, // we can concatenate them and pad the rest with undefs. unsigned NumConcats = VT.getSizeInBits() / InVT1.getSizeInBits(); assert(NumConcats >= 2 && "Concat needs at least two inputs!"); SmallVector ConcatOps(NumConcats, DAG.getUNDEF(InVT1)); ConcatOps[0] = VecIn1; ConcatOps[1] = VecIn2 ? VecIn2 : DAG.getUNDEF(InVT1); VecIn1 = DAG.getNode(ISD::CONCAT_VECTORS, DL, VT, ConcatOps); VecIn2 = SDValue(); } else if (InVT1.getSizeInBits() == VT.getSizeInBits() * 2) { if (!TLI.isExtractSubvectorCheap(VT, InVT1, NumElems)) return SDValue(); if (!VecIn2.getNode()) { // If we only have one input vector, and it's twice the size of the // output, split it in two. VecIn2 = DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, VT, VecIn1, DAG.getConstant(NumElems, DL, IdxTy)); VecIn1 = DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, VT, VecIn1, ZeroIdx); // Since we now have shorter input vectors, adjust the offset of the // second vector's start. Vec2Offset = NumElems; } else if (InVT2.getSizeInBits() <= InVT1.getSizeInBits()) { // VecIn1 is wider than the output, and we have another, possibly // smaller input. Pad the smaller input with undefs, shuffle at the // input vector width, and extract the output. // The shuffle type is different than VT, so check legality again. if (LegalOperations && !TLI.isOperationLegal(ISD::VECTOR_SHUFFLE, InVT1)) return SDValue(); // Legalizing INSERT_SUBVECTOR is tricky - you basically have to // lower it back into a BUILD_VECTOR. So if the inserted type is // illegal, don't even try. if (InVT1 != InVT2) { if (!TLI.isTypeLegal(InVT2)) return SDValue(); VecIn2 = DAG.getNode(ISD::INSERT_SUBVECTOR, DL, InVT1, DAG.getUNDEF(InVT1), VecIn2, ZeroIdx); } ShuffleNumElems = NumElems * 2; } else { // Both VecIn1 and VecIn2 are wider than the output, and VecIn2 is wider // than VecIn1. We can't handle this for now - this case will disappear // when we start sorting the vectors by type. return SDValue(); } } else if (InVT2.getSizeInBits() * 2 == VT.getSizeInBits() && InVT1.getSizeInBits() == VT.getSizeInBits()) { SmallVector ConcatOps(2, DAG.getUNDEF(InVT2)); ConcatOps[0] = VecIn2; VecIn2 = DAG.getNode(ISD::CONCAT_VECTORS, DL, VT, ConcatOps); } else { // TODO: Support cases where the length mismatch isn't exactly by a // factor of 2. // TODO: Move this check upwards, so that if we have bad type // mismatches, we don't create any DAG nodes. return SDValue(); } } // Initialize mask to undef. SmallVector Mask(ShuffleNumElems, -1); // Only need to run up to the number of elements actually used, not the // total number of elements in the shuffle - if we are shuffling a wider // vector, the high lanes should be set to undef. for (unsigned i = 0; i != NumElems; ++i) { if (VectorMask[i] <= 0) continue; unsigned ExtIndex = N->getOperand(i).getConstantOperandVal(1); if (VectorMask[i] == (int)LeftIdx) { Mask[i] = ExtIndex; } else if (VectorMask[i] == (int)LeftIdx + 1) { Mask[i] = Vec2Offset + ExtIndex; } } // The type the input vectors may have changed above. InVT1 = VecIn1.getValueType(); // If we already have a VecIn2, it should have the same type as VecIn1. // If we don't, get an undef/zero vector of the appropriate type. VecIn2 = VecIn2.getNode() ? VecIn2 : DAG.getUNDEF(InVT1); assert(InVT1 == VecIn2.getValueType() && "Unexpected second input type."); SDValue Shuffle = DAG.getVectorShuffle(InVT1, DL, VecIn1, VecIn2, Mask); if (ShuffleNumElems > NumElems) Shuffle = DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, VT, Shuffle, ZeroIdx); return Shuffle; } static SDValue reduceBuildVecToShuffleWithZero(SDNode *BV, SelectionDAG &DAG) { assert(BV->getOpcode() == ISD::BUILD_VECTOR && "Expected build vector"); // First, determine where the build vector is not undef. // TODO: We could extend this to handle zero elements as well as undefs. int NumBVOps = BV->getNumOperands(); int ZextElt = -1; for (int i = 0; i != NumBVOps; ++i) { SDValue Op = BV->getOperand(i); if (Op.isUndef()) continue; if (ZextElt == -1) ZextElt = i; else return SDValue(); } // Bail out if there's no non-undef element. if (ZextElt == -1) return SDValue(); // The build vector contains some number of undef elements and exactly // one other element. That other element must be a zero-extended scalar // extracted from a vector at a constant index to turn this into a shuffle. + // Also, require that the build vector does not implicitly truncate/extend + // its elements. // TODO: This could be enhanced to allow ANY_EXTEND as well as ZERO_EXTEND. + EVT VT = BV->getValueType(0); SDValue Zext = BV->getOperand(ZextElt); if (Zext.getOpcode() != ISD::ZERO_EXTEND || !Zext.hasOneUse() || Zext.getOperand(0).getOpcode() != ISD::EXTRACT_VECTOR_ELT || - !isa(Zext.getOperand(0).getOperand(1))) + !isa(Zext.getOperand(0).getOperand(1)) || + Zext.getValueSizeInBits() != VT.getScalarSizeInBits()) return SDValue(); - // The zero-extend must be a multiple of the source size. + // The zero-extend must be a multiple of the source size, and we must be + // building a vector of the same size as the source of the extract element. SDValue Extract = Zext.getOperand(0); unsigned DestSize = Zext.getValueSizeInBits(); unsigned SrcSize = Extract.getValueSizeInBits(); - if (DestSize % SrcSize != 0) + if (DestSize % SrcSize != 0 || + Extract.getOperand(0).getValueSizeInBits() != VT.getSizeInBits()) return SDValue(); // Create a shuffle mask that will combine the extracted element with zeros // and undefs. - int ZextRatio = DestSize / SrcSize; + int ZextRatio = DestSize / SrcSize; int NumMaskElts = NumBVOps * ZextRatio; SmallVector ShufMask(NumMaskElts, -1); for (int i = 0; i != NumMaskElts; ++i) { if (i / ZextRatio == ZextElt) { // The low bits of the (potentially translated) extracted element map to // the source vector. The high bits map to zero. We will use a zero vector // as the 2nd source operand of the shuffle, so use the 1st element of // that vector (mask value is number-of-elements) for the high bits. if (i % ZextRatio == 0) ShufMask[i] = Extract.getConstantOperandVal(1); else ShufMask[i] = NumMaskElts; } // Undef elements of the build vector remain undef because we initialize // the shuffle mask with -1. } // Turn this into a shuffle with zero if that's legal. EVT VecVT = Extract.getOperand(0).getValueType(); if (!DAG.getTargetLoweringInfo().isShuffleMaskLegal(ShufMask, VecVT)) return SDValue(); // buildvec undef, ..., (zext (extractelt V, IndexC)), undef... --> // bitcast (shuffle V, ZeroVec, VectorMask) SDLoc DL(BV); SDValue ZeroVec = DAG.getConstant(0, DL, VecVT); SDValue Shuf = DAG.getVectorShuffle(VecVT, DL, Extract.getOperand(0), ZeroVec, ShufMask); - return DAG.getBitcast(BV->getValueType(0), Shuf); + return DAG.getBitcast(VT, Shuf); } // Check to see if this is a BUILD_VECTOR of a bunch of EXTRACT_VECTOR_ELT // operations. If the types of the vectors we're extracting from allow it, // turn this into a vector_shuffle node. SDValue DAGCombiner::reduceBuildVecToShuffle(SDNode *N) { SDLoc DL(N); EVT VT = N->getValueType(0); // Only type-legal BUILD_VECTOR nodes are converted to shuffle nodes. if (!isTypeLegal(VT)) return SDValue(); if (SDValue V = reduceBuildVecToShuffleWithZero(N, DAG)) return V; // May only combine to shuffle after legalize if shuffle is legal. if (LegalOperations && !TLI.isOperationLegal(ISD::VECTOR_SHUFFLE, VT)) return SDValue(); bool UsesZeroVector = false; unsigned NumElems = N->getNumOperands(); // Record, for each element of the newly built vector, which input vector // that element comes from. -1 stands for undef, 0 for the zero vector, // and positive values for the input vectors. // VectorMask maps each element to its vector number, and VecIn maps vector // numbers to their initial SDValues. SmallVector VectorMask(NumElems, -1); SmallVector VecIn; VecIn.push_back(SDValue()); for (unsigned i = 0; i != NumElems; ++i) { SDValue Op = N->getOperand(i); if (Op.isUndef()) continue; // See if we can use a blend with a zero vector. // TODO: Should we generalize this to a blend with an arbitrary constant // vector? if (isNullConstant(Op) || isNullFPConstant(Op)) { UsesZeroVector = true; VectorMask[i] = 0; continue; } // Not an undef or zero. If the input is something other than an // EXTRACT_VECTOR_ELT with an in-range constant index, bail out. if (Op.getOpcode() != ISD::EXTRACT_VECTOR_ELT || !isa(Op.getOperand(1))) return SDValue(); SDValue ExtractedFromVec = Op.getOperand(0); APInt ExtractIdx = cast(Op.getOperand(1))->getAPIntValue(); if (ExtractIdx.uge(ExtractedFromVec.getValueType().getVectorNumElements())) return SDValue(); // All inputs must have the same element type as the output. if (VT.getVectorElementType() != ExtractedFromVec.getValueType().getVectorElementType()) return SDValue(); // Have we seen this input vector before? // The vectors are expected to be tiny (usually 1 or 2 elements), so using // a map back from SDValues to numbers isn't worth it. unsigned Idx = std::distance( VecIn.begin(), std::find(VecIn.begin(), VecIn.end(), ExtractedFromVec)); if (Idx == VecIn.size()) VecIn.push_back(ExtractedFromVec); VectorMask[i] = Idx; } // If we didn't find at least one input vector, bail out. if (VecIn.size() < 2) return SDValue(); // If all the Operands of BUILD_VECTOR extract from same // vector, then split the vector efficiently based on the maximum // vector access index and adjust the VectorMask and // VecIn accordingly. if (VecIn.size() == 2) { unsigned MaxIndex = 0; unsigned NearestPow2 = 0; SDValue Vec = VecIn.back(); EVT InVT = Vec.getValueType(); MVT IdxTy = TLI.getVectorIdxTy(DAG.getDataLayout()); SmallVector IndexVec(NumElems, 0); for (unsigned i = 0; i < NumElems; i++) { if (VectorMask[i] <= 0) continue; unsigned Index = N->getOperand(i).getConstantOperandVal(1); IndexVec[i] = Index; MaxIndex = std::max(MaxIndex, Index); } NearestPow2 = PowerOf2Ceil(MaxIndex); if (InVT.isSimple() && NearestPow2 > 2 && MaxIndex < NearestPow2 && NumElems * 2 < NearestPow2) { unsigned SplitSize = NearestPow2 / 2; EVT SplitVT = EVT::getVectorVT(*DAG.getContext(), InVT.getVectorElementType(), SplitSize); if (TLI.isTypeLegal(SplitVT)) { SDValue VecIn2 = DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, SplitVT, Vec, DAG.getConstant(SplitSize, DL, IdxTy)); SDValue VecIn1 = DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, SplitVT, Vec, DAG.getConstant(0, DL, IdxTy)); VecIn.pop_back(); VecIn.push_back(VecIn1); VecIn.push_back(VecIn2); for (unsigned i = 0; i < NumElems; i++) { if (VectorMask[i] <= 0) continue; VectorMask[i] = (IndexVec[i] < SplitSize) ? 1 : 2; } } } } // TODO: We want to sort the vectors by descending length, so that adjacent // pairs have similar length, and the longer vector is always first in the // pair. // TODO: Should this fire if some of the input vectors has illegal type (like // it does now), or should we let legalization run its course first? // Shuffle phase: // Take pairs of vectors, and shuffle them so that the result has elements // from these vectors in the correct places. // For example, given: // t10: i32 = extract_vector_elt t1, Constant:i64<0> // t11: i32 = extract_vector_elt t2, Constant:i64<0> // t12: i32 = extract_vector_elt t3, Constant:i64<0> // t13: i32 = extract_vector_elt t1, Constant:i64<1> // t14: v4i32 = BUILD_VECTOR t10, t11, t12, t13 // We will generate: // t20: v4i32 = vector_shuffle<0,4,u,1> t1, t2 // t21: v4i32 = vector_shuffle t3, undef SmallVector Shuffles; for (unsigned In = 0, Len = (VecIn.size() / 2); In < Len; ++In) { unsigned LeftIdx = 2 * In + 1; SDValue VecLeft = VecIn[LeftIdx]; SDValue VecRight = (LeftIdx + 1) < VecIn.size() ? VecIn[LeftIdx + 1] : SDValue(); if (SDValue Shuffle = createBuildVecShuffle(DL, N, VectorMask, VecLeft, VecRight, LeftIdx)) Shuffles.push_back(Shuffle); else return SDValue(); } // If we need the zero vector as an "ingredient" in the blend tree, add it // to the list of shuffles. if (UsesZeroVector) Shuffles.push_back(VT.isInteger() ? DAG.getConstant(0, DL, VT) : DAG.getConstantFP(0.0, DL, VT)); // If we only have one shuffle, we're done. if (Shuffles.size() == 1) return Shuffles[0]; // Update the vector mask to point to the post-shuffle vectors. for (int &Vec : VectorMask) if (Vec == 0) Vec = Shuffles.size() - 1; else Vec = (Vec - 1) / 2; // More than one shuffle. Generate a binary tree of blends, e.g. if from // the previous step we got the set of shuffles t10, t11, t12, t13, we will // generate: // t10: v8i32 = vector_shuffle<0,8,u,u,u,u,u,u> t1, t2 // t11: v8i32 = vector_shuffle t3, t4 // t12: v8i32 = vector_shuffle t5, t6 // t13: v8i32 = vector_shuffle t7, t8 // t20: v8i32 = vector_shuffle<0,1,10,11,u,u,u,u> t10, t11 // t21: v8i32 = vector_shuffle t12, t13 // t30: v8i32 = vector_shuffle<0,1,2,3,12,13,14,15> t20, t21 // Make sure the initial size of the shuffle list is even. if (Shuffles.size() % 2) Shuffles.push_back(DAG.getUNDEF(VT)); for (unsigned CurSize = Shuffles.size(); CurSize > 1; CurSize /= 2) { if (CurSize % 2) { Shuffles[CurSize] = DAG.getUNDEF(VT); CurSize++; } for (unsigned In = 0, Len = CurSize / 2; In < Len; ++In) { int Left = 2 * In; int Right = 2 * In + 1; SmallVector Mask(NumElems, -1); for (unsigned i = 0; i != NumElems; ++i) { if (VectorMask[i] == Left) { Mask[i] = i; VectorMask[i] = In; } else if (VectorMask[i] == Right) { Mask[i] = i + NumElems; VectorMask[i] = In; } } Shuffles[In] = DAG.getVectorShuffle(VT, DL, Shuffles[Left], Shuffles[Right], Mask); } } return Shuffles[0]; } // Try to turn a build vector of zero extends of extract vector elts into a // a vector zero extend and possibly an extract subvector. // TODO: Support sign extend or any extend? // TODO: Allow undef elements? // TODO: Don't require the extracts to start at element 0. SDValue DAGCombiner::convertBuildVecZextToZext(SDNode *N) { if (LegalOperations) return SDValue(); EVT VT = N->getValueType(0); SDValue Op0 = N->getOperand(0); auto checkElem = [&](SDValue Op) -> int64_t { if (Op.getOpcode() == ISD::ZERO_EXTEND && Op.getOperand(0).getOpcode() == ISD::EXTRACT_VECTOR_ELT && Op0.getOperand(0).getOperand(0) == Op.getOperand(0).getOperand(0)) if (auto *C = dyn_cast(Op.getOperand(0).getOperand(1))) return C->getZExtValue(); return -1; }; // Make sure the first element matches // (zext (extract_vector_elt X, C)) int64_t Offset = checkElem(Op0); if (Offset < 0) return SDValue(); unsigned NumElems = N->getNumOperands(); SDValue In = Op0.getOperand(0).getOperand(0); EVT InSVT = In.getValueType().getScalarType(); EVT InVT = EVT::getVectorVT(*DAG.getContext(), InSVT, NumElems); // Don't create an illegal input type after type legalization. if (LegalTypes && !TLI.isTypeLegal(InVT)) return SDValue(); // Ensure all the elements come from the same vector and are adjacent. for (unsigned i = 1; i != NumElems; ++i) { if ((Offset + i) != checkElem(N->getOperand(i))) return SDValue(); } SDLoc DL(N); In = DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, InVT, In, Op0.getOperand(0).getOperand(1)); return DAG.getNode(ISD::ZERO_EXTEND, DL, VT, In); } SDValue DAGCombiner::visitBUILD_VECTOR(SDNode *N) { EVT VT = N->getValueType(0); // A vector built entirely of undefs is undef. if (ISD::allOperandsUndef(N)) return DAG.getUNDEF(VT); // If this is a splat of a bitcast from another vector, change to a // concat_vector. // For example: // (build_vector (i64 (bitcast (v2i32 X))), (i64 (bitcast (v2i32 X)))) -> // (v2i64 (bitcast (concat_vectors (v2i32 X), (v2i32 X)))) // // If X is a build_vector itself, the concat can become a larger build_vector. // TODO: Maybe this is useful for non-splat too? if (!LegalOperations) { if (SDValue Splat = cast(N)->getSplatValue()) { Splat = peekThroughBitcasts(Splat); EVT SrcVT = Splat.getValueType(); if (SrcVT.isVector()) { unsigned NumElts = N->getNumOperands() * SrcVT.getVectorNumElements(); EVT NewVT = EVT::getVectorVT(*DAG.getContext(), SrcVT.getVectorElementType(), NumElts); if (!LegalTypes || TLI.isTypeLegal(NewVT)) { SmallVector Ops(N->getNumOperands(), Splat); SDValue Concat = DAG.getNode(ISD::CONCAT_VECTORS, SDLoc(N), NewVT, Ops); return DAG.getBitcast(VT, Concat); } } } } // Check if we can express BUILD VECTOR via subvector extract. if (!LegalTypes && (N->getNumOperands() > 1)) { SDValue Op0 = N->getOperand(0); auto checkElem = [&](SDValue Op) -> uint64_t { if ((Op.getOpcode() == ISD::EXTRACT_VECTOR_ELT) && (Op0.getOperand(0) == Op.getOperand(0))) if (auto CNode = dyn_cast(Op.getOperand(1))) return CNode->getZExtValue(); return -1; }; int Offset = checkElem(Op0); for (unsigned i = 0; i < N->getNumOperands(); ++i) { if (Offset + i != checkElem(N->getOperand(i))) { Offset = -1; break; } } if ((Offset == 0) && (Op0.getOperand(0).getValueType() == N->getValueType(0))) return Op0.getOperand(0); if ((Offset != -1) && ((Offset % N->getValueType(0).getVectorNumElements()) == 0)) // IDX must be multiple of output size. return DAG.getNode(ISD::EXTRACT_SUBVECTOR, SDLoc(N), N->getValueType(0), Op0.getOperand(0), Op0.getOperand(1)); } if (SDValue V = convertBuildVecZextToZext(N)) return V; if (SDValue V = reduceBuildVecExtToExtBuildVec(N)) return V; if (SDValue V = reduceBuildVecToShuffle(N)) return V; return SDValue(); } static SDValue combineConcatVectorOfScalars(SDNode *N, SelectionDAG &DAG) { const TargetLowering &TLI = DAG.getTargetLoweringInfo(); EVT OpVT = N->getOperand(0).getValueType(); // If the operands are legal vectors, leave them alone. if (TLI.isTypeLegal(OpVT)) return SDValue(); SDLoc DL(N); EVT VT = N->getValueType(0); SmallVector Ops; EVT SVT = EVT::getIntegerVT(*DAG.getContext(), OpVT.getSizeInBits()); SDValue ScalarUndef = DAG.getNode(ISD::UNDEF, DL, SVT); // Keep track of what we encounter. bool AnyInteger = false; bool AnyFP = false; for (const SDValue &Op : N->ops()) { if (ISD::BITCAST == Op.getOpcode() && !Op.getOperand(0).getValueType().isVector()) Ops.push_back(Op.getOperand(0)); else if (ISD::UNDEF == Op.getOpcode()) Ops.push_back(ScalarUndef); else return SDValue(); // Note whether we encounter an integer or floating point scalar. // If it's neither, bail out, it could be something weird like x86mmx. EVT LastOpVT = Ops.back().getValueType(); if (LastOpVT.isFloatingPoint()) AnyFP = true; else if (LastOpVT.isInteger()) AnyInteger = true; else return SDValue(); } // If any of the operands is a floating point scalar bitcast to a vector, // use floating point types throughout, and bitcast everything. // Replace UNDEFs by another scalar UNDEF node, of the final desired type. if (AnyFP) { SVT = EVT::getFloatingPointVT(OpVT.getSizeInBits()); ScalarUndef = DAG.getNode(ISD::UNDEF, DL, SVT); if (AnyInteger) { for (SDValue &Op : Ops) { if (Op.getValueType() == SVT) continue; if (Op.isUndef()) Op = ScalarUndef; else Op = DAG.getBitcast(SVT, Op); } } } EVT VecVT = EVT::getVectorVT(*DAG.getContext(), SVT, VT.getSizeInBits() / SVT.getSizeInBits()); return DAG.getBitcast(VT, DAG.getBuildVector(VecVT, DL, Ops)); } // Check to see if this is a CONCAT_VECTORS of a bunch of EXTRACT_SUBVECTOR // operations. If so, and if the EXTRACT_SUBVECTOR vector inputs come from at // most two distinct vectors the same size as the result, attempt to turn this // into a legal shuffle. static SDValue combineConcatVectorOfExtracts(SDNode *N, SelectionDAG &DAG) { EVT VT = N->getValueType(0); EVT OpVT = N->getOperand(0).getValueType(); int NumElts = VT.getVectorNumElements(); int NumOpElts = OpVT.getVectorNumElements(); SDValue SV0 = DAG.getUNDEF(VT), SV1 = DAG.getUNDEF(VT); SmallVector Mask; for (SDValue Op : N->ops()) { Op = peekThroughBitcasts(Op); // UNDEF nodes convert to UNDEF shuffle mask values. if (Op.isUndef()) { Mask.append((unsigned)NumOpElts, -1); continue; } if (Op.getOpcode() != ISD::EXTRACT_SUBVECTOR) return SDValue(); // What vector are we extracting the subvector from and at what index? SDValue ExtVec = Op.getOperand(0); // We want the EVT of the original extraction to correctly scale the // extraction index. EVT ExtVT = ExtVec.getValueType(); ExtVec = peekThroughBitcasts(ExtVec); // UNDEF nodes convert to UNDEF shuffle mask values. if (ExtVec.isUndef()) { Mask.append((unsigned)NumOpElts, -1); continue; } if (!isa(Op.getOperand(1))) return SDValue(); int ExtIdx = Op.getConstantOperandVal(1); // Ensure that we are extracting a subvector from a vector the same // size as the result. if (ExtVT.getSizeInBits() != VT.getSizeInBits()) return SDValue(); // Scale the subvector index to account for any bitcast. int NumExtElts = ExtVT.getVectorNumElements(); if (0 == (NumExtElts % NumElts)) ExtIdx /= (NumExtElts / NumElts); else if (0 == (NumElts % NumExtElts)) ExtIdx *= (NumElts / NumExtElts); else return SDValue(); // At most we can reference 2 inputs in the final shuffle. if (SV0.isUndef() || SV0 == ExtVec) { SV0 = ExtVec; for (int i = 0; i != NumOpElts; ++i) Mask.push_back(i + ExtIdx); } else if (SV1.isUndef() || SV1 == ExtVec) { SV1 = ExtVec; for (int i = 0; i != NumOpElts; ++i) Mask.push_back(i + ExtIdx + NumElts); } else { return SDValue(); } } if (!DAG.getTargetLoweringInfo().isShuffleMaskLegal(Mask, VT)) return SDValue(); return DAG.getVectorShuffle(VT, SDLoc(N), DAG.getBitcast(VT, SV0), DAG.getBitcast(VT, SV1), Mask); } SDValue DAGCombiner::visitCONCAT_VECTORS(SDNode *N) { // If we only have one input vector, we don't need to do any concatenation. if (N->getNumOperands() == 1) return N->getOperand(0); // Check if all of the operands are undefs. EVT VT = N->getValueType(0); if (ISD::allOperandsUndef(N)) return DAG.getUNDEF(VT); // Optimize concat_vectors where all but the first of the vectors are undef. if (std::all_of(std::next(N->op_begin()), N->op_end(), [](const SDValue &Op) { return Op.isUndef(); })) { SDValue In = N->getOperand(0); assert(In.getValueType().isVector() && "Must concat vectors"); SDValue Scalar = peekThroughOneUseBitcasts(In); // concat_vectors(scalar_to_vector(scalar), undef) -> // scalar_to_vector(scalar) if (!LegalOperations && Scalar.getOpcode() == ISD::SCALAR_TO_VECTOR && Scalar.hasOneUse()) { EVT SVT = Scalar.getValueType().getVectorElementType(); if (SVT == Scalar.getOperand(0).getValueType()) Scalar = Scalar.getOperand(0); } // concat_vectors(scalar, undef) -> scalar_to_vector(scalar) if (!Scalar.getValueType().isVector()) { // If the bitcast type isn't legal, it might be a trunc of a legal type; // look through the trunc so we can still do the transform: // concat_vectors(trunc(scalar), undef) -> scalar_to_vector(scalar) if (Scalar->getOpcode() == ISD::TRUNCATE && !TLI.isTypeLegal(Scalar.getValueType()) && TLI.isTypeLegal(Scalar->getOperand(0).getValueType())) Scalar = Scalar->getOperand(0); EVT SclTy = Scalar.getValueType(); if (!SclTy.isFloatingPoint() && !SclTy.isInteger()) return SDValue(); // Bail out if the vector size is not a multiple of the scalar size. if (VT.getSizeInBits() % SclTy.getSizeInBits()) return SDValue(); unsigned VNTNumElms = VT.getSizeInBits() / SclTy.getSizeInBits(); if (VNTNumElms < 2) return SDValue(); EVT NVT = EVT::getVectorVT(*DAG.getContext(), SclTy, VNTNumElms); if (!TLI.isTypeLegal(NVT) || !TLI.isTypeLegal(Scalar.getValueType())) return SDValue(); SDValue Res = DAG.getNode(ISD::SCALAR_TO_VECTOR, SDLoc(N), NVT, Scalar); return DAG.getBitcast(VT, Res); } } // Fold any combination of BUILD_VECTOR or UNDEF nodes into one BUILD_VECTOR. // We have already tested above for an UNDEF only concatenation. // fold (concat_vectors (BUILD_VECTOR A, B, ...), (BUILD_VECTOR C, D, ...)) // -> (BUILD_VECTOR A, B, ..., C, D, ...) auto IsBuildVectorOrUndef = [](const SDValue &Op) { return ISD::UNDEF == Op.getOpcode() || ISD::BUILD_VECTOR == Op.getOpcode(); }; if (llvm::all_of(N->ops(), IsBuildVectorOrUndef)) { SmallVector Opnds; EVT SVT = VT.getScalarType(); EVT MinVT = SVT; if (!SVT.isFloatingPoint()) { // If BUILD_VECTOR are from built from integer, they may have different // operand types. Get the smallest type and truncate all operands to it. bool FoundMinVT = false; for (const SDValue &Op : N->ops()) if (ISD::BUILD_VECTOR == Op.getOpcode()) { EVT OpSVT = Op.getOperand(0).getValueType(); MinVT = (!FoundMinVT || OpSVT.bitsLE(MinVT)) ? OpSVT : MinVT; FoundMinVT = true; } assert(FoundMinVT && "Concat vector type mismatch"); } for (const SDValue &Op : N->ops()) { EVT OpVT = Op.getValueType(); unsigned NumElts = OpVT.getVectorNumElements(); if (ISD::UNDEF == Op.getOpcode()) Opnds.append(NumElts, DAG.getUNDEF(MinVT)); if (ISD::BUILD_VECTOR == Op.getOpcode()) { if (SVT.isFloatingPoint()) { assert(SVT == OpVT.getScalarType() && "Concat vector type mismatch"); Opnds.append(Op->op_begin(), Op->op_begin() + NumElts); } else { for (unsigned i = 0; i != NumElts; ++i) Opnds.push_back( DAG.getNode(ISD::TRUNCATE, SDLoc(N), MinVT, Op.getOperand(i))); } } } assert(VT.getVectorNumElements() == Opnds.size() && "Concat vector type mismatch"); return DAG.getBuildVector(VT, SDLoc(N), Opnds); } // Fold CONCAT_VECTORS of only bitcast scalars (or undef) to BUILD_VECTOR. if (SDValue V = combineConcatVectorOfScalars(N, DAG)) return V; // Fold CONCAT_VECTORS of EXTRACT_SUBVECTOR (or undef) to VECTOR_SHUFFLE. if (Level < AfterLegalizeVectorOps && TLI.isTypeLegal(VT)) if (SDValue V = combineConcatVectorOfExtracts(N, DAG)) return V; // Type legalization of vectors and DAG canonicalization of SHUFFLE_VECTOR // nodes often generate nop CONCAT_VECTOR nodes. // Scan the CONCAT_VECTOR operands and look for a CONCAT operations that // place the incoming vectors at the exact same location. SDValue SingleSource = SDValue(); unsigned PartNumElem = N->getOperand(0).getValueType().getVectorNumElements(); for (unsigned i = 0, e = N->getNumOperands(); i != e; ++i) { SDValue Op = N->getOperand(i); if (Op.isUndef()) continue; // Check if this is the identity extract: if (Op.getOpcode() != ISD::EXTRACT_SUBVECTOR) return SDValue(); // Find the single incoming vector for the extract_subvector. if (SingleSource.getNode()) { if (Op.getOperand(0) != SingleSource) return SDValue(); } else { SingleSource = Op.getOperand(0); // Check the source type is the same as the type of the result. // If not, this concat may extend the vector, so we can not // optimize it away. if (SingleSource.getValueType() != N->getValueType(0)) return SDValue(); } unsigned IdentityIndex = i * PartNumElem; ConstantSDNode *CS = dyn_cast(Op.getOperand(1)); // The extract index must be constant. if (!CS) return SDValue(); // Check that we are reading from the identity index. if (CS->getZExtValue() != IdentityIndex) return SDValue(); } if (SingleSource.getNode()) return SingleSource; return SDValue(); } /// If we are extracting a subvector produced by a wide binary operator try /// to use a narrow binary operator and/or avoid concatenation and extraction. static SDValue narrowExtractedVectorBinOp(SDNode *Extract, SelectionDAG &DAG) { // TODO: Refactor with the caller (visitEXTRACT_SUBVECTOR), so we can share // some of these bailouts with other transforms. // The extract index must be a constant, so we can map it to a concat operand. auto *ExtractIndexC = dyn_cast(Extract->getOperand(1)); if (!ExtractIndexC) return SDValue(); // We are looking for an optionally bitcasted wide vector binary operator // feeding an extract subvector. SDValue BinOp = peekThroughBitcasts(Extract->getOperand(0)); if (!ISD::isBinaryOp(BinOp.getNode())) return SDValue(); // The binop must be a vector type, so we can extract some fraction of it. EVT WideBVT = BinOp.getValueType(); if (!WideBVT.isVector()) return SDValue(); EVT VT = Extract->getValueType(0); unsigned ExtractIndex = ExtractIndexC->getZExtValue(); assert(ExtractIndex % VT.getVectorNumElements() == 0 && "Extract index is not a multiple of the vector length."); // Bail out if this is not a proper multiple width extraction. unsigned WideWidth = WideBVT.getSizeInBits(); unsigned NarrowWidth = VT.getSizeInBits(); if (WideWidth % NarrowWidth != 0) return SDValue(); // Bail out if we are extracting a fraction of a single operation. This can // occur because we potentially looked through a bitcast of the binop. unsigned NarrowingRatio = WideWidth / NarrowWidth; unsigned WideNumElts = WideBVT.getVectorNumElements(); if (WideNumElts % NarrowingRatio != 0) return SDValue(); // Bail out if the target does not support a narrower version of the binop. EVT NarrowBVT = EVT::getVectorVT(*DAG.getContext(), WideBVT.getScalarType(), WideNumElts / NarrowingRatio); unsigned BOpcode = BinOp.getOpcode(); const TargetLowering &TLI = DAG.getTargetLoweringInfo(); if (!TLI.isOperationLegalOrCustomOrPromote(BOpcode, NarrowBVT)) return SDValue(); // If extraction is cheap, we don't need to look at the binop operands // for concat ops. The narrow binop alone makes this transform profitable. // We can't just reuse the original extract index operand because we may have // bitcasted. unsigned ConcatOpNum = ExtractIndex / VT.getVectorNumElements(); unsigned ExtBOIdx = ConcatOpNum * NarrowBVT.getVectorNumElements(); EVT ExtBOIdxVT = Extract->getOperand(1).getValueType(); if (TLI.isExtractSubvectorCheap(NarrowBVT, WideBVT, ExtBOIdx) && BinOp.hasOneUse() && Extract->getOperand(0)->hasOneUse()) { // extract (binop B0, B1), N --> binop (extract B0, N), (extract B1, N) SDLoc DL(Extract); SDValue NewExtIndex = DAG.getConstant(ExtBOIdx, DL, ExtBOIdxVT); SDValue X = DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, NarrowBVT, BinOp.getOperand(0), NewExtIndex); SDValue Y = DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, NarrowBVT, BinOp.getOperand(1), NewExtIndex); SDValue NarrowBinOp = DAG.getNode(BOpcode, DL, NarrowBVT, X, Y, BinOp.getNode()->getFlags()); return DAG.getBitcast(VT, NarrowBinOp); } // Only handle the case where we are doubling and then halving. A larger ratio // may require more than two narrow binops to replace the wide binop. if (NarrowingRatio != 2) return SDValue(); // TODO: The motivating case for this transform is an x86 AVX1 target. That // target has temptingly almost legal versions of bitwise logic ops in 256-bit // flavors, but no other 256-bit integer support. This could be extended to // handle any binop, but that may require fixing/adding other folds to avoid // codegen regressions. if (BOpcode != ISD::AND && BOpcode != ISD::OR && BOpcode != ISD::XOR) return SDValue(); // We need at least one concatenation operation of a binop operand to make // this transform worthwhile. The concat must double the input vector sizes. // TODO: Should we also handle INSERT_SUBVECTOR patterns? SDValue LHS = peekThroughBitcasts(BinOp.getOperand(0)); SDValue RHS = peekThroughBitcasts(BinOp.getOperand(1)); bool ConcatL = LHS.getOpcode() == ISD::CONCAT_VECTORS && LHS.getNumOperands() == 2; bool ConcatR = RHS.getOpcode() == ISD::CONCAT_VECTORS && RHS.getNumOperands() == 2; if (!ConcatL && !ConcatR) return SDValue(); // If one of the binop operands was not the result of a concat, we must // extract a half-sized operand for our new narrow binop. SDLoc DL(Extract); // extract (binop (concat X1, X2), (concat Y1, Y2)), N --> binop XN, YN // extract (binop (concat X1, X2), Y), N --> binop XN, (extract Y, N) // extract (binop X, (concat Y1, Y2)), N --> binop (extract X, N), YN SDValue X = ConcatL ? DAG.getBitcast(NarrowBVT, LHS.getOperand(ConcatOpNum)) : DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, NarrowBVT, BinOp.getOperand(0), DAG.getConstant(ExtBOIdx, DL, ExtBOIdxVT)); SDValue Y = ConcatR ? DAG.getBitcast(NarrowBVT, RHS.getOperand(ConcatOpNum)) : DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, NarrowBVT, BinOp.getOperand(1), DAG.getConstant(ExtBOIdx, DL, ExtBOIdxVT)); SDValue NarrowBinOp = DAG.getNode(BOpcode, DL, NarrowBVT, X, Y); return DAG.getBitcast(VT, NarrowBinOp); } /// If we are extracting a subvector from a wide vector load, convert to a /// narrow load to eliminate the extraction: /// (extract_subvector (load wide vector)) --> (load narrow vector) static SDValue narrowExtractedVectorLoad(SDNode *Extract, SelectionDAG &DAG) { // TODO: Add support for big-endian. The offset calculation must be adjusted. if (DAG.getDataLayout().isBigEndian()) return SDValue(); auto *Ld = dyn_cast(Extract->getOperand(0)); auto *ExtIdx = dyn_cast(Extract->getOperand(1)); if (!Ld || Ld->getExtensionType() || Ld->isVolatile() || !ExtIdx) return SDValue(); // Allow targets to opt-out. EVT VT = Extract->getValueType(0); const TargetLowering &TLI = DAG.getTargetLoweringInfo(); if (!TLI.shouldReduceLoadWidth(Ld, Ld->getExtensionType(), VT)) return SDValue(); // The narrow load will be offset from the base address of the old load if // we are extracting from something besides index 0 (little-endian). SDLoc DL(Extract); SDValue BaseAddr = Ld->getOperand(1); unsigned Offset = ExtIdx->getZExtValue() * VT.getScalarType().getStoreSize(); // TODO: Use "BaseIndexOffset" to make this more effective. SDValue NewAddr = DAG.getMemBasePlusOffset(BaseAddr, Offset, DL); MachineFunction &MF = DAG.getMachineFunction(); MachineMemOperand *MMO = MF.getMachineMemOperand(Ld->getMemOperand(), Offset, VT.getStoreSize()); SDValue NewLd = DAG.getLoad(VT, DL, Ld->getChain(), NewAddr, MMO); DAG.makeEquivalentMemoryOrdering(Ld, NewLd); return NewLd; } SDValue DAGCombiner::visitEXTRACT_SUBVECTOR(SDNode* N) { EVT NVT = N->getValueType(0); SDValue V = N->getOperand(0); // Extract from UNDEF is UNDEF. if (V.isUndef()) return DAG.getUNDEF(NVT); if (TLI.isOperationLegalOrCustomOrPromote(ISD::LOAD, NVT)) if (SDValue NarrowLoad = narrowExtractedVectorLoad(N, DAG)) return NarrowLoad; // Combine: // (extract_subvec (concat V1, V2, ...), i) // Into: // Vi if possible // Only operand 0 is checked as 'concat' assumes all inputs of the same // type. if (V.getOpcode() == ISD::CONCAT_VECTORS && isa(N->getOperand(1)) && V.getOperand(0).getValueType() == NVT) { unsigned Idx = N->getConstantOperandVal(1); unsigned NumElems = NVT.getVectorNumElements(); assert((Idx % NumElems) == 0 && "IDX in concat is not a multiple of the result vector length."); return V->getOperand(Idx / NumElems); } V = peekThroughBitcasts(V); // If the input is a build vector. Try to make a smaller build vector. if (V.getOpcode() == ISD::BUILD_VECTOR) { if (auto *Idx = dyn_cast(N->getOperand(1))) { EVT InVT = V.getValueType(); unsigned ExtractSize = NVT.getSizeInBits(); unsigned EltSize = InVT.getScalarSizeInBits(); // Only do this if we won't split any elements. if (ExtractSize % EltSize == 0) { unsigned NumElems = ExtractSize / EltSize; EVT EltVT = InVT.getVectorElementType(); EVT ExtractVT = NumElems == 1 ? EltVT : EVT::getVectorVT(*DAG.getContext(), EltVT, NumElems); if ((Level < AfterLegalizeDAG || (NumElems == 1 || TLI.isOperationLegal(ISD::BUILD_VECTOR, ExtractVT))) && (!LegalTypes || TLI.isTypeLegal(ExtractVT))) { unsigned IdxVal = (Idx->getZExtValue() * NVT.getScalarSizeInBits()) / EltSize; if (NumElems == 1) { SDValue Src = V->getOperand(IdxVal); if (EltVT != Src.getValueType()) Src = DAG.getNode(ISD::TRUNCATE, SDLoc(N), InVT, Src); return DAG.getBitcast(NVT, Src); } // Extract the pieces from the original build_vector. SDValue BuildVec = DAG.getBuildVector(ExtractVT, SDLoc(N), makeArrayRef(V->op_begin() + IdxVal, NumElems)); return DAG.getBitcast(NVT, BuildVec); } } } } if (V.getOpcode() == ISD::INSERT_SUBVECTOR) { // Handle only simple case where vector being inserted and vector // being extracted are of same size. EVT SmallVT = V.getOperand(1).getValueType(); if (!NVT.bitsEq(SmallVT)) return SDValue(); // Only handle cases where both indexes are constants. auto *ExtIdx = dyn_cast(N->getOperand(1)); auto *InsIdx = dyn_cast(V.getOperand(2)); if (InsIdx && ExtIdx) { // Combine: // (extract_subvec (insert_subvec V1, V2, InsIdx), ExtIdx) // Into: // indices are equal or bit offsets are equal => V1 // otherwise => (extract_subvec V1, ExtIdx) if (InsIdx->getZExtValue() * SmallVT.getScalarSizeInBits() == ExtIdx->getZExtValue() * NVT.getScalarSizeInBits()) return DAG.getBitcast(NVT, V.getOperand(1)); return DAG.getNode( ISD::EXTRACT_SUBVECTOR, SDLoc(N), NVT, DAG.getBitcast(N->getOperand(0).getValueType(), V.getOperand(0)), N->getOperand(1)); } } if (SDValue NarrowBOp = narrowExtractedVectorBinOp(N, DAG)) return NarrowBOp; if (SimplifyDemandedVectorElts(SDValue(N, 0))) return SDValue(N, 0); return SDValue(); } // Tries to turn a shuffle of two CONCAT_VECTORS into a single concat, // or turn a shuffle of a single concat into simpler shuffle then concat. static SDValue partitionShuffleOfConcats(SDNode *N, SelectionDAG &DAG) { EVT VT = N->getValueType(0); unsigned NumElts = VT.getVectorNumElements(); SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); ShuffleVectorSDNode *SVN = cast(N); SmallVector Ops; EVT ConcatVT = N0.getOperand(0).getValueType(); unsigned NumElemsPerConcat = ConcatVT.getVectorNumElements(); unsigned NumConcats = NumElts / NumElemsPerConcat; // Special case: shuffle(concat(A,B)) can be more efficiently represented // as concat(shuffle(A,B),UNDEF) if the shuffle doesn't set any of the high // half vector elements. if (NumElemsPerConcat * 2 == NumElts && N1.isUndef() && std::all_of(SVN->getMask().begin() + NumElemsPerConcat, SVN->getMask().end(), [](int i) { return i == -1; })) { N0 = DAG.getVectorShuffle(ConcatVT, SDLoc(N), N0.getOperand(0), N0.getOperand(1), makeArrayRef(SVN->getMask().begin(), NumElemsPerConcat)); N1 = DAG.getUNDEF(ConcatVT); return DAG.getNode(ISD::CONCAT_VECTORS, SDLoc(N), VT, N0, N1); } // Look at every vector that's inserted. We're looking for exact // subvector-sized copies from a concatenated vector for (unsigned I = 0; I != NumConcats; ++I) { // Make sure we're dealing with a copy. unsigned Begin = I * NumElemsPerConcat; bool AllUndef = true, NoUndef = true; for (unsigned J = Begin; J != Begin + NumElemsPerConcat; ++J) { if (SVN->getMaskElt(J) >= 0) AllUndef = false; else NoUndef = false; } if (NoUndef) { if (SVN->getMaskElt(Begin) % NumElemsPerConcat != 0) return SDValue(); for (unsigned J = 1; J != NumElemsPerConcat; ++J) if (SVN->getMaskElt(Begin + J - 1) + 1 != SVN->getMaskElt(Begin + J)) return SDValue(); unsigned FirstElt = SVN->getMaskElt(Begin) / NumElemsPerConcat; if (FirstElt < N0.getNumOperands()) Ops.push_back(N0.getOperand(FirstElt)); else Ops.push_back(N1.getOperand(FirstElt - N0.getNumOperands())); } else if (AllUndef) { Ops.push_back(DAG.getUNDEF(N0.getOperand(0).getValueType())); } else { // Mixed with general masks and undefs, can't do optimization. return SDValue(); } } return DAG.getNode(ISD::CONCAT_VECTORS, SDLoc(N), VT, Ops); } // Attempt to combine a shuffle of 2 inputs of 'scalar sources' - // BUILD_VECTOR or SCALAR_TO_VECTOR into a single BUILD_VECTOR. // // SHUFFLE(BUILD_VECTOR(), BUILD_VECTOR()) -> BUILD_VECTOR() is always // a simplification in some sense, but it isn't appropriate in general: some // BUILD_VECTORs are substantially cheaper than others. The general case // of a BUILD_VECTOR requires inserting each element individually (or // performing the equivalent in a temporary stack variable). A BUILD_VECTOR of // all constants is a single constant pool load. A BUILD_VECTOR where each // element is identical is a splat. A BUILD_VECTOR where most of the operands // are undef lowers to a small number of element insertions. // // To deal with this, we currently use a bunch of mostly arbitrary heuristics. // We don't fold shuffles where one side is a non-zero constant, and we don't // fold shuffles if the resulting (non-splat) BUILD_VECTOR would have duplicate // non-constant operands. This seems to work out reasonably well in practice. static SDValue combineShuffleOfScalars(ShuffleVectorSDNode *SVN, SelectionDAG &DAG, const TargetLowering &TLI) { EVT VT = SVN->getValueType(0); unsigned NumElts = VT.getVectorNumElements(); SDValue N0 = SVN->getOperand(0); SDValue N1 = SVN->getOperand(1); if (!N0->hasOneUse()) return SDValue(); // If only one of N1,N2 is constant, bail out if it is not ALL_ZEROS as // discussed above. if (!N1.isUndef()) { if (!N1->hasOneUse()) return SDValue(); bool N0AnyConst = isAnyConstantBuildVector(N0); bool N1AnyConst = isAnyConstantBuildVector(N1); if (N0AnyConst && !N1AnyConst && !ISD::isBuildVectorAllZeros(N0.getNode())) return SDValue(); if (!N0AnyConst && N1AnyConst && !ISD::isBuildVectorAllZeros(N1.getNode())) return SDValue(); } // If both inputs are splats of the same value then we can safely merge this // to a single BUILD_VECTOR with undef elements based on the shuffle mask. bool IsSplat = false; auto *BV0 = dyn_cast(N0); auto *BV1 = dyn_cast(N1); if (BV0 && BV1) if (SDValue Splat0 = BV0->getSplatValue()) IsSplat = (Splat0 == BV1->getSplatValue()); SmallVector Ops; SmallSet DuplicateOps; for (int M : SVN->getMask()) { SDValue Op = DAG.getUNDEF(VT.getScalarType()); if (M >= 0) { int Idx = M < (int)NumElts ? M : M - NumElts; SDValue &S = (M < (int)NumElts ? N0 : N1); if (S.getOpcode() == ISD::BUILD_VECTOR) { Op = S.getOperand(Idx); } else if (S.getOpcode() == ISD::SCALAR_TO_VECTOR) { assert(Idx == 0 && "Unexpected SCALAR_TO_VECTOR operand index."); Op = S.getOperand(0); } else { // Operand can't be combined - bail out. return SDValue(); } } // Don't duplicate a non-constant BUILD_VECTOR operand unless we're // generating a splat; semantically, this is fine, but it's likely to // generate low-quality code if the target can't reconstruct an appropriate // shuffle. if (!Op.isUndef() && !isa(Op) && !isa(Op)) if (!IsSplat && !DuplicateOps.insert(Op).second) return SDValue(); Ops.push_back(Op); } // BUILD_VECTOR requires all inputs to be of the same type, find the // maximum type and extend them all. EVT SVT = VT.getScalarType(); if (SVT.isInteger()) for (SDValue &Op : Ops) SVT = (SVT.bitsLT(Op.getValueType()) ? Op.getValueType() : SVT); if (SVT != VT.getScalarType()) for (SDValue &Op : Ops) Op = TLI.isZExtFree(Op.getValueType(), SVT) ? DAG.getZExtOrTrunc(Op, SDLoc(SVN), SVT) : DAG.getSExtOrTrunc(Op, SDLoc(SVN), SVT); return DAG.getBuildVector(VT, SDLoc(SVN), Ops); } // Match shuffles that can be converted to any_vector_extend_in_reg. // This is often generated during legalization. // e.g. v4i32 <0,u,1,u> -> (v2i64 any_vector_extend_in_reg(v4i32 src)) // TODO Add support for ZERO_EXTEND_VECTOR_INREG when we have a test case. static SDValue combineShuffleToVectorExtend(ShuffleVectorSDNode *SVN, SelectionDAG &DAG, const TargetLowering &TLI, bool LegalOperations) { EVT VT = SVN->getValueType(0); bool IsBigEndian = DAG.getDataLayout().isBigEndian(); // TODO Add support for big-endian when we have a test case. if (!VT.isInteger() || IsBigEndian) return SDValue(); unsigned NumElts = VT.getVectorNumElements(); unsigned EltSizeInBits = VT.getScalarSizeInBits(); ArrayRef Mask = SVN->getMask(); SDValue N0 = SVN->getOperand(0); // shuffle<0,-1,1,-1> == (v2i64 anyextend_vector_inreg(v4i32)) auto isAnyExtend = [&Mask, &NumElts](unsigned Scale) { for (unsigned i = 0; i != NumElts; ++i) { if (Mask[i] < 0) continue; if ((i % Scale) == 0 && Mask[i] == (int)(i / Scale)) continue; return false; } return true; }; // Attempt to match a '*_extend_vector_inreg' shuffle, we just search for // power-of-2 extensions as they are the most likely. for (unsigned Scale = 2; Scale < NumElts; Scale *= 2) { // Check for non power of 2 vector sizes if (NumElts % Scale != 0) continue; if (!isAnyExtend(Scale)) continue; EVT OutSVT = EVT::getIntegerVT(*DAG.getContext(), EltSizeInBits * Scale); EVT OutVT = EVT::getVectorVT(*DAG.getContext(), OutSVT, NumElts / Scale); // Never create an illegal type. Only create unsupported operations if we // are pre-legalization. if (TLI.isTypeLegal(OutVT)) if (!LegalOperations || TLI.isOperationLegalOrCustom(ISD::ANY_EXTEND_VECTOR_INREG, OutVT)) return DAG.getBitcast(VT, DAG.getNode(ISD::ANY_EXTEND_VECTOR_INREG, SDLoc(SVN), OutVT, N0)); } return SDValue(); } // Detect 'truncate_vector_inreg' style shuffles that pack the lower parts of // each source element of a large type into the lowest elements of a smaller // destination type. This is often generated during legalization. // If the source node itself was a '*_extend_vector_inreg' node then we should // then be able to remove it. static SDValue combineTruncationShuffle(ShuffleVectorSDNode *SVN, SelectionDAG &DAG) { EVT VT = SVN->getValueType(0); bool IsBigEndian = DAG.getDataLayout().isBigEndian(); // TODO Add support for big-endian when we have a test case. if (!VT.isInteger() || IsBigEndian) return SDValue(); SDValue N0 = peekThroughBitcasts(SVN->getOperand(0)); unsigned Opcode = N0.getOpcode(); if (Opcode != ISD::ANY_EXTEND_VECTOR_INREG && Opcode != ISD::SIGN_EXTEND_VECTOR_INREG && Opcode != ISD::ZERO_EXTEND_VECTOR_INREG) return SDValue(); SDValue N00 = N0.getOperand(0); ArrayRef Mask = SVN->getMask(); unsigned NumElts = VT.getVectorNumElements(); unsigned EltSizeInBits = VT.getScalarSizeInBits(); unsigned ExtSrcSizeInBits = N00.getScalarValueSizeInBits(); unsigned ExtDstSizeInBits = N0.getScalarValueSizeInBits(); if (ExtDstSizeInBits % ExtSrcSizeInBits != 0) return SDValue(); unsigned ExtScale = ExtDstSizeInBits / ExtSrcSizeInBits; // (v4i32 truncate_vector_inreg(v2i64)) == shuffle<0,2-1,-1> // (v8i16 truncate_vector_inreg(v4i32)) == shuffle<0,2,4,6,-1,-1,-1,-1> // (v8i16 truncate_vector_inreg(v2i64)) == shuffle<0,4,-1,-1,-1,-1,-1,-1> auto isTruncate = [&Mask, &NumElts](unsigned Scale) { for (unsigned i = 0; i != NumElts; ++i) { if (Mask[i] < 0) continue; if ((i * Scale) < NumElts && Mask[i] == (int)(i * Scale)) continue; return false; } return true; }; // At the moment we just handle the case where we've truncated back to the // same size as before the extension. // TODO: handle more extension/truncation cases as cases arise. if (EltSizeInBits != ExtSrcSizeInBits) return SDValue(); // We can remove *extend_vector_inreg only if the truncation happens at // the same scale as the extension. if (isTruncate(ExtScale)) return DAG.getBitcast(VT, N00); return SDValue(); } // Combine shuffles of splat-shuffles of the form: // shuffle (shuffle V, undef, splat-mask), undef, M // If splat-mask contains undef elements, we need to be careful about // introducing undef's in the folded mask which are not the result of composing // the masks of the shuffles. static SDValue combineShuffleOfSplat(ArrayRef UserMask, ShuffleVectorSDNode *Splat, SelectionDAG &DAG) { ArrayRef SplatMask = Splat->getMask(); assert(UserMask.size() == SplatMask.size() && "Mask length mismatch"); // Prefer simplifying to the splat-shuffle, if possible. This is legal if // every undef mask element in the splat-shuffle has a corresponding undef // element in the user-shuffle's mask or if the composition of mask elements // would result in undef. // Examples for (shuffle (shuffle v, undef, SplatMask), undef, UserMask): // * UserMask=[0,2,u,u], SplatMask=[2,u,2,u] -> [2,2,u,u] // In this case it is not legal to simplify to the splat-shuffle because we // may be exposing the users of the shuffle an undef element at index 1 // which was not there before the combine. // * UserMask=[0,u,2,u], SplatMask=[2,u,2,u] -> [2,u,2,u] // In this case the composition of masks yields SplatMask, so it's ok to // simplify to the splat-shuffle. // * UserMask=[3,u,2,u], SplatMask=[2,u,2,u] -> [u,u,2,u] // In this case the composed mask includes all undef elements of SplatMask // and in addition sets element zero to undef. It is safe to simplify to // the splat-shuffle. auto CanSimplifyToExistingSplat = [](ArrayRef UserMask, ArrayRef SplatMask) { for (unsigned i = 0, e = UserMask.size(); i != e; ++i) if (UserMask[i] != -1 && SplatMask[i] == -1 && SplatMask[UserMask[i]] != -1) return false; return true; }; if (CanSimplifyToExistingSplat(UserMask, SplatMask)) return SDValue(Splat, 0); // Create a new shuffle with a mask that is composed of the two shuffles' // masks. SmallVector NewMask; for (int Idx : UserMask) NewMask.push_back(Idx == -1 ? -1 : SplatMask[Idx]); return DAG.getVectorShuffle(Splat->getValueType(0), SDLoc(Splat), Splat->getOperand(0), Splat->getOperand(1), NewMask); } /// If the shuffle mask is taking exactly one element from the first vector /// operand and passing through all other elements from the second vector /// operand, return the index of the mask element that is choosing an element /// from the first operand. Otherwise, return -1. static int getShuffleMaskIndexOfOneElementFromOp0IntoOp1(ArrayRef Mask) { int MaskSize = Mask.size(); int EltFromOp0 = -1; // TODO: This does not match if there are undef elements in the shuffle mask. // Should we ignore undefs in the shuffle mask instead? The trade-off is // removing an instruction (a shuffle), but losing the knowledge that some // vector lanes are not needed. for (int i = 0; i != MaskSize; ++i) { if (Mask[i] >= 0 && Mask[i] < MaskSize) { // We're looking for a shuffle of exactly one element from operand 0. if (EltFromOp0 != -1) return -1; EltFromOp0 = i; } else if (Mask[i] != i + MaskSize) { // Nothing from operand 1 can change lanes. return -1; } } return EltFromOp0; } /// If a shuffle inserts exactly one element from a source vector operand into /// another vector operand and we can access the specified element as a scalar, /// then we can eliminate the shuffle. static SDValue replaceShuffleOfInsert(ShuffleVectorSDNode *Shuf, SelectionDAG &DAG) { // First, check if we are taking one element of a vector and shuffling that // element into another vector. ArrayRef Mask = Shuf->getMask(); SmallVector CommutedMask(Mask.begin(), Mask.end()); SDValue Op0 = Shuf->getOperand(0); SDValue Op1 = Shuf->getOperand(1); int ShufOp0Index = getShuffleMaskIndexOfOneElementFromOp0IntoOp1(Mask); if (ShufOp0Index == -1) { // Commute mask and check again. ShuffleVectorSDNode::commuteMask(CommutedMask); ShufOp0Index = getShuffleMaskIndexOfOneElementFromOp0IntoOp1(CommutedMask); if (ShufOp0Index == -1) return SDValue(); // Commute operands to match the commuted shuffle mask. std::swap(Op0, Op1); Mask = CommutedMask; } // The shuffle inserts exactly one element from operand 0 into operand 1. // Now see if we can access that element as a scalar via a real insert element // instruction. // TODO: We can try harder to locate the element as a scalar. Examples: it // could be an operand of SCALAR_TO_VECTOR, BUILD_VECTOR, or a constant. assert(Mask[ShufOp0Index] >= 0 && Mask[ShufOp0Index] < (int)Mask.size() && "Shuffle mask value must be from operand 0"); if (Op0.getOpcode() != ISD::INSERT_VECTOR_ELT) return SDValue(); auto *InsIndexC = dyn_cast(Op0.getOperand(2)); if (!InsIndexC || InsIndexC->getSExtValue() != Mask[ShufOp0Index]) return SDValue(); // There's an existing insertelement with constant insertion index, so we // don't need to check the legality/profitability of a replacement operation // that differs at most in the constant value. The target should be able to // lower any of those in a similar way. If not, legalization will expand this // to a scalar-to-vector plus shuffle. // // Note that the shuffle may move the scalar from the position that the insert // element used. Therefore, our new insert element occurs at the shuffle's // mask index value, not the insert's index value. // shuffle (insertelt v1, x, C), v2, mask --> insertelt v2, x, C' SDValue NewInsIndex = DAG.getConstant(ShufOp0Index, SDLoc(Shuf), Op0.getOperand(2).getValueType()); return DAG.getNode(ISD::INSERT_VECTOR_ELT, SDLoc(Shuf), Op0.getValueType(), Op1, Op0.getOperand(1), NewInsIndex); } SDValue DAGCombiner::visitVECTOR_SHUFFLE(SDNode *N) { EVT VT = N->getValueType(0); unsigned NumElts = VT.getVectorNumElements(); SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); assert(N0.getValueType() == VT && "Vector shuffle must be normalized in DAG"); // Canonicalize shuffle undef, undef -> undef if (N0.isUndef() && N1.isUndef()) return DAG.getUNDEF(VT); ShuffleVectorSDNode *SVN = cast(N); // Canonicalize shuffle v, v -> v, undef if (N0 == N1) { SmallVector NewMask; for (unsigned i = 0; i != NumElts; ++i) { int Idx = SVN->getMaskElt(i); if (Idx >= (int)NumElts) Idx -= NumElts; NewMask.push_back(Idx); } return DAG.getVectorShuffle(VT, SDLoc(N), N0, DAG.getUNDEF(VT), NewMask); } // Canonicalize shuffle undef, v -> v, undef. Commute the shuffle mask. if (N0.isUndef()) return DAG.getCommutedVectorShuffle(*SVN); // Remove references to rhs if it is undef if (N1.isUndef()) { bool Changed = false; SmallVector NewMask; for (unsigned i = 0; i != NumElts; ++i) { int Idx = SVN->getMaskElt(i); if (Idx >= (int)NumElts) { Idx = -1; Changed = true; } NewMask.push_back(Idx); } if (Changed) return DAG.getVectorShuffle(VT, SDLoc(N), N0, N1, NewMask); } if (SDValue InsElt = replaceShuffleOfInsert(SVN, DAG)) return InsElt; // A shuffle of a single vector that is a splat can always be folded. if (auto *N0Shuf = dyn_cast(N0)) if (N1->isUndef() && N0Shuf->isSplat()) return combineShuffleOfSplat(SVN->getMask(), N0Shuf, DAG); // If it is a splat, check if the argument vector is another splat or a // build_vector. if (SVN->isSplat() && SVN->getSplatIndex() < (int)NumElts) { SDNode *V = N0.getNode(); // If this is a bit convert that changes the element type of the vector but // not the number of vector elements, look through it. Be careful not to // look though conversions that change things like v4f32 to v2f64. if (V->getOpcode() == ISD::BITCAST) { SDValue ConvInput = V->getOperand(0); if (ConvInput.getValueType().isVector() && ConvInput.getValueType().getVectorNumElements() == NumElts) V = ConvInput.getNode(); } if (V->getOpcode() == ISD::BUILD_VECTOR) { assert(V->getNumOperands() == NumElts && "BUILD_VECTOR has wrong number of operands"); SDValue Base; bool AllSame = true; for (unsigned i = 0; i != NumElts; ++i) { if (!V->getOperand(i).isUndef()) { Base = V->getOperand(i); break; } } // Splat of , return if (!Base.getNode()) return N0; for (unsigned i = 0; i != NumElts; ++i) { if (V->getOperand(i) != Base) { AllSame = false; break; } } // Splat of , return if (AllSame) return N0; // Canonicalize any other splat as a build_vector. const SDValue &Splatted = V->getOperand(SVN->getSplatIndex()); SmallVector Ops(NumElts, Splatted); SDValue NewBV = DAG.getBuildVector(V->getValueType(0), SDLoc(N), Ops); // We may have jumped through bitcasts, so the type of the // BUILD_VECTOR may not match the type of the shuffle. if (V->getValueType(0) != VT) NewBV = DAG.getBitcast(VT, NewBV); return NewBV; } } // Simplify source operands based on shuffle mask. if (SimplifyDemandedVectorElts(SDValue(N, 0))) return SDValue(N, 0); // Match shuffles that can be converted to any_vector_extend_in_reg. if (SDValue V = combineShuffleToVectorExtend(SVN, DAG, TLI, LegalOperations)) return V; // Combine "truncate_vector_in_reg" style shuffles. if (SDValue V = combineTruncationShuffle(SVN, DAG)) return V; if (N0.getOpcode() == ISD::CONCAT_VECTORS && Level < AfterLegalizeVectorOps && (N1.isUndef() || (N1.getOpcode() == ISD::CONCAT_VECTORS && N0.getOperand(0).getValueType() == N1.getOperand(0).getValueType()))) { if (SDValue V = partitionShuffleOfConcats(N, DAG)) return V; } // Attempt to combine a shuffle of 2 inputs of 'scalar sources' - // BUILD_VECTOR or SCALAR_TO_VECTOR into a single BUILD_VECTOR. if (Level < AfterLegalizeDAG && TLI.isTypeLegal(VT)) if (SDValue Res = combineShuffleOfScalars(SVN, DAG, TLI)) return Res; // If this shuffle only has a single input that is a bitcasted shuffle, // attempt to merge the 2 shuffles and suitably bitcast the inputs/output // back to their original types. if (N0.getOpcode() == ISD::BITCAST && N0.hasOneUse() && N1.isUndef() && Level < AfterLegalizeVectorOps && TLI.isTypeLegal(VT)) { auto ScaleShuffleMask = [](ArrayRef Mask, int Scale) { if (Scale == 1) return SmallVector(Mask.begin(), Mask.end()); SmallVector NewMask; for (int M : Mask) for (int s = 0; s != Scale; ++s) NewMask.push_back(M < 0 ? -1 : Scale * M + s); return NewMask; }; SDValue BC0 = peekThroughOneUseBitcasts(N0); if (BC0.getOpcode() == ISD::VECTOR_SHUFFLE && BC0.hasOneUse()) { EVT SVT = VT.getScalarType(); EVT InnerVT = BC0->getValueType(0); EVT InnerSVT = InnerVT.getScalarType(); // Determine which shuffle works with the smaller scalar type. EVT ScaleVT = SVT.bitsLT(InnerSVT) ? VT : InnerVT; EVT ScaleSVT = ScaleVT.getScalarType(); if (TLI.isTypeLegal(ScaleVT) && 0 == (InnerSVT.getSizeInBits() % ScaleSVT.getSizeInBits()) && 0 == (SVT.getSizeInBits() % ScaleSVT.getSizeInBits())) { int InnerScale = InnerSVT.getSizeInBits() / ScaleSVT.getSizeInBits(); int OuterScale = SVT.getSizeInBits() / ScaleSVT.getSizeInBits(); // Scale the shuffle masks to the smaller scalar type. ShuffleVectorSDNode *InnerSVN = cast(BC0); SmallVector InnerMask = ScaleShuffleMask(InnerSVN->getMask(), InnerScale); SmallVector OuterMask = ScaleShuffleMask(SVN->getMask(), OuterScale); // Merge the shuffle masks. SmallVector NewMask; for (int M : OuterMask) NewMask.push_back(M < 0 ? -1 : InnerMask[M]); // Test for shuffle mask legality over both commutations. SDValue SV0 = BC0->getOperand(0); SDValue SV1 = BC0->getOperand(1); bool LegalMask = TLI.isShuffleMaskLegal(NewMask, ScaleVT); if (!LegalMask) { std::swap(SV0, SV1); ShuffleVectorSDNode::commuteMask(NewMask); LegalMask = TLI.isShuffleMaskLegal(NewMask, ScaleVT); } if (LegalMask) { SV0 = DAG.getBitcast(ScaleVT, SV0); SV1 = DAG.getBitcast(ScaleVT, SV1); return DAG.getBitcast( VT, DAG.getVectorShuffle(ScaleVT, SDLoc(N), SV0, SV1, NewMask)); } } } } // Canonicalize shuffles according to rules: // shuffle(A, shuffle(A, B)) -> shuffle(shuffle(A,B), A) // shuffle(B, shuffle(A, B)) -> shuffle(shuffle(A,B), B) // shuffle(B, shuffle(A, Undef)) -> shuffle(shuffle(A, Undef), B) if (N1.getOpcode() == ISD::VECTOR_SHUFFLE && N0.getOpcode() != ISD::VECTOR_SHUFFLE && Level < AfterLegalizeDAG && TLI.isTypeLegal(VT)) { // The incoming shuffle must be of the same type as the result of the // current shuffle. assert(N1->getOperand(0).getValueType() == VT && "Shuffle types don't match"); SDValue SV0 = N1->getOperand(0); SDValue SV1 = N1->getOperand(1); bool HasSameOp0 = N0 == SV0; bool IsSV1Undef = SV1.isUndef(); if (HasSameOp0 || IsSV1Undef || N0 == SV1) // Commute the operands of this shuffle so that next rule // will trigger. return DAG.getCommutedVectorShuffle(*SVN); } // Try to fold according to rules: // shuffle(shuffle(A, B, M0), C, M1) -> shuffle(A, B, M2) // shuffle(shuffle(A, B, M0), C, M1) -> shuffle(A, C, M2) // shuffle(shuffle(A, B, M0), C, M1) -> shuffle(B, C, M2) // Don't try to fold shuffles with illegal type. // Only fold if this shuffle is the only user of the other shuffle. if (N0.getOpcode() == ISD::VECTOR_SHUFFLE && N->isOnlyUserOf(N0.getNode()) && Level < AfterLegalizeDAG && TLI.isTypeLegal(VT)) { ShuffleVectorSDNode *OtherSV = cast(N0); // Don't try to fold splats; they're likely to simplify somehow, or they // might be free. if (OtherSV->isSplat()) return SDValue(); // The incoming shuffle must be of the same type as the result of the // current shuffle. assert(OtherSV->getOperand(0).getValueType() == VT && "Shuffle types don't match"); SDValue SV0, SV1; SmallVector Mask; // Compute the combined shuffle mask for a shuffle with SV0 as the first // operand, and SV1 as the second operand. for (unsigned i = 0; i != NumElts; ++i) { int Idx = SVN->getMaskElt(i); if (Idx < 0) { // Propagate Undef. Mask.push_back(Idx); continue; } SDValue CurrentVec; if (Idx < (int)NumElts) { // This shuffle index refers to the inner shuffle N0. Lookup the inner // shuffle mask to identify which vector is actually referenced. Idx = OtherSV->getMaskElt(Idx); if (Idx < 0) { // Propagate Undef. Mask.push_back(Idx); continue; } CurrentVec = (Idx < (int) NumElts) ? OtherSV->getOperand(0) : OtherSV->getOperand(1); } else { // This shuffle index references an element within N1. CurrentVec = N1; } // Simple case where 'CurrentVec' is UNDEF. if (CurrentVec.isUndef()) { Mask.push_back(-1); continue; } // Canonicalize the shuffle index. We don't know yet if CurrentVec // will be the first or second operand of the combined shuffle. Idx = Idx % NumElts; if (!SV0.getNode() || SV0 == CurrentVec) { // Ok. CurrentVec is the left hand side. // Update the mask accordingly. SV0 = CurrentVec; Mask.push_back(Idx); continue; } // Bail out if we cannot convert the shuffle pair into a single shuffle. if (SV1.getNode() && SV1 != CurrentVec) return SDValue(); // Ok. CurrentVec is the right hand side. // Update the mask accordingly. SV1 = CurrentVec; Mask.push_back(Idx + NumElts); } // Check if all indices in Mask are Undef. In case, propagate Undef. bool isUndefMask = true; for (unsigned i = 0; i != NumElts && isUndefMask; ++i) isUndefMask &= Mask[i] < 0; if (isUndefMask) return DAG.getUNDEF(VT); if (!SV0.getNode()) SV0 = DAG.getUNDEF(VT); if (!SV1.getNode()) SV1 = DAG.getUNDEF(VT); // Avoid introducing shuffles with illegal mask. if (!TLI.isShuffleMaskLegal(Mask, VT)) { ShuffleVectorSDNode::commuteMask(Mask); if (!TLI.isShuffleMaskLegal(Mask, VT)) return SDValue(); // shuffle(shuffle(A, B, M0), C, M1) -> shuffle(B, A, M2) // shuffle(shuffle(A, B, M0), C, M1) -> shuffle(C, A, M2) // shuffle(shuffle(A, B, M0), C, M1) -> shuffle(C, B, M2) std::swap(SV0, SV1); } // shuffle(shuffle(A, B, M0), C, M1) -> shuffle(A, B, M2) // shuffle(shuffle(A, B, M0), C, M1) -> shuffle(A, C, M2) // shuffle(shuffle(A, B, M0), C, M1) -> shuffle(B, C, M2) return DAG.getVectorShuffle(VT, SDLoc(N), SV0, SV1, Mask); } return SDValue(); } SDValue DAGCombiner::visitSCALAR_TO_VECTOR(SDNode *N) { SDValue InVal = N->getOperand(0); EVT VT = N->getValueType(0); // Replace a SCALAR_TO_VECTOR(EXTRACT_VECTOR_ELT(V,C0)) pattern // with a VECTOR_SHUFFLE and possible truncate. if (InVal.getOpcode() == ISD::EXTRACT_VECTOR_ELT) { SDValue InVec = InVal->getOperand(0); SDValue EltNo = InVal->getOperand(1); auto InVecT = InVec.getValueType(); if (ConstantSDNode *C0 = dyn_cast(EltNo)) { SmallVector NewMask(InVecT.getVectorNumElements(), -1); int Elt = C0->getZExtValue(); NewMask[0] = Elt; SDValue Val; // If we have an implict truncate do truncate here as long as it's legal. // if it's not legal, this should if (VT.getScalarType() != InVal.getValueType() && InVal.getValueType().isScalarInteger() && isTypeLegal(VT.getScalarType())) { Val = DAG.getNode(ISD::TRUNCATE, SDLoc(InVal), VT.getScalarType(), InVal); return DAG.getNode(ISD::SCALAR_TO_VECTOR, SDLoc(N), VT, Val); } if (VT.getScalarType() == InVecT.getScalarType() && VT.getVectorNumElements() <= InVecT.getVectorNumElements() && TLI.isShuffleMaskLegal(NewMask, VT)) { Val = DAG.getVectorShuffle(InVecT, SDLoc(N), InVec, DAG.getUNDEF(InVecT), NewMask); // If the initial vector is the correct size this shuffle is a // valid result. if (VT == InVecT) return Val; // If not we must truncate the vector. if (VT.getVectorNumElements() != InVecT.getVectorNumElements()) { MVT IdxTy = TLI.getVectorIdxTy(DAG.getDataLayout()); SDValue ZeroIdx = DAG.getConstant(0, SDLoc(N), IdxTy); EVT SubVT = EVT::getVectorVT(*DAG.getContext(), InVecT.getVectorElementType(), VT.getVectorNumElements()); Val = DAG.getNode(ISD::EXTRACT_SUBVECTOR, SDLoc(N), SubVT, Val, ZeroIdx); return Val; } } } } return SDValue(); } SDValue DAGCombiner::visitINSERT_SUBVECTOR(SDNode *N) { EVT VT = N->getValueType(0); SDValue N0 = N->getOperand(0); SDValue N1 = N->getOperand(1); SDValue N2 = N->getOperand(2); // If inserting an UNDEF, just return the original vector. if (N1.isUndef()) return N0; // If this is an insert of an extracted vector into an undef vector, we can // just use the input to the extract. if (N0.isUndef() && N1.getOpcode() == ISD::EXTRACT_SUBVECTOR && N1.getOperand(1) == N2 && N1.getOperand(0).getValueType() == VT) return N1.getOperand(0); // If we are inserting a bitcast value into an undef, with the same // number of elements, just use the bitcast input of the extract. // i.e. INSERT_SUBVECTOR UNDEF (BITCAST N1) N2 -> // BITCAST (INSERT_SUBVECTOR UNDEF N1 N2) if (N0.isUndef() && N1.getOpcode() == ISD::BITCAST && N1.getOperand(0).getOpcode() == ISD::EXTRACT_SUBVECTOR && N1.getOperand(0).getOperand(1) == N2 && N1.getOperand(0).getOperand(0).getValueType().getVectorNumElements() == VT.getVectorNumElements() && N1.getOperand(0).getOperand(0).getValueType().getSizeInBits() == VT.getSizeInBits()) { return DAG.getBitcast(VT, N1.getOperand(0).getOperand(0)); } // If both N1 and N2 are bitcast values on which insert_subvector // would makes sense, pull the bitcast through. // i.e. INSERT_SUBVECTOR (BITCAST N0) (BITCAST N1) N2 -> // BITCAST (INSERT_SUBVECTOR N0 N1 N2) if (N0.getOpcode() == ISD::BITCAST && N1.getOpcode() == ISD::BITCAST) { SDValue CN0 = N0.getOperand(0); SDValue CN1 = N1.getOperand(0); EVT CN0VT = CN0.getValueType(); EVT CN1VT = CN1.getValueType(); if (CN0VT.isVector() && CN1VT.isVector() && CN0VT.getVectorElementType() == CN1VT.getVectorElementType() && CN0VT.getVectorNumElements() == VT.getVectorNumElements()) { SDValue NewINSERT = DAG.getNode(ISD::INSERT_SUBVECTOR, SDLoc(N), CN0.getValueType(), CN0, CN1, N2); return DAG.getBitcast(VT, NewINSERT); } } // Combine INSERT_SUBVECTORs where we are inserting to the same index. // INSERT_SUBVECTOR( INSERT_SUBVECTOR( Vec, SubOld, Idx ), SubNew, Idx ) // --> INSERT_SUBVECTOR( Vec, SubNew, Idx ) if (N0.getOpcode() == ISD::INSERT_SUBVECTOR && N0.getOperand(1).getValueType() == N1.getValueType() && N0.getOperand(2) == N2) return DAG.getNode(ISD::INSERT_SUBVECTOR, SDLoc(N), VT, N0.getOperand(0), N1, N2); // Eliminate an intermediate insert into an undef vector: // insert_subvector undef, (insert_subvector undef, X, 0), N2 --> // insert_subvector undef, X, N2 if (N0.isUndef() && N1.getOpcode() == ISD::INSERT_SUBVECTOR && N1.getOperand(0).isUndef() && isNullConstant(N1.getOperand(2))) return DAG.getNode(ISD::INSERT_SUBVECTOR, SDLoc(N), VT, N0, N1.getOperand(1), N2); if (!isa(N2)) return SDValue(); unsigned InsIdx = cast(N2)->getZExtValue(); // Canonicalize insert_subvector dag nodes. // Example: // (insert_subvector (insert_subvector A, Idx0), Idx1) // -> (insert_subvector (insert_subvector A, Idx1), Idx0) if (N0.getOpcode() == ISD::INSERT_SUBVECTOR && N0.hasOneUse() && N1.getValueType() == N0.getOperand(1).getValueType() && isa(N0.getOperand(2))) { unsigned OtherIdx = N0.getConstantOperandVal(2); if (InsIdx < OtherIdx) { // Swap nodes. SDValue NewOp = DAG.getNode(ISD::INSERT_SUBVECTOR, SDLoc(N), VT, N0.getOperand(0), N1, N2); AddToWorklist(NewOp.getNode()); return DAG.getNode(ISD::INSERT_SUBVECTOR, SDLoc(N0.getNode()), VT, NewOp, N0.getOperand(1), N0.getOperand(2)); } } // If the input vector is a concatenation, and the insert replaces // one of the pieces, we can optimize into a single concat_vectors. if (N0.getOpcode() == ISD::CONCAT_VECTORS && N0.hasOneUse() && N0.getOperand(0).getValueType() == N1.getValueType()) { unsigned Factor = N1.getValueType().getVectorNumElements(); SmallVector Ops(N0->op_begin(), N0->op_end()); Ops[cast(N2)->getZExtValue() / Factor] = N1; return DAG.getNode(ISD::CONCAT_VECTORS, SDLoc(N), VT, Ops); } // Simplify source operands based on insertion. if (SimplifyDemandedVectorElts(SDValue(N, 0))) return SDValue(N, 0); return SDValue(); } SDValue DAGCombiner::visitFP_TO_FP16(SDNode *N) { SDValue N0 = N->getOperand(0); // fold (fp_to_fp16 (fp16_to_fp op)) -> op if (N0->getOpcode() == ISD::FP16_TO_FP) return N0->getOperand(0); return SDValue(); } SDValue DAGCombiner::visitFP16_TO_FP(SDNode *N) { SDValue N0 = N->getOperand(0); // fold fp16_to_fp(op & 0xffff) -> fp16_to_fp(op) if (N0->getOpcode() == ISD::AND) { ConstantSDNode *AndConst = getAsNonOpaqueConstant(N0.getOperand(1)); if (AndConst && AndConst->getAPIntValue() == 0xffff) { return DAG.getNode(ISD::FP16_TO_FP, SDLoc(N), N->getValueType(0), N0.getOperand(0)); } } return SDValue(); } /// Returns a vector_shuffle if it able to transform an AND to a vector_shuffle /// with the destination vector and a zero vector. /// e.g. AND V, <0xffffffff, 0, 0xffffffff, 0>. ==> /// vector_shuffle V, Zero, <0, 4, 2, 4> SDValue DAGCombiner::XformToShuffleWithZero(SDNode *N) { assert(N->getOpcode() == ISD::AND && "Unexpected opcode!"); EVT VT = N->getValueType(0); SDValue LHS = N->getOperand(0); SDValue RHS = peekThroughBitcasts(N->getOperand(1)); SDLoc DL(N); // Make sure we're not running after operation legalization where it // may have custom lowered the vector shuffles. if (LegalOperations) return SDValue(); if (RHS.getOpcode() != ISD::BUILD_VECTOR) return SDValue(); EVT RVT = RHS.getValueType(); unsigned NumElts = RHS.getNumOperands(); // Attempt to create a valid clear mask, splitting the mask into // sub elements and checking to see if each is // all zeros or all ones - suitable for shuffle masking. auto BuildClearMask = [&](int Split) { int NumSubElts = NumElts * Split; int NumSubBits = RVT.getScalarSizeInBits() / Split; SmallVector Indices; for (int i = 0; i != NumSubElts; ++i) { int EltIdx = i / Split; int SubIdx = i % Split; SDValue Elt = RHS.getOperand(EltIdx); if (Elt.isUndef()) { Indices.push_back(-1); continue; } APInt Bits; if (isa(Elt)) Bits = cast(Elt)->getAPIntValue(); else if (isa(Elt)) Bits = cast(Elt)->getValueAPF().bitcastToAPInt(); else return SDValue(); // Extract the sub element from the constant bit mask. if (DAG.getDataLayout().isBigEndian()) { Bits.lshrInPlace((Split - SubIdx - 1) * NumSubBits); } else { Bits.lshrInPlace(SubIdx * NumSubBits); } if (Split > 1) Bits = Bits.trunc(NumSubBits); if (Bits.isAllOnesValue()) Indices.push_back(i); else if (Bits == 0) Indices.push_back(i + NumSubElts); else return SDValue(); } // Let's see if the target supports this vector_shuffle. EVT ClearSVT = EVT::getIntegerVT(*DAG.getContext(), NumSubBits); EVT ClearVT = EVT::getVectorVT(*DAG.getContext(), ClearSVT, NumSubElts); if (!TLI.isVectorClearMaskLegal(Indices, ClearVT)) return SDValue(); SDValue Zero = DAG.getConstant(0, DL, ClearVT); return DAG.getBitcast(VT, DAG.getVectorShuffle(ClearVT, DL, DAG.getBitcast(ClearVT, LHS), Zero, Indices)); }; // Determine maximum split level (byte level masking). int MaxSplit = 1; if (RVT.getScalarSizeInBits() % 8 == 0) MaxSplit = RVT.getScalarSizeInBits() / 8; for (int Split = 1; Split <= MaxSplit; ++Split) if (RVT.getScalarSizeInBits() % Split == 0) if (SDValue S = BuildClearMask(Split)) return S; return SDValue(); } /// Visit a binary vector operation, like ADD. SDValue DAGCombiner::SimplifyVBinOp(SDNode *N) { assert(N->getValueType(0).isVector() && "SimplifyVBinOp only works on vectors!"); SDValue LHS = N->getOperand(0); SDValue RHS = N->getOperand(1); SDValue Ops[] = {LHS, RHS}; // See if we can constant fold the vector operation. if (SDValue Fold = DAG.FoldConstantVectorArithmetic( N->getOpcode(), SDLoc(LHS), LHS.getValueType(), Ops, N->getFlags())) return Fold; // Type legalization might introduce new shuffles in the DAG. // Fold (VBinOp (shuffle (A, Undef, Mask)), (shuffle (B, Undef, Mask))) // -> (shuffle (VBinOp (A, B)), Undef, Mask). if (LegalTypes && isa(LHS) && isa(RHS) && LHS.hasOneUse() && RHS.hasOneUse() && LHS.getOperand(1).isUndef() && RHS.getOperand(1).isUndef()) { ShuffleVectorSDNode *SVN0 = cast(LHS); ShuffleVectorSDNode *SVN1 = cast(RHS); if (SVN0->getMask().equals(SVN1->getMask())) { EVT VT = N->getValueType(0); SDValue UndefVector = LHS.getOperand(1); SDValue NewBinOp = DAG.getNode(N->getOpcode(), SDLoc(N), VT, LHS.getOperand(0), RHS.getOperand(0), N->getFlags()); AddUsersToWorklist(N); return DAG.getVectorShuffle(VT, SDLoc(N), NewBinOp, UndefVector, SVN0->getMask()); } } return SDValue(); } SDValue DAGCombiner::SimplifySelect(const SDLoc &DL, SDValue N0, SDValue N1, SDValue N2) { assert(N0.getOpcode() ==ISD::SETCC && "First argument must be a SetCC node!"); SDValue SCC = SimplifySelectCC(DL, N0.getOperand(0), N0.getOperand(1), N1, N2, cast(N0.getOperand(2))->get()); // If we got a simplified select_cc node back from SimplifySelectCC, then // break it down into a new SETCC node, and a new SELECT node, and then return // the SELECT node, since we were called with a SELECT node. if (SCC.getNode()) { // Check to see if we got a select_cc back (to turn into setcc/select). // Otherwise, just return whatever node we got back, like fabs. if (SCC.getOpcode() == ISD::SELECT_CC) { SDValue SETCC = DAG.getNode(ISD::SETCC, SDLoc(N0), N0.getValueType(), SCC.getOperand(0), SCC.getOperand(1), SCC.getOperand(4)); AddToWorklist(SETCC.getNode()); return DAG.getSelect(SDLoc(SCC), SCC.getValueType(), SETCC, SCC.getOperand(2), SCC.getOperand(3)); } return SCC; } return SDValue(); } /// Given a SELECT or a SELECT_CC node, where LHS and RHS are the two values /// being selected between, see if we can simplify the select. Callers of this /// should assume that TheSelect is deleted if this returns true. As such, they /// should return the appropriate thing (e.g. the node) back to the top-level of /// the DAG combiner loop to avoid it being looked at. bool DAGCombiner::SimplifySelectOps(SDNode *TheSelect, SDValue LHS, SDValue RHS) { // fold (select (setcc x, [+-]0.0, *lt), NaN, (fsqrt x)) // The select + setcc is redundant, because fsqrt returns NaN for X < 0. if (const ConstantFPSDNode *NaN = isConstOrConstSplatFP(LHS)) { if (NaN->isNaN() && RHS.getOpcode() == ISD::FSQRT) { // We have: (select (setcc ?, ?, ?), NaN, (fsqrt ?)) SDValue Sqrt = RHS; ISD::CondCode CC; SDValue CmpLHS; const ConstantFPSDNode *Zero = nullptr; if (TheSelect->getOpcode() == ISD::SELECT_CC) { CC = cast(TheSelect->getOperand(4))->get(); CmpLHS = TheSelect->getOperand(0); Zero = isConstOrConstSplatFP(TheSelect->getOperand(1)); } else { // SELECT or VSELECT SDValue Cmp = TheSelect->getOperand(0); if (Cmp.getOpcode() == ISD::SETCC) { CC = cast(Cmp.getOperand(2))->get(); CmpLHS = Cmp.getOperand(0); Zero = isConstOrConstSplatFP(Cmp.getOperand(1)); } } if (Zero && Zero->isZero() && Sqrt.getOperand(0) == CmpLHS && (CC == ISD::SETOLT || CC == ISD::SETULT || CC == ISD::SETLT)) { // We have: (select (setcc x, [+-]0.0, *lt), NaN, (fsqrt x)) CombineTo(TheSelect, Sqrt); return true; } } } // Cannot simplify select with vector condition if (TheSelect->getOperand(0).getValueType().isVector()) return false; // If this is a select from two identical things, try to pull the operation // through the select. if (LHS.getOpcode() != RHS.getOpcode() || !LHS.hasOneUse() || !RHS.hasOneUse()) return false; // If this is a load and the token chain is identical, replace the select // of two loads with a load through a select of the address to load from. // This triggers in things like "select bool X, 10.0, 123.0" after the FP // constants have been dropped into the constant pool. if (LHS.getOpcode() == ISD::LOAD) { LoadSDNode *LLD = cast(LHS); LoadSDNode *RLD = cast(RHS); // Token chains must be identical. if (LHS.getOperand(0) != RHS.getOperand(0) || // Do not let this transformation reduce the number of volatile loads. LLD->isVolatile() || RLD->isVolatile() || // FIXME: If either is a pre/post inc/dec load, // we'd need to split out the address adjustment. LLD->isIndexed() || RLD->isIndexed() || // If this is an EXTLOAD, the VT's must match. LLD->getMemoryVT() != RLD->getMemoryVT() || // If this is an EXTLOAD, the kind of extension must match. (LLD->getExtensionType() != RLD->getExtensionType() && // The only exception is if one of the extensions is anyext. LLD->getExtensionType() != ISD::EXTLOAD && RLD->getExtensionType() != ISD::EXTLOAD) || // FIXME: this discards src value information. This is // over-conservative. It would be beneficial to be able to remember // both potential memory locations. Since we are discarding // src value info, don't do the transformation if the memory // locations are not in the default address space. LLD->getPointerInfo().getAddrSpace() != 0 || RLD->getPointerInfo().getAddrSpace() != 0 || !TLI.isOperationLegalOrCustom(TheSelect->getOpcode(), LLD->getBasePtr().getValueType())) return false; // The loads must not depend on one another. if (LLD->isPredecessorOf(RLD) || RLD->isPredecessorOf(LLD)) return false; // Check that the select condition doesn't reach either load. If so, // folding this will induce a cycle into the DAG. If not, this is safe to // xform, so create a select of the addresses. SmallPtrSet Visited; SmallVector Worklist; // Always fail if LLD and RLD are not independent. TheSelect is a // predecessor to all Nodes in question so we need not search past it. Visited.insert(TheSelect); Worklist.push_back(LLD); Worklist.push_back(RLD); if (SDNode::hasPredecessorHelper(LLD, Visited, Worklist) || SDNode::hasPredecessorHelper(RLD, Visited, Worklist)) return false; SDValue Addr; if (TheSelect->getOpcode() == ISD::SELECT) { // We cannot do this optimization if any pair of {RLD, LLD} is a // predecessor to {RLD, LLD, CondNode}. As we've already compared the // Loads, we only need to check if CondNode is a successor to one of the // loads. We can further avoid this if there's no use of their chain // value. SDNode *CondNode = TheSelect->getOperand(0).getNode(); Worklist.push_back(CondNode); if ((LLD->hasAnyUseOfValue(1) && SDNode::hasPredecessorHelper(LLD, Visited, Worklist)) || (RLD->hasAnyUseOfValue(1) && SDNode::hasPredecessorHelper(RLD, Visited, Worklist))) return false; Addr = DAG.getSelect(SDLoc(TheSelect), LLD->getBasePtr().getValueType(), TheSelect->getOperand(0), LLD->getBasePtr(), RLD->getBasePtr()); } else { // Otherwise SELECT_CC // We cannot do this optimization if any pair of {RLD, LLD} is a // predecessor to {RLD, LLD, CondLHS, CondRHS}. As we've already compared // the Loads, we only need to check if CondLHS/CondRHS is a successor to // one of the loads. We can further avoid this if there's no use of their // chain value. SDNode *CondLHS = TheSelect->getOperand(0).getNode(); SDNode *CondRHS = TheSelect->getOperand(1).getNode(); Worklist.push_back(CondLHS); Worklist.push_back(CondRHS); if ((LLD->hasAnyUseOfValue(1) && SDNode::hasPredecessorHelper(LLD, Visited, Worklist)) || (RLD->hasAnyUseOfValue(1) && SDNode::hasPredecessorHelper(RLD, Visited, Worklist))) return false; Addr = DAG.getNode(ISD::SELECT_CC, SDLoc(TheSelect), LLD->getBasePtr().getValueType(), TheSelect->getOperand(0), TheSelect->getOperand(1), LLD->getBasePtr(), RLD->getBasePtr(), TheSelect->getOperand(4)); } SDValue Load; // It is safe to replace the two loads if they have different alignments, // but the new load must be the minimum (most restrictive) alignment of the // inputs. unsigned Alignment = std::min(LLD->getAlignment(), RLD->getAlignment()); MachineMemOperand::Flags MMOFlags = LLD->getMemOperand()->getFlags(); if (!RLD->isInvariant()) MMOFlags &= ~MachineMemOperand::MOInvariant; if (!RLD->isDereferenceable()) MMOFlags &= ~MachineMemOperand::MODereferenceable; if (LLD->getExtensionType() == ISD::NON_EXTLOAD) { // FIXME: Discards pointer and AA info. Load = DAG.getLoad(TheSelect->getValueType(0), SDLoc(TheSelect), LLD->getChain(), Addr, MachinePointerInfo(), Alignment, MMOFlags); } else { // FIXME: Discards pointer and AA info. Load = DAG.getExtLoad( LLD->getExtensionType() == ISD::EXTLOAD ? RLD->getExtensionType() : LLD->getExtensionType(), SDLoc(TheSelect), TheSelect->getValueType(0), LLD->getChain(), Addr, MachinePointerInfo(), LLD->getMemoryVT(), Alignment, MMOFlags); } // Users of the select now use the result of the load. CombineTo(TheSelect, Load); // Users of the old loads now use the new load's chain. We know the // old-load value is dead now. CombineTo(LHS.getNode(), Load.getValue(0), Load.getValue(1)); CombineTo(RHS.getNode(), Load.getValue(0), Load.getValue(1)); return true; } return false; } /// Try to fold an expression of the form (N0 cond N1) ? N2 : N3 to a shift and /// bitwise 'and'. SDValue DAGCombiner::foldSelectCCToShiftAnd(const SDLoc &DL, SDValue N0, SDValue N1, SDValue N2, SDValue N3, ISD::CondCode CC) { // If this is a select where the false operand is zero and the compare is a // check of the sign bit, see if we can perform the "gzip trick": // select_cc setlt X, 0, A, 0 -> and (sra X, size(X)-1), A // select_cc setgt X, 0, A, 0 -> and (not (sra X, size(X)-1)), A EVT XType = N0.getValueType(); EVT AType = N2.getValueType(); if (!isNullConstant(N3) || !XType.bitsGE(AType)) return SDValue(); // If the comparison is testing for a positive value, we have to invert // the sign bit mask, so only do that transform if the target has a bitwise // 'and not' instruction (the invert is free). if (CC == ISD::SETGT && TLI.hasAndNot(N2)) { // (X > -1) ? A : 0 // (X > 0) ? X : 0 <-- This is canonical signed max. if (!(isAllOnesConstant(N1) || (isNullConstant(N1) && N0 == N2))) return SDValue(); } else if (CC == ISD::SETLT) { // (X < 0) ? A : 0 // (X < 1) ? X : 0 <-- This is un-canonicalized signed min. if (!(isNullConstant(N1) || (isOneConstant(N1) && N0 == N2))) return SDValue(); } else { return SDValue(); } // and (sra X, size(X)-1), A -> "and (srl X, C2), A" iff A is a single-bit // constant. EVT ShiftAmtTy = getShiftAmountTy(N0.getValueType()); auto *N2C = dyn_cast(N2.getNode()); if (N2C && ((N2C->getAPIntValue() & (N2C->getAPIntValue() - 1)) == 0)) { unsigned ShCt = XType.getSizeInBits() - N2C->getAPIntValue().logBase2() - 1; SDValue ShiftAmt = DAG.getConstant(ShCt, DL, ShiftAmtTy); SDValue Shift = DAG.getNode(ISD::SRL, DL, XType, N0, ShiftAmt); AddToWorklist(Shift.getNode()); if (XType.bitsGT(AType)) { Shift = DAG.getNode(ISD::TRUNCATE, DL, AType, Shift); AddToWorklist(Shift.getNode()); } if (CC == ISD::SETGT) Shift = DAG.getNOT(DL, Shift, AType); return DAG.getNode(ISD::AND, DL, AType, Shift, N2); } SDValue ShiftAmt = DAG.getConstant(XType.getSizeInBits() - 1, DL, ShiftAmtTy); SDValue Shift = DAG.getNode(ISD::SRA, DL, XType, N0, ShiftAmt); AddToWorklist(Shift.getNode()); if (XType.bitsGT(AType)) { Shift = DAG.getNode(ISD::TRUNCATE, DL, AType, Shift); AddToWorklist(Shift.getNode()); } if (CC == ISD::SETGT) Shift = DAG.getNOT(DL, Shift, AType); return DAG.getNode(ISD::AND, DL, AType, Shift, N2); } /// Turn "(a cond b) ? 1.0f : 2.0f" into "load (tmp + ((a cond b) ? 0 : 4)" /// where "tmp" is a constant pool entry containing an array with 1.0 and 2.0 /// in it. This may be a win when the constant is not otherwise available /// because it replaces two constant pool loads with one. SDValue DAGCombiner::convertSelectOfFPConstantsToLoadOffset( const SDLoc &DL, SDValue N0, SDValue N1, SDValue N2, SDValue N3, ISD::CondCode CC) { if (!TLI.reduceSelectOfFPConstantLoads(N0.getValueType().isFloatingPoint())) return SDValue(); // If we are before legalize types, we want the other legalization to happen // first (for example, to avoid messing with soft float). auto *TV = dyn_cast(N2); auto *FV = dyn_cast(N3); EVT VT = N2.getValueType(); if (!TV || !FV || !TLI.isTypeLegal(VT)) return SDValue(); // If a constant can be materialized without loads, this does not make sense. if (TLI.getOperationAction(ISD::ConstantFP, VT) == TargetLowering::Legal || TLI.isFPImmLegal(TV->getValueAPF(), TV->getValueType(0)) || TLI.isFPImmLegal(FV->getValueAPF(), FV->getValueType(0))) return SDValue(); // If both constants have multiple uses, then we won't need to do an extra // load. The values are likely around in registers for other users. if (!TV->hasOneUse() && !FV->hasOneUse()) return SDValue(); Constant *Elts[] = { const_cast(FV->getConstantFPValue()), const_cast(TV->getConstantFPValue()) }; Type *FPTy = Elts[0]->getType(); const DataLayout &TD = DAG.getDataLayout(); // Create a ConstantArray of the two constants. Constant *CA = ConstantArray::get(ArrayType::get(FPTy, 2), Elts); SDValue CPIdx = DAG.getConstantPool(CA, TLI.getPointerTy(DAG.getDataLayout()), TD.getPrefTypeAlignment(FPTy)); unsigned Alignment = cast(CPIdx)->getAlignment(); // Get offsets to the 0 and 1 elements of the array, so we can select between // them. SDValue Zero = DAG.getIntPtrConstant(0, DL); unsigned EltSize = (unsigned)TD.getTypeAllocSize(Elts[0]->getType()); SDValue One = DAG.getIntPtrConstant(EltSize, SDLoc(FV)); SDValue Cond = DAG.getSetCC(DL, getSetCCResultType(N0.getValueType()), N0, N1, CC); AddToWorklist(Cond.getNode()); SDValue CstOffset = DAG.getSelect(DL, Zero.getValueType(), Cond, One, Zero); AddToWorklist(CstOffset.getNode()); CPIdx = DAG.getNode(ISD::ADD, DL, CPIdx.getValueType(), CPIdx, CstOffset); AddToWorklist(CPIdx.getNode()); return DAG.getLoad(TV->getValueType(0), DL, DAG.getEntryNode(), CPIdx, MachinePointerInfo::getConstantPool( DAG.getMachineFunction()), Alignment); } /// Simplify an expression of the form (N0 cond N1) ? N2 : N3 /// where 'cond' is the comparison specified by CC. SDValue DAGCombiner::SimplifySelectCC(const SDLoc &DL, SDValue N0, SDValue N1, SDValue N2, SDValue N3, ISD::CondCode CC, bool NotExtCompare) { // (x ? y : y) -> y. if (N2 == N3) return N2; EVT CmpOpVT = N0.getValueType(); EVT VT = N2.getValueType(); auto *N1C = dyn_cast(N1.getNode()); auto *N2C = dyn_cast(N2.getNode()); auto *N3C = dyn_cast(N3.getNode()); // Determine if the condition we're dealing with is constant. SDValue SCC = SimplifySetCC(getSetCCResultType(CmpOpVT), N0, N1, CC, DL, false); if (SCC.getNode()) AddToWorklist(SCC.getNode()); if (auto *SCCC = dyn_cast_or_null(SCC.getNode())) { // fold select_cc true, x, y -> x // fold select_cc false, x, y -> y return !SCCC->isNullValue() ? N2 : N3; } if (SDValue V = convertSelectOfFPConstantsToLoadOffset(DL, N0, N1, N2, N3, CC)) return V; if (SDValue V = foldSelectCCToShiftAnd(DL, N0, N1, N2, N3, CC)) return V; // fold (select_cc seteq (and x, y), 0, 0, A) -> (and (shr (shl x)) A) // where y is has a single bit set. // A plaintext description would be, we can turn the SELECT_CC into an AND // when the condition can be materialized as an all-ones register. Any // single bit-test can be materialized as an all-ones register with // shift-left and shift-right-arith. if (CC == ISD::SETEQ && N0->getOpcode() == ISD::AND && N0->getValueType(0) == VT && isNullConstant(N1) && isNullConstant(N2)) { SDValue AndLHS = N0->getOperand(0); auto *ConstAndRHS = dyn_cast(N0->getOperand(1)); if (ConstAndRHS && ConstAndRHS->getAPIntValue().countPopulation() == 1) { // Shift the tested bit over the sign bit. const APInt &AndMask = ConstAndRHS->getAPIntValue(); SDValue ShlAmt = DAG.getConstant(AndMask.countLeadingZeros(), SDLoc(AndLHS), getShiftAmountTy(AndLHS.getValueType())); SDValue Shl = DAG.getNode(ISD::SHL, SDLoc(N0), VT, AndLHS, ShlAmt); // Now arithmetic right shift it all the way over, so the result is either // all-ones, or zero. SDValue ShrAmt = DAG.getConstant(AndMask.getBitWidth() - 1, SDLoc(Shl), getShiftAmountTy(Shl.getValueType())); SDValue Shr = DAG.getNode(ISD::SRA, SDLoc(N0), VT, Shl, ShrAmt); return DAG.getNode(ISD::AND, DL, VT, Shr, N3); } } // fold select C, 16, 0 -> shl C, 4 bool Fold = N2C && isNullConstant(N3) && N2C->getAPIntValue().isPowerOf2(); bool Swap = N3C && isNullConstant(N2) && N3C->getAPIntValue().isPowerOf2(); if ((Fold || Swap) && TLI.getBooleanContents(CmpOpVT) == TargetLowering::ZeroOrOneBooleanContent && (!LegalOperations || TLI.isOperationLegal(ISD::SETCC, CmpOpVT))) { if (Swap) { CC = ISD::getSetCCInverse(CC, CmpOpVT.isInteger()); std::swap(N2C, N3C); } // If the caller doesn't want us to simplify this into a zext of a compare, // don't do it. if (NotExtCompare && N2C->isOne()) return SDValue(); SDValue Temp, SCC; // zext (setcc n0, n1) if (LegalTypes) { SCC = DAG.getSetCC(DL, getSetCCResultType(CmpOpVT), N0, N1, CC); if (VT.bitsLT(SCC.getValueType())) Temp = DAG.getZeroExtendInReg(SCC, SDLoc(N2), VT); else Temp = DAG.getNode(ISD::ZERO_EXTEND, SDLoc(N2), VT, SCC); } else { SCC = DAG.getSetCC(SDLoc(N0), MVT::i1, N0, N1, CC); Temp = DAG.getNode(ISD::ZERO_EXTEND, SDLoc(N2), VT, SCC); } AddToWorklist(SCC.getNode()); AddToWorklist(Temp.getNode()); if (N2C->isOne()) return Temp; // shl setcc result by log2 n2c return DAG.getNode(ISD::SHL, DL, N2.getValueType(), Temp, DAG.getConstant(N2C->getAPIntValue().logBase2(), SDLoc(Temp), getShiftAmountTy(Temp.getValueType()))); } // Check to see if this is an integer abs. // select_cc setg[te] X, 0, X, -X -> // select_cc setgt X, -1, X, -X -> // select_cc setl[te] X, 0, -X, X -> // select_cc setlt X, 1, -X, X -> // Y = sra (X, size(X)-1); xor (add (X, Y), Y) if (N1C) { ConstantSDNode *SubC = nullptr; if (((N1C->isNullValue() && (CC == ISD::SETGT || CC == ISD::SETGE)) || (N1C->isAllOnesValue() && CC == ISD::SETGT)) && N0 == N2 && N3.getOpcode() == ISD::SUB && N0 == N3.getOperand(1)) SubC = dyn_cast(N3.getOperand(0)); else if (((N1C->isNullValue() && (CC == ISD::SETLT || CC == ISD::SETLE)) || (N1C->isOne() && CC == ISD::SETLT)) && N0 == N3 && N2.getOpcode() == ISD::SUB && N0 == N2.getOperand(1)) SubC = dyn_cast(N2.getOperand(0)); if (SubC && SubC->isNullValue() && CmpOpVT.isInteger()) { SDLoc DL(N0); SDValue Shift = DAG.getNode(ISD::SRA, DL, CmpOpVT, N0, DAG.getConstant(CmpOpVT.getSizeInBits() - 1, DL, getShiftAmountTy(CmpOpVT))); SDValue Add = DAG.getNode(ISD::ADD, DL, CmpOpVT, N0, Shift); AddToWorklist(Shift.getNode()); AddToWorklist(Add.getNode()); return DAG.getNode(ISD::XOR, DL, CmpOpVT, Add, Shift); } } // select_cc seteq X, 0, sizeof(X), ctlz(X) -> ctlz(X) // select_cc seteq X, 0, sizeof(X), ctlz_zero_undef(X) -> ctlz(X) // select_cc seteq X, 0, sizeof(X), cttz(X) -> cttz(X) // select_cc seteq X, 0, sizeof(X), cttz_zero_undef(X) -> cttz(X) // select_cc setne X, 0, ctlz(X), sizeof(X) -> ctlz(X) // select_cc setne X, 0, ctlz_zero_undef(X), sizeof(X) -> ctlz(X) // select_cc setne X, 0, cttz(X), sizeof(X) -> cttz(X) // select_cc setne X, 0, cttz_zero_undef(X), sizeof(X) -> cttz(X) if (N1C && N1C->isNullValue() && (CC == ISD::SETEQ || CC == ISD::SETNE)) { SDValue ValueOnZero = N2; SDValue Count = N3; // If the condition is NE instead of E, swap the operands. if (CC == ISD::SETNE) std::swap(ValueOnZero, Count); // Check if the value on zero is a constant equal to the bits in the type. if (auto *ValueOnZeroC = dyn_cast(ValueOnZero)) { if (ValueOnZeroC->getAPIntValue() == VT.getSizeInBits()) { // If the other operand is cttz/cttz_zero_undef of N0, and cttz is // legal, combine to just cttz. if ((Count.getOpcode() == ISD::CTTZ || Count.getOpcode() == ISD::CTTZ_ZERO_UNDEF) && N0 == Count.getOperand(0) && (!LegalOperations || TLI.isOperationLegal(ISD::CTTZ, VT))) return DAG.getNode(ISD::CTTZ, DL, VT, N0); // If the other operand is ctlz/ctlz_zero_undef of N0, and ctlz is // legal, combine to just ctlz. if ((Count.getOpcode() == ISD::CTLZ || Count.getOpcode() == ISD::CTLZ_ZERO_UNDEF) && N0 == Count.getOperand(0) && (!LegalOperations || TLI.isOperationLegal(ISD::CTLZ, VT))) return DAG.getNode(ISD::CTLZ, DL, VT, N0); } } } return SDValue(); } /// This is a stub for TargetLowering::SimplifySetCC. SDValue DAGCombiner::SimplifySetCC(EVT VT, SDValue N0, SDValue N1, ISD::CondCode Cond, const SDLoc &DL, bool foldBooleans) { TargetLowering::DAGCombinerInfo DagCombineInfo(DAG, Level, false, this); return TLI.SimplifySetCC(VT, N0, N1, Cond, foldBooleans, DagCombineInfo, DL); } /// Given an ISD::SDIV node expressing a divide by constant, return /// a DAG expression to select that will generate the same value by multiplying /// by a magic number. /// Ref: "Hacker's Delight" or "The PowerPC Compiler Writer's Guide". SDValue DAGCombiner::BuildSDIV(SDNode *N) { // when optimising for minimum size, we don't want to expand a div to a mul // and a shift. if (DAG.getMachineFunction().getFunction().optForMinSize()) return SDValue(); SmallVector Built; if (SDValue S = TLI.BuildSDIV(N, DAG, LegalOperations, Built)) { for (SDNode *N : Built) AddToWorklist(N); return S; } return SDValue(); } /// Given an ISD::SDIV node expressing a divide by constant power of 2, return a /// DAG expression that will generate the same value by right shifting. SDValue DAGCombiner::BuildSDIVPow2(SDNode *N) { ConstantSDNode *C = isConstOrConstSplat(N->getOperand(1)); if (!C) return SDValue(); // Avoid division by zero. if (C->isNullValue()) return SDValue(); SmallVector Built; if (SDValue S = TLI.BuildSDIVPow2(N, C->getAPIntValue(), DAG, Built)) { for (SDNode *N : Built) AddToWorklist(N); return S; } return SDValue(); } /// Given an ISD::UDIV node expressing a divide by constant, return a DAG /// expression that will generate the same value by multiplying by a magic /// number. /// Ref: "Hacker's Delight" or "The PowerPC Compiler Writer's Guide". SDValue DAGCombiner::BuildUDIV(SDNode *N) { // when optimising for minimum size, we don't want to expand a div to a mul // and a shift. if (DAG.getMachineFunction().getFunction().optForMinSize()) return SDValue(); SmallVector Built; if (SDValue S = TLI.BuildUDIV(N, DAG, LegalOperations, Built)) { for (SDNode *N : Built) AddToWorklist(N); return S; } return SDValue(); } /// Determines the LogBase2 value for a non-null input value using the /// transform: LogBase2(V) = (EltBits - 1) - ctlz(V). SDValue DAGCombiner::BuildLogBase2(SDValue V, const SDLoc &DL) { EVT VT = V.getValueType(); unsigned EltBits = VT.getScalarSizeInBits(); SDValue Ctlz = DAG.getNode(ISD::CTLZ, DL, VT, V); SDValue Base = DAG.getConstant(EltBits - 1, DL, VT); SDValue LogBase2 = DAG.getNode(ISD::SUB, DL, VT, Base, Ctlz); return LogBase2; } /// Newton iteration for a function: F(X) is X_{i+1} = X_i - F(X_i)/F'(X_i) /// For the reciprocal, we need to find the zero of the function: /// F(X) = A X - 1 [which has a zero at X = 1/A] /// => /// X_{i+1} = X_i (2 - A X_i) = X_i + X_i (1 - A X_i) [this second form /// does not require additional intermediate precision] SDValue DAGCombiner::BuildReciprocalEstimate(SDValue Op, SDNodeFlags Flags) { if (Level >= AfterLegalizeDAG) return SDValue(); // TODO: Handle half and/or extended types? EVT VT = Op.getValueType(); if (VT.getScalarType() != MVT::f32 && VT.getScalarType() != MVT::f64) return SDValue(); // If estimates are explicitly disabled for this function, we're done. MachineFunction &MF = DAG.getMachineFunction(); int Enabled = TLI.getRecipEstimateDivEnabled(VT, MF); if (Enabled == TLI.ReciprocalEstimate::Disabled) return SDValue(); // Estimates may be explicitly enabled for this type with a custom number of // refinement steps. int Iterations = TLI.getDivRefinementSteps(VT, MF); if (SDValue Est = TLI.getRecipEstimate(Op, DAG, Enabled, Iterations)) { AddToWorklist(Est.getNode()); if (Iterations) { EVT VT = Op.getValueType(); SDLoc DL(Op); SDValue FPOne = DAG.getConstantFP(1.0, DL, VT); // Newton iterations: Est = Est + Est (1 - Arg * Est) for (int i = 0; i < Iterations; ++i) { SDValue NewEst = DAG.getNode(ISD::FMUL, DL, VT, Op, Est, Flags); AddToWorklist(NewEst.getNode()); NewEst = DAG.getNode(ISD::FSUB, DL, VT, FPOne, NewEst, Flags); AddToWorklist(NewEst.getNode()); NewEst = DAG.getNode(ISD::FMUL, DL, VT, Est, NewEst, Flags); AddToWorklist(NewEst.getNode()); Est = DAG.getNode(ISD::FADD, DL, VT, Est, NewEst, Flags); AddToWorklist(Est.getNode()); } } return Est; } return SDValue(); } /// Newton iteration for a function: F(X) is X_{i+1} = X_i - F(X_i)/F'(X_i) /// For the reciprocal sqrt, we need to find the zero of the function: /// F(X) = 1/X^2 - A [which has a zero at X = 1/sqrt(A)] /// => /// X_{i+1} = X_i (1.5 - A X_i^2 / 2) /// As a result, we precompute A/2 prior to the iteration loop. SDValue DAGCombiner::buildSqrtNROneConst(SDValue Arg, SDValue Est, unsigned Iterations, SDNodeFlags Flags, bool Reciprocal) { EVT VT = Arg.getValueType(); SDLoc DL(Arg); SDValue ThreeHalves = DAG.getConstantFP(1.5, DL, VT); // We now need 0.5 * Arg which we can write as (1.5 * Arg - Arg) so that // this entire sequence requires only one FP constant. SDValue HalfArg = DAG.getNode(ISD::FMUL, DL, VT, ThreeHalves, Arg, Flags); AddToWorklist(HalfArg.getNode()); HalfArg = DAG.getNode(ISD::FSUB, DL, VT, HalfArg, Arg, Flags); AddToWorklist(HalfArg.getNode()); // Newton iterations: Est = Est * (1.5 - HalfArg * Est * Est) for (unsigned i = 0; i < Iterations; ++i) { SDValue NewEst = DAG.getNode(ISD::FMUL, DL, VT, Est, Est, Flags); AddToWorklist(NewEst.getNode()); NewEst = DAG.getNode(ISD::FMUL, DL, VT, HalfArg, NewEst, Flags); AddToWorklist(NewEst.getNode()); NewEst = DAG.getNode(ISD::FSUB, DL, VT, ThreeHalves, NewEst, Flags); AddToWorklist(NewEst.getNode()); Est = DAG.getNode(ISD::FMUL, DL, VT, Est, NewEst, Flags); AddToWorklist(Est.getNode()); } // If non-reciprocal square root is requested, multiply the result by Arg. if (!Reciprocal) { Est = DAG.getNode(ISD::FMUL, DL, VT, Est, Arg, Flags); AddToWorklist(Est.getNode()); } return Est; } /// Newton iteration for a function: F(X) is X_{i+1} = X_i - F(X_i)/F'(X_i) /// For the reciprocal sqrt, we need to find the zero of the function: /// F(X) = 1/X^2 - A [which has a zero at X = 1/sqrt(A)] /// => /// X_{i+1} = (-0.5 * X_i) * (A * X_i * X_i + (-3.0)) SDValue DAGCombiner::buildSqrtNRTwoConst(SDValue Arg, SDValue Est, unsigned Iterations, SDNodeFlags Flags, bool Reciprocal) { EVT VT = Arg.getValueType(); SDLoc DL(Arg); SDValue MinusThree = DAG.getConstantFP(-3.0, DL, VT); SDValue MinusHalf = DAG.getConstantFP(-0.5, DL, VT); // This routine must enter the loop below to work correctly // when (Reciprocal == false). assert(Iterations > 0); // Newton iterations for reciprocal square root: // E = (E * -0.5) * ((A * E) * E + -3.0) for (unsigned i = 0; i < Iterations; ++i) { SDValue AE = DAG.getNode(ISD::FMUL, DL, VT, Arg, Est, Flags); AddToWorklist(AE.getNode()); SDValue AEE = DAG.getNode(ISD::FMUL, DL, VT, AE, Est, Flags); AddToWorklist(AEE.getNode()); SDValue RHS = DAG.getNode(ISD::FADD, DL, VT, AEE, MinusThree, Flags); AddToWorklist(RHS.getNode()); // When calculating a square root at the last iteration build: // S = ((A * E) * -0.5) * ((A * E) * E + -3.0) // (notice a common subexpression) SDValue LHS; if (Reciprocal || (i + 1) < Iterations) { // RSQRT: LHS = (E * -0.5) LHS = DAG.getNode(ISD::FMUL, DL, VT, Est, MinusHalf, Flags); } else { // SQRT: LHS = (A * E) * -0.5 LHS = DAG.getNode(ISD::FMUL, DL, VT, AE, MinusHalf, Flags); } AddToWorklist(LHS.getNode()); Est = DAG.getNode(ISD::FMUL, DL, VT, LHS, RHS, Flags); AddToWorklist(Est.getNode()); } return Est; } /// Build code to calculate either rsqrt(Op) or sqrt(Op). In the latter case /// Op*rsqrt(Op) is actually computed, so additional postprocessing is needed if /// Op can be zero. SDValue DAGCombiner::buildSqrtEstimateImpl(SDValue Op, SDNodeFlags Flags, bool Reciprocal) { if (Level >= AfterLegalizeDAG) return SDValue(); // TODO: Handle half and/or extended types? EVT VT = Op.getValueType(); if (VT.getScalarType() != MVT::f32 && VT.getScalarType() != MVT::f64) return SDValue(); // If estimates are explicitly disabled for this function, we're done. MachineFunction &MF = DAG.getMachineFunction(); int Enabled = TLI.getRecipEstimateSqrtEnabled(VT, MF); if (Enabled == TLI.ReciprocalEstimate::Disabled) return SDValue(); // Estimates may be explicitly enabled for this type with a custom number of // refinement steps. int Iterations = TLI.getSqrtRefinementSteps(VT, MF); bool UseOneConstNR = false; if (SDValue Est = TLI.getSqrtEstimate(Op, DAG, Enabled, Iterations, UseOneConstNR, Reciprocal)) { AddToWorklist(Est.getNode()); if (Iterations) { Est = UseOneConstNR ? buildSqrtNROneConst(Op, Est, Iterations, Flags, Reciprocal) : buildSqrtNRTwoConst(Op, Est, Iterations, Flags, Reciprocal); if (!Reciprocal) { // The estimate is now completely wrong if the input was exactly 0.0 or // possibly a denormal. Force the answer to 0.0 for those cases. EVT VT = Op.getValueType(); SDLoc DL(Op); EVT CCVT = getSetCCResultType(VT); ISD::NodeType SelOpcode = VT.isVector() ? ISD::VSELECT : ISD::SELECT; const Function &F = DAG.getMachineFunction().getFunction(); Attribute Denorms = F.getFnAttribute("denormal-fp-math"); if (Denorms.getValueAsString().equals("ieee")) { // fabs(X) < SmallestNormal ? 0.0 : Est const fltSemantics &FltSem = DAG.EVTToAPFloatSemantics(VT); APFloat SmallestNorm = APFloat::getSmallestNormalized(FltSem); SDValue NormC = DAG.getConstantFP(SmallestNorm, DL, VT); SDValue FPZero = DAG.getConstantFP(0.0, DL, VT); SDValue Fabs = DAG.getNode(ISD::FABS, DL, VT, Op); SDValue IsDenorm = DAG.getSetCC(DL, CCVT, Fabs, NormC, ISD::SETLT); Est = DAG.getNode(SelOpcode, DL, VT, IsDenorm, FPZero, Est); AddToWorklist(Fabs.getNode()); AddToWorklist(IsDenorm.getNode()); AddToWorklist(Est.getNode()); } else { // X == 0.0 ? 0.0 : Est SDValue FPZero = DAG.getConstantFP(0.0, DL, VT); SDValue IsZero = DAG.getSetCC(DL, CCVT, Op, FPZero, ISD::SETEQ); Est = DAG.getNode(SelOpcode, DL, VT, IsZero, FPZero, Est); AddToWorklist(IsZero.getNode()); AddToWorklist(Est.getNode()); } } } return Est; } return SDValue(); } SDValue DAGCombiner::buildRsqrtEstimate(SDValue Op, SDNodeFlags Flags) { return buildSqrtEstimateImpl(Op, Flags, true); } SDValue DAGCombiner::buildSqrtEstimate(SDValue Op, SDNodeFlags Flags) { return buildSqrtEstimateImpl(Op, Flags, false); } /// Return true if there is any possibility that the two addresses overlap. bool DAGCombiner::isAlias(LSBaseSDNode *Op0, LSBaseSDNode *Op1) const { // If they are the same then they must be aliases. if (Op0->getBasePtr() == Op1->getBasePtr()) return true; // If they are both volatile then they cannot be reordered. if (Op0->isVolatile() && Op1->isVolatile()) return true; // If one operation reads from invariant memory, and the other may store, they // cannot alias. These should really be checking the equivalent of mayWrite, // but it only matters for memory nodes other than load /store. if (Op0->isInvariant() && Op1->writeMem()) return false; if (Op1->isInvariant() && Op0->writeMem()) return false; unsigned NumBytes0 = Op0->getMemoryVT().getStoreSize(); unsigned NumBytes1 = Op1->getMemoryVT().getStoreSize(); // Check for BaseIndexOffset matching. BaseIndexOffset BasePtr0 = BaseIndexOffset::match(Op0, DAG); BaseIndexOffset BasePtr1 = BaseIndexOffset::match(Op1, DAG); int64_t PtrDiff; if (BasePtr0.getBase().getNode() && BasePtr1.getBase().getNode()) { if (BasePtr0.equalBaseIndex(BasePtr1, DAG, PtrDiff)) return !((NumBytes0 <= PtrDiff) || (PtrDiff + NumBytes1 <= 0)); // If both BasePtr0 and BasePtr1 are FrameIndexes, we will not be // able to calculate their relative offset if at least one arises // from an alloca. However, these allocas cannot overlap and we // can infer there is no alias. if (auto *A = dyn_cast(BasePtr0.getBase())) if (auto *B = dyn_cast(BasePtr1.getBase())) { MachineFrameInfo &MFI = DAG.getMachineFunction().getFrameInfo(); // If the base are the same frame index but the we couldn't find a // constant offset, (indices are different) be conservative. if (A != B && (!MFI.isFixedObjectIndex(A->getIndex()) || !MFI.isFixedObjectIndex(B->getIndex()))) return false; } bool IsFI0 = isa(BasePtr0.getBase()); bool IsFI1 = isa(BasePtr1.getBase()); bool IsGV0 = isa(BasePtr0.getBase()); bool IsGV1 = isa(BasePtr1.getBase()); bool IsCV0 = isa(BasePtr0.getBase()); bool IsCV1 = isa(BasePtr1.getBase()); // If of mismatched base types or checkable indices we can check // they do not alias. if ((BasePtr0.getIndex() == BasePtr1.getIndex() || (IsFI0 != IsFI1) || (IsGV0 != IsGV1) || (IsCV0 != IsCV1)) && (IsFI0 || IsGV0 || IsCV0) && (IsFI1 || IsGV1 || IsCV1)) return false; } // If we know required SrcValue1 and SrcValue2 have relatively large // alignment compared to the size and offset of the access, we may be able // to prove they do not alias. This check is conservative for now to catch // cases created by splitting vector types. int64_t SrcValOffset0 = Op0->getSrcValueOffset(); int64_t SrcValOffset1 = Op1->getSrcValueOffset(); unsigned OrigAlignment0 = Op0->getOriginalAlignment(); unsigned OrigAlignment1 = Op1->getOriginalAlignment(); if (OrigAlignment0 == OrigAlignment1 && SrcValOffset0 != SrcValOffset1 && NumBytes0 == NumBytes1 && OrigAlignment0 > NumBytes0) { int64_t OffAlign0 = SrcValOffset0 % OrigAlignment0; int64_t OffAlign1 = SrcValOffset1 % OrigAlignment1; // There is no overlap between these relatively aligned accesses of // similar size. Return no alias. if ((OffAlign0 + NumBytes0) <= OffAlign1 || (OffAlign1 + NumBytes1) <= OffAlign0) return false; } bool UseAA = CombinerGlobalAA.getNumOccurrences() > 0 ? CombinerGlobalAA : DAG.getSubtarget().useAA(); #ifndef NDEBUG if (CombinerAAOnlyFunc.getNumOccurrences() && CombinerAAOnlyFunc != DAG.getMachineFunction().getName()) UseAA = false; #endif if (UseAA && AA && Op0->getMemOperand()->getValue() && Op1->getMemOperand()->getValue()) { // Use alias analysis information. int64_t MinOffset = std::min(SrcValOffset0, SrcValOffset1); int64_t Overlap0 = NumBytes0 + SrcValOffset0 - MinOffset; int64_t Overlap1 = NumBytes1 + SrcValOffset1 - MinOffset; AliasResult AAResult = AA->alias(MemoryLocation(Op0->getMemOperand()->getValue(), Overlap0, UseTBAA ? Op0->getAAInfo() : AAMDNodes()), MemoryLocation(Op1->getMemOperand()->getValue(), Overlap1, UseTBAA ? Op1->getAAInfo() : AAMDNodes()) ); if (AAResult == NoAlias) return false; } // Otherwise we have to assume they alias. return true; } /// Walk up chain skipping non-aliasing memory nodes, /// looking for aliasing nodes and adding them to the Aliases vector. void DAGCombiner::GatherAllAliases(SDNode *N, SDValue OriginalChain, SmallVectorImpl &Aliases) { SmallVector Chains; // List of chains to visit. SmallPtrSet Visited; // Visited node set. // Get alias information for node. bool IsLoad = isa(N) && !cast(N)->isVolatile(); // Starting off. Chains.push_back(OriginalChain); unsigned Depth = 0; // Look at each chain and determine if it is an alias. If so, add it to the // aliases list. If not, then continue up the chain looking for the next // candidate. while (!Chains.empty()) { SDValue Chain = Chains.pop_back_val(); // For TokenFactor nodes, look at each operand and only continue up the // chain until we reach the depth limit. // // FIXME: The depth check could be made to return the last non-aliasing // chain we found before we hit a tokenfactor rather than the original // chain. if (Depth > TLI.getGatherAllAliasesMaxDepth()) { Aliases.clear(); Aliases.push_back(OriginalChain); return; } // Don't bother if we've been before. if (!Visited.insert(Chain.getNode()).second) continue; switch (Chain.getOpcode()) { case ISD::EntryToken: // Entry token is ideal chain operand, but handled in FindBetterChain. break; case ISD::LOAD: case ISD::STORE: { // Get alias information for Chain. bool IsOpLoad = isa(Chain.getNode()) && !cast(Chain.getNode())->isVolatile(); // If chain is alias then stop here. if (!(IsLoad && IsOpLoad) && isAlias(cast(N), cast(Chain.getNode()))) { Aliases.push_back(Chain); } else { // Look further up the chain. Chains.push_back(Chain.getOperand(0)); ++Depth; } break; } case ISD::TokenFactor: // We have to check each of the operands of the token factor for "small" // token factors, so we queue them up. Adding the operands to the queue // (stack) in reverse order maintains the original order and increases the // likelihood that getNode will find a matching token factor (CSE.) if (Chain.getNumOperands() > 16) { Aliases.push_back(Chain); break; } for (unsigned n = Chain.getNumOperands(); n;) Chains.push_back(Chain.getOperand(--n)); ++Depth; break; case ISD::CopyFromReg: // Forward past CopyFromReg. Chains.push_back(Chain.getOperand(0)); ++Depth; break; default: // For all other instructions we will just have to take what we can get. Aliases.push_back(Chain); break; } } } /// Walk up chain skipping non-aliasing memory nodes, looking for a better chain /// (aliasing node.) SDValue DAGCombiner::FindBetterChain(SDNode *N, SDValue OldChain) { if (OptLevel == CodeGenOpt::None) return OldChain; // Ops for replacing token factor. SmallVector Aliases; // Accumulate all the aliases to this node. GatherAllAliases(N, OldChain, Aliases); // If no operands then chain to entry token. if (Aliases.size() == 0) return DAG.getEntryNode(); // If a single operand then chain to it. We don't need to revisit it. if (Aliases.size() == 1) return Aliases[0]; // Construct a custom tailored token factor. return DAG.getNode(ISD::TokenFactor, SDLoc(N), MVT::Other, Aliases); } // TODO: Replace with with std::monostate when we move to C++17. struct UnitT { } Unit; bool operator==(const UnitT &, const UnitT &) { return true; } bool operator!=(const UnitT &, const UnitT &) { return false; } // This function tries to collect a bunch of potentially interesting // nodes to improve the chains of, all at once. This might seem // redundant, as this function gets called when visiting every store // node, so why not let the work be done on each store as it's visited? // // I believe this is mainly important because MergeConsecutiveStores // is unable to deal with merging stores of different sizes, so unless // we improve the chains of all the potential candidates up-front // before running MergeConsecutiveStores, it might only see some of // the nodes that will eventually be candidates, and then not be able // to go from a partially-merged state to the desired final // fully-merged state. bool DAGCombiner::parallelizeChainedStores(StoreSDNode *St) { SmallVector ChainedStores; StoreSDNode *STChain = St; // Intervals records which offsets from BaseIndex have been covered. In // the common case, every store writes to the immediately previous address // space and thus merged with the previous interval at insertion time. using IMap = llvm::IntervalMap>; IMap::Allocator A; IMap Intervals(A); // This holds the base pointer, index, and the offset in bytes from the base // pointer. const BaseIndexOffset BasePtr = BaseIndexOffset::match(St, DAG); // We must have a base and an offset. if (!BasePtr.getBase().getNode()) return false; // Do not handle stores to undef base pointers. if (BasePtr.getBase().isUndef()) return false; // Add ST's interval. Intervals.insert(0, (St->getMemoryVT().getSizeInBits() + 7) / 8, Unit); while (StoreSDNode *Chain = dyn_cast(STChain->getChain())) { // If the chain has more than one use, then we can't reorder the mem ops. if (!SDValue(Chain, 0)->hasOneUse()) break; if (Chain->isVolatile() || Chain->isIndexed()) break; // Find the base pointer and offset for this memory node. const BaseIndexOffset Ptr = BaseIndexOffset::match(Chain, DAG); // Check that the base pointer is the same as the original one. int64_t Offset; if (!BasePtr.equalBaseIndex(Ptr, DAG, Offset)) break; int64_t Length = (Chain->getMemoryVT().getSizeInBits() + 7) / 8; // Make sure we don't overlap with other intervals by checking the ones to // the left or right before inserting. auto I = Intervals.find(Offset); // If there's a next interval, we should end before it. if (I != Intervals.end() && I.start() < (Offset + Length)) break; // If there's a previous interval, we should start after it. if (I != Intervals.begin() && (--I).stop() <= Offset) break; Intervals.insert(Offset, Offset + Length, Unit); ChainedStores.push_back(Chain); STChain = Chain; } // If we didn't find a chained store, exit. if (ChainedStores.size() == 0) return false; // Improve all chained stores (St and ChainedStores members) starting from // where the store chain ended and return single TokenFactor. SDValue NewChain = STChain->getChain(); SmallVector TFOps; for (unsigned I = ChainedStores.size(); I;) { StoreSDNode *S = ChainedStores[--I]; SDValue BetterChain = FindBetterChain(S, NewChain); S = cast(DAG.UpdateNodeOperands( S, BetterChain, S->getOperand(1), S->getOperand(2), S->getOperand(3))); TFOps.push_back(SDValue(S, 0)); ChainedStores[I] = S; } // Improve St's chain. Use a new node to avoid creating a loop from CombineTo. SDValue BetterChain = FindBetterChain(St, NewChain); SDValue NewST; if (St->isTruncatingStore()) NewST = DAG.getTruncStore(BetterChain, SDLoc(St), St->getValue(), St->getBasePtr(), St->getMemoryVT(), St->getMemOperand()); else NewST = DAG.getStore(BetterChain, SDLoc(St), St->getValue(), St->getBasePtr(), St->getMemOperand()); TFOps.push_back(NewST); // If we improved every element of TFOps, then we've lost the dependence on // NewChain to successors of St and we need to add it back to TFOps. Do so at // the beginning to keep relative order consistent with FindBetterChains. auto hasImprovedChain = [&](SDValue ST) -> bool { return ST->getOperand(0) != NewChain; }; bool AddNewChain = llvm::all_of(TFOps, hasImprovedChain); if (AddNewChain) TFOps.insert(TFOps.begin(), NewChain); SDValue TF = DAG.getNode(ISD::TokenFactor, SDLoc(STChain), MVT::Other, TFOps); CombineTo(St, TF); AddToWorklist(STChain); // Add TF operands worklist in reverse order. for (auto I = TF->getNumOperands(); I;) AddToWorklist(TF->getOperand(--I).getNode()); AddToWorklist(TF.getNode()); return true; } bool DAGCombiner::findBetterNeighborChains(StoreSDNode *St) { if (OptLevel == CodeGenOpt::None) return false; const BaseIndexOffset BasePtr = BaseIndexOffset::match(St, DAG); // We must have a base and an offset. if (!BasePtr.getBase().getNode()) return false; // Do not handle stores to undef base pointers. if (BasePtr.getBase().isUndef()) return false; // Directly improve a chain of disjoint stores starting at St. if (parallelizeChainedStores(St)) return true; // Improve St's Chain.. SDValue BetterChain = FindBetterChain(St, St->getChain()); if (St->getChain() != BetterChain) { replaceStoreChain(St, BetterChain); return true; } return false; } /// This is the entry point for the file. void SelectionDAG::Combine(CombineLevel Level, AliasAnalysis *AA, CodeGenOpt::Level OptLevel) { /// This is the main entry point to this class. DAGCombiner(*this, AA, OptLevel).Run(Level); } Index: vendor/llvm/dist-release_80/lib/DebugInfo/DWARF/DWARFDebugLoc.cpp =================================================================== --- vendor/llvm/dist-release_80/lib/DebugInfo/DWARF/DWARFDebugLoc.cpp (revision 343793) +++ vendor/llvm/dist-release_80/lib/DebugInfo/DWARF/DWARFDebugLoc.cpp (revision 343794) @@ -1,278 +1,279 @@ //===- DWARFDebugLoc.cpp --------------------------------------------------===// // // The LLVM Compiler Infrastructure // // This file is distributed under the University of Illinois Open Source // License. See LICENSE.TXT for details. // //===----------------------------------------------------------------------===// #include "llvm/DebugInfo/DWARF/DWARFDebugLoc.h" #include "llvm/ADT/StringRef.h" #include "llvm/BinaryFormat/Dwarf.h" #include "llvm/DebugInfo/DWARF/DWARFContext.h" #include "llvm/DebugInfo/DWARF/DWARFExpression.h" #include "llvm/DebugInfo/DWARF/DWARFRelocMap.h" #include "llvm/DebugInfo/DWARF/DWARFUnit.h" #include "llvm/Support/Compiler.h" #include "llvm/Support/Format.h" #include "llvm/Support/WithColor.h" #include "llvm/Support/raw_ostream.h" #include #include #include using namespace llvm; // When directly dumping the .debug_loc without a compile unit, we have to guess // at the DWARF version. This only affects DW_OP_call_ref, which is a rare // expression that LLVM doesn't produce. Guessing the wrong version means we // won't be able to pretty print expressions in DWARF2 binaries produced by // non-LLVM tools. static void dumpExpression(raw_ostream &OS, ArrayRef Data, bool IsLittleEndian, unsigned AddressSize, const MCRegisterInfo *MRI) { DWARFDataExtractor Extractor(StringRef(Data.data(), Data.size()), IsLittleEndian, AddressSize); DWARFExpression(Extractor, dwarf::DWARF_VERSION, AddressSize).print(OS, MRI); } void DWARFDebugLoc::LocationList::dump(raw_ostream &OS, bool IsLittleEndian, unsigned AddressSize, const MCRegisterInfo *MRI, uint64_t BaseAddress, unsigned Indent) const { for (const Entry &E : Entries) { OS << '\n'; OS.indent(Indent); OS << format("[0x%*.*" PRIx64 ", ", AddressSize * 2, AddressSize * 2, BaseAddress + E.Begin); OS << format(" 0x%*.*" PRIx64 ")", AddressSize * 2, AddressSize * 2, BaseAddress + E.End); OS << ": "; dumpExpression(OS, E.Loc, IsLittleEndian, AddressSize, MRI); } } DWARFDebugLoc::LocationList const * DWARFDebugLoc::getLocationListAtOffset(uint64_t Offset) const { auto It = std::lower_bound( Locations.begin(), Locations.end(), Offset, [](const LocationList &L, uint64_t Offset) { return L.Offset < Offset; }); if (It != Locations.end() && It->Offset == Offset) return &(*It); return nullptr; } void DWARFDebugLoc::dump(raw_ostream &OS, const MCRegisterInfo *MRI, Optional Offset) const { auto DumpLocationList = [&](const LocationList &L) { OS << format("0x%8.8x: ", L.Offset); L.dump(OS, IsLittleEndian, AddressSize, MRI, 0, 12); OS << "\n\n"; }; if (Offset) { if (auto *L = getLocationListAtOffset(*Offset)) DumpLocationList(*L); return; } for (const LocationList &L : Locations) { DumpLocationList(L); } } Optional DWARFDebugLoc::parseOneLocationList(DWARFDataExtractor Data, unsigned *Offset) { LocationList LL; LL.Offset = *Offset; // 2.6.2 Location Lists // A location list entry consists of: while (true) { Entry E; if (!Data.isValidOffsetForDataOfSize(*Offset, 2 * Data.getAddressSize())) { WithColor::error() << "location list overflows the debug_loc section.\n"; return None; } // 1. A beginning address offset. ... E.Begin = Data.getRelocatedAddress(Offset); // 2. An ending address offset. ... E.End = Data.getRelocatedAddress(Offset); // The end of any given location list is marked by an end of list entry, // which consists of a 0 for the beginning address offset and a 0 for the // ending address offset. if (E.Begin == 0 && E.End == 0) return LL; if (!Data.isValidOffsetForDataOfSize(*Offset, 2)) { WithColor::error() << "location list overflows the debug_loc section.\n"; return None; } unsigned Bytes = Data.getU16(Offset); if (!Data.isValidOffsetForDataOfSize(*Offset, Bytes)) { WithColor::error() << "location list overflows the debug_loc section.\n"; return None; } // A single location description describing the location of the object... StringRef str = Data.getData().substr(*Offset, Bytes); *Offset += Bytes; E.Loc.reserve(str.size()); llvm::copy(str, std::back_inserter(E.Loc)); LL.Entries.push_back(std::move(E)); } } void DWARFDebugLoc::parse(const DWARFDataExtractor &data) { IsLittleEndian = data.isLittleEndian(); AddressSize = data.getAddressSize(); uint32_t Offset = 0; while (data.isValidOffset(Offset + data.getAddressSize() - 1)) { if (auto LL = parseOneLocationList(data, &Offset)) Locations.push_back(std::move(*LL)); else break; } if (data.isValidOffset(Offset)) WithColor::error() << "failed to consume entire .debug_loc section\n"; } Optional DWARFDebugLoclists::parseOneLocationList(DataExtractor Data, unsigned *Offset, unsigned Version) { LocationList LL; LL.Offset = *Offset; // dwarf::DW_LLE_end_of_list_entry is 0 and indicates the end of the list. while (auto Kind = static_cast(Data.getU8(Offset))) { Entry E; E.Kind = Kind; switch (Kind) { case dwarf::DW_LLE_startx_length: E.Value0 = Data.getULEB128(Offset); // Pre-DWARF 5 has different interpretation of the length field. We have // to support both pre- and standartized styles for the compatibility. if (Version < 5) E.Value1 = Data.getU32(Offset); else E.Value1 = Data.getULEB128(Offset); break; case dwarf::DW_LLE_start_length: E.Value0 = Data.getAddress(Offset); E.Value1 = Data.getULEB128(Offset); break; case dwarf::DW_LLE_offset_pair: E.Value0 = Data.getULEB128(Offset); E.Value1 = Data.getULEB128(Offset); break; case dwarf::DW_LLE_base_address: E.Value0 = Data.getAddress(Offset); break; default: WithColor::error() << "dumping support for LLE of kind " << (int)Kind << " not implemented\n"; return None; } if (Kind != dwarf::DW_LLE_base_address) { - unsigned Bytes = Data.getU16(Offset); + unsigned Bytes = + Version >= 5 ? Data.getULEB128(Offset) : Data.getU16(Offset); // A single location description describing the location of the object... StringRef str = Data.getData().substr(*Offset, Bytes); *Offset += Bytes; E.Loc.resize(str.size()); llvm::copy(str, E.Loc.begin()); } LL.Entries.push_back(std::move(E)); } return LL; } void DWARFDebugLoclists::parse(DataExtractor data, unsigned Version) { IsLittleEndian = data.isLittleEndian(); AddressSize = data.getAddressSize(); uint32_t Offset = 0; while (data.isValidOffset(Offset)) { if (auto LL = parseOneLocationList(data, &Offset, Version)) Locations.push_back(std::move(*LL)); else return; } } DWARFDebugLoclists::LocationList const * DWARFDebugLoclists::getLocationListAtOffset(uint64_t Offset) const { auto It = std::lower_bound( Locations.begin(), Locations.end(), Offset, [](const LocationList &L, uint64_t Offset) { return L.Offset < Offset; }); if (It != Locations.end() && It->Offset == Offset) return &(*It); return nullptr; } void DWARFDebugLoclists::LocationList::dump(raw_ostream &OS, uint64_t BaseAddr, bool IsLittleEndian, unsigned AddressSize, const MCRegisterInfo *MRI, unsigned Indent) const { for (const Entry &E : Entries) { switch (E.Kind) { case dwarf::DW_LLE_startx_length: OS << '\n'; OS.indent(Indent); OS << "Addr idx " << E.Value0 << " (w/ length " << E.Value1 << "): "; break; case dwarf::DW_LLE_start_length: OS << '\n'; OS.indent(Indent); OS << format("[0x%*.*" PRIx64 ", 0x%*.*" PRIx64 "): ", AddressSize * 2, AddressSize * 2, E.Value0, AddressSize * 2, AddressSize * 2, E.Value0 + E.Value1); break; case dwarf::DW_LLE_offset_pair: OS << '\n'; OS.indent(Indent); OS << format("[0x%*.*" PRIx64 ", 0x%*.*" PRIx64 "): ", AddressSize * 2, AddressSize * 2, BaseAddr + E.Value0, AddressSize * 2, AddressSize * 2, BaseAddr + E.Value1); break; case dwarf::DW_LLE_base_address: BaseAddr = E.Value0; break; default: llvm_unreachable("unreachable locations list kind"); } dumpExpression(OS, E.Loc, IsLittleEndian, AddressSize, MRI); } } void DWARFDebugLoclists::dump(raw_ostream &OS, uint64_t BaseAddr, const MCRegisterInfo *MRI, Optional Offset) const { auto DumpLocationList = [&](const LocationList &L) { OS << format("0x%8.8x: ", L.Offset); L.dump(OS, BaseAddr, IsLittleEndian, AddressSize, MRI, /*Indent=*/12); OS << "\n\n"; }; if (Offset) { if (auto *L = getLocationListAtOffset(*Offset)) DumpLocationList(*L); return; } for (const LocationList &L : Locations) { DumpLocationList(L); } } Index: vendor/llvm/dist-release_80/lib/IR/AutoUpgrade.cpp =================================================================== --- vendor/llvm/dist-release_80/lib/IR/AutoUpgrade.cpp (revision 343793) +++ vendor/llvm/dist-release_80/lib/IR/AutoUpgrade.cpp (revision 343794) @@ -1,3830 +1,3831 @@ //===-- AutoUpgrade.cpp - Implement auto-upgrade helper functions ---------===// // // The LLVM Compiler Infrastructure // // This file is distributed under the University of Illinois Open Source // License. See LICENSE.TXT for details. // //===----------------------------------------------------------------------===// // // This file implements the auto-upgrade helper functions. // This is where deprecated IR intrinsics and other IR features are updated to // current specifications. // //===----------------------------------------------------------------------===// #include "llvm/IR/AutoUpgrade.h" #include "llvm/ADT/StringSwitch.h" #include "llvm/IR/Constants.h" #include "llvm/IR/DIBuilder.h" #include "llvm/IR/DebugInfo.h" #include "llvm/IR/DiagnosticInfo.h" #include "llvm/IR/Function.h" #include "llvm/IR/IRBuilder.h" #include "llvm/IR/Instruction.h" #include "llvm/IR/IntrinsicInst.h" #include "llvm/IR/LLVMContext.h" #include "llvm/IR/Module.h" #include "llvm/IR/Verifier.h" #include "llvm/Support/ErrorHandling.h" #include "llvm/Support/Regex.h" #include using namespace llvm; static void rename(GlobalValue *GV) { GV->setName(GV->getName() + ".old"); } // Upgrade the declarations of the SSE4.1 ptest intrinsics whose arguments have // changed their type from v4f32 to v2i64. static bool UpgradePTESTIntrinsic(Function* F, Intrinsic::ID IID, Function *&NewFn) { // Check whether this is an old version of the function, which received // v4f32 arguments. Type *Arg0Type = F->getFunctionType()->getParamType(0); if (Arg0Type != VectorType::get(Type::getFloatTy(F->getContext()), 4)) return false; // Yes, it's old, replace it with new version. rename(F); NewFn = Intrinsic::getDeclaration(F->getParent(), IID); return true; } // Upgrade the declarations of intrinsic functions whose 8-bit immediate mask // arguments have changed their type from i32 to i8. static bool UpgradeX86IntrinsicsWith8BitMask(Function *F, Intrinsic::ID IID, Function *&NewFn) { // Check that the last argument is an i32. Type *LastArgType = F->getFunctionType()->getParamType( F->getFunctionType()->getNumParams() - 1); if (!LastArgType->isIntegerTy(32)) return false; // Move this function aside and map down. rename(F); NewFn = Intrinsic::getDeclaration(F->getParent(), IID); return true; } static bool ShouldUpgradeX86Intrinsic(Function *F, StringRef Name) { // All of the intrinsics matches below should be marked with which llvm // version started autoupgrading them. At some point in the future we would // like to use this information to remove upgrade code for some older // intrinsics. It is currently undecided how we will determine that future // point. if (Name == "addcarryx.u32" || // Added in 8.0 Name == "addcarryx.u64" || // Added in 8.0 Name == "addcarry.u32" || // Added in 8.0 Name == "addcarry.u64" || // Added in 8.0 Name == "subborrow.u32" || // Added in 8.0 Name == "subborrow.u64" || // Added in 8.0 Name.startswith("sse2.padds.") || // Added in 8.0 Name.startswith("sse2.psubs.") || // Added in 8.0 Name.startswith("sse2.paddus.") || // Added in 8.0 Name.startswith("sse2.psubus.") || // Added in 8.0 Name.startswith("avx2.padds.") || // Added in 8.0 Name.startswith("avx2.psubs.") || // Added in 8.0 Name.startswith("avx2.paddus.") || // Added in 8.0 Name.startswith("avx2.psubus.") || // Added in 8.0 Name.startswith("avx512.padds.") || // Added in 8.0 Name.startswith("avx512.psubs.") || // Added in 8.0 Name.startswith("avx512.mask.padds.") || // Added in 8.0 Name.startswith("avx512.mask.psubs.") || // Added in 8.0 Name.startswith("avx512.mask.paddus.") || // Added in 8.0 Name.startswith("avx512.mask.psubus.") || // Added in 8.0 Name=="ssse3.pabs.b.128" || // Added in 6.0 Name=="ssse3.pabs.w.128" || // Added in 6.0 Name=="ssse3.pabs.d.128" || // Added in 6.0 Name.startswith("fma4.vfmadd.s") || // Added in 7.0 Name.startswith("fma.vfmadd.") || // Added in 7.0 Name.startswith("fma.vfmsub.") || // Added in 7.0 Name.startswith("fma.vfmaddsub.") || // Added in 7.0 Name.startswith("fma.vfmsubadd.") || // Added in 7.0 Name.startswith("fma.vfnmadd.") || // Added in 7.0 Name.startswith("fma.vfnmsub.") || // Added in 7.0 Name.startswith("avx512.mask.vfmadd.") || // Added in 7.0 Name.startswith("avx512.mask.vfnmadd.") || // Added in 7.0 Name.startswith("avx512.mask.vfnmsub.") || // Added in 7.0 Name.startswith("avx512.mask3.vfmadd.") || // Added in 7.0 Name.startswith("avx512.maskz.vfmadd.") || // Added in 7.0 Name.startswith("avx512.mask3.vfmsub.") || // Added in 7.0 Name.startswith("avx512.mask3.vfnmsub.") || // Added in 7.0 Name.startswith("avx512.mask.vfmaddsub.") || // Added in 7.0 Name.startswith("avx512.maskz.vfmaddsub.") || // Added in 7.0 Name.startswith("avx512.mask3.vfmaddsub.") || // Added in 7.0 Name.startswith("avx512.mask3.vfmsubadd.") || // Added in 7.0 Name.startswith("avx512.mask.shuf.i") || // Added in 6.0 Name.startswith("avx512.mask.shuf.f") || // Added in 6.0 Name.startswith("avx512.kunpck") || //added in 6.0 Name.startswith("avx2.pabs.") || // Added in 6.0 Name.startswith("avx512.mask.pabs.") || // Added in 6.0 Name.startswith("avx512.broadcastm") || // Added in 6.0 Name == "sse.sqrt.ss" || // Added in 7.0 Name == "sse2.sqrt.sd" || // Added in 7.0 Name.startswith("avx512.mask.sqrt.p") || // Added in 7.0 Name.startswith("avx.sqrt.p") || // Added in 7.0 Name.startswith("sse2.sqrt.p") || // Added in 7.0 Name.startswith("sse.sqrt.p") || // Added in 7.0 Name.startswith("avx512.mask.pbroadcast") || // Added in 6.0 Name.startswith("sse2.pcmpeq.") || // Added in 3.1 Name.startswith("sse2.pcmpgt.") || // Added in 3.1 Name.startswith("avx2.pcmpeq.") || // Added in 3.1 Name.startswith("avx2.pcmpgt.") || // Added in 3.1 Name.startswith("avx512.mask.pcmpeq.") || // Added in 3.9 Name.startswith("avx512.mask.pcmpgt.") || // Added in 3.9 Name.startswith("avx.vperm2f128.") || // Added in 6.0 Name == "avx2.vperm2i128" || // Added in 6.0 Name == "sse.add.ss" || // Added in 4.0 Name == "sse2.add.sd" || // Added in 4.0 Name == "sse.sub.ss" || // Added in 4.0 Name == "sse2.sub.sd" || // Added in 4.0 Name == "sse.mul.ss" || // Added in 4.0 Name == "sse2.mul.sd" || // Added in 4.0 Name == "sse.div.ss" || // Added in 4.0 Name == "sse2.div.sd" || // Added in 4.0 Name == "sse41.pmaxsb" || // Added in 3.9 Name == "sse2.pmaxs.w" || // Added in 3.9 Name == "sse41.pmaxsd" || // Added in 3.9 Name == "sse2.pmaxu.b" || // Added in 3.9 Name == "sse41.pmaxuw" || // Added in 3.9 Name == "sse41.pmaxud" || // Added in 3.9 Name == "sse41.pminsb" || // Added in 3.9 Name == "sse2.pmins.w" || // Added in 3.9 Name == "sse41.pminsd" || // Added in 3.9 Name == "sse2.pminu.b" || // Added in 3.9 Name == "sse41.pminuw" || // Added in 3.9 Name == "sse41.pminud" || // Added in 3.9 Name == "avx512.kand.w" || // Added in 7.0 Name == "avx512.kandn.w" || // Added in 7.0 Name == "avx512.knot.w" || // Added in 7.0 Name == "avx512.kor.w" || // Added in 7.0 Name == "avx512.kxor.w" || // Added in 7.0 Name == "avx512.kxnor.w" || // Added in 7.0 Name == "avx512.kortestc.w" || // Added in 7.0 Name == "avx512.kortestz.w" || // Added in 7.0 Name.startswith("avx512.mask.pshuf.b.") || // Added in 4.0 Name.startswith("avx2.pmax") || // Added in 3.9 Name.startswith("avx2.pmin") || // Added in 3.9 Name.startswith("avx512.mask.pmax") || // Added in 4.0 Name.startswith("avx512.mask.pmin") || // Added in 4.0 Name.startswith("avx2.vbroadcast") || // Added in 3.8 Name.startswith("avx2.pbroadcast") || // Added in 3.8 Name.startswith("avx.vpermil.") || // Added in 3.1 Name.startswith("sse2.pshuf") || // Added in 3.9 Name.startswith("avx512.pbroadcast") || // Added in 3.9 Name.startswith("avx512.mask.broadcast.s") || // Added in 3.9 Name.startswith("avx512.mask.movddup") || // Added in 3.9 Name.startswith("avx512.mask.movshdup") || // Added in 3.9 Name.startswith("avx512.mask.movsldup") || // Added in 3.9 Name.startswith("avx512.mask.pshuf.d.") || // Added in 3.9 Name.startswith("avx512.mask.pshufl.w.") || // Added in 3.9 Name.startswith("avx512.mask.pshufh.w.") || // Added in 3.9 Name.startswith("avx512.mask.shuf.p") || // Added in 4.0 Name.startswith("avx512.mask.vpermil.p") || // Added in 3.9 Name.startswith("avx512.mask.perm.df.") || // Added in 3.9 Name.startswith("avx512.mask.perm.di.") || // Added in 3.9 Name.startswith("avx512.mask.punpckl") || // Added in 3.9 Name.startswith("avx512.mask.punpckh") || // Added in 3.9 Name.startswith("avx512.mask.unpckl.") || // Added in 3.9 Name.startswith("avx512.mask.unpckh.") || // Added in 3.9 Name.startswith("avx512.mask.pand.") || // Added in 3.9 Name.startswith("avx512.mask.pandn.") || // Added in 3.9 Name.startswith("avx512.mask.por.") || // Added in 3.9 Name.startswith("avx512.mask.pxor.") || // Added in 3.9 Name.startswith("avx512.mask.and.") || // Added in 3.9 Name.startswith("avx512.mask.andn.") || // Added in 3.9 Name.startswith("avx512.mask.or.") || // Added in 3.9 Name.startswith("avx512.mask.xor.") || // Added in 3.9 Name.startswith("avx512.mask.padd.") || // Added in 4.0 Name.startswith("avx512.mask.psub.") || // Added in 4.0 Name.startswith("avx512.mask.pmull.") || // Added in 4.0 Name.startswith("avx512.mask.cvtdq2pd.") || // Added in 4.0 Name.startswith("avx512.mask.cvtudq2pd.") || // Added in 4.0 Name == "avx512.mask.cvtudq2ps.128" || // Added in 7.0 Name == "avx512.mask.cvtudq2ps.256" || // Added in 7.0 Name == "avx512.mask.cvtqq2pd.128" || // Added in 7.0 Name == "avx512.mask.cvtqq2pd.256" || // Added in 7.0 Name == "avx512.mask.cvtuqq2pd.128" || // Added in 7.0 Name == "avx512.mask.cvtuqq2pd.256" || // Added in 7.0 Name == "avx512.mask.cvtdq2ps.128" || // Added in 7.0 Name == "avx512.mask.cvtdq2ps.256" || // Added in 7.0 Name == "avx512.mask.cvtpd2dq.256" || // Added in 7.0 Name == "avx512.mask.cvtpd2ps.256" || // Added in 7.0 Name == "avx512.mask.cvttpd2dq.256" || // Added in 7.0 Name == "avx512.mask.cvttps2dq.128" || // Added in 7.0 Name == "avx512.mask.cvttps2dq.256" || // Added in 7.0 Name == "avx512.mask.cvtps2pd.128" || // Added in 7.0 Name == "avx512.mask.cvtps2pd.256" || // Added in 7.0 Name == "avx512.cvtusi2sd" || // Added in 7.0 Name.startswith("avx512.mask.permvar.") || // Added in 7.0 Name.startswith("avx512.mask.permvar.") || // Added in 7.0 Name == "sse2.pmulu.dq" || // Added in 7.0 Name == "sse41.pmuldq" || // Added in 7.0 Name == "avx2.pmulu.dq" || // Added in 7.0 Name == "avx2.pmul.dq" || // Added in 7.0 Name == "avx512.pmulu.dq.512" || // Added in 7.0 Name == "avx512.pmul.dq.512" || // Added in 7.0 Name.startswith("avx512.mask.pmul.dq.") || // Added in 4.0 Name.startswith("avx512.mask.pmulu.dq.") || // Added in 4.0 Name.startswith("avx512.mask.pmul.hr.sw.") || // Added in 7.0 Name.startswith("avx512.mask.pmulh.w.") || // Added in 7.0 Name.startswith("avx512.mask.pmulhu.w.") || // Added in 7.0 Name.startswith("avx512.mask.pmaddw.d.") || // Added in 7.0 Name.startswith("avx512.mask.pmaddubs.w.") || // Added in 7.0 Name.startswith("avx512.mask.packsswb.") || // Added in 5.0 Name.startswith("avx512.mask.packssdw.") || // Added in 5.0 Name.startswith("avx512.mask.packuswb.") || // Added in 5.0 Name.startswith("avx512.mask.packusdw.") || // Added in 5.0 Name.startswith("avx512.mask.cmp.b") || // Added in 5.0 Name.startswith("avx512.mask.cmp.d") || // Added in 5.0 Name.startswith("avx512.mask.cmp.q") || // Added in 5.0 Name.startswith("avx512.mask.cmp.w") || // Added in 5.0 Name.startswith("avx512.mask.cmp.p") || // Added in 7.0 Name.startswith("avx512.mask.ucmp.") || // Added in 5.0 Name.startswith("avx512.cvtb2mask.") || // Added in 7.0 Name.startswith("avx512.cvtw2mask.") || // Added in 7.0 Name.startswith("avx512.cvtd2mask.") || // Added in 7.0 Name.startswith("avx512.cvtq2mask.") || // Added in 7.0 Name.startswith("avx512.mask.vpermilvar.") || // Added in 4.0 Name.startswith("avx512.mask.psll.d") || // Added in 4.0 Name.startswith("avx512.mask.psll.q") || // Added in 4.0 Name.startswith("avx512.mask.psll.w") || // Added in 4.0 Name.startswith("avx512.mask.psra.d") || // Added in 4.0 Name.startswith("avx512.mask.psra.q") || // Added in 4.0 Name.startswith("avx512.mask.psra.w") || // Added in 4.0 Name.startswith("avx512.mask.psrl.d") || // Added in 4.0 Name.startswith("avx512.mask.psrl.q") || // Added in 4.0 Name.startswith("avx512.mask.psrl.w") || // Added in 4.0 Name.startswith("avx512.mask.pslli") || // Added in 4.0 Name.startswith("avx512.mask.psrai") || // Added in 4.0 Name.startswith("avx512.mask.psrli") || // Added in 4.0 Name.startswith("avx512.mask.psllv") || // Added in 4.0 Name.startswith("avx512.mask.psrav") || // Added in 4.0 Name.startswith("avx512.mask.psrlv") || // Added in 4.0 Name.startswith("sse41.pmovsx") || // Added in 3.8 Name.startswith("sse41.pmovzx") || // Added in 3.9 Name.startswith("avx2.pmovsx") || // Added in 3.9 Name.startswith("avx2.pmovzx") || // Added in 3.9 Name.startswith("avx512.mask.pmovsx") || // Added in 4.0 Name.startswith("avx512.mask.pmovzx") || // Added in 4.0 Name.startswith("avx512.mask.lzcnt.") || // Added in 5.0 Name.startswith("avx512.mask.pternlog.") || // Added in 7.0 Name.startswith("avx512.maskz.pternlog.") || // Added in 7.0 Name.startswith("avx512.mask.vpmadd52") || // Added in 7.0 Name.startswith("avx512.maskz.vpmadd52") || // Added in 7.0 Name.startswith("avx512.mask.vpermi2var.") || // Added in 7.0 Name.startswith("avx512.mask.vpermt2var.") || // Added in 7.0 Name.startswith("avx512.maskz.vpermt2var.") || // Added in 7.0 Name.startswith("avx512.mask.vpdpbusd.") || // Added in 7.0 Name.startswith("avx512.maskz.vpdpbusd.") || // Added in 7.0 Name.startswith("avx512.mask.vpdpbusds.") || // Added in 7.0 Name.startswith("avx512.maskz.vpdpbusds.") || // Added in 7.0 Name.startswith("avx512.mask.vpdpwssd.") || // Added in 7.0 Name.startswith("avx512.maskz.vpdpwssd.") || // Added in 7.0 Name.startswith("avx512.mask.vpdpwssds.") || // Added in 7.0 Name.startswith("avx512.maskz.vpdpwssds.") || // Added in 7.0 Name.startswith("avx512.mask.dbpsadbw.") || // Added in 7.0 Name.startswith("avx512.mask.vpshld.") || // Added in 7.0 Name.startswith("avx512.mask.vpshrd.") || // Added in 7.0 Name.startswith("avx512.mask.vpshldv.") || // Added in 8.0 Name.startswith("avx512.mask.vpshrdv.") || // Added in 8.0 Name.startswith("avx512.maskz.vpshldv.") || // Added in 8.0 Name.startswith("avx512.maskz.vpshrdv.") || // Added in 8.0 Name.startswith("avx512.vpshld.") || // Added in 8.0 Name.startswith("avx512.vpshrd.") || // Added in 8.0 Name.startswith("avx512.mask.add.p") || // Added in 7.0. 128/256 in 4.0 Name.startswith("avx512.mask.sub.p") || // Added in 7.0. 128/256 in 4.0 Name.startswith("avx512.mask.mul.p") || // Added in 7.0. 128/256 in 4.0 Name.startswith("avx512.mask.div.p") || // Added in 7.0. 128/256 in 4.0 Name.startswith("avx512.mask.max.p") || // Added in 7.0. 128/256 in 5.0 Name.startswith("avx512.mask.min.p") || // Added in 7.0. 128/256 in 5.0 Name.startswith("avx512.mask.fpclass.p") || // Added in 7.0 Name.startswith("avx512.mask.vpshufbitqmb.") || // Added in 8.0 Name.startswith("avx512.mask.pmultishift.qb.") || // Added in 8.0 Name == "sse.cvtsi2ss" || // Added in 7.0 Name == "sse.cvtsi642ss" || // Added in 7.0 Name == "sse2.cvtsi2sd" || // Added in 7.0 Name == "sse2.cvtsi642sd" || // Added in 7.0 Name == "sse2.cvtss2sd" || // Added in 7.0 Name == "sse2.cvtdq2pd" || // Added in 3.9 Name == "sse2.cvtdq2ps" || // Added in 7.0 Name == "sse2.cvtps2pd" || // Added in 3.9 Name == "avx.cvtdq2.pd.256" || // Added in 3.9 Name == "avx.cvtdq2.ps.256" || // Added in 7.0 Name == "avx.cvt.ps2.pd.256" || // Added in 3.9 Name.startswith("avx.vinsertf128.") || // Added in 3.7 Name == "avx2.vinserti128" || // Added in 3.7 Name.startswith("avx512.mask.insert") || // Added in 4.0 Name.startswith("avx.vextractf128.") || // Added in 3.7 Name == "avx2.vextracti128" || // Added in 3.7 Name.startswith("avx512.mask.vextract") || // Added in 4.0 Name.startswith("sse4a.movnt.") || // Added in 3.9 Name.startswith("avx.movnt.") || // Added in 3.2 Name.startswith("avx512.storent.") || // Added in 3.9 Name == "sse41.movntdqa" || // Added in 5.0 Name == "avx2.movntdqa" || // Added in 5.0 Name == "avx512.movntdqa" || // Added in 5.0 Name == "sse2.storel.dq" || // Added in 3.9 Name.startswith("sse.storeu.") || // Added in 3.9 Name.startswith("sse2.storeu.") || // Added in 3.9 Name.startswith("avx.storeu.") || // Added in 3.9 Name.startswith("avx512.mask.storeu.") || // Added in 3.9 Name.startswith("avx512.mask.store.p") || // Added in 3.9 Name.startswith("avx512.mask.store.b.") || // Added in 3.9 Name.startswith("avx512.mask.store.w.") || // Added in 3.9 Name.startswith("avx512.mask.store.d.") || // Added in 3.9 Name.startswith("avx512.mask.store.q.") || // Added in 3.9 Name == "avx512.mask.store.ss" || // Added in 7.0 Name.startswith("avx512.mask.loadu.") || // Added in 3.9 Name.startswith("avx512.mask.load.") || // Added in 3.9 Name.startswith("avx512.mask.expand.load.") || // Added in 7.0 Name.startswith("avx512.mask.compress.store.") || // Added in 7.0 Name == "sse42.crc32.64.8" || // Added in 3.4 Name.startswith("avx.vbroadcast.s") || // Added in 3.5 Name.startswith("avx512.vbroadcast.s") || // Added in 7.0 Name.startswith("avx512.mask.palignr.") || // Added in 3.9 Name.startswith("avx512.mask.valign.") || // Added in 4.0 Name.startswith("sse2.psll.dq") || // Added in 3.7 Name.startswith("sse2.psrl.dq") || // Added in 3.7 Name.startswith("avx2.psll.dq") || // Added in 3.7 Name.startswith("avx2.psrl.dq") || // Added in 3.7 Name.startswith("avx512.psll.dq") || // Added in 3.9 Name.startswith("avx512.psrl.dq") || // Added in 3.9 Name == "sse41.pblendw" || // Added in 3.7 Name.startswith("sse41.blendp") || // Added in 3.7 Name.startswith("avx.blend.p") || // Added in 3.7 Name == "avx2.pblendw" || // Added in 3.7 Name.startswith("avx2.pblendd.") || // Added in 3.7 Name.startswith("avx.vbroadcastf128") || // Added in 4.0 Name == "avx2.vbroadcasti128" || // Added in 3.7 Name.startswith("avx512.mask.broadcastf") || // Added in 6.0 Name.startswith("avx512.mask.broadcasti") || // Added in 6.0 Name == "xop.vpcmov" || // Added in 3.8 Name == "xop.vpcmov.256" || // Added in 5.0 Name.startswith("avx512.mask.move.s") || // Added in 4.0 Name.startswith("avx512.cvtmask2") || // Added in 5.0 (Name.startswith("xop.vpcom") && // Added in 3.2 F->arg_size() == 2) || Name.startswith("xop.vprot") || // Added in 8.0 Name.startswith("avx512.prol") || // Added in 8.0 Name.startswith("avx512.pror") || // Added in 8.0 Name.startswith("avx512.mask.prorv.") || // Added in 8.0 Name.startswith("avx512.mask.pror.") || // Added in 8.0 Name.startswith("avx512.mask.prolv.") || // Added in 8.0 Name.startswith("avx512.mask.prol.") || // Added in 8.0 Name.startswith("avx512.ptestm") || //Added in 6.0 Name.startswith("avx512.ptestnm") || //Added in 6.0 Name.startswith("sse2.pavg") || // Added in 6.0 Name.startswith("avx2.pavg") || // Added in 6.0 Name.startswith("avx512.mask.pavg")) // Added in 6.0 return true; return false; } static bool UpgradeX86IntrinsicFunction(Function *F, StringRef Name, Function *&NewFn) { // Only handle intrinsics that start with "x86.". if (!Name.startswith("x86.")) return false; // Remove "x86." prefix. Name = Name.substr(4); if (ShouldUpgradeX86Intrinsic(F, Name)) { NewFn = nullptr; return true; } if (Name == "rdtscp") { // Added in 8.0 // If this intrinsic has 0 operands, it's the new version. if (F->getFunctionType()->getNumParams() == 0) return false; rename(F); NewFn = Intrinsic::getDeclaration(F->getParent(), Intrinsic::x86_rdtscp); return true; } // SSE4.1 ptest functions may have an old signature. if (Name.startswith("sse41.ptest")) { // Added in 3.2 if (Name.substr(11) == "c") return UpgradePTESTIntrinsic(F, Intrinsic::x86_sse41_ptestc, NewFn); if (Name.substr(11) == "z") return UpgradePTESTIntrinsic(F, Intrinsic::x86_sse41_ptestz, NewFn); if (Name.substr(11) == "nzc") return UpgradePTESTIntrinsic(F, Intrinsic::x86_sse41_ptestnzc, NewFn); } // Several blend and other instructions with masks used the wrong number of // bits. if (Name == "sse41.insertps") // Added in 3.6 return UpgradeX86IntrinsicsWith8BitMask(F, Intrinsic::x86_sse41_insertps, NewFn); if (Name == "sse41.dppd") // Added in 3.6 return UpgradeX86IntrinsicsWith8BitMask(F, Intrinsic::x86_sse41_dppd, NewFn); if (Name == "sse41.dpps") // Added in 3.6 return UpgradeX86IntrinsicsWith8BitMask(F, Intrinsic::x86_sse41_dpps, NewFn); if (Name == "sse41.mpsadbw") // Added in 3.6 return UpgradeX86IntrinsicsWith8BitMask(F, Intrinsic::x86_sse41_mpsadbw, NewFn); if (Name == "avx.dp.ps.256") // Added in 3.6 return UpgradeX86IntrinsicsWith8BitMask(F, Intrinsic::x86_avx_dp_ps_256, NewFn); if (Name == "avx2.mpsadbw") // Added in 3.6 return UpgradeX86IntrinsicsWith8BitMask(F, Intrinsic::x86_avx2_mpsadbw, NewFn); // frcz.ss/sd may need to have an argument dropped. Added in 3.2 if (Name.startswith("xop.vfrcz.ss") && F->arg_size() == 2) { rename(F); NewFn = Intrinsic::getDeclaration(F->getParent(), Intrinsic::x86_xop_vfrcz_ss); return true; } if (Name.startswith("xop.vfrcz.sd") && F->arg_size() == 2) { rename(F); NewFn = Intrinsic::getDeclaration(F->getParent(), Intrinsic::x86_xop_vfrcz_sd); return true; } // Upgrade any XOP PERMIL2 index operand still using a float/double vector. if (Name.startswith("xop.vpermil2")) { // Added in 3.9 auto Idx = F->getFunctionType()->getParamType(2); if (Idx->isFPOrFPVectorTy()) { rename(F); unsigned IdxSize = Idx->getPrimitiveSizeInBits(); unsigned EltSize = Idx->getScalarSizeInBits(); Intrinsic::ID Permil2ID; if (EltSize == 64 && IdxSize == 128) Permil2ID = Intrinsic::x86_xop_vpermil2pd; else if (EltSize == 32 && IdxSize == 128) Permil2ID = Intrinsic::x86_xop_vpermil2ps; else if (EltSize == 64 && IdxSize == 256) Permil2ID = Intrinsic::x86_xop_vpermil2pd_256; else Permil2ID = Intrinsic::x86_xop_vpermil2ps_256; NewFn = Intrinsic::getDeclaration(F->getParent(), Permil2ID); return true; } } + if (Name == "seh.recoverfp") { + NewFn = Intrinsic::getDeclaration(F->getParent(), Intrinsic::eh_recoverfp); + return true; + } + return false; } static bool UpgradeIntrinsicFunction1(Function *F, Function *&NewFn) { assert(F && "Illegal to upgrade a non-existent Function."); // Quickly eliminate it, if it's not a candidate. StringRef Name = F->getName(); if (Name.size() <= 8 || !Name.startswith("llvm.")) return false; Name = Name.substr(5); // Strip off "llvm." switch (Name[0]) { default: break; case 'a': { if (Name.startswith("arm.rbit") || Name.startswith("aarch64.rbit")) { NewFn = Intrinsic::getDeclaration(F->getParent(), Intrinsic::bitreverse, F->arg_begin()->getType()); return true; } if (Name.startswith("arm.neon.vclz")) { Type* args[2] = { F->arg_begin()->getType(), Type::getInt1Ty(F->getContext()) }; // Can't use Intrinsic::getDeclaration here as it adds a ".i1" to // the end of the name. Change name from llvm.arm.neon.vclz.* to // llvm.ctlz.* FunctionType* fType = FunctionType::get(F->getReturnType(), args, false); NewFn = Function::Create(fType, F->getLinkage(), F->getAddressSpace(), "llvm.ctlz." + Name.substr(14), F->getParent()); return true; } if (Name.startswith("arm.neon.vcnt")) { NewFn = Intrinsic::getDeclaration(F->getParent(), Intrinsic::ctpop, F->arg_begin()->getType()); return true; } Regex vldRegex("^arm\\.neon\\.vld([1234]|[234]lane)\\.v[a-z0-9]*$"); if (vldRegex.match(Name)) { auto fArgs = F->getFunctionType()->params(); SmallVector Tys(fArgs.begin(), fArgs.end()); // Can't use Intrinsic::getDeclaration here as the return types might // then only be structurally equal. FunctionType* fType = FunctionType::get(F->getReturnType(), Tys, false); NewFn = Function::Create(fType, F->getLinkage(), F->getAddressSpace(), "llvm." + Name + ".p0i8", F->getParent()); return true; } Regex vstRegex("^arm\\.neon\\.vst([1234]|[234]lane)\\.v[a-z0-9]*$"); if (vstRegex.match(Name)) { static const Intrinsic::ID StoreInts[] = {Intrinsic::arm_neon_vst1, Intrinsic::arm_neon_vst2, Intrinsic::arm_neon_vst3, Intrinsic::arm_neon_vst4}; static const Intrinsic::ID StoreLaneInts[] = { Intrinsic::arm_neon_vst2lane, Intrinsic::arm_neon_vst3lane, Intrinsic::arm_neon_vst4lane }; auto fArgs = F->getFunctionType()->params(); Type *Tys[] = {fArgs[0], fArgs[1]}; if (Name.find("lane") == StringRef::npos) NewFn = Intrinsic::getDeclaration(F->getParent(), StoreInts[fArgs.size() - 3], Tys); else NewFn = Intrinsic::getDeclaration(F->getParent(), StoreLaneInts[fArgs.size() - 5], Tys); return true; } if (Name == "aarch64.thread.pointer" || Name == "arm.thread.pointer") { NewFn = Intrinsic::getDeclaration(F->getParent(), Intrinsic::thread_pointer); - return true; - } - if (Name == "x86.seh.recoverfp") { - NewFn = Intrinsic::getDeclaration(F->getParent(), Intrinsic::eh_recoverfp); return true; } break; } case 'c': { if (Name.startswith("ctlz.") && F->arg_size() == 1) { rename(F); NewFn = Intrinsic::getDeclaration(F->getParent(), Intrinsic::ctlz, F->arg_begin()->getType()); return true; } if (Name.startswith("cttz.") && F->arg_size() == 1) { rename(F); NewFn = Intrinsic::getDeclaration(F->getParent(), Intrinsic::cttz, F->arg_begin()->getType()); return true; } break; } case 'd': { if (Name == "dbg.value" && F->arg_size() == 4) { rename(F); NewFn = Intrinsic::getDeclaration(F->getParent(), Intrinsic::dbg_value); return true; } break; } case 'i': case 'l': { bool IsLifetimeStart = Name.startswith("lifetime.start"); if (IsLifetimeStart || Name.startswith("invariant.start")) { Intrinsic::ID ID = IsLifetimeStart ? Intrinsic::lifetime_start : Intrinsic::invariant_start; auto Args = F->getFunctionType()->params(); Type* ObjectPtr[1] = {Args[1]}; if (F->getName() != Intrinsic::getName(ID, ObjectPtr)) { rename(F); NewFn = Intrinsic::getDeclaration(F->getParent(), ID, ObjectPtr); return true; } } bool IsLifetimeEnd = Name.startswith("lifetime.end"); if (IsLifetimeEnd || Name.startswith("invariant.end")) { Intrinsic::ID ID = IsLifetimeEnd ? Intrinsic::lifetime_end : Intrinsic::invariant_end; auto Args = F->getFunctionType()->params(); Type* ObjectPtr[1] = {Args[IsLifetimeEnd ? 1 : 2]}; if (F->getName() != Intrinsic::getName(ID, ObjectPtr)) { rename(F); NewFn = Intrinsic::getDeclaration(F->getParent(), ID, ObjectPtr); return true; } } if (Name.startswith("invariant.group.barrier")) { // Rename invariant.group.barrier to launder.invariant.group auto Args = F->getFunctionType()->params(); Type* ObjectPtr[1] = {Args[0]}; rename(F); NewFn = Intrinsic::getDeclaration(F->getParent(), Intrinsic::launder_invariant_group, ObjectPtr); return true; } break; } case 'm': { if (Name.startswith("masked.load.")) { Type *Tys[] = { F->getReturnType(), F->arg_begin()->getType() }; if (F->getName() != Intrinsic::getName(Intrinsic::masked_load, Tys)) { rename(F); NewFn = Intrinsic::getDeclaration(F->getParent(), Intrinsic::masked_load, Tys); return true; } } if (Name.startswith("masked.store.")) { auto Args = F->getFunctionType()->params(); Type *Tys[] = { Args[0], Args[1] }; if (F->getName() != Intrinsic::getName(Intrinsic::masked_store, Tys)) { rename(F); NewFn = Intrinsic::getDeclaration(F->getParent(), Intrinsic::masked_store, Tys); return true; } } // Renaming gather/scatter intrinsics with no address space overloading // to the new overload which includes an address space if (Name.startswith("masked.gather.")) { Type *Tys[] = {F->getReturnType(), F->arg_begin()->getType()}; if (F->getName() != Intrinsic::getName(Intrinsic::masked_gather, Tys)) { rename(F); NewFn = Intrinsic::getDeclaration(F->getParent(), Intrinsic::masked_gather, Tys); return true; } } if (Name.startswith("masked.scatter.")) { auto Args = F->getFunctionType()->params(); Type *Tys[] = {Args[0], Args[1]}; if (F->getName() != Intrinsic::getName(Intrinsic::masked_scatter, Tys)) { rename(F); NewFn = Intrinsic::getDeclaration(F->getParent(), Intrinsic::masked_scatter, Tys); return true; } } // Updating the memory intrinsics (memcpy/memmove/memset) that have an // alignment parameter to embedding the alignment as an attribute of // the pointer args. if (Name.startswith("memcpy.") && F->arg_size() == 5) { rename(F); // Get the types of dest, src, and len ArrayRef ParamTypes = F->getFunctionType()->params().slice(0, 3); NewFn = Intrinsic::getDeclaration(F->getParent(), Intrinsic::memcpy, ParamTypes); return true; } if (Name.startswith("memmove.") && F->arg_size() == 5) { rename(F); // Get the types of dest, src, and len ArrayRef ParamTypes = F->getFunctionType()->params().slice(0, 3); NewFn = Intrinsic::getDeclaration(F->getParent(), Intrinsic::memmove, ParamTypes); return true; } if (Name.startswith("memset.") && F->arg_size() == 5) { rename(F); // Get the types of dest, and len const auto *FT = F->getFunctionType(); Type *ParamTypes[2] = { FT->getParamType(0), // Dest FT->getParamType(2) // len }; NewFn = Intrinsic::getDeclaration(F->getParent(), Intrinsic::memset, ParamTypes); return true; } break; } case 'n': { if (Name.startswith("nvvm.")) { Name = Name.substr(5); // The following nvvm intrinsics correspond exactly to an LLVM intrinsic. Intrinsic::ID IID = StringSwitch(Name) .Cases("brev32", "brev64", Intrinsic::bitreverse) .Case("clz.i", Intrinsic::ctlz) .Case("popc.i", Intrinsic::ctpop) .Default(Intrinsic::not_intrinsic); if (IID != Intrinsic::not_intrinsic && F->arg_size() == 1) { NewFn = Intrinsic::getDeclaration(F->getParent(), IID, {F->getReturnType()}); return true; } // The following nvvm intrinsics correspond exactly to an LLVM idiom, but // not to an intrinsic alone. We expand them in UpgradeIntrinsicCall. // // TODO: We could add lohi.i2d. bool Expand = StringSwitch(Name) .Cases("abs.i", "abs.ll", true) .Cases("clz.ll", "popc.ll", "h2f", true) .Cases("max.i", "max.ll", "max.ui", "max.ull", true) .Cases("min.i", "min.ll", "min.ui", "min.ull", true) .Default(false); if (Expand) { NewFn = nullptr; return true; } } break; } case 'o': // We only need to change the name to match the mangling including the // address space. if (Name.startswith("objectsize.")) { Type *Tys[2] = { F->getReturnType(), F->arg_begin()->getType() }; if (F->arg_size() == 2 || F->getName() != Intrinsic::getName(Intrinsic::objectsize, Tys)) { rename(F); NewFn = Intrinsic::getDeclaration(F->getParent(), Intrinsic::objectsize, Tys); return true; } } break; case 's': if (Name == "stackprotectorcheck") { NewFn = nullptr; return true; } break; case 'x': if (UpgradeX86IntrinsicFunction(F, Name, NewFn)) return true; } // Remangle our intrinsic since we upgrade the mangling auto Result = llvm::Intrinsic::remangleIntrinsicFunction(F); if (Result != None) { NewFn = Result.getValue(); return true; } // This may not belong here. This function is effectively being overloaded // to both detect an intrinsic which needs upgrading, and to provide the // upgraded form of the intrinsic. We should perhaps have two separate // functions for this. return false; } bool llvm::UpgradeIntrinsicFunction(Function *F, Function *&NewFn) { NewFn = nullptr; bool Upgraded = UpgradeIntrinsicFunction1(F, NewFn); assert(F != NewFn && "Intrinsic function upgraded to the same function"); // Upgrade intrinsic attributes. This does not change the function. if (NewFn) F = NewFn; if (Intrinsic::ID id = F->getIntrinsicID()) F->setAttributes(Intrinsic::getAttributes(F->getContext(), id)); return Upgraded; } bool llvm::UpgradeGlobalVariable(GlobalVariable *GV) { // Nothing to do yet. return false; } // Handles upgrading SSE2/AVX2/AVX512BW PSLLDQ intrinsics by converting them // to byte shuffles. static Value *UpgradeX86PSLLDQIntrinsics(IRBuilder<> &Builder, Value *Op, unsigned Shift) { Type *ResultTy = Op->getType(); unsigned NumElts = ResultTy->getVectorNumElements() * 8; // Bitcast from a 64-bit element type to a byte element type. Type *VecTy = VectorType::get(Builder.getInt8Ty(), NumElts); Op = Builder.CreateBitCast(Op, VecTy, "cast"); // We'll be shuffling in zeroes. Value *Res = Constant::getNullValue(VecTy); // If shift is less than 16, emit a shuffle to move the bytes. Otherwise, // we'll just return the zero vector. if (Shift < 16) { uint32_t Idxs[64]; // 256/512-bit version is split into 2/4 16-byte lanes. for (unsigned l = 0; l != NumElts; l += 16) for (unsigned i = 0; i != 16; ++i) { unsigned Idx = NumElts + i - Shift; if (Idx < NumElts) Idx -= NumElts - 16; // end of lane, switch operand. Idxs[l + i] = Idx + l; } Res = Builder.CreateShuffleVector(Res, Op, makeArrayRef(Idxs, NumElts)); } // Bitcast back to a 64-bit element type. return Builder.CreateBitCast(Res, ResultTy, "cast"); } // Handles upgrading SSE2/AVX2/AVX512BW PSRLDQ intrinsics by converting them // to byte shuffles. static Value *UpgradeX86PSRLDQIntrinsics(IRBuilder<> &Builder, Value *Op, unsigned Shift) { Type *ResultTy = Op->getType(); unsigned NumElts = ResultTy->getVectorNumElements() * 8; // Bitcast from a 64-bit element type to a byte element type. Type *VecTy = VectorType::get(Builder.getInt8Ty(), NumElts); Op = Builder.CreateBitCast(Op, VecTy, "cast"); // We'll be shuffling in zeroes. Value *Res = Constant::getNullValue(VecTy); // If shift is less than 16, emit a shuffle to move the bytes. Otherwise, // we'll just return the zero vector. if (Shift < 16) { uint32_t Idxs[64]; // 256/512-bit version is split into 2/4 16-byte lanes. for (unsigned l = 0; l != NumElts; l += 16) for (unsigned i = 0; i != 16; ++i) { unsigned Idx = i + Shift; if (Idx >= 16) Idx += NumElts - 16; // end of lane, switch operand. Idxs[l + i] = Idx + l; } Res = Builder.CreateShuffleVector(Op, Res, makeArrayRef(Idxs, NumElts)); } // Bitcast back to a 64-bit element type. return Builder.CreateBitCast(Res, ResultTy, "cast"); } static Value *getX86MaskVec(IRBuilder<> &Builder, Value *Mask, unsigned NumElts) { llvm::VectorType *MaskTy = llvm::VectorType::get(Builder.getInt1Ty(), cast(Mask->getType())->getBitWidth()); Mask = Builder.CreateBitCast(Mask, MaskTy); // If we have less than 8 elements, then the starting mask was an i8 and // we need to extract down to the right number of elements. if (NumElts < 8) { uint32_t Indices[4]; for (unsigned i = 0; i != NumElts; ++i) Indices[i] = i; Mask = Builder.CreateShuffleVector(Mask, Mask, makeArrayRef(Indices, NumElts), "extract"); } return Mask; } static Value *EmitX86Select(IRBuilder<> &Builder, Value *Mask, Value *Op0, Value *Op1) { // If the mask is all ones just emit the first operation. if (const auto *C = dyn_cast(Mask)) if (C->isAllOnesValue()) return Op0; Mask = getX86MaskVec(Builder, Mask, Op0->getType()->getVectorNumElements()); return Builder.CreateSelect(Mask, Op0, Op1); } static Value *EmitX86ScalarSelect(IRBuilder<> &Builder, Value *Mask, Value *Op0, Value *Op1) { // If the mask is all ones just emit the first operation. if (const auto *C = dyn_cast(Mask)) if (C->isAllOnesValue()) return Op0; llvm::VectorType *MaskTy = llvm::VectorType::get(Builder.getInt1Ty(), Mask->getType()->getIntegerBitWidth()); Mask = Builder.CreateBitCast(Mask, MaskTy); Mask = Builder.CreateExtractElement(Mask, (uint64_t)0); return Builder.CreateSelect(Mask, Op0, Op1); } // Handle autoupgrade for masked PALIGNR and VALIGND/Q intrinsics. // PALIGNR handles large immediates by shifting while VALIGN masks the immediate // so we need to handle both cases. VALIGN also doesn't have 128-bit lanes. static Value *UpgradeX86ALIGNIntrinsics(IRBuilder<> &Builder, Value *Op0, Value *Op1, Value *Shift, Value *Passthru, Value *Mask, bool IsVALIGN) { unsigned ShiftVal = cast(Shift)->getZExtValue(); unsigned NumElts = Op0->getType()->getVectorNumElements(); assert((IsVALIGN || NumElts % 16 == 0) && "Illegal NumElts for PALIGNR!"); assert((!IsVALIGN || NumElts <= 16) && "NumElts too large for VALIGN!"); assert(isPowerOf2_32(NumElts) && "NumElts not a power of 2!"); // Mask the immediate for VALIGN. if (IsVALIGN) ShiftVal &= (NumElts - 1); // If palignr is shifting the pair of vectors more than the size of two // lanes, emit zero. if (ShiftVal >= 32) return llvm::Constant::getNullValue(Op0->getType()); // If palignr is shifting the pair of input vectors more than one lane, // but less than two lanes, convert to shifting in zeroes. if (ShiftVal > 16) { ShiftVal -= 16; Op1 = Op0; Op0 = llvm::Constant::getNullValue(Op0->getType()); } uint32_t Indices[64]; // 256-bit palignr operates on 128-bit lanes so we need to handle that for (unsigned l = 0; l < NumElts; l += 16) { for (unsigned i = 0; i != 16; ++i) { unsigned Idx = ShiftVal + i; if (!IsVALIGN && Idx >= 16) // Disable wrap for VALIGN. Idx += NumElts - 16; // End of lane, switch operand. Indices[l + i] = Idx + l; } } Value *Align = Builder.CreateShuffleVector(Op1, Op0, makeArrayRef(Indices, NumElts), "palignr"); return EmitX86Select(Builder, Mask, Align, Passthru); } static Value *UpgradeX86VPERMT2Intrinsics(IRBuilder<> &Builder, CallInst &CI, bool ZeroMask, bool IndexForm) { Type *Ty = CI.getType(); unsigned VecWidth = Ty->getPrimitiveSizeInBits(); unsigned EltWidth = Ty->getScalarSizeInBits(); bool IsFloat = Ty->isFPOrFPVectorTy(); Intrinsic::ID IID; if (VecWidth == 128 && EltWidth == 32 && IsFloat) IID = Intrinsic::x86_avx512_vpermi2var_ps_128; else if (VecWidth == 128 && EltWidth == 32 && !IsFloat) IID = Intrinsic::x86_avx512_vpermi2var_d_128; else if (VecWidth == 128 && EltWidth == 64 && IsFloat) IID = Intrinsic::x86_avx512_vpermi2var_pd_128; else if (VecWidth == 128 && EltWidth == 64 && !IsFloat) IID = Intrinsic::x86_avx512_vpermi2var_q_128; else if (VecWidth == 256 && EltWidth == 32 && IsFloat) IID = Intrinsic::x86_avx512_vpermi2var_ps_256; else if (VecWidth == 256 && EltWidth == 32 && !IsFloat) IID = Intrinsic::x86_avx512_vpermi2var_d_256; else if (VecWidth == 256 && EltWidth == 64 && IsFloat) IID = Intrinsic::x86_avx512_vpermi2var_pd_256; else if (VecWidth == 256 && EltWidth == 64 && !IsFloat) IID = Intrinsic::x86_avx512_vpermi2var_q_256; else if (VecWidth == 512 && EltWidth == 32 && IsFloat) IID = Intrinsic::x86_avx512_vpermi2var_ps_512; else if (VecWidth == 512 && EltWidth == 32 && !IsFloat) IID = Intrinsic::x86_avx512_vpermi2var_d_512; else if (VecWidth == 512 && EltWidth == 64 && IsFloat) IID = Intrinsic::x86_avx512_vpermi2var_pd_512; else if (VecWidth == 512 && EltWidth == 64 && !IsFloat) IID = Intrinsic::x86_avx512_vpermi2var_q_512; else if (VecWidth == 128 && EltWidth == 16) IID = Intrinsic::x86_avx512_vpermi2var_hi_128; else if (VecWidth == 256 && EltWidth == 16) IID = Intrinsic::x86_avx512_vpermi2var_hi_256; else if (VecWidth == 512 && EltWidth == 16) IID = Intrinsic::x86_avx512_vpermi2var_hi_512; else if (VecWidth == 128 && EltWidth == 8) IID = Intrinsic::x86_avx512_vpermi2var_qi_128; else if (VecWidth == 256 && EltWidth == 8) IID = Intrinsic::x86_avx512_vpermi2var_qi_256; else if (VecWidth == 512 && EltWidth == 8) IID = Intrinsic::x86_avx512_vpermi2var_qi_512; else llvm_unreachable("Unexpected intrinsic"); Value *Args[] = { CI.getArgOperand(0) , CI.getArgOperand(1), CI.getArgOperand(2) }; // If this isn't index form we need to swap operand 0 and 1. if (!IndexForm) std::swap(Args[0], Args[1]); Value *V = Builder.CreateCall(Intrinsic::getDeclaration(CI.getModule(), IID), Args); Value *PassThru = ZeroMask ? ConstantAggregateZero::get(Ty) : Builder.CreateBitCast(CI.getArgOperand(1), Ty); return EmitX86Select(Builder, CI.getArgOperand(3), V, PassThru); } static Value *UpgradeX86AddSubSatIntrinsics(IRBuilder<> &Builder, CallInst &CI, bool IsSigned, bool IsAddition) { Type *Ty = CI.getType(); Value *Op0 = CI.getOperand(0); Value *Op1 = CI.getOperand(1); Intrinsic::ID IID = IsSigned ? (IsAddition ? Intrinsic::sadd_sat : Intrinsic::ssub_sat) : (IsAddition ? Intrinsic::uadd_sat : Intrinsic::usub_sat); Function *Intrin = Intrinsic::getDeclaration(CI.getModule(), IID, Ty); Value *Res = Builder.CreateCall(Intrin, {Op0, Op1}); if (CI.getNumArgOperands() == 4) { // For masked intrinsics. Value *VecSrc = CI.getOperand(2); Value *Mask = CI.getOperand(3); Res = EmitX86Select(Builder, Mask, Res, VecSrc); } return Res; } static Value *upgradeX86Rotate(IRBuilder<> &Builder, CallInst &CI, bool IsRotateRight) { Type *Ty = CI.getType(); Value *Src = CI.getArgOperand(0); Value *Amt = CI.getArgOperand(1); // Amount may be scalar immediate, in which case create a splat vector. // Funnel shifts amounts are treated as modulo and types are all power-of-2 so // we only care about the lowest log2 bits anyway. if (Amt->getType() != Ty) { unsigned NumElts = Ty->getVectorNumElements(); Amt = Builder.CreateIntCast(Amt, Ty->getScalarType(), false); Amt = Builder.CreateVectorSplat(NumElts, Amt); } Intrinsic::ID IID = IsRotateRight ? Intrinsic::fshr : Intrinsic::fshl; Function *Intrin = Intrinsic::getDeclaration(CI.getModule(), IID, Ty); Value *Res = Builder.CreateCall(Intrin, {Src, Src, Amt}); if (CI.getNumArgOperands() == 4) { // For masked intrinsics. Value *VecSrc = CI.getOperand(2); Value *Mask = CI.getOperand(3); Res = EmitX86Select(Builder, Mask, Res, VecSrc); } return Res; } static Value *upgradeX86ConcatShift(IRBuilder<> &Builder, CallInst &CI, bool IsShiftRight, bool ZeroMask) { Type *Ty = CI.getType(); Value *Op0 = CI.getArgOperand(0); Value *Op1 = CI.getArgOperand(1); Value *Amt = CI.getArgOperand(2); if (IsShiftRight) std::swap(Op0, Op1); // Amount may be scalar immediate, in which case create a splat vector. // Funnel shifts amounts are treated as modulo and types are all power-of-2 so // we only care about the lowest log2 bits anyway. if (Amt->getType() != Ty) { unsigned NumElts = Ty->getVectorNumElements(); Amt = Builder.CreateIntCast(Amt, Ty->getScalarType(), false); Amt = Builder.CreateVectorSplat(NumElts, Amt); } Intrinsic::ID IID = IsShiftRight ? Intrinsic::fshr : Intrinsic::fshl; Function *Intrin = Intrinsic::getDeclaration(CI.getModule(), IID, Ty); Value *Res = Builder.CreateCall(Intrin, {Op0, Op1, Amt}); unsigned NumArgs = CI.getNumArgOperands(); if (NumArgs >= 4) { // For masked intrinsics. Value *VecSrc = NumArgs == 5 ? CI.getArgOperand(3) : ZeroMask ? ConstantAggregateZero::get(CI.getType()) : CI.getArgOperand(0); Value *Mask = CI.getOperand(NumArgs - 1); Res = EmitX86Select(Builder, Mask, Res, VecSrc); } return Res; } static Value *UpgradeMaskedStore(IRBuilder<> &Builder, Value *Ptr, Value *Data, Value *Mask, bool Aligned) { // Cast the pointer to the right type. Ptr = Builder.CreateBitCast(Ptr, llvm::PointerType::getUnqual(Data->getType())); unsigned Align = Aligned ? cast(Data->getType())->getBitWidth() / 8 : 1; // If the mask is all ones just emit a regular store. if (const auto *C = dyn_cast(Mask)) if (C->isAllOnesValue()) return Builder.CreateAlignedStore(Data, Ptr, Align); // Convert the mask from an integer type to a vector of i1. unsigned NumElts = Data->getType()->getVectorNumElements(); Mask = getX86MaskVec(Builder, Mask, NumElts); return Builder.CreateMaskedStore(Data, Ptr, Align, Mask); } static Value *UpgradeMaskedLoad(IRBuilder<> &Builder, Value *Ptr, Value *Passthru, Value *Mask, bool Aligned) { // Cast the pointer to the right type. Ptr = Builder.CreateBitCast(Ptr, llvm::PointerType::getUnqual(Passthru->getType())); unsigned Align = Aligned ? cast(Passthru->getType())->getBitWidth() / 8 : 1; // If the mask is all ones just emit a regular store. if (const auto *C = dyn_cast(Mask)) if (C->isAllOnesValue()) return Builder.CreateAlignedLoad(Ptr, Align); // Convert the mask from an integer type to a vector of i1. unsigned NumElts = Passthru->getType()->getVectorNumElements(); Mask = getX86MaskVec(Builder, Mask, NumElts); return Builder.CreateMaskedLoad(Ptr, Align, Mask, Passthru); } static Value *upgradeAbs(IRBuilder<> &Builder, CallInst &CI) { Value *Op0 = CI.getArgOperand(0); llvm::Type *Ty = Op0->getType(); Value *Zero = llvm::Constant::getNullValue(Ty); Value *Cmp = Builder.CreateICmp(ICmpInst::ICMP_SGT, Op0, Zero); Value *Neg = Builder.CreateNeg(Op0); Value *Res = Builder.CreateSelect(Cmp, Op0, Neg); if (CI.getNumArgOperands() == 3) Res = EmitX86Select(Builder,CI.getArgOperand(2), Res, CI.getArgOperand(1)); return Res; } static Value *upgradeIntMinMax(IRBuilder<> &Builder, CallInst &CI, ICmpInst::Predicate Pred) { Value *Op0 = CI.getArgOperand(0); Value *Op1 = CI.getArgOperand(1); Value *Cmp = Builder.CreateICmp(Pred, Op0, Op1); Value *Res = Builder.CreateSelect(Cmp, Op0, Op1); if (CI.getNumArgOperands() == 4) Res = EmitX86Select(Builder, CI.getArgOperand(3), Res, CI.getArgOperand(2)); return Res; } static Value *upgradePMULDQ(IRBuilder<> &Builder, CallInst &CI, bool IsSigned) { Type *Ty = CI.getType(); // Arguments have a vXi32 type so cast to vXi64. Value *LHS = Builder.CreateBitCast(CI.getArgOperand(0), Ty); Value *RHS = Builder.CreateBitCast(CI.getArgOperand(1), Ty); if (IsSigned) { // Shift left then arithmetic shift right. Constant *ShiftAmt = ConstantInt::get(Ty, 32); LHS = Builder.CreateShl(LHS, ShiftAmt); LHS = Builder.CreateAShr(LHS, ShiftAmt); RHS = Builder.CreateShl(RHS, ShiftAmt); RHS = Builder.CreateAShr(RHS, ShiftAmt); } else { // Clear the upper bits. Constant *Mask = ConstantInt::get(Ty, 0xffffffff); LHS = Builder.CreateAnd(LHS, Mask); RHS = Builder.CreateAnd(RHS, Mask); } Value *Res = Builder.CreateMul(LHS, RHS); if (CI.getNumArgOperands() == 4) Res = EmitX86Select(Builder, CI.getArgOperand(3), Res, CI.getArgOperand(2)); return Res; } // Applying mask on vector of i1's and make sure result is at least 8 bits wide. static Value *ApplyX86MaskOn1BitsVec(IRBuilder<> &Builder, Value *Vec, Value *Mask) { unsigned NumElts = Vec->getType()->getVectorNumElements(); if (Mask) { const auto *C = dyn_cast(Mask); if (!C || !C->isAllOnesValue()) Vec = Builder.CreateAnd(Vec, getX86MaskVec(Builder, Mask, NumElts)); } if (NumElts < 8) { uint32_t Indices[8]; for (unsigned i = 0; i != NumElts; ++i) Indices[i] = i; for (unsigned i = NumElts; i != 8; ++i) Indices[i] = NumElts + i % NumElts; Vec = Builder.CreateShuffleVector(Vec, Constant::getNullValue(Vec->getType()), Indices); } return Builder.CreateBitCast(Vec, Builder.getIntNTy(std::max(NumElts, 8U))); } static Value *upgradeMaskedCompare(IRBuilder<> &Builder, CallInst &CI, unsigned CC, bool Signed) { Value *Op0 = CI.getArgOperand(0); unsigned NumElts = Op0->getType()->getVectorNumElements(); Value *Cmp; if (CC == 3) { Cmp = Constant::getNullValue(llvm::VectorType::get(Builder.getInt1Ty(), NumElts)); } else if (CC == 7) { Cmp = Constant::getAllOnesValue(llvm::VectorType::get(Builder.getInt1Ty(), NumElts)); } else { ICmpInst::Predicate Pred; switch (CC) { default: llvm_unreachable("Unknown condition code"); case 0: Pred = ICmpInst::ICMP_EQ; break; case 1: Pred = Signed ? ICmpInst::ICMP_SLT : ICmpInst::ICMP_ULT; break; case 2: Pred = Signed ? ICmpInst::ICMP_SLE : ICmpInst::ICMP_ULE; break; case 4: Pred = ICmpInst::ICMP_NE; break; case 5: Pred = Signed ? ICmpInst::ICMP_SGE : ICmpInst::ICMP_UGE; break; case 6: Pred = Signed ? ICmpInst::ICMP_SGT : ICmpInst::ICMP_UGT; break; } Cmp = Builder.CreateICmp(Pred, Op0, CI.getArgOperand(1)); } Value *Mask = CI.getArgOperand(CI.getNumArgOperands() - 1); return ApplyX86MaskOn1BitsVec(Builder, Cmp, Mask); } // Replace a masked intrinsic with an older unmasked intrinsic. static Value *UpgradeX86MaskedShift(IRBuilder<> &Builder, CallInst &CI, Intrinsic::ID IID) { Function *Intrin = Intrinsic::getDeclaration(CI.getModule(), IID); Value *Rep = Builder.CreateCall(Intrin, { CI.getArgOperand(0), CI.getArgOperand(1) }); return EmitX86Select(Builder, CI.getArgOperand(3), Rep, CI.getArgOperand(2)); } static Value* upgradeMaskedMove(IRBuilder<> &Builder, CallInst &CI) { Value* A = CI.getArgOperand(0); Value* B = CI.getArgOperand(1); Value* Src = CI.getArgOperand(2); Value* Mask = CI.getArgOperand(3); Value* AndNode = Builder.CreateAnd(Mask, APInt(8, 1)); Value* Cmp = Builder.CreateIsNotNull(AndNode); Value* Extract1 = Builder.CreateExtractElement(B, (uint64_t)0); Value* Extract2 = Builder.CreateExtractElement(Src, (uint64_t)0); Value* Select = Builder.CreateSelect(Cmp, Extract1, Extract2); return Builder.CreateInsertElement(A, Select, (uint64_t)0); } static Value* UpgradeMaskToInt(IRBuilder<> &Builder, CallInst &CI) { Value* Op = CI.getArgOperand(0); Type* ReturnOp = CI.getType(); unsigned NumElts = CI.getType()->getVectorNumElements(); Value *Mask = getX86MaskVec(Builder, Op, NumElts); return Builder.CreateSExt(Mask, ReturnOp, "vpmovm2"); } // Replace intrinsic with unmasked version and a select. static bool upgradeAVX512MaskToSelect(StringRef Name, IRBuilder<> &Builder, CallInst &CI, Value *&Rep) { Name = Name.substr(12); // Remove avx512.mask. unsigned VecWidth = CI.getType()->getPrimitiveSizeInBits(); unsigned EltWidth = CI.getType()->getScalarSizeInBits(); Intrinsic::ID IID; if (Name.startswith("max.p")) { if (VecWidth == 128 && EltWidth == 32) IID = Intrinsic::x86_sse_max_ps; else if (VecWidth == 128 && EltWidth == 64) IID = Intrinsic::x86_sse2_max_pd; else if (VecWidth == 256 && EltWidth == 32) IID = Intrinsic::x86_avx_max_ps_256; else if (VecWidth == 256 && EltWidth == 64) IID = Intrinsic::x86_avx_max_pd_256; else llvm_unreachable("Unexpected intrinsic"); } else if (Name.startswith("min.p")) { if (VecWidth == 128 && EltWidth == 32) IID = Intrinsic::x86_sse_min_ps; else if (VecWidth == 128 && EltWidth == 64) IID = Intrinsic::x86_sse2_min_pd; else if (VecWidth == 256 && EltWidth == 32) IID = Intrinsic::x86_avx_min_ps_256; else if (VecWidth == 256 && EltWidth == 64) IID = Intrinsic::x86_avx_min_pd_256; else llvm_unreachable("Unexpected intrinsic"); } else if (Name.startswith("pshuf.b.")) { if (VecWidth == 128) IID = Intrinsic::x86_ssse3_pshuf_b_128; else if (VecWidth == 256) IID = Intrinsic::x86_avx2_pshuf_b; else if (VecWidth == 512) IID = Intrinsic::x86_avx512_pshuf_b_512; else llvm_unreachable("Unexpected intrinsic"); } else if (Name.startswith("pmul.hr.sw.")) { if (VecWidth == 128) IID = Intrinsic::x86_ssse3_pmul_hr_sw_128; else if (VecWidth == 256) IID = Intrinsic::x86_avx2_pmul_hr_sw; else if (VecWidth == 512) IID = Intrinsic::x86_avx512_pmul_hr_sw_512; else llvm_unreachable("Unexpected intrinsic"); } else if (Name.startswith("pmulh.w.")) { if (VecWidth == 128) IID = Intrinsic::x86_sse2_pmulh_w; else if (VecWidth == 256) IID = Intrinsic::x86_avx2_pmulh_w; else if (VecWidth == 512) IID = Intrinsic::x86_avx512_pmulh_w_512; else llvm_unreachable("Unexpected intrinsic"); } else if (Name.startswith("pmulhu.w.")) { if (VecWidth == 128) IID = Intrinsic::x86_sse2_pmulhu_w; else if (VecWidth == 256) IID = Intrinsic::x86_avx2_pmulhu_w; else if (VecWidth == 512) IID = Intrinsic::x86_avx512_pmulhu_w_512; else llvm_unreachable("Unexpected intrinsic"); } else if (Name.startswith("pmaddw.d.")) { if (VecWidth == 128) IID = Intrinsic::x86_sse2_pmadd_wd; else if (VecWidth == 256) IID = Intrinsic::x86_avx2_pmadd_wd; else if (VecWidth == 512) IID = Intrinsic::x86_avx512_pmaddw_d_512; else llvm_unreachable("Unexpected intrinsic"); } else if (Name.startswith("pmaddubs.w.")) { if (VecWidth == 128) IID = Intrinsic::x86_ssse3_pmadd_ub_sw_128; else if (VecWidth == 256) IID = Intrinsic::x86_avx2_pmadd_ub_sw; else if (VecWidth == 512) IID = Intrinsic::x86_avx512_pmaddubs_w_512; else llvm_unreachable("Unexpected intrinsic"); } else if (Name.startswith("packsswb.")) { if (VecWidth == 128) IID = Intrinsic::x86_sse2_packsswb_128; else if (VecWidth == 256) IID = Intrinsic::x86_avx2_packsswb; else if (VecWidth == 512) IID = Intrinsic::x86_avx512_packsswb_512; else llvm_unreachable("Unexpected intrinsic"); } else if (Name.startswith("packssdw.")) { if (VecWidth == 128) IID = Intrinsic::x86_sse2_packssdw_128; else if (VecWidth == 256) IID = Intrinsic::x86_avx2_packssdw; else if (VecWidth == 512) IID = Intrinsic::x86_avx512_packssdw_512; else llvm_unreachable("Unexpected intrinsic"); } else if (Name.startswith("packuswb.")) { if (VecWidth == 128) IID = Intrinsic::x86_sse2_packuswb_128; else if (VecWidth == 256) IID = Intrinsic::x86_avx2_packuswb; else if (VecWidth == 512) IID = Intrinsic::x86_avx512_packuswb_512; else llvm_unreachable("Unexpected intrinsic"); } else if (Name.startswith("packusdw.")) { if (VecWidth == 128) IID = Intrinsic::x86_sse41_packusdw; else if (VecWidth == 256) IID = Intrinsic::x86_avx2_packusdw; else if (VecWidth == 512) IID = Intrinsic::x86_avx512_packusdw_512; else llvm_unreachable("Unexpected intrinsic"); } else if (Name.startswith("vpermilvar.")) { if (VecWidth == 128 && EltWidth == 32) IID = Intrinsic::x86_avx_vpermilvar_ps; else if (VecWidth == 128 && EltWidth == 64) IID = Intrinsic::x86_avx_vpermilvar_pd; else if (VecWidth == 256 && EltWidth == 32) IID = Intrinsic::x86_avx_vpermilvar_ps_256; else if (VecWidth == 256 && EltWidth == 64) IID = Intrinsic::x86_avx_vpermilvar_pd_256; else if (VecWidth == 512 && EltWidth == 32) IID = Intrinsic::x86_avx512_vpermilvar_ps_512; else if (VecWidth == 512 && EltWidth == 64) IID = Intrinsic::x86_avx512_vpermilvar_pd_512; else llvm_unreachable("Unexpected intrinsic"); } else if (Name == "cvtpd2dq.256") { IID = Intrinsic::x86_avx_cvt_pd2dq_256; } else if (Name == "cvtpd2ps.256") { IID = Intrinsic::x86_avx_cvt_pd2_ps_256; } else if (Name == "cvttpd2dq.256") { IID = Intrinsic::x86_avx_cvtt_pd2dq_256; } else if (Name == "cvttps2dq.128") { IID = Intrinsic::x86_sse2_cvttps2dq; } else if (Name == "cvttps2dq.256") { IID = Intrinsic::x86_avx_cvtt_ps2dq_256; } else if (Name.startswith("permvar.")) { bool IsFloat = CI.getType()->isFPOrFPVectorTy(); if (VecWidth == 256 && EltWidth == 32 && IsFloat) IID = Intrinsic::x86_avx2_permps; else if (VecWidth == 256 && EltWidth == 32 && !IsFloat) IID = Intrinsic::x86_avx2_permd; else if (VecWidth == 256 && EltWidth == 64 && IsFloat) IID = Intrinsic::x86_avx512_permvar_df_256; else if (VecWidth == 256 && EltWidth == 64 && !IsFloat) IID = Intrinsic::x86_avx512_permvar_di_256; else if (VecWidth == 512 && EltWidth == 32 && IsFloat) IID = Intrinsic::x86_avx512_permvar_sf_512; else if (VecWidth == 512 && EltWidth == 32 && !IsFloat) IID = Intrinsic::x86_avx512_permvar_si_512; else if (VecWidth == 512 && EltWidth == 64 && IsFloat) IID = Intrinsic::x86_avx512_permvar_df_512; else if (VecWidth == 512 && EltWidth == 64 && !IsFloat) IID = Intrinsic::x86_avx512_permvar_di_512; else if (VecWidth == 128 && EltWidth == 16) IID = Intrinsic::x86_avx512_permvar_hi_128; else if (VecWidth == 256 && EltWidth == 16) IID = Intrinsic::x86_avx512_permvar_hi_256; else if (VecWidth == 512 && EltWidth == 16) IID = Intrinsic::x86_avx512_permvar_hi_512; else if (VecWidth == 128 && EltWidth == 8) IID = Intrinsic::x86_avx512_permvar_qi_128; else if (VecWidth == 256 && EltWidth == 8) IID = Intrinsic::x86_avx512_permvar_qi_256; else if (VecWidth == 512 && EltWidth == 8) IID = Intrinsic::x86_avx512_permvar_qi_512; else llvm_unreachable("Unexpected intrinsic"); } else if (Name.startswith("dbpsadbw.")) { if (VecWidth == 128) IID = Intrinsic::x86_avx512_dbpsadbw_128; else if (VecWidth == 256) IID = Intrinsic::x86_avx512_dbpsadbw_256; else if (VecWidth == 512) IID = Intrinsic::x86_avx512_dbpsadbw_512; else llvm_unreachable("Unexpected intrinsic"); } else if (Name.startswith("pmultishift.qb.")) { if (VecWidth == 128) IID = Intrinsic::x86_avx512_pmultishift_qb_128; else if (VecWidth == 256) IID = Intrinsic::x86_avx512_pmultishift_qb_256; else if (VecWidth == 512) IID = Intrinsic::x86_avx512_pmultishift_qb_512; else llvm_unreachable("Unexpected intrinsic"); } else return false; SmallVector Args(CI.arg_operands().begin(), CI.arg_operands().end()); Args.pop_back(); Args.pop_back(); Rep = Builder.CreateCall(Intrinsic::getDeclaration(CI.getModule(), IID), Args); unsigned NumArgs = CI.getNumArgOperands(); Rep = EmitX86Select(Builder, CI.getArgOperand(NumArgs - 1), Rep, CI.getArgOperand(NumArgs - 2)); return true; } /// Upgrade comment in call to inline asm that represents an objc retain release /// marker. void llvm::UpgradeInlineAsmString(std::string *AsmStr) { size_t Pos; if (AsmStr->find("mov\tfp") == 0 && AsmStr->find("objc_retainAutoreleaseReturnValue") != std::string::npos && (Pos = AsmStr->find("# marker")) != std::string::npos) { AsmStr->replace(Pos, 1, ";"); } return; } /// Upgrade a call to an old intrinsic. All argument and return casting must be /// provided to seamlessly integrate with existing context. void llvm::UpgradeIntrinsicCall(CallInst *CI, Function *NewFn) { Function *F = CI->getCalledFunction(); LLVMContext &C = CI->getContext(); IRBuilder<> Builder(C); Builder.SetInsertPoint(CI->getParent(), CI->getIterator()); assert(F && "Intrinsic call is not direct?"); if (!NewFn) { // Get the Function's name. StringRef Name = F->getName(); assert(Name.startswith("llvm.") && "Intrinsic doesn't start with 'llvm.'"); Name = Name.substr(5); bool IsX86 = Name.startswith("x86."); if (IsX86) Name = Name.substr(4); bool IsNVVM = Name.startswith("nvvm."); if (IsNVVM) Name = Name.substr(5); if (IsX86 && Name.startswith("sse4a.movnt.")) { Module *M = F->getParent(); SmallVector Elts; Elts.push_back( ConstantAsMetadata::get(ConstantInt::get(Type::getInt32Ty(C), 1))); MDNode *Node = MDNode::get(C, Elts); Value *Arg0 = CI->getArgOperand(0); Value *Arg1 = CI->getArgOperand(1); // Nontemporal (unaligned) store of the 0'th element of the float/double // vector. Type *SrcEltTy = cast(Arg1->getType())->getElementType(); PointerType *EltPtrTy = PointerType::getUnqual(SrcEltTy); Value *Addr = Builder.CreateBitCast(Arg0, EltPtrTy, "cast"); Value *Extract = Builder.CreateExtractElement(Arg1, (uint64_t)0, "extractelement"); StoreInst *SI = Builder.CreateAlignedStore(Extract, Addr, 1); SI->setMetadata(M->getMDKindID("nontemporal"), Node); // Remove intrinsic. CI->eraseFromParent(); return; } if (IsX86 && (Name.startswith("avx.movnt.") || Name.startswith("avx512.storent."))) { Module *M = F->getParent(); SmallVector Elts; Elts.push_back( ConstantAsMetadata::get(ConstantInt::get(Type::getInt32Ty(C), 1))); MDNode *Node = MDNode::get(C, Elts); Value *Arg0 = CI->getArgOperand(0); Value *Arg1 = CI->getArgOperand(1); // Convert the type of the pointer to a pointer to the stored type. Value *BC = Builder.CreateBitCast(Arg0, PointerType::getUnqual(Arg1->getType()), "cast"); VectorType *VTy = cast(Arg1->getType()); StoreInst *SI = Builder.CreateAlignedStore(Arg1, BC, VTy->getBitWidth() / 8); SI->setMetadata(M->getMDKindID("nontemporal"), Node); // Remove intrinsic. CI->eraseFromParent(); return; } if (IsX86 && Name == "sse2.storel.dq") { Value *Arg0 = CI->getArgOperand(0); Value *Arg1 = CI->getArgOperand(1); Type *NewVecTy = VectorType::get(Type::getInt64Ty(C), 2); Value *BC0 = Builder.CreateBitCast(Arg1, NewVecTy, "cast"); Value *Elt = Builder.CreateExtractElement(BC0, (uint64_t)0); Value *BC = Builder.CreateBitCast(Arg0, PointerType::getUnqual(Elt->getType()), "cast"); Builder.CreateAlignedStore(Elt, BC, 1); // Remove intrinsic. CI->eraseFromParent(); return; } if (IsX86 && (Name.startswith("sse.storeu.") || Name.startswith("sse2.storeu.") || Name.startswith("avx.storeu."))) { Value *Arg0 = CI->getArgOperand(0); Value *Arg1 = CI->getArgOperand(1); Arg0 = Builder.CreateBitCast(Arg0, PointerType::getUnqual(Arg1->getType()), "cast"); Builder.CreateAlignedStore(Arg1, Arg0, 1); // Remove intrinsic. CI->eraseFromParent(); return; } if (IsX86 && Name == "avx512.mask.store.ss") { Value *Mask = Builder.CreateAnd(CI->getArgOperand(2), Builder.getInt8(1)); UpgradeMaskedStore(Builder, CI->getArgOperand(0), CI->getArgOperand(1), Mask, false); // Remove intrinsic. CI->eraseFromParent(); return; } if (IsX86 && (Name.startswith("avx512.mask.store"))) { // "avx512.mask.storeu." or "avx512.mask.store." bool Aligned = Name[17] != 'u'; // "avx512.mask.storeu". UpgradeMaskedStore(Builder, CI->getArgOperand(0), CI->getArgOperand(1), CI->getArgOperand(2), Aligned); // Remove intrinsic. CI->eraseFromParent(); return; } Value *Rep; // Upgrade packed integer vector compare intrinsics to compare instructions. if (IsX86 && (Name.startswith("sse2.pcmp") || Name.startswith("avx2.pcmp"))) { // "sse2.pcpmpeq." "sse2.pcmpgt." "avx2.pcmpeq." or "avx2.pcmpgt." bool CmpEq = Name[9] == 'e'; Rep = Builder.CreateICmp(CmpEq ? ICmpInst::ICMP_EQ : ICmpInst::ICMP_SGT, CI->getArgOperand(0), CI->getArgOperand(1)); Rep = Builder.CreateSExt(Rep, CI->getType(), ""); } else if (IsX86 && (Name.startswith("avx512.broadcastm"))) { Type *ExtTy = Type::getInt32Ty(C); if (CI->getOperand(0)->getType()->isIntegerTy(8)) ExtTy = Type::getInt64Ty(C); unsigned NumElts = CI->getType()->getPrimitiveSizeInBits() / ExtTy->getPrimitiveSizeInBits(); Rep = Builder.CreateZExt(CI->getArgOperand(0), ExtTy); Rep = Builder.CreateVectorSplat(NumElts, Rep); } else if (IsX86 && (Name == "sse.sqrt.ss" || Name == "sse2.sqrt.sd")) { Value *Vec = CI->getArgOperand(0); Value *Elt0 = Builder.CreateExtractElement(Vec, (uint64_t)0); Function *Intr = Intrinsic::getDeclaration(F->getParent(), Intrinsic::sqrt, Elt0->getType()); Elt0 = Builder.CreateCall(Intr, Elt0); Rep = Builder.CreateInsertElement(Vec, Elt0, (uint64_t)0); } else if (IsX86 && (Name.startswith("avx.sqrt.p") || Name.startswith("sse2.sqrt.p") || Name.startswith("sse.sqrt.p"))) { Rep = Builder.CreateCall(Intrinsic::getDeclaration(F->getParent(), Intrinsic::sqrt, CI->getType()), {CI->getArgOperand(0)}); } else if (IsX86 && (Name.startswith("avx512.mask.sqrt.p"))) { if (CI->getNumArgOperands() == 4 && (!isa(CI->getArgOperand(3)) || cast(CI->getArgOperand(3))->getZExtValue() != 4)) { Intrinsic::ID IID = Name[18] == 's' ? Intrinsic::x86_avx512_sqrt_ps_512 : Intrinsic::x86_avx512_sqrt_pd_512; Value *Args[] = { CI->getArgOperand(0), CI->getArgOperand(3) }; Rep = Builder.CreateCall(Intrinsic::getDeclaration(CI->getModule(), IID), Args); } else { Rep = Builder.CreateCall(Intrinsic::getDeclaration(F->getParent(), Intrinsic::sqrt, CI->getType()), {CI->getArgOperand(0)}); } Rep = EmitX86Select(Builder, CI->getArgOperand(2), Rep, CI->getArgOperand(1)); } else if (IsX86 && (Name.startswith("avx512.ptestm") || Name.startswith("avx512.ptestnm"))) { Value *Op0 = CI->getArgOperand(0); Value *Op1 = CI->getArgOperand(1); Value *Mask = CI->getArgOperand(2); Rep = Builder.CreateAnd(Op0, Op1); llvm::Type *Ty = Op0->getType(); Value *Zero = llvm::Constant::getNullValue(Ty); ICmpInst::Predicate Pred = Name.startswith("avx512.ptestm") ? ICmpInst::ICMP_NE : ICmpInst::ICMP_EQ; Rep = Builder.CreateICmp(Pred, Rep, Zero); Rep = ApplyX86MaskOn1BitsVec(Builder, Rep, Mask); } else if (IsX86 && (Name.startswith("avx512.mask.pbroadcast"))){ unsigned NumElts = CI->getArgOperand(1)->getType()->getVectorNumElements(); Rep = Builder.CreateVectorSplat(NumElts, CI->getArgOperand(0)); Rep = EmitX86Select(Builder, CI->getArgOperand(2), Rep, CI->getArgOperand(1)); } else if (IsX86 && (Name.startswith("avx512.kunpck"))) { unsigned NumElts = CI->getType()->getScalarSizeInBits(); Value *LHS = getX86MaskVec(Builder, CI->getArgOperand(0), NumElts); Value *RHS = getX86MaskVec(Builder, CI->getArgOperand(1), NumElts); uint32_t Indices[64]; for (unsigned i = 0; i != NumElts; ++i) Indices[i] = i; // First extract half of each vector. This gives better codegen than // doing it in a single shuffle. LHS = Builder.CreateShuffleVector(LHS, LHS, makeArrayRef(Indices, NumElts / 2)); RHS = Builder.CreateShuffleVector(RHS, RHS, makeArrayRef(Indices, NumElts / 2)); // Concat the vectors. // NOTE: Operands have to be swapped to match intrinsic definition. Rep = Builder.CreateShuffleVector(RHS, LHS, makeArrayRef(Indices, NumElts)); Rep = Builder.CreateBitCast(Rep, CI->getType()); } else if (IsX86 && Name == "avx512.kand.w") { Value *LHS = getX86MaskVec(Builder, CI->getArgOperand(0), 16); Value *RHS = getX86MaskVec(Builder, CI->getArgOperand(1), 16); Rep = Builder.CreateAnd(LHS, RHS); Rep = Builder.CreateBitCast(Rep, CI->getType()); } else if (IsX86 && Name == "avx512.kandn.w") { Value *LHS = getX86MaskVec(Builder, CI->getArgOperand(0), 16); Value *RHS = getX86MaskVec(Builder, CI->getArgOperand(1), 16); LHS = Builder.CreateNot(LHS); Rep = Builder.CreateAnd(LHS, RHS); Rep = Builder.CreateBitCast(Rep, CI->getType()); } else if (IsX86 && Name == "avx512.kor.w") { Value *LHS = getX86MaskVec(Builder, CI->getArgOperand(0), 16); Value *RHS = getX86MaskVec(Builder, CI->getArgOperand(1), 16); Rep = Builder.CreateOr(LHS, RHS); Rep = Builder.CreateBitCast(Rep, CI->getType()); } else if (IsX86 && Name == "avx512.kxor.w") { Value *LHS = getX86MaskVec(Builder, CI->getArgOperand(0), 16); Value *RHS = getX86MaskVec(Builder, CI->getArgOperand(1), 16); Rep = Builder.CreateXor(LHS, RHS); Rep = Builder.CreateBitCast(Rep, CI->getType()); } else if (IsX86 && Name == "avx512.kxnor.w") { Value *LHS = getX86MaskVec(Builder, CI->getArgOperand(0), 16); Value *RHS = getX86MaskVec(Builder, CI->getArgOperand(1), 16); LHS = Builder.CreateNot(LHS); Rep = Builder.CreateXor(LHS, RHS); Rep = Builder.CreateBitCast(Rep, CI->getType()); } else if (IsX86 && Name == "avx512.knot.w") { Rep = getX86MaskVec(Builder, CI->getArgOperand(0), 16); Rep = Builder.CreateNot(Rep); Rep = Builder.CreateBitCast(Rep, CI->getType()); } else if (IsX86 && (Name == "avx512.kortestz.w" || Name == "avx512.kortestc.w")) { Value *LHS = getX86MaskVec(Builder, CI->getArgOperand(0), 16); Value *RHS = getX86MaskVec(Builder, CI->getArgOperand(1), 16); Rep = Builder.CreateOr(LHS, RHS); Rep = Builder.CreateBitCast(Rep, Builder.getInt16Ty()); Value *C; if (Name[14] == 'c') C = ConstantInt::getAllOnesValue(Builder.getInt16Ty()); else C = ConstantInt::getNullValue(Builder.getInt16Ty()); Rep = Builder.CreateICmpEQ(Rep, C); Rep = Builder.CreateZExt(Rep, Builder.getInt32Ty()); } else if (IsX86 && (Name == "sse.add.ss" || Name == "sse2.add.sd" || Name == "sse.sub.ss" || Name == "sse2.sub.sd" || Name == "sse.mul.ss" || Name == "sse2.mul.sd" || Name == "sse.div.ss" || Name == "sse2.div.sd")) { Type *I32Ty = Type::getInt32Ty(C); Value *Elt0 = Builder.CreateExtractElement(CI->getArgOperand(0), ConstantInt::get(I32Ty, 0)); Value *Elt1 = Builder.CreateExtractElement(CI->getArgOperand(1), ConstantInt::get(I32Ty, 0)); Value *EltOp; if (Name.contains(".add.")) EltOp = Builder.CreateFAdd(Elt0, Elt1); else if (Name.contains(".sub.")) EltOp = Builder.CreateFSub(Elt0, Elt1); else if (Name.contains(".mul.")) EltOp = Builder.CreateFMul(Elt0, Elt1); else EltOp = Builder.CreateFDiv(Elt0, Elt1); Rep = Builder.CreateInsertElement(CI->getArgOperand(0), EltOp, ConstantInt::get(I32Ty, 0)); } else if (IsX86 && Name.startswith("avx512.mask.pcmp")) { // "avx512.mask.pcmpeq." or "avx512.mask.pcmpgt." bool CmpEq = Name[16] == 'e'; Rep = upgradeMaskedCompare(Builder, *CI, CmpEq ? 0 : 6, true); } else if (IsX86 && Name.startswith("avx512.mask.vpshufbitqmb.")) { Type *OpTy = CI->getArgOperand(0)->getType(); unsigned VecWidth = OpTy->getPrimitiveSizeInBits(); Intrinsic::ID IID; switch (VecWidth) { default: llvm_unreachable("Unexpected intrinsic"); case 128: IID = Intrinsic::x86_avx512_vpshufbitqmb_128; break; case 256: IID = Intrinsic::x86_avx512_vpshufbitqmb_256; break; case 512: IID = Intrinsic::x86_avx512_vpshufbitqmb_512; break; } Rep = Builder.CreateCall(Intrinsic::getDeclaration(F->getParent(), IID), { CI->getOperand(0), CI->getArgOperand(1) }); Rep = ApplyX86MaskOn1BitsVec(Builder, Rep, CI->getArgOperand(2)); } else if (IsX86 && Name.startswith("avx512.mask.fpclass.p")) { Type *OpTy = CI->getArgOperand(0)->getType(); unsigned VecWidth = OpTy->getPrimitiveSizeInBits(); unsigned EltWidth = OpTy->getScalarSizeInBits(); Intrinsic::ID IID; if (VecWidth == 128 && EltWidth == 32) IID = Intrinsic::x86_avx512_fpclass_ps_128; else if (VecWidth == 256 && EltWidth == 32) IID = Intrinsic::x86_avx512_fpclass_ps_256; else if (VecWidth == 512 && EltWidth == 32) IID = Intrinsic::x86_avx512_fpclass_ps_512; else if (VecWidth == 128 && EltWidth == 64) IID = Intrinsic::x86_avx512_fpclass_pd_128; else if (VecWidth == 256 && EltWidth == 64) IID = Intrinsic::x86_avx512_fpclass_pd_256; else if (VecWidth == 512 && EltWidth == 64) IID = Intrinsic::x86_avx512_fpclass_pd_512; else llvm_unreachable("Unexpected intrinsic"); Rep = Builder.CreateCall(Intrinsic::getDeclaration(F->getParent(), IID), { CI->getOperand(0), CI->getArgOperand(1) }); Rep = ApplyX86MaskOn1BitsVec(Builder, Rep, CI->getArgOperand(2)); } else if (IsX86 && Name.startswith("avx512.mask.cmp.p")) { Type *OpTy = CI->getArgOperand(0)->getType(); unsigned VecWidth = OpTy->getPrimitiveSizeInBits(); unsigned EltWidth = OpTy->getScalarSizeInBits(); Intrinsic::ID IID; if (VecWidth == 128 && EltWidth == 32) IID = Intrinsic::x86_avx512_cmp_ps_128; else if (VecWidth == 256 && EltWidth == 32) IID = Intrinsic::x86_avx512_cmp_ps_256; else if (VecWidth == 512 && EltWidth == 32) IID = Intrinsic::x86_avx512_cmp_ps_512; else if (VecWidth == 128 && EltWidth == 64) IID = Intrinsic::x86_avx512_cmp_pd_128; else if (VecWidth == 256 && EltWidth == 64) IID = Intrinsic::x86_avx512_cmp_pd_256; else if (VecWidth == 512 && EltWidth == 64) IID = Intrinsic::x86_avx512_cmp_pd_512; else llvm_unreachable("Unexpected intrinsic"); SmallVector Args; Args.push_back(CI->getArgOperand(0)); Args.push_back(CI->getArgOperand(1)); Args.push_back(CI->getArgOperand(2)); if (CI->getNumArgOperands() == 5) Args.push_back(CI->getArgOperand(4)); Rep = Builder.CreateCall(Intrinsic::getDeclaration(F->getParent(), IID), Args); Rep = ApplyX86MaskOn1BitsVec(Builder, Rep, CI->getArgOperand(3)); } else if (IsX86 && Name.startswith("avx512.mask.cmp.") && Name[16] != 'p') { // Integer compare intrinsics. unsigned Imm = cast(CI->getArgOperand(2))->getZExtValue(); Rep = upgradeMaskedCompare(Builder, *CI, Imm, true); } else if (IsX86 && Name.startswith("avx512.mask.ucmp.")) { unsigned Imm = cast(CI->getArgOperand(2))->getZExtValue(); Rep = upgradeMaskedCompare(Builder, *CI, Imm, false); } else if (IsX86 && (Name.startswith("avx512.cvtb2mask.") || Name.startswith("avx512.cvtw2mask.") || Name.startswith("avx512.cvtd2mask.") || Name.startswith("avx512.cvtq2mask."))) { Value *Op = CI->getArgOperand(0); Value *Zero = llvm::Constant::getNullValue(Op->getType()); Rep = Builder.CreateICmp(ICmpInst::ICMP_SLT, Op, Zero); Rep = ApplyX86MaskOn1BitsVec(Builder, Rep, nullptr); } else if(IsX86 && (Name == "ssse3.pabs.b.128" || Name == "ssse3.pabs.w.128" || Name == "ssse3.pabs.d.128" || Name.startswith("avx2.pabs") || Name.startswith("avx512.mask.pabs"))) { Rep = upgradeAbs(Builder, *CI); } else if (IsX86 && (Name == "sse41.pmaxsb" || Name == "sse2.pmaxs.w" || Name == "sse41.pmaxsd" || Name.startswith("avx2.pmaxs") || Name.startswith("avx512.mask.pmaxs"))) { Rep = upgradeIntMinMax(Builder, *CI, ICmpInst::ICMP_SGT); } else if (IsX86 && (Name == "sse2.pmaxu.b" || Name == "sse41.pmaxuw" || Name == "sse41.pmaxud" || Name.startswith("avx2.pmaxu") || Name.startswith("avx512.mask.pmaxu"))) { Rep = upgradeIntMinMax(Builder, *CI, ICmpInst::ICMP_UGT); } else if (IsX86 && (Name == "sse41.pminsb" || Name == "sse2.pmins.w" || Name == "sse41.pminsd" || Name.startswith("avx2.pmins") || Name.startswith("avx512.mask.pmins"))) { Rep = upgradeIntMinMax(Builder, *CI, ICmpInst::ICMP_SLT); } else if (IsX86 && (Name == "sse2.pminu.b" || Name == "sse41.pminuw" || Name == "sse41.pminud" || Name.startswith("avx2.pminu") || Name.startswith("avx512.mask.pminu"))) { Rep = upgradeIntMinMax(Builder, *CI, ICmpInst::ICMP_ULT); } else if (IsX86 && (Name == "sse2.pmulu.dq" || Name == "avx2.pmulu.dq" || Name == "avx512.pmulu.dq.512" || Name.startswith("avx512.mask.pmulu.dq."))) { Rep = upgradePMULDQ(Builder, *CI, /*Signed*/false); } else if (IsX86 && (Name == "sse41.pmuldq" || Name == "avx2.pmul.dq" || Name == "avx512.pmul.dq.512" || Name.startswith("avx512.mask.pmul.dq."))) { Rep = upgradePMULDQ(Builder, *CI, /*Signed*/true); } else if (IsX86 && (Name == "sse.cvtsi2ss" || Name == "sse2.cvtsi2sd" || Name == "sse.cvtsi642ss" || Name == "sse2.cvtsi642sd")) { Rep = Builder.CreateSIToFP(CI->getArgOperand(1), CI->getType()->getVectorElementType()); Rep = Builder.CreateInsertElement(CI->getArgOperand(0), Rep, (uint64_t)0); } else if (IsX86 && Name == "avx512.cvtusi2sd") { Rep = Builder.CreateUIToFP(CI->getArgOperand(1), CI->getType()->getVectorElementType()); Rep = Builder.CreateInsertElement(CI->getArgOperand(0), Rep, (uint64_t)0); } else if (IsX86 && Name == "sse2.cvtss2sd") { Rep = Builder.CreateExtractElement(CI->getArgOperand(1), (uint64_t)0); Rep = Builder.CreateFPExt(Rep, CI->getType()->getVectorElementType()); Rep = Builder.CreateInsertElement(CI->getArgOperand(0), Rep, (uint64_t)0); } else if (IsX86 && (Name == "sse2.cvtdq2pd" || Name == "sse2.cvtdq2ps" || Name == "avx.cvtdq2.pd.256" || Name == "avx.cvtdq2.ps.256" || Name.startswith("avx512.mask.cvtdq2pd.") || Name.startswith("avx512.mask.cvtudq2pd.") || Name == "avx512.mask.cvtdq2ps.128" || Name == "avx512.mask.cvtdq2ps.256" || Name == "avx512.mask.cvtudq2ps.128" || Name == "avx512.mask.cvtudq2ps.256" || Name == "avx512.mask.cvtqq2pd.128" || Name == "avx512.mask.cvtqq2pd.256" || Name == "avx512.mask.cvtuqq2pd.128" || Name == "avx512.mask.cvtuqq2pd.256" || Name == "sse2.cvtps2pd" || Name == "avx.cvt.ps2.pd.256" || Name == "avx512.mask.cvtps2pd.128" || Name == "avx512.mask.cvtps2pd.256")) { Type *DstTy = CI->getType(); Rep = CI->getArgOperand(0); unsigned NumDstElts = DstTy->getVectorNumElements(); if (NumDstElts < Rep->getType()->getVectorNumElements()) { assert(NumDstElts == 2 && "Unexpected vector size"); uint32_t ShuffleMask[2] = { 0, 1 }; Rep = Builder.CreateShuffleVector(Rep, Rep, ShuffleMask); } bool IsPS2PD = (StringRef::npos != Name.find("ps2")); bool IsUnsigned = (StringRef::npos != Name.find("cvtu")); if (IsPS2PD) Rep = Builder.CreateFPExt(Rep, DstTy, "cvtps2pd"); else if (IsUnsigned) Rep = Builder.CreateUIToFP(Rep, DstTy, "cvt"); else Rep = Builder.CreateSIToFP(Rep, DstTy, "cvt"); if (CI->getNumArgOperands() == 3) Rep = EmitX86Select(Builder, CI->getArgOperand(2), Rep, CI->getArgOperand(1)); } else if (IsX86 && (Name.startswith("avx512.mask.loadu."))) { Rep = UpgradeMaskedLoad(Builder, CI->getArgOperand(0), CI->getArgOperand(1), CI->getArgOperand(2), /*Aligned*/false); } else if (IsX86 && (Name.startswith("avx512.mask.load."))) { Rep = UpgradeMaskedLoad(Builder, CI->getArgOperand(0), CI->getArgOperand(1),CI->getArgOperand(2), /*Aligned*/true); } else if (IsX86 && Name.startswith("avx512.mask.expand.load.")) { Type *ResultTy = CI->getType(); Type *PtrTy = ResultTy->getVectorElementType(); // Cast the pointer to element type. Value *Ptr = Builder.CreateBitCast(CI->getOperand(0), llvm::PointerType::getUnqual(PtrTy)); Value *MaskVec = getX86MaskVec(Builder, CI->getArgOperand(2), ResultTy->getVectorNumElements()); Function *ELd = Intrinsic::getDeclaration(F->getParent(), Intrinsic::masked_expandload, ResultTy); Rep = Builder.CreateCall(ELd, { Ptr, MaskVec, CI->getOperand(1) }); } else if (IsX86 && Name.startswith("avx512.mask.compress.store.")) { Type *ResultTy = CI->getArgOperand(1)->getType(); Type *PtrTy = ResultTy->getVectorElementType(); // Cast the pointer to element type. Value *Ptr = Builder.CreateBitCast(CI->getOperand(0), llvm::PointerType::getUnqual(PtrTy)); Value *MaskVec = getX86MaskVec(Builder, CI->getArgOperand(2), ResultTy->getVectorNumElements()); Function *CSt = Intrinsic::getDeclaration(F->getParent(), Intrinsic::masked_compressstore, ResultTy); Rep = Builder.CreateCall(CSt, { CI->getArgOperand(1), Ptr, MaskVec }); } else if (IsX86 && Name.startswith("xop.vpcom")) { Intrinsic::ID intID; if (Name.endswith("ub")) intID = Intrinsic::x86_xop_vpcomub; else if (Name.endswith("uw")) intID = Intrinsic::x86_xop_vpcomuw; else if (Name.endswith("ud")) intID = Intrinsic::x86_xop_vpcomud; else if (Name.endswith("uq")) intID = Intrinsic::x86_xop_vpcomuq; else if (Name.endswith("b")) intID = Intrinsic::x86_xop_vpcomb; else if (Name.endswith("w")) intID = Intrinsic::x86_xop_vpcomw; else if (Name.endswith("d")) intID = Intrinsic::x86_xop_vpcomd; else if (Name.endswith("q")) intID = Intrinsic::x86_xop_vpcomq; else llvm_unreachable("Unknown suffix"); Name = Name.substr(9); // strip off "xop.vpcom" unsigned Imm; if (Name.startswith("lt")) Imm = 0; else if (Name.startswith("le")) Imm = 1; else if (Name.startswith("gt")) Imm = 2; else if (Name.startswith("ge")) Imm = 3; else if (Name.startswith("eq")) Imm = 4; else if (Name.startswith("ne")) Imm = 5; else if (Name.startswith("false")) Imm = 6; else if (Name.startswith("true")) Imm = 7; else llvm_unreachable("Unknown condition"); Function *VPCOM = Intrinsic::getDeclaration(F->getParent(), intID); Rep = Builder.CreateCall(VPCOM, {CI->getArgOperand(0), CI->getArgOperand(1), Builder.getInt8(Imm)}); } else if (IsX86 && Name.startswith("xop.vpcmov")) { Value *Sel = CI->getArgOperand(2); Value *NotSel = Builder.CreateNot(Sel); Value *Sel0 = Builder.CreateAnd(CI->getArgOperand(0), Sel); Value *Sel1 = Builder.CreateAnd(CI->getArgOperand(1), NotSel); Rep = Builder.CreateOr(Sel0, Sel1); } else if (IsX86 && (Name.startswith("xop.vprot") || Name.startswith("avx512.prol") || Name.startswith("avx512.mask.prol"))) { Rep = upgradeX86Rotate(Builder, *CI, false); } else if (IsX86 && (Name.startswith("avx512.pror") || Name.startswith("avx512.mask.pror"))) { Rep = upgradeX86Rotate(Builder, *CI, true); } else if (IsX86 && (Name.startswith("avx512.vpshld.") || Name.startswith("avx512.mask.vpshld") || Name.startswith("avx512.maskz.vpshld"))) { bool ZeroMask = Name[11] == 'z'; Rep = upgradeX86ConcatShift(Builder, *CI, false, ZeroMask); } else if (IsX86 && (Name.startswith("avx512.vpshrd.") || Name.startswith("avx512.mask.vpshrd") || Name.startswith("avx512.maskz.vpshrd"))) { bool ZeroMask = Name[11] == 'z'; Rep = upgradeX86ConcatShift(Builder, *CI, true, ZeroMask); } else if (IsX86 && Name == "sse42.crc32.64.8") { Function *CRC32 = Intrinsic::getDeclaration(F->getParent(), Intrinsic::x86_sse42_crc32_32_8); Value *Trunc0 = Builder.CreateTrunc(CI->getArgOperand(0), Type::getInt32Ty(C)); Rep = Builder.CreateCall(CRC32, {Trunc0, CI->getArgOperand(1)}); Rep = Builder.CreateZExt(Rep, CI->getType(), ""); } else if (IsX86 && (Name.startswith("avx.vbroadcast.s") || Name.startswith("avx512.vbroadcast.s"))) { // Replace broadcasts with a series of insertelements. Type *VecTy = CI->getType(); Type *EltTy = VecTy->getVectorElementType(); unsigned EltNum = VecTy->getVectorNumElements(); Value *Cast = Builder.CreateBitCast(CI->getArgOperand(0), EltTy->getPointerTo()); Value *Load = Builder.CreateLoad(EltTy, Cast); Type *I32Ty = Type::getInt32Ty(C); Rep = UndefValue::get(VecTy); for (unsigned I = 0; I < EltNum; ++I) Rep = Builder.CreateInsertElement(Rep, Load, ConstantInt::get(I32Ty, I)); } else if (IsX86 && (Name.startswith("sse41.pmovsx") || Name.startswith("sse41.pmovzx") || Name.startswith("avx2.pmovsx") || Name.startswith("avx2.pmovzx") || Name.startswith("avx512.mask.pmovsx") || Name.startswith("avx512.mask.pmovzx"))) { VectorType *SrcTy = cast(CI->getArgOperand(0)->getType()); VectorType *DstTy = cast(CI->getType()); unsigned NumDstElts = DstTy->getNumElements(); // Extract a subvector of the first NumDstElts lanes and sign/zero extend. SmallVector ShuffleMask(NumDstElts); for (unsigned i = 0; i != NumDstElts; ++i) ShuffleMask[i] = i; Value *SV = Builder.CreateShuffleVector( CI->getArgOperand(0), UndefValue::get(SrcTy), ShuffleMask); bool DoSext = (StringRef::npos != Name.find("pmovsx")); Rep = DoSext ? Builder.CreateSExt(SV, DstTy) : Builder.CreateZExt(SV, DstTy); // If there are 3 arguments, it's a masked intrinsic so we need a select. if (CI->getNumArgOperands() == 3) Rep = EmitX86Select(Builder, CI->getArgOperand(2), Rep, CI->getArgOperand(1)); } else if (IsX86 && (Name.startswith("avx.vbroadcastf128") || Name == "avx2.vbroadcasti128")) { // Replace vbroadcastf128/vbroadcasti128 with a vector load+shuffle. Type *EltTy = CI->getType()->getVectorElementType(); unsigned NumSrcElts = 128 / EltTy->getPrimitiveSizeInBits(); Type *VT = VectorType::get(EltTy, NumSrcElts); Value *Op = Builder.CreatePointerCast(CI->getArgOperand(0), PointerType::getUnqual(VT)); Value *Load = Builder.CreateAlignedLoad(Op, 1); if (NumSrcElts == 2) Rep = Builder.CreateShuffleVector(Load, UndefValue::get(Load->getType()), { 0, 1, 0, 1 }); else Rep = Builder.CreateShuffleVector(Load, UndefValue::get(Load->getType()), { 0, 1, 2, 3, 0, 1, 2, 3 }); } else if (IsX86 && (Name.startswith("avx512.mask.shuf.i") || Name.startswith("avx512.mask.shuf.f"))) { unsigned Imm = cast(CI->getArgOperand(2))->getZExtValue(); Type *VT = CI->getType(); unsigned NumLanes = VT->getPrimitiveSizeInBits() / 128; unsigned NumElementsInLane = 128 / VT->getScalarSizeInBits(); unsigned ControlBitsMask = NumLanes - 1; unsigned NumControlBits = NumLanes / 2; SmallVector ShuffleMask(0); for (unsigned l = 0; l != NumLanes; ++l) { unsigned LaneMask = (Imm >> (l * NumControlBits)) & ControlBitsMask; // We actually need the other source. if (l >= NumLanes / 2) LaneMask += NumLanes; for (unsigned i = 0; i != NumElementsInLane; ++i) ShuffleMask.push_back(LaneMask * NumElementsInLane + i); } Rep = Builder.CreateShuffleVector(CI->getArgOperand(0), CI->getArgOperand(1), ShuffleMask); Rep = EmitX86Select(Builder, CI->getArgOperand(4), Rep, CI->getArgOperand(3)); }else if (IsX86 && (Name.startswith("avx512.mask.broadcastf") || Name.startswith("avx512.mask.broadcasti"))) { unsigned NumSrcElts = CI->getArgOperand(0)->getType()->getVectorNumElements(); unsigned NumDstElts = CI->getType()->getVectorNumElements(); SmallVector ShuffleMask(NumDstElts); for (unsigned i = 0; i != NumDstElts; ++i) ShuffleMask[i] = i % NumSrcElts; Rep = Builder.CreateShuffleVector(CI->getArgOperand(0), CI->getArgOperand(0), ShuffleMask); Rep = EmitX86Select(Builder, CI->getArgOperand(2), Rep, CI->getArgOperand(1)); } else if (IsX86 && (Name.startswith("avx2.pbroadcast") || Name.startswith("avx2.vbroadcast") || Name.startswith("avx512.pbroadcast") || Name.startswith("avx512.mask.broadcast.s"))) { // Replace vp?broadcasts with a vector shuffle. Value *Op = CI->getArgOperand(0); unsigned NumElts = CI->getType()->getVectorNumElements(); Type *MaskTy = VectorType::get(Type::getInt32Ty(C), NumElts); Rep = Builder.CreateShuffleVector(Op, UndefValue::get(Op->getType()), Constant::getNullValue(MaskTy)); if (CI->getNumArgOperands() == 3) Rep = EmitX86Select(Builder, CI->getArgOperand(2), Rep, CI->getArgOperand(1)); } else if (IsX86 && (Name.startswith("sse2.padds.") || Name.startswith("sse2.psubs.") || Name.startswith("avx2.padds.") || Name.startswith("avx2.psubs.") || Name.startswith("avx512.padds.") || Name.startswith("avx512.psubs.") || Name.startswith("avx512.mask.padds.") || Name.startswith("avx512.mask.psubs."))) { bool IsAdd = Name.contains(".padds"); Rep = UpgradeX86AddSubSatIntrinsics(Builder, *CI, true, IsAdd); } else if (IsX86 && (Name.startswith("sse2.paddus.") || Name.startswith("sse2.psubus.") || Name.startswith("avx2.paddus.") || Name.startswith("avx2.psubus.") || Name.startswith("avx512.mask.paddus.") || Name.startswith("avx512.mask.psubus."))) { bool IsAdd = Name.contains(".paddus"); Rep = UpgradeX86AddSubSatIntrinsics(Builder, *CI, false, IsAdd); } else if (IsX86 && Name.startswith("avx512.mask.palignr.")) { Rep = UpgradeX86ALIGNIntrinsics(Builder, CI->getArgOperand(0), CI->getArgOperand(1), CI->getArgOperand(2), CI->getArgOperand(3), CI->getArgOperand(4), false); } else if (IsX86 && Name.startswith("avx512.mask.valign.")) { Rep = UpgradeX86ALIGNIntrinsics(Builder, CI->getArgOperand(0), CI->getArgOperand(1), CI->getArgOperand(2), CI->getArgOperand(3), CI->getArgOperand(4), true); } else if (IsX86 && (Name == "sse2.psll.dq" || Name == "avx2.psll.dq")) { // 128/256-bit shift left specified in bits. unsigned Shift = cast(CI->getArgOperand(1))->getZExtValue(); Rep = UpgradeX86PSLLDQIntrinsics(Builder, CI->getArgOperand(0), Shift / 8); // Shift is in bits. } else if (IsX86 && (Name == "sse2.psrl.dq" || Name == "avx2.psrl.dq")) { // 128/256-bit shift right specified in bits. unsigned Shift = cast(CI->getArgOperand(1))->getZExtValue(); Rep = UpgradeX86PSRLDQIntrinsics(Builder, CI->getArgOperand(0), Shift / 8); // Shift is in bits. } else if (IsX86 && (Name == "sse2.psll.dq.bs" || Name == "avx2.psll.dq.bs" || Name == "avx512.psll.dq.512")) { // 128/256/512-bit shift left specified in bytes. unsigned Shift = cast(CI->getArgOperand(1))->getZExtValue(); Rep = UpgradeX86PSLLDQIntrinsics(Builder, CI->getArgOperand(0), Shift); } else if (IsX86 && (Name == "sse2.psrl.dq.bs" || Name == "avx2.psrl.dq.bs" || Name == "avx512.psrl.dq.512")) { // 128/256/512-bit shift right specified in bytes. unsigned Shift = cast(CI->getArgOperand(1))->getZExtValue(); Rep = UpgradeX86PSRLDQIntrinsics(Builder, CI->getArgOperand(0), Shift); } else if (IsX86 && (Name == "sse41.pblendw" || Name.startswith("sse41.blendp") || Name.startswith("avx.blend.p") || Name == "avx2.pblendw" || Name.startswith("avx2.pblendd."))) { Value *Op0 = CI->getArgOperand(0); Value *Op1 = CI->getArgOperand(1); unsigned Imm = cast (CI->getArgOperand(2))->getZExtValue(); VectorType *VecTy = cast(CI->getType()); unsigned NumElts = VecTy->getNumElements(); SmallVector Idxs(NumElts); for (unsigned i = 0; i != NumElts; ++i) Idxs[i] = ((Imm >> (i%8)) & 1) ? i + NumElts : i; Rep = Builder.CreateShuffleVector(Op0, Op1, Idxs); } else if (IsX86 && (Name.startswith("avx.vinsertf128.") || Name == "avx2.vinserti128" || Name.startswith("avx512.mask.insert"))) { Value *Op0 = CI->getArgOperand(0); Value *Op1 = CI->getArgOperand(1); unsigned Imm = cast(CI->getArgOperand(2))->getZExtValue(); unsigned DstNumElts = CI->getType()->getVectorNumElements(); unsigned SrcNumElts = Op1->getType()->getVectorNumElements(); unsigned Scale = DstNumElts / SrcNumElts; // Mask off the high bits of the immediate value; hardware ignores those. Imm = Imm % Scale; // Extend the second operand into a vector the size of the destination. Value *UndefV = UndefValue::get(Op1->getType()); SmallVector Idxs(DstNumElts); for (unsigned i = 0; i != SrcNumElts; ++i) Idxs[i] = i; for (unsigned i = SrcNumElts; i != DstNumElts; ++i) Idxs[i] = SrcNumElts; Rep = Builder.CreateShuffleVector(Op1, UndefV, Idxs); // Insert the second operand into the first operand. // Note that there is no guarantee that instruction lowering will actually // produce a vinsertf128 instruction for the created shuffles. In // particular, the 0 immediate case involves no lane changes, so it can // be handled as a blend. // Example of shuffle mask for 32-bit elements: // Imm = 1 // Imm = 0 // First fill with identify mask. for (unsigned i = 0; i != DstNumElts; ++i) Idxs[i] = i; // Then replace the elements where we need to insert. for (unsigned i = 0; i != SrcNumElts; ++i) Idxs[i + Imm * SrcNumElts] = i + DstNumElts; Rep = Builder.CreateShuffleVector(Op0, Rep, Idxs); // If the intrinsic has a mask operand, handle that. if (CI->getNumArgOperands() == 5) Rep = EmitX86Select(Builder, CI->getArgOperand(4), Rep, CI->getArgOperand(3)); } else if (IsX86 && (Name.startswith("avx.vextractf128.") || Name == "avx2.vextracti128" || Name.startswith("avx512.mask.vextract"))) { Value *Op0 = CI->getArgOperand(0); unsigned Imm = cast(CI->getArgOperand(1))->getZExtValue(); unsigned DstNumElts = CI->getType()->getVectorNumElements(); unsigned SrcNumElts = Op0->getType()->getVectorNumElements(); unsigned Scale = SrcNumElts / DstNumElts; // Mask off the high bits of the immediate value; hardware ignores those. Imm = Imm % Scale; // Get indexes for the subvector of the input vector. SmallVector Idxs(DstNumElts); for (unsigned i = 0; i != DstNumElts; ++i) { Idxs[i] = i + (Imm * DstNumElts); } Rep = Builder.CreateShuffleVector(Op0, Op0, Idxs); // If the intrinsic has a mask operand, handle that. if (CI->getNumArgOperands() == 4) Rep = EmitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2)); } else if (!IsX86 && Name == "stackprotectorcheck") { Rep = nullptr; } else if (IsX86 && (Name.startswith("avx512.mask.perm.df.") || Name.startswith("avx512.mask.perm.di."))) { Value *Op0 = CI->getArgOperand(0); unsigned Imm = cast(CI->getArgOperand(1))->getZExtValue(); VectorType *VecTy = cast(CI->getType()); unsigned NumElts = VecTy->getNumElements(); SmallVector Idxs(NumElts); for (unsigned i = 0; i != NumElts; ++i) Idxs[i] = (i & ~0x3) + ((Imm >> (2 * (i & 0x3))) & 3); Rep = Builder.CreateShuffleVector(Op0, Op0, Idxs); if (CI->getNumArgOperands() == 4) Rep = EmitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2)); } else if (IsX86 && (Name.startswith("avx.vperm2f128.") || Name == "avx2.vperm2i128")) { // The immediate permute control byte looks like this: // [1:0] - select 128 bits from sources for low half of destination // [2] - ignore // [3] - zero low half of destination // [5:4] - select 128 bits from sources for high half of destination // [6] - ignore // [7] - zero high half of destination uint8_t Imm = cast(CI->getArgOperand(2))->getZExtValue(); unsigned NumElts = CI->getType()->getVectorNumElements(); unsigned HalfSize = NumElts / 2; SmallVector ShuffleMask(NumElts); // Determine which operand(s) are actually in use for this instruction. Value *V0 = (Imm & 0x02) ? CI->getArgOperand(1) : CI->getArgOperand(0); Value *V1 = (Imm & 0x20) ? CI->getArgOperand(1) : CI->getArgOperand(0); // If needed, replace operands based on zero mask. V0 = (Imm & 0x08) ? ConstantAggregateZero::get(CI->getType()) : V0; V1 = (Imm & 0x80) ? ConstantAggregateZero::get(CI->getType()) : V1; // Permute low half of result. unsigned StartIndex = (Imm & 0x01) ? HalfSize : 0; for (unsigned i = 0; i < HalfSize; ++i) ShuffleMask[i] = StartIndex + i; // Permute high half of result. StartIndex = (Imm & 0x10) ? HalfSize : 0; for (unsigned i = 0; i < HalfSize; ++i) ShuffleMask[i + HalfSize] = NumElts + StartIndex + i; Rep = Builder.CreateShuffleVector(V0, V1, ShuffleMask); } else if (IsX86 && (Name.startswith("avx.vpermil.") || Name == "sse2.pshuf.d" || Name.startswith("avx512.mask.vpermil.p") || Name.startswith("avx512.mask.pshuf.d."))) { Value *Op0 = CI->getArgOperand(0); unsigned Imm = cast(CI->getArgOperand(1))->getZExtValue(); VectorType *VecTy = cast(CI->getType()); unsigned NumElts = VecTy->getNumElements(); // Calculate the size of each index in the immediate. unsigned IdxSize = 64 / VecTy->getScalarSizeInBits(); unsigned IdxMask = ((1 << IdxSize) - 1); SmallVector Idxs(NumElts); // Lookup the bits for this element, wrapping around the immediate every // 8-bits. Elements are grouped into sets of 2 or 4 elements so we need // to offset by the first index of each group. for (unsigned i = 0; i != NumElts; ++i) Idxs[i] = ((Imm >> ((i * IdxSize) % 8)) & IdxMask) | (i & ~IdxMask); Rep = Builder.CreateShuffleVector(Op0, Op0, Idxs); if (CI->getNumArgOperands() == 4) Rep = EmitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2)); } else if (IsX86 && (Name == "sse2.pshufl.w" || Name.startswith("avx512.mask.pshufl.w."))) { Value *Op0 = CI->getArgOperand(0); unsigned Imm = cast(CI->getArgOperand(1))->getZExtValue(); unsigned NumElts = CI->getType()->getVectorNumElements(); SmallVector Idxs(NumElts); for (unsigned l = 0; l != NumElts; l += 8) { for (unsigned i = 0; i != 4; ++i) Idxs[i + l] = ((Imm >> (2 * i)) & 0x3) + l; for (unsigned i = 4; i != 8; ++i) Idxs[i + l] = i + l; } Rep = Builder.CreateShuffleVector(Op0, Op0, Idxs); if (CI->getNumArgOperands() == 4) Rep = EmitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2)); } else if (IsX86 && (Name == "sse2.pshufh.w" || Name.startswith("avx512.mask.pshufh.w."))) { Value *Op0 = CI->getArgOperand(0); unsigned Imm = cast(CI->getArgOperand(1))->getZExtValue(); unsigned NumElts = CI->getType()->getVectorNumElements(); SmallVector Idxs(NumElts); for (unsigned l = 0; l != NumElts; l += 8) { for (unsigned i = 0; i != 4; ++i) Idxs[i + l] = i + l; for (unsigned i = 0; i != 4; ++i) Idxs[i + l + 4] = ((Imm >> (2 * i)) & 0x3) + 4 + l; } Rep = Builder.CreateShuffleVector(Op0, Op0, Idxs); if (CI->getNumArgOperands() == 4) Rep = EmitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2)); } else if (IsX86 && Name.startswith("avx512.mask.shuf.p")) { Value *Op0 = CI->getArgOperand(0); Value *Op1 = CI->getArgOperand(1); unsigned Imm = cast(CI->getArgOperand(2))->getZExtValue(); unsigned NumElts = CI->getType()->getVectorNumElements(); unsigned NumLaneElts = 128/CI->getType()->getScalarSizeInBits(); unsigned HalfLaneElts = NumLaneElts / 2; SmallVector Idxs(NumElts); for (unsigned i = 0; i != NumElts; ++i) { // Base index is the starting element of the lane. Idxs[i] = i - (i % NumLaneElts); // If we are half way through the lane switch to the other source. if ((i % NumLaneElts) >= HalfLaneElts) Idxs[i] += NumElts; // Now select the specific element. By adding HalfLaneElts bits from // the immediate. Wrapping around the immediate every 8-bits. Idxs[i] += (Imm >> ((i * HalfLaneElts) % 8)) & ((1 << HalfLaneElts) - 1); } Rep = Builder.CreateShuffleVector(Op0, Op1, Idxs); Rep = EmitX86Select(Builder, CI->getArgOperand(4), Rep, CI->getArgOperand(3)); } else if (IsX86 && (Name.startswith("avx512.mask.movddup") || Name.startswith("avx512.mask.movshdup") || Name.startswith("avx512.mask.movsldup"))) { Value *Op0 = CI->getArgOperand(0); unsigned NumElts = CI->getType()->getVectorNumElements(); unsigned NumLaneElts = 128/CI->getType()->getScalarSizeInBits(); unsigned Offset = 0; if (Name.startswith("avx512.mask.movshdup.")) Offset = 1; SmallVector Idxs(NumElts); for (unsigned l = 0; l != NumElts; l += NumLaneElts) for (unsigned i = 0; i != NumLaneElts; i += 2) { Idxs[i + l + 0] = i + l + Offset; Idxs[i + l + 1] = i + l + Offset; } Rep = Builder.CreateShuffleVector(Op0, Op0, Idxs); Rep = EmitX86Select(Builder, CI->getArgOperand(2), Rep, CI->getArgOperand(1)); } else if (IsX86 && (Name.startswith("avx512.mask.punpckl") || Name.startswith("avx512.mask.unpckl."))) { Value *Op0 = CI->getArgOperand(0); Value *Op1 = CI->getArgOperand(1); int NumElts = CI->getType()->getVectorNumElements(); int NumLaneElts = 128/CI->getType()->getScalarSizeInBits(); SmallVector Idxs(NumElts); for (int l = 0; l != NumElts; l += NumLaneElts) for (int i = 0; i != NumLaneElts; ++i) Idxs[i + l] = l + (i / 2) + NumElts * (i % 2); Rep = Builder.CreateShuffleVector(Op0, Op1, Idxs); Rep = EmitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2)); } else if (IsX86 && (Name.startswith("avx512.mask.punpckh") || Name.startswith("avx512.mask.unpckh."))) { Value *Op0 = CI->getArgOperand(0); Value *Op1 = CI->getArgOperand(1); int NumElts = CI->getType()->getVectorNumElements(); int NumLaneElts = 128/CI->getType()->getScalarSizeInBits(); SmallVector Idxs(NumElts); for (int l = 0; l != NumElts; l += NumLaneElts) for (int i = 0; i != NumLaneElts; ++i) Idxs[i + l] = (NumLaneElts / 2) + l + (i / 2) + NumElts * (i % 2); Rep = Builder.CreateShuffleVector(Op0, Op1, Idxs); Rep = EmitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2)); } else if (IsX86 && (Name.startswith("avx512.mask.and.") || Name.startswith("avx512.mask.pand."))) { VectorType *FTy = cast(CI->getType()); VectorType *ITy = VectorType::getInteger(FTy); Rep = Builder.CreateAnd(Builder.CreateBitCast(CI->getArgOperand(0), ITy), Builder.CreateBitCast(CI->getArgOperand(1), ITy)); Rep = Builder.CreateBitCast(Rep, FTy); Rep = EmitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2)); } else if (IsX86 && (Name.startswith("avx512.mask.andn.") || Name.startswith("avx512.mask.pandn."))) { VectorType *FTy = cast(CI->getType()); VectorType *ITy = VectorType::getInteger(FTy); Rep = Builder.CreateNot(Builder.CreateBitCast(CI->getArgOperand(0), ITy)); Rep = Builder.CreateAnd(Rep, Builder.CreateBitCast(CI->getArgOperand(1), ITy)); Rep = Builder.CreateBitCast(Rep, FTy); Rep = EmitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2)); } else if (IsX86 && (Name.startswith("avx512.mask.or.") || Name.startswith("avx512.mask.por."))) { VectorType *FTy = cast(CI->getType()); VectorType *ITy = VectorType::getInteger(FTy); Rep = Builder.CreateOr(Builder.CreateBitCast(CI->getArgOperand(0), ITy), Builder.CreateBitCast(CI->getArgOperand(1), ITy)); Rep = Builder.CreateBitCast(Rep, FTy); Rep = EmitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2)); } else if (IsX86 && (Name.startswith("avx512.mask.xor.") || Name.startswith("avx512.mask.pxor."))) { VectorType *FTy = cast(CI->getType()); VectorType *ITy = VectorType::getInteger(FTy); Rep = Builder.CreateXor(Builder.CreateBitCast(CI->getArgOperand(0), ITy), Builder.CreateBitCast(CI->getArgOperand(1), ITy)); Rep = Builder.CreateBitCast(Rep, FTy); Rep = EmitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2)); } else if (IsX86 && Name.startswith("avx512.mask.padd.")) { Rep = Builder.CreateAdd(CI->getArgOperand(0), CI->getArgOperand(1)); Rep = EmitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2)); } else if (IsX86 && Name.startswith("avx512.mask.psub.")) { Rep = Builder.CreateSub(CI->getArgOperand(0), CI->getArgOperand(1)); Rep = EmitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2)); } else if (IsX86 && Name.startswith("avx512.mask.pmull.")) { Rep = Builder.CreateMul(CI->getArgOperand(0), CI->getArgOperand(1)); Rep = EmitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2)); } else if (IsX86 && Name.startswith("avx512.mask.add.p")) { if (Name.endswith(".512")) { Intrinsic::ID IID; if (Name[17] == 's') IID = Intrinsic::x86_avx512_add_ps_512; else IID = Intrinsic::x86_avx512_add_pd_512; Rep = Builder.CreateCall(Intrinsic::getDeclaration(F->getParent(), IID), { CI->getArgOperand(0), CI->getArgOperand(1), CI->getArgOperand(4) }); } else { Rep = Builder.CreateFAdd(CI->getArgOperand(0), CI->getArgOperand(1)); } Rep = EmitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2)); } else if (IsX86 && Name.startswith("avx512.mask.div.p")) { if (Name.endswith(".512")) { Intrinsic::ID IID; if (Name[17] == 's') IID = Intrinsic::x86_avx512_div_ps_512; else IID = Intrinsic::x86_avx512_div_pd_512; Rep = Builder.CreateCall(Intrinsic::getDeclaration(F->getParent(), IID), { CI->getArgOperand(0), CI->getArgOperand(1), CI->getArgOperand(4) }); } else { Rep = Builder.CreateFDiv(CI->getArgOperand(0), CI->getArgOperand(1)); } Rep = EmitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2)); } else if (IsX86 && Name.startswith("avx512.mask.mul.p")) { if (Name.endswith(".512")) { Intrinsic::ID IID; if (Name[17] == 's') IID = Intrinsic::x86_avx512_mul_ps_512; else IID = Intrinsic::x86_avx512_mul_pd_512; Rep = Builder.CreateCall(Intrinsic::getDeclaration(F->getParent(), IID), { CI->getArgOperand(0), CI->getArgOperand(1), CI->getArgOperand(4) }); } else { Rep = Builder.CreateFMul(CI->getArgOperand(0), CI->getArgOperand(1)); } Rep = EmitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2)); } else if (IsX86 && Name.startswith("avx512.mask.sub.p")) { if (Name.endswith(".512")) { Intrinsic::ID IID; if (Name[17] == 's') IID = Intrinsic::x86_avx512_sub_ps_512; else IID = Intrinsic::x86_avx512_sub_pd_512; Rep = Builder.CreateCall(Intrinsic::getDeclaration(F->getParent(), IID), { CI->getArgOperand(0), CI->getArgOperand(1), CI->getArgOperand(4) }); } else { Rep = Builder.CreateFSub(CI->getArgOperand(0), CI->getArgOperand(1)); } Rep = EmitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2)); } else if (IsX86 && (Name.startswith("avx512.mask.max.p") || Name.startswith("avx512.mask.min.p")) && Name.drop_front(18) == ".512") { bool IsDouble = Name[17] == 'd'; bool IsMin = Name[13] == 'i'; static const Intrinsic::ID MinMaxTbl[2][2] = { { Intrinsic::x86_avx512_max_ps_512, Intrinsic::x86_avx512_max_pd_512 }, { Intrinsic::x86_avx512_min_ps_512, Intrinsic::x86_avx512_min_pd_512 } }; Intrinsic::ID IID = MinMaxTbl[IsMin][IsDouble]; Rep = Builder.CreateCall(Intrinsic::getDeclaration(F->getParent(), IID), { CI->getArgOperand(0), CI->getArgOperand(1), CI->getArgOperand(4) }); Rep = EmitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2)); } else if (IsX86 && Name.startswith("avx512.mask.lzcnt.")) { Rep = Builder.CreateCall(Intrinsic::getDeclaration(F->getParent(), Intrinsic::ctlz, CI->getType()), { CI->getArgOperand(0), Builder.getInt1(false) }); Rep = EmitX86Select(Builder, CI->getArgOperand(2), Rep, CI->getArgOperand(1)); } else if (IsX86 && Name.startswith("avx512.mask.psll")) { bool IsImmediate = Name[16] == 'i' || (Name.size() > 18 && Name[18] == 'i'); bool IsVariable = Name[16] == 'v'; char Size = Name[16] == '.' ? Name[17] : Name[17] == '.' ? Name[18] : Name[18] == '.' ? Name[19] : Name[20]; Intrinsic::ID IID; if (IsVariable && Name[17] != '.') { if (Size == 'd' && Name[17] == '2') // avx512.mask.psllv2.di IID = Intrinsic::x86_avx2_psllv_q; else if (Size == 'd' && Name[17] == '4') // avx512.mask.psllv4.di IID = Intrinsic::x86_avx2_psllv_q_256; else if (Size == 's' && Name[17] == '4') // avx512.mask.psllv4.si IID = Intrinsic::x86_avx2_psllv_d; else if (Size == 's' && Name[17] == '8') // avx512.mask.psllv8.si IID = Intrinsic::x86_avx2_psllv_d_256; else if (Size == 'h' && Name[17] == '8') // avx512.mask.psllv8.hi IID = Intrinsic::x86_avx512_psllv_w_128; else if (Size == 'h' && Name[17] == '1') // avx512.mask.psllv16.hi IID = Intrinsic::x86_avx512_psllv_w_256; else if (Name[17] == '3' && Name[18] == '2') // avx512.mask.psllv32hi IID = Intrinsic::x86_avx512_psllv_w_512; else llvm_unreachable("Unexpected size"); } else if (Name.endswith(".128")) { if (Size == 'd') // avx512.mask.psll.d.128, avx512.mask.psll.di.128 IID = IsImmediate ? Intrinsic::x86_sse2_pslli_d : Intrinsic::x86_sse2_psll_d; else if (Size == 'q') // avx512.mask.psll.q.128, avx512.mask.psll.qi.128 IID = IsImmediate ? Intrinsic::x86_sse2_pslli_q : Intrinsic::x86_sse2_psll_q; else if (Size == 'w') // avx512.mask.psll.w.128, avx512.mask.psll.wi.128 IID = IsImmediate ? Intrinsic::x86_sse2_pslli_w : Intrinsic::x86_sse2_psll_w; else llvm_unreachable("Unexpected size"); } else if (Name.endswith(".256")) { if (Size == 'd') // avx512.mask.psll.d.256, avx512.mask.psll.di.256 IID = IsImmediate ? Intrinsic::x86_avx2_pslli_d : Intrinsic::x86_avx2_psll_d; else if (Size == 'q') // avx512.mask.psll.q.256, avx512.mask.psll.qi.256 IID = IsImmediate ? Intrinsic::x86_avx2_pslli_q : Intrinsic::x86_avx2_psll_q; else if (Size == 'w') // avx512.mask.psll.w.256, avx512.mask.psll.wi.256 IID = IsImmediate ? Intrinsic::x86_avx2_pslli_w : Intrinsic::x86_avx2_psll_w; else llvm_unreachable("Unexpected size"); } else { if (Size == 'd') // psll.di.512, pslli.d, psll.d, psllv.d.512 IID = IsImmediate ? Intrinsic::x86_avx512_pslli_d_512 : IsVariable ? Intrinsic::x86_avx512_psllv_d_512 : Intrinsic::x86_avx512_psll_d_512; else if (Size == 'q') // psll.qi.512, pslli.q, psll.q, psllv.q.512 IID = IsImmediate ? Intrinsic::x86_avx512_pslli_q_512 : IsVariable ? Intrinsic::x86_avx512_psllv_q_512 : Intrinsic::x86_avx512_psll_q_512; else if (Size == 'w') // psll.wi.512, pslli.w, psll.w IID = IsImmediate ? Intrinsic::x86_avx512_pslli_w_512 : Intrinsic::x86_avx512_psll_w_512; else llvm_unreachable("Unexpected size"); } Rep = UpgradeX86MaskedShift(Builder, *CI, IID); } else if (IsX86 && Name.startswith("avx512.mask.psrl")) { bool IsImmediate = Name[16] == 'i' || (Name.size() > 18 && Name[18] == 'i'); bool IsVariable = Name[16] == 'v'; char Size = Name[16] == '.' ? Name[17] : Name[17] == '.' ? Name[18] : Name[18] == '.' ? Name[19] : Name[20]; Intrinsic::ID IID; if (IsVariable && Name[17] != '.') { if (Size == 'd' && Name[17] == '2') // avx512.mask.psrlv2.di IID = Intrinsic::x86_avx2_psrlv_q; else if (Size == 'd' && Name[17] == '4') // avx512.mask.psrlv4.di IID = Intrinsic::x86_avx2_psrlv_q_256; else if (Size == 's' && Name[17] == '4') // avx512.mask.psrlv4.si IID = Intrinsic::x86_avx2_psrlv_d; else if (Size == 's' && Name[17] == '8') // avx512.mask.psrlv8.si IID = Intrinsic::x86_avx2_psrlv_d_256; else if (Size == 'h' && Name[17] == '8') // avx512.mask.psrlv8.hi IID = Intrinsic::x86_avx512_psrlv_w_128; else if (Size == 'h' && Name[17] == '1') // avx512.mask.psrlv16.hi IID = Intrinsic::x86_avx512_psrlv_w_256; else if (Name[17] == '3' && Name[18] == '2') // avx512.mask.psrlv32hi IID = Intrinsic::x86_avx512_psrlv_w_512; else llvm_unreachable("Unexpected size"); } else if (Name.endswith(".128")) { if (Size == 'd') // avx512.mask.psrl.d.128, avx512.mask.psrl.di.128 IID = IsImmediate ? Intrinsic::x86_sse2_psrli_d : Intrinsic::x86_sse2_psrl_d; else if (Size == 'q') // avx512.mask.psrl.q.128, avx512.mask.psrl.qi.128 IID = IsImmediate ? Intrinsic::x86_sse2_psrli_q : Intrinsic::x86_sse2_psrl_q; else if (Size == 'w') // avx512.mask.psrl.w.128, avx512.mask.psrl.wi.128 IID = IsImmediate ? Intrinsic::x86_sse2_psrli_w : Intrinsic::x86_sse2_psrl_w; else llvm_unreachable("Unexpected size"); } else if (Name.endswith(".256")) { if (Size == 'd') // avx512.mask.psrl.d.256, avx512.mask.psrl.di.256 IID = IsImmediate ? Intrinsic::x86_avx2_psrli_d : Intrinsic::x86_avx2_psrl_d; else if (Size == 'q') // avx512.mask.psrl.q.256, avx512.mask.psrl.qi.256 IID = IsImmediate ? Intrinsic::x86_avx2_psrli_q : Intrinsic::x86_avx2_psrl_q; else if (Size == 'w') // avx512.mask.psrl.w.256, avx512.mask.psrl.wi.256 IID = IsImmediate ? Intrinsic::x86_avx2_psrli_w : Intrinsic::x86_avx2_psrl_w; else llvm_unreachable("Unexpected size"); } else { if (Size == 'd') // psrl.di.512, psrli.d, psrl.d, psrl.d.512 IID = IsImmediate ? Intrinsic::x86_avx512_psrli_d_512 : IsVariable ? Intrinsic::x86_avx512_psrlv_d_512 : Intrinsic::x86_avx512_psrl_d_512; else if (Size == 'q') // psrl.qi.512, psrli.q, psrl.q, psrl.q.512 IID = IsImmediate ? Intrinsic::x86_avx512_psrli_q_512 : IsVariable ? Intrinsic::x86_avx512_psrlv_q_512 : Intrinsic::x86_avx512_psrl_q_512; else if (Size == 'w') // psrl.wi.512, psrli.w, psrl.w) IID = IsImmediate ? Intrinsic::x86_avx512_psrli_w_512 : Intrinsic::x86_avx512_psrl_w_512; else llvm_unreachable("Unexpected size"); } Rep = UpgradeX86MaskedShift(Builder, *CI, IID); } else if (IsX86 && Name.startswith("avx512.mask.psra")) { bool IsImmediate = Name[16] == 'i' || (Name.size() > 18 && Name[18] == 'i'); bool IsVariable = Name[16] == 'v'; char Size = Name[16] == '.' ? Name[17] : Name[17] == '.' ? Name[18] : Name[18] == '.' ? Name[19] : Name[20]; Intrinsic::ID IID; if (IsVariable && Name[17] != '.') { if (Size == 's' && Name[17] == '4') // avx512.mask.psrav4.si IID = Intrinsic::x86_avx2_psrav_d; else if (Size == 's' && Name[17] == '8') // avx512.mask.psrav8.si IID = Intrinsic::x86_avx2_psrav_d_256; else if (Size == 'h' && Name[17] == '8') // avx512.mask.psrav8.hi IID = Intrinsic::x86_avx512_psrav_w_128; else if (Size == 'h' && Name[17] == '1') // avx512.mask.psrav16.hi IID = Intrinsic::x86_avx512_psrav_w_256; else if (Name[17] == '3' && Name[18] == '2') // avx512.mask.psrav32hi IID = Intrinsic::x86_avx512_psrav_w_512; else llvm_unreachable("Unexpected size"); } else if (Name.endswith(".128")) { if (Size == 'd') // avx512.mask.psra.d.128, avx512.mask.psra.di.128 IID = IsImmediate ? Intrinsic::x86_sse2_psrai_d : Intrinsic::x86_sse2_psra_d; else if (Size == 'q') // avx512.mask.psra.q.128, avx512.mask.psra.qi.128 IID = IsImmediate ? Intrinsic::x86_avx512_psrai_q_128 : IsVariable ? Intrinsic::x86_avx512_psrav_q_128 : Intrinsic::x86_avx512_psra_q_128; else if (Size == 'w') // avx512.mask.psra.w.128, avx512.mask.psra.wi.128 IID = IsImmediate ? Intrinsic::x86_sse2_psrai_w : Intrinsic::x86_sse2_psra_w; else llvm_unreachable("Unexpected size"); } else if (Name.endswith(".256")) { if (Size == 'd') // avx512.mask.psra.d.256, avx512.mask.psra.di.256 IID = IsImmediate ? Intrinsic::x86_avx2_psrai_d : Intrinsic::x86_avx2_psra_d; else if (Size == 'q') // avx512.mask.psra.q.256, avx512.mask.psra.qi.256 IID = IsImmediate ? Intrinsic::x86_avx512_psrai_q_256 : IsVariable ? Intrinsic::x86_avx512_psrav_q_256 : Intrinsic::x86_avx512_psra_q_256; else if (Size == 'w') // avx512.mask.psra.w.256, avx512.mask.psra.wi.256 IID = IsImmediate ? Intrinsic::x86_avx2_psrai_w : Intrinsic::x86_avx2_psra_w; else llvm_unreachable("Unexpected size"); } else { if (Size == 'd') // psra.di.512, psrai.d, psra.d, psrav.d.512 IID = IsImmediate ? Intrinsic::x86_avx512_psrai_d_512 : IsVariable ? Intrinsic::x86_avx512_psrav_d_512 : Intrinsic::x86_avx512_psra_d_512; else if (Size == 'q') // psra.qi.512, psrai.q, psra.q IID = IsImmediate ? Intrinsic::x86_avx512_psrai_q_512 : IsVariable ? Intrinsic::x86_avx512_psrav_q_512 : Intrinsic::x86_avx512_psra_q_512; else if (Size == 'w') // psra.wi.512, psrai.w, psra.w IID = IsImmediate ? Intrinsic::x86_avx512_psrai_w_512 : Intrinsic::x86_avx512_psra_w_512; else llvm_unreachable("Unexpected size"); } Rep = UpgradeX86MaskedShift(Builder, *CI, IID); } else if (IsX86 && Name.startswith("avx512.mask.move.s")) { Rep = upgradeMaskedMove(Builder, *CI); } else if (IsX86 && Name.startswith("avx512.cvtmask2")) { Rep = UpgradeMaskToInt(Builder, *CI); } else if (IsX86 && Name.endswith(".movntdqa")) { Module *M = F->getParent(); MDNode *Node = MDNode::get( C, ConstantAsMetadata::get(ConstantInt::get(Type::getInt32Ty(C), 1))); Value *Ptr = CI->getArgOperand(0); VectorType *VTy = cast(CI->getType()); // Convert the type of the pointer to a pointer to the stored type. Value *BC = Builder.CreateBitCast(Ptr, PointerType::getUnqual(VTy), "cast"); LoadInst *LI = Builder.CreateAlignedLoad(BC, VTy->getBitWidth() / 8); LI->setMetadata(M->getMDKindID("nontemporal"), Node); Rep = LI; } else if (IsX86 && (Name.startswith("sse2.pavg") || Name.startswith("avx2.pavg") || Name.startswith("avx512.mask.pavg"))) { // llvm.x86.sse2.pavg.b/w, llvm.x86.avx2.pavg.b/w, // llvm.x86.avx512.mask.pavg.b/w Value *A = CI->getArgOperand(0); Value *B = CI->getArgOperand(1); VectorType *ZextType = VectorType::getExtendedElementVectorType( cast(A->getType())); Value *ExtendedA = Builder.CreateZExt(A, ZextType); Value *ExtendedB = Builder.CreateZExt(B, ZextType); Value *Sum = Builder.CreateAdd(ExtendedA, ExtendedB); Value *AddOne = Builder.CreateAdd(Sum, ConstantInt::get(ZextType, 1)); Value *ShiftR = Builder.CreateLShr(AddOne, ConstantInt::get(ZextType, 1)); Rep = Builder.CreateTrunc(ShiftR, A->getType()); if (CI->getNumArgOperands() > 2) { Rep = EmitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2)); } } else if (IsX86 && (Name.startswith("fma.vfmadd.") || Name.startswith("fma.vfmsub.") || Name.startswith("fma.vfnmadd.") || Name.startswith("fma.vfnmsub."))) { bool NegMul = Name[6] == 'n'; bool NegAcc = NegMul ? Name[8] == 's' : Name[7] == 's'; bool IsScalar = NegMul ? Name[12] == 's' : Name[11] == 's'; Value *Ops[] = { CI->getArgOperand(0), CI->getArgOperand(1), CI->getArgOperand(2) }; if (IsScalar) { Ops[0] = Builder.CreateExtractElement(Ops[0], (uint64_t)0); Ops[1] = Builder.CreateExtractElement(Ops[1], (uint64_t)0); Ops[2] = Builder.CreateExtractElement(Ops[2], (uint64_t)0); } if (NegMul && !IsScalar) Ops[0] = Builder.CreateFNeg(Ops[0]); if (NegMul && IsScalar) Ops[1] = Builder.CreateFNeg(Ops[1]); if (NegAcc) Ops[2] = Builder.CreateFNeg(Ops[2]); Rep = Builder.CreateCall(Intrinsic::getDeclaration(CI->getModule(), Intrinsic::fma, Ops[0]->getType()), Ops); if (IsScalar) Rep = Builder.CreateInsertElement(CI->getArgOperand(0), Rep, (uint64_t)0); } else if (IsX86 && Name.startswith("fma4.vfmadd.s")) { Value *Ops[] = { CI->getArgOperand(0), CI->getArgOperand(1), CI->getArgOperand(2) }; Ops[0] = Builder.CreateExtractElement(Ops[0], (uint64_t)0); Ops[1] = Builder.CreateExtractElement(Ops[1], (uint64_t)0); Ops[2] = Builder.CreateExtractElement(Ops[2], (uint64_t)0); Rep = Builder.CreateCall(Intrinsic::getDeclaration(CI->getModule(), Intrinsic::fma, Ops[0]->getType()), Ops); Rep = Builder.CreateInsertElement(Constant::getNullValue(CI->getType()), Rep, (uint64_t)0); } else if (IsX86 && (Name.startswith("avx512.mask.vfmadd.s") || Name.startswith("avx512.maskz.vfmadd.s") || Name.startswith("avx512.mask3.vfmadd.s") || Name.startswith("avx512.mask3.vfmsub.s") || Name.startswith("avx512.mask3.vfnmsub.s"))) { bool IsMask3 = Name[11] == '3'; bool IsMaskZ = Name[11] == 'z'; // Drop the "avx512.mask." to make it easier. Name = Name.drop_front(IsMask3 || IsMaskZ ? 13 : 12); bool NegMul = Name[2] == 'n'; bool NegAcc = NegMul ? Name[4] == 's' : Name[3] == 's'; Value *A = CI->getArgOperand(0); Value *B = CI->getArgOperand(1); Value *C = CI->getArgOperand(2); if (NegMul && (IsMask3 || IsMaskZ)) A = Builder.CreateFNeg(A); if (NegMul && !(IsMask3 || IsMaskZ)) B = Builder.CreateFNeg(B); if (NegAcc) C = Builder.CreateFNeg(C); A = Builder.CreateExtractElement(A, (uint64_t)0); B = Builder.CreateExtractElement(B, (uint64_t)0); C = Builder.CreateExtractElement(C, (uint64_t)0); if (!isa(CI->getArgOperand(4)) || cast(CI->getArgOperand(4))->getZExtValue() != 4) { Value *Ops[] = { A, B, C, CI->getArgOperand(4) }; Intrinsic::ID IID; if (Name.back() == 'd') IID = Intrinsic::x86_avx512_vfmadd_f64; else IID = Intrinsic::x86_avx512_vfmadd_f32; Function *FMA = Intrinsic::getDeclaration(CI->getModule(), IID); Rep = Builder.CreateCall(FMA, Ops); } else { Function *FMA = Intrinsic::getDeclaration(CI->getModule(), Intrinsic::fma, A->getType()); Rep = Builder.CreateCall(FMA, { A, B, C }); } Value *PassThru = IsMaskZ ? Constant::getNullValue(Rep->getType()) : IsMask3 ? C : A; // For Mask3 with NegAcc, we need to create a new extractelement that // avoids the negation above. if (NegAcc && IsMask3) PassThru = Builder.CreateExtractElement(CI->getArgOperand(2), (uint64_t)0); Rep = EmitX86ScalarSelect(Builder, CI->getArgOperand(3), Rep, PassThru); Rep = Builder.CreateInsertElement(CI->getArgOperand(IsMask3 ? 2 : 0), Rep, (uint64_t)0); } else if (IsX86 && (Name.startswith("avx512.mask.vfmadd.p") || Name.startswith("avx512.mask.vfnmadd.p") || Name.startswith("avx512.mask.vfnmsub.p") || Name.startswith("avx512.mask3.vfmadd.p") || Name.startswith("avx512.mask3.vfmsub.p") || Name.startswith("avx512.mask3.vfnmsub.p") || Name.startswith("avx512.maskz.vfmadd.p"))) { bool IsMask3 = Name[11] == '3'; bool IsMaskZ = Name[11] == 'z'; // Drop the "avx512.mask." to make it easier. Name = Name.drop_front(IsMask3 || IsMaskZ ? 13 : 12); bool NegMul = Name[2] == 'n'; bool NegAcc = NegMul ? Name[4] == 's' : Name[3] == 's'; Value *A = CI->getArgOperand(0); Value *B = CI->getArgOperand(1); Value *C = CI->getArgOperand(2); if (NegMul && (IsMask3 || IsMaskZ)) A = Builder.CreateFNeg(A); if (NegMul && !(IsMask3 || IsMaskZ)) B = Builder.CreateFNeg(B); if (NegAcc) C = Builder.CreateFNeg(C); if (CI->getNumArgOperands() == 5 && (!isa(CI->getArgOperand(4)) || cast(CI->getArgOperand(4))->getZExtValue() != 4)) { Intrinsic::ID IID; // Check the character before ".512" in string. if (Name[Name.size()-5] == 's') IID = Intrinsic::x86_avx512_vfmadd_ps_512; else IID = Intrinsic::x86_avx512_vfmadd_pd_512; Rep = Builder.CreateCall(Intrinsic::getDeclaration(F->getParent(), IID), { A, B, C, CI->getArgOperand(4) }); } else { Function *FMA = Intrinsic::getDeclaration(CI->getModule(), Intrinsic::fma, A->getType()); Rep = Builder.CreateCall(FMA, { A, B, C }); } Value *PassThru = IsMaskZ ? llvm::Constant::getNullValue(CI->getType()) : IsMask3 ? CI->getArgOperand(2) : CI->getArgOperand(0); Rep = EmitX86Select(Builder, CI->getArgOperand(3), Rep, PassThru); } else if (IsX86 && (Name.startswith("fma.vfmaddsub.p") || Name.startswith("fma.vfmsubadd.p"))) { bool IsSubAdd = Name[7] == 's'; int NumElts = CI->getType()->getVectorNumElements(); Value *Ops[] = { CI->getArgOperand(0), CI->getArgOperand(1), CI->getArgOperand(2) }; Function *FMA = Intrinsic::getDeclaration(CI->getModule(), Intrinsic::fma, Ops[0]->getType()); Value *Odd = Builder.CreateCall(FMA, Ops); Ops[2] = Builder.CreateFNeg(Ops[2]); Value *Even = Builder.CreateCall(FMA, Ops); if (IsSubAdd) std::swap(Even, Odd); SmallVector Idxs(NumElts); for (int i = 0; i != NumElts; ++i) Idxs[i] = i + (i % 2) * NumElts; Rep = Builder.CreateShuffleVector(Even, Odd, Idxs); } else if (IsX86 && (Name.startswith("avx512.mask.vfmaddsub.p") || Name.startswith("avx512.mask3.vfmaddsub.p") || Name.startswith("avx512.maskz.vfmaddsub.p") || Name.startswith("avx512.mask3.vfmsubadd.p"))) { bool IsMask3 = Name[11] == '3'; bool IsMaskZ = Name[11] == 'z'; // Drop the "avx512.mask." to make it easier. Name = Name.drop_front(IsMask3 || IsMaskZ ? 13 : 12); bool IsSubAdd = Name[3] == 's'; if (CI->getNumArgOperands() == 5 && (!isa(CI->getArgOperand(4)) || cast(CI->getArgOperand(4))->getZExtValue() != 4)) { Intrinsic::ID IID; // Check the character before ".512" in string. if (Name[Name.size()-5] == 's') IID = Intrinsic::x86_avx512_vfmaddsub_ps_512; else IID = Intrinsic::x86_avx512_vfmaddsub_pd_512; Value *Ops[] = { CI->getArgOperand(0), CI->getArgOperand(1), CI->getArgOperand(2), CI->getArgOperand(4) }; if (IsSubAdd) Ops[2] = Builder.CreateFNeg(Ops[2]); Rep = Builder.CreateCall(Intrinsic::getDeclaration(F->getParent(), IID), {CI->getArgOperand(0), CI->getArgOperand(1), CI->getArgOperand(2), CI->getArgOperand(4)}); } else { int NumElts = CI->getType()->getVectorNumElements(); Value *Ops[] = { CI->getArgOperand(0), CI->getArgOperand(1), CI->getArgOperand(2) }; Function *FMA = Intrinsic::getDeclaration(CI->getModule(), Intrinsic::fma, Ops[0]->getType()); Value *Odd = Builder.CreateCall(FMA, Ops); Ops[2] = Builder.CreateFNeg(Ops[2]); Value *Even = Builder.CreateCall(FMA, Ops); if (IsSubAdd) std::swap(Even, Odd); SmallVector Idxs(NumElts); for (int i = 0; i != NumElts; ++i) Idxs[i] = i + (i % 2) * NumElts; Rep = Builder.CreateShuffleVector(Even, Odd, Idxs); } Value *PassThru = IsMaskZ ? llvm::Constant::getNullValue(CI->getType()) : IsMask3 ? CI->getArgOperand(2) : CI->getArgOperand(0); Rep = EmitX86Select(Builder, CI->getArgOperand(3), Rep, PassThru); } else if (IsX86 && (Name.startswith("avx512.mask.pternlog.") || Name.startswith("avx512.maskz.pternlog."))) { bool ZeroMask = Name[11] == 'z'; unsigned VecWidth = CI->getType()->getPrimitiveSizeInBits(); unsigned EltWidth = CI->getType()->getScalarSizeInBits(); Intrinsic::ID IID; if (VecWidth == 128 && EltWidth == 32) IID = Intrinsic::x86_avx512_pternlog_d_128; else if (VecWidth == 256 && EltWidth == 32) IID = Intrinsic::x86_avx512_pternlog_d_256; else if (VecWidth == 512 && EltWidth == 32) IID = Intrinsic::x86_avx512_pternlog_d_512; else if (VecWidth == 128 && EltWidth == 64) IID = Intrinsic::x86_avx512_pternlog_q_128; else if (VecWidth == 256 && EltWidth == 64) IID = Intrinsic::x86_avx512_pternlog_q_256; else if (VecWidth == 512 && EltWidth == 64) IID = Intrinsic::x86_avx512_pternlog_q_512; else llvm_unreachable("Unexpected intrinsic"); Value *Args[] = { CI->getArgOperand(0) , CI->getArgOperand(1), CI->getArgOperand(2), CI->getArgOperand(3) }; Rep = Builder.CreateCall(Intrinsic::getDeclaration(CI->getModule(), IID), Args); Value *PassThru = ZeroMask ? ConstantAggregateZero::get(CI->getType()) : CI->getArgOperand(0); Rep = EmitX86Select(Builder, CI->getArgOperand(4), Rep, PassThru); } else if (IsX86 && (Name.startswith("avx512.mask.vpmadd52") || Name.startswith("avx512.maskz.vpmadd52"))) { bool ZeroMask = Name[11] == 'z'; bool High = Name[20] == 'h' || Name[21] == 'h'; unsigned VecWidth = CI->getType()->getPrimitiveSizeInBits(); Intrinsic::ID IID; if (VecWidth == 128 && !High) IID = Intrinsic::x86_avx512_vpmadd52l_uq_128; else if (VecWidth == 256 && !High) IID = Intrinsic::x86_avx512_vpmadd52l_uq_256; else if (VecWidth == 512 && !High) IID = Intrinsic::x86_avx512_vpmadd52l_uq_512; else if (VecWidth == 128 && High) IID = Intrinsic::x86_avx512_vpmadd52h_uq_128; else if (VecWidth == 256 && High) IID = Intrinsic::x86_avx512_vpmadd52h_uq_256; else if (VecWidth == 512 && High) IID = Intrinsic::x86_avx512_vpmadd52h_uq_512; else llvm_unreachable("Unexpected intrinsic"); Value *Args[] = { CI->getArgOperand(0) , CI->getArgOperand(1), CI->getArgOperand(2) }; Rep = Builder.CreateCall(Intrinsic::getDeclaration(CI->getModule(), IID), Args); Value *PassThru = ZeroMask ? ConstantAggregateZero::get(CI->getType()) : CI->getArgOperand(0); Rep = EmitX86Select(Builder, CI->getArgOperand(3), Rep, PassThru); } else if (IsX86 && (Name.startswith("avx512.mask.vpermi2var.") || Name.startswith("avx512.mask.vpermt2var.") || Name.startswith("avx512.maskz.vpermt2var."))) { bool ZeroMask = Name[11] == 'z'; bool IndexForm = Name[17] == 'i'; Rep = UpgradeX86VPERMT2Intrinsics(Builder, *CI, ZeroMask, IndexForm); } else if (IsX86 && (Name.startswith("avx512.mask.vpdpbusd.") || Name.startswith("avx512.maskz.vpdpbusd.") || Name.startswith("avx512.mask.vpdpbusds.") || Name.startswith("avx512.maskz.vpdpbusds."))) { bool ZeroMask = Name[11] == 'z'; bool IsSaturating = Name[ZeroMask ? 21 : 20] == 's'; unsigned VecWidth = CI->getType()->getPrimitiveSizeInBits(); Intrinsic::ID IID; if (VecWidth == 128 && !IsSaturating) IID = Intrinsic::x86_avx512_vpdpbusd_128; else if (VecWidth == 256 && !IsSaturating) IID = Intrinsic::x86_avx512_vpdpbusd_256; else if (VecWidth == 512 && !IsSaturating) IID = Intrinsic::x86_avx512_vpdpbusd_512; else if (VecWidth == 128 && IsSaturating) IID = Intrinsic::x86_avx512_vpdpbusds_128; else if (VecWidth == 256 && IsSaturating) IID = Intrinsic::x86_avx512_vpdpbusds_256; else if (VecWidth == 512 && IsSaturating) IID = Intrinsic::x86_avx512_vpdpbusds_512; else llvm_unreachable("Unexpected intrinsic"); Value *Args[] = { CI->getArgOperand(0), CI->getArgOperand(1), CI->getArgOperand(2) }; Rep = Builder.CreateCall(Intrinsic::getDeclaration(CI->getModule(), IID), Args); Value *PassThru = ZeroMask ? ConstantAggregateZero::get(CI->getType()) : CI->getArgOperand(0); Rep = EmitX86Select(Builder, CI->getArgOperand(3), Rep, PassThru); } else if (IsX86 && (Name.startswith("avx512.mask.vpdpwssd.") || Name.startswith("avx512.maskz.vpdpwssd.") || Name.startswith("avx512.mask.vpdpwssds.") || Name.startswith("avx512.maskz.vpdpwssds."))) { bool ZeroMask = Name[11] == 'z'; bool IsSaturating = Name[ZeroMask ? 21 : 20] == 's'; unsigned VecWidth = CI->getType()->getPrimitiveSizeInBits(); Intrinsic::ID IID; if (VecWidth == 128 && !IsSaturating) IID = Intrinsic::x86_avx512_vpdpwssd_128; else if (VecWidth == 256 && !IsSaturating) IID = Intrinsic::x86_avx512_vpdpwssd_256; else if (VecWidth == 512 && !IsSaturating) IID = Intrinsic::x86_avx512_vpdpwssd_512; else if (VecWidth == 128 && IsSaturating) IID = Intrinsic::x86_avx512_vpdpwssds_128; else if (VecWidth == 256 && IsSaturating) IID = Intrinsic::x86_avx512_vpdpwssds_256; else if (VecWidth == 512 && IsSaturating) IID = Intrinsic::x86_avx512_vpdpwssds_512; else llvm_unreachable("Unexpected intrinsic"); Value *Args[] = { CI->getArgOperand(0), CI->getArgOperand(1), CI->getArgOperand(2) }; Rep = Builder.CreateCall(Intrinsic::getDeclaration(CI->getModule(), IID), Args); Value *PassThru = ZeroMask ? ConstantAggregateZero::get(CI->getType()) : CI->getArgOperand(0); Rep = EmitX86Select(Builder, CI->getArgOperand(3), Rep, PassThru); } else if (IsX86 && (Name == "addcarryx.u32" || Name == "addcarryx.u64" || Name == "addcarry.u32" || Name == "addcarry.u64" || Name == "subborrow.u32" || Name == "subborrow.u64")) { Intrinsic::ID IID; if (Name[0] == 'a' && Name.back() == '2') IID = Intrinsic::x86_addcarry_32; else if (Name[0] == 'a' && Name.back() == '4') IID = Intrinsic::x86_addcarry_64; else if (Name[0] == 's' && Name.back() == '2') IID = Intrinsic::x86_subborrow_32; else if (Name[0] == 's' && Name.back() == '4') IID = Intrinsic::x86_subborrow_64; else llvm_unreachable("Unexpected intrinsic"); // Make a call with 3 operands. Value *Args[] = { CI->getArgOperand(0), CI->getArgOperand(1), CI->getArgOperand(2)}; Value *NewCall = Builder.CreateCall( Intrinsic::getDeclaration(CI->getModule(), IID), Args); // Extract the second result and store it. Value *Data = Builder.CreateExtractValue(NewCall, 1); // Cast the pointer to the right type. Value *Ptr = Builder.CreateBitCast(CI->getArgOperand(3), llvm::PointerType::getUnqual(Data->getType())); Builder.CreateAlignedStore(Data, Ptr, 1); // Replace the original call result with the first result of the new call. Value *CF = Builder.CreateExtractValue(NewCall, 0); CI->replaceAllUsesWith(CF); Rep = nullptr; } else if (IsX86 && Name.startswith("avx512.mask.") && upgradeAVX512MaskToSelect(Name, Builder, *CI, Rep)) { // Rep will be updated by the call in the condition. } else if (IsNVVM && (Name == "abs.i" || Name == "abs.ll")) { Value *Arg = CI->getArgOperand(0); Value *Neg = Builder.CreateNeg(Arg, "neg"); Value *Cmp = Builder.CreateICmpSGE( Arg, llvm::Constant::getNullValue(Arg->getType()), "abs.cond"); Rep = Builder.CreateSelect(Cmp, Arg, Neg, "abs"); } else if (IsNVVM && (Name == "max.i" || Name == "max.ll" || Name == "max.ui" || Name == "max.ull")) { Value *Arg0 = CI->getArgOperand(0); Value *Arg1 = CI->getArgOperand(1); Value *Cmp = Name.endswith(".ui") || Name.endswith(".ull") ? Builder.CreateICmpUGE(Arg0, Arg1, "max.cond") : Builder.CreateICmpSGE(Arg0, Arg1, "max.cond"); Rep = Builder.CreateSelect(Cmp, Arg0, Arg1, "max"); } else if (IsNVVM && (Name == "min.i" || Name == "min.ll" || Name == "min.ui" || Name == "min.ull")) { Value *Arg0 = CI->getArgOperand(0); Value *Arg1 = CI->getArgOperand(1); Value *Cmp = Name.endswith(".ui") || Name.endswith(".ull") ? Builder.CreateICmpULE(Arg0, Arg1, "min.cond") : Builder.CreateICmpSLE(Arg0, Arg1, "min.cond"); Rep = Builder.CreateSelect(Cmp, Arg0, Arg1, "min"); } else if (IsNVVM && Name == "clz.ll") { // llvm.nvvm.clz.ll returns an i32, but llvm.ctlz.i64 and returns an i64. Value *Arg = CI->getArgOperand(0); Value *Ctlz = Builder.CreateCall( Intrinsic::getDeclaration(F->getParent(), Intrinsic::ctlz, {Arg->getType()}), {Arg, Builder.getFalse()}, "ctlz"); Rep = Builder.CreateTrunc(Ctlz, Builder.getInt32Ty(), "ctlz.trunc"); } else if (IsNVVM && Name == "popc.ll") { // llvm.nvvm.popc.ll returns an i32, but llvm.ctpop.i64 and returns an // i64. Value *Arg = CI->getArgOperand(0); Value *Popc = Builder.CreateCall( Intrinsic::getDeclaration(F->getParent(), Intrinsic::ctpop, {Arg->getType()}), Arg, "ctpop"); Rep = Builder.CreateTrunc(Popc, Builder.getInt32Ty(), "ctpop.trunc"); } else if (IsNVVM && Name == "h2f") { Rep = Builder.CreateCall(Intrinsic::getDeclaration( F->getParent(), Intrinsic::convert_from_fp16, {Builder.getFloatTy()}), CI->getArgOperand(0), "h2f"); } else { llvm_unreachable("Unknown function for CallInst upgrade."); } if (Rep) CI->replaceAllUsesWith(Rep); CI->eraseFromParent(); return; } const auto &DefaultCase = [&NewFn, &CI]() -> void { // Handle generic mangling change, but nothing else assert( (CI->getCalledFunction()->getName() != NewFn->getName()) && "Unknown function for CallInst upgrade and isn't just a name change"); CI->setCalledFunction(NewFn); }; CallInst *NewCall = nullptr; switch (NewFn->getIntrinsicID()) { default: { DefaultCase(); return; } case Intrinsic::arm_neon_vld1: case Intrinsic::arm_neon_vld2: case Intrinsic::arm_neon_vld3: case Intrinsic::arm_neon_vld4: case Intrinsic::arm_neon_vld2lane: case Intrinsic::arm_neon_vld3lane: case Intrinsic::arm_neon_vld4lane: case Intrinsic::arm_neon_vst1: case Intrinsic::arm_neon_vst2: case Intrinsic::arm_neon_vst3: case Intrinsic::arm_neon_vst4: case Intrinsic::arm_neon_vst2lane: case Intrinsic::arm_neon_vst3lane: case Intrinsic::arm_neon_vst4lane: { SmallVector Args(CI->arg_operands().begin(), CI->arg_operands().end()); NewCall = Builder.CreateCall(NewFn, Args); break; } case Intrinsic::bitreverse: NewCall = Builder.CreateCall(NewFn, {CI->getArgOperand(0)}); break; case Intrinsic::ctlz: case Intrinsic::cttz: assert(CI->getNumArgOperands() == 1 && "Mismatch between function args and call args"); NewCall = Builder.CreateCall(NewFn, {CI->getArgOperand(0), Builder.getFalse()}); break; case Intrinsic::objectsize: { Value *NullIsUnknownSize = CI->getNumArgOperands() == 2 ? Builder.getFalse() : CI->getArgOperand(2); NewCall = Builder.CreateCall( NewFn, {CI->getArgOperand(0), CI->getArgOperand(1), NullIsUnknownSize}); break; } case Intrinsic::ctpop: NewCall = Builder.CreateCall(NewFn, {CI->getArgOperand(0)}); break; case Intrinsic::convert_from_fp16: NewCall = Builder.CreateCall(NewFn, {CI->getArgOperand(0)}); break; case Intrinsic::dbg_value: // Upgrade from the old version that had an extra offset argument. assert(CI->getNumArgOperands() == 4); // Drop nonzero offsets instead of attempting to upgrade them. if (auto *Offset = dyn_cast_or_null(CI->getArgOperand(1))) if (Offset->isZeroValue()) { NewCall = Builder.CreateCall( NewFn, {CI->getArgOperand(0), CI->getArgOperand(2), CI->getArgOperand(3)}); break; } CI->eraseFromParent(); return; case Intrinsic::x86_xop_vfrcz_ss: case Intrinsic::x86_xop_vfrcz_sd: NewCall = Builder.CreateCall(NewFn, {CI->getArgOperand(1)}); break; case Intrinsic::x86_xop_vpermil2pd: case Intrinsic::x86_xop_vpermil2ps: case Intrinsic::x86_xop_vpermil2pd_256: case Intrinsic::x86_xop_vpermil2ps_256: { SmallVector Args(CI->arg_operands().begin(), CI->arg_operands().end()); VectorType *FltIdxTy = cast(Args[2]->getType()); VectorType *IntIdxTy = VectorType::getInteger(FltIdxTy); Args[2] = Builder.CreateBitCast(Args[2], IntIdxTy); NewCall = Builder.CreateCall(NewFn, Args); break; } case Intrinsic::x86_sse41_ptestc: case Intrinsic::x86_sse41_ptestz: case Intrinsic::x86_sse41_ptestnzc: { // The arguments for these intrinsics used to be v4f32, and changed // to v2i64. This is purely a nop, since those are bitwise intrinsics. // So, the only thing required is a bitcast for both arguments. // First, check the arguments have the old type. Value *Arg0 = CI->getArgOperand(0); if (Arg0->getType() != VectorType::get(Type::getFloatTy(C), 4)) return; // Old intrinsic, add bitcasts Value *Arg1 = CI->getArgOperand(1); Type *NewVecTy = VectorType::get(Type::getInt64Ty(C), 2); Value *BC0 = Builder.CreateBitCast(Arg0, NewVecTy, "cast"); Value *BC1 = Builder.CreateBitCast(Arg1, NewVecTy, "cast"); NewCall = Builder.CreateCall(NewFn, {BC0, BC1}); break; } case Intrinsic::x86_rdtscp: { // This used to take 1 arguments. If we have no arguments, it is already // upgraded. if (CI->getNumOperands() == 0) return; NewCall = Builder.CreateCall(NewFn); // Extract the second result and store it. Value *Data = Builder.CreateExtractValue(NewCall, 1); // Cast the pointer to the right type. Value *Ptr = Builder.CreateBitCast(CI->getArgOperand(0), llvm::PointerType::getUnqual(Data->getType())); Builder.CreateAlignedStore(Data, Ptr, 1); // Replace the original call result with the first result of the new call. Value *TSC = Builder.CreateExtractValue(NewCall, 0); std::string Name = CI->getName(); if (!Name.empty()) { CI->setName(Name + ".old"); NewCall->setName(Name); } CI->replaceAllUsesWith(TSC); CI->eraseFromParent(); return; } case Intrinsic::x86_sse41_insertps: case Intrinsic::x86_sse41_dppd: case Intrinsic::x86_sse41_dpps: case Intrinsic::x86_sse41_mpsadbw: case Intrinsic::x86_avx_dp_ps_256: case Intrinsic::x86_avx2_mpsadbw: { // Need to truncate the last argument from i32 to i8 -- this argument models // an inherently 8-bit immediate operand to these x86 instructions. SmallVector Args(CI->arg_operands().begin(), CI->arg_operands().end()); // Replace the last argument with a trunc. Args.back() = Builder.CreateTrunc(Args.back(), Type::getInt8Ty(C), "trunc"); NewCall = Builder.CreateCall(NewFn, Args); break; } case Intrinsic::thread_pointer: { NewCall = Builder.CreateCall(NewFn, {}); break; } case Intrinsic::invariant_start: case Intrinsic::invariant_end: case Intrinsic::masked_load: case Intrinsic::masked_store: case Intrinsic::masked_gather: case Intrinsic::masked_scatter: { SmallVector Args(CI->arg_operands().begin(), CI->arg_operands().end()); NewCall = Builder.CreateCall(NewFn, Args); break; } case Intrinsic::memcpy: case Intrinsic::memmove: case Intrinsic::memset: { // We have to make sure that the call signature is what we're expecting. // We only want to change the old signatures by removing the alignment arg: // @llvm.mem[cpy|move]...(i8*, i8*, i[32|i64], i32, i1) // -> @llvm.mem[cpy|move]...(i8*, i8*, i[32|i64], i1) // @llvm.memset...(i8*, i8, i[32|64], i32, i1) // -> @llvm.memset...(i8*, i8, i[32|64], i1) // Note: i8*'s in the above can be any pointer type if (CI->getNumArgOperands() != 5) { DefaultCase(); return; } // Remove alignment argument (3), and add alignment attributes to the // dest/src pointers. Value *Args[4] = {CI->getArgOperand(0), CI->getArgOperand(1), CI->getArgOperand(2), CI->getArgOperand(4)}; NewCall = Builder.CreateCall(NewFn, Args); auto *MemCI = cast(NewCall); // All mem intrinsics support dest alignment. const ConstantInt *Align = cast(CI->getArgOperand(3)); MemCI->setDestAlignment(Align->getZExtValue()); // Memcpy/Memmove also support source alignment. if (auto *MTI = dyn_cast(MemCI)) MTI->setSourceAlignment(Align->getZExtValue()); break; } } assert(NewCall && "Should have either set this variable or returned through " "the default case"); std::string Name = CI->getName(); if (!Name.empty()) { CI->setName(Name + ".old"); NewCall->setName(Name); } CI->replaceAllUsesWith(NewCall); CI->eraseFromParent(); } void llvm::UpgradeCallsToIntrinsic(Function *F) { assert(F && "Illegal attempt to upgrade a non-existent intrinsic."); // Check if this function should be upgraded and get the replacement function // if there is one. Function *NewFn; if (UpgradeIntrinsicFunction(F, NewFn)) { // Replace all users of the old function with the new function or new // instructions. This is not a range loop because the call is deleted. for (auto UI = F->user_begin(), UE = F->user_end(); UI != UE; ) if (CallInst *CI = dyn_cast(*UI++)) UpgradeIntrinsicCall(CI, NewFn); // Remove old function, no longer used, from the module. F->eraseFromParent(); } } MDNode *llvm::UpgradeTBAANode(MDNode &MD) { // Check if the tag uses struct-path aware TBAA format. if (isa(MD.getOperand(0)) && MD.getNumOperands() >= 3) return &MD; auto &Context = MD.getContext(); if (MD.getNumOperands() == 3) { Metadata *Elts[] = {MD.getOperand(0), MD.getOperand(1)}; MDNode *ScalarType = MDNode::get(Context, Elts); // Create a MDNode Metadata *Elts2[] = {ScalarType, ScalarType, ConstantAsMetadata::get( Constant::getNullValue(Type::getInt64Ty(Context))), MD.getOperand(2)}; return MDNode::get(Context, Elts2); } // Create a MDNode Metadata *Elts[] = {&MD, &MD, ConstantAsMetadata::get(Constant::getNullValue( Type::getInt64Ty(Context)))}; return MDNode::get(Context, Elts); } Instruction *llvm::UpgradeBitCastInst(unsigned Opc, Value *V, Type *DestTy, Instruction *&Temp) { if (Opc != Instruction::BitCast) return nullptr; Temp = nullptr; Type *SrcTy = V->getType(); if (SrcTy->isPtrOrPtrVectorTy() && DestTy->isPtrOrPtrVectorTy() && SrcTy->getPointerAddressSpace() != DestTy->getPointerAddressSpace()) { LLVMContext &Context = V->getContext(); // We have no information about target data layout, so we assume that // the maximum pointer size is 64bit. Type *MidTy = Type::getInt64Ty(Context); Temp = CastInst::Create(Instruction::PtrToInt, V, MidTy); return CastInst::Create(Instruction::IntToPtr, Temp, DestTy); } return nullptr; } Value *llvm::UpgradeBitCastExpr(unsigned Opc, Constant *C, Type *DestTy) { if (Opc != Instruction::BitCast) return nullptr; Type *SrcTy = C->getType(); if (SrcTy->isPtrOrPtrVectorTy() && DestTy->isPtrOrPtrVectorTy() && SrcTy->getPointerAddressSpace() != DestTy->getPointerAddressSpace()) { LLVMContext &Context = C->getContext(); // We have no information about target data layout, so we assume that // the maximum pointer size is 64bit. Type *MidTy = Type::getInt64Ty(Context); return ConstantExpr::getIntToPtr(ConstantExpr::getPtrToInt(C, MidTy), DestTy); } return nullptr; } /// Check the debug info version number, if it is out-dated, drop the debug /// info. Return true if module is modified. bool llvm::UpgradeDebugInfo(Module &M) { unsigned Version = getDebugMetadataVersionFromModule(M); if (Version == DEBUG_METADATA_VERSION) { bool BrokenDebugInfo = false; if (verifyModule(M, &llvm::errs(), &BrokenDebugInfo)) report_fatal_error("Broken module found, compilation aborted!"); if (!BrokenDebugInfo) // Everything is ok. return false; else { // Diagnose malformed debug info. DiagnosticInfoIgnoringInvalidDebugMetadata Diag(M); M.getContext().diagnose(Diag); } } bool Modified = StripDebugInfo(M); if (Modified && Version != DEBUG_METADATA_VERSION) { // Diagnose a version mismatch. DiagnosticInfoDebugMetadataVersion DiagVersion(M, Version); M.getContext().diagnose(DiagVersion); } return Modified; } bool llvm::UpgradeRetainReleaseMarker(Module &M) { bool Changed = false; NamedMDNode *ModRetainReleaseMarker = M.getNamedMetadata("clang.arc.retainAutoreleasedReturnValueMarker"); if (ModRetainReleaseMarker) { MDNode *Op = ModRetainReleaseMarker->getOperand(0); if (Op) { MDString *ID = dyn_cast_or_null(Op->getOperand(0)); if (ID) { SmallVector ValueComp; ID->getString().split(ValueComp, "#"); if (ValueComp.size() == 2) { std::string NewValue = ValueComp[0].str() + ";" + ValueComp[1].str(); Metadata *Ops[1] = {MDString::get(M.getContext(), NewValue)}; ModRetainReleaseMarker->setOperand(0, MDNode::get(M.getContext(), Ops)); Changed = true; } } } } return Changed; } bool llvm::UpgradeModuleFlags(Module &M) { NamedMDNode *ModFlags = M.getModuleFlagsMetadata(); if (!ModFlags) return false; bool HasObjCFlag = false, HasClassProperties = false, Changed = false; for (unsigned I = 0, E = ModFlags->getNumOperands(); I != E; ++I) { MDNode *Op = ModFlags->getOperand(I); if (Op->getNumOperands() != 3) continue; MDString *ID = dyn_cast_or_null(Op->getOperand(1)); if (!ID) continue; if (ID->getString() == "Objective-C Image Info Version") HasObjCFlag = true; if (ID->getString() == "Objective-C Class Properties") HasClassProperties = true; // Upgrade PIC/PIE Module Flags. The module flag behavior for these two // field was Error and now they are Max. if (ID->getString() == "PIC Level" || ID->getString() == "PIE Level") { if (auto *Behavior = mdconst::dyn_extract_or_null(Op->getOperand(0))) { if (Behavior->getLimitedValue() == Module::Error) { Type *Int32Ty = Type::getInt32Ty(M.getContext()); Metadata *Ops[3] = { ConstantAsMetadata::get(ConstantInt::get(Int32Ty, Module::Max)), MDString::get(M.getContext(), ID->getString()), Op->getOperand(2)}; ModFlags->setOperand(I, MDNode::get(M.getContext(), Ops)); Changed = true; } } } // Upgrade Objective-C Image Info Section. Removed the whitespce in the // section name so that llvm-lto will not complain about mismatching // module flags that is functionally the same. if (ID->getString() == "Objective-C Image Info Section") { if (auto *Value = dyn_cast_or_null(Op->getOperand(2))) { SmallVector ValueComp; Value->getString().split(ValueComp, " "); if (ValueComp.size() != 1) { std::string NewValue; for (auto &S : ValueComp) NewValue += S.str(); Metadata *Ops[3] = {Op->getOperand(0), Op->getOperand(1), MDString::get(M.getContext(), NewValue)}; ModFlags->setOperand(I, MDNode::get(M.getContext(), Ops)); Changed = true; } } } } // "Objective-C Class Properties" is recently added for Objective-C. We // upgrade ObjC bitcodes to contain a "Objective-C Class Properties" module // flag of value 0, so we can correclty downgrade this flag when trying to // link an ObjC bitcode without this module flag with an ObjC bitcode with // this module flag. if (HasObjCFlag && !HasClassProperties) { M.addModuleFlag(llvm::Module::Override, "Objective-C Class Properties", (uint32_t)0); Changed = true; } return Changed; } void llvm::UpgradeSectionAttributes(Module &M) { auto TrimSpaces = [](StringRef Section) -> std::string { SmallVector Components; Section.split(Components, ','); SmallString<32> Buffer; raw_svector_ostream OS(Buffer); for (auto Component : Components) OS << ',' << Component.trim(); return OS.str().substr(1); }; for (auto &GV : M.globals()) { if (!GV.hasSection()) continue; StringRef Section = GV.getSection(); if (!Section.startswith("__DATA, __objc_catlist")) continue; // __DATA, __objc_catlist, regular, no_dead_strip // __DATA,__objc_catlist,regular,no_dead_strip GV.setSection(TrimSpaces(Section)); } } static bool isOldLoopArgument(Metadata *MD) { auto *T = dyn_cast_or_null(MD); if (!T) return false; if (T->getNumOperands() < 1) return false; auto *S = dyn_cast_or_null(T->getOperand(0)); if (!S) return false; return S->getString().startswith("llvm.vectorizer."); } static MDString *upgradeLoopTag(LLVMContext &C, StringRef OldTag) { StringRef OldPrefix = "llvm.vectorizer."; assert(OldTag.startswith(OldPrefix) && "Expected old prefix"); if (OldTag == "llvm.vectorizer.unroll") return MDString::get(C, "llvm.loop.interleave.count"); return MDString::get( C, (Twine("llvm.loop.vectorize.") + OldTag.drop_front(OldPrefix.size())) .str()); } static Metadata *upgradeLoopArgument(Metadata *MD) { auto *T = dyn_cast_or_null(MD); if (!T) return MD; if (T->getNumOperands() < 1) return MD; auto *OldTag = dyn_cast_or_null(T->getOperand(0)); if (!OldTag) return MD; if (!OldTag->getString().startswith("llvm.vectorizer.")) return MD; // This has an old tag. Upgrade it. SmallVector Ops; Ops.reserve(T->getNumOperands()); Ops.push_back(upgradeLoopTag(T->getContext(), OldTag->getString())); for (unsigned I = 1, E = T->getNumOperands(); I != E; ++I) Ops.push_back(T->getOperand(I)); return MDTuple::get(T->getContext(), Ops); } MDNode *llvm::upgradeInstructionLoopAttachment(MDNode &N) { auto *T = dyn_cast(&N); if (!T) return &N; if (none_of(T->operands(), isOldLoopArgument)) return &N; SmallVector Ops; Ops.reserve(T->getNumOperands()); for (Metadata *MD : T->operands()) Ops.push_back(upgradeLoopArgument(MD)); return MDTuple::get(T->getContext(), Ops); } Index: vendor/llvm/dist-release_80/lib/Support/JSON.cpp =================================================================== --- vendor/llvm/dist-release_80/lib/Support/JSON.cpp (revision 343793) +++ vendor/llvm/dist-release_80/lib/Support/JSON.cpp (revision 343794) @@ -1,693 +1,699 @@ //=== JSON.cpp - JSON value, parsing and serialization - C++ -----------*-===// // // The LLVM Compiler Infrastructure // // This file is distributed under the University of Illinois Open Source // License. See LICENSE.TXT for details. // //===---------------------------------------------------------------------===// #include "llvm/Support/JSON.h" #include "llvm/Support/ConvertUTF.h" #include "llvm/Support/Format.h" #include namespace llvm { namespace json { Value &Object::operator[](const ObjectKey &K) { return try_emplace(K, nullptr).first->getSecond(); } Value &Object::operator[](ObjectKey &&K) { return try_emplace(std::move(K), nullptr).first->getSecond(); } Value *Object::get(StringRef K) { auto I = find(K); if (I == end()) return nullptr; return &I->second; } const Value *Object::get(StringRef K) const { auto I = find(K); if (I == end()) return nullptr; return &I->second; } llvm::Optional Object::getNull(StringRef K) const { if (auto *V = get(K)) return V->getAsNull(); return llvm::None; } llvm::Optional Object::getBoolean(StringRef K) const { if (auto *V = get(K)) return V->getAsBoolean(); return llvm::None; } llvm::Optional Object::getNumber(StringRef K) const { if (auto *V = get(K)) return V->getAsNumber(); return llvm::None; } llvm::Optional Object::getInteger(StringRef K) const { if (auto *V = get(K)) return V->getAsInteger(); return llvm::None; } llvm::Optional Object::getString(StringRef K) const { if (auto *V = get(K)) return V->getAsString(); return llvm::None; } const json::Object *Object::getObject(StringRef K) const { if (auto *V = get(K)) return V->getAsObject(); return nullptr; } json::Object *Object::getObject(StringRef K) { if (auto *V = get(K)) return V->getAsObject(); return nullptr; } const json::Array *Object::getArray(StringRef K) const { if (auto *V = get(K)) return V->getAsArray(); return nullptr; } json::Array *Object::getArray(StringRef K) { if (auto *V = get(K)) return V->getAsArray(); return nullptr; } bool operator==(const Object &LHS, const Object &RHS) { if (LHS.size() != RHS.size()) return false; for (const auto &L : LHS) { auto R = RHS.find(L.first); if (R == RHS.end() || L.second != R->second) return false; } return true; } Array::Array(std::initializer_list Elements) { V.reserve(Elements.size()); for (const Value &V : Elements) { emplace_back(nullptr); back().moveFrom(std::move(V)); } } Value::Value(std::initializer_list Elements) : Value(json::Array(Elements)) {} void Value::copyFrom(const Value &M) { Type = M.Type; switch (Type) { case T_Null: case T_Boolean: case T_Double: case T_Integer: memcpy(Union.buffer, M.Union.buffer, sizeof(Union.buffer)); break; case T_StringRef: create(M.as()); break; case T_String: create(M.as()); break; case T_Object: create(M.as()); break; case T_Array: create(M.as()); break; } } void Value::moveFrom(const Value &&M) { Type = M.Type; switch (Type) { case T_Null: case T_Boolean: case T_Double: case T_Integer: memcpy(Union.buffer, M.Union.buffer, sizeof(Union.buffer)); break; case T_StringRef: create(M.as()); break; case T_String: create(std::move(M.as())); M.Type = T_Null; break; case T_Object: create(std::move(M.as())); M.Type = T_Null; break; case T_Array: create(std::move(M.as())); M.Type = T_Null; break; } } void Value::destroy() { switch (Type) { case T_Null: case T_Boolean: case T_Double: case T_Integer: break; case T_StringRef: as().~StringRef(); break; case T_String: as().~basic_string(); break; case T_Object: as().~Object(); break; case T_Array: as().~Array(); break; } } bool operator==(const Value &L, const Value &R) { if (L.kind() != R.kind()) return false; switch (L.kind()) { case Value::Null: return *L.getAsNull() == *R.getAsNull(); case Value::Boolean: return *L.getAsBoolean() == *R.getAsBoolean(); case Value::Number: + // Workaround for https://gcc.gnu.org/bugzilla/show_bug.cgi?id=323 + // The same integer must convert to the same double, per the standard. + // However we see 64-vs-80-bit precision comparisons with gcc-7 -O3 -m32. + // So we avoid floating point promotion for exact comparisons. + if (L.Type == Value::T_Integer || R.Type == Value::T_Integer) + return L.getAsInteger() == R.getAsInteger(); return *L.getAsNumber() == *R.getAsNumber(); case Value::String: return *L.getAsString() == *R.getAsString(); case Value::Array: return *L.getAsArray() == *R.getAsArray(); case Value::Object: return *L.getAsObject() == *R.getAsObject(); } llvm_unreachable("Unknown value kind"); } namespace { // Simple recursive-descent JSON parser. class Parser { public: Parser(StringRef JSON) : Start(JSON.begin()), P(JSON.begin()), End(JSON.end()) {} bool checkUTF8() { size_t ErrOffset; if (isUTF8(StringRef(Start, End - Start), &ErrOffset)) return true; P = Start + ErrOffset; // For line/column calculation. return parseError("Invalid UTF-8 sequence"); } bool parseValue(Value &Out); bool assertEnd() { eatWhitespace(); if (P == End) return true; return parseError("Text after end of document"); } Error takeError() { assert(Err); return std::move(*Err); } private: void eatWhitespace() { while (P != End && (*P == ' ' || *P == '\r' || *P == '\n' || *P == '\t')) ++P; } // On invalid syntax, parseX() functions return false and set Err. bool parseNumber(char First, Value &Out); bool parseString(std::string &Out); bool parseUnicode(std::string &Out); bool parseError(const char *Msg); // always returns false char next() { return P == End ? 0 : *P++; } char peek() { return P == End ? 0 : *P; } static bool isNumber(char C) { return C == '0' || C == '1' || C == '2' || C == '3' || C == '4' || C == '5' || C == '6' || C == '7' || C == '8' || C == '9' || C == 'e' || C == 'E' || C == '+' || C == '-' || C == '.'; } Optional Err; const char *Start, *P, *End; }; bool Parser::parseValue(Value &Out) { eatWhitespace(); if (P == End) return parseError("Unexpected EOF"); switch (char C = next()) { // Bare null/true/false are easy - first char identifies them. case 'n': Out = nullptr; return (next() == 'u' && next() == 'l' && next() == 'l') || parseError("Invalid JSON value (null?)"); case 't': Out = true; return (next() == 'r' && next() == 'u' && next() == 'e') || parseError("Invalid JSON value (true?)"); case 'f': Out = false; return (next() == 'a' && next() == 'l' && next() == 's' && next() == 'e') || parseError("Invalid JSON value (false?)"); case '"': { std::string S; if (parseString(S)) { Out = std::move(S); return true; } return false; } case '[': { Out = Array{}; Array &A = *Out.getAsArray(); eatWhitespace(); if (peek() == ']') { ++P; return true; } for (;;) { A.emplace_back(nullptr); if (!parseValue(A.back())) return false; eatWhitespace(); switch (next()) { case ',': eatWhitespace(); continue; case ']': return true; default: return parseError("Expected , or ] after array element"); } } } case '{': { Out = Object{}; Object &O = *Out.getAsObject(); eatWhitespace(); if (peek() == '}') { ++P; return true; } for (;;) { if (next() != '"') return parseError("Expected object key"); std::string K; if (!parseString(K)) return false; eatWhitespace(); if (next() != ':') return parseError("Expected : after object key"); eatWhitespace(); if (!parseValue(O[std::move(K)])) return false; eatWhitespace(); switch (next()) { case ',': eatWhitespace(); continue; case '}': return true; default: return parseError("Expected , or } after object property"); } } } default: if (isNumber(C)) return parseNumber(C, Out); return parseError("Invalid JSON value"); } } bool Parser::parseNumber(char First, Value &Out) { // Read the number into a string. (Must be null-terminated for strto*). SmallString<24> S; S.push_back(First); while (isNumber(peek())) S.push_back(next()); char *End; // Try first to parse as integer, and if so preserve full 64 bits. // strtoll returns long long >= 64 bits, so check it's in range too. auto I = std::strtoll(S.c_str(), &End, 10); if (End == S.end() && I >= std::numeric_limits::min() && I <= std::numeric_limits::max()) { Out = int64_t(I); return true; } // If it's not an integer Out = std::strtod(S.c_str(), &End); return End == S.end() || parseError("Invalid JSON value (number?)"); } bool Parser::parseString(std::string &Out) { // leading quote was already consumed. for (char C = next(); C != '"'; C = next()) { if (LLVM_UNLIKELY(P == End)) return parseError("Unterminated string"); if (LLVM_UNLIKELY((C & 0x1f) == C)) return parseError("Control character in string"); if (LLVM_LIKELY(C != '\\')) { Out.push_back(C); continue; } // Handle escape sequence. switch (C = next()) { case '"': case '\\': case '/': Out.push_back(C); break; case 'b': Out.push_back('\b'); break; case 'f': Out.push_back('\f'); break; case 'n': Out.push_back('\n'); break; case 'r': Out.push_back('\r'); break; case 't': Out.push_back('\t'); break; case 'u': if (!parseUnicode(Out)) return false; break; default: return parseError("Invalid escape sequence"); } } return true; } static void encodeUtf8(uint32_t Rune, std::string &Out) { if (Rune < 0x80) { Out.push_back(Rune & 0x7F); } else if (Rune < 0x800) { uint8_t FirstByte = 0xC0 | ((Rune & 0x7C0) >> 6); uint8_t SecondByte = 0x80 | (Rune & 0x3F); Out.push_back(FirstByte); Out.push_back(SecondByte); } else if (Rune < 0x10000) { uint8_t FirstByte = 0xE0 | ((Rune & 0xF000) >> 12); uint8_t SecondByte = 0x80 | ((Rune & 0xFC0) >> 6); uint8_t ThirdByte = 0x80 | (Rune & 0x3F); Out.push_back(FirstByte); Out.push_back(SecondByte); Out.push_back(ThirdByte); } else if (Rune < 0x110000) { uint8_t FirstByte = 0xF0 | ((Rune & 0x1F0000) >> 18); uint8_t SecondByte = 0x80 | ((Rune & 0x3F000) >> 12); uint8_t ThirdByte = 0x80 | ((Rune & 0xFC0) >> 6); uint8_t FourthByte = 0x80 | (Rune & 0x3F); Out.push_back(FirstByte); Out.push_back(SecondByte); Out.push_back(ThirdByte); Out.push_back(FourthByte); } else { llvm_unreachable("Invalid codepoint"); } } // Parse a UTF-16 \uNNNN escape sequence. "\u" has already been consumed. // May parse several sequential escapes to ensure proper surrogate handling. // We do not use ConvertUTF.h, it can't accept and replace unpaired surrogates. // These are invalid Unicode but valid JSON (RFC 8259, section 8.2). bool Parser::parseUnicode(std::string &Out) { // Invalid UTF is not a JSON error (RFC 8529§8.2). It gets replaced by U+FFFD. auto Invalid = [&] { Out.append(/* UTF-8 */ {'\xef', '\xbf', '\xbd'}); }; // Decodes 4 hex digits from the stream into Out, returns false on error. auto Parse4Hex = [this](uint16_t &Out) -> bool { Out = 0; char Bytes[] = {next(), next(), next(), next()}; for (unsigned char C : Bytes) { if (!std::isxdigit(C)) return parseError("Invalid \\u escape sequence"); Out <<= 4; Out |= (C > '9') ? (C & ~0x20) - 'A' + 10 : (C - '0'); } return true; }; uint16_t First; // UTF-16 code unit from the first \u escape. if (!Parse4Hex(First)) return false; // We loop to allow proper surrogate-pair error handling. while (true) { // Case 1: the UTF-16 code unit is already a codepoint in the BMP. if (LLVM_LIKELY(First < 0xD800 || First >= 0xE000)) { encodeUtf8(First, Out); return true; } // Case 2: it's an (unpaired) trailing surrogate. if (LLVM_UNLIKELY(First >= 0xDC00)) { Invalid(); return true; } // Case 3: it's a leading surrogate. We expect a trailing one next. // Case 3a: there's no trailing \u escape. Don't advance in the stream. if (LLVM_UNLIKELY(P + 2 > End || *P != '\\' || *(P + 1) != 'u')) { Invalid(); // Leading surrogate was unpaired. return true; } P += 2; uint16_t Second; if (!Parse4Hex(Second)) return false; // Case 3b: there was another \u escape, but it wasn't a trailing surrogate. if (LLVM_UNLIKELY(Second < 0xDC00 || Second >= 0xE000)) { Invalid(); // Leading surrogate was unpaired. First = Second; // Second escape still needs to be processed. continue; } // Case 3c: a valid surrogate pair encoding an astral codepoint. encodeUtf8(0x10000 | ((First - 0xD800) << 10) | (Second - 0xDC00), Out); return true; } } bool Parser::parseError(const char *Msg) { int Line = 1; const char *StartOfLine = Start; for (const char *X = Start; X < P; ++X) { if (*X == 0x0A) { ++Line; StartOfLine = X + 1; } } Err.emplace( llvm::make_unique(Msg, Line, P - StartOfLine, P - Start)); return false; } } // namespace Expected parse(StringRef JSON) { Parser P(JSON); Value E = nullptr; if (P.checkUTF8()) if (P.parseValue(E)) if (P.assertEnd()) return std::move(E); return P.takeError(); } char ParseError::ID = 0; static std::vector sortedElements(const Object &O) { std::vector Elements; for (const auto &E : O) Elements.push_back(&E); llvm::sort(Elements, [](const Object::value_type *L, const Object::value_type *R) { return L->first < R->first; }); return Elements; } bool isUTF8(llvm::StringRef S, size_t *ErrOffset) { // Fast-path for ASCII, which is valid UTF-8. if (LLVM_LIKELY(isASCII(S))) return true; const UTF8 *Data = reinterpret_cast(S.data()), *Rest = Data; if (LLVM_LIKELY(isLegalUTF8String(&Rest, Data + S.size()))) return true; if (ErrOffset) *ErrOffset = Rest - Data; return false; } std::string fixUTF8(llvm::StringRef S) { // This isn't particularly efficient, but is only for error-recovery. std::vector Codepoints(S.size()); // 1 codepoint per byte suffices. const UTF8 *In8 = reinterpret_cast(S.data()); UTF32 *Out32 = Codepoints.data(); ConvertUTF8toUTF32(&In8, In8 + S.size(), &Out32, Out32 + Codepoints.size(), lenientConversion); Codepoints.resize(Out32 - Codepoints.data()); std::string Res(4 * Codepoints.size(), 0); // 4 bytes per codepoint suffice const UTF32 *In32 = Codepoints.data(); UTF8 *Out8 = reinterpret_cast(&Res[0]); ConvertUTF32toUTF8(&In32, In32 + Codepoints.size(), &Out8, Out8 + Res.size(), strictConversion); Res.resize(reinterpret_cast(Out8) - Res.data()); return Res; } } // namespace json } // namespace llvm static void quote(llvm::raw_ostream &OS, llvm::StringRef S) { OS << '\"'; for (unsigned char C : S) { if (C == 0x22 || C == 0x5C) OS << '\\'; if (C >= 0x20) { OS << C; continue; } OS << '\\'; switch (C) { // A few characters are common enough to make short escapes worthwhile. case '\t': OS << 't'; break; case '\n': OS << 'n'; break; case '\r': OS << 'r'; break; default: OS << 'u'; llvm::write_hex(OS, C, llvm::HexPrintStyle::Lower, 4); break; } } OS << '\"'; } enum IndenterAction { Indent, Outdent, Newline, Space, }; // Prints JSON. The indenter can be used to control formatting. template void llvm::json::Value::print(raw_ostream &OS, const Indenter &I) const { switch (Type) { case T_Null: OS << "null"; break; case T_Boolean: OS << (as() ? "true" : "false"); break; case T_Double: OS << format("%.*g", std::numeric_limits::max_digits10, as()); break; case T_Integer: OS << as(); break; case T_StringRef: quote(OS, as()); break; case T_String: quote(OS, as()); break; case T_Object: { bool Comma = false; OS << '{'; I(Indent); for (const auto *P : sortedElements(as())) { if (Comma) OS << ','; Comma = true; I(Newline); quote(OS, P->first); OS << ':'; I(Space); P->second.print(OS, I); } I(Outdent); if (Comma) I(Newline); OS << '}'; break; } case T_Array: { bool Comma = false; OS << '['; I(Indent); for (const auto &E : as()) { if (Comma) OS << ','; Comma = true; I(Newline); E.print(OS, I); } I(Outdent); if (Comma) I(Newline); OS << ']'; break; } } } void llvm::format_provider::format( const llvm::json::Value &E, raw_ostream &OS, StringRef Options) { if (Options.empty()) { OS << E; return; } unsigned IndentAmount = 0; if (Options.getAsInteger(/*Radix=*/10, IndentAmount)) llvm_unreachable("json::Value format options should be an integer"); unsigned IndentLevel = 0; E.print(OS, [&](IndenterAction A) { switch (A) { case Newline: OS << '\n'; OS.indent(IndentLevel); break; case Space: OS << ' '; break; case Indent: IndentLevel += IndentAmount; break; case Outdent: IndentLevel -= IndentAmount; break; }; }); } llvm::raw_ostream &llvm::json::operator<<(raw_ostream &OS, const Value &E) { E.print(OS, [](IndenterAction A) { /*ignore*/ }); return OS; } Index: vendor/llvm/dist-release_80/lib/Target/AArch64/AArch64SpeculationHardening.cpp =================================================================== --- vendor/llvm/dist-release_80/lib/Target/AArch64/AArch64SpeculationHardening.cpp (revision 343793) +++ vendor/llvm/dist-release_80/lib/Target/AArch64/AArch64SpeculationHardening.cpp (revision 343794) @@ -1,641 +1,702 @@ //===- AArch64SpeculationHardening.cpp - Harden Against Missspeculation --===// // // The LLVM Compiler Infrastructure // // This file is distributed under the University of Illinois Open Source // License. See LICENSE.TXT for details. // //===----------------------------------------------------------------------===// // // This file contains a pass to insert code to mitigate against side channel // vulnerabilities that may happen under control flow miss-speculation. // // The pass implements tracking of control flow miss-speculation into a "taint" // register. That taint register can then be used to mask off registers with // sensitive data when executing under miss-speculation, a.k.a. "transient // execution". // This pass is aimed at mitigating against SpectreV1-style vulnarabilities. // // It also implements speculative load hardening, i.e. using the taint register // to automatically mask off loaded data. // // As a possible follow-on improvement, also an intrinsics-based approach as // explained at https://lwn.net/Articles/759423/ could be implemented on top of // the current design. // // For AArch64, the following implementation choices are made to implement the // tracking of control flow miss-speculation into a taint register: // Some of these are different than the implementation choices made in // the similar pass implemented in X86SpeculativeLoadHardening.cpp, as // the instruction set characteristics result in different trade-offs. // - The speculation hardening is done after register allocation. With a // relative abundance of registers, one register is reserved (X16) to be // the taint register. X16 is expected to not clash with other register // reservation mechanisms with very high probability because: // . The AArch64 ABI doesn't guarantee X16 to be retained across any call. // . The only way to request X16 to be used as a programmer is through // inline assembly. In the rare case a function explicitly demands to // use X16/W16, this pass falls back to hardening against speculation // by inserting a DSB SYS/ISB barrier pair which will prevent control // flow speculation. // - It is easy to insert mask operations at this late stage as we have // mask operations available that don't set flags. // - The taint variable contains all-ones when no miss-speculation is detected, // and contains all-zeros when miss-speculation is detected. Therefore, when // masking, an AND instruction (which only changes the register to be masked, // no other side effects) can easily be inserted anywhere that's needed. // - The tracking of miss-speculation is done by using a data-flow conditional // select instruction (CSEL) to evaluate the flags that were also used to // make conditional branch direction decisions. Speculation of the CSEL // instruction can be limited with a CSDB instruction - so the combination of // CSEL + a later CSDB gives the guarantee that the flags as used in the CSEL // aren't speculated. When conditional branch direction gets miss-speculated, // the semantics of the inserted CSEL instruction is such that the taint // register will contain all zero bits. // One key requirement for this to work is that the conditional branch is // followed by an execution of the CSEL instruction, where the CSEL // instruction needs to use the same flags status as the conditional branch. // This means that the conditional branches must not be implemented as one // of the AArch64 conditional branches that do not use the flags as input // (CB(N)Z and TB(N)Z). This is implemented by ensuring in the instruction // selectors to not produce these instructions when speculation hardening // is enabled. This pass will assert if it does encounter such an instruction. // - On function call boundaries, the miss-speculation state is transferred from // the taint register X16 to be encoded in the SP register as value 0. // // For the aspect of automatically hardening loads, using the taint register, // (a.k.a. speculative load hardening, see // https://llvm.org/docs/SpeculativeLoadHardening.html), the following // implementation choices are made for AArch64: // - Many of the optimizations described at // https://llvm.org/docs/SpeculativeLoadHardening.html to harden fewer // loads haven't been implemented yet - but for some of them there are // FIXMEs in the code. // - loads that load into general purpose (X or W) registers get hardened by // masking the loaded data. For loads that load into other registers, the // address loaded from gets hardened. It is expected that hardening the // loaded data may be more efficient; but masking data in registers other // than X or W is not easy and may result in being slower than just // hardening the X address register loaded from. // - On AArch64, CSDB instructions are inserted between the masking of the // register and its first use, to ensure there's no non-control-flow // speculation that might undermine the hardening mechanism. // // Future extensions/improvements could be: // - Implement this functionality using full speculation barriers, akin to the // x86-slh-lfence option. This may be more useful for the intrinsics-based // approach than for the SLH approach to masking. // Note that this pass already inserts the full speculation barriers if the // function for some niche reason makes use of X16/W16. // - no indirect branch misprediction gets protected/instrumented; but this // could be done for some indirect branches, such as switch jump tables. //===----------------------------------------------------------------------===// #include "AArch64InstrInfo.h" #include "AArch64Subtarget.h" #include "Utils/AArch64BaseInfo.h" #include "llvm/ADT/BitVector.h" #include "llvm/ADT/SmallVector.h" #include "llvm/CodeGen/MachineBasicBlock.h" #include "llvm/CodeGen/MachineFunction.h" #include "llvm/CodeGen/MachineFunctionPass.h" #include "llvm/CodeGen/MachineInstr.h" #include "llvm/CodeGen/MachineInstrBuilder.h" #include "llvm/CodeGen/MachineOperand.h" #include "llvm/CodeGen/MachineRegisterInfo.h" +#include "llvm/CodeGen/RegisterScavenging.h" #include "llvm/IR/DebugLoc.h" #include "llvm/Pass.h" #include "llvm/Support/CodeGen.h" #include "llvm/Target/TargetMachine.h" #include using namespace llvm; #define DEBUG_TYPE "aarch64-speculation-hardening" #define AARCH64_SPECULATION_HARDENING_NAME "AArch64 speculation hardening pass" cl::opt HardenLoads("aarch64-slh-loads", cl::Hidden, cl::desc("Sanitize loads from memory."), cl::init(true)); namespace { class AArch64SpeculationHardening : public MachineFunctionPass { public: const TargetInstrInfo *TII; const TargetRegisterInfo *TRI; static char ID; AArch64SpeculationHardening() : MachineFunctionPass(ID) { initializeAArch64SpeculationHardeningPass(*PassRegistry::getPassRegistry()); } bool runOnMachineFunction(MachineFunction &Fn) override; StringRef getPassName() const override { return AARCH64_SPECULATION_HARDENING_NAME; } private: unsigned MisspeculatingTaintReg; unsigned MisspeculatingTaintReg32Bit; bool UseControlFlowSpeculationBarrier; BitVector RegsNeedingCSDBBeforeUse; BitVector RegsAlreadyMasked; bool functionUsesHardeningRegister(MachineFunction &MF) const; - bool instrumentControlFlow(MachineBasicBlock &MBB); + bool instrumentControlFlow(MachineBasicBlock &MBB, + bool &UsesFullSpeculationBarrier); bool endsWithCondControlFlow(MachineBasicBlock &MBB, MachineBasicBlock *&TBB, MachineBasicBlock *&FBB, AArch64CC::CondCode &CondCode) const; void insertTrackingCode(MachineBasicBlock &SplitEdgeBB, AArch64CC::CondCode &CondCode, DebugLoc DL) const; - void insertSPToRegTaintPropagation(MachineBasicBlock *MBB, + void insertSPToRegTaintPropagation(MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI) const; - void insertRegToSPTaintPropagation(MachineBasicBlock *MBB, + void insertRegToSPTaintPropagation(MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, unsigned TmpReg) const; + void insertFullSpeculationBarrier(MachineBasicBlock &MBB, + MachineBasicBlock::iterator MBBI, + DebugLoc DL) const; bool slhLoads(MachineBasicBlock &MBB); bool makeGPRSpeculationSafe(MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, MachineInstr &MI, unsigned Reg); - bool lowerSpeculationSafeValuePseudos(MachineBasicBlock &MBB); + bool lowerSpeculationSafeValuePseudos(MachineBasicBlock &MBB, + bool UsesFullSpeculationBarrier); bool expandSpeculationSafeValue(MachineBasicBlock &MBB, - MachineBasicBlock::iterator MBBI); + MachineBasicBlock::iterator MBBI, + bool UsesFullSpeculationBarrier); bool insertCSDB(MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, DebugLoc DL); }; } // end anonymous namespace char AArch64SpeculationHardening::ID = 0; INITIALIZE_PASS(AArch64SpeculationHardening, "aarch64-speculation-hardening", AARCH64_SPECULATION_HARDENING_NAME, false, false) bool AArch64SpeculationHardening::endsWithCondControlFlow( MachineBasicBlock &MBB, MachineBasicBlock *&TBB, MachineBasicBlock *&FBB, AArch64CC::CondCode &CondCode) const { SmallVector analyzeBranchCondCode; if (TII->analyzeBranch(MBB, TBB, FBB, analyzeBranchCondCode, false)) return false; // Ignore if the BB ends in an unconditional branch/fall-through. if (analyzeBranchCondCode.empty()) return false; // If the BB ends with a single conditional branch, FBB will be set to // nullptr (see API docs for TII->analyzeBranch). For the rest of the // analysis we want the FBB block to be set always. assert(TBB != nullptr); if (FBB == nullptr) FBB = MBB.getFallThrough(); // If both the true and the false condition jump to the same basic block, // there isn't need for any protection - whether the branch is speculated // correctly or not, we end up executing the architecturally correct code. if (TBB == FBB) return false; assert(MBB.succ_size() == 2); // translate analyzeBranchCondCode to CondCode. assert(analyzeBranchCondCode.size() == 1 && "unknown Cond array format"); CondCode = AArch64CC::CondCode(analyzeBranchCondCode[0].getImm()); return true; } +void AArch64SpeculationHardening::insertFullSpeculationBarrier( + MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, + DebugLoc DL) const { + // A full control flow speculation barrier consists of (DSB SYS + ISB) + BuildMI(MBB, MBBI, DL, TII->get(AArch64::DSB)).addImm(0xf); + BuildMI(MBB, MBBI, DL, TII->get(AArch64::ISB)).addImm(0xf); +} + void AArch64SpeculationHardening::insertTrackingCode( MachineBasicBlock &SplitEdgeBB, AArch64CC::CondCode &CondCode, DebugLoc DL) const { if (UseControlFlowSpeculationBarrier) { - // insert full control flow speculation barrier (DSB SYS + ISB) - BuildMI(SplitEdgeBB, SplitEdgeBB.begin(), DL, TII->get(AArch64::ISB)) - .addImm(0xf); - BuildMI(SplitEdgeBB, SplitEdgeBB.begin(), DL, TII->get(AArch64::DSB)) - .addImm(0xf); + insertFullSpeculationBarrier(SplitEdgeBB, SplitEdgeBB.begin(), DL); } else { BuildMI(SplitEdgeBB, SplitEdgeBB.begin(), DL, TII->get(AArch64::CSELXr)) .addDef(MisspeculatingTaintReg) .addUse(MisspeculatingTaintReg) .addUse(AArch64::XZR) .addImm(CondCode); SplitEdgeBB.addLiveIn(AArch64::NZCV); } } bool AArch64SpeculationHardening::instrumentControlFlow( - MachineBasicBlock &MBB) { + MachineBasicBlock &MBB, bool &UsesFullSpeculationBarrier) { LLVM_DEBUG(dbgs() << "Instrument control flow tracking on MBB: " << MBB); bool Modified = false; MachineBasicBlock *TBB = nullptr; MachineBasicBlock *FBB = nullptr; AArch64CC::CondCode CondCode; if (!endsWithCondControlFlow(MBB, TBB, FBB, CondCode)) { LLVM_DEBUG(dbgs() << "... doesn't end with CondControlFlow\n"); } else { // Now insert: // "CSEL MisSpeculatingR, MisSpeculatingR, XZR, cond" on the True edge and // "CSEL MisSpeculatingR, MisSpeculatingR, XZR, Invertcond" on the False // edge. AArch64CC::CondCode InvCondCode = AArch64CC::getInvertedCondCode(CondCode); MachineBasicBlock *SplitEdgeTBB = MBB.SplitCriticalEdge(TBB, *this); MachineBasicBlock *SplitEdgeFBB = MBB.SplitCriticalEdge(FBB, *this); assert(SplitEdgeTBB != nullptr); assert(SplitEdgeFBB != nullptr); DebugLoc DL; if (MBB.instr_end() != MBB.instr_begin()) DL = (--MBB.instr_end())->getDebugLoc(); insertTrackingCode(*SplitEdgeTBB, CondCode, DL); insertTrackingCode(*SplitEdgeFBB, InvCondCode, DL); LLVM_DEBUG(dbgs() << "SplitEdgeTBB: " << *SplitEdgeTBB << "\n"); LLVM_DEBUG(dbgs() << "SplitEdgeFBB: " << *SplitEdgeFBB << "\n"); Modified = true; } // Perform correct code generation around function calls and before returns. - { - SmallVector ReturnInstructions; - SmallVector CallInstructions; + // The below variables record the return/terminator instructions and the call + // instructions respectively; including which register is available as a + // temporary register just before the recorded instructions. + SmallVector, 4> ReturnInstructions; + SmallVector, 4> CallInstructions; + // if a temporary register is not available for at least one of the + // instructions for which we need to transfer taint to the stack pointer, we + // need to insert a full speculation barrier. + // TmpRegisterNotAvailableEverywhere tracks that condition. + bool TmpRegisterNotAvailableEverywhere = false; - for (MachineInstr &MI : MBB) { - if (MI.isReturn()) - ReturnInstructions.push_back(&MI); - else if (MI.isCall()) - CallInstructions.push_back(&MI); - } + RegScavenger RS; + RS.enterBasicBlock(MBB); - Modified |= - (ReturnInstructions.size() > 0) || (CallInstructions.size() > 0); + for (MachineBasicBlock::iterator I = MBB.begin(); I != MBB.end(); I++) { + MachineInstr &MI = *I; + if (!MI.isReturn() && !MI.isCall()) + continue; - for (MachineInstr *Return : ReturnInstructions) - insertRegToSPTaintPropagation(Return->getParent(), Return, AArch64::X17); - for (MachineInstr *Call : CallInstructions) { + // The RegScavenger represents registers available *after* the MI + // instruction pointed to by RS.getCurrentPosition(). + // We need to have a register that is available *before* the MI is executed. + if (I != MBB.begin()) + RS.forward(std::prev(I)); + // FIXME: The below just finds *a* unused register. Maybe code could be + // optimized more if this looks for the register that isn't used for the + // longest time around this place, to enable more scheduling freedom. Not + // sure if that would actually result in a big performance difference + // though. Maybe RegisterScavenger::findSurvivorBackwards has some logic + // already to do this - but it's unclear if that could easily be used here. + unsigned TmpReg = RS.FindUnusedReg(&AArch64::GPR64commonRegClass); + LLVM_DEBUG(dbgs() << "RS finds " + << ((TmpReg == 0) ? "no register " : "register "); + if (TmpReg != 0) dbgs() << printReg(TmpReg, TRI) << " "; + dbgs() << "to be available at MI " << MI); + if (TmpReg == 0) + TmpRegisterNotAvailableEverywhere = true; + if (MI.isReturn()) + ReturnInstructions.push_back({&MI, TmpReg}); + else if (MI.isCall()) + CallInstructions.push_back({&MI, TmpReg}); + } + + if (TmpRegisterNotAvailableEverywhere) { + // When a temporary register is not available everywhere in this basic + // basic block where a propagate-taint-to-sp operation is needed, just + // emit a full speculation barrier at the start of this basic block, which + // renders the taint/speculation tracking in this basic block unnecessary. + insertFullSpeculationBarrier(MBB, MBB.begin(), + (MBB.begin())->getDebugLoc()); + UsesFullSpeculationBarrier = true; + Modified = true; + } else { + for (auto MI_Reg : ReturnInstructions) { + assert(MI_Reg.second != 0); + LLVM_DEBUG( + dbgs() + << " About to insert Reg to SP taint propagation with temp register " + << printReg(MI_Reg.second, TRI) + << " on instruction: " << *MI_Reg.first); + insertRegToSPTaintPropagation(MBB, MI_Reg.first, MI_Reg.second); + Modified = true; + } + + for (auto MI_Reg : CallInstructions) { + assert(MI_Reg.second != 0); + LLVM_DEBUG(dbgs() << " About to insert Reg to SP and back taint " + "propagation with temp register " + << printReg(MI_Reg.second, TRI) + << " around instruction: " << *MI_Reg.first); // Just after the call: - MachineBasicBlock::iterator i = Call; - i++; - insertSPToRegTaintPropagation(Call->getParent(), i); + insertSPToRegTaintPropagation( + MBB, std::next((MachineBasicBlock::iterator)MI_Reg.first)); // Just before the call: - insertRegToSPTaintPropagation(Call->getParent(), Call, AArch64::X17); + insertRegToSPTaintPropagation(MBB, MI_Reg.first, MI_Reg.second); + Modified = true; } } - return Modified; } void AArch64SpeculationHardening::insertSPToRegTaintPropagation( - MachineBasicBlock *MBB, MachineBasicBlock::iterator MBBI) const { + MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI) const { // If full control flow speculation barriers are used, emit a control flow // barrier to block potential miss-speculation in flight coming in to this // function. if (UseControlFlowSpeculationBarrier) { - // insert full control flow speculation barrier (DSB SYS + ISB) - BuildMI(*MBB, MBBI, DebugLoc(), TII->get(AArch64::DSB)).addImm(0xf); - BuildMI(*MBB, MBBI, DebugLoc(), TII->get(AArch64::ISB)).addImm(0xf); + insertFullSpeculationBarrier(MBB, MBBI, DebugLoc()); return; } // CMP SP, #0 === SUBS xzr, SP, #0 - BuildMI(*MBB, MBBI, DebugLoc(), TII->get(AArch64::SUBSXri)) + BuildMI(MBB, MBBI, DebugLoc(), TII->get(AArch64::SUBSXri)) .addDef(AArch64::XZR) .addUse(AArch64::SP) .addImm(0) .addImm(0); // no shift // CSETM x16, NE === CSINV x16, xzr, xzr, EQ - BuildMI(*MBB, MBBI, DebugLoc(), TII->get(AArch64::CSINVXr)) + BuildMI(MBB, MBBI, DebugLoc(), TII->get(AArch64::CSINVXr)) .addDef(MisspeculatingTaintReg) .addUse(AArch64::XZR) .addUse(AArch64::XZR) .addImm(AArch64CC::EQ); } void AArch64SpeculationHardening::insertRegToSPTaintPropagation( - MachineBasicBlock *MBB, MachineBasicBlock::iterator MBBI, + MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, unsigned TmpReg) const { // If full control flow speculation barriers are used, there will not be // miss-speculation when returning from this function, and therefore, also // no need to encode potential miss-speculation into the stack pointer. if (UseControlFlowSpeculationBarrier) return; // mov Xtmp, SP === ADD Xtmp, SP, #0 - BuildMI(*MBB, MBBI, DebugLoc(), TII->get(AArch64::ADDXri)) + BuildMI(MBB, MBBI, DebugLoc(), TII->get(AArch64::ADDXri)) .addDef(TmpReg) .addUse(AArch64::SP) .addImm(0) .addImm(0); // no shift // and Xtmp, Xtmp, TaintReg === AND Xtmp, Xtmp, TaintReg, #0 - BuildMI(*MBB, MBBI, DebugLoc(), TII->get(AArch64::ANDXrs)) + BuildMI(MBB, MBBI, DebugLoc(), TII->get(AArch64::ANDXrs)) .addDef(TmpReg, RegState::Renamable) .addUse(TmpReg, RegState::Kill | RegState::Renamable) .addUse(MisspeculatingTaintReg, RegState::Kill) .addImm(0); // mov SP, Xtmp === ADD SP, Xtmp, #0 - BuildMI(*MBB, MBBI, DebugLoc(), TII->get(AArch64::ADDXri)) + BuildMI(MBB, MBBI, DebugLoc(), TII->get(AArch64::ADDXri)) .addDef(AArch64::SP) .addUse(TmpReg, RegState::Kill) .addImm(0) .addImm(0); // no shift } bool AArch64SpeculationHardening::functionUsesHardeningRegister( MachineFunction &MF) const { for (MachineBasicBlock &MBB : MF) { for (MachineInstr &MI : MBB) { // treat function calls specially, as the hardening register does not // need to remain live across function calls. if (MI.isCall()) continue; if (MI.readsRegister(MisspeculatingTaintReg, TRI) || MI.modifiesRegister(MisspeculatingTaintReg, TRI)) return true; } } return false; } // Make GPR register Reg speculation-safe by putting it through the // SpeculationSafeValue pseudo instruction, if we can't prove that // the value in the register has already been hardened. bool AArch64SpeculationHardening::makeGPRSpeculationSafe( MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, MachineInstr &MI, unsigned Reg) { assert(AArch64::GPR32allRegClass.contains(Reg) || AArch64::GPR64allRegClass.contains(Reg)); // Loads cannot directly load a value into the SP (nor WSP). // Therefore, if Reg is SP or WSP, it is because the instruction loads from // the stack through the stack pointer. // // Since the stack pointer is never dynamically controllable, don't harden it. if (Reg == AArch64::SP || Reg == AArch64::WSP) return false; // Do not harden the register again if already hardened before. if (RegsAlreadyMasked[Reg]) return false; const bool Is64Bit = AArch64::GPR64allRegClass.contains(Reg); LLVM_DEBUG(dbgs() << "About to harden register : " << Reg << "\n"); BuildMI(MBB, MBBI, MI.getDebugLoc(), TII->get(Is64Bit ? AArch64::SpeculationSafeValueX : AArch64::SpeculationSafeValueW)) .addDef(Reg) .addUse(Reg); RegsAlreadyMasked.set(Reg); return true; } bool AArch64SpeculationHardening::slhLoads(MachineBasicBlock &MBB) { bool Modified = false; LLVM_DEBUG(dbgs() << "slhLoads running on MBB: " << MBB); RegsAlreadyMasked.reset(); MachineBasicBlock::iterator MBBI = MBB.begin(), E = MBB.end(); MachineBasicBlock::iterator NextMBBI; for (; MBBI != E; MBBI = NextMBBI) { MachineInstr &MI = *MBBI; NextMBBI = std::next(MBBI); // Only harden loaded values or addresses used in loads. if (!MI.mayLoad()) continue; LLVM_DEBUG(dbgs() << "About to harden: " << MI); // For general purpose register loads, harden the registers loaded into. // For other loads, harden the address loaded from. // Masking the loaded value is expected to result in less performance // overhead, as the load can still execute speculatively in comparison to // when the address loaded from gets masked. However, masking is only // easy to do efficiently on GPR registers, so for loads into non-GPR // registers (e.g. floating point loads), mask the address loaded from. bool AllDefsAreGPR = llvm::all_of(MI.defs(), [&](MachineOperand &Op) { return Op.isReg() && (AArch64::GPR32allRegClass.contains(Op.getReg()) || AArch64::GPR64allRegClass.contains(Op.getReg())); }); // FIXME: it might be a worthwhile optimization to not mask loaded // values if all the registers involved in address calculation are already // hardened, leading to this load not able to execute on a miss-speculated // path. bool HardenLoadedData = AllDefsAreGPR; bool HardenAddressLoadedFrom = !HardenLoadedData; // First remove registers from AlreadyMaskedRegisters if their value is // updated by this instruction - it makes them contain a new value that is // not guaranteed to already have been masked. for (MachineOperand Op : MI.defs()) for (MCRegAliasIterator AI(Op.getReg(), TRI, true); AI.isValid(); ++AI) RegsAlreadyMasked.reset(*AI); // FIXME: loads from the stack with an immediate offset from the stack // pointer probably shouldn't be hardened, which could result in a // significant optimization. See section "Don’t check loads from // compile-time constant stack offsets", in // https://llvm.org/docs/SpeculativeLoadHardening.html if (HardenLoadedData) for (auto Def : MI.defs()) { if (Def.isDead()) // Do not mask a register that is not used further. continue; // FIXME: For pre/post-increment addressing modes, the base register // used in address calculation is also defined by this instruction. // It might be a worthwhile optimization to not harden that // base register increment/decrement when the increment/decrement is // an immediate. Modified |= makeGPRSpeculationSafe(MBB, NextMBBI, MI, Def.getReg()); } if (HardenAddressLoadedFrom) for (auto Use : MI.uses()) { if (!Use.isReg()) continue; unsigned Reg = Use.getReg(); // Some loads of floating point data have implicit defs/uses on a // super register of that floating point data. Some examples: // $s0 = LDRSui $sp, 22, implicit-def $q0 // $q0 = LD1i64 $q0, 1, renamable $x0 // We need to filter out these uses for non-GPR register which occur // because the load partially fills a non-GPR register with the loaded // data. Just skipping all non-GPR registers is safe (for now) as all // AArch64 load instructions only use GPR registers to perform the // address calculation. FIXME: However that might change once we can // produce SVE gather instructions. if (!(AArch64::GPR32allRegClass.contains(Reg) || AArch64::GPR64allRegClass.contains(Reg))) continue; Modified |= makeGPRSpeculationSafe(MBB, MBBI, MI, Reg); } } return Modified; } /// \brief If MBBI references a pseudo instruction that should be expanded /// here, do the expansion and return true. Otherwise return false. bool AArch64SpeculationHardening::expandSpeculationSafeValue( - MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI) { + MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, + bool UsesFullSpeculationBarrier) { MachineInstr &MI = *MBBI; unsigned Opcode = MI.getOpcode(); bool Is64Bit = true; switch (Opcode) { default: break; case AArch64::SpeculationSafeValueW: Is64Bit = false; LLVM_FALLTHROUGH; case AArch64::SpeculationSafeValueX: // Just remove the SpeculationSafe pseudo's if control flow // miss-speculation isn't happening because we're already inserting barriers // to guarantee that. - if (!UseControlFlowSpeculationBarrier) { + if (!UseControlFlowSpeculationBarrier && !UsesFullSpeculationBarrier) { unsigned DstReg = MI.getOperand(0).getReg(); unsigned SrcReg = MI.getOperand(1).getReg(); // Mark this register and all its aliasing registers as needing to be // value speculation hardened before its next use, by using a CSDB // barrier instruction. for (MachineOperand Op : MI.defs()) for (MCRegAliasIterator AI(Op.getReg(), TRI, true); AI.isValid(); ++AI) RegsNeedingCSDBBeforeUse.set(*AI); // Mask off with taint state. BuildMI(MBB, MBBI, MI.getDebugLoc(), Is64Bit ? TII->get(AArch64::ANDXrs) : TII->get(AArch64::ANDWrs)) .addDef(DstReg) .addUse(SrcReg, RegState::Kill) .addUse(Is64Bit ? MisspeculatingTaintReg : MisspeculatingTaintReg32Bit) .addImm(0); } MI.eraseFromParent(); return true; } return false; } bool AArch64SpeculationHardening::insertCSDB(MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, DebugLoc DL) { assert(!UseControlFlowSpeculationBarrier && "No need to insert CSDBs when " "control flow miss-speculation " "is already blocked"); // insert data value speculation barrier (CSDB) BuildMI(MBB, MBBI, DL, TII->get(AArch64::HINT)).addImm(0x14); RegsNeedingCSDBBeforeUse.reset(); return true; } bool AArch64SpeculationHardening::lowerSpeculationSafeValuePseudos( - MachineBasicBlock &MBB) { + MachineBasicBlock &MBB, bool UsesFullSpeculationBarrier) { bool Modified = false; RegsNeedingCSDBBeforeUse.reset(); // The following loop iterates over all instructions in the basic block, // and performs 2 operations: // 1. Insert a CSDB at this location if needed. // 2. Expand the SpeculationSafeValuePseudo if the current instruction is // one. // // The insertion of the CSDB is done as late as possible (i.e. just before // the use of a masked register), in the hope that that will reduce the // total number of CSDBs in a block when there are multiple masked registers // in the block. MachineBasicBlock::iterator MBBI = MBB.begin(), E = MBB.end(); DebugLoc DL; while (MBBI != E) { MachineInstr &MI = *MBBI; DL = MI.getDebugLoc(); MachineBasicBlock::iterator NMBBI = std::next(MBBI); // First check if a CSDB needs to be inserted due to earlier registers // that were masked and that are used by the next instruction. // Also emit the barrier on any potential control flow changes. bool NeedToEmitBarrier = false; if (RegsNeedingCSDBBeforeUse.any() && (MI.isCall() || MI.isTerminator())) NeedToEmitBarrier = true; if (!NeedToEmitBarrier) for (MachineOperand Op : MI.uses()) if (Op.isReg() && RegsNeedingCSDBBeforeUse[Op.getReg()]) { NeedToEmitBarrier = true; break; } - if (NeedToEmitBarrier) + if (NeedToEmitBarrier && !UsesFullSpeculationBarrier) Modified |= insertCSDB(MBB, MBBI, DL); - Modified |= expandSpeculationSafeValue(MBB, MBBI); + Modified |= + expandSpeculationSafeValue(MBB, MBBI, UsesFullSpeculationBarrier); MBBI = NMBBI; } - if (RegsNeedingCSDBBeforeUse.any()) + if (RegsNeedingCSDBBeforeUse.any() && !UsesFullSpeculationBarrier) Modified |= insertCSDB(MBB, MBBI, DL); return Modified; } bool AArch64SpeculationHardening::runOnMachineFunction(MachineFunction &MF) { if (!MF.getFunction().hasFnAttribute(Attribute::SpeculativeLoadHardening)) return false; MisspeculatingTaintReg = AArch64::X16; MisspeculatingTaintReg32Bit = AArch64::W16; TII = MF.getSubtarget().getInstrInfo(); TRI = MF.getSubtarget().getRegisterInfo(); RegsNeedingCSDBBeforeUse.resize(TRI->getNumRegs()); RegsAlreadyMasked.resize(TRI->getNumRegs()); UseControlFlowSpeculationBarrier = functionUsesHardeningRegister(MF); bool Modified = false; // Step 1: Enable automatic insertion of SpeculationSafeValue. if (HardenLoads) { LLVM_DEBUG( dbgs() << "***** AArch64SpeculationHardening - automatic insertion of " "SpeculationSafeValue intrinsics *****\n"); for (auto &MBB : MF) Modified |= slhLoads(MBB); } - // 2.a Add instrumentation code to function entry and exits. + // 2. Add instrumentation code to function entry and exits. LLVM_DEBUG( dbgs() << "***** AArch64SpeculationHardening - track control flow *****\n"); SmallVector EntryBlocks; EntryBlocks.push_back(&MF.front()); for (const LandingPadInfo &LPI : MF.getLandingPads()) EntryBlocks.push_back(LPI.LandingPadBlock); for (auto Entry : EntryBlocks) insertSPToRegTaintPropagation( - Entry, Entry->SkipPHIsLabelsAndDebug(Entry->begin())); + *Entry, Entry->SkipPHIsLabelsAndDebug(Entry->begin())); - // 2.b Add instrumentation code to every basic block. - for (auto &MBB : MF) - Modified |= instrumentControlFlow(MBB); - - LLVM_DEBUG(dbgs() << "***** AArch64SpeculationHardening - Lowering " - "SpeculationSafeValue Pseudos *****\n"); - // Step 3: Lower SpeculationSafeValue pseudo instructions. - for (auto &MBB : MF) - Modified |= lowerSpeculationSafeValuePseudos(MBB); + // 3. Add instrumentation code to every basic block. + for (auto &MBB : MF) { + bool UsesFullSpeculationBarrier = false; + Modified |= instrumentControlFlow(MBB, UsesFullSpeculationBarrier); + Modified |= + lowerSpeculationSafeValuePseudos(MBB, UsesFullSpeculationBarrier); + } return Modified; } /// \brief Returns an instance of the pseudo instruction expansion pass. FunctionPass *llvm::createAArch64SpeculationHardeningPass() { return new AArch64SpeculationHardening(); } Index: vendor/llvm/dist-release_80/lib/Target/Mips/AsmParser/MipsAsmParser.cpp =================================================================== --- vendor/llvm/dist-release_80/lib/Target/Mips/AsmParser/MipsAsmParser.cpp (revision 343793) +++ vendor/llvm/dist-release_80/lib/Target/Mips/AsmParser/MipsAsmParser.cpp (revision 343794) @@ -1,8232 +1,8229 @@ //===-- MipsAsmParser.cpp - Parse Mips assembly to MCInst instructions ----===// // // The LLVM Compiler Infrastructure // // This file is distributed under the University of Illinois Open Source // License. See LICENSE.TXT for details. // //===----------------------------------------------------------------------===// #include "MCTargetDesc/MipsABIFlagsSection.h" #include "MCTargetDesc/MipsABIInfo.h" #include "MCTargetDesc/MipsBaseInfo.h" #include "MCTargetDesc/MipsMCExpr.h" #include "MCTargetDesc/MipsMCTargetDesc.h" #include "MipsTargetStreamer.h" #include "llvm/ADT/APFloat.h" #include "llvm/ADT/STLExtras.h" #include "llvm/ADT/SmallVector.h" #include "llvm/ADT/StringRef.h" #include "llvm/ADT/StringSwitch.h" #include "llvm/ADT/Triple.h" #include "llvm/ADT/Twine.h" #include "llvm/BinaryFormat/ELF.h" #include "llvm/MC/MCContext.h" #include "llvm/MC/MCExpr.h" #include "llvm/MC/MCInst.h" #include "llvm/MC/MCInstrDesc.h" #include "llvm/MC/MCObjectFileInfo.h" #include "llvm/MC/MCParser/MCAsmLexer.h" #include "llvm/MC/MCParser/MCAsmParser.h" #include "llvm/MC/MCParser/MCAsmParserExtension.h" #include "llvm/MC/MCParser/MCParsedAsmOperand.h" #include "llvm/MC/MCParser/MCTargetAsmParser.h" #include "llvm/MC/MCSectionELF.h" #include "llvm/MC/MCStreamer.h" #include "llvm/MC/MCSubtargetInfo.h" #include "llvm/MC/MCSymbol.h" #include "llvm/MC/MCSymbolELF.h" #include "llvm/MC/MCValue.h" #include "llvm/MC/SubtargetFeature.h" #include "llvm/Support/Casting.h" #include "llvm/Support/CommandLine.h" #include "llvm/Support/Compiler.h" #include "llvm/Support/Debug.h" #include "llvm/Support/ErrorHandling.h" #include "llvm/Support/MathExtras.h" #include "llvm/Support/SMLoc.h" #include "llvm/Support/SourceMgr.h" #include "llvm/Support/TargetRegistry.h" #include "llvm/Support/raw_ostream.h" #include #include #include #include #include #include using namespace llvm; #define DEBUG_TYPE "mips-asm-parser" namespace llvm { class MCInstrInfo; } // end namespace llvm -static cl::opt -EmitJalrReloc("mips-jalr-reloc", cl::Hidden, - cl::desc("MIPS: Emit R_{MICRO}MIPS_JALR relocation with jalr"), - cl::init(true)); +extern cl::opt EmitJalrReloc; namespace { class MipsAssemblerOptions { public: MipsAssemblerOptions(const FeatureBitset &Features_) : Features(Features_) {} MipsAssemblerOptions(const MipsAssemblerOptions *Opts) { ATReg = Opts->getATRegIndex(); Reorder = Opts->isReorder(); Macro = Opts->isMacro(); Features = Opts->getFeatures(); } unsigned getATRegIndex() const { return ATReg; } bool setATRegIndex(unsigned Reg) { if (Reg > 31) return false; ATReg = Reg; return true; } bool isReorder() const { return Reorder; } void setReorder() { Reorder = true; } void setNoReorder() { Reorder = false; } bool isMacro() const { return Macro; } void setMacro() { Macro = true; } void setNoMacro() { Macro = false; } const FeatureBitset &getFeatures() const { return Features; } void setFeatures(const FeatureBitset &Features_) { Features = Features_; } // Set of features that are either architecture features or referenced // by them (e.g.: FeatureNaN2008 implied by FeatureMips32r6). // The full table can be found in MipsGenSubtargetInfo.inc (MipsFeatureKV[]). // The reason we need this mask is explained in the selectArch function. // FIXME: Ideally we would like TableGen to generate this information. static const FeatureBitset AllArchRelatedMask; private: unsigned ATReg = 1; bool Reorder = true; bool Macro = true; FeatureBitset Features; }; } // end anonymous namespace const FeatureBitset MipsAssemblerOptions::AllArchRelatedMask = { Mips::FeatureMips1, Mips::FeatureMips2, Mips::FeatureMips3, Mips::FeatureMips3_32, Mips::FeatureMips3_32r2, Mips::FeatureMips4, Mips::FeatureMips4_32, Mips::FeatureMips4_32r2, Mips::FeatureMips5, Mips::FeatureMips5_32r2, Mips::FeatureMips32, Mips::FeatureMips32r2, Mips::FeatureMips32r3, Mips::FeatureMips32r5, Mips::FeatureMips32r6, Mips::FeatureMips64, Mips::FeatureMips64r2, Mips::FeatureMips64r3, Mips::FeatureMips64r5, Mips::FeatureMips64r6, Mips::FeatureCnMips, Mips::FeatureFP64Bit, Mips::FeatureGP64Bit, Mips::FeatureNaN2008 }; namespace { class MipsAsmParser : public MCTargetAsmParser { MipsTargetStreamer &getTargetStreamer() { MCTargetStreamer &TS = *getParser().getStreamer().getTargetStreamer(); return static_cast(TS); } MipsABIInfo ABI; SmallVector, 2> AssemblerOptions; MCSymbol *CurrentFn; // Pointer to the function being parsed. It may be a // nullptr, which indicates that no function is currently // selected. This usually happens after an '.end func' // directive. bool IsLittleEndian; bool IsPicEnabled; bool IsCpRestoreSet; int CpRestoreOffset; unsigned CpSaveLocation; /// If true, then CpSaveLocation is a register, otherwise it's an offset. bool CpSaveLocationIsRegister; // Map of register aliases created via the .set directive. StringMap RegisterSets; // Print a warning along with its fix-it message at the given range. void printWarningWithFixIt(const Twine &Msg, const Twine &FixMsg, SMRange Range, bool ShowColors = true); void ConvertXWPOperands(MCInst &Inst, const OperandVector &Operands); #define GET_ASSEMBLER_HEADER #include "MipsGenAsmMatcher.inc" unsigned checkEarlyTargetMatchPredicate(MCInst &Inst, const OperandVector &Operands) override; unsigned checkTargetMatchPredicate(MCInst &Inst) override; bool MatchAndEmitInstruction(SMLoc IDLoc, unsigned &Opcode, OperandVector &Operands, MCStreamer &Out, uint64_t &ErrorInfo, bool MatchingInlineAsm) override; /// Parse a register as used in CFI directives bool ParseRegister(unsigned &RegNo, SMLoc &StartLoc, SMLoc &EndLoc) override; bool parseParenSuffix(StringRef Name, OperandVector &Operands); bool parseBracketSuffix(StringRef Name, OperandVector &Operands); bool mnemonicIsValid(StringRef Mnemonic, unsigned VariantID); bool ParseInstruction(ParseInstructionInfo &Info, StringRef Name, SMLoc NameLoc, OperandVector &Operands) override; bool ParseDirective(AsmToken DirectiveID) override; OperandMatchResultTy parseMemOperand(OperandVector &Operands); OperandMatchResultTy matchAnyRegisterNameWithoutDollar(OperandVector &Operands, StringRef Identifier, SMLoc S); OperandMatchResultTy matchAnyRegisterWithoutDollar(OperandVector &Operands, const AsmToken &Token, SMLoc S); OperandMatchResultTy matchAnyRegisterWithoutDollar(OperandVector &Operands, SMLoc S); OperandMatchResultTy parseAnyRegister(OperandVector &Operands); OperandMatchResultTy parseImm(OperandVector &Operands); OperandMatchResultTy parseJumpTarget(OperandVector &Operands); OperandMatchResultTy parseInvNum(OperandVector &Operands); OperandMatchResultTy parseRegisterList(OperandVector &Operands); bool searchSymbolAlias(OperandVector &Operands); bool parseOperand(OperandVector &, StringRef Mnemonic); enum MacroExpanderResultTy { MER_NotAMacro, MER_Success, MER_Fail, }; // Expands assembly pseudo instructions. MacroExpanderResultTy tryExpandInstruction(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI); bool expandJalWithRegs(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI); bool loadImmediate(int64_t ImmValue, unsigned DstReg, unsigned SrcReg, bool Is32BitImm, bool IsAddress, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI); bool loadAndAddSymbolAddress(const MCExpr *SymExpr, unsigned DstReg, unsigned SrcReg, bool Is32BitSym, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI); bool emitPartialAddress(MipsTargetStreamer &TOut, SMLoc IDLoc, MCSymbol *Sym); bool expandLoadImm(MCInst &Inst, bool Is32BitImm, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI); bool expandLoadImmReal(MCInst &Inst, bool IsSingle, bool IsGPR, bool Is64FPU, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI); bool expandLoadAddress(unsigned DstReg, unsigned BaseReg, const MCOperand &Offset, bool Is32BitAddress, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI); bool expandUncondBranchMMPseudo(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI); void expandMemInst(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI, bool IsLoad); bool expandLoadStoreMultiple(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI); bool expandAliasImmediate(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI); bool expandBranchImm(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI); bool expandCondBranches(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI); bool expandDivRem(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI, const bool IsMips64, const bool Signed); bool expandTrunc(MCInst &Inst, bool IsDouble, bool Is64FPU, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI); bool expandUlh(MCInst &Inst, bool Signed, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI); bool expandUsh(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI); bool expandUxw(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI); bool expandRotation(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI); bool expandRotationImm(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI); bool expandDRotation(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI); bool expandDRotationImm(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI); bool expandAbs(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI); bool expandMulImm(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI); bool expandMulO(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI); bool expandMulOU(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI); bool expandDMULMacro(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI); bool expandLoadStoreDMacro(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI, bool IsLoad); bool expandSeq(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI); bool expandSeqI(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI); bool expandMXTRAlias(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI); bool reportParseError(Twine ErrorMsg); bool reportParseError(SMLoc Loc, Twine ErrorMsg); bool parseMemOffset(const MCExpr *&Res, bool isParenExpr); bool isEvaluated(const MCExpr *Expr); bool parseSetMips0Directive(); bool parseSetArchDirective(); bool parseSetFeature(uint64_t Feature); bool isPicAndNotNxxAbi(); // Used by .cpload, .cprestore, and .cpsetup. bool parseDirectiveCpLoad(SMLoc Loc); bool parseDirectiveCpRestore(SMLoc Loc); bool parseDirectiveCPSetup(); bool parseDirectiveCPReturn(); bool parseDirectiveNaN(); bool parseDirectiveSet(); bool parseDirectiveOption(); bool parseInsnDirective(); bool parseRSectionDirective(StringRef Section); bool parseSSectionDirective(StringRef Section, unsigned Type); bool parseSetAtDirective(); bool parseSetNoAtDirective(); bool parseSetMacroDirective(); bool parseSetNoMacroDirective(); bool parseSetMsaDirective(); bool parseSetNoMsaDirective(); bool parseSetNoDspDirective(); bool parseSetReorderDirective(); bool parseSetNoReorderDirective(); bool parseSetMips16Directive(); bool parseSetNoMips16Directive(); bool parseSetFpDirective(); bool parseSetOddSPRegDirective(); bool parseSetNoOddSPRegDirective(); bool parseSetPopDirective(); bool parseSetPushDirective(); bool parseSetSoftFloatDirective(); bool parseSetHardFloatDirective(); bool parseSetMtDirective(); bool parseSetNoMtDirective(); bool parseSetNoCRCDirective(); bool parseSetNoVirtDirective(); bool parseSetNoGINVDirective(); bool parseSetAssignment(); bool parseDirectiveGpWord(); bool parseDirectiveGpDWord(); bool parseDirectiveDtpRelWord(); bool parseDirectiveDtpRelDWord(); bool parseDirectiveTpRelWord(); bool parseDirectiveTpRelDWord(); bool parseDirectiveModule(); bool parseDirectiveModuleFP(); bool parseFpABIValue(MipsABIFlagsSection::FpABIKind &FpABI, StringRef Directive); bool parseInternalDirectiveReallowModule(); bool eatComma(StringRef ErrorStr); int matchCPURegisterName(StringRef Symbol); int matchHWRegsRegisterName(StringRef Symbol); int matchFPURegisterName(StringRef Name); int matchFCCRegisterName(StringRef Name); int matchACRegisterName(StringRef Name); int matchMSA128RegisterName(StringRef Name); int matchMSA128CtrlRegisterName(StringRef Name); unsigned getReg(int RC, int RegNo); /// Returns the internal register number for the current AT. Also checks if /// the current AT is unavailable (set to $0) and gives an error if it is. /// This should be used in pseudo-instruction expansions which need AT. unsigned getATReg(SMLoc Loc); bool canUseATReg(); bool processInstruction(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI); // Helper function that checks if the value of a vector index is within the // boundaries of accepted values for each RegisterKind // Example: INSERT.B $w0[n], $1 => 16 > n >= 0 bool validateMSAIndex(int Val, int RegKind); // Selects a new architecture by updating the FeatureBits with the necessary // info including implied dependencies. // Internally, it clears all the feature bits related to *any* architecture // and selects the new one using the ToggleFeature functionality of the // MCSubtargetInfo object that handles implied dependencies. The reason we // clear all the arch related bits manually is because ToggleFeature only // clears the features that imply the feature being cleared and not the // features implied by the feature being cleared. This is easier to see // with an example: // -------------------------------------------------- // | Feature | Implies | // | -------------------------------------------------| // | FeatureMips1 | None | // | FeatureMips2 | FeatureMips1 | // | FeatureMips3 | FeatureMips2 | FeatureMipsGP64 | // | FeatureMips4 | FeatureMips3 | // | ... | | // -------------------------------------------------- // // Setting Mips3 is equivalent to set: (FeatureMips3 | FeatureMips2 | // FeatureMipsGP64 | FeatureMips1) // Clearing Mips3 is equivalent to clear (FeatureMips3 | FeatureMips4). void selectArch(StringRef ArchFeature) { MCSubtargetInfo &STI = copySTI(); FeatureBitset FeatureBits = STI.getFeatureBits(); FeatureBits &= ~MipsAssemblerOptions::AllArchRelatedMask; STI.setFeatureBits(FeatureBits); setAvailableFeatures( ComputeAvailableFeatures(STI.ToggleFeature(ArchFeature))); AssemblerOptions.back()->setFeatures(STI.getFeatureBits()); } void setFeatureBits(uint64_t Feature, StringRef FeatureString) { if (!(getSTI().getFeatureBits()[Feature])) { MCSubtargetInfo &STI = copySTI(); setAvailableFeatures( ComputeAvailableFeatures(STI.ToggleFeature(FeatureString))); AssemblerOptions.back()->setFeatures(STI.getFeatureBits()); } } void clearFeatureBits(uint64_t Feature, StringRef FeatureString) { if (getSTI().getFeatureBits()[Feature]) { MCSubtargetInfo &STI = copySTI(); setAvailableFeatures( ComputeAvailableFeatures(STI.ToggleFeature(FeatureString))); AssemblerOptions.back()->setFeatures(STI.getFeatureBits()); } } void setModuleFeatureBits(uint64_t Feature, StringRef FeatureString) { setFeatureBits(Feature, FeatureString); AssemblerOptions.front()->setFeatures(getSTI().getFeatureBits()); } void clearModuleFeatureBits(uint64_t Feature, StringRef FeatureString) { clearFeatureBits(Feature, FeatureString); AssemblerOptions.front()->setFeatures(getSTI().getFeatureBits()); } public: enum MipsMatchResultTy { Match_RequiresDifferentSrcAndDst = FIRST_TARGET_MATCH_RESULT_TY, Match_RequiresDifferentOperands, Match_RequiresNoZeroRegister, Match_RequiresSameSrcAndDst, Match_NoFCCRegisterForCurrentISA, Match_NonZeroOperandForSync, Match_NonZeroOperandForMTCX, Match_RequiresPosSizeRange0_32, Match_RequiresPosSizeRange33_64, Match_RequiresPosSizeUImm6, #define GET_OPERAND_DIAGNOSTIC_TYPES #include "MipsGenAsmMatcher.inc" #undef GET_OPERAND_DIAGNOSTIC_TYPES }; MipsAsmParser(const MCSubtargetInfo &sti, MCAsmParser &parser, const MCInstrInfo &MII, const MCTargetOptions &Options) : MCTargetAsmParser(Options, sti, MII), ABI(MipsABIInfo::computeTargetABI(Triple(sti.getTargetTriple()), sti.getCPU(), Options)) { MCAsmParserExtension::Initialize(parser); parser.addAliasForDirective(".asciiz", ".asciz"); parser.addAliasForDirective(".hword", ".2byte"); parser.addAliasForDirective(".word", ".4byte"); parser.addAliasForDirective(".dword", ".8byte"); // Initialize the set of available features. setAvailableFeatures(ComputeAvailableFeatures(getSTI().getFeatureBits())); // Remember the initial assembler options. The user can not modify these. AssemblerOptions.push_back( llvm::make_unique(getSTI().getFeatureBits())); // Create an assembler options environment for the user to modify. AssemblerOptions.push_back( llvm::make_unique(getSTI().getFeatureBits())); getTargetStreamer().updateABIInfo(*this); if (!isABI_O32() && !useOddSPReg() != 0) report_fatal_error("-mno-odd-spreg requires the O32 ABI"); CurrentFn = nullptr; IsPicEnabled = getContext().getObjectFileInfo()->isPositionIndependent(); IsCpRestoreSet = false; CpRestoreOffset = -1; const Triple &TheTriple = sti.getTargetTriple(); IsLittleEndian = TheTriple.isLittleEndian(); if (getSTI().getCPU() == "mips64r6" && inMicroMipsMode()) report_fatal_error("microMIPS64R6 is not supported", false); if (!isABI_O32() && inMicroMipsMode()) report_fatal_error("microMIPS64 is not supported", false); } /// True if all of $fcc0 - $fcc7 exist for the current ISA. bool hasEightFccRegisters() const { return hasMips4() || hasMips32(); } bool isGP64bit() const { return getSTI().getFeatureBits()[Mips::FeatureGP64Bit]; } bool isFP64bit() const { return getSTI().getFeatureBits()[Mips::FeatureFP64Bit]; } const MipsABIInfo &getABI() const { return ABI; } bool isABI_N32() const { return ABI.IsN32(); } bool isABI_N64() const { return ABI.IsN64(); } bool isABI_O32() const { return ABI.IsO32(); } bool isABI_FPXX() const { return getSTI().getFeatureBits()[Mips::FeatureFPXX]; } bool useOddSPReg() const { return !(getSTI().getFeatureBits()[Mips::FeatureNoOddSPReg]); } bool inMicroMipsMode() const { return getSTI().getFeatureBits()[Mips::FeatureMicroMips]; } bool hasMips1() const { return getSTI().getFeatureBits()[Mips::FeatureMips1]; } bool hasMips2() const { return getSTI().getFeatureBits()[Mips::FeatureMips2]; } bool hasMips3() const { return getSTI().getFeatureBits()[Mips::FeatureMips3]; } bool hasMips4() const { return getSTI().getFeatureBits()[Mips::FeatureMips4]; } bool hasMips5() const { return getSTI().getFeatureBits()[Mips::FeatureMips5]; } bool hasMips32() const { return getSTI().getFeatureBits()[Mips::FeatureMips32]; } bool hasMips64() const { return getSTI().getFeatureBits()[Mips::FeatureMips64]; } bool hasMips32r2() const { return getSTI().getFeatureBits()[Mips::FeatureMips32r2]; } bool hasMips64r2() const { return getSTI().getFeatureBits()[Mips::FeatureMips64r2]; } bool hasMips32r3() const { return (getSTI().getFeatureBits()[Mips::FeatureMips32r3]); } bool hasMips64r3() const { return (getSTI().getFeatureBits()[Mips::FeatureMips64r3]); } bool hasMips32r5() const { return (getSTI().getFeatureBits()[Mips::FeatureMips32r5]); } bool hasMips64r5() const { return (getSTI().getFeatureBits()[Mips::FeatureMips64r5]); } bool hasMips32r6() const { return getSTI().getFeatureBits()[Mips::FeatureMips32r6]; } bool hasMips64r6() const { return getSTI().getFeatureBits()[Mips::FeatureMips64r6]; } bool hasDSP() const { return getSTI().getFeatureBits()[Mips::FeatureDSP]; } bool hasDSPR2() const { return getSTI().getFeatureBits()[Mips::FeatureDSPR2]; } bool hasDSPR3() const { return getSTI().getFeatureBits()[Mips::FeatureDSPR3]; } bool hasMSA() const { return getSTI().getFeatureBits()[Mips::FeatureMSA]; } bool hasCnMips() const { return (getSTI().getFeatureBits()[Mips::FeatureCnMips]); } bool inPicMode() { return IsPicEnabled; } bool inMips16Mode() const { return getSTI().getFeatureBits()[Mips::FeatureMips16]; } bool useTraps() const { return getSTI().getFeatureBits()[Mips::FeatureUseTCCInDIV]; } bool useSoftFloat() const { return getSTI().getFeatureBits()[Mips::FeatureSoftFloat]; } bool hasMT() const { return getSTI().getFeatureBits()[Mips::FeatureMT]; } bool hasCRC() const { return getSTI().getFeatureBits()[Mips::FeatureCRC]; } bool hasVirt() const { return getSTI().getFeatureBits()[Mips::FeatureVirt]; } bool hasGINV() const { return getSTI().getFeatureBits()[Mips::FeatureGINV]; } /// Warn if RegIndex is the same as the current AT. void warnIfRegIndexIsAT(unsigned RegIndex, SMLoc Loc); void warnIfNoMacro(SMLoc Loc); bool isLittle() const { return IsLittleEndian; } const MCExpr *createTargetUnaryExpr(const MCExpr *E, AsmToken::TokenKind OperatorToken, MCContext &Ctx) override { switch(OperatorToken) { default: llvm_unreachable("Unknown token"); return nullptr; case AsmToken::PercentCall16: return MipsMCExpr::create(MipsMCExpr::MEK_GOT_CALL, E, Ctx); case AsmToken::PercentCall_Hi: return MipsMCExpr::create(MipsMCExpr::MEK_CALL_HI16, E, Ctx); case AsmToken::PercentCall_Lo: return MipsMCExpr::create(MipsMCExpr::MEK_CALL_LO16, E, Ctx); case AsmToken::PercentDtprel_Hi: return MipsMCExpr::create(MipsMCExpr::MEK_DTPREL_HI, E, Ctx); case AsmToken::PercentDtprel_Lo: return MipsMCExpr::create(MipsMCExpr::MEK_DTPREL_LO, E, Ctx); case AsmToken::PercentGot: return MipsMCExpr::create(MipsMCExpr::MEK_GOT, E, Ctx); case AsmToken::PercentGot_Disp: return MipsMCExpr::create(MipsMCExpr::MEK_GOT_DISP, E, Ctx); case AsmToken::PercentGot_Hi: return MipsMCExpr::create(MipsMCExpr::MEK_GOT_HI16, E, Ctx); case AsmToken::PercentGot_Lo: return MipsMCExpr::create(MipsMCExpr::MEK_GOT_LO16, E, Ctx); case AsmToken::PercentGot_Ofst: return MipsMCExpr::create(MipsMCExpr::MEK_GOT_OFST, E, Ctx); case AsmToken::PercentGot_Page: return MipsMCExpr::create(MipsMCExpr::MEK_GOT_PAGE, E, Ctx); case AsmToken::PercentGottprel: return MipsMCExpr::create(MipsMCExpr::MEK_GOTTPREL, E, Ctx); case AsmToken::PercentGp_Rel: return MipsMCExpr::create(MipsMCExpr::MEK_GPREL, E, Ctx); case AsmToken::PercentHi: return MipsMCExpr::create(MipsMCExpr::MEK_HI, E, Ctx); case AsmToken::PercentHigher: return MipsMCExpr::create(MipsMCExpr::MEK_HIGHER, E, Ctx); case AsmToken::PercentHighest: return MipsMCExpr::create(MipsMCExpr::MEK_HIGHEST, E, Ctx); case AsmToken::PercentLo: return MipsMCExpr::create(MipsMCExpr::MEK_LO, E, Ctx); case AsmToken::PercentNeg: return MipsMCExpr::create(MipsMCExpr::MEK_NEG, E, Ctx); case AsmToken::PercentPcrel_Hi: return MipsMCExpr::create(MipsMCExpr::MEK_PCREL_HI16, E, Ctx); case AsmToken::PercentPcrel_Lo: return MipsMCExpr::create(MipsMCExpr::MEK_PCREL_LO16, E, Ctx); case AsmToken::PercentTlsgd: return MipsMCExpr::create(MipsMCExpr::MEK_TLSGD, E, Ctx); case AsmToken::PercentTlsldm: return MipsMCExpr::create(MipsMCExpr::MEK_TLSLDM, E, Ctx); case AsmToken::PercentTprel_Hi: return MipsMCExpr::create(MipsMCExpr::MEK_TPREL_HI, E, Ctx); case AsmToken::PercentTprel_Lo: return MipsMCExpr::create(MipsMCExpr::MEK_TPREL_LO, E, Ctx); } } }; /// MipsOperand - Instances of this class represent a parsed Mips machine /// instruction. class MipsOperand : public MCParsedAsmOperand { public: /// Broad categories of register classes /// The exact class is finalized by the render method. enum RegKind { RegKind_GPR = 1, /// GPR32 and GPR64 (depending on isGP64bit()) RegKind_FGR = 2, /// FGR32, FGR64, AFGR64 (depending on context and /// isFP64bit()) RegKind_FCC = 4, /// FCC RegKind_MSA128 = 8, /// MSA128[BHWD] (makes no difference which) RegKind_MSACtrl = 16, /// MSA control registers RegKind_COP2 = 32, /// COP2 RegKind_ACC = 64, /// HI32DSP, LO32DSP, and ACC64DSP (depending on /// context). RegKind_CCR = 128, /// CCR RegKind_HWRegs = 256, /// HWRegs RegKind_COP3 = 512, /// COP3 RegKind_COP0 = 1024, /// COP0 /// Potentially any (e.g. $1) RegKind_Numeric = RegKind_GPR | RegKind_FGR | RegKind_FCC | RegKind_MSA128 | RegKind_MSACtrl | RegKind_COP2 | RegKind_ACC | RegKind_CCR | RegKind_HWRegs | RegKind_COP3 | RegKind_COP0 }; private: enum KindTy { k_Immediate, /// An immediate (possibly involving symbol references) k_Memory, /// Base + Offset Memory Address k_RegisterIndex, /// A register index in one or more RegKind. k_Token, /// A simple token k_RegList, /// A physical register list } Kind; public: MipsOperand(KindTy K, MipsAsmParser &Parser) : MCParsedAsmOperand(), Kind(K), AsmParser(Parser) {} ~MipsOperand() override { switch (Kind) { case k_Memory: delete Mem.Base; break; case k_RegList: delete RegList.List; break; case k_Immediate: case k_RegisterIndex: case k_Token: break; } } private: /// For diagnostics, and checking the assembler temporary MipsAsmParser &AsmParser; struct Token { const char *Data; unsigned Length; }; struct RegIdxOp { unsigned Index; /// Index into the register class RegKind Kind; /// Bitfield of the kinds it could possibly be struct Token Tok; /// The input token this operand originated from. const MCRegisterInfo *RegInfo; }; struct ImmOp { const MCExpr *Val; }; struct MemOp { MipsOperand *Base; const MCExpr *Off; }; struct RegListOp { SmallVector *List; }; union { struct Token Tok; struct RegIdxOp RegIdx; struct ImmOp Imm; struct MemOp Mem; struct RegListOp RegList; }; SMLoc StartLoc, EndLoc; /// Internal constructor for register kinds static std::unique_ptr CreateReg(unsigned Index, StringRef Str, RegKind RegKind, const MCRegisterInfo *RegInfo, SMLoc S, SMLoc E, MipsAsmParser &Parser) { auto Op = llvm::make_unique(k_RegisterIndex, Parser); Op->RegIdx.Index = Index; Op->RegIdx.RegInfo = RegInfo; Op->RegIdx.Kind = RegKind; Op->RegIdx.Tok.Data = Str.data(); Op->RegIdx.Tok.Length = Str.size(); Op->StartLoc = S; Op->EndLoc = E; return Op; } public: /// Coerce the register to GPR32 and return the real register for the current /// target. unsigned getGPR32Reg() const { assert(isRegIdx() && (RegIdx.Kind & RegKind_GPR) && "Invalid access!"); AsmParser.warnIfRegIndexIsAT(RegIdx.Index, StartLoc); unsigned ClassID = Mips::GPR32RegClassID; return RegIdx.RegInfo->getRegClass(ClassID).getRegister(RegIdx.Index); } /// Coerce the register to GPR32 and return the real register for the current /// target. unsigned getGPRMM16Reg() const { assert(isRegIdx() && (RegIdx.Kind & RegKind_GPR) && "Invalid access!"); unsigned ClassID = Mips::GPR32RegClassID; return RegIdx.RegInfo->getRegClass(ClassID).getRegister(RegIdx.Index); } /// Coerce the register to GPR64 and return the real register for the current /// target. unsigned getGPR64Reg() const { assert(isRegIdx() && (RegIdx.Kind & RegKind_GPR) && "Invalid access!"); unsigned ClassID = Mips::GPR64RegClassID; return RegIdx.RegInfo->getRegClass(ClassID).getRegister(RegIdx.Index); } private: /// Coerce the register to AFGR64 and return the real register for the current /// target. unsigned getAFGR64Reg() const { assert(isRegIdx() && (RegIdx.Kind & RegKind_FGR) && "Invalid access!"); if (RegIdx.Index % 2 != 0) AsmParser.Warning(StartLoc, "Float register should be even."); return RegIdx.RegInfo->getRegClass(Mips::AFGR64RegClassID) .getRegister(RegIdx.Index / 2); } /// Coerce the register to FGR64 and return the real register for the current /// target. unsigned getFGR64Reg() const { assert(isRegIdx() && (RegIdx.Kind & RegKind_FGR) && "Invalid access!"); return RegIdx.RegInfo->getRegClass(Mips::FGR64RegClassID) .getRegister(RegIdx.Index); } /// Coerce the register to FGR32 and return the real register for the current /// target. unsigned getFGR32Reg() const { assert(isRegIdx() && (RegIdx.Kind & RegKind_FGR) && "Invalid access!"); return RegIdx.RegInfo->getRegClass(Mips::FGR32RegClassID) .getRegister(RegIdx.Index); } /// Coerce the register to FGRH32 and return the real register for the current /// target. unsigned getFGRH32Reg() const { assert(isRegIdx() && (RegIdx.Kind & RegKind_FGR) && "Invalid access!"); return RegIdx.RegInfo->getRegClass(Mips::FGRH32RegClassID) .getRegister(RegIdx.Index); } /// Coerce the register to FCC and return the real register for the current /// target. unsigned getFCCReg() const { assert(isRegIdx() && (RegIdx.Kind & RegKind_FCC) && "Invalid access!"); return RegIdx.RegInfo->getRegClass(Mips::FCCRegClassID) .getRegister(RegIdx.Index); } /// Coerce the register to MSA128 and return the real register for the current /// target. unsigned getMSA128Reg() const { assert(isRegIdx() && (RegIdx.Kind & RegKind_MSA128) && "Invalid access!"); // It doesn't matter which of the MSA128[BHWD] classes we use. They are all // identical unsigned ClassID = Mips::MSA128BRegClassID; return RegIdx.RegInfo->getRegClass(ClassID).getRegister(RegIdx.Index); } /// Coerce the register to MSACtrl and return the real register for the /// current target. unsigned getMSACtrlReg() const { assert(isRegIdx() && (RegIdx.Kind & RegKind_MSACtrl) && "Invalid access!"); unsigned ClassID = Mips::MSACtrlRegClassID; return RegIdx.RegInfo->getRegClass(ClassID).getRegister(RegIdx.Index); } /// Coerce the register to COP0 and return the real register for the /// current target. unsigned getCOP0Reg() const { assert(isRegIdx() && (RegIdx.Kind & RegKind_COP0) && "Invalid access!"); unsigned ClassID = Mips::COP0RegClassID; return RegIdx.RegInfo->getRegClass(ClassID).getRegister(RegIdx.Index); } /// Coerce the register to COP2 and return the real register for the /// current target. unsigned getCOP2Reg() const { assert(isRegIdx() && (RegIdx.Kind & RegKind_COP2) && "Invalid access!"); unsigned ClassID = Mips::COP2RegClassID; return RegIdx.RegInfo->getRegClass(ClassID).getRegister(RegIdx.Index); } /// Coerce the register to COP3 and return the real register for the /// current target. unsigned getCOP3Reg() const { assert(isRegIdx() && (RegIdx.Kind & RegKind_COP3) && "Invalid access!"); unsigned ClassID = Mips::COP3RegClassID; return RegIdx.RegInfo->getRegClass(ClassID).getRegister(RegIdx.Index); } /// Coerce the register to ACC64DSP and return the real register for the /// current target. unsigned getACC64DSPReg() const { assert(isRegIdx() && (RegIdx.Kind & RegKind_ACC) && "Invalid access!"); unsigned ClassID = Mips::ACC64DSPRegClassID; return RegIdx.RegInfo->getRegClass(ClassID).getRegister(RegIdx.Index); } /// Coerce the register to HI32DSP and return the real register for the /// current target. unsigned getHI32DSPReg() const { assert(isRegIdx() && (RegIdx.Kind & RegKind_ACC) && "Invalid access!"); unsigned ClassID = Mips::HI32DSPRegClassID; return RegIdx.RegInfo->getRegClass(ClassID).getRegister(RegIdx.Index); } /// Coerce the register to LO32DSP and return the real register for the /// current target. unsigned getLO32DSPReg() const { assert(isRegIdx() && (RegIdx.Kind & RegKind_ACC) && "Invalid access!"); unsigned ClassID = Mips::LO32DSPRegClassID; return RegIdx.RegInfo->getRegClass(ClassID).getRegister(RegIdx.Index); } /// Coerce the register to CCR and return the real register for the /// current target. unsigned getCCRReg() const { assert(isRegIdx() && (RegIdx.Kind & RegKind_CCR) && "Invalid access!"); unsigned ClassID = Mips::CCRRegClassID; return RegIdx.RegInfo->getRegClass(ClassID).getRegister(RegIdx.Index); } /// Coerce the register to HWRegs and return the real register for the /// current target. unsigned getHWRegsReg() const { assert(isRegIdx() && (RegIdx.Kind & RegKind_HWRegs) && "Invalid access!"); unsigned ClassID = Mips::HWRegsRegClassID; return RegIdx.RegInfo->getRegClass(ClassID).getRegister(RegIdx.Index); } public: void addExpr(MCInst &Inst, const MCExpr *Expr) const { // Add as immediate when possible. Null MCExpr = 0. if (!Expr) Inst.addOperand(MCOperand::createImm(0)); else if (const MCConstantExpr *CE = dyn_cast(Expr)) Inst.addOperand(MCOperand::createImm(CE->getValue())); else Inst.addOperand(MCOperand::createExpr(Expr)); } void addRegOperands(MCInst &Inst, unsigned N) const { llvm_unreachable("Use a custom parser instead"); } /// Render the operand to an MCInst as a GPR32 /// Asserts if the wrong number of operands are requested, or the operand /// is not a k_RegisterIndex compatible with RegKind_GPR void addGPR32ZeroAsmRegOperands(MCInst &Inst, unsigned N) const { assert(N == 1 && "Invalid number of operands!"); Inst.addOperand(MCOperand::createReg(getGPR32Reg())); } void addGPR32NonZeroAsmRegOperands(MCInst &Inst, unsigned N) const { assert(N == 1 && "Invalid number of operands!"); Inst.addOperand(MCOperand::createReg(getGPR32Reg())); } void addGPR32AsmRegOperands(MCInst &Inst, unsigned N) const { assert(N == 1 && "Invalid number of operands!"); Inst.addOperand(MCOperand::createReg(getGPR32Reg())); } void addGPRMM16AsmRegOperands(MCInst &Inst, unsigned N) const { assert(N == 1 && "Invalid number of operands!"); Inst.addOperand(MCOperand::createReg(getGPRMM16Reg())); } void addGPRMM16AsmRegZeroOperands(MCInst &Inst, unsigned N) const { assert(N == 1 && "Invalid number of operands!"); Inst.addOperand(MCOperand::createReg(getGPRMM16Reg())); } void addGPRMM16AsmRegMovePOperands(MCInst &Inst, unsigned N) const { assert(N == 1 && "Invalid number of operands!"); Inst.addOperand(MCOperand::createReg(getGPRMM16Reg())); } void addGPRMM16AsmRegMovePPairFirstOperands(MCInst &Inst, unsigned N) const { assert(N == 1 && "Invalid number of operands!"); Inst.addOperand(MCOperand::createReg(getGPRMM16Reg())); } void addGPRMM16AsmRegMovePPairSecondOperands(MCInst &Inst, unsigned N) const { assert(N == 1 && "Invalid number of operands!"); Inst.addOperand(MCOperand::createReg(getGPRMM16Reg())); } /// Render the operand to an MCInst as a GPR64 /// Asserts if the wrong number of operands are requested, or the operand /// is not a k_RegisterIndex compatible with RegKind_GPR void addGPR64AsmRegOperands(MCInst &Inst, unsigned N) const { assert(N == 1 && "Invalid number of operands!"); Inst.addOperand(MCOperand::createReg(getGPR64Reg())); } void addAFGR64AsmRegOperands(MCInst &Inst, unsigned N) const { assert(N == 1 && "Invalid number of operands!"); Inst.addOperand(MCOperand::createReg(getAFGR64Reg())); } void addStrictlyAFGR64AsmRegOperands(MCInst &Inst, unsigned N) const { assert(N == 1 && "Invalid number of operands!"); Inst.addOperand(MCOperand::createReg(getAFGR64Reg())); } void addStrictlyFGR64AsmRegOperands(MCInst &Inst, unsigned N) const { assert(N == 1 && "Invalid number of operands!"); Inst.addOperand(MCOperand::createReg(getFGR64Reg())); } void addFGR64AsmRegOperands(MCInst &Inst, unsigned N) const { assert(N == 1 && "Invalid number of operands!"); Inst.addOperand(MCOperand::createReg(getFGR64Reg())); } void addFGR32AsmRegOperands(MCInst &Inst, unsigned N) const { assert(N == 1 && "Invalid number of operands!"); Inst.addOperand(MCOperand::createReg(getFGR32Reg())); // FIXME: We ought to do this for -integrated-as without -via-file-asm too. // FIXME: This should propagate failure up to parseStatement. if (!AsmParser.useOddSPReg() && RegIdx.Index & 1) AsmParser.getParser().printError( StartLoc, "-mno-odd-spreg prohibits the use of odd FPU " "registers"); } void addStrictlyFGR32AsmRegOperands(MCInst &Inst, unsigned N) const { assert(N == 1 && "Invalid number of operands!"); Inst.addOperand(MCOperand::createReg(getFGR32Reg())); // FIXME: We ought to do this for -integrated-as without -via-file-asm too. if (!AsmParser.useOddSPReg() && RegIdx.Index & 1) AsmParser.Error(StartLoc, "-mno-odd-spreg prohibits the use of odd FPU " "registers"); } void addFGRH32AsmRegOperands(MCInst &Inst, unsigned N) const { assert(N == 1 && "Invalid number of operands!"); Inst.addOperand(MCOperand::createReg(getFGRH32Reg())); } void addFCCAsmRegOperands(MCInst &Inst, unsigned N) const { assert(N == 1 && "Invalid number of operands!"); Inst.addOperand(MCOperand::createReg(getFCCReg())); } void addMSA128AsmRegOperands(MCInst &Inst, unsigned N) const { assert(N == 1 && "Invalid number of operands!"); Inst.addOperand(MCOperand::createReg(getMSA128Reg())); } void addMSACtrlAsmRegOperands(MCInst &Inst, unsigned N) const { assert(N == 1 && "Invalid number of operands!"); Inst.addOperand(MCOperand::createReg(getMSACtrlReg())); } void addCOP0AsmRegOperands(MCInst &Inst, unsigned N) const { assert(N == 1 && "Invalid number of operands!"); Inst.addOperand(MCOperand::createReg(getCOP0Reg())); } void addCOP2AsmRegOperands(MCInst &Inst, unsigned N) const { assert(N == 1 && "Invalid number of operands!"); Inst.addOperand(MCOperand::createReg(getCOP2Reg())); } void addCOP3AsmRegOperands(MCInst &Inst, unsigned N) const { assert(N == 1 && "Invalid number of operands!"); Inst.addOperand(MCOperand::createReg(getCOP3Reg())); } void addACC64DSPAsmRegOperands(MCInst &Inst, unsigned N) const { assert(N == 1 && "Invalid number of operands!"); Inst.addOperand(MCOperand::createReg(getACC64DSPReg())); } void addHI32DSPAsmRegOperands(MCInst &Inst, unsigned N) const { assert(N == 1 && "Invalid number of operands!"); Inst.addOperand(MCOperand::createReg(getHI32DSPReg())); } void addLO32DSPAsmRegOperands(MCInst &Inst, unsigned N) const { assert(N == 1 && "Invalid number of operands!"); Inst.addOperand(MCOperand::createReg(getLO32DSPReg())); } void addCCRAsmRegOperands(MCInst &Inst, unsigned N) const { assert(N == 1 && "Invalid number of operands!"); Inst.addOperand(MCOperand::createReg(getCCRReg())); } void addHWRegsAsmRegOperands(MCInst &Inst, unsigned N) const { assert(N == 1 && "Invalid number of operands!"); Inst.addOperand(MCOperand::createReg(getHWRegsReg())); } template void addConstantUImmOperands(MCInst &Inst, unsigned N) const { assert(N == 1 && "Invalid number of operands!"); uint64_t Imm = getConstantImm() - Offset; Imm &= (1ULL << Bits) - 1; Imm += Offset; Imm += AdjustOffset; Inst.addOperand(MCOperand::createImm(Imm)); } template void addSImmOperands(MCInst &Inst, unsigned N) const { if (isImm() && !isConstantImm()) { addExpr(Inst, getImm()); return; } addConstantSImmOperands(Inst, N); } template void addUImmOperands(MCInst &Inst, unsigned N) const { if (isImm() && !isConstantImm()) { addExpr(Inst, getImm()); return; } addConstantUImmOperands(Inst, N); } template void addConstantSImmOperands(MCInst &Inst, unsigned N) const { assert(N == 1 && "Invalid number of operands!"); int64_t Imm = getConstantImm() - Offset; Imm = SignExtend64(Imm); Imm += Offset; Imm += AdjustOffset; Inst.addOperand(MCOperand::createImm(Imm)); } void addImmOperands(MCInst &Inst, unsigned N) const { assert(N == 1 && "Invalid number of operands!"); const MCExpr *Expr = getImm(); addExpr(Inst, Expr); } void addMemOperands(MCInst &Inst, unsigned N) const { assert(N == 2 && "Invalid number of operands!"); Inst.addOperand(MCOperand::createReg(AsmParser.getABI().ArePtrs64bit() ? getMemBase()->getGPR64Reg() : getMemBase()->getGPR32Reg())); const MCExpr *Expr = getMemOff(); addExpr(Inst, Expr); } void addMicroMipsMemOperands(MCInst &Inst, unsigned N) const { assert(N == 2 && "Invalid number of operands!"); Inst.addOperand(MCOperand::createReg(getMemBase()->getGPRMM16Reg())); const MCExpr *Expr = getMemOff(); addExpr(Inst, Expr); } void addRegListOperands(MCInst &Inst, unsigned N) const { assert(N == 1 && "Invalid number of operands!"); for (auto RegNo : getRegList()) Inst.addOperand(MCOperand::createReg(RegNo)); } bool isReg() const override { // As a special case until we sort out the definition of div/divu, accept // $0/$zero here so that MCK_ZERO works correctly. return isGPRAsmReg() && RegIdx.Index == 0; } bool isRegIdx() const { return Kind == k_RegisterIndex; } bool isImm() const override { return Kind == k_Immediate; } bool isConstantImm() const { int64_t Res; return isImm() && getImm()->evaluateAsAbsolute(Res); } bool isConstantImmz() const { return isConstantImm() && getConstantImm() == 0; } template bool isConstantUImm() const { return isConstantImm() && isUInt(getConstantImm() - Offset); } template bool isSImm() const { return isConstantImm() ? isInt(getConstantImm()) : isImm(); } template bool isUImm() const { return isConstantImm() ? isUInt(getConstantImm()) : isImm(); } template bool isAnyImm() const { return isConstantImm() ? (isInt(getConstantImm()) || isUInt(getConstantImm())) : isImm(); } template bool isConstantSImm() const { return isConstantImm() && isInt(getConstantImm() - Offset); } template bool isConstantUImmRange() const { return isConstantImm() && getConstantImm() >= Bottom && getConstantImm() <= Top; } bool isToken() const override { // Note: It's not possible to pretend that other operand kinds are tokens. // The matcher emitter checks tokens first. return Kind == k_Token; } bool isMem() const override { return Kind == k_Memory; } bool isConstantMemOff() const { return isMem() && isa(getMemOff()); } // Allow relocation operators. // FIXME: This predicate and others need to look through binary expressions // and determine whether a Value is a constant or not. template bool isMemWithSimmOffset() const { if (!isMem()) return false; if (!getMemBase()->isGPRAsmReg()) return false; if (isa(getMemOff()) || (isConstantMemOff() && isShiftedInt(getConstantMemOff()))) return true; MCValue Res; bool IsReloc = getMemOff()->evaluateAsRelocatable(Res, nullptr, nullptr); return IsReloc && isShiftedInt(Res.getConstant()); } bool isMemWithPtrSizeOffset() const { if (!isMem()) return false; if (!getMemBase()->isGPRAsmReg()) return false; const unsigned PtrBits = AsmParser.getABI().ArePtrs64bit() ? 64 : 32; if (isa(getMemOff()) || (isConstantMemOff() && isIntN(PtrBits, getConstantMemOff()))) return true; MCValue Res; bool IsReloc = getMemOff()->evaluateAsRelocatable(Res, nullptr, nullptr); return IsReloc && isIntN(PtrBits, Res.getConstant()); } bool isMemWithGRPMM16Base() const { return isMem() && getMemBase()->isMM16AsmReg(); } template bool isMemWithUimmOffsetSP() const { return isMem() && isConstantMemOff() && isUInt(getConstantMemOff()) && getMemBase()->isRegIdx() && (getMemBase()->getGPR32Reg() == Mips::SP); } template bool isMemWithUimmWordAlignedOffsetSP() const { return isMem() && isConstantMemOff() && isUInt(getConstantMemOff()) && (getConstantMemOff() % 4 == 0) && getMemBase()->isRegIdx() && (getMemBase()->getGPR32Reg() == Mips::SP); } template bool isMemWithSimmWordAlignedOffsetGP() const { return isMem() && isConstantMemOff() && isInt(getConstantMemOff()) && (getConstantMemOff() % 4 == 0) && getMemBase()->isRegIdx() && (getMemBase()->getGPR32Reg() == Mips::GP); } template bool isScaledUImm() const { return isConstantImm() && isShiftedUInt(getConstantImm()); } template bool isScaledSImm() const { if (isConstantImm() && isShiftedInt(getConstantImm())) return true; // Operand can also be a symbol or symbol plus // offset in case of relocations. if (Kind != k_Immediate) return false; MCValue Res; bool Success = getImm()->evaluateAsRelocatable(Res, nullptr, nullptr); return Success && isShiftedInt(Res.getConstant()); } bool isRegList16() const { if (!isRegList()) return false; int Size = RegList.List->size(); if (Size < 2 || Size > 5) return false; unsigned R0 = RegList.List->front(); unsigned R1 = RegList.List->back(); if (!((R0 == Mips::S0 && R1 == Mips::RA) || (R0 == Mips::S0_64 && R1 == Mips::RA_64))) return false; int PrevReg = *RegList.List->begin(); for (int i = 1; i < Size - 1; i++) { int Reg = (*(RegList.List))[i]; if ( Reg != PrevReg + 1) return false; PrevReg = Reg; } return true; } bool isInvNum() const { return Kind == k_Immediate; } bool isLSAImm() const { if (!isConstantImm()) return false; int64_t Val = getConstantImm(); return 1 <= Val && Val <= 4; } bool isRegList() const { return Kind == k_RegList; } StringRef getToken() const { assert(Kind == k_Token && "Invalid access!"); return StringRef(Tok.Data, Tok.Length); } unsigned getReg() const override { // As a special case until we sort out the definition of div/divu, accept // $0/$zero here so that MCK_ZERO works correctly. if (Kind == k_RegisterIndex && RegIdx.Index == 0 && RegIdx.Kind & RegKind_GPR) return getGPR32Reg(); // FIXME: GPR64 too llvm_unreachable("Invalid access!"); return 0; } const MCExpr *getImm() const { assert((Kind == k_Immediate) && "Invalid access!"); return Imm.Val; } int64_t getConstantImm() const { const MCExpr *Val = getImm(); int64_t Value = 0; (void)Val->evaluateAsAbsolute(Value); return Value; } MipsOperand *getMemBase() const { assert((Kind == k_Memory) && "Invalid access!"); return Mem.Base; } const MCExpr *getMemOff() const { assert((Kind == k_Memory) && "Invalid access!"); return Mem.Off; } int64_t getConstantMemOff() const { return static_cast(getMemOff())->getValue(); } const SmallVectorImpl &getRegList() const { assert((Kind == k_RegList) && "Invalid access!"); return *(RegList.List); } static std::unique_ptr CreateToken(StringRef Str, SMLoc S, MipsAsmParser &Parser) { auto Op = llvm::make_unique(k_Token, Parser); Op->Tok.Data = Str.data(); Op->Tok.Length = Str.size(); Op->StartLoc = S; Op->EndLoc = S; return Op; } /// Create a numeric register (e.g. $1). The exact register remains /// unresolved until an instruction successfully matches static std::unique_ptr createNumericReg(unsigned Index, StringRef Str, const MCRegisterInfo *RegInfo, SMLoc S, SMLoc E, MipsAsmParser &Parser) { LLVM_DEBUG(dbgs() << "createNumericReg(" << Index << ", ...)\n"); return CreateReg(Index, Str, RegKind_Numeric, RegInfo, S, E, Parser); } /// Create a register that is definitely a GPR. /// This is typically only used for named registers such as $gp. static std::unique_ptr createGPRReg(unsigned Index, StringRef Str, const MCRegisterInfo *RegInfo, SMLoc S, SMLoc E, MipsAsmParser &Parser) { return CreateReg(Index, Str, RegKind_GPR, RegInfo, S, E, Parser); } /// Create a register that is definitely a FGR. /// This is typically only used for named registers such as $f0. static std::unique_ptr createFGRReg(unsigned Index, StringRef Str, const MCRegisterInfo *RegInfo, SMLoc S, SMLoc E, MipsAsmParser &Parser) { return CreateReg(Index, Str, RegKind_FGR, RegInfo, S, E, Parser); } /// Create a register that is definitely a HWReg. /// This is typically only used for named registers such as $hwr_cpunum. static std::unique_ptr createHWRegsReg(unsigned Index, StringRef Str, const MCRegisterInfo *RegInfo, SMLoc S, SMLoc E, MipsAsmParser &Parser) { return CreateReg(Index, Str, RegKind_HWRegs, RegInfo, S, E, Parser); } /// Create a register that is definitely an FCC. /// This is typically only used for named registers such as $fcc0. static std::unique_ptr createFCCReg(unsigned Index, StringRef Str, const MCRegisterInfo *RegInfo, SMLoc S, SMLoc E, MipsAsmParser &Parser) { return CreateReg(Index, Str, RegKind_FCC, RegInfo, S, E, Parser); } /// Create a register that is definitely an ACC. /// This is typically only used for named registers such as $ac0. static std::unique_ptr createACCReg(unsigned Index, StringRef Str, const MCRegisterInfo *RegInfo, SMLoc S, SMLoc E, MipsAsmParser &Parser) { return CreateReg(Index, Str, RegKind_ACC, RegInfo, S, E, Parser); } /// Create a register that is definitely an MSA128. /// This is typically only used for named registers such as $w0. static std::unique_ptr createMSA128Reg(unsigned Index, StringRef Str, const MCRegisterInfo *RegInfo, SMLoc S, SMLoc E, MipsAsmParser &Parser) { return CreateReg(Index, Str, RegKind_MSA128, RegInfo, S, E, Parser); } /// Create a register that is definitely an MSACtrl. /// This is typically only used for named registers such as $msaaccess. static std::unique_ptr createMSACtrlReg(unsigned Index, StringRef Str, const MCRegisterInfo *RegInfo, SMLoc S, SMLoc E, MipsAsmParser &Parser) { return CreateReg(Index, Str, RegKind_MSACtrl, RegInfo, S, E, Parser); } static std::unique_ptr CreateImm(const MCExpr *Val, SMLoc S, SMLoc E, MipsAsmParser &Parser) { auto Op = llvm::make_unique(k_Immediate, Parser); Op->Imm.Val = Val; Op->StartLoc = S; Op->EndLoc = E; return Op; } static std::unique_ptr CreateMem(std::unique_ptr Base, const MCExpr *Off, SMLoc S, SMLoc E, MipsAsmParser &Parser) { auto Op = llvm::make_unique(k_Memory, Parser); Op->Mem.Base = Base.release(); Op->Mem.Off = Off; Op->StartLoc = S; Op->EndLoc = E; return Op; } static std::unique_ptr CreateRegList(SmallVectorImpl &Regs, SMLoc StartLoc, SMLoc EndLoc, MipsAsmParser &Parser) { assert(Regs.size() > 0 && "Empty list not allowed"); auto Op = llvm::make_unique(k_RegList, Parser); Op->RegList.List = new SmallVector(Regs.begin(), Regs.end()); Op->StartLoc = StartLoc; Op->EndLoc = EndLoc; return Op; } bool isGPRZeroAsmReg() const { return isRegIdx() && RegIdx.Kind & RegKind_GPR && RegIdx.Index == 0; } bool isGPRNonZeroAsmReg() const { return isRegIdx() && RegIdx.Kind & RegKind_GPR && RegIdx.Index > 0 && RegIdx.Index <= 31; } bool isGPRAsmReg() const { return isRegIdx() && RegIdx.Kind & RegKind_GPR && RegIdx.Index <= 31; } bool isMM16AsmReg() const { if (!(isRegIdx() && RegIdx.Kind)) return false; return ((RegIdx.Index >= 2 && RegIdx.Index <= 7) || RegIdx.Index == 16 || RegIdx.Index == 17); } bool isMM16AsmRegZero() const { if (!(isRegIdx() && RegIdx.Kind)) return false; return (RegIdx.Index == 0 || (RegIdx.Index >= 2 && RegIdx.Index <= 7) || RegIdx.Index == 17); } bool isMM16AsmRegMoveP() const { if (!(isRegIdx() && RegIdx.Kind)) return false; return (RegIdx.Index == 0 || (RegIdx.Index >= 2 && RegIdx.Index <= 3) || (RegIdx.Index >= 16 && RegIdx.Index <= 20)); } bool isMM16AsmRegMovePPairFirst() const { if (!(isRegIdx() && RegIdx.Kind)) return false; return RegIdx.Index >= 4 && RegIdx.Index <= 6; } bool isMM16AsmRegMovePPairSecond() const { if (!(isRegIdx() && RegIdx.Kind)) return false; return (RegIdx.Index == 21 || RegIdx.Index == 22 || (RegIdx.Index >= 5 && RegIdx.Index <= 7)); } bool isFGRAsmReg() const { // AFGR64 is $0-$15 but we handle this in getAFGR64() return isRegIdx() && RegIdx.Kind & RegKind_FGR && RegIdx.Index <= 31; } bool isStrictlyFGRAsmReg() const { // AFGR64 is $0-$15 but we handle this in getAFGR64() return isRegIdx() && RegIdx.Kind == RegKind_FGR && RegIdx.Index <= 31; } bool isHWRegsAsmReg() const { return isRegIdx() && RegIdx.Kind & RegKind_HWRegs && RegIdx.Index <= 31; } bool isCCRAsmReg() const { return isRegIdx() && RegIdx.Kind & RegKind_CCR && RegIdx.Index <= 31; } bool isFCCAsmReg() const { if (!(isRegIdx() && RegIdx.Kind & RegKind_FCC)) return false; return RegIdx.Index <= 7; } bool isACCAsmReg() const { return isRegIdx() && RegIdx.Kind & RegKind_ACC && RegIdx.Index <= 3; } bool isCOP0AsmReg() const { return isRegIdx() && RegIdx.Kind & RegKind_COP0 && RegIdx.Index <= 31; } bool isCOP2AsmReg() const { return isRegIdx() && RegIdx.Kind & RegKind_COP2 && RegIdx.Index <= 31; } bool isCOP3AsmReg() const { return isRegIdx() && RegIdx.Kind & RegKind_COP3 && RegIdx.Index <= 31; } bool isMSA128AsmReg() const { return isRegIdx() && RegIdx.Kind & RegKind_MSA128 && RegIdx.Index <= 31; } bool isMSACtrlAsmReg() const { return isRegIdx() && RegIdx.Kind & RegKind_MSACtrl && RegIdx.Index <= 7; } /// getStartLoc - Get the location of the first token of this operand. SMLoc getStartLoc() const override { return StartLoc; } /// getEndLoc - Get the location of the last token of this operand. SMLoc getEndLoc() const override { return EndLoc; } void print(raw_ostream &OS) const override { switch (Kind) { case k_Immediate: OS << "Imm<"; OS << *Imm.Val; OS << ">"; break; case k_Memory: OS << "Mem<"; Mem.Base->print(OS); OS << ", "; OS << *Mem.Off; OS << ">"; break; case k_RegisterIndex: OS << "RegIdx<" << RegIdx.Index << ":" << RegIdx.Kind << ", " << StringRef(RegIdx.Tok.Data, RegIdx.Tok.Length) << ">"; break; case k_Token: OS << getToken(); break; case k_RegList: OS << "RegList< "; for (auto Reg : (*RegList.List)) OS << Reg << " "; OS << ">"; break; } } bool isValidForTie(const MipsOperand &Other) const { if (Kind != Other.Kind) return false; switch (Kind) { default: llvm_unreachable("Unexpected kind"); return false; case k_RegisterIndex: { StringRef Token(RegIdx.Tok.Data, RegIdx.Tok.Length); StringRef OtherToken(Other.RegIdx.Tok.Data, Other.RegIdx.Tok.Length); return Token == OtherToken; } } } }; // class MipsOperand } // end anonymous namespace namespace llvm { extern const MCInstrDesc MipsInsts[]; } // end namespace llvm static const MCInstrDesc &getInstDesc(unsigned Opcode) { return MipsInsts[Opcode]; } static bool hasShortDelaySlot(MCInst &Inst) { switch (Inst.getOpcode()) { case Mips::BEQ_MM: case Mips::BNE_MM: case Mips::BLTZ_MM: case Mips::BGEZ_MM: case Mips::BLEZ_MM: case Mips::BGTZ_MM: case Mips::JRC16_MM: case Mips::JALS_MM: case Mips::JALRS_MM: case Mips::JALRS16_MM: case Mips::BGEZALS_MM: case Mips::BLTZALS_MM: return true; case Mips::J_MM: return !Inst.getOperand(0).isReg(); default: return false; } } static const MCSymbol *getSingleMCSymbol(const MCExpr *Expr) { if (const MCSymbolRefExpr *SRExpr = dyn_cast(Expr)) { return &SRExpr->getSymbol(); } if (const MCBinaryExpr *BExpr = dyn_cast(Expr)) { const MCSymbol *LHSSym = getSingleMCSymbol(BExpr->getLHS()); const MCSymbol *RHSSym = getSingleMCSymbol(BExpr->getRHS()); if (LHSSym) return LHSSym; if (RHSSym) return RHSSym; return nullptr; } if (const MCUnaryExpr *UExpr = dyn_cast(Expr)) return getSingleMCSymbol(UExpr->getSubExpr()); return nullptr; } static unsigned countMCSymbolRefExpr(const MCExpr *Expr) { if (isa(Expr)) return 1; if (const MCBinaryExpr *BExpr = dyn_cast(Expr)) return countMCSymbolRefExpr(BExpr->getLHS()) + countMCSymbolRefExpr(BExpr->getRHS()); if (const MCUnaryExpr *UExpr = dyn_cast(Expr)) return countMCSymbolRefExpr(UExpr->getSubExpr()); return 0; } bool MipsAsmParser::processInstruction(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI) { MipsTargetStreamer &TOut = getTargetStreamer(); const MCInstrDesc &MCID = getInstDesc(Inst.getOpcode()); bool ExpandedJalSym = false; Inst.setLoc(IDLoc); if (MCID.isBranch() || MCID.isCall()) { const unsigned Opcode = Inst.getOpcode(); MCOperand Offset; switch (Opcode) { default: break; case Mips::BBIT0: case Mips::BBIT032: case Mips::BBIT1: case Mips::BBIT132: assert(hasCnMips() && "instruction only valid for octeon cpus"); LLVM_FALLTHROUGH; case Mips::BEQ: case Mips::BNE: case Mips::BEQ_MM: case Mips::BNE_MM: assert(MCID.getNumOperands() == 3 && "unexpected number of operands"); Offset = Inst.getOperand(2); if (!Offset.isImm()) break; // We'll deal with this situation later on when applying fixups. if (!isIntN(inMicroMipsMode() ? 17 : 18, Offset.getImm())) return Error(IDLoc, "branch target out of range"); if (OffsetToAlignment(Offset.getImm(), 1LL << (inMicroMipsMode() ? 1 : 2))) return Error(IDLoc, "branch to misaligned address"); break; case Mips::BGEZ: case Mips::BGTZ: case Mips::BLEZ: case Mips::BLTZ: case Mips::BGEZAL: case Mips::BLTZAL: case Mips::BC1F: case Mips::BC1T: case Mips::BGEZ_MM: case Mips::BGTZ_MM: case Mips::BLEZ_MM: case Mips::BLTZ_MM: case Mips::BGEZAL_MM: case Mips::BLTZAL_MM: case Mips::BC1F_MM: case Mips::BC1T_MM: case Mips::BC1EQZC_MMR6: case Mips::BC1NEZC_MMR6: case Mips::BC2EQZC_MMR6: case Mips::BC2NEZC_MMR6: assert(MCID.getNumOperands() == 2 && "unexpected number of operands"); Offset = Inst.getOperand(1); if (!Offset.isImm()) break; // We'll deal with this situation later on when applying fixups. if (!isIntN(inMicroMipsMode() ? 17 : 18, Offset.getImm())) return Error(IDLoc, "branch target out of range"); if (OffsetToAlignment(Offset.getImm(), 1LL << (inMicroMipsMode() ? 1 : 2))) return Error(IDLoc, "branch to misaligned address"); break; case Mips::BGEC: case Mips::BGEC_MMR6: case Mips::BLTC: case Mips::BLTC_MMR6: case Mips::BGEUC: case Mips::BGEUC_MMR6: case Mips::BLTUC: case Mips::BLTUC_MMR6: case Mips::BEQC: case Mips::BEQC_MMR6: case Mips::BNEC: case Mips::BNEC_MMR6: assert(MCID.getNumOperands() == 3 && "unexpected number of operands"); Offset = Inst.getOperand(2); if (!Offset.isImm()) break; // We'll deal with this situation later on when applying fixups. if (!isIntN(18, Offset.getImm())) return Error(IDLoc, "branch target out of range"); if (OffsetToAlignment(Offset.getImm(), 1LL << 2)) return Error(IDLoc, "branch to misaligned address"); break; case Mips::BLEZC: case Mips::BLEZC_MMR6: case Mips::BGEZC: case Mips::BGEZC_MMR6: case Mips::BGTZC: case Mips::BGTZC_MMR6: case Mips::BLTZC: case Mips::BLTZC_MMR6: assert(MCID.getNumOperands() == 2 && "unexpected number of operands"); Offset = Inst.getOperand(1); if (!Offset.isImm()) break; // We'll deal with this situation later on when applying fixups. if (!isIntN(18, Offset.getImm())) return Error(IDLoc, "branch target out of range"); if (OffsetToAlignment(Offset.getImm(), 1LL << 2)) return Error(IDLoc, "branch to misaligned address"); break; case Mips::BEQZC: case Mips::BEQZC_MMR6: case Mips::BNEZC: case Mips::BNEZC_MMR6: assert(MCID.getNumOperands() == 2 && "unexpected number of operands"); Offset = Inst.getOperand(1); if (!Offset.isImm()) break; // We'll deal with this situation later on when applying fixups. if (!isIntN(23, Offset.getImm())) return Error(IDLoc, "branch target out of range"); if (OffsetToAlignment(Offset.getImm(), 1LL << 2)) return Error(IDLoc, "branch to misaligned address"); break; case Mips::BEQZ16_MM: case Mips::BEQZC16_MMR6: case Mips::BNEZ16_MM: case Mips::BNEZC16_MMR6: assert(MCID.getNumOperands() == 2 && "unexpected number of operands"); Offset = Inst.getOperand(1); if (!Offset.isImm()) break; // We'll deal with this situation later on when applying fixups. if (!isInt<8>(Offset.getImm())) return Error(IDLoc, "branch target out of range"); if (OffsetToAlignment(Offset.getImm(), 2LL)) return Error(IDLoc, "branch to misaligned address"); break; } } // SSNOP is deprecated on MIPS32r6/MIPS64r6 // We still accept it but it is a normal nop. if (hasMips32r6() && Inst.getOpcode() == Mips::SSNOP) { std::string ISA = hasMips64r6() ? "MIPS64r6" : "MIPS32r6"; Warning(IDLoc, "ssnop is deprecated for " + ISA + " and is equivalent to a " "nop instruction"); } if (hasCnMips()) { const unsigned Opcode = Inst.getOpcode(); MCOperand Opnd; int Imm; switch (Opcode) { default: break; case Mips::BBIT0: case Mips::BBIT032: case Mips::BBIT1: case Mips::BBIT132: assert(MCID.getNumOperands() == 3 && "unexpected number of operands"); // The offset is handled above Opnd = Inst.getOperand(1); if (!Opnd.isImm()) return Error(IDLoc, "expected immediate operand kind"); Imm = Opnd.getImm(); if (Imm < 0 || Imm > (Opcode == Mips::BBIT0 || Opcode == Mips::BBIT1 ? 63 : 31)) return Error(IDLoc, "immediate operand value out of range"); if (Imm > 31) { Inst.setOpcode(Opcode == Mips::BBIT0 ? Mips::BBIT032 : Mips::BBIT132); Inst.getOperand(1).setImm(Imm - 32); } break; case Mips::SEQi: case Mips::SNEi: assert(MCID.getNumOperands() == 3 && "unexpected number of operands"); Opnd = Inst.getOperand(2); if (!Opnd.isImm()) return Error(IDLoc, "expected immediate operand kind"); Imm = Opnd.getImm(); if (!isInt<10>(Imm)) return Error(IDLoc, "immediate operand value out of range"); break; } } // Warn on division by zero. We're checking here as all instructions get // processed here, not just the macros that need expansion. // // The MIPS backend models most of the divison instructions and macros as // three operand instructions. The pre-R6 divide instructions however have // two operands and explicitly define HI/LO as part of the instruction, // not in the operands. unsigned FirstOp = 1; unsigned SecondOp = 2; switch (Inst.getOpcode()) { default: break; case Mips::SDivIMacro: case Mips::UDivIMacro: case Mips::DSDivIMacro: case Mips::DUDivIMacro: if (Inst.getOperand(2).getImm() == 0) { if (Inst.getOperand(1).getReg() == Mips::ZERO || Inst.getOperand(1).getReg() == Mips::ZERO_64) Warning(IDLoc, "dividing zero by zero"); else Warning(IDLoc, "division by zero"); } break; case Mips::DSDIV: case Mips::SDIV: case Mips::UDIV: case Mips::DUDIV: case Mips::UDIV_MM: case Mips::SDIV_MM: FirstOp = 0; SecondOp = 1; LLVM_FALLTHROUGH; case Mips::SDivMacro: case Mips::DSDivMacro: case Mips::UDivMacro: case Mips::DUDivMacro: case Mips::DIV: case Mips::DIVU: case Mips::DDIV: case Mips::DDIVU: case Mips::DIVU_MMR6: case Mips::DIV_MMR6: if (Inst.getOperand(SecondOp).getReg() == Mips::ZERO || Inst.getOperand(SecondOp).getReg() == Mips::ZERO_64) { if (Inst.getOperand(FirstOp).getReg() == Mips::ZERO || Inst.getOperand(FirstOp).getReg() == Mips::ZERO_64) Warning(IDLoc, "dividing zero by zero"); else Warning(IDLoc, "division by zero"); } break; } // For PIC code convert unconditional jump to unconditional branch. if ((Inst.getOpcode() == Mips::J || Inst.getOpcode() == Mips::J_MM) && inPicMode()) { MCInst BInst; BInst.setOpcode(inMicroMipsMode() ? Mips::BEQ_MM : Mips::BEQ); BInst.addOperand(MCOperand::createReg(Mips::ZERO)); BInst.addOperand(MCOperand::createReg(Mips::ZERO)); BInst.addOperand(Inst.getOperand(0)); Inst = BInst; } // This expansion is not in a function called by tryExpandInstruction() // because the pseudo-instruction doesn't have a distinct opcode. if ((Inst.getOpcode() == Mips::JAL || Inst.getOpcode() == Mips::JAL_MM) && inPicMode()) { warnIfNoMacro(IDLoc); const MCExpr *JalExpr = Inst.getOperand(0).getExpr(); // We can do this expansion if there's only 1 symbol in the argument // expression. if (countMCSymbolRefExpr(JalExpr) > 1) return Error(IDLoc, "jal doesn't support multiple symbols in PIC mode"); // FIXME: This is checking the expression can be handled by the later stages // of the assembler. We ought to leave it to those later stages. const MCSymbol *JalSym = getSingleMCSymbol(JalExpr); // FIXME: Add support for label+offset operands (currently causes an error). // FIXME: Add support for forward-declared local symbols. // FIXME: Add expansion for when the LargeGOT option is enabled. if (JalSym->isInSection() || JalSym->isTemporary() || (JalSym->isELF() && cast(JalSym)->getBinding() == ELF::STB_LOCAL)) { if (isABI_O32()) { // If it's a local symbol and the O32 ABI is being used, we expand to: // lw $25, 0($gp) // R_(MICRO)MIPS_GOT16 label // addiu $25, $25, 0 // R_(MICRO)MIPS_LO16 label // jalr $25 const MCExpr *Got16RelocExpr = MipsMCExpr::create(MipsMCExpr::MEK_GOT, JalExpr, getContext()); const MCExpr *Lo16RelocExpr = MipsMCExpr::create(MipsMCExpr::MEK_LO, JalExpr, getContext()); TOut.emitRRX(Mips::LW, Mips::T9, Mips::GP, MCOperand::createExpr(Got16RelocExpr), IDLoc, STI); TOut.emitRRX(Mips::ADDiu, Mips::T9, Mips::T9, MCOperand::createExpr(Lo16RelocExpr), IDLoc, STI); } else if (isABI_N32() || isABI_N64()) { // If it's a local symbol and the N32/N64 ABIs are being used, // we expand to: // lw/ld $25, 0($gp) // R_(MICRO)MIPS_GOT_DISP label // jalr $25 const MCExpr *GotDispRelocExpr = MipsMCExpr::create(MipsMCExpr::MEK_GOT_DISP, JalExpr, getContext()); TOut.emitRRX(ABI.ArePtrs64bit() ? Mips::LD : Mips::LW, Mips::T9, Mips::GP, MCOperand::createExpr(GotDispRelocExpr), IDLoc, STI); } } else { // If it's an external/weak symbol, we expand to: // lw/ld $25, 0($gp) // R_(MICRO)MIPS_CALL16 label // jalr $25 const MCExpr *Call16RelocExpr = MipsMCExpr::create(MipsMCExpr::MEK_GOT_CALL, JalExpr, getContext()); TOut.emitRRX(ABI.ArePtrs64bit() ? Mips::LD : Mips::LW, Mips::T9, Mips::GP, MCOperand::createExpr(Call16RelocExpr), IDLoc, STI); } MCInst JalrInst; if (IsCpRestoreSet && inMicroMipsMode()) JalrInst.setOpcode(Mips::JALRS_MM); else JalrInst.setOpcode(inMicroMipsMode() ? Mips::JALR_MM : Mips::JALR); JalrInst.addOperand(MCOperand::createReg(Mips::RA)); JalrInst.addOperand(MCOperand::createReg(Mips::T9)); if (EmitJalrReloc) { // As an optimization hint for the linker, before the JALR we add: // .reloc tmplabel, R_{MICRO}MIPS_JALR, symbol // tmplabel: MCSymbol *TmpLabel = getContext().createTempSymbol(); const MCExpr *TmpExpr = MCSymbolRefExpr::create(TmpLabel, getContext()); const MCExpr *RelocJalrExpr = MCSymbolRefExpr::create(JalSym, MCSymbolRefExpr::VK_None, getContext(), IDLoc); TOut.getStreamer().EmitRelocDirective(*TmpExpr, inMicroMipsMode() ? "R_MICROMIPS_JALR" : "R_MIPS_JALR", RelocJalrExpr, IDLoc, *STI); TOut.getStreamer().EmitLabel(TmpLabel); } Inst = JalrInst; ExpandedJalSym = true; } bool IsPCRelativeLoad = (MCID.TSFlags & MipsII::IsPCRelativeLoad) != 0; if ((MCID.mayLoad() || MCID.mayStore()) && !IsPCRelativeLoad) { // Check the offset of memory operand, if it is a symbol // reference or immediate we may have to expand instructions. for (unsigned i = 0; i < MCID.getNumOperands(); i++) { const MCOperandInfo &OpInfo = MCID.OpInfo[i]; if ((OpInfo.OperandType == MCOI::OPERAND_MEMORY) || (OpInfo.OperandType == MCOI::OPERAND_UNKNOWN)) { MCOperand &Op = Inst.getOperand(i); if (Op.isImm()) { int64_t MemOffset = Op.getImm(); if (MemOffset < -32768 || MemOffset > 32767) { // Offset can't exceed 16bit value. expandMemInst(Inst, IDLoc, Out, STI, MCID.mayLoad()); return getParser().hasPendingError(); } } else if (Op.isExpr()) { const MCExpr *Expr = Op.getExpr(); if (Expr->getKind() == MCExpr::SymbolRef) { const MCSymbolRefExpr *SR = static_cast(Expr); if (SR->getKind() == MCSymbolRefExpr::VK_None) { // Expand symbol. expandMemInst(Inst, IDLoc, Out, STI, MCID.mayLoad()); return getParser().hasPendingError(); } } else if (!isEvaluated(Expr)) { expandMemInst(Inst, IDLoc, Out, STI, MCID.mayLoad()); return getParser().hasPendingError(); } } } } // for } // if load/store if (inMicroMipsMode()) { if (MCID.mayLoad() && Inst.getOpcode() != Mips::LWP_MM) { // Try to create 16-bit GP relative load instruction. for (unsigned i = 0; i < MCID.getNumOperands(); i++) { const MCOperandInfo &OpInfo = MCID.OpInfo[i]; if ((OpInfo.OperandType == MCOI::OPERAND_MEMORY) || (OpInfo.OperandType == MCOI::OPERAND_UNKNOWN)) { MCOperand &Op = Inst.getOperand(i); if (Op.isImm()) { int MemOffset = Op.getImm(); MCOperand &DstReg = Inst.getOperand(0); MCOperand &BaseReg = Inst.getOperand(1); if (isInt<9>(MemOffset) && (MemOffset % 4 == 0) && getContext().getRegisterInfo()->getRegClass( Mips::GPRMM16RegClassID).contains(DstReg.getReg()) && (BaseReg.getReg() == Mips::GP || BaseReg.getReg() == Mips::GP_64)) { TOut.emitRRI(Mips::LWGP_MM, DstReg.getReg(), Mips::GP, MemOffset, IDLoc, STI); return false; } } } } // for } // if load // TODO: Handle this with the AsmOperandClass.PredicateMethod. MCOperand Opnd; int Imm; switch (Inst.getOpcode()) { default: break; case Mips::ADDIUSP_MM: Opnd = Inst.getOperand(0); if (!Opnd.isImm()) return Error(IDLoc, "expected immediate operand kind"); Imm = Opnd.getImm(); if (Imm < -1032 || Imm > 1028 || (Imm < 8 && Imm > -12) || Imm % 4 != 0) return Error(IDLoc, "immediate operand value out of range"); break; case Mips::SLL16_MM: case Mips::SRL16_MM: Opnd = Inst.getOperand(2); if (!Opnd.isImm()) return Error(IDLoc, "expected immediate operand kind"); Imm = Opnd.getImm(); if (Imm < 1 || Imm > 8) return Error(IDLoc, "immediate operand value out of range"); break; case Mips::LI16_MM: Opnd = Inst.getOperand(1); if (!Opnd.isImm()) return Error(IDLoc, "expected immediate operand kind"); Imm = Opnd.getImm(); if (Imm < -1 || Imm > 126) return Error(IDLoc, "immediate operand value out of range"); break; case Mips::ADDIUR2_MM: Opnd = Inst.getOperand(2); if (!Opnd.isImm()) return Error(IDLoc, "expected immediate operand kind"); Imm = Opnd.getImm(); if (!(Imm == 1 || Imm == -1 || ((Imm % 4 == 0) && Imm < 28 && Imm > 0))) return Error(IDLoc, "immediate operand value out of range"); break; case Mips::ANDI16_MM: Opnd = Inst.getOperand(2); if (!Opnd.isImm()) return Error(IDLoc, "expected immediate operand kind"); Imm = Opnd.getImm(); if (!(Imm == 128 || (Imm >= 1 && Imm <= 4) || Imm == 7 || Imm == 8 || Imm == 15 || Imm == 16 || Imm == 31 || Imm == 32 || Imm == 63 || Imm == 64 || Imm == 255 || Imm == 32768 || Imm == 65535)) return Error(IDLoc, "immediate operand value out of range"); break; case Mips::LBU16_MM: Opnd = Inst.getOperand(2); if (!Opnd.isImm()) return Error(IDLoc, "expected immediate operand kind"); Imm = Opnd.getImm(); if (Imm < -1 || Imm > 14) return Error(IDLoc, "immediate operand value out of range"); break; case Mips::SB16_MM: case Mips::SB16_MMR6: Opnd = Inst.getOperand(2); if (!Opnd.isImm()) return Error(IDLoc, "expected immediate operand kind"); Imm = Opnd.getImm(); if (Imm < 0 || Imm > 15) return Error(IDLoc, "immediate operand value out of range"); break; case Mips::LHU16_MM: case Mips::SH16_MM: case Mips::SH16_MMR6: Opnd = Inst.getOperand(2); if (!Opnd.isImm()) return Error(IDLoc, "expected immediate operand kind"); Imm = Opnd.getImm(); if (Imm < 0 || Imm > 30 || (Imm % 2 != 0)) return Error(IDLoc, "immediate operand value out of range"); break; case Mips::LW16_MM: case Mips::SW16_MM: case Mips::SW16_MMR6: Opnd = Inst.getOperand(2); if (!Opnd.isImm()) return Error(IDLoc, "expected immediate operand kind"); Imm = Opnd.getImm(); if (Imm < 0 || Imm > 60 || (Imm % 4 != 0)) return Error(IDLoc, "immediate operand value out of range"); break; case Mips::ADDIUPC_MM: Opnd = Inst.getOperand(1); if (!Opnd.isImm()) return Error(IDLoc, "expected immediate operand kind"); Imm = Opnd.getImm(); if ((Imm % 4 != 0) || !isInt<25>(Imm)) return Error(IDLoc, "immediate operand value out of range"); break; case Mips::LWP_MM: case Mips::SWP_MM: if (Inst.getOperand(0).getReg() == Mips::RA) return Error(IDLoc, "invalid operand for instruction"); break; case Mips::MOVEP_MM: case Mips::MOVEP_MMR6: { unsigned R0 = Inst.getOperand(0).getReg(); unsigned R1 = Inst.getOperand(1).getReg(); bool RegPair = ((R0 == Mips::A1 && R1 == Mips::A2) || (R0 == Mips::A1 && R1 == Mips::A3) || (R0 == Mips::A2 && R1 == Mips::A3) || (R0 == Mips::A0 && R1 == Mips::S5) || (R0 == Mips::A0 && R1 == Mips::S6) || (R0 == Mips::A0 && R1 == Mips::A1) || (R0 == Mips::A0 && R1 == Mips::A2) || (R0 == Mips::A0 && R1 == Mips::A3)); if (!RegPair) return Error(IDLoc, "invalid operand for instruction"); break; } } } bool FillDelaySlot = MCID.hasDelaySlot() && AssemblerOptions.back()->isReorder(); if (FillDelaySlot) TOut.emitDirectiveSetNoReorder(); MacroExpanderResultTy ExpandResult = tryExpandInstruction(Inst, IDLoc, Out, STI); switch (ExpandResult) { case MER_NotAMacro: Out.EmitInstruction(Inst, *STI); break; case MER_Success: break; case MER_Fail: return true; } // We know we emitted an instruction on the MER_NotAMacro or MER_Success path. // If we're in microMIPS mode then we must also set EF_MIPS_MICROMIPS. if (inMicroMipsMode()) { TOut.setUsesMicroMips(); TOut.updateABIInfo(*this); } // If this instruction has a delay slot and .set reorder is active, // emit a NOP after it. if (FillDelaySlot) { TOut.emitEmptyDelaySlot(hasShortDelaySlot(Inst), IDLoc, STI); TOut.emitDirectiveSetReorder(); } if ((Inst.getOpcode() == Mips::JalOneReg || Inst.getOpcode() == Mips::JalTwoReg || ExpandedJalSym) && isPicAndNotNxxAbi()) { if (IsCpRestoreSet) { // We need a NOP between the JALR and the LW: // If .set reorder has been used, we've already emitted a NOP. // If .set noreorder has been used, we need to emit a NOP at this point. if (!AssemblerOptions.back()->isReorder()) TOut.emitEmptyDelaySlot(hasShortDelaySlot(Inst), IDLoc, STI); // Load the $gp from the stack. TOut.emitGPRestore(CpRestoreOffset, IDLoc, STI); } else Warning(IDLoc, "no .cprestore used in PIC mode"); } return false; } MipsAsmParser::MacroExpanderResultTy MipsAsmParser::tryExpandInstruction(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI) { switch (Inst.getOpcode()) { default: return MER_NotAMacro; case Mips::LoadImm32: return expandLoadImm(Inst, true, IDLoc, Out, STI) ? MER_Fail : MER_Success; case Mips::LoadImm64: return expandLoadImm(Inst, false, IDLoc, Out, STI) ? MER_Fail : MER_Success; case Mips::LoadAddrImm32: case Mips::LoadAddrImm64: assert(Inst.getOperand(0).isReg() && "expected register operand kind"); assert((Inst.getOperand(1).isImm() || Inst.getOperand(1).isExpr()) && "expected immediate operand kind"); return expandLoadAddress(Inst.getOperand(0).getReg(), Mips::NoRegister, Inst.getOperand(1), Inst.getOpcode() == Mips::LoadAddrImm32, IDLoc, Out, STI) ? MER_Fail : MER_Success; case Mips::LoadAddrReg32: case Mips::LoadAddrReg64: assert(Inst.getOperand(0).isReg() && "expected register operand kind"); assert(Inst.getOperand(1).isReg() && "expected register operand kind"); assert((Inst.getOperand(2).isImm() || Inst.getOperand(2).isExpr()) && "expected immediate operand kind"); return expandLoadAddress(Inst.getOperand(0).getReg(), Inst.getOperand(1).getReg(), Inst.getOperand(2), Inst.getOpcode() == Mips::LoadAddrReg32, IDLoc, Out, STI) ? MER_Fail : MER_Success; case Mips::B_MM_Pseudo: case Mips::B_MMR6_Pseudo: return expandUncondBranchMMPseudo(Inst, IDLoc, Out, STI) ? MER_Fail : MER_Success; case Mips::SWM_MM: case Mips::LWM_MM: return expandLoadStoreMultiple(Inst, IDLoc, Out, STI) ? MER_Fail : MER_Success; case Mips::JalOneReg: case Mips::JalTwoReg: return expandJalWithRegs(Inst, IDLoc, Out, STI) ? MER_Fail : MER_Success; case Mips::BneImm: case Mips::BeqImm: case Mips::BEQLImmMacro: case Mips::BNELImmMacro: return expandBranchImm(Inst, IDLoc, Out, STI) ? MER_Fail : MER_Success; case Mips::BLT: case Mips::BLE: case Mips::BGE: case Mips::BGT: case Mips::BLTU: case Mips::BLEU: case Mips::BGEU: case Mips::BGTU: case Mips::BLTL: case Mips::BLEL: case Mips::BGEL: case Mips::BGTL: case Mips::BLTUL: case Mips::BLEUL: case Mips::BGEUL: case Mips::BGTUL: case Mips::BLTImmMacro: case Mips::BLEImmMacro: case Mips::BGEImmMacro: case Mips::BGTImmMacro: case Mips::BLTUImmMacro: case Mips::BLEUImmMacro: case Mips::BGEUImmMacro: case Mips::BGTUImmMacro: case Mips::BLTLImmMacro: case Mips::BLELImmMacro: case Mips::BGELImmMacro: case Mips::BGTLImmMacro: case Mips::BLTULImmMacro: case Mips::BLEULImmMacro: case Mips::BGEULImmMacro: case Mips::BGTULImmMacro: return expandCondBranches(Inst, IDLoc, Out, STI) ? MER_Fail : MER_Success; case Mips::SDivMacro: case Mips::SDivIMacro: case Mips::SRemMacro: case Mips::SRemIMacro: return expandDivRem(Inst, IDLoc, Out, STI, false, true) ? MER_Fail : MER_Success; case Mips::DSDivMacro: case Mips::DSDivIMacro: case Mips::DSRemMacro: case Mips::DSRemIMacro: return expandDivRem(Inst, IDLoc, Out, STI, true, true) ? MER_Fail : MER_Success; case Mips::UDivMacro: case Mips::UDivIMacro: case Mips::URemMacro: case Mips::URemIMacro: return expandDivRem(Inst, IDLoc, Out, STI, false, false) ? MER_Fail : MER_Success; case Mips::DUDivMacro: case Mips::DUDivIMacro: case Mips::DURemMacro: case Mips::DURemIMacro: return expandDivRem(Inst, IDLoc, Out, STI, true, false) ? MER_Fail : MER_Success; case Mips::PseudoTRUNC_W_S: return expandTrunc(Inst, false, false, IDLoc, Out, STI) ? MER_Fail : MER_Success; case Mips::PseudoTRUNC_W_D32: return expandTrunc(Inst, true, false, IDLoc, Out, STI) ? MER_Fail : MER_Success; case Mips::PseudoTRUNC_W_D: return expandTrunc(Inst, true, true, IDLoc, Out, STI) ? MER_Fail : MER_Success; case Mips::LoadImmSingleGPR: return expandLoadImmReal(Inst, true, true, false, IDLoc, Out, STI) ? MER_Fail : MER_Success; case Mips::LoadImmSingleFGR: return expandLoadImmReal(Inst, true, false, false, IDLoc, Out, STI) ? MER_Fail : MER_Success; case Mips::LoadImmDoubleGPR: return expandLoadImmReal(Inst, false, true, false, IDLoc, Out, STI) ? MER_Fail : MER_Success; case Mips::LoadImmDoubleFGR: return expandLoadImmReal(Inst, false, false, true, IDLoc, Out, STI) ? MER_Fail : MER_Success; case Mips::LoadImmDoubleFGR_32: return expandLoadImmReal(Inst, false, false, false, IDLoc, Out, STI) ? MER_Fail : MER_Success; case Mips::Ulh: return expandUlh(Inst, true, IDLoc, Out, STI) ? MER_Fail : MER_Success; case Mips::Ulhu: return expandUlh(Inst, false, IDLoc, Out, STI) ? MER_Fail : MER_Success; case Mips::Ush: return expandUsh(Inst, IDLoc, Out, STI) ? MER_Fail : MER_Success; case Mips::Ulw: case Mips::Usw: return expandUxw(Inst, IDLoc, Out, STI) ? MER_Fail : MER_Success; case Mips::NORImm: case Mips::NORImm64: return expandAliasImmediate(Inst, IDLoc, Out, STI) ? MER_Fail : MER_Success; case Mips::SLTImm64: if (isInt<16>(Inst.getOperand(2).getImm())) { Inst.setOpcode(Mips::SLTi64); return MER_NotAMacro; } return expandAliasImmediate(Inst, IDLoc, Out, STI) ? MER_Fail : MER_Success; case Mips::SLTUImm64: if (isInt<16>(Inst.getOperand(2).getImm())) { Inst.setOpcode(Mips::SLTiu64); return MER_NotAMacro; } return expandAliasImmediate(Inst, IDLoc, Out, STI) ? MER_Fail : MER_Success; case Mips::ADDi: case Mips::ADDi_MM: case Mips::ADDiu: case Mips::ADDiu_MM: case Mips::SLTi: case Mips::SLTi_MM: case Mips::SLTiu: case Mips::SLTiu_MM: if ((Inst.getNumOperands() == 3) && Inst.getOperand(0).isReg() && Inst.getOperand(1).isReg() && Inst.getOperand(2).isImm()) { int64_t ImmValue = Inst.getOperand(2).getImm(); if (isInt<16>(ImmValue)) return MER_NotAMacro; return expandAliasImmediate(Inst, IDLoc, Out, STI) ? MER_Fail : MER_Success; } return MER_NotAMacro; case Mips::ANDi: case Mips::ANDi_MM: case Mips::ANDi64: case Mips::ORi: case Mips::ORi_MM: case Mips::ORi64: case Mips::XORi: case Mips::XORi_MM: case Mips::XORi64: if ((Inst.getNumOperands() == 3) && Inst.getOperand(0).isReg() && Inst.getOperand(1).isReg() && Inst.getOperand(2).isImm()) { int64_t ImmValue = Inst.getOperand(2).getImm(); if (isUInt<16>(ImmValue)) return MER_NotAMacro; return expandAliasImmediate(Inst, IDLoc, Out, STI) ? MER_Fail : MER_Success; } return MER_NotAMacro; case Mips::ROL: case Mips::ROR: return expandRotation(Inst, IDLoc, Out, STI) ? MER_Fail : MER_Success; case Mips::ROLImm: case Mips::RORImm: return expandRotationImm(Inst, IDLoc, Out, STI) ? MER_Fail : MER_Success; case Mips::DROL: case Mips::DROR: return expandDRotation(Inst, IDLoc, Out, STI) ? MER_Fail : MER_Success; case Mips::DROLImm: case Mips::DRORImm: return expandDRotationImm(Inst, IDLoc, Out, STI) ? MER_Fail : MER_Success; case Mips::ABSMacro: return expandAbs(Inst, IDLoc, Out, STI) ? MER_Fail : MER_Success; case Mips::MULImmMacro: case Mips::DMULImmMacro: return expandMulImm(Inst, IDLoc, Out, STI) ? MER_Fail : MER_Success; case Mips::MULOMacro: case Mips::DMULOMacro: return expandMulO(Inst, IDLoc, Out, STI) ? MER_Fail : MER_Success; case Mips::MULOUMacro: case Mips::DMULOUMacro: return expandMulOU(Inst, IDLoc, Out, STI) ? MER_Fail : MER_Success; case Mips::DMULMacro: return expandDMULMacro(Inst, IDLoc, Out, STI) ? MER_Fail : MER_Success; case Mips::LDMacro: case Mips::SDMacro: return expandLoadStoreDMacro(Inst, IDLoc, Out, STI, Inst.getOpcode() == Mips::LDMacro) ? MER_Fail : MER_Success; case Mips::SEQMacro: return expandSeq(Inst, IDLoc, Out, STI) ? MER_Fail : MER_Success; case Mips::SEQIMacro: return expandSeqI(Inst, IDLoc, Out, STI) ? MER_Fail : MER_Success; case Mips::MFTC0: case Mips::MTTC0: case Mips::MFTGPR: case Mips::MTTGPR: case Mips::MFTLO: case Mips::MTTLO: case Mips::MFTHI: case Mips::MTTHI: case Mips::MFTACX: case Mips::MTTACX: case Mips::MFTDSP: case Mips::MTTDSP: case Mips::MFTC1: case Mips::MTTC1: case Mips::MFTHC1: case Mips::MTTHC1: case Mips::CFTC1: case Mips::CTTC1: return expandMXTRAlias(Inst, IDLoc, Out, STI) ? MER_Fail : MER_Success; } } bool MipsAsmParser::expandJalWithRegs(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI) { MipsTargetStreamer &TOut = getTargetStreamer(); // Create a JALR instruction which is going to replace the pseudo-JAL. MCInst JalrInst; JalrInst.setLoc(IDLoc); const MCOperand FirstRegOp = Inst.getOperand(0); const unsigned Opcode = Inst.getOpcode(); if (Opcode == Mips::JalOneReg) { // jal $rs => jalr $rs if (IsCpRestoreSet && inMicroMipsMode()) { JalrInst.setOpcode(Mips::JALRS16_MM); JalrInst.addOperand(FirstRegOp); } else if (inMicroMipsMode()) { JalrInst.setOpcode(hasMips32r6() ? Mips::JALRC16_MMR6 : Mips::JALR16_MM); JalrInst.addOperand(FirstRegOp); } else { JalrInst.setOpcode(Mips::JALR); JalrInst.addOperand(MCOperand::createReg(Mips::RA)); JalrInst.addOperand(FirstRegOp); } } else if (Opcode == Mips::JalTwoReg) { // jal $rd, $rs => jalr $rd, $rs if (IsCpRestoreSet && inMicroMipsMode()) JalrInst.setOpcode(Mips::JALRS_MM); else JalrInst.setOpcode(inMicroMipsMode() ? Mips::JALR_MM : Mips::JALR); JalrInst.addOperand(FirstRegOp); const MCOperand SecondRegOp = Inst.getOperand(1); JalrInst.addOperand(SecondRegOp); } Out.EmitInstruction(JalrInst, *STI); // If .set reorder is active and branch instruction has a delay slot, // emit a NOP after it. const MCInstrDesc &MCID = getInstDesc(JalrInst.getOpcode()); if (MCID.hasDelaySlot() && AssemblerOptions.back()->isReorder()) TOut.emitEmptyDelaySlot(hasShortDelaySlot(JalrInst), IDLoc, STI); return false; } /// Can the value be represented by a unsigned N-bit value and a shift left? template static bool isShiftedUIntAtAnyPosition(uint64_t x) { unsigned BitNum = findFirstSet(x); return (x == x >> BitNum << BitNum) && isUInt(x >> BitNum); } /// Load (or add) an immediate into a register. /// /// @param ImmValue The immediate to load. /// @param DstReg The register that will hold the immediate. /// @param SrcReg A register to add to the immediate or Mips::NoRegister /// for a simple initialization. /// @param Is32BitImm Is ImmValue 32-bit or 64-bit? /// @param IsAddress True if the immediate represents an address. False if it /// is an integer. /// @param IDLoc Location of the immediate in the source file. bool MipsAsmParser::loadImmediate(int64_t ImmValue, unsigned DstReg, unsigned SrcReg, bool Is32BitImm, bool IsAddress, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI) { MipsTargetStreamer &TOut = getTargetStreamer(); if (!Is32BitImm && !isGP64bit()) { Error(IDLoc, "instruction requires a 64-bit architecture"); return true; } if (Is32BitImm) { if (isInt<32>(ImmValue) || isUInt<32>(ImmValue)) { // Sign extend up to 64-bit so that the predicates match the hardware // behaviour. In particular, isInt<16>(0xffff8000) and similar should be // true. ImmValue = SignExtend64<32>(ImmValue); } else { Error(IDLoc, "instruction requires a 32-bit immediate"); return true; } } unsigned ZeroReg = IsAddress ? ABI.GetNullPtr() : ABI.GetZeroReg(); unsigned AdduOp = !Is32BitImm ? Mips::DADDu : Mips::ADDu; bool UseSrcReg = false; if (SrcReg != Mips::NoRegister) UseSrcReg = true; unsigned TmpReg = DstReg; if (UseSrcReg && getContext().getRegisterInfo()->isSuperOrSubRegisterEq(DstReg, SrcReg)) { // At this point we need AT to perform the expansions and we exit if it is // not available. unsigned ATReg = getATReg(IDLoc); if (!ATReg) return true; TmpReg = ATReg; } if (isInt<16>(ImmValue)) { if (!UseSrcReg) SrcReg = ZeroReg; // This doesn't quite follow the usual ABI expectations for N32 but matches // traditional assembler behaviour. N32 would normally use addiu for both // integers and addresses. if (IsAddress && !Is32BitImm) { TOut.emitRRI(Mips::DADDiu, DstReg, SrcReg, ImmValue, IDLoc, STI); return false; } TOut.emitRRI(Mips::ADDiu, DstReg, SrcReg, ImmValue, IDLoc, STI); return false; } if (isUInt<16>(ImmValue)) { unsigned TmpReg = DstReg; if (SrcReg == DstReg) { TmpReg = getATReg(IDLoc); if (!TmpReg) return true; } TOut.emitRRI(Mips::ORi, TmpReg, ZeroReg, ImmValue, IDLoc, STI); if (UseSrcReg) TOut.emitRRR(ABI.GetPtrAdduOp(), DstReg, TmpReg, SrcReg, IDLoc, STI); return false; } if (isInt<32>(ImmValue) || isUInt<32>(ImmValue)) { warnIfNoMacro(IDLoc); uint16_t Bits31To16 = (ImmValue >> 16) & 0xffff; uint16_t Bits15To0 = ImmValue & 0xffff; if (!Is32BitImm && !isInt<32>(ImmValue)) { // Traditional behaviour seems to special case this particular value. It's // not clear why other masks are handled differently. if (ImmValue == 0xffffffff) { TOut.emitRI(Mips::LUi, TmpReg, 0xffff, IDLoc, STI); TOut.emitRRI(Mips::DSRL32, TmpReg, TmpReg, 0, IDLoc, STI); if (UseSrcReg) TOut.emitRRR(AdduOp, DstReg, TmpReg, SrcReg, IDLoc, STI); return false; } // Expand to an ORi instead of a LUi to avoid sign-extending into the // upper 32 bits. TOut.emitRRI(Mips::ORi, TmpReg, ZeroReg, Bits31To16, IDLoc, STI); TOut.emitRRI(Mips::DSLL, TmpReg, TmpReg, 16, IDLoc, STI); if (Bits15To0) TOut.emitRRI(Mips::ORi, TmpReg, TmpReg, Bits15To0, IDLoc, STI); if (UseSrcReg) TOut.emitRRR(AdduOp, DstReg, TmpReg, SrcReg, IDLoc, STI); return false; } TOut.emitRI(Mips::LUi, TmpReg, Bits31To16, IDLoc, STI); if (Bits15To0) TOut.emitRRI(Mips::ORi, TmpReg, TmpReg, Bits15To0, IDLoc, STI); if (UseSrcReg) TOut.emitRRR(AdduOp, DstReg, TmpReg, SrcReg, IDLoc, STI); return false; } if (isShiftedUIntAtAnyPosition<16>(ImmValue)) { if (Is32BitImm) { Error(IDLoc, "instruction requires a 32-bit immediate"); return true; } // Traditionally, these immediates are shifted as little as possible and as // such we align the most significant bit to bit 15 of our temporary. unsigned FirstSet = findFirstSet((uint64_t)ImmValue); unsigned LastSet = findLastSet((uint64_t)ImmValue); unsigned ShiftAmount = FirstSet - (15 - (LastSet - FirstSet)); uint16_t Bits = (ImmValue >> ShiftAmount) & 0xffff; TOut.emitRRI(Mips::ORi, TmpReg, ZeroReg, Bits, IDLoc, STI); TOut.emitRRI(Mips::DSLL, TmpReg, TmpReg, ShiftAmount, IDLoc, STI); if (UseSrcReg) TOut.emitRRR(AdduOp, DstReg, TmpReg, SrcReg, IDLoc, STI); return false; } warnIfNoMacro(IDLoc); // The remaining case is packed with a sequence of dsll and ori with zeros // being omitted and any neighbouring dsll's being coalesced. // The highest 32-bit's are equivalent to a 32-bit immediate load. // Load bits 32-63 of ImmValue into bits 0-31 of the temporary register. if (loadImmediate(ImmValue >> 32, TmpReg, Mips::NoRegister, true, false, IDLoc, Out, STI)) return false; // Shift and accumulate into the register. If a 16-bit chunk is zero, then // skip it and defer the shift to the next chunk. unsigned ShiftCarriedForwards = 16; for (int BitNum = 16; BitNum >= 0; BitNum -= 16) { uint16_t ImmChunk = (ImmValue >> BitNum) & 0xffff; if (ImmChunk != 0) { TOut.emitDSLL(TmpReg, TmpReg, ShiftCarriedForwards, IDLoc, STI); TOut.emitRRI(Mips::ORi, TmpReg, TmpReg, ImmChunk, IDLoc, STI); ShiftCarriedForwards = 0; } ShiftCarriedForwards += 16; } ShiftCarriedForwards -= 16; // Finish any remaining shifts left by trailing zeros. if (ShiftCarriedForwards) TOut.emitDSLL(TmpReg, TmpReg, ShiftCarriedForwards, IDLoc, STI); if (UseSrcReg) TOut.emitRRR(AdduOp, DstReg, TmpReg, SrcReg, IDLoc, STI); return false; } bool MipsAsmParser::expandLoadImm(MCInst &Inst, bool Is32BitImm, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI) { const MCOperand &ImmOp = Inst.getOperand(1); assert(ImmOp.isImm() && "expected immediate operand kind"); const MCOperand &DstRegOp = Inst.getOperand(0); assert(DstRegOp.isReg() && "expected register operand kind"); if (loadImmediate(ImmOp.getImm(), DstRegOp.getReg(), Mips::NoRegister, Is32BitImm, false, IDLoc, Out, STI)) return true; return false; } bool MipsAsmParser::expandLoadAddress(unsigned DstReg, unsigned BaseReg, const MCOperand &Offset, bool Is32BitAddress, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI) { // la can't produce a usable address when addresses are 64-bit. if (Is32BitAddress && ABI.ArePtrs64bit()) { // FIXME: Demote this to a warning and continue as if we had 'dla' instead. // We currently can't do this because we depend on the equality // operator and N64 can end up with a GPR32/GPR64 mismatch. Error(IDLoc, "la used to load 64-bit address"); // Continue as if we had 'dla' instead. Is32BitAddress = false; return true; } // dla requires 64-bit addresses. if (!Is32BitAddress && !hasMips3()) { Error(IDLoc, "instruction requires a 64-bit architecture"); return true; } if (!Offset.isImm()) return loadAndAddSymbolAddress(Offset.getExpr(), DstReg, BaseReg, Is32BitAddress, IDLoc, Out, STI); if (!ABI.ArePtrs64bit()) { // Continue as if we had 'la' whether we had 'la' or 'dla'. Is32BitAddress = true; } return loadImmediate(Offset.getImm(), DstReg, BaseReg, Is32BitAddress, true, IDLoc, Out, STI); } bool MipsAsmParser::loadAndAddSymbolAddress(const MCExpr *SymExpr, unsigned DstReg, unsigned SrcReg, bool Is32BitSym, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI) { // FIXME: These expansions do not respect -mxgot. MipsTargetStreamer &TOut = getTargetStreamer(); bool UseSrcReg = SrcReg != Mips::NoRegister; warnIfNoMacro(IDLoc); if (inPicMode() && ABI.IsO32()) { MCValue Res; if (!SymExpr->evaluateAsRelocatable(Res, nullptr, nullptr)) { Error(IDLoc, "expected relocatable expression"); return true; } if (Res.getSymB() != nullptr) { Error(IDLoc, "expected relocatable expression with only one symbol"); return true; } // The case where the result register is $25 is somewhat special. If the // symbol in the final relocation is external and not modified with a // constant then we must use R_MIPS_CALL16 instead of R_MIPS_GOT16. if ((DstReg == Mips::T9 || DstReg == Mips::T9_64) && !UseSrcReg && Res.getConstant() == 0 && !(Res.getSymA()->getSymbol().isInSection() || Res.getSymA()->getSymbol().isTemporary() || (Res.getSymA()->getSymbol().isELF() && cast(Res.getSymA()->getSymbol()).getBinding() == ELF::STB_LOCAL))) { const MCExpr *CallExpr = MipsMCExpr::create(MipsMCExpr::MEK_GOT_CALL, SymExpr, getContext()); TOut.emitRRX(Mips::LW, DstReg, ABI.GetGlobalPtr(), MCOperand::createExpr(CallExpr), IDLoc, STI); return false; } // The remaining cases are: // External GOT: lw $tmp, %got(symbol+offset)($gp) // >addiu $tmp, $tmp, %lo(offset) // >addiu $rd, $tmp, $rs // Local GOT: lw $tmp, %got(symbol+offset)($gp) // addiu $tmp, $tmp, %lo(symbol+offset)($gp) // >addiu $rd, $tmp, $rs // The addiu's marked with a '>' may be omitted if they are redundant. If // this happens then the last instruction must use $rd as the result // register. const MipsMCExpr *GotExpr = MipsMCExpr::create(MipsMCExpr::MEK_GOT, SymExpr, getContext()); const MCExpr *LoExpr = nullptr; if (Res.getSymA()->getSymbol().isInSection() || Res.getSymA()->getSymbol().isTemporary()) LoExpr = MipsMCExpr::create(MipsMCExpr::MEK_LO, SymExpr, getContext()); else if (Res.getConstant() != 0) { // External symbols fully resolve the symbol with just the %got(symbol) // but we must still account for any offset to the symbol for expressions // like symbol+8. LoExpr = MCConstantExpr::create(Res.getConstant(), getContext()); } unsigned TmpReg = DstReg; if (UseSrcReg && getContext().getRegisterInfo()->isSuperOrSubRegisterEq(DstReg, SrcReg)) { // If $rs is the same as $rd, we need to use AT. // If it is not available we exit. unsigned ATReg = getATReg(IDLoc); if (!ATReg) return true; TmpReg = ATReg; } TOut.emitRRX(Mips::LW, TmpReg, ABI.GetGlobalPtr(), MCOperand::createExpr(GotExpr), IDLoc, STI); if (LoExpr) TOut.emitRRX(Mips::ADDiu, TmpReg, TmpReg, MCOperand::createExpr(LoExpr), IDLoc, STI); if (UseSrcReg) TOut.emitRRR(Mips::ADDu, DstReg, TmpReg, SrcReg, IDLoc, STI); return false; } if (inPicMode() && ABI.ArePtrs64bit()) { MCValue Res; if (!SymExpr->evaluateAsRelocatable(Res, nullptr, nullptr)) { Error(IDLoc, "expected relocatable expression"); return true; } if (Res.getSymB() != nullptr) { Error(IDLoc, "expected relocatable expression with only one symbol"); return true; } // The case where the result register is $25 is somewhat special. If the // symbol in the final relocation is external and not modified with a // constant then we must use R_MIPS_CALL16 instead of R_MIPS_GOT_DISP. if ((DstReg == Mips::T9 || DstReg == Mips::T9_64) && !UseSrcReg && Res.getConstant() == 0 && !(Res.getSymA()->getSymbol().isInSection() || Res.getSymA()->getSymbol().isTemporary() || (Res.getSymA()->getSymbol().isELF() && cast(Res.getSymA()->getSymbol()).getBinding() == ELF::STB_LOCAL))) { const MCExpr *CallExpr = MipsMCExpr::create(MipsMCExpr::MEK_GOT_CALL, SymExpr, getContext()); TOut.emitRRX(Mips::LD, DstReg, ABI.GetGlobalPtr(), MCOperand::createExpr(CallExpr), IDLoc, STI); return false; } // The remaining cases are: // Small offset: ld $tmp, %got_disp(symbol)($gp) // >daddiu $tmp, $tmp, offset // >daddu $rd, $tmp, $rs // The daddiu's marked with a '>' may be omitted if they are redundant. If // this happens then the last instruction must use $rd as the result // register. const MipsMCExpr *GotExpr = MipsMCExpr::create(MipsMCExpr::MEK_GOT_DISP, Res.getSymA(), getContext()); const MCExpr *LoExpr = nullptr; if (Res.getConstant() != 0) { // Symbols fully resolve with just the %got_disp(symbol) but we // must still account for any offset to the symbol for // expressions like symbol+8. LoExpr = MCConstantExpr::create(Res.getConstant(), getContext()); // FIXME: Offsets greater than 16 bits are not yet implemented. // FIXME: The correct range is a 32-bit sign-extended number. if (Res.getConstant() < -0x8000 || Res.getConstant() > 0x7fff) { Error(IDLoc, "macro instruction uses large offset, which is not " "currently supported"); return true; } } unsigned TmpReg = DstReg; if (UseSrcReg && getContext().getRegisterInfo()->isSuperOrSubRegisterEq(DstReg, SrcReg)) { // If $rs is the same as $rd, we need to use AT. // If it is not available we exit. unsigned ATReg = getATReg(IDLoc); if (!ATReg) return true; TmpReg = ATReg; } TOut.emitRRX(Mips::LD, TmpReg, ABI.GetGlobalPtr(), MCOperand::createExpr(GotExpr), IDLoc, STI); if (LoExpr) TOut.emitRRX(Mips::DADDiu, TmpReg, TmpReg, MCOperand::createExpr(LoExpr), IDLoc, STI); if (UseSrcReg) TOut.emitRRR(Mips::DADDu, DstReg, TmpReg, SrcReg, IDLoc, STI); return false; } const MipsMCExpr *HiExpr = MipsMCExpr::create(MipsMCExpr::MEK_HI, SymExpr, getContext()); const MipsMCExpr *LoExpr = MipsMCExpr::create(MipsMCExpr::MEK_LO, SymExpr, getContext()); // This is the 64-bit symbol address expansion. if (ABI.ArePtrs64bit() && isGP64bit()) { // We need AT for the 64-bit expansion in the cases where the optional // source register is the destination register and for the superscalar // scheduled form. // // If it is not available we exit if the destination is the same as the // source register. const MipsMCExpr *HighestExpr = MipsMCExpr::create(MipsMCExpr::MEK_HIGHEST, SymExpr, getContext()); const MipsMCExpr *HigherExpr = MipsMCExpr::create(MipsMCExpr::MEK_HIGHER, SymExpr, getContext()); bool RdRegIsRsReg = getContext().getRegisterInfo()->isSuperOrSubRegisterEq(DstReg, SrcReg); if (canUseATReg() && UseSrcReg && RdRegIsRsReg) { unsigned ATReg = getATReg(IDLoc); // If $rs is the same as $rd: // (d)la $rd, sym($rd) => lui $at, %highest(sym) // daddiu $at, $at, %higher(sym) // dsll $at, $at, 16 // daddiu $at, $at, %hi(sym) // dsll $at, $at, 16 // daddiu $at, $at, %lo(sym) // daddu $rd, $at, $rd TOut.emitRX(Mips::LUi, ATReg, MCOperand::createExpr(HighestExpr), IDLoc, STI); TOut.emitRRX(Mips::DADDiu, ATReg, ATReg, MCOperand::createExpr(HigherExpr), IDLoc, STI); TOut.emitRRI(Mips::DSLL, ATReg, ATReg, 16, IDLoc, STI); TOut.emitRRX(Mips::DADDiu, ATReg, ATReg, MCOperand::createExpr(HiExpr), IDLoc, STI); TOut.emitRRI(Mips::DSLL, ATReg, ATReg, 16, IDLoc, STI); TOut.emitRRX(Mips::DADDiu, ATReg, ATReg, MCOperand::createExpr(LoExpr), IDLoc, STI); TOut.emitRRR(Mips::DADDu, DstReg, ATReg, SrcReg, IDLoc, STI); return false; } else if (canUseATReg() && !RdRegIsRsReg) { unsigned ATReg = getATReg(IDLoc); // If the $rs is different from $rd or if $rs isn't specified and we // have $at available: // (d)la $rd, sym/sym($rs) => lui $rd, %highest(sym) // lui $at, %hi(sym) // daddiu $rd, $rd, %higher(sym) // daddiu $at, $at, %lo(sym) // dsll32 $rd, $rd, 0 // daddu $rd, $rd, $at // (daddu $rd, $rd, $rs) // // Which is preferred for superscalar issue. TOut.emitRX(Mips::LUi, DstReg, MCOperand::createExpr(HighestExpr), IDLoc, STI); TOut.emitRX(Mips::LUi, ATReg, MCOperand::createExpr(HiExpr), IDLoc, STI); TOut.emitRRX(Mips::DADDiu, DstReg, DstReg, MCOperand::createExpr(HigherExpr), IDLoc, STI); TOut.emitRRX(Mips::DADDiu, ATReg, ATReg, MCOperand::createExpr(LoExpr), IDLoc, STI); TOut.emitRRI(Mips::DSLL32, DstReg, DstReg, 0, IDLoc, STI); TOut.emitRRR(Mips::DADDu, DstReg, DstReg, ATReg, IDLoc, STI); if (UseSrcReg) TOut.emitRRR(Mips::DADDu, DstReg, DstReg, SrcReg, IDLoc, STI); return false; } else if (!canUseATReg() && !RdRegIsRsReg) { // Otherwise, synthesize the address in the destination register // serially: // (d)la $rd, sym/sym($rs) => lui $rd, %highest(sym) // daddiu $rd, $rd, %higher(sym) // dsll $rd, $rd, 16 // daddiu $rd, $rd, %hi(sym) // dsll $rd, $rd, 16 // daddiu $rd, $rd, %lo(sym) TOut.emitRX(Mips::LUi, DstReg, MCOperand::createExpr(HighestExpr), IDLoc, STI); TOut.emitRRX(Mips::DADDiu, DstReg, DstReg, MCOperand::createExpr(HigherExpr), IDLoc, STI); TOut.emitRRI(Mips::DSLL, DstReg, DstReg, 16, IDLoc, STI); TOut.emitRRX(Mips::DADDiu, DstReg, DstReg, MCOperand::createExpr(HiExpr), IDLoc, STI); TOut.emitRRI(Mips::DSLL, DstReg, DstReg, 16, IDLoc, STI); TOut.emitRRX(Mips::DADDiu, DstReg, DstReg, MCOperand::createExpr(LoExpr), IDLoc, STI); if (UseSrcReg) TOut.emitRRR(Mips::DADDu, DstReg, DstReg, SrcReg, IDLoc, STI); return false; } else { // We have a case where SrcReg == DstReg and we don't have $at // available. We can't expand this case, so error out appropriately. assert(SrcReg == DstReg && !canUseATReg() && "Could have expanded dla but didn't?"); reportParseError(IDLoc, "pseudo-instruction requires $at, which is not available"); return true; } } // And now, the 32-bit symbol address expansion: // If $rs is the same as $rd: // (d)la $rd, sym($rd) => lui $at, %hi(sym) // ori $at, $at, %lo(sym) // addu $rd, $at, $rd // Otherwise, if the $rs is different from $rd or if $rs isn't specified: // (d)la $rd, sym/sym($rs) => lui $rd, %hi(sym) // ori $rd, $rd, %lo(sym) // (addu $rd, $rd, $rs) unsigned TmpReg = DstReg; if (UseSrcReg && getContext().getRegisterInfo()->isSuperOrSubRegisterEq(DstReg, SrcReg)) { // If $rs is the same as $rd, we need to use AT. // If it is not available we exit. unsigned ATReg = getATReg(IDLoc); if (!ATReg) return true; TmpReg = ATReg; } TOut.emitRX(Mips::LUi, TmpReg, MCOperand::createExpr(HiExpr), IDLoc, STI); TOut.emitRRX(Mips::ADDiu, TmpReg, TmpReg, MCOperand::createExpr(LoExpr), IDLoc, STI); if (UseSrcReg) TOut.emitRRR(Mips::ADDu, DstReg, TmpReg, SrcReg, IDLoc, STI); else assert( getContext().getRegisterInfo()->isSuperOrSubRegisterEq(DstReg, TmpReg)); return false; } // Each double-precision register DO-D15 overlaps with two of the single // precision registers F0-F31. As an example, all of the following hold true: // D0 + 1 == F1, F1 + 1 == D1, F1 + 1 == F2, depending on the context. static unsigned nextReg(unsigned Reg) { if (MipsMCRegisterClasses[Mips::FGR32RegClassID].contains(Reg)) return Reg == (unsigned)Mips::F31 ? (unsigned)Mips::F0 : Reg + 1; switch (Reg) { default: llvm_unreachable("Unknown register in assembly macro expansion!"); case Mips::ZERO: return Mips::AT; case Mips::AT: return Mips::V0; case Mips::V0: return Mips::V1; case Mips::V1: return Mips::A0; case Mips::A0: return Mips::A1; case Mips::A1: return Mips::A2; case Mips::A2: return Mips::A3; case Mips::A3: return Mips::T0; case Mips::T0: return Mips::T1; case Mips::T1: return Mips::T2; case Mips::T2: return Mips::T3; case Mips::T3: return Mips::T4; case Mips::T4: return Mips::T5; case Mips::T5: return Mips::T6; case Mips::T6: return Mips::T7; case Mips::T7: return Mips::S0; case Mips::S0: return Mips::S1; case Mips::S1: return Mips::S2; case Mips::S2: return Mips::S3; case Mips::S3: return Mips::S4; case Mips::S4: return Mips::S5; case Mips::S5: return Mips::S6; case Mips::S6: return Mips::S7; case Mips::S7: return Mips::T8; case Mips::T8: return Mips::T9; case Mips::T9: return Mips::K0; case Mips::K0: return Mips::K1; case Mips::K1: return Mips::GP; case Mips::GP: return Mips::SP; case Mips::SP: return Mips::FP; case Mips::FP: return Mips::RA; case Mips::RA: return Mips::ZERO; case Mips::D0: return Mips::F1; case Mips::D1: return Mips::F3; case Mips::D2: return Mips::F5; case Mips::D3: return Mips::F7; case Mips::D4: return Mips::F9; case Mips::D5: return Mips::F11; case Mips::D6: return Mips::F13; case Mips::D7: return Mips::F15; case Mips::D8: return Mips::F17; case Mips::D9: return Mips::F19; case Mips::D10: return Mips::F21; case Mips::D11: return Mips::F23; case Mips::D12: return Mips::F25; case Mips::D13: return Mips::F27; case Mips::D14: return Mips::F29; case Mips::D15: return Mips::F31; } } // FIXME: This method is too general. In principle we should compute the number // of instructions required to synthesize the immediate inline compared to // synthesizing the address inline and relying on non .text sections. // For static O32 and N32 this may yield a small benefit, for static N64 this is // likely to yield a much larger benefit as we have to synthesize a 64bit // address to load a 64 bit value. bool MipsAsmParser::emitPartialAddress(MipsTargetStreamer &TOut, SMLoc IDLoc, MCSymbol *Sym) { unsigned ATReg = getATReg(IDLoc); if (!ATReg) return true; if(IsPicEnabled) { const MCExpr *GotSym = MCSymbolRefExpr::create(Sym, MCSymbolRefExpr::VK_None, getContext()); const MipsMCExpr *GotExpr = MipsMCExpr::create(MipsMCExpr::MEK_GOT, GotSym, getContext()); if(isABI_O32() || isABI_N32()) { TOut.emitRRX(Mips::LW, ATReg, Mips::GP, MCOperand::createExpr(GotExpr), IDLoc, STI); } else { //isABI_N64() TOut.emitRRX(Mips::LD, ATReg, Mips::GP, MCOperand::createExpr(GotExpr), IDLoc, STI); } } else { //!IsPicEnabled const MCExpr *HiSym = MCSymbolRefExpr::create(Sym, MCSymbolRefExpr::VK_None, getContext()); const MipsMCExpr *HiExpr = MipsMCExpr::create(MipsMCExpr::MEK_HI, HiSym, getContext()); // FIXME: This is technically correct but gives a different result to gas, // but gas is incomplete there (it has a fixme noting it doesn't work with // 64-bit addresses). // FIXME: With -msym32 option, the address expansion for N64 should probably // use the O32 / N32 case. It's safe to use the 64 address expansion as the // symbol's value is considered sign extended. if(isABI_O32() || isABI_N32()) { TOut.emitRX(Mips::LUi, ATReg, MCOperand::createExpr(HiExpr), IDLoc, STI); } else { //isABI_N64() const MCExpr *HighestSym = MCSymbolRefExpr::create(Sym, MCSymbolRefExpr::VK_None, getContext()); const MipsMCExpr *HighestExpr = MipsMCExpr::create(MipsMCExpr::MEK_HIGHEST, HighestSym, getContext()); const MCExpr *HigherSym = MCSymbolRefExpr::create(Sym, MCSymbolRefExpr::VK_None, getContext()); const MipsMCExpr *HigherExpr = MipsMCExpr::create(MipsMCExpr::MEK_HIGHER, HigherSym, getContext()); TOut.emitRX(Mips::LUi, ATReg, MCOperand::createExpr(HighestExpr), IDLoc, STI); TOut.emitRRX(Mips::DADDiu, ATReg, ATReg, MCOperand::createExpr(HigherExpr), IDLoc, STI); TOut.emitRRI(Mips::DSLL, ATReg, ATReg, 16, IDLoc, STI); TOut.emitRRX(Mips::DADDiu, ATReg, ATReg, MCOperand::createExpr(HiExpr), IDLoc, STI); TOut.emitRRI(Mips::DSLL, ATReg, ATReg, 16, IDLoc, STI); } } return false; } bool MipsAsmParser::expandLoadImmReal(MCInst &Inst, bool IsSingle, bool IsGPR, bool Is64FPU, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI) { MipsTargetStreamer &TOut = getTargetStreamer(); assert(Inst.getNumOperands() == 2 && "Invalid operand count"); assert(Inst.getOperand(0).isReg() && Inst.getOperand(1).isImm() && "Invalid instruction operand."); unsigned FirstReg = Inst.getOperand(0).getReg(); uint64_t ImmOp64 = Inst.getOperand(1).getImm(); uint32_t HiImmOp64 = (ImmOp64 & 0xffffffff00000000) >> 32; // If ImmOp64 is AsmToken::Integer type (all bits set to zero in the // exponent field), convert it to double (e.g. 1 to 1.0) if ((HiImmOp64 & 0x7ff00000) == 0) { APFloat RealVal(APFloat::IEEEdouble(), ImmOp64); ImmOp64 = RealVal.bitcastToAPInt().getZExtValue(); } uint32_t LoImmOp64 = ImmOp64 & 0xffffffff; HiImmOp64 = (ImmOp64 & 0xffffffff00000000) >> 32; if (IsSingle) { // Conversion of a double in an uint64_t to a float in a uint32_t, // retaining the bit pattern of a float. uint32_t ImmOp32; double doubleImm = BitsToDouble(ImmOp64); float tmp_float = static_cast(doubleImm); ImmOp32 = FloatToBits(tmp_float); if (IsGPR) { if (loadImmediate(ImmOp32, FirstReg, Mips::NoRegister, true, true, IDLoc, Out, STI)) return true; return false; } else { unsigned ATReg = getATReg(IDLoc); if (!ATReg) return true; if (LoImmOp64 == 0) { if (loadImmediate(ImmOp32, ATReg, Mips::NoRegister, true, true, IDLoc, Out, STI)) return true; TOut.emitRR(Mips::MTC1, FirstReg, ATReg, IDLoc, STI); return false; } MCSection *CS = getStreamer().getCurrentSectionOnly(); // FIXME: Enhance this expansion to use the .lit4 & .lit8 sections // where appropriate. MCSection *ReadOnlySection = getContext().getELFSection( ".rodata", ELF::SHT_PROGBITS, ELF::SHF_ALLOC); MCSymbol *Sym = getContext().createTempSymbol(); const MCExpr *LoSym = MCSymbolRefExpr::create(Sym, MCSymbolRefExpr::VK_None, getContext()); const MipsMCExpr *LoExpr = MipsMCExpr::create(MipsMCExpr::MEK_LO, LoSym, getContext()); getStreamer().SwitchSection(ReadOnlySection); getStreamer().EmitLabel(Sym, IDLoc); getStreamer().EmitIntValue(ImmOp32, 4); getStreamer().SwitchSection(CS); if(emitPartialAddress(TOut, IDLoc, Sym)) return true; TOut.emitRRX(Mips::LWC1, FirstReg, ATReg, MCOperand::createExpr(LoExpr), IDLoc, STI); } return false; } // if(!IsSingle) unsigned ATReg = getATReg(IDLoc); if (!ATReg) return true; if (IsGPR) { if (LoImmOp64 == 0) { if(isABI_N32() || isABI_N64()) { if (loadImmediate(HiImmOp64, FirstReg, Mips::NoRegister, false, true, IDLoc, Out, STI)) return true; return false; } else { if (loadImmediate(HiImmOp64, FirstReg, Mips::NoRegister, true, true, IDLoc, Out, STI)) return true; if (loadImmediate(0, nextReg(FirstReg), Mips::NoRegister, true, true, IDLoc, Out, STI)) return true; return false; } } MCSection *CS = getStreamer().getCurrentSectionOnly(); MCSection *ReadOnlySection = getContext().getELFSection( ".rodata", ELF::SHT_PROGBITS, ELF::SHF_ALLOC); MCSymbol *Sym = getContext().createTempSymbol(); const MCExpr *LoSym = MCSymbolRefExpr::create(Sym, MCSymbolRefExpr::VK_None, getContext()); const MipsMCExpr *LoExpr = MipsMCExpr::create(MipsMCExpr::MEK_LO, LoSym, getContext()); getStreamer().SwitchSection(ReadOnlySection); getStreamer().EmitLabel(Sym, IDLoc); getStreamer().EmitIntValue(HiImmOp64, 4); getStreamer().EmitIntValue(LoImmOp64, 4); getStreamer().SwitchSection(CS); if(emitPartialAddress(TOut, IDLoc, Sym)) return true; if(isABI_N64()) TOut.emitRRX(Mips::DADDiu, ATReg, ATReg, MCOperand::createExpr(LoExpr), IDLoc, STI); else TOut.emitRRX(Mips::ADDiu, ATReg, ATReg, MCOperand::createExpr(LoExpr), IDLoc, STI); if(isABI_N32() || isABI_N64()) TOut.emitRRI(Mips::LD, FirstReg, ATReg, 0, IDLoc, STI); else { TOut.emitRRI(Mips::LW, FirstReg, ATReg, 0, IDLoc, STI); TOut.emitRRI(Mips::LW, nextReg(FirstReg), ATReg, 4, IDLoc, STI); } return false; } else { // if(!IsGPR && !IsSingle) if ((LoImmOp64 == 0) && !((HiImmOp64 & 0xffff0000) && (HiImmOp64 & 0x0000ffff))) { // FIXME: In the case where the constant is zero, we can load the // register directly from the zero register. if (loadImmediate(HiImmOp64, ATReg, Mips::NoRegister, true, true, IDLoc, Out, STI)) return true; if (isABI_N32() || isABI_N64()) TOut.emitRR(Mips::DMTC1, FirstReg, ATReg, IDLoc, STI); else if (hasMips32r2()) { TOut.emitRR(Mips::MTC1, FirstReg, Mips::ZERO, IDLoc, STI); TOut.emitRRR(Mips::MTHC1_D32, FirstReg, FirstReg, ATReg, IDLoc, STI); } else { TOut.emitRR(Mips::MTC1, nextReg(FirstReg), ATReg, IDLoc, STI); TOut.emitRR(Mips::MTC1, FirstReg, Mips::ZERO, IDLoc, STI); } return false; } MCSection *CS = getStreamer().getCurrentSectionOnly(); // FIXME: Enhance this expansion to use the .lit4 & .lit8 sections // where appropriate. MCSection *ReadOnlySection = getContext().getELFSection( ".rodata", ELF::SHT_PROGBITS, ELF::SHF_ALLOC); MCSymbol *Sym = getContext().createTempSymbol(); const MCExpr *LoSym = MCSymbolRefExpr::create(Sym, MCSymbolRefExpr::VK_None, getContext()); const MipsMCExpr *LoExpr = MipsMCExpr::create(MipsMCExpr::MEK_LO, LoSym, getContext()); getStreamer().SwitchSection(ReadOnlySection); getStreamer().EmitLabel(Sym, IDLoc); getStreamer().EmitIntValue(HiImmOp64, 4); getStreamer().EmitIntValue(LoImmOp64, 4); getStreamer().SwitchSection(CS); if(emitPartialAddress(TOut, IDLoc, Sym)) return true; TOut.emitRRX(Is64FPU ? Mips::LDC164 : Mips::LDC1, FirstReg, ATReg, MCOperand::createExpr(LoExpr), IDLoc, STI); } return false; } bool MipsAsmParser::expandUncondBranchMMPseudo(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI) { MipsTargetStreamer &TOut = getTargetStreamer(); assert(getInstDesc(Inst.getOpcode()).getNumOperands() == 1 && "unexpected number of operands"); MCOperand Offset = Inst.getOperand(0); if (Offset.isExpr()) { Inst.clear(); Inst.setOpcode(Mips::BEQ_MM); Inst.addOperand(MCOperand::createReg(Mips::ZERO)); Inst.addOperand(MCOperand::createReg(Mips::ZERO)); Inst.addOperand(MCOperand::createExpr(Offset.getExpr())); } else { assert(Offset.isImm() && "expected immediate operand kind"); if (isInt<11>(Offset.getImm())) { // If offset fits into 11 bits then this instruction becomes microMIPS // 16-bit unconditional branch instruction. if (inMicroMipsMode()) Inst.setOpcode(hasMips32r6() ? Mips::BC16_MMR6 : Mips::B16_MM); } else { if (!isInt<17>(Offset.getImm())) return Error(IDLoc, "branch target out of range"); if (OffsetToAlignment(Offset.getImm(), 1LL << 1)) return Error(IDLoc, "branch to misaligned address"); Inst.clear(); Inst.setOpcode(Mips::BEQ_MM); Inst.addOperand(MCOperand::createReg(Mips::ZERO)); Inst.addOperand(MCOperand::createReg(Mips::ZERO)); Inst.addOperand(MCOperand::createImm(Offset.getImm())); } } Out.EmitInstruction(Inst, *STI); // If .set reorder is active and branch instruction has a delay slot, // emit a NOP after it. const MCInstrDesc &MCID = getInstDesc(Inst.getOpcode()); if (MCID.hasDelaySlot() && AssemblerOptions.back()->isReorder()) TOut.emitEmptyDelaySlot(true, IDLoc, STI); return false; } bool MipsAsmParser::expandBranchImm(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI) { MipsTargetStreamer &TOut = getTargetStreamer(); const MCOperand &DstRegOp = Inst.getOperand(0); assert(DstRegOp.isReg() && "expected register operand kind"); const MCOperand &ImmOp = Inst.getOperand(1); assert(ImmOp.isImm() && "expected immediate operand kind"); const MCOperand &MemOffsetOp = Inst.getOperand(2); assert((MemOffsetOp.isImm() || MemOffsetOp.isExpr()) && "expected immediate or expression operand"); bool IsLikely = false; unsigned OpCode = 0; switch(Inst.getOpcode()) { case Mips::BneImm: OpCode = Mips::BNE; break; case Mips::BeqImm: OpCode = Mips::BEQ; break; case Mips::BEQLImmMacro: OpCode = Mips::BEQL; IsLikely = true; break; case Mips::BNELImmMacro: OpCode = Mips::BNEL; IsLikely = true; break; default: llvm_unreachable("Unknown immediate branch pseudo-instruction."); break; } int64_t ImmValue = ImmOp.getImm(); if (ImmValue == 0) { if (IsLikely) { TOut.emitRRX(OpCode, DstRegOp.getReg(), Mips::ZERO, MCOperand::createExpr(MemOffsetOp.getExpr()), IDLoc, STI); TOut.emitRRI(Mips::SLL, Mips::ZERO, Mips::ZERO, 0, IDLoc, STI); } else TOut.emitRRX(OpCode, DstRegOp.getReg(), Mips::ZERO, MemOffsetOp, IDLoc, STI); } else { warnIfNoMacro(IDLoc); unsigned ATReg = getATReg(IDLoc); if (!ATReg) return true; if (loadImmediate(ImmValue, ATReg, Mips::NoRegister, !isGP64bit(), true, IDLoc, Out, STI)) return true; if (IsLikely) { TOut.emitRRX(OpCode, DstRegOp.getReg(), ATReg, MCOperand::createExpr(MemOffsetOp.getExpr()), IDLoc, STI); TOut.emitRRI(Mips::SLL, Mips::ZERO, Mips::ZERO, 0, IDLoc, STI); } else TOut.emitRRX(OpCode, DstRegOp.getReg(), ATReg, MemOffsetOp, IDLoc, STI); } return false; } void MipsAsmParser::expandMemInst(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI, bool IsLoad) { const MCOperand &DstRegOp = Inst.getOperand(0); assert(DstRegOp.isReg() && "expected register operand kind"); const MCOperand &BaseRegOp = Inst.getOperand(1); assert(BaseRegOp.isReg() && "expected register operand kind"); const MCOperand &OffsetOp = Inst.getOperand(2); MipsTargetStreamer &TOut = getTargetStreamer(); unsigned DstReg = DstRegOp.getReg(); unsigned BaseReg = BaseRegOp.getReg(); unsigned TmpReg = DstReg; const MCInstrDesc &Desc = getInstDesc(Inst.getOpcode()); int16_t DstRegClass = Desc.OpInfo[0].RegClass; unsigned DstRegClassID = getContext().getRegisterInfo()->getRegClass(DstRegClass).getID(); bool IsGPR = (DstRegClassID == Mips::GPR32RegClassID) || (DstRegClassID == Mips::GPR64RegClassID); if (!IsLoad || !IsGPR || (BaseReg == DstReg)) { // At this point we need AT to perform the expansions // and we exit if it is not available. TmpReg = getATReg(IDLoc); if (!TmpReg) return; } if (OffsetOp.isImm()) { int64_t LoOffset = OffsetOp.getImm() & 0xffff; int64_t HiOffset = OffsetOp.getImm() & ~0xffff; // If msb of LoOffset is 1(negative number) we must increment // HiOffset to account for the sign-extension of the low part. if (LoOffset & 0x8000) HiOffset += 0x10000; bool IsLargeOffset = HiOffset != 0; if (IsLargeOffset) { bool Is32BitImm = (HiOffset >> 32) == 0; if (loadImmediate(HiOffset, TmpReg, Mips::NoRegister, Is32BitImm, true, IDLoc, Out, STI)) return; } if (BaseReg != Mips::ZERO && BaseReg != Mips::ZERO_64) TOut.emitRRR(isGP64bit() ? Mips::DADDu : Mips::ADDu, TmpReg, TmpReg, BaseReg, IDLoc, STI); TOut.emitRRI(Inst.getOpcode(), DstReg, TmpReg, LoOffset, IDLoc, STI); } else { assert(OffsetOp.isExpr() && "expected expression operand kind"); const MCExpr *ExprOffset = OffsetOp.getExpr(); MCOperand LoOperand = MCOperand::createExpr( MipsMCExpr::create(MipsMCExpr::MEK_LO, ExprOffset, getContext())); MCOperand HiOperand = MCOperand::createExpr( MipsMCExpr::create(MipsMCExpr::MEK_HI, ExprOffset, getContext())); if (IsLoad) TOut.emitLoadWithSymOffset(Inst.getOpcode(), DstReg, BaseReg, HiOperand, LoOperand, TmpReg, IDLoc, STI); else TOut.emitStoreWithSymOffset(Inst.getOpcode(), DstReg, BaseReg, HiOperand, LoOperand, TmpReg, IDLoc, STI); } } bool MipsAsmParser::expandLoadStoreMultiple(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI) { unsigned OpNum = Inst.getNumOperands(); unsigned Opcode = Inst.getOpcode(); unsigned NewOpcode = Opcode == Mips::SWM_MM ? Mips::SWM32_MM : Mips::LWM32_MM; assert(Inst.getOperand(OpNum - 1).isImm() && Inst.getOperand(OpNum - 2).isReg() && Inst.getOperand(OpNum - 3).isReg() && "Invalid instruction operand."); if (OpNum < 8 && Inst.getOperand(OpNum - 1).getImm() <= 60 && Inst.getOperand(OpNum - 1).getImm() >= 0 && (Inst.getOperand(OpNum - 2).getReg() == Mips::SP || Inst.getOperand(OpNum - 2).getReg() == Mips::SP_64) && (Inst.getOperand(OpNum - 3).getReg() == Mips::RA || Inst.getOperand(OpNum - 3).getReg() == Mips::RA_64)) { // It can be implemented as SWM16 or LWM16 instruction. if (inMicroMipsMode() && hasMips32r6()) NewOpcode = Opcode == Mips::SWM_MM ? Mips::SWM16_MMR6 : Mips::LWM16_MMR6; else NewOpcode = Opcode == Mips::SWM_MM ? Mips::SWM16_MM : Mips::LWM16_MM; } Inst.setOpcode(NewOpcode); Out.EmitInstruction(Inst, *STI); return false; } bool MipsAsmParser::expandCondBranches(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI) { MipsTargetStreamer &TOut = getTargetStreamer(); bool EmittedNoMacroWarning = false; unsigned PseudoOpcode = Inst.getOpcode(); unsigned SrcReg = Inst.getOperand(0).getReg(); const MCOperand &TrgOp = Inst.getOperand(1); const MCExpr *OffsetExpr = Inst.getOperand(2).getExpr(); unsigned ZeroSrcOpcode, ZeroTrgOpcode; bool ReverseOrderSLT, IsUnsigned, IsLikely, AcceptsEquality; unsigned TrgReg; if (TrgOp.isReg()) TrgReg = TrgOp.getReg(); else if (TrgOp.isImm()) { warnIfNoMacro(IDLoc); EmittedNoMacroWarning = true; TrgReg = getATReg(IDLoc); if (!TrgReg) return true; switch(PseudoOpcode) { default: llvm_unreachable("unknown opcode for branch pseudo-instruction"); case Mips::BLTImmMacro: PseudoOpcode = Mips::BLT; break; case Mips::BLEImmMacro: PseudoOpcode = Mips::BLE; break; case Mips::BGEImmMacro: PseudoOpcode = Mips::BGE; break; case Mips::BGTImmMacro: PseudoOpcode = Mips::BGT; break; case Mips::BLTUImmMacro: PseudoOpcode = Mips::BLTU; break; case Mips::BLEUImmMacro: PseudoOpcode = Mips::BLEU; break; case Mips::BGEUImmMacro: PseudoOpcode = Mips::BGEU; break; case Mips::BGTUImmMacro: PseudoOpcode = Mips::BGTU; break; case Mips::BLTLImmMacro: PseudoOpcode = Mips::BLTL; break; case Mips::BLELImmMacro: PseudoOpcode = Mips::BLEL; break; case Mips::BGELImmMacro: PseudoOpcode = Mips::BGEL; break; case Mips::BGTLImmMacro: PseudoOpcode = Mips::BGTL; break; case Mips::BLTULImmMacro: PseudoOpcode = Mips::BLTUL; break; case Mips::BLEULImmMacro: PseudoOpcode = Mips::BLEUL; break; case Mips::BGEULImmMacro: PseudoOpcode = Mips::BGEUL; break; case Mips::BGTULImmMacro: PseudoOpcode = Mips::BGTUL; break; } if (loadImmediate(TrgOp.getImm(), TrgReg, Mips::NoRegister, !isGP64bit(), false, IDLoc, Out, STI)) return true; } switch (PseudoOpcode) { case Mips::BLT: case Mips::BLTU: case Mips::BLTL: case Mips::BLTUL: AcceptsEquality = false; ReverseOrderSLT = false; IsUnsigned = ((PseudoOpcode == Mips::BLTU) || (PseudoOpcode == Mips::BLTUL)); IsLikely = ((PseudoOpcode == Mips::BLTL) || (PseudoOpcode == Mips::BLTUL)); ZeroSrcOpcode = Mips::BGTZ; ZeroTrgOpcode = Mips::BLTZ; break; case Mips::BLE: case Mips::BLEU: case Mips::BLEL: case Mips::BLEUL: AcceptsEquality = true; ReverseOrderSLT = true; IsUnsigned = ((PseudoOpcode == Mips::BLEU) || (PseudoOpcode == Mips::BLEUL)); IsLikely = ((PseudoOpcode == Mips::BLEL) || (PseudoOpcode == Mips::BLEUL)); ZeroSrcOpcode = Mips::BGEZ; ZeroTrgOpcode = Mips::BLEZ; break; case Mips::BGE: case Mips::BGEU: case Mips::BGEL: case Mips::BGEUL: AcceptsEquality = true; ReverseOrderSLT = false; IsUnsigned = ((PseudoOpcode == Mips::BGEU) || (PseudoOpcode == Mips::BGEUL)); IsLikely = ((PseudoOpcode == Mips::BGEL) || (PseudoOpcode == Mips::BGEUL)); ZeroSrcOpcode = Mips::BLEZ; ZeroTrgOpcode = Mips::BGEZ; break; case Mips::BGT: case Mips::BGTU: case Mips::BGTL: case Mips::BGTUL: AcceptsEquality = false; ReverseOrderSLT = true; IsUnsigned = ((PseudoOpcode == Mips::BGTU) || (PseudoOpcode == Mips::BGTUL)); IsLikely = ((PseudoOpcode == Mips::BGTL) || (PseudoOpcode == Mips::BGTUL)); ZeroSrcOpcode = Mips::BLTZ; ZeroTrgOpcode = Mips::BGTZ; break; default: llvm_unreachable("unknown opcode for branch pseudo-instruction"); } bool IsTrgRegZero = (TrgReg == Mips::ZERO); bool IsSrcRegZero = (SrcReg == Mips::ZERO); if (IsSrcRegZero && IsTrgRegZero) { // FIXME: All of these Opcode-specific if's are needed for compatibility // with GAS' behaviour. However, they may not generate the most efficient // code in some circumstances. if (PseudoOpcode == Mips::BLT) { TOut.emitRX(Mips::BLTZ, Mips::ZERO, MCOperand::createExpr(OffsetExpr), IDLoc, STI); return false; } if (PseudoOpcode == Mips::BLE) { TOut.emitRX(Mips::BLEZ, Mips::ZERO, MCOperand::createExpr(OffsetExpr), IDLoc, STI); Warning(IDLoc, "branch is always taken"); return false; } if (PseudoOpcode == Mips::BGE) { TOut.emitRX(Mips::BGEZ, Mips::ZERO, MCOperand::createExpr(OffsetExpr), IDLoc, STI); Warning(IDLoc, "branch is always taken"); return false; } if (PseudoOpcode == Mips::BGT) { TOut.emitRX(Mips::BGTZ, Mips::ZERO, MCOperand::createExpr(OffsetExpr), IDLoc, STI); return false; } if (PseudoOpcode == Mips::BGTU) { TOut.emitRRX(Mips::BNE, Mips::ZERO, Mips::ZERO, MCOperand::createExpr(OffsetExpr), IDLoc, STI); return false; } if (AcceptsEquality) { // If both registers are $0 and the pseudo-branch accepts equality, it // will always be taken, so we emit an unconditional branch. TOut.emitRRX(Mips::BEQ, Mips::ZERO, Mips::ZERO, MCOperand::createExpr(OffsetExpr), IDLoc, STI); Warning(IDLoc, "branch is always taken"); return false; } // If both registers are $0 and the pseudo-branch does not accept // equality, it will never be taken, so we don't have to emit anything. return false; } if (IsSrcRegZero || IsTrgRegZero) { if ((IsSrcRegZero && PseudoOpcode == Mips::BGTU) || (IsTrgRegZero && PseudoOpcode == Mips::BLTU)) { // If the $rs is $0 and the pseudo-branch is BGTU (0 > x) or // if the $rt is $0 and the pseudo-branch is BLTU (x < 0), // the pseudo-branch will never be taken, so we don't emit anything. // This only applies to unsigned pseudo-branches. return false; } if ((IsSrcRegZero && PseudoOpcode == Mips::BLEU) || (IsTrgRegZero && PseudoOpcode == Mips::BGEU)) { // If the $rs is $0 and the pseudo-branch is BLEU (0 <= x) or // if the $rt is $0 and the pseudo-branch is BGEU (x >= 0), // the pseudo-branch will always be taken, so we emit an unconditional // branch. // This only applies to unsigned pseudo-branches. TOut.emitRRX(Mips::BEQ, Mips::ZERO, Mips::ZERO, MCOperand::createExpr(OffsetExpr), IDLoc, STI); Warning(IDLoc, "branch is always taken"); return false; } if (IsUnsigned) { // If the $rs is $0 and the pseudo-branch is BLTU (0 < x) or // if the $rt is $0 and the pseudo-branch is BGTU (x > 0), // the pseudo-branch will be taken only when the non-zero register is // different from 0, so we emit a BNEZ. // // If the $rs is $0 and the pseudo-branch is BGEU (0 >= x) or // if the $rt is $0 and the pseudo-branch is BLEU (x <= 0), // the pseudo-branch will be taken only when the non-zero register is // equal to 0, so we emit a BEQZ. // // Because only BLEU and BGEU branch on equality, we can use the // AcceptsEquality variable to decide when to emit the BEQZ. TOut.emitRRX(AcceptsEquality ? Mips::BEQ : Mips::BNE, IsSrcRegZero ? TrgReg : SrcReg, Mips::ZERO, MCOperand::createExpr(OffsetExpr), IDLoc, STI); return false; } // If we have a signed pseudo-branch and one of the registers is $0, // we can use an appropriate compare-to-zero branch. We select which one // to use in the switch statement above. TOut.emitRX(IsSrcRegZero ? ZeroSrcOpcode : ZeroTrgOpcode, IsSrcRegZero ? TrgReg : SrcReg, MCOperand::createExpr(OffsetExpr), IDLoc, STI); return false; } // If neither the SrcReg nor the TrgReg are $0, we need AT to perform the // expansions. If it is not available, we return. unsigned ATRegNum = getATReg(IDLoc); if (!ATRegNum) return true; if (!EmittedNoMacroWarning) warnIfNoMacro(IDLoc); // SLT fits well with 2 of our 4 pseudo-branches: // BLT, where $rs < $rt, translates into "slt $at, $rs, $rt" and // BGT, where $rs > $rt, translates into "slt $at, $rt, $rs". // If the result of the SLT is 1, we branch, and if it's 0, we don't. // This is accomplished by using a BNEZ with the result of the SLT. // // The other 2 pseudo-branches are opposites of the above 2 (BGE with BLT // and BLE with BGT), so we change the BNEZ into a BEQZ. // Because only BGE and BLE branch on equality, we can use the // AcceptsEquality variable to decide when to emit the BEQZ. // Note that the order of the SLT arguments doesn't change between // opposites. // // The same applies to the unsigned variants, except that SLTu is used // instead of SLT. TOut.emitRRR(IsUnsigned ? Mips::SLTu : Mips::SLT, ATRegNum, ReverseOrderSLT ? TrgReg : SrcReg, ReverseOrderSLT ? SrcReg : TrgReg, IDLoc, STI); TOut.emitRRX(IsLikely ? (AcceptsEquality ? Mips::BEQL : Mips::BNEL) : (AcceptsEquality ? Mips::BEQ : Mips::BNE), ATRegNum, Mips::ZERO, MCOperand::createExpr(OffsetExpr), IDLoc, STI); return false; } // Expand a integer division macro. // // Notably we don't have to emit a warning when encountering $rt as the $zero // register, or 0 as an immediate. processInstruction() has already done that. // // The destination register can only be $zero when expanding (S)DivIMacro or // D(S)DivMacro. bool MipsAsmParser::expandDivRem(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI, const bool IsMips64, const bool Signed) { MipsTargetStreamer &TOut = getTargetStreamer(); warnIfNoMacro(IDLoc); const MCOperand &RdRegOp = Inst.getOperand(0); assert(RdRegOp.isReg() && "expected register operand kind"); unsigned RdReg = RdRegOp.getReg(); const MCOperand &RsRegOp = Inst.getOperand(1); assert(RsRegOp.isReg() && "expected register operand kind"); unsigned RsReg = RsRegOp.getReg(); unsigned RtReg; int64_t ImmValue; const MCOperand &RtOp = Inst.getOperand(2); assert((RtOp.isReg() || RtOp.isImm()) && "expected register or immediate operand kind"); if (RtOp.isReg()) RtReg = RtOp.getReg(); else ImmValue = RtOp.getImm(); unsigned DivOp; unsigned ZeroReg; unsigned SubOp; if (IsMips64) { DivOp = Signed ? Mips::DSDIV : Mips::DUDIV; ZeroReg = Mips::ZERO_64; SubOp = Mips::DSUB; } else { DivOp = Signed ? Mips::SDIV : Mips::UDIV; ZeroReg = Mips::ZERO; SubOp = Mips::SUB; } bool UseTraps = useTraps(); unsigned Opcode = Inst.getOpcode(); bool isDiv = Opcode == Mips::SDivMacro || Opcode == Mips::SDivIMacro || Opcode == Mips::UDivMacro || Opcode == Mips::UDivIMacro || Opcode == Mips::DSDivMacro || Opcode == Mips::DSDivIMacro || Opcode == Mips::DUDivMacro || Opcode == Mips::DUDivIMacro; bool isRem = Opcode == Mips::SRemMacro || Opcode == Mips::SRemIMacro || Opcode == Mips::URemMacro || Opcode == Mips::URemIMacro || Opcode == Mips::DSRemMacro || Opcode == Mips::DSRemIMacro || Opcode == Mips::DURemMacro || Opcode == Mips::DURemIMacro; if (RtOp.isImm()) { unsigned ATReg = getATReg(IDLoc); if (!ATReg) return true; if (ImmValue == 0) { if (UseTraps) TOut.emitRRI(Mips::TEQ, ZeroReg, ZeroReg, 0x7, IDLoc, STI); else TOut.emitII(Mips::BREAK, 0x7, 0, IDLoc, STI); return false; } if (isRem && (ImmValue == 1 || (Signed && (ImmValue == -1)))) { TOut.emitRRR(Mips::OR, RdReg, ZeroReg, ZeroReg, IDLoc, STI); return false; } else if (isDiv && ImmValue == 1) { TOut.emitRRR(Mips::OR, RdReg, RsReg, Mips::ZERO, IDLoc, STI); return false; } else if (isDiv && Signed && ImmValue == -1) { TOut.emitRRR(SubOp, RdReg, ZeroReg, RsReg, IDLoc, STI); return false; } else { if (loadImmediate(ImmValue, ATReg, Mips::NoRegister, isInt<32>(ImmValue), false, Inst.getLoc(), Out, STI)) return true; TOut.emitRR(DivOp, RsReg, ATReg, IDLoc, STI); TOut.emitR(isDiv ? Mips::MFLO : Mips::MFHI, RdReg, IDLoc, STI); return false; } return true; } // If the macro expansion of (d)div(u) or (d)rem(u) would always trap or // break, insert the trap/break and exit. This gives a different result to // GAS. GAS has an inconsistency/missed optimization in that not all cases // are handled equivalently. As the observed behaviour is the same, we're ok. if (RtReg == Mips::ZERO || RtReg == Mips::ZERO_64) { if (UseTraps) { TOut.emitRRI(Mips::TEQ, ZeroReg, ZeroReg, 0x7, IDLoc, STI); return false; } TOut.emitII(Mips::BREAK, 0x7, 0, IDLoc, STI); return false; } // (d)rem(u) $0, $X, $Y is a special case. Like div $zero, $X, $Y, it does // not expand to macro sequence. if (isRem && (RdReg == Mips::ZERO || RdReg == Mips::ZERO_64)) { TOut.emitRR(DivOp, RsReg, RtReg, IDLoc, STI); return false; } // Temporary label for first branch traget MCContext &Context = TOut.getStreamer().getContext(); MCSymbol *BrTarget; MCOperand LabelOp; if (UseTraps) { TOut.emitRRI(Mips::TEQ, RtReg, ZeroReg, 0x7, IDLoc, STI); } else { // Branch to the li instruction. BrTarget = Context.createTempSymbol(); LabelOp = MCOperand::createExpr(MCSymbolRefExpr::create(BrTarget, Context)); TOut.emitRRX(Mips::BNE, RtReg, ZeroReg, LabelOp, IDLoc, STI); } TOut.emitRR(DivOp, RsReg, RtReg, IDLoc, STI); if (!UseTraps) TOut.emitII(Mips::BREAK, 0x7, 0, IDLoc, STI); if (!Signed) { if (!UseTraps) TOut.getStreamer().EmitLabel(BrTarget); TOut.emitR(isDiv ? Mips::MFLO : Mips::MFHI, RdReg, IDLoc, STI); return false; } unsigned ATReg = getATReg(IDLoc); if (!ATReg) return true; if (!UseTraps) TOut.getStreamer().EmitLabel(BrTarget); TOut.emitRRI(Mips::ADDiu, ATReg, ZeroReg, -1, IDLoc, STI); // Temporary label for the second branch target. MCSymbol *BrTargetEnd = Context.createTempSymbol(); MCOperand LabelOpEnd = MCOperand::createExpr(MCSymbolRefExpr::create(BrTargetEnd, Context)); // Branch to the mflo instruction. TOut.emitRRX(Mips::BNE, RtReg, ATReg, LabelOpEnd, IDLoc, STI); if (IsMips64) { TOut.emitRRI(Mips::ADDiu, ATReg, ZeroReg, 1, IDLoc, STI); TOut.emitDSLL(ATReg, ATReg, 63, IDLoc, STI); } else { TOut.emitRI(Mips::LUi, ATReg, (uint16_t)0x8000, IDLoc, STI); } if (UseTraps) TOut.emitRRI(Mips::TEQ, RsReg, ATReg, 0x6, IDLoc, STI); else { // Branch to the mflo instruction. TOut.emitRRX(Mips::BNE, RsReg, ATReg, LabelOpEnd, IDLoc, STI); TOut.emitNop(IDLoc, STI); TOut.emitII(Mips::BREAK, 0x6, 0, IDLoc, STI); } TOut.getStreamer().EmitLabel(BrTargetEnd); TOut.emitR(isDiv ? Mips::MFLO : Mips::MFHI, RdReg, IDLoc, STI); return false; } bool MipsAsmParser::expandTrunc(MCInst &Inst, bool IsDouble, bool Is64FPU, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI) { MipsTargetStreamer &TOut = getTargetStreamer(); assert(Inst.getNumOperands() == 3 && "Invalid operand count"); assert(Inst.getOperand(0).isReg() && Inst.getOperand(1).isReg() && Inst.getOperand(2).isReg() && "Invalid instruction operand."); unsigned FirstReg = Inst.getOperand(0).getReg(); unsigned SecondReg = Inst.getOperand(1).getReg(); unsigned ThirdReg = Inst.getOperand(2).getReg(); if (hasMips1() && !hasMips2()) { unsigned ATReg = getATReg(IDLoc); if (!ATReg) return true; TOut.emitRR(Mips::CFC1, ThirdReg, Mips::RA, IDLoc, STI); TOut.emitRR(Mips::CFC1, ThirdReg, Mips::RA, IDLoc, STI); TOut.emitNop(IDLoc, STI); TOut.emitRRI(Mips::ORi, ATReg, ThirdReg, 0x3, IDLoc, STI); TOut.emitRRI(Mips::XORi, ATReg, ATReg, 0x2, IDLoc, STI); TOut.emitRR(Mips::CTC1, Mips::RA, ATReg, IDLoc, STI); TOut.emitNop(IDLoc, STI); TOut.emitRR(IsDouble ? (Is64FPU ? Mips::CVT_W_D64 : Mips::CVT_W_D32) : Mips::CVT_W_S, FirstReg, SecondReg, IDLoc, STI); TOut.emitRR(Mips::CTC1, Mips::RA, ThirdReg, IDLoc, STI); TOut.emitNop(IDLoc, STI); return false; } TOut.emitRR(IsDouble ? (Is64FPU ? Mips::TRUNC_W_D64 : Mips::TRUNC_W_D32) : Mips::TRUNC_W_S, FirstReg, SecondReg, IDLoc, STI); return false; } bool MipsAsmParser::expandUlh(MCInst &Inst, bool Signed, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI) { if (hasMips32r6() || hasMips64r6()) { return Error(IDLoc, "instruction not supported on mips32r6 or mips64r6"); } const MCOperand &DstRegOp = Inst.getOperand(0); assert(DstRegOp.isReg() && "expected register operand kind"); const MCOperand &SrcRegOp = Inst.getOperand(1); assert(SrcRegOp.isReg() && "expected register operand kind"); const MCOperand &OffsetImmOp = Inst.getOperand(2); assert(OffsetImmOp.isImm() && "expected immediate operand kind"); MipsTargetStreamer &TOut = getTargetStreamer(); unsigned DstReg = DstRegOp.getReg(); unsigned SrcReg = SrcRegOp.getReg(); int64_t OffsetValue = OffsetImmOp.getImm(); // NOTE: We always need AT for ULHU, as it is always used as the source // register for one of the LBu's. warnIfNoMacro(IDLoc); unsigned ATReg = getATReg(IDLoc); if (!ATReg) return true; bool IsLargeOffset = !(isInt<16>(OffsetValue + 1) && isInt<16>(OffsetValue)); if (IsLargeOffset) { if (loadImmediate(OffsetValue, ATReg, SrcReg, !ABI.ArePtrs64bit(), true, IDLoc, Out, STI)) return true; } int64_t FirstOffset = IsLargeOffset ? 0 : OffsetValue; int64_t SecondOffset = IsLargeOffset ? 1 : (OffsetValue + 1); if (isLittle()) std::swap(FirstOffset, SecondOffset); unsigned FirstLbuDstReg = IsLargeOffset ? DstReg : ATReg; unsigned SecondLbuDstReg = IsLargeOffset ? ATReg : DstReg; unsigned LbuSrcReg = IsLargeOffset ? ATReg : SrcReg; unsigned SllReg = IsLargeOffset ? DstReg : ATReg; TOut.emitRRI(Signed ? Mips::LB : Mips::LBu, FirstLbuDstReg, LbuSrcReg, FirstOffset, IDLoc, STI); TOut.emitRRI(Mips::LBu, SecondLbuDstReg, LbuSrcReg, SecondOffset, IDLoc, STI); TOut.emitRRI(Mips::SLL, SllReg, SllReg, 8, IDLoc, STI); TOut.emitRRR(Mips::OR, DstReg, DstReg, ATReg, IDLoc, STI); return false; } bool MipsAsmParser::expandUsh(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI) { if (hasMips32r6() || hasMips64r6()) { return Error(IDLoc, "instruction not supported on mips32r6 or mips64r6"); } const MCOperand &DstRegOp = Inst.getOperand(0); assert(DstRegOp.isReg() && "expected register operand kind"); const MCOperand &SrcRegOp = Inst.getOperand(1); assert(SrcRegOp.isReg() && "expected register operand kind"); const MCOperand &OffsetImmOp = Inst.getOperand(2); assert(OffsetImmOp.isImm() && "expected immediate operand kind"); MipsTargetStreamer &TOut = getTargetStreamer(); unsigned DstReg = DstRegOp.getReg(); unsigned SrcReg = SrcRegOp.getReg(); int64_t OffsetValue = OffsetImmOp.getImm(); warnIfNoMacro(IDLoc); unsigned ATReg = getATReg(IDLoc); if (!ATReg) return true; bool IsLargeOffset = !(isInt<16>(OffsetValue + 1) && isInt<16>(OffsetValue)); if (IsLargeOffset) { if (loadImmediate(OffsetValue, ATReg, SrcReg, !ABI.ArePtrs64bit(), true, IDLoc, Out, STI)) return true; } int64_t FirstOffset = IsLargeOffset ? 1 : (OffsetValue + 1); int64_t SecondOffset = IsLargeOffset ? 0 : OffsetValue; if (isLittle()) std::swap(FirstOffset, SecondOffset); if (IsLargeOffset) { TOut.emitRRI(Mips::SB, DstReg, ATReg, FirstOffset, IDLoc, STI); TOut.emitRRI(Mips::SRL, DstReg, DstReg, 8, IDLoc, STI); TOut.emitRRI(Mips::SB, DstReg, ATReg, SecondOffset, IDLoc, STI); TOut.emitRRI(Mips::LBu, ATReg, ATReg, 0, IDLoc, STI); TOut.emitRRI(Mips::SLL, DstReg, DstReg, 8, IDLoc, STI); TOut.emitRRR(Mips::OR, DstReg, DstReg, ATReg, IDLoc, STI); } else { TOut.emitRRI(Mips::SB, DstReg, SrcReg, FirstOffset, IDLoc, STI); TOut.emitRRI(Mips::SRL, ATReg, DstReg, 8, IDLoc, STI); TOut.emitRRI(Mips::SB, ATReg, SrcReg, SecondOffset, IDLoc, STI); } return false; } bool MipsAsmParser::expandUxw(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI) { if (hasMips32r6() || hasMips64r6()) { return Error(IDLoc, "instruction not supported on mips32r6 or mips64r6"); } const MCOperand &DstRegOp = Inst.getOperand(0); assert(DstRegOp.isReg() && "expected register operand kind"); const MCOperand &SrcRegOp = Inst.getOperand(1); assert(SrcRegOp.isReg() && "expected register operand kind"); const MCOperand &OffsetImmOp = Inst.getOperand(2); assert(OffsetImmOp.isImm() && "expected immediate operand kind"); MipsTargetStreamer &TOut = getTargetStreamer(); unsigned DstReg = DstRegOp.getReg(); unsigned SrcReg = SrcRegOp.getReg(); int64_t OffsetValue = OffsetImmOp.getImm(); // Compute left/right load/store offsets. bool IsLargeOffset = !(isInt<16>(OffsetValue + 3) && isInt<16>(OffsetValue)); int64_t LxlOffset = IsLargeOffset ? 0 : OffsetValue; int64_t LxrOffset = IsLargeOffset ? 3 : (OffsetValue + 3); if (isLittle()) std::swap(LxlOffset, LxrOffset); bool IsLoadInst = (Inst.getOpcode() == Mips::Ulw); bool DoMove = IsLoadInst && (SrcReg == DstReg) && !IsLargeOffset; unsigned TmpReg = SrcReg; if (IsLargeOffset || DoMove) { warnIfNoMacro(IDLoc); TmpReg = getATReg(IDLoc); if (!TmpReg) return true; } if (IsLargeOffset) { if (loadImmediate(OffsetValue, TmpReg, SrcReg, !ABI.ArePtrs64bit(), true, IDLoc, Out, STI)) return true; } if (DoMove) std::swap(DstReg, TmpReg); unsigned XWL = IsLoadInst ? Mips::LWL : Mips::SWL; unsigned XWR = IsLoadInst ? Mips::LWR : Mips::SWR; TOut.emitRRI(XWL, DstReg, TmpReg, LxlOffset, IDLoc, STI); TOut.emitRRI(XWR, DstReg, TmpReg, LxrOffset, IDLoc, STI); if (DoMove) TOut.emitRRR(Mips::OR, TmpReg, DstReg, Mips::ZERO, IDLoc, STI); return false; } bool MipsAsmParser::expandAliasImmediate(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI) { MipsTargetStreamer &TOut = getTargetStreamer(); assert(Inst.getNumOperands() == 3 && "Invalid operand count"); assert(Inst.getOperand(0).isReg() && Inst.getOperand(1).isReg() && Inst.getOperand(2).isImm() && "Invalid instruction operand."); unsigned ATReg = Mips::NoRegister; unsigned FinalDstReg = Mips::NoRegister; unsigned DstReg = Inst.getOperand(0).getReg(); unsigned SrcReg = Inst.getOperand(1).getReg(); int64_t ImmValue = Inst.getOperand(2).getImm(); bool Is32Bit = isInt<32>(ImmValue) || (!isGP64bit() && isUInt<32>(ImmValue)); unsigned FinalOpcode = Inst.getOpcode(); if (DstReg == SrcReg) { ATReg = getATReg(Inst.getLoc()); if (!ATReg) return true; FinalDstReg = DstReg; DstReg = ATReg; } if (!loadImmediate(ImmValue, DstReg, Mips::NoRegister, Is32Bit, false, Inst.getLoc(), Out, STI)) { switch (FinalOpcode) { default: llvm_unreachable("unimplemented expansion"); case Mips::ADDi: FinalOpcode = Mips::ADD; break; case Mips::ADDiu: FinalOpcode = Mips::ADDu; break; case Mips::ANDi: FinalOpcode = Mips::AND; break; case Mips::NORImm: FinalOpcode = Mips::NOR; break; case Mips::ORi: FinalOpcode = Mips::OR; break; case Mips::SLTi: FinalOpcode = Mips::SLT; break; case Mips::SLTiu: FinalOpcode = Mips::SLTu; break; case Mips::XORi: FinalOpcode = Mips::XOR; break; case Mips::ADDi_MM: FinalOpcode = Mips::ADD_MM; break; case Mips::ADDiu_MM: FinalOpcode = Mips::ADDu_MM; break; case Mips::ANDi_MM: FinalOpcode = Mips::AND_MM; break; case Mips::ORi_MM: FinalOpcode = Mips::OR_MM; break; case Mips::SLTi_MM: FinalOpcode = Mips::SLT_MM; break; case Mips::SLTiu_MM: FinalOpcode = Mips::SLTu_MM; break; case Mips::XORi_MM: FinalOpcode = Mips::XOR_MM; break; case Mips::ANDi64: FinalOpcode = Mips::AND64; break; case Mips::NORImm64: FinalOpcode = Mips::NOR64; break; case Mips::ORi64: FinalOpcode = Mips::OR64; break; case Mips::SLTImm64: FinalOpcode = Mips::SLT64; break; case Mips::SLTUImm64: FinalOpcode = Mips::SLTu64; break; case Mips::XORi64: FinalOpcode = Mips::XOR64; break; } if (FinalDstReg == Mips::NoRegister) TOut.emitRRR(FinalOpcode, DstReg, DstReg, SrcReg, IDLoc, STI); else TOut.emitRRR(FinalOpcode, FinalDstReg, FinalDstReg, DstReg, IDLoc, STI); return false; } return true; } bool MipsAsmParser::expandRotation(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI) { MipsTargetStreamer &TOut = getTargetStreamer(); unsigned ATReg = Mips::NoRegister; unsigned DReg = Inst.getOperand(0).getReg(); unsigned SReg = Inst.getOperand(1).getReg(); unsigned TReg = Inst.getOperand(2).getReg(); unsigned TmpReg = DReg; unsigned FirstShift = Mips::NOP; unsigned SecondShift = Mips::NOP; if (hasMips32r2()) { if (DReg == SReg) { TmpReg = getATReg(Inst.getLoc()); if (!TmpReg) return true; } if (Inst.getOpcode() == Mips::ROL) { TOut.emitRRR(Mips::SUBu, TmpReg, Mips::ZERO, TReg, Inst.getLoc(), STI); TOut.emitRRR(Mips::ROTRV, DReg, SReg, TmpReg, Inst.getLoc(), STI); return false; } if (Inst.getOpcode() == Mips::ROR) { TOut.emitRRR(Mips::ROTRV, DReg, SReg, TReg, Inst.getLoc(), STI); return false; } return true; } if (hasMips32()) { switch (Inst.getOpcode()) { default: llvm_unreachable("unexpected instruction opcode"); case Mips::ROL: FirstShift = Mips::SRLV; SecondShift = Mips::SLLV; break; case Mips::ROR: FirstShift = Mips::SLLV; SecondShift = Mips::SRLV; break; } ATReg = getATReg(Inst.getLoc()); if (!ATReg) return true; TOut.emitRRR(Mips::SUBu, ATReg, Mips::ZERO, TReg, Inst.getLoc(), STI); TOut.emitRRR(FirstShift, ATReg, SReg, ATReg, Inst.getLoc(), STI); TOut.emitRRR(SecondShift, DReg, SReg, TReg, Inst.getLoc(), STI); TOut.emitRRR(Mips::OR, DReg, DReg, ATReg, Inst.getLoc(), STI); return false; } return true; } bool MipsAsmParser::expandRotationImm(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI) { MipsTargetStreamer &TOut = getTargetStreamer(); unsigned ATReg = Mips::NoRegister; unsigned DReg = Inst.getOperand(0).getReg(); unsigned SReg = Inst.getOperand(1).getReg(); int64_t ImmValue = Inst.getOperand(2).getImm(); unsigned FirstShift = Mips::NOP; unsigned SecondShift = Mips::NOP; if (hasMips32r2()) { if (Inst.getOpcode() == Mips::ROLImm) { uint64_t MaxShift = 32; uint64_t ShiftValue = ImmValue; if (ImmValue != 0) ShiftValue = MaxShift - ImmValue; TOut.emitRRI(Mips::ROTR, DReg, SReg, ShiftValue, Inst.getLoc(), STI); return false; } if (Inst.getOpcode() == Mips::RORImm) { TOut.emitRRI(Mips::ROTR, DReg, SReg, ImmValue, Inst.getLoc(), STI); return false; } return true; } if (hasMips32()) { if (ImmValue == 0) { TOut.emitRRI(Mips::SRL, DReg, SReg, 0, Inst.getLoc(), STI); return false; } switch (Inst.getOpcode()) { default: llvm_unreachable("unexpected instruction opcode"); case Mips::ROLImm: FirstShift = Mips::SLL; SecondShift = Mips::SRL; break; case Mips::RORImm: FirstShift = Mips::SRL; SecondShift = Mips::SLL; break; } ATReg = getATReg(Inst.getLoc()); if (!ATReg) return true; TOut.emitRRI(FirstShift, ATReg, SReg, ImmValue, Inst.getLoc(), STI); TOut.emitRRI(SecondShift, DReg, SReg, 32 - ImmValue, Inst.getLoc(), STI); TOut.emitRRR(Mips::OR, DReg, DReg, ATReg, Inst.getLoc(), STI); return false; } return true; } bool MipsAsmParser::expandDRotation(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI) { MipsTargetStreamer &TOut = getTargetStreamer(); unsigned ATReg = Mips::NoRegister; unsigned DReg = Inst.getOperand(0).getReg(); unsigned SReg = Inst.getOperand(1).getReg(); unsigned TReg = Inst.getOperand(2).getReg(); unsigned TmpReg = DReg; unsigned FirstShift = Mips::NOP; unsigned SecondShift = Mips::NOP; if (hasMips64r2()) { if (TmpReg == SReg) { TmpReg = getATReg(Inst.getLoc()); if (!TmpReg) return true; } if (Inst.getOpcode() == Mips::DROL) { TOut.emitRRR(Mips::DSUBu, TmpReg, Mips::ZERO, TReg, Inst.getLoc(), STI); TOut.emitRRR(Mips::DROTRV, DReg, SReg, TmpReg, Inst.getLoc(), STI); return false; } if (Inst.getOpcode() == Mips::DROR) { TOut.emitRRR(Mips::DROTRV, DReg, SReg, TReg, Inst.getLoc(), STI); return false; } return true; } if (hasMips64()) { switch (Inst.getOpcode()) { default: llvm_unreachable("unexpected instruction opcode"); case Mips::DROL: FirstShift = Mips::DSRLV; SecondShift = Mips::DSLLV; break; case Mips::DROR: FirstShift = Mips::DSLLV; SecondShift = Mips::DSRLV; break; } ATReg = getATReg(Inst.getLoc()); if (!ATReg) return true; TOut.emitRRR(Mips::DSUBu, ATReg, Mips::ZERO, TReg, Inst.getLoc(), STI); TOut.emitRRR(FirstShift, ATReg, SReg, ATReg, Inst.getLoc(), STI); TOut.emitRRR(SecondShift, DReg, SReg, TReg, Inst.getLoc(), STI); TOut.emitRRR(Mips::OR, DReg, DReg, ATReg, Inst.getLoc(), STI); return false; } return true; } bool MipsAsmParser::expandDRotationImm(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI) { MipsTargetStreamer &TOut = getTargetStreamer(); unsigned ATReg = Mips::NoRegister; unsigned DReg = Inst.getOperand(0).getReg(); unsigned SReg = Inst.getOperand(1).getReg(); int64_t ImmValue = Inst.getOperand(2).getImm() % 64; unsigned FirstShift = Mips::NOP; unsigned SecondShift = Mips::NOP; MCInst TmpInst; if (hasMips64r2()) { unsigned FinalOpcode = Mips::NOP; if (ImmValue == 0) FinalOpcode = Mips::DROTR; else if (ImmValue % 32 == 0) FinalOpcode = Mips::DROTR32; else if ((ImmValue >= 1) && (ImmValue <= 32)) { if (Inst.getOpcode() == Mips::DROLImm) FinalOpcode = Mips::DROTR32; else FinalOpcode = Mips::DROTR; } else if (ImmValue >= 33) { if (Inst.getOpcode() == Mips::DROLImm) FinalOpcode = Mips::DROTR; else FinalOpcode = Mips::DROTR32; } uint64_t ShiftValue = ImmValue % 32; if (Inst.getOpcode() == Mips::DROLImm) ShiftValue = (32 - ImmValue % 32) % 32; TOut.emitRRI(FinalOpcode, DReg, SReg, ShiftValue, Inst.getLoc(), STI); return false; } if (hasMips64()) { if (ImmValue == 0) { TOut.emitRRI(Mips::DSRL, DReg, SReg, 0, Inst.getLoc(), STI); return false; } switch (Inst.getOpcode()) { default: llvm_unreachable("unexpected instruction opcode"); case Mips::DROLImm: if ((ImmValue >= 1) && (ImmValue <= 31)) { FirstShift = Mips::DSLL; SecondShift = Mips::DSRL32; } if (ImmValue == 32) { FirstShift = Mips::DSLL32; SecondShift = Mips::DSRL32; } if ((ImmValue >= 33) && (ImmValue <= 63)) { FirstShift = Mips::DSLL32; SecondShift = Mips::DSRL; } break; case Mips::DRORImm: if ((ImmValue >= 1) && (ImmValue <= 31)) { FirstShift = Mips::DSRL; SecondShift = Mips::DSLL32; } if (ImmValue == 32) { FirstShift = Mips::DSRL32; SecondShift = Mips::DSLL32; } if ((ImmValue >= 33) && (ImmValue <= 63)) { FirstShift = Mips::DSRL32; SecondShift = Mips::DSLL; } break; } ATReg = getATReg(Inst.getLoc()); if (!ATReg) return true; TOut.emitRRI(FirstShift, ATReg, SReg, ImmValue % 32, Inst.getLoc(), STI); TOut.emitRRI(SecondShift, DReg, SReg, (32 - ImmValue % 32) % 32, Inst.getLoc(), STI); TOut.emitRRR(Mips::OR, DReg, DReg, ATReg, Inst.getLoc(), STI); return false; } return true; } bool MipsAsmParser::expandAbs(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI) { MipsTargetStreamer &TOut = getTargetStreamer(); unsigned FirstRegOp = Inst.getOperand(0).getReg(); unsigned SecondRegOp = Inst.getOperand(1).getReg(); TOut.emitRI(Mips::BGEZ, SecondRegOp, 8, IDLoc, STI); if (FirstRegOp != SecondRegOp) TOut.emitRRR(Mips::ADDu, FirstRegOp, SecondRegOp, Mips::ZERO, IDLoc, STI); else TOut.emitEmptyDelaySlot(false, IDLoc, STI); TOut.emitRRR(Mips::SUB, FirstRegOp, Mips::ZERO, SecondRegOp, IDLoc, STI); return false; } bool MipsAsmParser::expandMulImm(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI) { MipsTargetStreamer &TOut = getTargetStreamer(); unsigned ATReg = Mips::NoRegister; unsigned DstReg = Inst.getOperand(0).getReg(); unsigned SrcReg = Inst.getOperand(1).getReg(); int32_t ImmValue = Inst.getOperand(2).getImm(); ATReg = getATReg(IDLoc); if (!ATReg) return true; loadImmediate(ImmValue, ATReg, Mips::NoRegister, true, false, IDLoc, Out, STI); TOut.emitRR(Inst.getOpcode() == Mips::MULImmMacro ? Mips::MULT : Mips::DMULT, SrcReg, ATReg, IDLoc, STI); TOut.emitR(Mips::MFLO, DstReg, IDLoc, STI); return false; } bool MipsAsmParser::expandMulO(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI) { MipsTargetStreamer &TOut = getTargetStreamer(); unsigned ATReg = Mips::NoRegister; unsigned DstReg = Inst.getOperand(0).getReg(); unsigned SrcReg = Inst.getOperand(1).getReg(); unsigned TmpReg = Inst.getOperand(2).getReg(); ATReg = getATReg(Inst.getLoc()); if (!ATReg) return true; TOut.emitRR(Inst.getOpcode() == Mips::MULOMacro ? Mips::MULT : Mips::DMULT, SrcReg, TmpReg, IDLoc, STI); TOut.emitR(Mips::MFLO, DstReg, IDLoc, STI); TOut.emitRRI(Inst.getOpcode() == Mips::MULOMacro ? Mips::SRA : Mips::DSRA32, DstReg, DstReg, 0x1F, IDLoc, STI); TOut.emitR(Mips::MFHI, ATReg, IDLoc, STI); if (useTraps()) { TOut.emitRRI(Mips::TNE, DstReg, ATReg, 6, IDLoc, STI); } else { MCContext & Context = TOut.getStreamer().getContext(); MCSymbol * BrTarget = Context.createTempSymbol(); MCOperand LabelOp = MCOperand::createExpr(MCSymbolRefExpr::create(BrTarget, Context)); TOut.emitRRX(Mips::BEQ, DstReg, ATReg, LabelOp, IDLoc, STI); if (AssemblerOptions.back()->isReorder()) TOut.emitNop(IDLoc, STI); TOut.emitII(Mips::BREAK, 6, 0, IDLoc, STI); TOut.getStreamer().EmitLabel(BrTarget); } TOut.emitR(Mips::MFLO, DstReg, IDLoc, STI); return false; } bool MipsAsmParser::expandMulOU(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI) { MipsTargetStreamer &TOut = getTargetStreamer(); unsigned ATReg = Mips::NoRegister; unsigned DstReg = Inst.getOperand(0).getReg(); unsigned SrcReg = Inst.getOperand(1).getReg(); unsigned TmpReg = Inst.getOperand(2).getReg(); ATReg = getATReg(IDLoc); if (!ATReg) return true; TOut.emitRR(Inst.getOpcode() == Mips::MULOUMacro ? Mips::MULTu : Mips::DMULTu, SrcReg, TmpReg, IDLoc, STI); TOut.emitR(Mips::MFHI, ATReg, IDLoc, STI); TOut.emitR(Mips::MFLO, DstReg, IDLoc, STI); if (useTraps()) { TOut.emitRRI(Mips::TNE, ATReg, Mips::ZERO, 6, IDLoc, STI); } else { MCContext & Context = TOut.getStreamer().getContext(); MCSymbol * BrTarget = Context.createTempSymbol(); MCOperand LabelOp = MCOperand::createExpr(MCSymbolRefExpr::create(BrTarget, Context)); TOut.emitRRX(Mips::BEQ, ATReg, Mips::ZERO, LabelOp, IDLoc, STI); if (AssemblerOptions.back()->isReorder()) TOut.emitNop(IDLoc, STI); TOut.emitII(Mips::BREAK, 6, 0, IDLoc, STI); TOut.getStreamer().EmitLabel(BrTarget); } return false; } bool MipsAsmParser::expandDMULMacro(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI) { MipsTargetStreamer &TOut = getTargetStreamer(); unsigned DstReg = Inst.getOperand(0).getReg(); unsigned SrcReg = Inst.getOperand(1).getReg(); unsigned TmpReg = Inst.getOperand(2).getReg(); TOut.emitRR(Mips::DMULTu, SrcReg, TmpReg, IDLoc, STI); TOut.emitR(Mips::MFLO, DstReg, IDLoc, STI); return false; } // Expand 'ld $ offset($reg2)' to 'lw $, offset($reg2); // lw $>, offset+4($reg2)' // or expand 'sd $ offset($reg2)' to 'sw $, offset($reg2); // sw $>, offset+4($reg2)' // for O32. bool MipsAsmParser::expandLoadStoreDMacro(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI, bool IsLoad) { if (!isABI_O32()) return true; warnIfNoMacro(IDLoc); MipsTargetStreamer &TOut = getTargetStreamer(); unsigned Opcode = IsLoad ? Mips::LW : Mips::SW; unsigned FirstReg = Inst.getOperand(0).getReg(); unsigned SecondReg = nextReg(FirstReg); unsigned BaseReg = Inst.getOperand(1).getReg(); if (!SecondReg) return true; warnIfRegIndexIsAT(FirstReg, IDLoc); assert(Inst.getOperand(2).isImm() && "Offset for load macro is not immediate!"); MCOperand &FirstOffset = Inst.getOperand(2); signed NextOffset = FirstOffset.getImm() + 4; MCOperand SecondOffset = MCOperand::createImm(NextOffset); if (!isInt<16>(FirstOffset.getImm()) || !isInt<16>(NextOffset)) return true; // For loads, clobber the base register with the second load instead of the // first if the BaseReg == FirstReg. if (FirstReg != BaseReg || !IsLoad) { TOut.emitRRX(Opcode, FirstReg, BaseReg, FirstOffset, IDLoc, STI); TOut.emitRRX(Opcode, SecondReg, BaseReg, SecondOffset, IDLoc, STI); } else { TOut.emitRRX(Opcode, SecondReg, BaseReg, SecondOffset, IDLoc, STI); TOut.emitRRX(Opcode, FirstReg, BaseReg, FirstOffset, IDLoc, STI); } return false; } bool MipsAsmParser::expandSeq(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI) { warnIfNoMacro(IDLoc); MipsTargetStreamer &TOut = getTargetStreamer(); if (Inst.getOperand(1).getReg() != Mips::ZERO && Inst.getOperand(2).getReg() != Mips::ZERO) { TOut.emitRRR(Mips::XOR, Inst.getOperand(0).getReg(), Inst.getOperand(1).getReg(), Inst.getOperand(2).getReg(), IDLoc, STI); TOut.emitRRI(Mips::SLTiu, Inst.getOperand(0).getReg(), Inst.getOperand(0).getReg(), 1, IDLoc, STI); return false; } unsigned Reg = 0; if (Inst.getOperand(1).getReg() == Mips::ZERO) { Reg = Inst.getOperand(2).getReg(); } else { Reg = Inst.getOperand(1).getReg(); } TOut.emitRRI(Mips::SLTiu, Inst.getOperand(0).getReg(), Reg, 1, IDLoc, STI); return false; } bool MipsAsmParser::expandSeqI(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI) { warnIfNoMacro(IDLoc); MipsTargetStreamer &TOut = getTargetStreamer(); unsigned Opc; int64_t Imm = Inst.getOperand(2).getImm(); unsigned Reg = Inst.getOperand(1).getReg(); if (Imm == 0) { TOut.emitRRI(Mips::SLTiu, Inst.getOperand(0).getReg(), Inst.getOperand(1).getReg(), 1, IDLoc, STI); return false; } else { if (Reg == Mips::ZERO) { Warning(IDLoc, "comparison is always false"); TOut.emitRRR(isGP64bit() ? Mips::DADDu : Mips::ADDu, Inst.getOperand(0).getReg(), Reg, Reg, IDLoc, STI); return false; } if (Imm > -0x8000 && Imm < 0) { Imm = -Imm; Opc = isGP64bit() ? Mips::DADDiu : Mips::ADDiu; } else { Opc = Mips::XORi; } } if (!isUInt<16>(Imm)) { unsigned ATReg = getATReg(IDLoc); if (!ATReg) return true; if (loadImmediate(Imm, ATReg, Mips::NoRegister, true, isGP64bit(), IDLoc, Out, STI)) return true; TOut.emitRRR(Mips::XOR, Inst.getOperand(0).getReg(), Inst.getOperand(1).getReg(), ATReg, IDLoc, STI); TOut.emitRRI(Mips::SLTiu, Inst.getOperand(0).getReg(), Inst.getOperand(0).getReg(), 1, IDLoc, STI); return false; } TOut.emitRRI(Opc, Inst.getOperand(0).getReg(), Inst.getOperand(1).getReg(), Imm, IDLoc, STI); TOut.emitRRI(Mips::SLTiu, Inst.getOperand(0).getReg(), Inst.getOperand(0).getReg(), 1, IDLoc, STI); return false; } // Map the DSP accumulator and control register to the corresponding gpr // operand. Unlike the other alias, the m(f|t)t(lo|hi|acx) instructions // do not map the DSP registers contigously to gpr registers. static unsigned getRegisterForMxtrDSP(MCInst &Inst, bool IsMFDSP) { switch (Inst.getOpcode()) { case Mips::MFTLO: case Mips::MTTLO: switch (Inst.getOperand(IsMFDSP ? 1 : 0).getReg()) { case Mips::AC0: return Mips::ZERO; case Mips::AC1: return Mips::A0; case Mips::AC2: return Mips::T0; case Mips::AC3: return Mips::T4; default: llvm_unreachable("Unknown register for 'mttr' alias!"); } case Mips::MFTHI: case Mips::MTTHI: switch (Inst.getOperand(IsMFDSP ? 1 : 0).getReg()) { case Mips::AC0: return Mips::AT; case Mips::AC1: return Mips::A1; case Mips::AC2: return Mips::T1; case Mips::AC3: return Mips::T5; default: llvm_unreachable("Unknown register for 'mttr' alias!"); } case Mips::MFTACX: case Mips::MTTACX: switch (Inst.getOperand(IsMFDSP ? 1 : 0).getReg()) { case Mips::AC0: return Mips::V0; case Mips::AC1: return Mips::A2; case Mips::AC2: return Mips::T2; case Mips::AC3: return Mips::T6; default: llvm_unreachable("Unknown register for 'mttr' alias!"); } case Mips::MFTDSP: case Mips::MTTDSP: return Mips::S0; default: llvm_unreachable("Unknown instruction for 'mttr' dsp alias!"); } } // Map the floating point register operand to the corresponding register // operand. static unsigned getRegisterForMxtrFP(MCInst &Inst, bool IsMFTC1) { switch (Inst.getOperand(IsMFTC1 ? 1 : 0).getReg()) { case Mips::F0: return Mips::ZERO; case Mips::F1: return Mips::AT; case Mips::F2: return Mips::V0; case Mips::F3: return Mips::V1; case Mips::F4: return Mips::A0; case Mips::F5: return Mips::A1; case Mips::F6: return Mips::A2; case Mips::F7: return Mips::A3; case Mips::F8: return Mips::T0; case Mips::F9: return Mips::T1; case Mips::F10: return Mips::T2; case Mips::F11: return Mips::T3; case Mips::F12: return Mips::T4; case Mips::F13: return Mips::T5; case Mips::F14: return Mips::T6; case Mips::F15: return Mips::T7; case Mips::F16: return Mips::S0; case Mips::F17: return Mips::S1; case Mips::F18: return Mips::S2; case Mips::F19: return Mips::S3; case Mips::F20: return Mips::S4; case Mips::F21: return Mips::S5; case Mips::F22: return Mips::S6; case Mips::F23: return Mips::S7; case Mips::F24: return Mips::T8; case Mips::F25: return Mips::T9; case Mips::F26: return Mips::K0; case Mips::F27: return Mips::K1; case Mips::F28: return Mips::GP; case Mips::F29: return Mips::SP; case Mips::F30: return Mips::FP; case Mips::F31: return Mips::RA; default: llvm_unreachable("Unknown register for mttc1 alias!"); } } // Map the coprocessor operand the corresponding gpr register operand. static unsigned getRegisterForMxtrC0(MCInst &Inst, bool IsMFTC0) { switch (Inst.getOperand(IsMFTC0 ? 1 : 0).getReg()) { case Mips::COP00: return Mips::ZERO; case Mips::COP01: return Mips::AT; case Mips::COP02: return Mips::V0; case Mips::COP03: return Mips::V1; case Mips::COP04: return Mips::A0; case Mips::COP05: return Mips::A1; case Mips::COP06: return Mips::A2; case Mips::COP07: return Mips::A3; case Mips::COP08: return Mips::T0; case Mips::COP09: return Mips::T1; case Mips::COP010: return Mips::T2; case Mips::COP011: return Mips::T3; case Mips::COP012: return Mips::T4; case Mips::COP013: return Mips::T5; case Mips::COP014: return Mips::T6; case Mips::COP015: return Mips::T7; case Mips::COP016: return Mips::S0; case Mips::COP017: return Mips::S1; case Mips::COP018: return Mips::S2; case Mips::COP019: return Mips::S3; case Mips::COP020: return Mips::S4; case Mips::COP021: return Mips::S5; case Mips::COP022: return Mips::S6; case Mips::COP023: return Mips::S7; case Mips::COP024: return Mips::T8; case Mips::COP025: return Mips::T9; case Mips::COP026: return Mips::K0; case Mips::COP027: return Mips::K1; case Mips::COP028: return Mips::GP; case Mips::COP029: return Mips::SP; case Mips::COP030: return Mips::FP; case Mips::COP031: return Mips::RA; default: llvm_unreachable("Unknown register for mttc0 alias!"); } } /// Expand an alias of 'mftr' or 'mttr' into the full instruction, by producing /// an mftr or mttr with the correctly mapped gpr register, u, sel and h bits. bool MipsAsmParser::expandMXTRAlias(MCInst &Inst, SMLoc IDLoc, MCStreamer &Out, const MCSubtargetInfo *STI) { MipsTargetStreamer &TOut = getTargetStreamer(); unsigned rd = 0; unsigned u = 1; unsigned sel = 0; unsigned h = 0; bool IsMFTR = false; switch (Inst.getOpcode()) { case Mips::MFTC0: IsMFTR = true; LLVM_FALLTHROUGH; case Mips::MTTC0: u = 0; rd = getRegisterForMxtrC0(Inst, IsMFTR); sel = Inst.getOperand(2).getImm(); break; case Mips::MFTGPR: IsMFTR = true; LLVM_FALLTHROUGH; case Mips::MTTGPR: rd = Inst.getOperand(IsMFTR ? 1 : 0).getReg(); break; case Mips::MFTLO: case Mips::MFTHI: case Mips::MFTACX: case Mips::MFTDSP: IsMFTR = true; LLVM_FALLTHROUGH; case Mips::MTTLO: case Mips::MTTHI: case Mips::MTTACX: case Mips::MTTDSP: rd = getRegisterForMxtrDSP(Inst, IsMFTR); sel = 1; break; case Mips::MFTHC1: h = 1; LLVM_FALLTHROUGH; case Mips::MFTC1: IsMFTR = true; rd = getRegisterForMxtrFP(Inst, IsMFTR); sel = 2; break; case Mips::MTTHC1: h = 1; LLVM_FALLTHROUGH; case Mips::MTTC1: rd = getRegisterForMxtrFP(Inst, IsMFTR); sel = 2; break; case Mips::CFTC1: IsMFTR = true; LLVM_FALLTHROUGH; case Mips::CTTC1: rd = getRegisterForMxtrFP(Inst, IsMFTR); sel = 3; break; } unsigned Op0 = IsMFTR ? Inst.getOperand(0).getReg() : rd; unsigned Op1 = IsMFTR ? rd : (Inst.getOpcode() != Mips::MTTDSP ? Inst.getOperand(1).getReg() : Inst.getOperand(0).getReg()); TOut.emitRRIII(IsMFTR ? Mips::MFTR : Mips::MTTR, Op0, Op1, u, sel, h, IDLoc, STI); return false; } unsigned MipsAsmParser::checkEarlyTargetMatchPredicate(MCInst &Inst, const OperandVector &Operands) { switch (Inst.getOpcode()) { default: return Match_Success; case Mips::DATI: case Mips::DAHI: if (static_cast(*Operands[1]) .isValidForTie(static_cast(*Operands[2]))) return Match_Success; return Match_RequiresSameSrcAndDst; } } unsigned MipsAsmParser::checkTargetMatchPredicate(MCInst &Inst) { switch (Inst.getOpcode()) { // As described by the MIPSR6 spec, daui must not use the zero operand for // its source operand. case Mips::DAUI: if (Inst.getOperand(1).getReg() == Mips::ZERO || Inst.getOperand(1).getReg() == Mips::ZERO_64) return Match_RequiresNoZeroRegister; return Match_Success; // As described by the Mips32r2 spec, the registers Rd and Rs for // jalr.hb must be different. // It also applies for registers Rt and Rs of microMIPSr6 jalrc.hb instruction // and registers Rd and Base for microMIPS lwp instruction case Mips::JALR_HB: case Mips::JALR_HB64: case Mips::JALRC_HB_MMR6: case Mips::JALRC_MMR6: if (Inst.getOperand(0).getReg() == Inst.getOperand(1).getReg()) return Match_RequiresDifferentSrcAndDst; return Match_Success; case Mips::LWP_MM: if (Inst.getOperand(0).getReg() == Inst.getOperand(2).getReg()) return Match_RequiresDifferentSrcAndDst; return Match_Success; case Mips::SYNC: if (Inst.getOperand(0).getImm() != 0 && !hasMips32()) return Match_NonZeroOperandForSync; return Match_Success; case Mips::MFC0: case Mips::MTC0: case Mips::MTC2: case Mips::MFC2: if (Inst.getOperand(2).getImm() != 0 && !hasMips32()) return Match_NonZeroOperandForMTCX; return Match_Success; // As described the MIPSR6 spec, the compact branches that compare registers // must: // a) Not use the zero register. // b) Not use the same register twice. // c) rs < rt for bnec, beqc. // NB: For this case, the encoding will swap the operands as their // ordering doesn't matter. GAS performs this transformation too. // Hence, that constraint does not have to be enforced. // // The compact branches that branch iff the signed addition of two registers // would overflow must have rs >= rt. That can be handled like beqc/bnec with // operand swapping. They do not have restriction of using the zero register. case Mips::BLEZC: case Mips::BLEZC_MMR6: case Mips::BGEZC: case Mips::BGEZC_MMR6: case Mips::BGTZC: case Mips::BGTZC_MMR6: case Mips::BLTZC: case Mips::BLTZC_MMR6: case Mips::BEQZC: case Mips::BEQZC_MMR6: case Mips::BNEZC: case Mips::BNEZC_MMR6: case Mips::BLEZC64: case Mips::BGEZC64: case Mips::BGTZC64: case Mips::BLTZC64: case Mips::BEQZC64: case Mips::BNEZC64: if (Inst.getOperand(0).getReg() == Mips::ZERO || Inst.getOperand(0).getReg() == Mips::ZERO_64) return Match_RequiresNoZeroRegister; return Match_Success; case Mips::BGEC: case Mips::BGEC_MMR6: case Mips::BLTC: case Mips::BLTC_MMR6: case Mips::BGEUC: case Mips::BGEUC_MMR6: case Mips::BLTUC: case Mips::BLTUC_MMR6: case Mips::BEQC: case Mips::BEQC_MMR6: case Mips::BNEC: case Mips::BNEC_MMR6: case Mips::BGEC64: case Mips::BLTC64: case Mips::BGEUC64: case Mips::BLTUC64: case Mips::BEQC64: case Mips::BNEC64: if (Inst.getOperand(0).getReg() == Mips::ZERO || Inst.getOperand(0).getReg() == Mips::ZERO_64) return Match_RequiresNoZeroRegister; if (Inst.getOperand(1).getReg() == Mips::ZERO || Inst.getOperand(1).getReg() == Mips::ZERO_64) return Match_RequiresNoZeroRegister; if (Inst.getOperand(0).getReg() == Inst.getOperand(1).getReg()) return Match_RequiresDifferentOperands; return Match_Success; case Mips::DINS: { assert(Inst.getOperand(2).isImm() && Inst.getOperand(3).isImm() && "Operands must be immediates for dins!"); const signed Pos = Inst.getOperand(2).getImm(); const signed Size = Inst.getOperand(3).getImm(); if ((0 > (Pos + Size)) || ((Pos + Size) > 32)) return Match_RequiresPosSizeRange0_32; return Match_Success; } case Mips::DINSM: case Mips::DINSU: { assert(Inst.getOperand(2).isImm() && Inst.getOperand(3).isImm() && "Operands must be immediates for dinsm/dinsu!"); const signed Pos = Inst.getOperand(2).getImm(); const signed Size = Inst.getOperand(3).getImm(); if ((32 >= (Pos + Size)) || ((Pos + Size) > 64)) return Match_RequiresPosSizeRange33_64; return Match_Success; } case Mips::DEXT: { assert(Inst.getOperand(2).isImm() && Inst.getOperand(3).isImm() && "Operands must be immediates for DEXTM!"); const signed Pos = Inst.getOperand(2).getImm(); const signed Size = Inst.getOperand(3).getImm(); if ((1 > (Pos + Size)) || ((Pos + Size) > 63)) return Match_RequiresPosSizeUImm6; return Match_Success; } case Mips::DEXTM: case Mips::DEXTU: { assert(Inst.getOperand(2).isImm() && Inst.getOperand(3).isImm() && "Operands must be immediates for dextm/dextu!"); const signed Pos = Inst.getOperand(2).getImm(); const signed Size = Inst.getOperand(3).getImm(); if ((32 > (Pos + Size)) || ((Pos + Size) > 64)) return Match_RequiresPosSizeRange33_64; return Match_Success; } case Mips::CRC32B: case Mips::CRC32CB: case Mips::CRC32H: case Mips::CRC32CH: case Mips::CRC32W: case Mips::CRC32CW: case Mips::CRC32D: case Mips::CRC32CD: if (Inst.getOperand(0).getReg() != Inst.getOperand(2).getReg()) return Match_RequiresSameSrcAndDst; return Match_Success; } uint64_t TSFlags = getInstDesc(Inst.getOpcode()).TSFlags; if ((TSFlags & MipsII::HasFCCRegOperand) && (Inst.getOperand(0).getReg() != Mips::FCC0) && !hasEightFccRegisters()) return Match_NoFCCRegisterForCurrentISA; return Match_Success; } static SMLoc RefineErrorLoc(const SMLoc Loc, const OperandVector &Operands, uint64_t ErrorInfo) { if (ErrorInfo != ~0ULL && ErrorInfo < Operands.size()) { SMLoc ErrorLoc = Operands[ErrorInfo]->getStartLoc(); if (ErrorLoc == SMLoc()) return Loc; return ErrorLoc; } return Loc; } bool MipsAsmParser::MatchAndEmitInstruction(SMLoc IDLoc, unsigned &Opcode, OperandVector &Operands, MCStreamer &Out, uint64_t &ErrorInfo, bool MatchingInlineAsm) { MCInst Inst; unsigned MatchResult = MatchInstructionImpl(Operands, Inst, ErrorInfo, MatchingInlineAsm); switch (MatchResult) { case Match_Success: if (processInstruction(Inst, IDLoc, Out, STI)) return true; return false; case Match_MissingFeature: Error(IDLoc, "instruction requires a CPU feature not currently enabled"); return true; case Match_InvalidOperand: { SMLoc ErrorLoc = IDLoc; if (ErrorInfo != ~0ULL) { if (ErrorInfo >= Operands.size()) return Error(IDLoc, "too few operands for instruction"); ErrorLoc = Operands[ErrorInfo]->getStartLoc(); if (ErrorLoc == SMLoc()) ErrorLoc = IDLoc; } return Error(ErrorLoc, "invalid operand for instruction"); } case Match_NonZeroOperandForSync: return Error(IDLoc, "s-type must be zero or unspecified for pre-MIPS32 ISAs"); case Match_NonZeroOperandForMTCX: return Error(IDLoc, "selector must be zero for pre-MIPS32 ISAs"); case Match_MnemonicFail: return Error(IDLoc, "invalid instruction"); case Match_RequiresDifferentSrcAndDst: return Error(IDLoc, "source and destination must be different"); case Match_RequiresDifferentOperands: return Error(IDLoc, "registers must be different"); case Match_RequiresNoZeroRegister: return Error(IDLoc, "invalid operand ($zero) for instruction"); case Match_RequiresSameSrcAndDst: return Error(IDLoc, "source and destination must match"); case Match_NoFCCRegisterForCurrentISA: return Error(RefineErrorLoc(IDLoc, Operands, ErrorInfo), "non-zero fcc register doesn't exist in current ISA level"); case Match_Immz: return Error(RefineErrorLoc(IDLoc, Operands, ErrorInfo), "expected '0'"); case Match_UImm1_0: return Error(RefineErrorLoc(IDLoc, Operands, ErrorInfo), "expected 1-bit unsigned immediate"); case Match_UImm2_0: return Error(RefineErrorLoc(IDLoc, Operands, ErrorInfo), "expected 2-bit unsigned immediate"); case Match_UImm2_1: return Error(RefineErrorLoc(IDLoc, Operands, ErrorInfo), "expected immediate in range 1 .. 4"); case Match_UImm3_0: return Error(RefineErrorLoc(IDLoc, Operands, ErrorInfo), "expected 3-bit unsigned immediate"); case Match_UImm4_0: return Error(RefineErrorLoc(IDLoc, Operands, ErrorInfo), "expected 4-bit unsigned immediate"); case Match_SImm4_0: return Error(RefineErrorLoc(IDLoc, Operands, ErrorInfo), "expected 4-bit signed immediate"); case Match_UImm5_0: return Error(RefineErrorLoc(IDLoc, Operands, ErrorInfo), "expected 5-bit unsigned immediate"); case Match_SImm5_0: return Error(RefineErrorLoc(IDLoc, Operands, ErrorInfo), "expected 5-bit signed immediate"); case Match_UImm5_1: return Error(RefineErrorLoc(IDLoc, Operands, ErrorInfo), "expected immediate in range 1 .. 32"); case Match_UImm5_32: return Error(RefineErrorLoc(IDLoc, Operands, ErrorInfo), "expected immediate in range 32 .. 63"); case Match_UImm5_33: return Error(RefineErrorLoc(IDLoc, Operands, ErrorInfo), "expected immediate in range 33 .. 64"); case Match_UImm5_0_Report_UImm6: // This is used on UImm5 operands that have a corresponding UImm5_32 // operand to avoid confusing the user. return Error(RefineErrorLoc(IDLoc, Operands, ErrorInfo), "expected 6-bit unsigned immediate"); case Match_UImm5_Lsl2: return Error(RefineErrorLoc(IDLoc, Operands, ErrorInfo), "expected both 7-bit unsigned immediate and multiple of 4"); case Match_UImmRange2_64: return Error(RefineErrorLoc(IDLoc, Operands, ErrorInfo), "expected immediate in range 2 .. 64"); case Match_UImm6_0: return Error(RefineErrorLoc(IDLoc, Operands, ErrorInfo), "expected 6-bit unsigned immediate"); case Match_UImm6_Lsl2: return Error(RefineErrorLoc(IDLoc, Operands, ErrorInfo), "expected both 8-bit unsigned immediate and multiple of 4"); case Match_SImm6_0: return Error(RefineErrorLoc(IDLoc, Operands, ErrorInfo), "expected 6-bit signed immediate"); case Match_UImm7_0: return Error(RefineErrorLoc(IDLoc, Operands, ErrorInfo), "expected 7-bit unsigned immediate"); case Match_UImm7_N1: return Error(RefineErrorLoc(IDLoc, Operands, ErrorInfo), "expected immediate in range -1 .. 126"); case Match_SImm7_Lsl2: return Error(RefineErrorLoc(IDLoc, Operands, ErrorInfo), "expected both 9-bit signed immediate and multiple of 4"); case Match_UImm8_0: return Error(RefineErrorLoc(IDLoc, Operands, ErrorInfo), "expected 8-bit unsigned immediate"); case Match_UImm10_0: return Error(RefineErrorLoc(IDLoc, Operands, ErrorInfo), "expected 10-bit unsigned immediate"); case Match_SImm10_0: return Error(RefineErrorLoc(IDLoc, Operands, ErrorInfo), "expected 10-bit signed immediate"); case Match_SImm11_0: return Error(RefineErrorLoc(IDLoc, Operands, ErrorInfo), "expected 11-bit signed immediate"); case Match_UImm16: case Match_UImm16_Relaxed: case Match_UImm16_AltRelaxed: return Error(RefineErrorLoc(IDLoc, Operands, ErrorInfo), "expected 16-bit unsigned immediate"); case Match_SImm16: case Match_SImm16_Relaxed: return Error(RefineErrorLoc(IDLoc, Operands, ErrorInfo), "expected 16-bit signed immediate"); case Match_SImm19_Lsl2: return Error(RefineErrorLoc(IDLoc, Operands, ErrorInfo), "expected both 19-bit signed immediate and multiple of 4"); case Match_UImm20_0: return Error(RefineErrorLoc(IDLoc, Operands, ErrorInfo), "expected 20-bit unsigned immediate"); case Match_UImm26_0: return Error(RefineErrorLoc(IDLoc, Operands, ErrorInfo), "expected 26-bit unsigned immediate"); case Match_SImm32: case Match_SImm32_Relaxed: return Error(RefineErrorLoc(IDLoc, Operands, ErrorInfo), "expected 32-bit signed immediate"); case Match_UImm32_Coerced: return Error(RefineErrorLoc(IDLoc, Operands, ErrorInfo), "expected 32-bit immediate"); case Match_MemSImm9: return Error(RefineErrorLoc(IDLoc, Operands, ErrorInfo), "expected memory with 9-bit signed offset"); case Match_MemSImm10: return Error(RefineErrorLoc(IDLoc, Operands, ErrorInfo), "expected memory with 10-bit signed offset"); case Match_MemSImm10Lsl1: return Error(RefineErrorLoc(IDLoc, Operands, ErrorInfo), "expected memory with 11-bit signed offset and multiple of 2"); case Match_MemSImm10Lsl2: return Error(RefineErrorLoc(IDLoc, Operands, ErrorInfo), "expected memory with 12-bit signed offset and multiple of 4"); case Match_MemSImm10Lsl3: return Error(RefineErrorLoc(IDLoc, Operands, ErrorInfo), "expected memory with 13-bit signed offset and multiple of 8"); case Match_MemSImm11: return Error(RefineErrorLoc(IDLoc, Operands, ErrorInfo), "expected memory with 11-bit signed offset"); case Match_MemSImm12: return Error(RefineErrorLoc(IDLoc, Operands, ErrorInfo), "expected memory with 12-bit signed offset"); case Match_MemSImm16: return Error(RefineErrorLoc(IDLoc, Operands, ErrorInfo), "expected memory with 16-bit signed offset"); case Match_MemSImmPtr: return Error(RefineErrorLoc(IDLoc, Operands, ErrorInfo), "expected memory with 32-bit signed offset"); case Match_RequiresPosSizeRange0_32: { SMLoc ErrorStart = Operands[3]->getStartLoc(); SMLoc ErrorEnd = Operands[4]->getEndLoc(); return Error(ErrorStart, "size plus position are not in the range 0 .. 32", SMRange(ErrorStart, ErrorEnd)); } case Match_RequiresPosSizeUImm6: { SMLoc ErrorStart = Operands[3]->getStartLoc(); SMLoc ErrorEnd = Operands[4]->getEndLoc(); return Error(ErrorStart, "size plus position are not in the range 1 .. 63", SMRange(ErrorStart, ErrorEnd)); } case Match_RequiresPosSizeRange33_64: { SMLoc ErrorStart = Operands[3]->getStartLoc(); SMLoc ErrorEnd = Operands[4]->getEndLoc(); return Error(ErrorStart, "size plus position are not in the range 33 .. 64", SMRange(ErrorStart, ErrorEnd)); } } llvm_unreachable("Implement any new match types added!"); } void MipsAsmParser::warnIfRegIndexIsAT(unsigned RegIndex, SMLoc Loc) { if (RegIndex != 0 && AssemblerOptions.back()->getATRegIndex() == RegIndex) Warning(Loc, "used $at (currently $" + Twine(RegIndex) + ") without \".set noat\""); } void MipsAsmParser::warnIfNoMacro(SMLoc Loc) { if (!AssemblerOptions.back()->isMacro()) Warning(Loc, "macro instruction expanded into multiple instructions"); } void MipsAsmParser::ConvertXWPOperands(MCInst &Inst, const OperandVector &Operands) { assert( (Inst.getOpcode() == Mips::LWP_MM || Inst.getOpcode() == Mips::SWP_MM) && "Unexpected instruction!"); ((MipsOperand &)*Operands[1]).addGPR32ZeroAsmRegOperands(Inst, 1); int NextReg = nextReg(((MipsOperand &)*Operands[1]).getGPR32Reg()); Inst.addOperand(MCOperand::createReg(NextReg)); ((MipsOperand &)*Operands[2]).addMemOperands(Inst, 2); } void MipsAsmParser::printWarningWithFixIt(const Twine &Msg, const Twine &FixMsg, SMRange Range, bool ShowColors) { getSourceManager().PrintMessage(Range.Start, SourceMgr::DK_Warning, Msg, Range, SMFixIt(Range, FixMsg), ShowColors); } int MipsAsmParser::matchCPURegisterName(StringRef Name) { int CC; CC = StringSwitch(Name) .Case("zero", 0) .Cases("at", "AT", 1) .Case("a0", 4) .Case("a1", 5) .Case("a2", 6) .Case("a3", 7) .Case("v0", 2) .Case("v1", 3) .Case("s0", 16) .Case("s1", 17) .Case("s2", 18) .Case("s3", 19) .Case("s4", 20) .Case("s5", 21) .Case("s6", 22) .Case("s7", 23) .Case("k0", 26) .Case("k1", 27) .Case("gp", 28) .Case("sp", 29) .Case("fp", 30) .Case("s8", 30) .Case("ra", 31) .Case("t0", 8) .Case("t1", 9) .Case("t2", 10) .Case("t3", 11) .Case("t4", 12) .Case("t5", 13) .Case("t6", 14) .Case("t7", 15) .Case("t8", 24) .Case("t9", 25) .Default(-1); if (!(isABI_N32() || isABI_N64())) return CC; if (12 <= CC && CC <= 15) { // Name is one of t4-t7 AsmToken RegTok = getLexer().peekTok(); SMRange RegRange = RegTok.getLocRange(); StringRef FixedName = StringSwitch(Name) .Case("t4", "t0") .Case("t5", "t1") .Case("t6", "t2") .Case("t7", "t3") .Default(""); assert(FixedName != "" && "Register name is not one of t4-t7."); printWarningWithFixIt("register names $t4-$t7 are only available in O32.", "Did you mean $" + FixedName + "?", RegRange); } // Although SGI documentation just cuts out t0-t3 for n32/n64, // GNU pushes the values of t0-t3 to override the o32/o64 values for t4-t7 // We are supporting both cases, so for t0-t3 we'll just push them to t4-t7. if (8 <= CC && CC <= 11) CC += 4; if (CC == -1) CC = StringSwitch(Name) .Case("a4", 8) .Case("a5", 9) .Case("a6", 10) .Case("a7", 11) .Case("kt0", 26) .Case("kt1", 27) .Default(-1); return CC; } int MipsAsmParser::matchHWRegsRegisterName(StringRef Name) { int CC; CC = StringSwitch(Name) .Case("hwr_cpunum", 0) .Case("hwr_synci_step", 1) .Case("hwr_cc", 2) .Case("hwr_ccres", 3) .Case("hwr_ulr", 29) .Default(-1); return CC; } int MipsAsmParser::matchFPURegisterName(StringRef Name) { if (Name[0] == 'f') { StringRef NumString = Name.substr(1); unsigned IntVal; if (NumString.getAsInteger(10, IntVal)) return -1; // This is not an integer. if (IntVal > 31) // Maximum index for fpu register. return -1; return IntVal; } return -1; } int MipsAsmParser::matchFCCRegisterName(StringRef Name) { if (Name.startswith("fcc")) { StringRef NumString = Name.substr(3); unsigned IntVal; if (NumString.getAsInteger(10, IntVal)) return -1; // This is not an integer. if (IntVal > 7) // There are only 8 fcc registers. return -1; return IntVal; } return -1; } int MipsAsmParser::matchACRegisterName(StringRef Name) { if (Name.startswith("ac")) { StringRef NumString = Name.substr(2); unsigned IntVal; if (NumString.getAsInteger(10, IntVal)) return -1; // This is not an integer. if (IntVal > 3) // There are only 3 acc registers. return -1; return IntVal; } return -1; } int MipsAsmParser::matchMSA128RegisterName(StringRef Name) { unsigned IntVal; if (Name.front() != 'w' || Name.drop_front(1).getAsInteger(10, IntVal)) return -1; if (IntVal > 31) return -1; return IntVal; } int MipsAsmParser::matchMSA128CtrlRegisterName(StringRef Name) { int CC; CC = StringSwitch(Name) .Case("msair", 0) .Case("msacsr", 1) .Case("msaaccess", 2) .Case("msasave", 3) .Case("msamodify", 4) .Case("msarequest", 5) .Case("msamap", 6) .Case("msaunmap", 7) .Default(-1); return CC; } bool MipsAsmParser::canUseATReg() { return AssemblerOptions.back()->getATRegIndex() != 0; } unsigned MipsAsmParser::getATReg(SMLoc Loc) { unsigned ATIndex = AssemblerOptions.back()->getATRegIndex(); if (ATIndex == 0) { reportParseError(Loc, "pseudo-instruction requires $at, which is not available"); return 0; } unsigned AT = getReg( (isGP64bit()) ? Mips::GPR64RegClassID : Mips::GPR32RegClassID, ATIndex); return AT; } unsigned MipsAsmParser::getReg(int RC, int RegNo) { return *(getContext().getRegisterInfo()->getRegClass(RC).begin() + RegNo); } bool MipsAsmParser::parseOperand(OperandVector &Operands, StringRef Mnemonic) { MCAsmParser &Parser = getParser(); LLVM_DEBUG(dbgs() << "parseOperand\n"); // Check if the current operand has a custom associated parser, if so, try to // custom parse the operand, or fallback to the general approach. OperandMatchResultTy ResTy = MatchOperandParserImpl(Operands, Mnemonic); if (ResTy == MatchOperand_Success) return false; // If there wasn't a custom match, try the generic matcher below. Otherwise, // there was a match, but an error occurred, in which case, just return that // the operand parsing failed. if (ResTy == MatchOperand_ParseFail) return true; LLVM_DEBUG(dbgs() << ".. Generic Parser\n"); switch (getLexer().getKind()) { case AsmToken::Dollar: { // Parse the register. SMLoc S = Parser.getTok().getLoc(); // Almost all registers have been parsed by custom parsers. There is only // one exception to this. $zero (and it's alias $0) will reach this point // for div, divu, and similar instructions because it is not an operand // to the instruction definition but an explicit register. Special case // this situation for now. if (parseAnyRegister(Operands) != MatchOperand_NoMatch) return false; // Maybe it is a symbol reference. StringRef Identifier; if (Parser.parseIdentifier(Identifier)) return true; SMLoc E = SMLoc::getFromPointer(Parser.getTok().getLoc().getPointer() - 1); MCSymbol *Sym = getContext().getOrCreateSymbol("$" + Identifier); // Otherwise create a symbol reference. const MCExpr *Res = MCSymbolRefExpr::create(Sym, MCSymbolRefExpr::VK_None, getContext()); Operands.push_back(MipsOperand::CreateImm(Res, S, E, *this)); return false; } default: { LLVM_DEBUG(dbgs() << ".. generic integer expression\n"); const MCExpr *Expr; SMLoc S = Parser.getTok().getLoc(); // Start location of the operand. if (getParser().parseExpression(Expr)) return true; SMLoc E = SMLoc::getFromPointer(Parser.getTok().getLoc().getPointer() - 1); Operands.push_back(MipsOperand::CreateImm(Expr, S, E, *this)); return false; } } // switch(getLexer().getKind()) return true; } bool MipsAsmParser::isEvaluated(const MCExpr *Expr) { switch (Expr->getKind()) { case MCExpr::Constant: return true; case MCExpr::SymbolRef: return (cast(Expr)->getKind() != MCSymbolRefExpr::VK_None); case MCExpr::Binary: { const MCBinaryExpr *BE = cast(Expr); if (!isEvaluated(BE->getLHS())) return false; return isEvaluated(BE->getRHS()); } case MCExpr::Unary: return isEvaluated(cast(Expr)->getSubExpr()); case MCExpr::Target: return true; } return false; } bool MipsAsmParser::ParseRegister(unsigned &RegNo, SMLoc &StartLoc, SMLoc &EndLoc) { SmallVector, 1> Operands; OperandMatchResultTy ResTy = parseAnyRegister(Operands); if (ResTy == MatchOperand_Success) { assert(Operands.size() == 1); MipsOperand &Operand = static_cast(*Operands.front()); StartLoc = Operand.getStartLoc(); EndLoc = Operand.getEndLoc(); // AFAIK, we only support numeric registers and named GPR's in CFI // directives. // Don't worry about eating tokens before failing. Using an unrecognised // register is a parse error. if (Operand.isGPRAsmReg()) { // Resolve to GPR32 or GPR64 appropriately. RegNo = isGP64bit() ? Operand.getGPR64Reg() : Operand.getGPR32Reg(); } return (RegNo == (unsigned)-1); } assert(Operands.size() == 0); return (RegNo == (unsigned)-1); } bool MipsAsmParser::parseMemOffset(const MCExpr *&Res, bool isParenExpr) { SMLoc S; if (isParenExpr) return getParser().parseParenExprOfDepth(0, Res, S); return getParser().parseExpression(Res); } OperandMatchResultTy MipsAsmParser::parseMemOperand(OperandVector &Operands) { MCAsmParser &Parser = getParser(); LLVM_DEBUG(dbgs() << "parseMemOperand\n"); const MCExpr *IdVal = nullptr; SMLoc S; bool isParenExpr = false; OperandMatchResultTy Res = MatchOperand_NoMatch; // First operand is the offset. S = Parser.getTok().getLoc(); if (getLexer().getKind() == AsmToken::LParen) { Parser.Lex(); isParenExpr = true; } if (getLexer().getKind() != AsmToken::Dollar) { if (parseMemOffset(IdVal, isParenExpr)) return MatchOperand_ParseFail; const AsmToken &Tok = Parser.getTok(); // Get the next token. if (Tok.isNot(AsmToken::LParen)) { MipsOperand &Mnemonic = static_cast(*Operands[0]); if (Mnemonic.getToken() == "la" || Mnemonic.getToken() == "dla") { SMLoc E = SMLoc::getFromPointer(Parser.getTok().getLoc().getPointer() - 1); Operands.push_back(MipsOperand::CreateImm(IdVal, S, E, *this)); return MatchOperand_Success; } if (Tok.is(AsmToken::EndOfStatement)) { SMLoc E = SMLoc::getFromPointer(Parser.getTok().getLoc().getPointer() - 1); // Zero register assumed, add a memory operand with ZERO as its base. // "Base" will be managed by k_Memory. auto Base = MipsOperand::createGPRReg( 0, "0", getContext().getRegisterInfo(), S, E, *this); Operands.push_back( MipsOperand::CreateMem(std::move(Base), IdVal, S, E, *this)); return MatchOperand_Success; } MCBinaryExpr::Opcode Opcode; // GAS and LLVM treat comparison operators different. GAS will generate -1 // or 0, while LLVM will generate 0 or 1. Since a comparsion operator is // highly unlikely to be found in a memory offset expression, we don't // handle them. switch (Tok.getKind()) { case AsmToken::Plus: Opcode = MCBinaryExpr::Add; Parser.Lex(); break; case AsmToken::Minus: Opcode = MCBinaryExpr::Sub; Parser.Lex(); break; case AsmToken::Star: Opcode = MCBinaryExpr::Mul; Parser.Lex(); break; case AsmToken::Pipe: Opcode = MCBinaryExpr::Or; Parser.Lex(); break; case AsmToken::Amp: Opcode = MCBinaryExpr::And; Parser.Lex(); break; case AsmToken::LessLess: Opcode = MCBinaryExpr::Shl; Parser.Lex(); break; case AsmToken::GreaterGreater: Opcode = MCBinaryExpr::LShr; Parser.Lex(); break; case AsmToken::Caret: Opcode = MCBinaryExpr::Xor; Parser.Lex(); break; case AsmToken::Slash: Opcode = MCBinaryExpr::Div; Parser.Lex(); break; case AsmToken::Percent: Opcode = MCBinaryExpr::Mod; Parser.Lex(); break; default: Error(Parser.getTok().getLoc(), "'(' or expression expected"); return MatchOperand_ParseFail; } const MCExpr * NextExpr; if (getParser().parseExpression(NextExpr)) return MatchOperand_ParseFail; IdVal = MCBinaryExpr::create(Opcode, IdVal, NextExpr, getContext()); } Parser.Lex(); // Eat the '(' token. } Res = parseAnyRegister(Operands); if (Res != MatchOperand_Success) return Res; if (Parser.getTok().isNot(AsmToken::RParen)) { Error(Parser.getTok().getLoc(), "')' expected"); return MatchOperand_ParseFail; } SMLoc E = SMLoc::getFromPointer(Parser.getTok().getLoc().getPointer() - 1); Parser.Lex(); // Eat the ')' token. if (!IdVal) IdVal = MCConstantExpr::create(0, getContext()); // Replace the register operand with the memory operand. std::unique_ptr op( static_cast(Operands.back().release())); // Remove the register from the operands. // "op" will be managed by k_Memory. Operands.pop_back(); // Add the memory operand. if (const MCBinaryExpr *BE = dyn_cast(IdVal)) { int64_t Imm; if (IdVal->evaluateAsAbsolute(Imm)) IdVal = MCConstantExpr::create(Imm, getContext()); else if (BE->getLHS()->getKind() != MCExpr::SymbolRef) IdVal = MCBinaryExpr::create(BE->getOpcode(), BE->getRHS(), BE->getLHS(), getContext()); } Operands.push_back(MipsOperand::CreateMem(std::move(op), IdVal, S, E, *this)); return MatchOperand_Success; } bool MipsAsmParser::searchSymbolAlias(OperandVector &Operands) { MCAsmParser &Parser = getParser(); MCSymbol *Sym = getContext().lookupSymbol(Parser.getTok().getIdentifier()); if (!Sym) return false; SMLoc S = Parser.getTok().getLoc(); if (Sym->isVariable()) { const MCExpr *Expr = Sym->getVariableValue(); if (Expr->getKind() == MCExpr::SymbolRef) { const MCSymbolRefExpr *Ref = static_cast(Expr); StringRef DefSymbol = Ref->getSymbol().getName(); if (DefSymbol.startswith("$")) { OperandMatchResultTy ResTy = matchAnyRegisterNameWithoutDollar(Operands, DefSymbol.substr(1), S); if (ResTy == MatchOperand_Success) { Parser.Lex(); return true; } if (ResTy == MatchOperand_ParseFail) llvm_unreachable("Should never ParseFail"); } } } else if (Sym->isUnset()) { // If symbol is unset, it might be created in the `parseSetAssignment` // routine as an alias for a numeric register name. // Lookup in the aliases list. auto Entry = RegisterSets.find(Sym->getName()); if (Entry != RegisterSets.end()) { OperandMatchResultTy ResTy = matchAnyRegisterWithoutDollar(Operands, Entry->getValue(), S); if (ResTy == MatchOperand_Success) { Parser.Lex(); return true; } } } return false; } OperandMatchResultTy MipsAsmParser::matchAnyRegisterNameWithoutDollar(OperandVector &Operands, StringRef Identifier, SMLoc S) { int Index = matchCPURegisterName(Identifier); if (Index != -1) { Operands.push_back(MipsOperand::createGPRReg( Index, Identifier, getContext().getRegisterInfo(), S, getLexer().getLoc(), *this)); return MatchOperand_Success; } Index = matchHWRegsRegisterName(Identifier); if (Index != -1) { Operands.push_back(MipsOperand::createHWRegsReg( Index, Identifier, getContext().getRegisterInfo(), S, getLexer().getLoc(), *this)); return MatchOperand_Success; } Index = matchFPURegisterName(Identifier); if (Index != -1) { Operands.push_back(MipsOperand::createFGRReg( Index, Identifier, getContext().getRegisterInfo(), S, getLexer().getLoc(), *this)); return MatchOperand_Success; } Index = matchFCCRegisterName(Identifier); if (Index != -1) { Operands.push_back(MipsOperand::createFCCReg( Index, Identifier, getContext().getRegisterInfo(), S, getLexer().getLoc(), *this)); return MatchOperand_Success; } Index = matchACRegisterName(Identifier); if (Index != -1) { Operands.push_back(MipsOperand::createACCReg( Index, Identifier, getContext().getRegisterInfo(), S, getLexer().getLoc(), *this)); return MatchOperand_Success; } Index = matchMSA128RegisterName(Identifier); if (Index != -1) { Operands.push_back(MipsOperand::createMSA128Reg( Index, Identifier, getContext().getRegisterInfo(), S, getLexer().getLoc(), *this)); return MatchOperand_Success; } Index = matchMSA128CtrlRegisterName(Identifier); if (Index != -1) { Operands.push_back(MipsOperand::createMSACtrlReg( Index, Identifier, getContext().getRegisterInfo(), S, getLexer().getLoc(), *this)); return MatchOperand_Success; } return MatchOperand_NoMatch; } OperandMatchResultTy MipsAsmParser::matchAnyRegisterWithoutDollar(OperandVector &Operands, const AsmToken &Token, SMLoc S) { if (Token.is(AsmToken::Identifier)) { LLVM_DEBUG(dbgs() << ".. identifier\n"); StringRef Identifier = Token.getIdentifier(); OperandMatchResultTy ResTy = matchAnyRegisterNameWithoutDollar(Operands, Identifier, S); return ResTy; } else if (Token.is(AsmToken::Integer)) { LLVM_DEBUG(dbgs() << ".. integer\n"); int64_t RegNum = Token.getIntVal(); if (RegNum < 0 || RegNum > 31) { // Show the error, but treat invalid register // number as a normal one to continue parsing // and catch other possible errors. Error(getLexer().getLoc(), "invalid register number"); } Operands.push_back(MipsOperand::createNumericReg( RegNum, Token.getString(), getContext().getRegisterInfo(), S, Token.getLoc(), *this)); return MatchOperand_Success; } LLVM_DEBUG(dbgs() << Token.getKind() << "\n"); return MatchOperand_NoMatch; } OperandMatchResultTy MipsAsmParser::matchAnyRegisterWithoutDollar(OperandVector &Operands, SMLoc S) { auto Token = getLexer().peekTok(false); return matchAnyRegisterWithoutDollar(Operands, Token, S); } OperandMatchResultTy MipsAsmParser::parseAnyRegister(OperandVector &Operands) { MCAsmParser &Parser = getParser(); LLVM_DEBUG(dbgs() << "parseAnyRegister\n"); auto Token = Parser.getTok(); SMLoc S = Token.getLoc(); if (Token.isNot(AsmToken::Dollar)) { LLVM_DEBUG(dbgs() << ".. !$ -> try sym aliasing\n"); if (Token.is(AsmToken::Identifier)) { if (searchSymbolAlias(Operands)) return MatchOperand_Success; } LLVM_DEBUG(dbgs() << ".. !symalias -> NoMatch\n"); return MatchOperand_NoMatch; } LLVM_DEBUG(dbgs() << ".. $\n"); OperandMatchResultTy ResTy = matchAnyRegisterWithoutDollar(Operands, S); if (ResTy == MatchOperand_Success) { Parser.Lex(); // $ Parser.Lex(); // identifier } return ResTy; } OperandMatchResultTy MipsAsmParser::parseJumpTarget(OperandVector &Operands) { MCAsmParser &Parser = getParser(); LLVM_DEBUG(dbgs() << "parseJumpTarget\n"); SMLoc S = getLexer().getLoc(); // Registers are a valid target and have priority over symbols. OperandMatchResultTy ResTy = parseAnyRegister(Operands); if (ResTy != MatchOperand_NoMatch) return ResTy; // Integers and expressions are acceptable const MCExpr *Expr = nullptr; if (Parser.parseExpression(Expr)) { // We have no way of knowing if a symbol was consumed so we must ParseFail return MatchOperand_ParseFail; } Operands.push_back( MipsOperand::CreateImm(Expr, S, getLexer().getLoc(), *this)); return MatchOperand_Success; } OperandMatchResultTy MipsAsmParser::parseInvNum(OperandVector &Operands) { MCAsmParser &Parser = getParser(); const MCExpr *IdVal; // If the first token is '$' we may have register operand. We have to reject // cases where it is not a register. Complicating the matter is that // register names are not reserved across all ABIs. // Peek past the dollar to see if it's a register name for this ABI. SMLoc S = Parser.getTok().getLoc(); if (Parser.getTok().is(AsmToken::Dollar)) { return matchCPURegisterName(Parser.getLexer().peekTok().getString()) == -1 ? MatchOperand_ParseFail : MatchOperand_NoMatch; } if (getParser().parseExpression(IdVal)) return MatchOperand_ParseFail; const MCConstantExpr *MCE = dyn_cast(IdVal); if (!MCE) return MatchOperand_NoMatch; int64_t Val = MCE->getValue(); SMLoc E = SMLoc::getFromPointer(Parser.getTok().getLoc().getPointer() - 1); Operands.push_back(MipsOperand::CreateImm( MCConstantExpr::create(0 - Val, getContext()), S, E, *this)); return MatchOperand_Success; } OperandMatchResultTy MipsAsmParser::parseRegisterList(OperandVector &Operands) { MCAsmParser &Parser = getParser(); SmallVector Regs; unsigned RegNo; unsigned PrevReg = Mips::NoRegister; bool RegRange = false; SmallVector, 8> TmpOperands; if (Parser.getTok().isNot(AsmToken::Dollar)) return MatchOperand_ParseFail; SMLoc S = Parser.getTok().getLoc(); while (parseAnyRegister(TmpOperands) == MatchOperand_Success) { SMLoc E = getLexer().getLoc(); MipsOperand &Reg = static_cast(*TmpOperands.back()); RegNo = isGP64bit() ? Reg.getGPR64Reg() : Reg.getGPR32Reg(); if (RegRange) { // Remove last register operand because registers from register range // should be inserted first. if ((isGP64bit() && RegNo == Mips::RA_64) || (!isGP64bit() && RegNo == Mips::RA)) { Regs.push_back(RegNo); } else { unsigned TmpReg = PrevReg + 1; while (TmpReg <= RegNo) { if ((((TmpReg < Mips::S0) || (TmpReg > Mips::S7)) && !isGP64bit()) || (((TmpReg < Mips::S0_64) || (TmpReg > Mips::S7_64)) && isGP64bit())) { Error(E, "invalid register operand"); return MatchOperand_ParseFail; } PrevReg = TmpReg; Regs.push_back(TmpReg++); } } RegRange = false; } else { if ((PrevReg == Mips::NoRegister) && ((isGP64bit() && (RegNo != Mips::S0_64) && (RegNo != Mips::RA_64)) || (!isGP64bit() && (RegNo != Mips::S0) && (RegNo != Mips::RA)))) { Error(E, "$16 or $31 expected"); return MatchOperand_ParseFail; } else if (!(((RegNo == Mips::FP || RegNo == Mips::RA || (RegNo >= Mips::S0 && RegNo <= Mips::S7)) && !isGP64bit()) || ((RegNo == Mips::FP_64 || RegNo == Mips::RA_64 || (RegNo >= Mips::S0_64 && RegNo <= Mips::S7_64)) && isGP64bit()))) { Error(E, "invalid register operand"); return MatchOperand_ParseFail; } else if ((PrevReg != Mips::NoRegister) && (RegNo != PrevReg + 1) && ((RegNo != Mips::FP && RegNo != Mips::RA && !isGP64bit()) || (RegNo != Mips::FP_64 && RegNo != Mips::RA_64 && isGP64bit()))) { Error(E, "consecutive register numbers expected"); return MatchOperand_ParseFail; } Regs.push_back(RegNo); } if (Parser.getTok().is(AsmToken::Minus)) RegRange = true; if (!Parser.getTok().isNot(AsmToken::Minus) && !Parser.getTok().isNot(AsmToken::Comma)) { Error(E, "',' or '-' expected"); return MatchOperand_ParseFail; } Lex(); // Consume comma or minus if (Parser.getTok().isNot(AsmToken::Dollar)) break; PrevReg = RegNo; } SMLoc E = Parser.getTok().getLoc(); Operands.push_back(MipsOperand::CreateRegList(Regs, S, E, *this)); parseMemOperand(Operands); return MatchOperand_Success; } /// Sometimes (i.e. load/stores) the operand may be followed immediately by /// either this. /// ::= '(', register, ')' /// handle it before we iterate so we don't get tripped up by the lack of /// a comma. bool MipsAsmParser::parseParenSuffix(StringRef Name, OperandVector &Operands) { MCAsmParser &Parser = getParser(); if (getLexer().is(AsmToken::LParen)) { Operands.push_back( MipsOperand::CreateToken("(", getLexer().getLoc(), *this)); Parser.Lex(); if (parseOperand(Operands, Name)) { SMLoc Loc = getLexer().getLoc(); return Error(Loc, "unexpected token in argument list"); } if (Parser.getTok().isNot(AsmToken::RParen)) { SMLoc Loc = getLexer().getLoc(); return Error(Loc, "unexpected token, expected ')'"); } Operands.push_back( MipsOperand::CreateToken(")", getLexer().getLoc(), *this)); Parser.Lex(); } return false; } /// Sometimes (i.e. in MSA) the operand may be followed immediately by /// either one of these. /// ::= '[', register, ']' /// ::= '[', integer, ']' /// handle it before we iterate so we don't get tripped up by the lack of /// a comma. bool MipsAsmParser::parseBracketSuffix(StringRef Name, OperandVector &Operands) { MCAsmParser &Parser = getParser(); if (getLexer().is(AsmToken::LBrac)) { Operands.push_back( MipsOperand::CreateToken("[", getLexer().getLoc(), *this)); Parser.Lex(); if (parseOperand(Operands, Name)) { SMLoc Loc = getLexer().getLoc(); return Error(Loc, "unexpected token in argument list"); } if (Parser.getTok().isNot(AsmToken::RBrac)) { SMLoc Loc = getLexer().getLoc(); return Error(Loc, "unexpected token, expected ']'"); } Operands.push_back( MipsOperand::CreateToken("]", getLexer().getLoc(), *this)); Parser.Lex(); } return false; } static std::string MipsMnemonicSpellCheck(StringRef S, uint64_t FBS, unsigned VariantID = 0); bool MipsAsmParser::ParseInstruction(ParseInstructionInfo &Info, StringRef Name, SMLoc NameLoc, OperandVector &Operands) { MCAsmParser &Parser = getParser(); LLVM_DEBUG(dbgs() << "ParseInstruction\n"); // We have reached first instruction, module directive are now forbidden. getTargetStreamer().forbidModuleDirective(); // Check if we have valid mnemonic if (!mnemonicIsValid(Name, 0)) { uint64_t FBS = ComputeAvailableFeatures(getSTI().getFeatureBits()); std::string Suggestion = MipsMnemonicSpellCheck(Name, FBS); return Error(NameLoc, "unknown instruction" + Suggestion); } // First operand in MCInst is instruction mnemonic. Operands.push_back(MipsOperand::CreateToken(Name, NameLoc, *this)); // Read the remaining operands. if (getLexer().isNot(AsmToken::EndOfStatement)) { // Read the first operand. if (parseOperand(Operands, Name)) { SMLoc Loc = getLexer().getLoc(); return Error(Loc, "unexpected token in argument list"); } if (getLexer().is(AsmToken::LBrac) && parseBracketSuffix(Name, Operands)) return true; // AFAIK, parenthesis suffixes are never on the first operand while (getLexer().is(AsmToken::Comma)) { Parser.Lex(); // Eat the comma. // Parse and remember the operand. if (parseOperand(Operands, Name)) { SMLoc Loc = getLexer().getLoc(); return Error(Loc, "unexpected token in argument list"); } // Parse bracket and parenthesis suffixes before we iterate if (getLexer().is(AsmToken::LBrac)) { if (parseBracketSuffix(Name, Operands)) return true; } else if (getLexer().is(AsmToken::LParen) && parseParenSuffix(Name, Operands)) return true; } } if (getLexer().isNot(AsmToken::EndOfStatement)) { SMLoc Loc = getLexer().getLoc(); return Error(Loc, "unexpected token in argument list"); } Parser.Lex(); // Consume the EndOfStatement. return false; } // FIXME: Given that these have the same name, these should both be // consistent on affecting the Parser. bool MipsAsmParser::reportParseError(Twine ErrorMsg) { SMLoc Loc = getLexer().getLoc(); return Error(Loc, ErrorMsg); } bool MipsAsmParser::reportParseError(SMLoc Loc, Twine ErrorMsg) { return Error(Loc, ErrorMsg); } bool MipsAsmParser::parseSetNoAtDirective() { MCAsmParser &Parser = getParser(); // Line should look like: ".set noat". // Set the $at register to $0. AssemblerOptions.back()->setATRegIndex(0); Parser.Lex(); // Eat "noat". // If this is not the end of the statement, report an error. if (getLexer().isNot(AsmToken::EndOfStatement)) { reportParseError("unexpected token, expected end of statement"); return false; } getTargetStreamer().emitDirectiveSetNoAt(); Parser.Lex(); // Consume the EndOfStatement. return false; } bool MipsAsmParser::parseSetAtDirective() { // Line can be: ".set at", which sets $at to $1 // or ".set at=$reg", which sets $at to $reg. MCAsmParser &Parser = getParser(); Parser.Lex(); // Eat "at". if (getLexer().is(AsmToken::EndOfStatement)) { // No register was specified, so we set $at to $1. AssemblerOptions.back()->setATRegIndex(1); getTargetStreamer().emitDirectiveSetAt(); Parser.Lex(); // Consume the EndOfStatement. return false; } if (getLexer().isNot(AsmToken::Equal)) { reportParseError("unexpected token, expected equals sign"); return false; } Parser.Lex(); // Eat "=". if (getLexer().isNot(AsmToken::Dollar)) { if (getLexer().is(AsmToken::EndOfStatement)) { reportParseError("no register specified"); return false; } else { reportParseError("unexpected token, expected dollar sign '$'"); return false; } } Parser.Lex(); // Eat "$". // Find out what "reg" is. unsigned AtRegNo; const AsmToken &Reg = Parser.getTok(); if (Reg.is(AsmToken::Identifier)) { AtRegNo = matchCPURegisterName(Reg.getIdentifier()); } else if (Reg.is(AsmToken::Integer)) { AtRegNo = Reg.getIntVal(); } else { reportParseError("unexpected token, expected identifier or integer"); return false; } // Check if $reg is a valid register. If it is, set $at to $reg. if (!AssemblerOptions.back()->setATRegIndex(AtRegNo)) { reportParseError("invalid register"); return false; } Parser.Lex(); // Eat "reg". // If this is not the end of the statement, report an error. if (getLexer().isNot(AsmToken::EndOfStatement)) { reportParseError("unexpected token, expected end of statement"); return false; } getTargetStreamer().emitDirectiveSetAtWithArg(AtRegNo); Parser.Lex(); // Consume the EndOfStatement. return false; } bool MipsAsmParser::parseSetReorderDirective() { MCAsmParser &Parser = getParser(); Parser.Lex(); // If this is not the end of the statement, report an error. if (getLexer().isNot(AsmToken::EndOfStatement)) { reportParseError("unexpected token, expected end of statement"); return false; } AssemblerOptions.back()->setReorder(); getTargetStreamer().emitDirectiveSetReorder(); Parser.Lex(); // Consume the EndOfStatement. return false; } bool MipsAsmParser::parseSetNoReorderDirective() { MCAsmParser &Parser = getParser(); Parser.Lex(); // If this is not the end of the statement, report an error. if (getLexer().isNot(AsmToken::EndOfStatement)) { reportParseError("unexpected token, expected end of statement"); return false; } AssemblerOptions.back()->setNoReorder(); getTargetStreamer().emitDirectiveSetNoReorder(); Parser.Lex(); // Consume the EndOfStatement. return false; } bool MipsAsmParser::parseSetMacroDirective() { MCAsmParser &Parser = getParser(); Parser.Lex(); // If this is not the end of the statement, report an error. if (getLexer().isNot(AsmToken::EndOfStatement)) { reportParseError("unexpected token, expected end of statement"); return false; } AssemblerOptions.back()->setMacro(); getTargetStreamer().emitDirectiveSetMacro(); Parser.Lex(); // Consume the EndOfStatement. return false; } bool MipsAsmParser::parseSetNoMacroDirective() { MCAsmParser &Parser = getParser(); Parser.Lex(); // If this is not the end of the statement, report an error. if (getLexer().isNot(AsmToken::EndOfStatement)) { reportParseError("unexpected token, expected end of statement"); return false; } if (AssemblerOptions.back()->isReorder()) { reportParseError("`noreorder' must be set before `nomacro'"); return false; } AssemblerOptions.back()->setNoMacro(); getTargetStreamer().emitDirectiveSetNoMacro(); Parser.Lex(); // Consume the EndOfStatement. return false; } bool MipsAsmParser::parseSetMsaDirective() { MCAsmParser &Parser = getParser(); Parser.Lex(); // If this is not the end of the statement, report an error. if (getLexer().isNot(AsmToken::EndOfStatement)) return reportParseError("unexpected token, expected end of statement"); setFeatureBits(Mips::FeatureMSA, "msa"); getTargetStreamer().emitDirectiveSetMsa(); return false; } bool MipsAsmParser::parseSetNoMsaDirective() { MCAsmParser &Parser = getParser(); Parser.Lex(); // If this is not the end of the statement, report an error. if (getLexer().isNot(AsmToken::EndOfStatement)) return reportParseError("unexpected token, expected end of statement"); clearFeatureBits(Mips::FeatureMSA, "msa"); getTargetStreamer().emitDirectiveSetNoMsa(); return false; } bool MipsAsmParser::parseSetNoDspDirective() { MCAsmParser &Parser = getParser(); Parser.Lex(); // Eat "nodsp". // If this is not the end of the statement, report an error. if (getLexer().isNot(AsmToken::EndOfStatement)) { reportParseError("unexpected token, expected end of statement"); return false; } clearFeatureBits(Mips::FeatureDSP, "dsp"); getTargetStreamer().emitDirectiveSetNoDsp(); return false; } bool MipsAsmParser::parseSetMips16Directive() { MCAsmParser &Parser = getParser(); Parser.Lex(); // Eat "mips16". // If this is not the end of the statement, report an error. if (getLexer().isNot(AsmToken::EndOfStatement)) { reportParseError("unexpected token, expected end of statement"); return false; } setFeatureBits(Mips::FeatureMips16, "mips16"); getTargetStreamer().emitDirectiveSetMips16(); Parser.Lex(); // Consume the EndOfStatement. return false; } bool MipsAsmParser::parseSetNoMips16Directive() { MCAsmParser &Parser = getParser(); Parser.Lex(); // Eat "nomips16". // If this is not the end of the statement, report an error. if (getLexer().isNot(AsmToken::EndOfStatement)) { reportParseError("unexpected token, expected end of statement"); return false; } clearFeatureBits(Mips::FeatureMips16, "mips16"); getTargetStreamer().emitDirectiveSetNoMips16(); Parser.Lex(); // Consume the EndOfStatement. return false; } bool MipsAsmParser::parseSetFpDirective() { MCAsmParser &Parser = getParser(); MipsABIFlagsSection::FpABIKind FpAbiVal; // Line can be: .set fp=32 // .set fp=xx // .set fp=64 Parser.Lex(); // Eat fp token AsmToken Tok = Parser.getTok(); if (Tok.isNot(AsmToken::Equal)) { reportParseError("unexpected token, expected equals sign '='"); return false; } Parser.Lex(); // Eat '=' token. Tok = Parser.getTok(); if (!parseFpABIValue(FpAbiVal, ".set")) return false; if (getLexer().isNot(AsmToken::EndOfStatement)) { reportParseError("unexpected token, expected end of statement"); return false; } getTargetStreamer().emitDirectiveSetFp(FpAbiVal); Parser.Lex(); // Consume the EndOfStatement. return false; } bool MipsAsmParser::parseSetOddSPRegDirective() { MCAsmParser &Parser = getParser(); Parser.Lex(); // Eat "oddspreg". if (getLexer().isNot(AsmToken::EndOfStatement)) { reportParseError("unexpected token, expected end of statement"); return false; } clearFeatureBits(Mips::FeatureNoOddSPReg, "nooddspreg"); getTargetStreamer().emitDirectiveSetOddSPReg(); return false; } bool MipsAsmParser::parseSetNoOddSPRegDirective() { MCAsmParser &Parser = getParser(); Parser.Lex(); // Eat "nooddspreg". if (getLexer().isNot(AsmToken::EndOfStatement)) { reportParseError("unexpected token, expected end of statement"); return false; } setFeatureBits(Mips::FeatureNoOddSPReg, "nooddspreg"); getTargetStreamer().emitDirectiveSetNoOddSPReg(); return false; } bool MipsAsmParser::parseSetMtDirective() { MCAsmParser &Parser = getParser(); Parser.Lex(); // Eat "mt". // If this is not the end of the statement, report an error. if (getLexer().isNot(AsmToken::EndOfStatement)) { reportParseError("unexpected token, expected end of statement"); return false; } setFeatureBits(Mips::FeatureMT, "mt"); getTargetStreamer().emitDirectiveSetMt(); Parser.Lex(); // Consume the EndOfStatement. return false; } bool MipsAsmParser::parseSetNoMtDirective() { MCAsmParser &Parser = getParser(); Parser.Lex(); // Eat "nomt". // If this is not the end of the statement, report an error. if (getLexer().isNot(AsmToken::EndOfStatement)) { reportParseError("unexpected token, expected end of statement"); return false; } clearFeatureBits(Mips::FeatureMT, "mt"); getTargetStreamer().emitDirectiveSetNoMt(); Parser.Lex(); // Consume the EndOfStatement. return false; } bool MipsAsmParser::parseSetNoCRCDirective() { MCAsmParser &Parser = getParser(); Parser.Lex(); // Eat "nocrc". // If this is not the end of the statement, report an error. if (getLexer().isNot(AsmToken::EndOfStatement)) { reportParseError("unexpected token, expected end of statement"); return false; } clearFeatureBits(Mips::FeatureCRC, "crc"); getTargetStreamer().emitDirectiveSetNoCRC(); Parser.Lex(); // Consume the EndOfStatement. return false; } bool MipsAsmParser::parseSetNoVirtDirective() { MCAsmParser &Parser = getParser(); Parser.Lex(); // Eat "novirt". // If this is not the end of the statement, report an error. if (getLexer().isNot(AsmToken::EndOfStatement)) { reportParseError("unexpected token, expected end of statement"); return false; } clearFeatureBits(Mips::FeatureVirt, "virt"); getTargetStreamer().emitDirectiveSetNoVirt(); Parser.Lex(); // Consume the EndOfStatement. return false; } bool MipsAsmParser::parseSetNoGINVDirective() { MCAsmParser &Parser = getParser(); Parser.Lex(); // Eat "noginv". // If this is not the end of the statement, report an error. if (getLexer().isNot(AsmToken::EndOfStatement)) { reportParseError("unexpected token, expected end of statement"); return false; } clearFeatureBits(Mips::FeatureGINV, "ginv"); getTargetStreamer().emitDirectiveSetNoGINV(); Parser.Lex(); // Consume the EndOfStatement. return false; } bool MipsAsmParser::parseSetPopDirective() { MCAsmParser &Parser = getParser(); SMLoc Loc = getLexer().getLoc(); Parser.Lex(); if (getLexer().isNot(AsmToken::EndOfStatement)) return reportParseError("unexpected token, expected end of statement"); // Always keep an element on the options "stack" to prevent the user // from changing the initial options. This is how we remember them. if (AssemblerOptions.size() == 2) return reportParseError(Loc, ".set pop with no .set push"); MCSubtargetInfo &STI = copySTI(); AssemblerOptions.pop_back(); setAvailableFeatures( ComputeAvailableFeatures(AssemblerOptions.back()->getFeatures())); STI.setFeatureBits(AssemblerOptions.back()->getFeatures()); getTargetStreamer().emitDirectiveSetPop(); return false; } bool MipsAsmParser::parseSetPushDirective() { MCAsmParser &Parser = getParser(); Parser.Lex(); if (getLexer().isNot(AsmToken::EndOfStatement)) return reportParseError("unexpected token, expected end of statement"); // Create a copy of the current assembler options environment and push it. AssemblerOptions.push_back( llvm::make_unique(AssemblerOptions.back().get())); getTargetStreamer().emitDirectiveSetPush(); return false; } bool MipsAsmParser::parseSetSoftFloatDirective() { MCAsmParser &Parser = getParser(); Parser.Lex(); if (getLexer().isNot(AsmToken::EndOfStatement)) return reportParseError("unexpected token, expected end of statement"); setFeatureBits(Mips::FeatureSoftFloat, "soft-float"); getTargetStreamer().emitDirectiveSetSoftFloat(); return false; } bool MipsAsmParser::parseSetHardFloatDirective() { MCAsmParser &Parser = getParser(); Parser.Lex(); if (getLexer().isNot(AsmToken::EndOfStatement)) return reportParseError("unexpected token, expected end of statement"); clearFeatureBits(Mips::FeatureSoftFloat, "soft-float"); getTargetStreamer().emitDirectiveSetHardFloat(); return false; } bool MipsAsmParser::parseSetAssignment() { StringRef Name; const MCExpr *Value; MCAsmParser &Parser = getParser(); if (Parser.parseIdentifier(Name)) return reportParseError("expected identifier after .set"); if (getLexer().isNot(AsmToken::Comma)) return reportParseError("unexpected token, expected comma"); Lex(); // Eat comma if (getLexer().is(AsmToken::Dollar) && getLexer().peekTok().is(AsmToken::Integer)) { // Parse assignment of a numeric register: // .set r1,$1 Parser.Lex(); // Eat $. RegisterSets[Name] = Parser.getTok(); Parser.Lex(); // Eat identifier. getContext().getOrCreateSymbol(Name); } else if (!Parser.parseExpression(Value)) { // Parse assignment of an expression including // symbolic registers: // .set $tmp, $BB0-$BB1 // .set r2, $f2 MCSymbol *Sym = getContext().getOrCreateSymbol(Name); Sym->setVariableValue(Value); } else { return reportParseError("expected valid expression after comma"); } return false; } bool MipsAsmParser::parseSetMips0Directive() { MCAsmParser &Parser = getParser(); Parser.Lex(); if (getLexer().isNot(AsmToken::EndOfStatement)) return reportParseError("unexpected token, expected end of statement"); // Reset assembler options to their initial values. MCSubtargetInfo &STI = copySTI(); setAvailableFeatures( ComputeAvailableFeatures(AssemblerOptions.front()->getFeatures())); STI.setFeatureBits(AssemblerOptions.front()->getFeatures()); AssemblerOptions.back()->setFeatures(AssemblerOptions.front()->getFeatures()); getTargetStreamer().emitDirectiveSetMips0(); return false; } bool MipsAsmParser::parseSetArchDirective() { MCAsmParser &Parser = getParser(); Parser.Lex(); if (getLexer().isNot(AsmToken::Equal)) return reportParseError("unexpected token, expected equals sign"); Parser.Lex(); StringRef Arch; if (Parser.parseIdentifier(Arch)) return reportParseError("expected arch identifier"); StringRef ArchFeatureName = StringSwitch(Arch) .Case("mips1", "mips1") .Case("mips2", "mips2") .Case("mips3", "mips3") .Case("mips4", "mips4") .Case("mips5", "mips5") .Case("mips32", "mips32") .Case("mips32r2", "mips32r2") .Case("mips32r3", "mips32r3") .Case("mips32r5", "mips32r5") .Case("mips32r6", "mips32r6") .Case("mips64", "mips64") .Case("mips64r2", "mips64r2") .Case("mips64r3", "mips64r3") .Case("mips64r5", "mips64r5") .Case("mips64r6", "mips64r6") .Case("octeon", "cnmips") .Case("r4000", "mips3") // This is an implementation of Mips3. .Default(""); if (ArchFeatureName.empty()) return reportParseError("unsupported architecture"); if (ArchFeatureName == "mips64r6" && inMicroMipsMode()) return reportParseError("mips64r6 does not support microMIPS"); selectArch(ArchFeatureName); getTargetStreamer().emitDirectiveSetArch(Arch); return false; } bool MipsAsmParser::parseSetFeature(uint64_t Feature) { MCAsmParser &Parser = getParser(); Parser.Lex(); if (getLexer().isNot(AsmToken::EndOfStatement)) return reportParseError("unexpected token, expected end of statement"); switch (Feature) { default: llvm_unreachable("Unimplemented feature"); case Mips::FeatureDSP: setFeatureBits(Mips::FeatureDSP, "dsp"); getTargetStreamer().emitDirectiveSetDsp(); break; case Mips::FeatureDSPR2: setFeatureBits(Mips::FeatureDSPR2, "dspr2"); getTargetStreamer().emitDirectiveSetDspr2(); break; case Mips::FeatureMicroMips: setFeatureBits(Mips::FeatureMicroMips, "micromips"); getTargetStreamer().emitDirectiveSetMicroMips(); break; case Mips::FeatureMips1: selectArch("mips1"); getTargetStreamer().emitDirectiveSetMips1(); break; case Mips::FeatureMips2: selectArch("mips2"); getTargetStreamer().emitDirectiveSetMips2(); break; case Mips::FeatureMips3: selectArch("mips3"); getTargetStreamer().emitDirectiveSetMips3(); break; case Mips::FeatureMips4: selectArch("mips4"); getTargetStreamer().emitDirectiveSetMips4(); break; case Mips::FeatureMips5: selectArch("mips5"); getTargetStreamer().emitDirectiveSetMips5(); break; case Mips::FeatureMips32: selectArch("mips32"); getTargetStreamer().emitDirectiveSetMips32(); break; case Mips::FeatureMips32r2: selectArch("mips32r2"); getTargetStreamer().emitDirectiveSetMips32R2(); break; case Mips::FeatureMips32r3: selectArch("mips32r3"); getTargetStreamer().emitDirectiveSetMips32R3(); break; case Mips::FeatureMips32r5: selectArch("mips32r5"); getTargetStreamer().emitDirectiveSetMips32R5(); break; case Mips::FeatureMips32r6: selectArch("mips32r6"); getTargetStreamer().emitDirectiveSetMips32R6(); break; case Mips::FeatureMips64: selectArch("mips64"); getTargetStreamer().emitDirectiveSetMips64(); break; case Mips::FeatureMips64r2: selectArch("mips64r2"); getTargetStreamer().emitDirectiveSetMips64R2(); break; case Mips::FeatureMips64r3: selectArch("mips64r3"); getTargetStreamer().emitDirectiveSetMips64R3(); break; case Mips::FeatureMips64r5: selectArch("mips64r5"); getTargetStreamer().emitDirectiveSetMips64R5(); break; case Mips::FeatureMips64r6: selectArch("mips64r6"); getTargetStreamer().emitDirectiveSetMips64R6(); break; case Mips::FeatureCRC: setFeatureBits(Mips::FeatureCRC, "crc"); getTargetStreamer().emitDirectiveSetCRC(); break; case Mips::FeatureVirt: setFeatureBits(Mips::FeatureVirt, "virt"); getTargetStreamer().emitDirectiveSetVirt(); break; case Mips::FeatureGINV: setFeatureBits(Mips::FeatureGINV, "ginv"); getTargetStreamer().emitDirectiveSetGINV(); break; } return false; } bool MipsAsmParser::eatComma(StringRef ErrorStr) { MCAsmParser &Parser = getParser(); if (getLexer().isNot(AsmToken::Comma)) { SMLoc Loc = getLexer().getLoc(); return Error(Loc, ErrorStr); } Parser.Lex(); // Eat the comma. return true; } // Used to determine if .cpload, .cprestore, and .cpsetup have any effect. // In this class, it is only used for .cprestore. // FIXME: Only keep track of IsPicEnabled in one place, instead of in both // MipsTargetELFStreamer and MipsAsmParser. bool MipsAsmParser::isPicAndNotNxxAbi() { return inPicMode() && !(isABI_N32() || isABI_N64()); } bool MipsAsmParser::parseDirectiveCpLoad(SMLoc Loc) { if (AssemblerOptions.back()->isReorder()) Warning(Loc, ".cpload should be inside a noreorder section"); if (inMips16Mode()) { reportParseError(".cpload is not supported in Mips16 mode"); return false; } SmallVector, 1> Reg; OperandMatchResultTy ResTy = parseAnyRegister(Reg); if (ResTy == MatchOperand_NoMatch || ResTy == MatchOperand_ParseFail) { reportParseError("expected register containing function address"); return false; } MipsOperand &RegOpnd = static_cast(*Reg[0]); if (!RegOpnd.isGPRAsmReg()) { reportParseError(RegOpnd.getStartLoc(), "invalid register"); return false; } // If this is not the end of the statement, report an error. if (getLexer().isNot(AsmToken::EndOfStatement)) { reportParseError("unexpected token, expected end of statement"); return false; } getTargetStreamer().emitDirectiveCpLoad(RegOpnd.getGPR32Reg()); return false; } bool MipsAsmParser::parseDirectiveCpRestore(SMLoc Loc) { MCAsmParser &Parser = getParser(); // Note that .cprestore is ignored if used with the N32 and N64 ABIs or if it // is used in non-PIC mode. if (inMips16Mode()) { reportParseError(".cprestore is not supported in Mips16 mode"); return false; } // Get the stack offset value. const MCExpr *StackOffset; int64_t StackOffsetVal; if (Parser.parseExpression(StackOffset)) { reportParseError("expected stack offset value"); return false; } if (!StackOffset->evaluateAsAbsolute(StackOffsetVal)) { reportParseError("stack offset is not an absolute expression"); return false; } if (StackOffsetVal < 0) { Warning(Loc, ".cprestore with negative stack offset has no effect"); IsCpRestoreSet = false; } else { IsCpRestoreSet = true; CpRestoreOffset = StackOffsetVal; } // If this is not the end of the statement, report an error. if (getLexer().isNot(AsmToken::EndOfStatement)) { reportParseError("unexpected token, expected end of statement"); return false; } if (!getTargetStreamer().emitDirectiveCpRestore( CpRestoreOffset, [&]() { return getATReg(Loc); }, Loc, STI)) return true; Parser.Lex(); // Consume the EndOfStatement. return false; } bool MipsAsmParser::parseDirectiveCPSetup() { MCAsmParser &Parser = getParser(); unsigned FuncReg; unsigned Save; bool SaveIsReg = true; SmallVector, 1> TmpReg; OperandMatchResultTy ResTy = parseAnyRegister(TmpReg); if (ResTy == MatchOperand_NoMatch) { reportParseError("expected register containing function address"); return false; } MipsOperand &FuncRegOpnd = static_cast(*TmpReg[0]); if (!FuncRegOpnd.isGPRAsmReg()) { reportParseError(FuncRegOpnd.getStartLoc(), "invalid register"); return false; } FuncReg = FuncRegOpnd.getGPR32Reg(); TmpReg.clear(); if (!eatComma("unexpected token, expected comma")) return true; ResTy = parseAnyRegister(TmpReg); if (ResTy == MatchOperand_NoMatch) { const MCExpr *OffsetExpr; int64_t OffsetVal; SMLoc ExprLoc = getLexer().getLoc(); if (Parser.parseExpression(OffsetExpr) || !OffsetExpr->evaluateAsAbsolute(OffsetVal)) { reportParseError(ExprLoc, "expected save register or stack offset"); return false; } Save = OffsetVal; SaveIsReg = false; } else { MipsOperand &SaveOpnd = static_cast(*TmpReg[0]); if (!SaveOpnd.isGPRAsmReg()) { reportParseError(SaveOpnd.getStartLoc(), "invalid register"); return false; } Save = SaveOpnd.getGPR32Reg(); } if (!eatComma("unexpected token, expected comma")) return true; const MCExpr *Expr; if (Parser.parseExpression(Expr)) { reportParseError("expected expression"); return false; } if (Expr->getKind() != MCExpr::SymbolRef) { reportParseError("expected symbol"); return false; } const MCSymbolRefExpr *Ref = static_cast(Expr); CpSaveLocation = Save; CpSaveLocationIsRegister = SaveIsReg; getTargetStreamer().emitDirectiveCpsetup(FuncReg, Save, Ref->getSymbol(), SaveIsReg); return false; } bool MipsAsmParser::parseDirectiveCPReturn() { getTargetStreamer().emitDirectiveCpreturn(CpSaveLocation, CpSaveLocationIsRegister); return false; } bool MipsAsmParser::parseDirectiveNaN() { MCAsmParser &Parser = getParser(); if (getLexer().isNot(AsmToken::EndOfStatement)) { const AsmToken &Tok = Parser.getTok(); if (Tok.getString() == "2008") { Parser.Lex(); getTargetStreamer().emitDirectiveNaN2008(); return false; } else if (Tok.getString() == "legacy") { Parser.Lex(); getTargetStreamer().emitDirectiveNaNLegacy(); return false; } } // If we don't recognize the option passed to the .nan // directive (e.g. no option or unknown option), emit an error. reportParseError("invalid option in .nan directive"); return false; } bool MipsAsmParser::parseDirectiveSet() { const AsmToken &Tok = getParser().getTok(); StringRef IdVal = Tok.getString(); SMLoc Loc = Tok.getLoc(); if (IdVal == "noat") return parseSetNoAtDirective(); if (IdVal == "at") return parseSetAtDirective(); if (IdVal == "arch") return parseSetArchDirective(); if (IdVal == "bopt") { Warning(Loc, "'bopt' feature is unsupported"); getParser().Lex(); return false; } if (IdVal == "nobopt") { // We're already running in nobopt mode, so nothing to do. getParser().Lex(); return false; } if (IdVal == "fp") return parseSetFpDirective(); if (IdVal == "oddspreg") return parseSetOddSPRegDirective(); if (IdVal == "nooddspreg") return parseSetNoOddSPRegDirective(); if (IdVal == "pop") return parseSetPopDirective(); if (IdVal == "push") return parseSetPushDirective(); if (IdVal == "reorder") return parseSetReorderDirective(); if (IdVal == "noreorder") return parseSetNoReorderDirective(); if (IdVal == "macro") return parseSetMacroDirective(); if (IdVal == "nomacro") return parseSetNoMacroDirective(); if (IdVal == "mips16") return parseSetMips16Directive(); if (IdVal == "nomips16") return parseSetNoMips16Directive(); if (IdVal == "nomicromips") { clearFeatureBits(Mips::FeatureMicroMips, "micromips"); getTargetStreamer().emitDirectiveSetNoMicroMips(); getParser().eatToEndOfStatement(); return false; } if (IdVal == "micromips") { if (hasMips64r6()) { Error(Loc, ".set micromips directive is not supported with MIPS64R6"); return false; } return parseSetFeature(Mips::FeatureMicroMips); } if (IdVal == "mips0") return parseSetMips0Directive(); if (IdVal == "mips1") return parseSetFeature(Mips::FeatureMips1); if (IdVal == "mips2") return parseSetFeature(Mips::FeatureMips2); if (IdVal == "mips3") return parseSetFeature(Mips::FeatureMips3); if (IdVal == "mips4") return parseSetFeature(Mips::FeatureMips4); if (IdVal == "mips5") return parseSetFeature(Mips::FeatureMips5); if (IdVal == "mips32") return parseSetFeature(Mips::FeatureMips32); if (IdVal == "mips32r2") return parseSetFeature(Mips::FeatureMips32r2); if (IdVal == "mips32r3") return parseSetFeature(Mips::FeatureMips32r3); if (IdVal == "mips32r5") return parseSetFeature(Mips::FeatureMips32r5); if (IdVal == "mips32r6") return parseSetFeature(Mips::FeatureMips32r6); if (IdVal == "mips64") return parseSetFeature(Mips::FeatureMips64); if (IdVal == "mips64r2") return parseSetFeature(Mips::FeatureMips64r2); if (IdVal == "mips64r3") return parseSetFeature(Mips::FeatureMips64r3); if (IdVal == "mips64r5") return parseSetFeature(Mips::FeatureMips64r5); if (IdVal == "mips64r6") { if (inMicroMipsMode()) { Error(Loc, "MIPS64R6 is not supported with microMIPS"); return false; } return parseSetFeature(Mips::FeatureMips64r6); } if (IdVal == "dsp") return parseSetFeature(Mips::FeatureDSP); if (IdVal == "dspr2") return parseSetFeature(Mips::FeatureDSPR2); if (IdVal == "nodsp") return parseSetNoDspDirective(); if (IdVal == "msa") return parseSetMsaDirective(); if (IdVal == "nomsa") return parseSetNoMsaDirective(); if (IdVal == "mt") return parseSetMtDirective(); if (IdVal == "nomt") return parseSetNoMtDirective(); if (IdVal == "softfloat") return parseSetSoftFloatDirective(); if (IdVal == "hardfloat") return parseSetHardFloatDirective(); if (IdVal == "crc") return parseSetFeature(Mips::FeatureCRC); if (IdVal == "nocrc") return parseSetNoCRCDirective(); if (IdVal == "virt") return parseSetFeature(Mips::FeatureVirt); if (IdVal == "novirt") return parseSetNoVirtDirective(); if (IdVal == "ginv") return parseSetFeature(Mips::FeatureGINV); if (IdVal == "noginv") return parseSetNoGINVDirective(); // It is just an identifier, look for an assignment. return parseSetAssignment(); } /// parseDirectiveGpWord /// ::= .gpword local_sym bool MipsAsmParser::parseDirectiveGpWord() { MCAsmParser &Parser = getParser(); const MCExpr *Value; // EmitGPRel32Value requires an expression, so we are using base class // method to evaluate the expression. if (getParser().parseExpression(Value)) return true; getParser().getStreamer().EmitGPRel32Value(Value); if (getLexer().isNot(AsmToken::EndOfStatement)) return Error(getLexer().getLoc(), "unexpected token, expected end of statement"); Parser.Lex(); // Eat EndOfStatement token. return false; } /// parseDirectiveGpDWord /// ::= .gpdword local_sym bool MipsAsmParser::parseDirectiveGpDWord() { MCAsmParser &Parser = getParser(); const MCExpr *Value; // EmitGPRel64Value requires an expression, so we are using base class // method to evaluate the expression. if (getParser().parseExpression(Value)) return true; getParser().getStreamer().EmitGPRel64Value(Value); if (getLexer().isNot(AsmToken::EndOfStatement)) return Error(getLexer().getLoc(), "unexpected token, expected end of statement"); Parser.Lex(); // Eat EndOfStatement token. return false; } /// parseDirectiveDtpRelWord /// ::= .dtprelword tls_sym bool MipsAsmParser::parseDirectiveDtpRelWord() { MCAsmParser &Parser = getParser(); const MCExpr *Value; // EmitDTPRel32Value requires an expression, so we are using base class // method to evaluate the expression. if (getParser().parseExpression(Value)) return true; getParser().getStreamer().EmitDTPRel32Value(Value); if (getLexer().isNot(AsmToken::EndOfStatement)) return Error(getLexer().getLoc(), "unexpected token, expected end of statement"); Parser.Lex(); // Eat EndOfStatement token. return false; } /// parseDirectiveDtpRelDWord /// ::= .dtpreldword tls_sym bool MipsAsmParser::parseDirectiveDtpRelDWord() { MCAsmParser &Parser = getParser(); const MCExpr *Value; // EmitDTPRel64Value requires an expression, so we are using base class // method to evaluate the expression. if (getParser().parseExpression(Value)) return true; getParser().getStreamer().EmitDTPRel64Value(Value); if (getLexer().isNot(AsmToken::EndOfStatement)) return Error(getLexer().getLoc(), "unexpected token, expected end of statement"); Parser.Lex(); // Eat EndOfStatement token. return false; } /// parseDirectiveTpRelWord /// ::= .tprelword tls_sym bool MipsAsmParser::parseDirectiveTpRelWord() { MCAsmParser &Parser = getParser(); const MCExpr *Value; // EmitTPRel32Value requires an expression, so we are using base class // method to evaluate the expression. if (getParser().parseExpression(Value)) return true; getParser().getStreamer().EmitTPRel32Value(Value); if (getLexer().isNot(AsmToken::EndOfStatement)) return Error(getLexer().getLoc(), "unexpected token, expected end of statement"); Parser.Lex(); // Eat EndOfStatement token. return false; } /// parseDirectiveTpRelDWord /// ::= .tpreldword tls_sym bool MipsAsmParser::parseDirectiveTpRelDWord() { MCAsmParser &Parser = getParser(); const MCExpr *Value; // EmitTPRel64Value requires an expression, so we are using base class // method to evaluate the expression. if (getParser().parseExpression(Value)) return true; getParser().getStreamer().EmitTPRel64Value(Value); if (getLexer().isNot(AsmToken::EndOfStatement)) return Error(getLexer().getLoc(), "unexpected token, expected end of statement"); Parser.Lex(); // Eat EndOfStatement token. return false; } bool MipsAsmParser::parseDirectiveOption() { MCAsmParser &Parser = getParser(); // Get the option token. AsmToken Tok = Parser.getTok(); // At the moment only identifiers are supported. if (Tok.isNot(AsmToken::Identifier)) { return Error(Parser.getTok().getLoc(), "unexpected token, expected identifier"); } StringRef Option = Tok.getIdentifier(); if (Option == "pic0") { // MipsAsmParser needs to know if the current PIC mode changes. IsPicEnabled = false; getTargetStreamer().emitDirectiveOptionPic0(); Parser.Lex(); if (Parser.getTok().isNot(AsmToken::EndOfStatement)) { return Error(Parser.getTok().getLoc(), "unexpected token, expected end of statement"); } return false; } if (Option == "pic2") { // MipsAsmParser needs to know if the current PIC mode changes. IsPicEnabled = true; getTargetStreamer().emitDirectiveOptionPic2(); Parser.Lex(); if (Parser.getTok().isNot(AsmToken::EndOfStatement)) { return Error(Parser.getTok().getLoc(), "unexpected token, expected end of statement"); } return false; } // Unknown option. Warning(Parser.getTok().getLoc(), "unknown option, expected 'pic0' or 'pic2'"); Parser.eatToEndOfStatement(); return false; } /// parseInsnDirective /// ::= .insn bool MipsAsmParser::parseInsnDirective() { // If this is not the end of the statement, report an error. if (getLexer().isNot(AsmToken::EndOfStatement)) { reportParseError("unexpected token, expected end of statement"); return false; } // The actual label marking happens in // MipsELFStreamer::createPendingLabelRelocs(). getTargetStreamer().emitDirectiveInsn(); getParser().Lex(); // Eat EndOfStatement token. return false; } /// parseRSectionDirective /// ::= .rdata bool MipsAsmParser::parseRSectionDirective(StringRef Section) { // If this is not the end of the statement, report an error. if (getLexer().isNot(AsmToken::EndOfStatement)) { reportParseError("unexpected token, expected end of statement"); return false; } MCSection *ELFSection = getContext().getELFSection( Section, ELF::SHT_PROGBITS, ELF::SHF_ALLOC); getParser().getStreamer().SwitchSection(ELFSection); getParser().Lex(); // Eat EndOfStatement token. return false; } /// parseSSectionDirective /// ::= .sbss /// ::= .sdata bool MipsAsmParser::parseSSectionDirective(StringRef Section, unsigned Type) { // If this is not the end of the statement, report an error. if (getLexer().isNot(AsmToken::EndOfStatement)) { reportParseError("unexpected token, expected end of statement"); return false; } MCSection *ELFSection = getContext().getELFSection( Section, Type, ELF::SHF_WRITE | ELF::SHF_ALLOC | ELF::SHF_MIPS_GPREL); getParser().getStreamer().SwitchSection(ELFSection); getParser().Lex(); // Eat EndOfStatement token. return false; } /// parseDirectiveModule /// ::= .module oddspreg /// ::= .module nooddspreg /// ::= .module fp=value /// ::= .module softfloat /// ::= .module hardfloat /// ::= .module mt /// ::= .module crc /// ::= .module nocrc /// ::= .module virt /// ::= .module novirt /// ::= .module ginv /// ::= .module noginv bool MipsAsmParser::parseDirectiveModule() { MCAsmParser &Parser = getParser(); MCAsmLexer &Lexer = getLexer(); SMLoc L = Lexer.getLoc(); if (!getTargetStreamer().isModuleDirectiveAllowed()) { // TODO : get a better message. reportParseError(".module directive must appear before any code"); return false; } StringRef Option; if (Parser.parseIdentifier(Option)) { reportParseError("expected .module option identifier"); return false; } if (Option == "oddspreg") { clearModuleFeatureBits(Mips::FeatureNoOddSPReg, "nooddspreg"); // Synchronize the abiflags information with the FeatureBits information we // changed above. getTargetStreamer().updateABIInfo(*this); // If printing assembly, use the recently updated abiflags information. // If generating ELF, don't do anything (the .MIPS.abiflags section gets // emitted at the end). getTargetStreamer().emitDirectiveModuleOddSPReg(); // If this is not the end of the statement, report an error. if (getLexer().isNot(AsmToken::EndOfStatement)) { reportParseError("unexpected token, expected end of statement"); return false; } return false; // parseDirectiveModule has finished successfully. } else if (Option == "nooddspreg") { if (!isABI_O32()) { return Error(L, "'.module nooddspreg' requires the O32 ABI"); } setModuleFeatureBits(Mips::FeatureNoOddSPReg, "nooddspreg"); // Synchronize the abiflags information with the FeatureBits information we // changed above. getTargetStreamer().updateABIInfo(*this); // If printing assembly, use the recently updated abiflags information. // If generating ELF, don't do anything (the .MIPS.abiflags section gets // emitted at the end). getTargetStreamer().emitDirectiveModuleOddSPReg(); // If this is not the end of the statement, report an error. if (getLexer().isNot(AsmToken::EndOfStatement)) { reportParseError("unexpected token, expected end of statement"); return false; } return false; // parseDirectiveModule has finished successfully. } else if (Option == "fp") { return parseDirectiveModuleFP(); } else if (Option == "softfloat") { setModuleFeatureBits(Mips::FeatureSoftFloat, "soft-float"); // Synchronize the ABI Flags information with the FeatureBits information we // updated above. getTargetStreamer().updateABIInfo(*this); // If printing assembly, use the recently updated ABI Flags information. // If generating ELF, don't do anything (the .MIPS.abiflags section gets // emitted later). getTargetStreamer().emitDirectiveModuleSoftFloat(); // If this is not the end of the statement, report an error. if (getLexer().isNot(AsmToken::EndOfStatement)) { reportParseError("unexpected token, expected end of statement"); return false; } return false; // parseDirectiveModule has finished successfully. } else if (Option == "hardfloat") { clearModuleFeatureBits(Mips::FeatureSoftFloat, "soft-float"); // Synchronize the ABI Flags information with the FeatureBits information we // updated above. getTargetStreamer().updateABIInfo(*this); // If printing assembly, use the recently updated ABI Flags information. // If generating ELF, don't do anything (the .MIPS.abiflags section gets // emitted later). getTargetStreamer().emitDirectiveModuleHardFloat(); // If this is not the end of the statement, report an error. if (getLexer().isNot(AsmToken::EndOfStatement)) { reportParseError("unexpected token, expected end of statement"); return false; } return false; // parseDirectiveModule has finished successfully. } else if (Option == "mt") { setModuleFeatureBits(Mips::FeatureMT, "mt"); // Synchronize the ABI Flags information with the FeatureBits information we // updated above. getTargetStreamer().updateABIInfo(*this); // If printing assembly, use the recently updated ABI Flags information. // If generating ELF, don't do anything (the .MIPS.abiflags section gets // emitted later). getTargetStreamer().emitDirectiveModuleMT(); // If this is not the end of the statement, report an error. if (getLexer().isNot(AsmToken::EndOfStatement)) { reportParseError("unexpected token, expected end of statement"); return false; } return false; // parseDirectiveModule has finished successfully. } else if (Option == "crc") { setModuleFeatureBits(Mips::FeatureCRC, "crc"); // Synchronize the ABI Flags information with the FeatureBits information we // updated above. getTargetStreamer().updateABIInfo(*this); // If printing assembly, use the recently updated ABI Flags information. // If generating ELF, don't do anything (the .MIPS.abiflags section gets // emitted later). getTargetStreamer().emitDirectiveModuleCRC(); // If this is not the end of the statement, report an error. if (getLexer().isNot(AsmToken::EndOfStatement)) { reportParseError("unexpected token, expected end of statement"); return false; } return false; // parseDirectiveModule has finished successfully. } else if (Option == "nocrc") { clearModuleFeatureBits(Mips::FeatureCRC, "crc"); // Synchronize the ABI Flags information with the FeatureBits information we // updated above. getTargetStreamer().updateABIInfo(*this); // If printing assembly, use the recently updated ABI Flags information. // If generating ELF, don't do anything (the .MIPS.abiflags section gets // emitted later). getTargetStreamer().emitDirectiveModuleNoCRC(); // If this is not the end of the statement, report an error. if (getLexer().isNot(AsmToken::EndOfStatement)) { reportParseError("unexpected token, expected end of statement"); return false; } return false; // parseDirectiveModule has finished successfully. } else if (Option == "virt") { setModuleFeatureBits(Mips::FeatureVirt, "virt"); // Synchronize the ABI Flags information with the FeatureBits information we // updated above. getTargetStreamer().updateABIInfo(*this); // If printing assembly, use the recently updated ABI Flags information. // If generating ELF, don't do anything (the .MIPS.abiflags section gets // emitted later). getTargetStreamer().emitDirectiveModuleVirt(); // If this is not the end of the statement, report an error. if (getLexer().isNot(AsmToken::EndOfStatement)) { reportParseError("unexpected token, expected end of statement"); return false; } return false; // parseDirectiveModule has finished successfully. } else if (Option == "novirt") { clearModuleFeatureBits(Mips::FeatureVirt, "virt"); // Synchronize the ABI Flags information with the FeatureBits information we // updated above. getTargetStreamer().updateABIInfo(*this); // If printing assembly, use the recently updated ABI Flags information. // If generating ELF, don't do anything (the .MIPS.abiflags section gets // emitted later). getTargetStreamer().emitDirectiveModuleNoVirt(); // If this is not the end of the statement, report an error. if (getLexer().isNot(AsmToken::EndOfStatement)) { reportParseError("unexpected token, expected end of statement"); return false; } return false; // parseDirectiveModule has finished successfully. } else if (Option == "ginv") { setModuleFeatureBits(Mips::FeatureGINV, "ginv"); // Synchronize the ABI Flags information with the FeatureBits information we // updated above. getTargetStreamer().updateABIInfo(*this); // If printing assembly, use the recently updated ABI Flags information. // If generating ELF, don't do anything (the .MIPS.abiflags section gets // emitted later). getTargetStreamer().emitDirectiveModuleGINV(); // If this is not the end of the statement, report an error. if (getLexer().isNot(AsmToken::EndOfStatement)) { reportParseError("unexpected token, expected end of statement"); return false; } return false; // parseDirectiveModule has finished successfully. } else if (Option == "noginv") { clearModuleFeatureBits(Mips::FeatureGINV, "ginv"); // Synchronize the ABI Flags information with the FeatureBits information we // updated above. getTargetStreamer().updateABIInfo(*this); // If printing assembly, use the recently updated ABI Flags information. // If generating ELF, don't do anything (the .MIPS.abiflags section gets // emitted later). getTargetStreamer().emitDirectiveModuleNoGINV(); // If this is not the end of the statement, report an error. if (getLexer().isNot(AsmToken::EndOfStatement)) { reportParseError("unexpected token, expected end of statement"); return false; } return false; // parseDirectiveModule has finished successfully. } else { return Error(L, "'" + Twine(Option) + "' is not a valid .module option."); } } /// parseDirectiveModuleFP /// ::= =32 /// ::= =xx /// ::= =64 bool MipsAsmParser::parseDirectiveModuleFP() { MCAsmParser &Parser = getParser(); MCAsmLexer &Lexer = getLexer(); if (Lexer.isNot(AsmToken::Equal)) { reportParseError("unexpected token, expected equals sign '='"); return false; } Parser.Lex(); // Eat '=' token. MipsABIFlagsSection::FpABIKind FpABI; if (!parseFpABIValue(FpABI, ".module")) return false; if (getLexer().isNot(AsmToken::EndOfStatement)) { reportParseError("unexpected token, expected end of statement"); return false; } // Synchronize the abiflags information with the FeatureBits information we // changed above. getTargetStreamer().updateABIInfo(*this); // If printing assembly, use the recently updated abiflags information. // If generating ELF, don't do anything (the .MIPS.abiflags section gets // emitted at the end). getTargetStreamer().emitDirectiveModuleFP(); Parser.Lex(); // Consume the EndOfStatement. return false; } bool MipsAsmParser::parseFpABIValue(MipsABIFlagsSection::FpABIKind &FpABI, StringRef Directive) { MCAsmParser &Parser = getParser(); MCAsmLexer &Lexer = getLexer(); bool ModuleLevelOptions = Directive == ".module"; if (Lexer.is(AsmToken::Identifier)) { StringRef Value = Parser.getTok().getString(); Parser.Lex(); if (Value != "xx") { reportParseError("unsupported value, expected 'xx', '32' or '64'"); return false; } if (!isABI_O32()) { reportParseError("'" + Directive + " fp=xx' requires the O32 ABI"); return false; } FpABI = MipsABIFlagsSection::FpABIKind::XX; if (ModuleLevelOptions) { setModuleFeatureBits(Mips::FeatureFPXX, "fpxx"); clearModuleFeatureBits(Mips::FeatureFP64Bit, "fp64"); } else { setFeatureBits(Mips::FeatureFPXX, "fpxx"); clearFeatureBits(Mips::FeatureFP64Bit, "fp64"); } return true; } if (Lexer.is(AsmToken::Integer)) { unsigned Value = Parser.getTok().getIntVal(); Parser.Lex(); if (Value != 32 && Value != 64) { reportParseError("unsupported value, expected 'xx', '32' or '64'"); return false; } if (Value == 32) { if (!isABI_O32()) { reportParseError("'" + Directive + " fp=32' requires the O32 ABI"); return false; } FpABI = MipsABIFlagsSection::FpABIKind::S32; if (ModuleLevelOptions) { clearModuleFeatureBits(Mips::FeatureFPXX, "fpxx"); clearModuleFeatureBits(Mips::FeatureFP64Bit, "fp64"); } else { clearFeatureBits(Mips::FeatureFPXX, "fpxx"); clearFeatureBits(Mips::FeatureFP64Bit, "fp64"); } } else { FpABI = MipsABIFlagsSection::FpABIKind::S64; if (ModuleLevelOptions) { clearModuleFeatureBits(Mips::FeatureFPXX, "fpxx"); setModuleFeatureBits(Mips::FeatureFP64Bit, "fp64"); } else { clearFeatureBits(Mips::FeatureFPXX, "fpxx"); setFeatureBits(Mips::FeatureFP64Bit, "fp64"); } } return true; } return false; } bool MipsAsmParser::ParseDirective(AsmToken DirectiveID) { // This returns false if this function recognizes the directive // regardless of whether it is successfully handles or reports an // error. Otherwise it returns true to give the generic parser a // chance at recognizing it. MCAsmParser &Parser = getParser(); StringRef IDVal = DirectiveID.getString(); if (IDVal == ".cpload") { parseDirectiveCpLoad(DirectiveID.getLoc()); return false; } if (IDVal == ".cprestore") { parseDirectiveCpRestore(DirectiveID.getLoc()); return false; } if (IDVal == ".ent") { StringRef SymbolName; if (Parser.parseIdentifier(SymbolName)) { reportParseError("expected identifier after .ent"); return false; } // There's an undocumented extension that allows an integer to // follow the name of the procedure which AFAICS is ignored by GAS. // Example: .ent foo,2 if (getLexer().isNot(AsmToken::EndOfStatement)) { if (getLexer().isNot(AsmToken::Comma)) { // Even though we accept this undocumented extension for compatibility // reasons, the additional integer argument does not actually change // the behaviour of the '.ent' directive, so we would like to discourage // its use. We do this by not referring to the extended version in // error messages which are not directly related to its use. reportParseError("unexpected token, expected end of statement"); return false; } Parser.Lex(); // Eat the comma. const MCExpr *DummyNumber; int64_t DummyNumberVal; // If the user was explicitly trying to use the extended version, // we still give helpful extension-related error messages. if (Parser.parseExpression(DummyNumber)) { reportParseError("expected number after comma"); return false; } if (!DummyNumber->evaluateAsAbsolute(DummyNumberVal)) { reportParseError("expected an absolute expression after comma"); return false; } } // If this is not the end of the statement, report an error. if (getLexer().isNot(AsmToken::EndOfStatement)) { reportParseError("unexpected token, expected end of statement"); return false; } MCSymbol *Sym = getContext().getOrCreateSymbol(SymbolName); getTargetStreamer().emitDirectiveEnt(*Sym); CurrentFn = Sym; IsCpRestoreSet = false; return false; } if (IDVal == ".end") { StringRef SymbolName; if (Parser.parseIdentifier(SymbolName)) { reportParseError("expected identifier after .end"); return false; } if (getLexer().isNot(AsmToken::EndOfStatement)) { reportParseError("unexpected token, expected end of statement"); return false; } if (CurrentFn == nullptr) { reportParseError(".end used without .ent"); return false; } if ((SymbolName != CurrentFn->getName())) { reportParseError(".end symbol does not match .ent symbol"); return false; } getTargetStreamer().emitDirectiveEnd(SymbolName); CurrentFn = nullptr; IsCpRestoreSet = false; return false; } if (IDVal == ".frame") { // .frame $stack_reg, frame_size_in_bytes, $return_reg SmallVector, 1> TmpReg; OperandMatchResultTy ResTy = parseAnyRegister(TmpReg); if (ResTy == MatchOperand_NoMatch || ResTy == MatchOperand_ParseFail) { reportParseError("expected stack register"); return false; } MipsOperand &StackRegOpnd = static_cast(*TmpReg[0]); if (!StackRegOpnd.isGPRAsmReg()) { reportParseError(StackRegOpnd.getStartLoc(), "expected general purpose register"); return false; } unsigned StackReg = StackRegOpnd.getGPR32Reg(); if (Parser.getTok().is(AsmToken::Comma)) Parser.Lex(); else { reportParseError("unexpected token, expected comma"); return false; } // Parse the frame size. const MCExpr *FrameSize; int64_t FrameSizeVal; if (Parser.parseExpression(FrameSize)) { reportParseError("expected frame size value"); return false; } if (!FrameSize->evaluateAsAbsolute(FrameSizeVal)) { reportParseError("frame size not an absolute expression"); return false; } if (Parser.getTok().is(AsmToken::Comma)) Parser.Lex(); else { reportParseError("unexpected token, expected comma"); return false; } // Parse the return register. TmpReg.clear(); ResTy = parseAnyRegister(TmpReg); if (ResTy == MatchOperand_NoMatch || ResTy == MatchOperand_ParseFail) { reportParseError("expected return register"); return false; } MipsOperand &ReturnRegOpnd = static_cast(*TmpReg[0]); if (!ReturnRegOpnd.isGPRAsmReg()) { reportParseError(ReturnRegOpnd.getStartLoc(), "expected general purpose register"); return false; } // If this is not the end of the statement, report an error. if (getLexer().isNot(AsmToken::EndOfStatement)) { reportParseError("unexpected token, expected end of statement"); return false; } getTargetStreamer().emitFrame(StackReg, FrameSizeVal, ReturnRegOpnd.getGPR32Reg()); IsCpRestoreSet = false; return false; } if (IDVal == ".set") { parseDirectiveSet(); return false; } if (IDVal == ".mask" || IDVal == ".fmask") { // .mask bitmask, frame_offset // bitmask: One bit for each register used. // frame_offset: Offset from Canonical Frame Address ($sp on entry) where // first register is expected to be saved. // Examples: // .mask 0x80000000, -4 // .fmask 0x80000000, -4 // // Parse the bitmask const MCExpr *BitMask; int64_t BitMaskVal; if (Parser.parseExpression(BitMask)) { reportParseError("expected bitmask value"); return false; } if (!BitMask->evaluateAsAbsolute(BitMaskVal)) { reportParseError("bitmask not an absolute expression"); return false; } if (Parser.getTok().is(AsmToken::Comma)) Parser.Lex(); else { reportParseError("unexpected token, expected comma"); return false; } // Parse the frame_offset const MCExpr *FrameOffset; int64_t FrameOffsetVal; if (Parser.parseExpression(FrameOffset)) { reportParseError("expected frame offset value"); return false; } if (!FrameOffset->evaluateAsAbsolute(FrameOffsetVal)) { reportParseError("frame offset not an absolute expression"); return false; } // If this is not the end of the statement, report an error. if (getLexer().isNot(AsmToken::EndOfStatement)) { reportParseError("unexpected token, expected end of statement"); return false; } if (IDVal == ".mask") getTargetStreamer().emitMask(BitMaskVal, FrameOffsetVal); else getTargetStreamer().emitFMask(BitMaskVal, FrameOffsetVal); return false; } if (IDVal == ".nan") return parseDirectiveNaN(); if (IDVal == ".gpword") { parseDirectiveGpWord(); return false; } if (IDVal == ".gpdword") { parseDirectiveGpDWord(); return false; } if (IDVal == ".dtprelword") { parseDirectiveDtpRelWord(); return false; } if (IDVal == ".dtpreldword") { parseDirectiveDtpRelDWord(); return false; } if (IDVal == ".tprelword") { parseDirectiveTpRelWord(); return false; } if (IDVal == ".tpreldword") { parseDirectiveTpRelDWord(); return false; } if (IDVal == ".option") { parseDirectiveOption(); return false; } if (IDVal == ".abicalls") { getTargetStreamer().emitDirectiveAbiCalls(); if (Parser.getTok().isNot(AsmToken::EndOfStatement)) { Error(Parser.getTok().getLoc(), "unexpected token, expected end of statement"); } return false; } if (IDVal == ".cpsetup") { parseDirectiveCPSetup(); return false; } if (IDVal == ".cpreturn") { parseDirectiveCPReturn(); return false; } if (IDVal == ".module") { parseDirectiveModule(); return false; } if (IDVal == ".llvm_internal_mips_reallow_module_directive") { parseInternalDirectiveReallowModule(); return false; } if (IDVal == ".insn") { parseInsnDirective(); return false; } if (IDVal == ".rdata") { parseRSectionDirective(".rodata"); return false; } if (IDVal == ".sbss") { parseSSectionDirective(IDVal, ELF::SHT_NOBITS); return false; } if (IDVal == ".sdata") { parseSSectionDirective(IDVal, ELF::SHT_PROGBITS); return false; } return true; } bool MipsAsmParser::parseInternalDirectiveReallowModule() { // If this is not the end of the statement, report an error. if (getLexer().isNot(AsmToken::EndOfStatement)) { reportParseError("unexpected token, expected end of statement"); return false; } getTargetStreamer().reallowModuleDirective(); getParser().Lex(); // Eat EndOfStatement token. return false; } extern "C" void LLVMInitializeMipsAsmParser() { RegisterMCAsmParser X(getTheMipsTarget()); RegisterMCAsmParser Y(getTheMipselTarget()); RegisterMCAsmParser A(getTheMips64Target()); RegisterMCAsmParser B(getTheMips64elTarget()); } #define GET_REGISTER_MATCHER #define GET_MATCHER_IMPLEMENTATION #define GET_MNEMONIC_SPELL_CHECKER #include "MipsGenAsmMatcher.inc" bool MipsAsmParser::mnemonicIsValid(StringRef Mnemonic, unsigned VariantID) { // Find the appropriate table for this asm variant. const MatchEntry *Start, *End; switch (VariantID) { default: llvm_unreachable("invalid variant!"); case 0: Start = std::begin(MatchTable0); End = std::end(MatchTable0); break; } // Search the table. auto MnemonicRange = std::equal_range(Start, End, Mnemonic, LessOpcode()); return MnemonicRange.first != MnemonicRange.second; } Index: vendor/llvm/dist-release_80/lib/Target/Mips/MCTargetDesc/MipsABIInfo.cpp =================================================================== --- vendor/llvm/dist-release_80/lib/Target/Mips/MCTargetDesc/MipsABIInfo.cpp (revision 343793) +++ vendor/llvm/dist-release_80/lib/Target/Mips/MCTargetDesc/MipsABIInfo.cpp (revision 343794) @@ -1,121 +1,128 @@ //===---- MipsABIInfo.cpp - Information about MIPS ABI's ------------------===// // // The LLVM Compiler Infrastructure // // This file is distributed under the University of Illinois Open Source // License. See LICENSE.TXT for details. // //===----------------------------------------------------------------------===// #include "MipsABIInfo.h" #include "MipsRegisterInfo.h" #include "llvm/ADT/StringRef.h" #include "llvm/ADT/StringSwitch.h" #include "llvm/MC/MCTargetOptions.h" using namespace llvm; +// Note: this option is defined here to be visible from libLLVMMipsAsmParser +// and libLLVMMipsCodeGen +cl::opt +EmitJalrReloc("mips-jalr-reloc", cl::Hidden, + cl::desc("MIPS: Emit R_{MICRO}MIPS_JALR relocation with jalr"), + cl::init(true)); + namespace { static const MCPhysReg O32IntRegs[4] = {Mips::A0, Mips::A1, Mips::A2, Mips::A3}; static const MCPhysReg Mips64IntRegs[8] = { Mips::A0_64, Mips::A1_64, Mips::A2_64, Mips::A3_64, Mips::T0_64, Mips::T1_64, Mips::T2_64, Mips::T3_64}; } ArrayRef MipsABIInfo::GetByValArgRegs() const { if (IsO32()) return makeArrayRef(O32IntRegs); if (IsN32() || IsN64()) return makeArrayRef(Mips64IntRegs); llvm_unreachable("Unhandled ABI"); } ArrayRef MipsABIInfo::GetVarArgRegs() const { if (IsO32()) return makeArrayRef(O32IntRegs); if (IsN32() || IsN64()) return makeArrayRef(Mips64IntRegs); llvm_unreachable("Unhandled ABI"); } unsigned MipsABIInfo::GetCalleeAllocdArgSizeInBytes(CallingConv::ID CC) const { if (IsO32()) return CC != CallingConv::Fast ? 16 : 0; if (IsN32() || IsN64()) return 0; llvm_unreachable("Unhandled ABI"); } MipsABIInfo MipsABIInfo::computeTargetABI(const Triple &TT, StringRef CPU, const MCTargetOptions &Options) { if (Options.getABIName().startswith("o32")) return MipsABIInfo::O32(); if (Options.getABIName().startswith("n32")) return MipsABIInfo::N32(); if (Options.getABIName().startswith("n64")) return MipsABIInfo::N64(); if (TT.getEnvironment() == llvm::Triple::GNUABIN32) return MipsABIInfo::N32(); assert(Options.getABIName().empty() && "Unknown ABI option for MIPS"); if (TT.isMIPS64()) return MipsABIInfo::N64(); return MipsABIInfo::O32(); } unsigned MipsABIInfo::GetStackPtr() const { return ArePtrs64bit() ? Mips::SP_64 : Mips::SP; } unsigned MipsABIInfo::GetFramePtr() const { return ArePtrs64bit() ? Mips::FP_64 : Mips::FP; } unsigned MipsABIInfo::GetBasePtr() const { return ArePtrs64bit() ? Mips::S7_64 : Mips::S7; } unsigned MipsABIInfo::GetGlobalPtr() const { return ArePtrs64bit() ? Mips::GP_64 : Mips::GP; } unsigned MipsABIInfo::GetNullPtr() const { return ArePtrs64bit() ? Mips::ZERO_64 : Mips::ZERO; } unsigned MipsABIInfo::GetZeroReg() const { return AreGprs64bit() ? Mips::ZERO_64 : Mips::ZERO; } unsigned MipsABIInfo::GetPtrAdduOp() const { return ArePtrs64bit() ? Mips::DADDu : Mips::ADDu; } unsigned MipsABIInfo::GetPtrAddiuOp() const { return ArePtrs64bit() ? Mips::DADDiu : Mips::ADDiu; } unsigned MipsABIInfo::GetPtrSubuOp() const { return ArePtrs64bit() ? Mips::DSUBu : Mips::SUBu; } unsigned MipsABIInfo::GetPtrAndOp() const { return ArePtrs64bit() ? Mips::AND64 : Mips::AND; } unsigned MipsABIInfo::GetGPRMoveOp() const { return ArePtrs64bit() ? Mips::OR64 : Mips::OR; } unsigned MipsABIInfo::GetEhDataReg(unsigned I) const { static const unsigned EhDataReg[] = { Mips::A0, Mips::A1, Mips::A2, Mips::A3 }; static const unsigned EhDataReg64[] = { Mips::A0_64, Mips::A1_64, Mips::A2_64, Mips::A3_64 }; return IsN64() ? EhDataReg64[I] : EhDataReg[I]; } Index: vendor/llvm/dist-release_80/lib/Target/Mips/MCTargetDesc/MipsBaseInfo.h =================================================================== --- vendor/llvm/dist-release_80/lib/Target/Mips/MCTargetDesc/MipsBaseInfo.h (revision 343793) +++ vendor/llvm/dist-release_80/lib/Target/Mips/MCTargetDesc/MipsBaseInfo.h (revision 343794) @@ -1,134 +1,137 @@ //===-- MipsBaseInfo.h - Top level definitions for MIPS MC ------*- C++ -*-===// // // The LLVM Compiler Infrastructure // // This file is distributed under the University of Illinois Open Source // License. See LICENSE.TXT for details. // //===----------------------------------------------------------------------===// // // This file contains small standalone helper functions and enum definitions for // the Mips target useful for the compiler back-end and the MC libraries. // //===----------------------------------------------------------------------===// #ifndef LLVM_LIB_TARGET_MIPS_MCTARGETDESC_MIPSBASEINFO_H #define LLVM_LIB_TARGET_MIPS_MCTARGETDESC_MIPSBASEINFO_H #include "MipsFixupKinds.h" #include "MipsMCTargetDesc.h" #include "llvm/MC/MCExpr.h" #include "llvm/Support/DataTypes.h" #include "llvm/Support/ErrorHandling.h" namespace llvm { /// MipsII - This namespace holds all of the target specific flags that /// instruction info tracks. /// namespace MipsII { /// Target Operand Flag enum. enum TOF { //===------------------------------------------------------------------===// // Mips Specific MachineOperand flags. MO_NO_FLAG, /// MO_GOT - Represents the offset into the global offset table at which /// the address the relocation entry symbol resides during execution. MO_GOT, /// MO_GOT_CALL - Represents the offset into the global offset table at /// which the address of a call site relocation entry symbol resides /// during execution. This is different from the above since this flag /// can only be present in call instructions. MO_GOT_CALL, /// MO_GPREL - Represents the offset from the current gp value to be used /// for the relocatable object file being produced. MO_GPREL, /// MO_ABS_HI/LO - Represents the hi or low part of an absolute symbol /// address. MO_ABS_HI, MO_ABS_LO, /// MO_TLSGD - Represents the offset into the global offset table at which // the module ID and TSL block offset reside during execution (General // Dynamic TLS). MO_TLSGD, /// MO_TLSLDM - Represents the offset into the global offset table at which // the module ID and TSL block offset reside during execution (Local // Dynamic TLS). MO_TLSLDM, MO_DTPREL_HI, MO_DTPREL_LO, /// MO_GOTTPREL - Represents the offset from the thread pointer (Initial // Exec TLS). MO_GOTTPREL, /// MO_TPREL_HI/LO - Represents the hi and low part of the offset from // the thread pointer (Local Exec TLS). MO_TPREL_HI, MO_TPREL_LO, // N32/64 Flags. MO_GPOFF_HI, MO_GPOFF_LO, MO_GOT_DISP, MO_GOT_PAGE, MO_GOT_OFST, /// MO_HIGHER/HIGHEST - Represents the highest or higher half word of a /// 64-bit symbol address. MO_HIGHER, MO_HIGHEST, /// MO_GOT_HI16/LO16, MO_CALL_HI16/LO16 - Relocations used for large GOTs. MO_GOT_HI16, MO_GOT_LO16, MO_CALL_HI16, - MO_CALL_LO16 + MO_CALL_LO16, + + /// Helper operand used to generate R_MIPS_JALR + MO_JALR }; enum { //===------------------------------------------------------------------===// // Instruction encodings. These are the standard/most common forms for // Mips instructions. // // Pseudo - This represents an instruction that is a pseudo instruction // or one that has not been implemented yet. It is illegal to code generate // it, but tolerated for intermediate implementation stages. Pseudo = 0, /// FrmR - This form is for instructions of the format R. FrmR = 1, /// FrmI - This form is for instructions of the format I. FrmI = 2, /// FrmJ - This form is for instructions of the format J. FrmJ = 3, /// FrmFR - This form is for instructions of the format FR. FrmFR = 4, /// FrmFI - This form is for instructions of the format FI. FrmFI = 5, /// FrmOther - This form is for instructions that have no specific format. FrmOther = 6, FormMask = 15, /// IsCTI - Instruction is a Control Transfer Instruction. IsCTI = 1 << 4, /// HasForbiddenSlot - Instruction has a forbidden slot. HasForbiddenSlot = 1 << 5, /// IsPCRelativeLoad - A Load instruction with implicit source register /// ($pc) with explicit offset and destination register IsPCRelativeLoad = 1 << 6, /// HasFCCRegOperand - Instruction uses an $fcc register. HasFCCRegOperand = 1 << 7 }; } } #endif Index: vendor/llvm/dist-release_80/lib/Target/Mips/MCTargetDesc/MipsMCCodeEmitter.cpp =================================================================== --- vendor/llvm/dist-release_80/lib/Target/Mips/MCTargetDesc/MipsMCCodeEmitter.cpp (revision 343793) +++ vendor/llvm/dist-release_80/lib/Target/Mips/MCTargetDesc/MipsMCCodeEmitter.cpp (revision 343794) @@ -1,1140 +1,1141 @@ //===-- MipsMCCodeEmitter.cpp - Convert Mips Code to Machine Code ---------===// // // The LLVM Compiler Infrastructure // // This file is distributed under the University of Illinois Open Source // License. See LICENSE.TXT for details. // //===----------------------------------------------------------------------===// // // This file implements the MipsMCCodeEmitter class. // //===----------------------------------------------------------------------===// #include "MipsMCCodeEmitter.h" #include "MCTargetDesc/MipsFixupKinds.h" #include "MCTargetDesc/MipsMCExpr.h" #include "MCTargetDesc/MipsMCTargetDesc.h" #include "llvm/ADT/APFloat.h" #include "llvm/ADT/APInt.h" #include "llvm/ADT/SmallVector.h" #include "llvm/MC/MCContext.h" #include "llvm/MC/MCExpr.h" #include "llvm/MC/MCFixup.h" #include "llvm/MC/MCInst.h" #include "llvm/MC/MCInstrDesc.h" #include "llvm/MC/MCInstrInfo.h" #include "llvm/MC/MCRegisterInfo.h" #include "llvm/MC/MCSubtargetInfo.h" #include "llvm/Support/Casting.h" #include "llvm/Support/ErrorHandling.h" #include "llvm/Support/raw_ostream.h" #include #include using namespace llvm; #define DEBUG_TYPE "mccodeemitter" #define GET_INSTRMAP_INFO #include "MipsGenInstrInfo.inc" #undef GET_INSTRMAP_INFO namespace llvm { MCCodeEmitter *createMipsMCCodeEmitterEB(const MCInstrInfo &MCII, const MCRegisterInfo &MRI, MCContext &Ctx) { return new MipsMCCodeEmitter(MCII, Ctx, false); } MCCodeEmitter *createMipsMCCodeEmitterEL(const MCInstrInfo &MCII, const MCRegisterInfo &MRI, MCContext &Ctx) { return new MipsMCCodeEmitter(MCII, Ctx, true); } } // end namespace llvm // If the D instruction has a shift amount that is greater // than 31 (checked in calling routine), lower it to a D32 instruction static void LowerLargeShift(MCInst& Inst) { assert(Inst.getNumOperands() == 3 && "Invalid no. of operands for shift!"); assert(Inst.getOperand(2).isImm()); int64_t Shift = Inst.getOperand(2).getImm(); if (Shift <= 31) return; // Do nothing Shift -= 32; // saminus32 Inst.getOperand(2).setImm(Shift); switch (Inst.getOpcode()) { default: // Calling function is not synchronized llvm_unreachable("Unexpected shift instruction"); case Mips::DSLL: Inst.setOpcode(Mips::DSLL32); return; case Mips::DSRL: Inst.setOpcode(Mips::DSRL32); return; case Mips::DSRA: Inst.setOpcode(Mips::DSRA32); return; case Mips::DROTR: Inst.setOpcode(Mips::DROTR32); return; } } // Fix a bad compact branch encoding for beqc/bnec. void MipsMCCodeEmitter::LowerCompactBranch(MCInst& Inst) const { // Encoding may be illegal !(rs < rt), but this situation is // easily fixed. unsigned RegOp0 = Inst.getOperand(0).getReg(); unsigned RegOp1 = Inst.getOperand(1).getReg(); unsigned Reg0 = Ctx.getRegisterInfo()->getEncodingValue(RegOp0); unsigned Reg1 = Ctx.getRegisterInfo()->getEncodingValue(RegOp1); if (Inst.getOpcode() == Mips::BNEC || Inst.getOpcode() == Mips::BEQC || Inst.getOpcode() == Mips::BNEC64 || Inst.getOpcode() == Mips::BEQC64) { assert(Reg0 != Reg1 && "Instruction has bad operands ($rs == $rt)!"); if (Reg0 < Reg1) return; } else if (Inst.getOpcode() == Mips::BNVC || Inst.getOpcode() == Mips::BOVC) { if (Reg0 >= Reg1) return; } else if (Inst.getOpcode() == Mips::BNVC_MMR6 || Inst.getOpcode() == Mips::BOVC_MMR6) { if (Reg1 >= Reg0) return; } else llvm_unreachable("Cannot rewrite unknown branch!"); Inst.getOperand(0).setReg(RegOp1); Inst.getOperand(1).setReg(RegOp0); } bool MipsMCCodeEmitter::isMicroMips(const MCSubtargetInfo &STI) const { return STI.getFeatureBits()[Mips::FeatureMicroMips]; } bool MipsMCCodeEmitter::isMips32r6(const MCSubtargetInfo &STI) const { return STI.getFeatureBits()[Mips::FeatureMips32r6]; } void MipsMCCodeEmitter::EmitByte(unsigned char C, raw_ostream &OS) const { OS << (char)C; } void MipsMCCodeEmitter::EmitInstruction(uint64_t Val, unsigned Size, const MCSubtargetInfo &STI, raw_ostream &OS) const { // Output the instruction encoding in little endian byte order. // Little-endian byte ordering: // mips32r2: 4 | 3 | 2 | 1 // microMIPS: 2 | 1 | 4 | 3 if (IsLittleEndian && Size == 4 && isMicroMips(STI)) { EmitInstruction(Val >> 16, 2, STI, OS); EmitInstruction(Val, 2, STI, OS); } else { for (unsigned i = 0; i < Size; ++i) { unsigned Shift = IsLittleEndian ? i * 8 : (Size - 1 - i) * 8; EmitByte((Val >> Shift) & 0xff, OS); } } } /// encodeInstruction - Emit the instruction. /// Size the instruction with Desc.getSize(). void MipsMCCodeEmitter:: encodeInstruction(const MCInst &MI, raw_ostream &OS, SmallVectorImpl &Fixups, const MCSubtargetInfo &STI) const { // Non-pseudo instructions that get changed for direct object // only based on operand values. // If this list of instructions get much longer we will move // the check to a function call. Until then, this is more efficient. MCInst TmpInst = MI; switch (MI.getOpcode()) { // If shift amount is >= 32 it the inst needs to be lowered further case Mips::DSLL: case Mips::DSRL: case Mips::DSRA: case Mips::DROTR: LowerLargeShift(TmpInst); break; // Compact branches, enforce encoding restrictions. case Mips::BEQC: case Mips::BNEC: case Mips::BEQC64: case Mips::BNEC64: case Mips::BOVC: case Mips::BOVC_MMR6: case Mips::BNVC: case Mips::BNVC_MMR6: LowerCompactBranch(TmpInst); } unsigned long N = Fixups.size(); uint32_t Binary = getBinaryCodeForInstr(TmpInst, Fixups, STI); // Check for unimplemented opcodes. // Unfortunately in MIPS both NOP and SLL will come in with Binary == 0 // so we have to special check for them. unsigned Opcode = TmpInst.getOpcode(); if ((Opcode != Mips::NOP) && (Opcode != Mips::SLL) && (Opcode != Mips::SLL_MM) && (Opcode != Mips::SLL_MMR6) && !Binary) llvm_unreachable("unimplemented opcode in encodeInstruction()"); int NewOpcode = -1; if (isMicroMips(STI)) { if (isMips32r6(STI)) { NewOpcode = Mips::MipsR62MicroMipsR6(Opcode, Mips::Arch_micromipsr6); if (NewOpcode == -1) NewOpcode = Mips::Std2MicroMipsR6(Opcode, Mips::Arch_micromipsr6); } else NewOpcode = Mips::Std2MicroMips(Opcode, Mips::Arch_micromips); // Check whether it is Dsp instruction. if (NewOpcode == -1) NewOpcode = Mips::Dsp2MicroMips(Opcode, Mips::Arch_mmdsp); if (NewOpcode != -1) { if (Fixups.size() > N) Fixups.pop_back(); Opcode = NewOpcode; TmpInst.setOpcode (NewOpcode); Binary = getBinaryCodeForInstr(TmpInst, Fixups, STI); } if (((MI.getOpcode() == Mips::MOVEP_MM) || (MI.getOpcode() == Mips::MOVEP_MMR6))) { unsigned RegPair = getMovePRegPairOpValue(MI, 0, Fixups, STI); Binary = (Binary & 0xFFFFFC7F) | (RegPair << 7); } } const MCInstrDesc &Desc = MCII.get(TmpInst.getOpcode()); // Get byte count of instruction unsigned Size = Desc.getSize(); if (!Size) llvm_unreachable("Desc.getSize() returns 0"); EmitInstruction(Binary, Size, STI, OS); } /// getBranchTargetOpValue - Return binary encoding of the branch /// target operand. If the machine operand requires relocation, /// record the relocation and return zero. unsigned MipsMCCodeEmitter:: getBranchTargetOpValue(const MCInst &MI, unsigned OpNo, SmallVectorImpl &Fixups, const MCSubtargetInfo &STI) const { const MCOperand &MO = MI.getOperand(OpNo); // If the destination is an immediate, divide by 4. if (MO.isImm()) return MO.getImm() >> 2; assert(MO.isExpr() && "getBranchTargetOpValue expects only expressions or immediates"); const MCExpr *FixupExpression = MCBinaryExpr::createAdd( MO.getExpr(), MCConstantExpr::create(-4, Ctx), Ctx); Fixups.push_back(MCFixup::create(0, FixupExpression, MCFixupKind(Mips::fixup_Mips_PC16))); return 0; } /// getBranchTargetOpValue1SImm16 - Return binary encoding of the branch /// target operand. If the machine operand requires relocation, /// record the relocation and return zero. unsigned MipsMCCodeEmitter:: getBranchTargetOpValue1SImm16(const MCInst &MI, unsigned OpNo, SmallVectorImpl &Fixups, const MCSubtargetInfo &STI) const { const MCOperand &MO = MI.getOperand(OpNo); // If the destination is an immediate, divide by 2. if (MO.isImm()) return MO.getImm() >> 1; assert(MO.isExpr() && "getBranchTargetOpValue expects only expressions or immediates"); const MCExpr *FixupExpression = MCBinaryExpr::createAdd( MO.getExpr(), MCConstantExpr::create(-4, Ctx), Ctx); Fixups.push_back(MCFixup::create(0, FixupExpression, MCFixupKind(Mips::fixup_Mips_PC16))); return 0; } /// getBranchTargetOpValueMMR6 - Return binary encoding of the branch /// target operand. If the machine operand requires relocation, /// record the relocation and return zero. unsigned MipsMCCodeEmitter:: getBranchTargetOpValueMMR6(const MCInst &MI, unsigned OpNo, SmallVectorImpl &Fixups, const MCSubtargetInfo &STI) const { const MCOperand &MO = MI.getOperand(OpNo); // If the destination is an immediate, divide by 2. if (MO.isImm()) return MO.getImm() >> 1; assert(MO.isExpr() && "getBranchTargetOpValueMMR6 expects only expressions or immediates"); const MCExpr *FixupExpression = MCBinaryExpr::createAdd( MO.getExpr(), MCConstantExpr::create(-2, Ctx), Ctx); Fixups.push_back(MCFixup::create(0, FixupExpression, MCFixupKind(Mips::fixup_Mips_PC16))); return 0; } /// getBranchTargetOpValueLsl2MMR6 - Return binary encoding of the branch /// target operand. If the machine operand requires relocation, /// record the relocation and return zero. unsigned MipsMCCodeEmitter:: getBranchTargetOpValueLsl2MMR6(const MCInst &MI, unsigned OpNo, SmallVectorImpl &Fixups, const MCSubtargetInfo &STI) const { const MCOperand &MO = MI.getOperand(OpNo); // If the destination is an immediate, divide by 4. if (MO.isImm()) return MO.getImm() >> 2; assert(MO.isExpr() && "getBranchTargetOpValueLsl2MMR6 expects only expressions or immediates"); const MCExpr *FixupExpression = MCBinaryExpr::createAdd( MO.getExpr(), MCConstantExpr::create(-4, Ctx), Ctx); Fixups.push_back(MCFixup::create(0, FixupExpression, MCFixupKind(Mips::fixup_Mips_PC16))); return 0; } /// getBranchTarget7OpValueMM - Return binary encoding of the microMIPS branch /// target operand. If the machine operand requires relocation, /// record the relocation and return zero. unsigned MipsMCCodeEmitter:: getBranchTarget7OpValueMM(const MCInst &MI, unsigned OpNo, SmallVectorImpl &Fixups, const MCSubtargetInfo &STI) const { const MCOperand &MO = MI.getOperand(OpNo); // If the destination is an immediate, divide by 2. if (MO.isImm()) return MO.getImm() >> 1; assert(MO.isExpr() && "getBranchTargetOpValueMM expects only expressions or immediates"); const MCExpr *Expr = MO.getExpr(); Fixups.push_back(MCFixup::create(0, Expr, MCFixupKind(Mips::fixup_MICROMIPS_PC7_S1))); return 0; } /// getBranchTargetOpValueMMPC10 - Return binary encoding of the microMIPS /// 10-bit branch target operand. If the machine operand requires relocation, /// record the relocation and return zero. unsigned MipsMCCodeEmitter:: getBranchTargetOpValueMMPC10(const MCInst &MI, unsigned OpNo, SmallVectorImpl &Fixups, const MCSubtargetInfo &STI) const { const MCOperand &MO = MI.getOperand(OpNo); // If the destination is an immediate, divide by 2. if (MO.isImm()) return MO.getImm() >> 1; assert(MO.isExpr() && "getBranchTargetOpValuePC10 expects only expressions or immediates"); const MCExpr *Expr = MO.getExpr(); Fixups.push_back(MCFixup::create(0, Expr, MCFixupKind(Mips::fixup_MICROMIPS_PC10_S1))); return 0; } /// getBranchTargetOpValue - Return binary encoding of the microMIPS branch /// target operand. If the machine operand requires relocation, /// record the relocation and return zero. unsigned MipsMCCodeEmitter:: getBranchTargetOpValueMM(const MCInst &MI, unsigned OpNo, SmallVectorImpl &Fixups, const MCSubtargetInfo &STI) const { const MCOperand &MO = MI.getOperand(OpNo); // If the destination is an immediate, divide by 2. if (MO.isImm()) return MO.getImm() >> 1; assert(MO.isExpr() && "getBranchTargetOpValueMM expects only expressions or immediates"); const MCExpr *Expr = MO.getExpr(); Fixups.push_back(MCFixup::create(0, Expr, MCFixupKind(Mips:: fixup_MICROMIPS_PC16_S1))); return 0; } /// getBranchTarget21OpValue - Return binary encoding of the branch /// target operand. If the machine operand requires relocation, /// record the relocation and return zero. unsigned MipsMCCodeEmitter:: getBranchTarget21OpValue(const MCInst &MI, unsigned OpNo, SmallVectorImpl &Fixups, const MCSubtargetInfo &STI) const { const MCOperand &MO = MI.getOperand(OpNo); // If the destination is an immediate, divide by 4. if (MO.isImm()) return MO.getImm() >> 2; assert(MO.isExpr() && "getBranchTarget21OpValue expects only expressions or immediates"); const MCExpr *FixupExpression = MCBinaryExpr::createAdd( MO.getExpr(), MCConstantExpr::create(-4, Ctx), Ctx); Fixups.push_back(MCFixup::create(0, FixupExpression, MCFixupKind(Mips::fixup_MIPS_PC21_S2))); return 0; } /// getBranchTarget21OpValueMM - Return binary encoding of the branch /// target operand for microMIPS. If the machine operand requires /// relocation, record the relocation and return zero. unsigned MipsMCCodeEmitter:: getBranchTarget21OpValueMM(const MCInst &MI, unsigned OpNo, SmallVectorImpl &Fixups, const MCSubtargetInfo &STI) const { const MCOperand &MO = MI.getOperand(OpNo); // If the destination is an immediate, divide by 4. if (MO.isImm()) return MO.getImm() >> 2; assert(MO.isExpr() && "getBranchTarget21OpValueMM expects only expressions or immediates"); const MCExpr *FixupExpression = MCBinaryExpr::createAdd( MO.getExpr(), MCConstantExpr::create(-4, Ctx), Ctx); Fixups.push_back(MCFixup::create(0, FixupExpression, MCFixupKind(Mips::fixup_MICROMIPS_PC21_S1))); return 0; } /// getBranchTarget26OpValue - Return binary encoding of the branch /// target operand. If the machine operand requires relocation, /// record the relocation and return zero. unsigned MipsMCCodeEmitter:: getBranchTarget26OpValue(const MCInst &MI, unsigned OpNo, SmallVectorImpl &Fixups, const MCSubtargetInfo &STI) const { const MCOperand &MO = MI.getOperand(OpNo); // If the destination is an immediate, divide by 4. if (MO.isImm()) return MO.getImm() >> 2; assert(MO.isExpr() && "getBranchTarget26OpValue expects only expressions or immediates"); const MCExpr *FixupExpression = MCBinaryExpr::createAdd( MO.getExpr(), MCConstantExpr::create(-4, Ctx), Ctx); Fixups.push_back(MCFixup::create(0, FixupExpression, MCFixupKind(Mips::fixup_MIPS_PC26_S2))); return 0; } /// getBranchTarget26OpValueMM - Return binary encoding of the branch /// target operand. If the machine operand requires relocation, /// record the relocation and return zero. unsigned MipsMCCodeEmitter::getBranchTarget26OpValueMM( const MCInst &MI, unsigned OpNo, SmallVectorImpl &Fixups, const MCSubtargetInfo &STI) const { const MCOperand &MO = MI.getOperand(OpNo); // If the destination is an immediate, divide by 2. if (MO.isImm()) return MO.getImm() >> 1; assert(MO.isExpr() && "getBranchTarget26OpValueMM expects only expressions or immediates"); const MCExpr *FixupExpression = MCBinaryExpr::createAdd( MO.getExpr(), MCConstantExpr::create(-4, Ctx), Ctx); Fixups.push_back(MCFixup::create(0, FixupExpression, MCFixupKind(Mips::fixup_MICROMIPS_PC26_S1))); return 0; } /// getJumpOffset16OpValue - Return binary encoding of the jump /// target operand. If the machine operand requires relocation, /// record the relocation and return zero. unsigned MipsMCCodeEmitter:: getJumpOffset16OpValue(const MCInst &MI, unsigned OpNo, SmallVectorImpl &Fixups, const MCSubtargetInfo &STI) const { const MCOperand &MO = MI.getOperand(OpNo); if (MO.isImm()) return MO.getImm(); assert(MO.isExpr() && "getJumpOffset16OpValue expects only expressions or an immediate"); // TODO: Push fixup. return 0; } /// getJumpTargetOpValue - Return binary encoding of the jump /// target operand. If the machine operand requires relocation, /// record the relocation and return zero. unsigned MipsMCCodeEmitter:: getJumpTargetOpValue(const MCInst &MI, unsigned OpNo, SmallVectorImpl &Fixups, const MCSubtargetInfo &STI) const { const MCOperand &MO = MI.getOperand(OpNo); // If the destination is an immediate, divide by 4. if (MO.isImm()) return MO.getImm()>>2; assert(MO.isExpr() && "getJumpTargetOpValue expects only expressions or an immediate"); const MCExpr *Expr = MO.getExpr(); Fixups.push_back(MCFixup::create(0, Expr, MCFixupKind(Mips::fixup_Mips_26))); return 0; } unsigned MipsMCCodeEmitter:: getJumpTargetOpValueMM(const MCInst &MI, unsigned OpNo, SmallVectorImpl &Fixups, const MCSubtargetInfo &STI) const { const MCOperand &MO = MI.getOperand(OpNo); // If the destination is an immediate, divide by 2. if (MO.isImm()) return MO.getImm() >> 1; assert(MO.isExpr() && "getJumpTargetOpValueMM expects only expressions or an immediate"); const MCExpr *Expr = MO.getExpr(); Fixups.push_back(MCFixup::create(0, Expr, MCFixupKind(Mips::fixup_MICROMIPS_26_S1))); return 0; } unsigned MipsMCCodeEmitter:: getUImm5Lsl2Encoding(const MCInst &MI, unsigned OpNo, SmallVectorImpl &Fixups, const MCSubtargetInfo &STI) const { const MCOperand &MO = MI.getOperand(OpNo); if (MO.isImm()) { // The immediate is encoded as 'immediate << 2'. unsigned Res = getMachineOpValue(MI, MO, Fixups, STI); assert((Res & 3) == 0); return Res >> 2; } assert(MO.isExpr() && "getUImm5Lsl2Encoding expects only expressions or an immediate"); return 0; } unsigned MipsMCCodeEmitter:: getSImm3Lsa2Value(const MCInst &MI, unsigned OpNo, SmallVectorImpl &Fixups, const MCSubtargetInfo &STI) const { const MCOperand &MO = MI.getOperand(OpNo); if (MO.isImm()) { int Value = MO.getImm(); return Value >> 2; } return 0; } unsigned MipsMCCodeEmitter:: getUImm6Lsl2Encoding(const MCInst &MI, unsigned OpNo, SmallVectorImpl &Fixups, const MCSubtargetInfo &STI) const { const MCOperand &MO = MI.getOperand(OpNo); if (MO.isImm()) { unsigned Value = MO.getImm(); return Value >> 2; } return 0; } unsigned MipsMCCodeEmitter:: getSImm9AddiuspValue(const MCInst &MI, unsigned OpNo, SmallVectorImpl &Fixups, const MCSubtargetInfo &STI) const { const MCOperand &MO = MI.getOperand(OpNo); if (MO.isImm()) { unsigned Binary = (MO.getImm() >> 2) & 0x0000ffff; return (((Binary & 0x8000) >> 7) | (Binary & 0x00ff)); } return 0; } unsigned MipsMCCodeEmitter:: getExprOpValue(const MCExpr *Expr, SmallVectorImpl &Fixups, const MCSubtargetInfo &STI) const { int64_t Res; if (Expr->evaluateAsAbsolute(Res)) return Res; MCExpr::ExprKind Kind = Expr->getKind(); if (Kind == MCExpr::Constant) { return cast(Expr)->getValue(); } if (Kind == MCExpr::Binary) { unsigned Res = getExprOpValue(cast(Expr)->getLHS(), Fixups, STI); Res += getExprOpValue(cast(Expr)->getRHS(), Fixups, STI); return Res; } if (Kind == MCExpr::Target) { const MipsMCExpr *MipsExpr = cast(Expr); Mips::Fixups FixupKind = Mips::Fixups(0); switch (MipsExpr->getKind()) { case MipsMCExpr::MEK_None: case MipsMCExpr::MEK_Special: llvm_unreachable("Unhandled fixup kind!"); break; case MipsMCExpr::MEK_DTPREL: - llvm_unreachable("MEK_DTPREL is used for TLS DIEExpr only"); - break; + // MEK_DTPREL is used for marking TLS DIEExpr only + // and contains a regular sub-expression. + return getExprOpValue(MipsExpr->getSubExpr(), Fixups, STI); case MipsMCExpr::MEK_CALL_HI16: FixupKind = Mips::fixup_Mips_CALL_HI16; break; case MipsMCExpr::MEK_CALL_LO16: FixupKind = Mips::fixup_Mips_CALL_LO16; break; case MipsMCExpr::MEK_DTPREL_HI: FixupKind = isMicroMips(STI) ? Mips::fixup_MICROMIPS_TLS_DTPREL_HI16 : Mips::fixup_Mips_DTPREL_HI; break; case MipsMCExpr::MEK_DTPREL_LO: FixupKind = isMicroMips(STI) ? Mips::fixup_MICROMIPS_TLS_DTPREL_LO16 : Mips::fixup_Mips_DTPREL_LO; break; case MipsMCExpr::MEK_GOTTPREL: FixupKind = isMicroMips(STI) ? Mips::fixup_MICROMIPS_GOTTPREL : Mips::fixup_Mips_GOTTPREL; break; case MipsMCExpr::MEK_GOT: FixupKind = isMicroMips(STI) ? Mips::fixup_MICROMIPS_GOT16 : Mips::fixup_Mips_GOT; break; case MipsMCExpr::MEK_GOT_CALL: FixupKind = isMicroMips(STI) ? Mips::fixup_MICROMIPS_CALL16 : Mips::fixup_Mips_CALL16; break; case MipsMCExpr::MEK_GOT_DISP: FixupKind = isMicroMips(STI) ? Mips::fixup_MICROMIPS_GOT_DISP : Mips::fixup_Mips_GOT_DISP; break; case MipsMCExpr::MEK_GOT_HI16: FixupKind = Mips::fixup_Mips_GOT_HI16; break; case MipsMCExpr::MEK_GOT_LO16: FixupKind = Mips::fixup_Mips_GOT_LO16; break; case MipsMCExpr::MEK_GOT_PAGE: FixupKind = isMicroMips(STI) ? Mips::fixup_MICROMIPS_GOT_PAGE : Mips::fixup_Mips_GOT_PAGE; break; case MipsMCExpr::MEK_GOT_OFST: FixupKind = isMicroMips(STI) ? Mips::fixup_MICROMIPS_GOT_OFST : Mips::fixup_Mips_GOT_OFST; break; case MipsMCExpr::MEK_GPREL: FixupKind = Mips::fixup_Mips_GPREL16; break; case MipsMCExpr::MEK_LO: // Check for %lo(%neg(%gp_rel(X))) if (MipsExpr->isGpOff()) FixupKind = isMicroMips(STI) ? Mips::fixup_MICROMIPS_GPOFF_LO : Mips::fixup_Mips_GPOFF_LO; else FixupKind = isMicroMips(STI) ? Mips::fixup_MICROMIPS_LO16 : Mips::fixup_Mips_LO16; break; case MipsMCExpr::MEK_HIGHEST: FixupKind = isMicroMips(STI) ? Mips::fixup_MICROMIPS_HIGHEST : Mips::fixup_Mips_HIGHEST; break; case MipsMCExpr::MEK_HIGHER: FixupKind = isMicroMips(STI) ? Mips::fixup_MICROMIPS_HIGHER : Mips::fixup_Mips_HIGHER; break; case MipsMCExpr::MEK_HI: // Check for %hi(%neg(%gp_rel(X))) if (MipsExpr->isGpOff()) FixupKind = isMicroMips(STI) ? Mips::fixup_MICROMIPS_GPOFF_HI : Mips::fixup_Mips_GPOFF_HI; else FixupKind = isMicroMips(STI) ? Mips::fixup_MICROMIPS_HI16 : Mips::fixup_Mips_HI16; break; case MipsMCExpr::MEK_PCREL_HI16: FixupKind = Mips::fixup_MIPS_PCHI16; break; case MipsMCExpr::MEK_PCREL_LO16: FixupKind = Mips::fixup_MIPS_PCLO16; break; case MipsMCExpr::MEK_TLSGD: FixupKind = isMicroMips(STI) ? Mips::fixup_MICROMIPS_TLS_GD : Mips::fixup_Mips_TLSGD; break; case MipsMCExpr::MEK_TLSLDM: FixupKind = isMicroMips(STI) ? Mips::fixup_MICROMIPS_TLS_LDM : Mips::fixup_Mips_TLSLDM; break; case MipsMCExpr::MEK_TPREL_HI: FixupKind = isMicroMips(STI) ? Mips::fixup_MICROMIPS_TLS_TPREL_HI16 : Mips::fixup_Mips_TPREL_HI; break; case MipsMCExpr::MEK_TPREL_LO: FixupKind = isMicroMips(STI) ? Mips::fixup_MICROMIPS_TLS_TPREL_LO16 : Mips::fixup_Mips_TPREL_LO; break; case MipsMCExpr::MEK_NEG: FixupKind = isMicroMips(STI) ? Mips::fixup_MICROMIPS_SUB : Mips::fixup_Mips_SUB; break; } Fixups.push_back(MCFixup::create(0, MipsExpr, MCFixupKind(FixupKind))); return 0; } if (Kind == MCExpr::SymbolRef) { Mips::Fixups FixupKind = Mips::Fixups(0); switch(cast(Expr)->getKind()) { default: llvm_unreachable("Unknown fixup kind!"); break; case MCSymbolRefExpr::VK_None: FixupKind = Mips::fixup_Mips_32; // FIXME: This is ok for O32/N32 but not N64. break; } // switch Fixups.push_back(MCFixup::create(0, Expr, MCFixupKind(FixupKind))); return 0; } return 0; } /// getMachineOpValue - Return binary encoding of operand. If the machine /// operand requires relocation, record the relocation and return zero. unsigned MipsMCCodeEmitter:: getMachineOpValue(const MCInst &MI, const MCOperand &MO, SmallVectorImpl &Fixups, const MCSubtargetInfo &STI) const { if (MO.isReg()) { unsigned Reg = MO.getReg(); unsigned RegNo = Ctx.getRegisterInfo()->getEncodingValue(Reg); return RegNo; } else if (MO.isImm()) { return static_cast(MO.getImm()); } else if (MO.isFPImm()) { return static_cast(APFloat(MO.getFPImm()) .bitcastToAPInt().getHiBits(32).getLimitedValue()); } // MO must be an Expr. assert(MO.isExpr()); return getExprOpValue(MO.getExpr(),Fixups, STI); } /// Return binary encoding of memory related operand. /// If the offset operand requires relocation, record the relocation. template unsigned MipsMCCodeEmitter::getMemEncoding(const MCInst &MI, unsigned OpNo, SmallVectorImpl &Fixups, const MCSubtargetInfo &STI) const { // Base register is encoded in bits 20-16, offset is encoded in bits 15-0. assert(MI.getOperand(OpNo).isReg()); unsigned RegBits = getMachineOpValue(MI, MI.getOperand(OpNo),Fixups, STI) << 16; unsigned OffBits = getMachineOpValue(MI, MI.getOperand(OpNo+1), Fixups, STI); // Apply the scale factor if there is one. OffBits >>= ShiftAmount; return (OffBits & 0xFFFF) | RegBits; } unsigned MipsMCCodeEmitter:: getMemEncodingMMImm4(const MCInst &MI, unsigned OpNo, SmallVectorImpl &Fixups, const MCSubtargetInfo &STI) const { // Base register is encoded in bits 6-4, offset is encoded in bits 3-0. assert(MI.getOperand(OpNo).isReg()); unsigned RegBits = getMachineOpValue(MI, MI.getOperand(OpNo), Fixups, STI) << 4; unsigned OffBits = getMachineOpValue(MI, MI.getOperand(OpNo+1), Fixups, STI); return (OffBits & 0xF) | RegBits; } unsigned MipsMCCodeEmitter:: getMemEncodingMMImm4Lsl1(const MCInst &MI, unsigned OpNo, SmallVectorImpl &Fixups, const MCSubtargetInfo &STI) const { // Base register is encoded in bits 6-4, offset is encoded in bits 3-0. assert(MI.getOperand(OpNo).isReg()); unsigned RegBits = getMachineOpValue(MI, MI.getOperand(OpNo), Fixups, STI) << 4; unsigned OffBits = getMachineOpValue(MI, MI.getOperand(OpNo+1), Fixups, STI) >> 1; return (OffBits & 0xF) | RegBits; } unsigned MipsMCCodeEmitter:: getMemEncodingMMImm4Lsl2(const MCInst &MI, unsigned OpNo, SmallVectorImpl &Fixups, const MCSubtargetInfo &STI) const { // Base register is encoded in bits 6-4, offset is encoded in bits 3-0. assert(MI.getOperand(OpNo).isReg()); unsigned RegBits = getMachineOpValue(MI, MI.getOperand(OpNo), Fixups, STI) << 4; unsigned OffBits = getMachineOpValue(MI, MI.getOperand(OpNo+1), Fixups, STI) >> 2; return (OffBits & 0xF) | RegBits; } unsigned MipsMCCodeEmitter:: getMemEncodingMMSPImm5Lsl2(const MCInst &MI, unsigned OpNo, SmallVectorImpl &Fixups, const MCSubtargetInfo &STI) const { // Register is encoded in bits 9-5, offset is encoded in bits 4-0. assert(MI.getOperand(OpNo).isReg() && (MI.getOperand(OpNo).getReg() == Mips::SP || MI.getOperand(OpNo).getReg() == Mips::SP_64) && "Unexpected base register!"); unsigned OffBits = getMachineOpValue(MI, MI.getOperand(OpNo+1), Fixups, STI) >> 2; return OffBits & 0x1F; } unsigned MipsMCCodeEmitter:: getMemEncodingMMGPImm7Lsl2(const MCInst &MI, unsigned OpNo, SmallVectorImpl &Fixups, const MCSubtargetInfo &STI) const { // Register is encoded in bits 9-7, offset is encoded in bits 6-0. assert(MI.getOperand(OpNo).isReg() && MI.getOperand(OpNo).getReg() == Mips::GP && "Unexpected base register!"); unsigned OffBits = getMachineOpValue(MI, MI.getOperand(OpNo+1), Fixups, STI) >> 2; return OffBits & 0x7F; } unsigned MipsMCCodeEmitter:: getMemEncodingMMImm9(const MCInst &MI, unsigned OpNo, SmallVectorImpl &Fixups, const MCSubtargetInfo &STI) const { // Base register is encoded in bits 20-16, offset is encoded in bits 8-0. assert(MI.getOperand(OpNo).isReg()); unsigned RegBits = getMachineOpValue(MI, MI.getOperand(OpNo), Fixups, STI) << 16; unsigned OffBits = getMachineOpValue(MI, MI.getOperand(OpNo + 1), Fixups, STI); return (OffBits & 0x1FF) | RegBits; } unsigned MipsMCCodeEmitter:: getMemEncodingMMImm11(const MCInst &MI, unsigned OpNo, SmallVectorImpl &Fixups, const MCSubtargetInfo &STI) const { // Base register is encoded in bits 20-16, offset is encoded in bits 10-0. assert(MI.getOperand(OpNo).isReg()); unsigned RegBits = getMachineOpValue(MI, MI.getOperand(OpNo), Fixups, STI) << 16; unsigned OffBits = getMachineOpValue(MI, MI.getOperand(OpNo+1), Fixups, STI); return (OffBits & 0x07FF) | RegBits; } unsigned MipsMCCodeEmitter:: getMemEncodingMMImm12(const MCInst &MI, unsigned OpNo, SmallVectorImpl &Fixups, const MCSubtargetInfo &STI) const { // opNum can be invalid if instruction had reglist as operand. // MemOperand is always last operand of instruction (base + offset). switch (MI.getOpcode()) { default: break; case Mips::SWM32_MM: case Mips::LWM32_MM: OpNo = MI.getNumOperands() - 2; break; } // Base register is encoded in bits 20-16, offset is encoded in bits 11-0. assert(MI.getOperand(OpNo).isReg()); unsigned RegBits = getMachineOpValue(MI, MI.getOperand(OpNo), Fixups, STI) << 16; unsigned OffBits = getMachineOpValue(MI, MI.getOperand(OpNo+1), Fixups, STI); return (OffBits & 0x0FFF) | RegBits; } unsigned MipsMCCodeEmitter:: getMemEncodingMMImm16(const MCInst &MI, unsigned OpNo, SmallVectorImpl &Fixups, const MCSubtargetInfo &STI) const { // Base register is encoded in bits 20-16, offset is encoded in bits 15-0. assert(MI.getOperand(OpNo).isReg()); unsigned RegBits = getMachineOpValue(MI, MI.getOperand(OpNo), Fixups, STI) << 16; unsigned OffBits = getMachineOpValue(MI, MI.getOperand(OpNo+1), Fixups, STI); return (OffBits & 0xFFFF) | RegBits; } unsigned MipsMCCodeEmitter:: getMemEncodingMMImm4sp(const MCInst &MI, unsigned OpNo, SmallVectorImpl &Fixups, const MCSubtargetInfo &STI) const { // opNum can be invalid if instruction had reglist as operand // MemOperand is always last operand of instruction (base + offset) switch (MI.getOpcode()) { default: break; case Mips::SWM16_MM: case Mips::SWM16_MMR6: case Mips::LWM16_MM: case Mips::LWM16_MMR6: OpNo = MI.getNumOperands() - 2; break; } // Offset is encoded in bits 4-0. assert(MI.getOperand(OpNo).isReg()); // Base register is always SP - thus it is not encoded. assert(MI.getOperand(OpNo+1).isImm()); unsigned OffBits = getMachineOpValue(MI, MI.getOperand(OpNo+1), Fixups, STI); return ((OffBits >> 2) & 0x0F); } // FIXME: should be called getMSBEncoding // unsigned MipsMCCodeEmitter::getSizeInsEncoding(const MCInst &MI, unsigned OpNo, SmallVectorImpl &Fixups, const MCSubtargetInfo &STI) const { assert(MI.getOperand(OpNo-1).isImm()); assert(MI.getOperand(OpNo).isImm()); unsigned Position = getMachineOpValue(MI, MI.getOperand(OpNo-1), Fixups, STI); unsigned Size = getMachineOpValue(MI, MI.getOperand(OpNo), Fixups, STI); return Position + Size - 1; } template unsigned MipsMCCodeEmitter::getUImmWithOffsetEncoding(const MCInst &MI, unsigned OpNo, SmallVectorImpl &Fixups, const MCSubtargetInfo &STI) const { assert(MI.getOperand(OpNo).isImm()); unsigned Value = getMachineOpValue(MI, MI.getOperand(OpNo), Fixups, STI); Value -= Offset; return Value; } unsigned MipsMCCodeEmitter::getSimm19Lsl2Encoding(const MCInst &MI, unsigned OpNo, SmallVectorImpl &Fixups, const MCSubtargetInfo &STI) const { const MCOperand &MO = MI.getOperand(OpNo); if (MO.isImm()) { // The immediate is encoded as 'immediate << 2'. unsigned Res = getMachineOpValue(MI, MO, Fixups, STI); assert((Res & 3) == 0); return Res >> 2; } assert(MO.isExpr() && "getSimm19Lsl2Encoding expects only expressions or an immediate"); const MCExpr *Expr = MO.getExpr(); Mips::Fixups FixupKind = isMicroMips(STI) ? Mips::fixup_MICROMIPS_PC19_S2 : Mips::fixup_MIPS_PC19_S2; Fixups.push_back(MCFixup::create(0, Expr, MCFixupKind(FixupKind))); return 0; } unsigned MipsMCCodeEmitter::getSimm18Lsl3Encoding(const MCInst &MI, unsigned OpNo, SmallVectorImpl &Fixups, const MCSubtargetInfo &STI) const { const MCOperand &MO = MI.getOperand(OpNo); if (MO.isImm()) { // The immediate is encoded as 'immediate << 3'. unsigned Res = getMachineOpValue(MI, MI.getOperand(OpNo), Fixups, STI); assert((Res & 7) == 0); return Res >> 3; } assert(MO.isExpr() && "getSimm18Lsl2Encoding expects only expressions or an immediate"); const MCExpr *Expr = MO.getExpr(); Mips::Fixups FixupKind = isMicroMips(STI) ? Mips::fixup_MICROMIPS_PC18_S3 : Mips::fixup_MIPS_PC18_S3; Fixups.push_back(MCFixup::create(0, Expr, MCFixupKind(FixupKind))); return 0; } unsigned MipsMCCodeEmitter::getUImm3Mod8Encoding(const MCInst &MI, unsigned OpNo, SmallVectorImpl &Fixups, const MCSubtargetInfo &STI) const { assert(MI.getOperand(OpNo).isImm()); const MCOperand &MO = MI.getOperand(OpNo); return MO.getImm() % 8; } unsigned MipsMCCodeEmitter::getUImm4AndValue(const MCInst &MI, unsigned OpNo, SmallVectorImpl &Fixups, const MCSubtargetInfo &STI) const { assert(MI.getOperand(OpNo).isImm()); const MCOperand &MO = MI.getOperand(OpNo); unsigned Value = MO.getImm(); switch (Value) { case 128: return 0x0; case 1: return 0x1; case 2: return 0x2; case 3: return 0x3; case 4: return 0x4; case 7: return 0x5; case 8: return 0x6; case 15: return 0x7; case 16: return 0x8; case 31: return 0x9; case 32: return 0xa; case 63: return 0xb; case 64: return 0xc; case 255: return 0xd; case 32768: return 0xe; case 65535: return 0xf; } llvm_unreachable("Unexpected value"); } unsigned MipsMCCodeEmitter::getRegisterListOpValue(const MCInst &MI, unsigned OpNo, SmallVectorImpl &Fixups, const MCSubtargetInfo &STI) const { unsigned res = 0; // Register list operand is always first operand of instruction and it is // placed before memory operand (register + imm). for (unsigned I = OpNo, E = MI.getNumOperands() - 2; I < E; ++I) { unsigned Reg = MI.getOperand(I).getReg(); unsigned RegNo = Ctx.getRegisterInfo()->getEncodingValue(Reg); if (RegNo != 31) res++; else res |= 0x10; } return res; } unsigned MipsMCCodeEmitter::getRegisterListOpValue16(const MCInst &MI, unsigned OpNo, SmallVectorImpl &Fixups, const MCSubtargetInfo &STI) const { return (MI.getNumOperands() - 4); } unsigned MipsMCCodeEmitter::getMovePRegPairOpValue(const MCInst &MI, unsigned OpNo, SmallVectorImpl &Fixups, const MCSubtargetInfo &STI) const { unsigned res = 0; if (MI.getOperand(0).getReg() == Mips::A1 && MI.getOperand(1).getReg() == Mips::A2) res = 0; else if (MI.getOperand(0).getReg() == Mips::A1 && MI.getOperand(1).getReg() == Mips::A3) res = 1; else if (MI.getOperand(0).getReg() == Mips::A2 && MI.getOperand(1).getReg() == Mips::A3) res = 2; else if (MI.getOperand(0).getReg() == Mips::A0 && MI.getOperand(1).getReg() == Mips::S5) res = 3; else if (MI.getOperand(0).getReg() == Mips::A0 && MI.getOperand(1).getReg() == Mips::S6) res = 4; else if (MI.getOperand(0).getReg() == Mips::A0 && MI.getOperand(1).getReg() == Mips::A1) res = 5; else if (MI.getOperand(0).getReg() == Mips::A0 && MI.getOperand(1).getReg() == Mips::A2) res = 6; else if (MI.getOperand(0).getReg() == Mips::A0 && MI.getOperand(1).getReg() == Mips::A3) res = 7; return res; } unsigned MipsMCCodeEmitter::getMovePRegSingleOpValue(const MCInst &MI, unsigned OpNo, SmallVectorImpl &Fixups, const MCSubtargetInfo &STI) const { assert(((OpNo == 2) || (OpNo == 3)) && "Unexpected OpNo for movep operand encoding!"); MCOperand Op = MI.getOperand(OpNo); assert(Op.isReg() && "Operand of movep is not a register!"); switch (Op.getReg()) { default: llvm_unreachable("Unknown register for movep!"); case Mips::ZERO: return 0; case Mips::S1: return 1; case Mips::V0: return 2; case Mips::V1: return 3; case Mips::S0: return 4; case Mips::S2: return 5; case Mips::S3: return 6; case Mips::S4: return 7; } } unsigned MipsMCCodeEmitter::getSimm23Lsl2Encoding(const MCInst &MI, unsigned OpNo, SmallVectorImpl &Fixups, const MCSubtargetInfo &STI) const { const MCOperand &MO = MI.getOperand(OpNo); assert(MO.isImm() && "getSimm23Lsl2Encoding expects only an immediate"); // The immediate is encoded as 'immediate >> 2'. unsigned Res = static_cast(MO.getImm()); assert((Res & 3) == 0); return Res >> 2; } #include "MipsGenMCCodeEmitter.inc" Index: vendor/llvm/dist-release_80/lib/Target/Mips/MCTargetDesc/MipsMCExpr.cpp =================================================================== --- vendor/llvm/dist-release_80/lib/Target/Mips/MCTargetDesc/MipsMCExpr.cpp (revision 343793) +++ vendor/llvm/dist-release_80/lib/Target/Mips/MCTargetDesc/MipsMCExpr.cpp (revision 343794) @@ -1,301 +1,303 @@ //===-- MipsMCExpr.cpp - Mips specific MC expression classes --------------===// // // The LLVM Compiler Infrastructure // // This file is distributed under the University of Illinois Open Source // License. See LICENSE.TXT for details. // //===----------------------------------------------------------------------===// #include "MipsMCExpr.h" #include "llvm/BinaryFormat/ELF.h" #include "llvm/MC/MCAsmInfo.h" #include "llvm/MC/MCAssembler.h" #include "llvm/MC/MCContext.h" #include "llvm/MC/MCStreamer.h" #include "llvm/MC/MCSymbolELF.h" #include "llvm/MC/MCValue.h" #include "llvm/Support/Casting.h" #include "llvm/Support/ErrorHandling.h" #include "llvm/Support/MathExtras.h" #include "llvm/Support/raw_ostream.h" #include using namespace llvm; #define DEBUG_TYPE "mipsmcexpr" const MipsMCExpr *MipsMCExpr::create(MipsMCExpr::MipsExprKind Kind, const MCExpr *Expr, MCContext &Ctx) { return new (Ctx) MipsMCExpr(Kind, Expr); } const MipsMCExpr *MipsMCExpr::createGpOff(MipsMCExpr::MipsExprKind Kind, const MCExpr *Expr, MCContext &Ctx) { return create(Kind, create(MEK_NEG, create(MEK_GPREL, Expr, Ctx), Ctx), Ctx); } void MipsMCExpr::printImpl(raw_ostream &OS, const MCAsmInfo *MAI) const { int64_t AbsVal; switch (Kind) { case MEK_None: case MEK_Special: llvm_unreachable("MEK_None and MEK_Special are invalid"); break; case MEK_DTPREL: - llvm_unreachable("MEK_DTPREL is used for TLS DIEExpr only"); - break; + // MEK_DTPREL is used for marking TLS DIEExpr only + // and contains a regular sub-expression. + getSubExpr()->print(OS, MAI, true); + return; case MEK_CALL_HI16: OS << "%call_hi"; break; case MEK_CALL_LO16: OS << "%call_lo"; break; case MEK_DTPREL_HI: OS << "%dtprel_hi"; break; case MEK_DTPREL_LO: OS << "%dtprel_lo"; break; case MEK_GOT: OS << "%got"; break; case MEK_GOTTPREL: OS << "%gottprel"; break; case MEK_GOT_CALL: OS << "%call16"; break; case MEK_GOT_DISP: OS << "%got_disp"; break; case MEK_GOT_HI16: OS << "%got_hi"; break; case MEK_GOT_LO16: OS << "%got_lo"; break; case MEK_GOT_PAGE: OS << "%got_page"; break; case MEK_GOT_OFST: OS << "%got_ofst"; break; case MEK_GPREL: OS << "%gp_rel"; break; case MEK_HI: OS << "%hi"; break; case MEK_HIGHER: OS << "%higher"; break; case MEK_HIGHEST: OS << "%highest"; break; case MEK_LO: OS << "%lo"; break; case MEK_NEG: OS << "%neg"; break; case MEK_PCREL_HI16: OS << "%pcrel_hi"; break; case MEK_PCREL_LO16: OS << "%pcrel_lo"; break; case MEK_TLSGD: OS << "%tlsgd"; break; case MEK_TLSLDM: OS << "%tlsldm"; break; case MEK_TPREL_HI: OS << "%tprel_hi"; break; case MEK_TPREL_LO: OS << "%tprel_lo"; break; } OS << '('; if (Expr->evaluateAsAbsolute(AbsVal)) OS << AbsVal; else Expr->print(OS, MAI, true); OS << ')'; } bool MipsMCExpr::evaluateAsRelocatableImpl(MCValue &Res, const MCAsmLayout *Layout, const MCFixup *Fixup) const { // Look for the %hi(%neg(%gp_rel(X))) and %lo(%neg(%gp_rel(X))) special cases. if (isGpOff()) { const MCExpr *SubExpr = cast(cast(getSubExpr())->getSubExpr()) ->getSubExpr(); if (!SubExpr->evaluateAsRelocatable(Res, Layout, Fixup)) return false; Res = MCValue::get(Res.getSymA(), Res.getSymB(), Res.getConstant(), MEK_Special); return true; } if (!getSubExpr()->evaluateAsRelocatable(Res, Layout, Fixup)) return false; if (Res.getRefKind() != MCSymbolRefExpr::VK_None) return false; // evaluateAsAbsolute() and evaluateAsValue() require that we evaluate the // %hi/%lo/etc. here. Fixup is a null pointer when either of these is the // caller. if (Res.isAbsolute() && Fixup == nullptr) { int64_t AbsVal = Res.getConstant(); switch (Kind) { case MEK_None: case MEK_Special: llvm_unreachable("MEK_None and MEK_Special are invalid"); case MEK_DTPREL: - llvm_unreachable("MEK_DTPREL is used for TLS DIEExpr only"); + // MEK_DTPREL is used for marking TLS DIEExpr only + // and contains a regular sub-expression. + return getSubExpr()->evaluateAsRelocatable(Res, Layout, Fixup); case MEK_DTPREL_HI: case MEK_DTPREL_LO: case MEK_GOT: case MEK_GOTTPREL: case MEK_GOT_CALL: case MEK_GOT_DISP: case MEK_GOT_HI16: case MEK_GOT_LO16: case MEK_GOT_OFST: case MEK_GOT_PAGE: case MEK_GPREL: case MEK_PCREL_HI16: case MEK_PCREL_LO16: case MEK_TLSGD: case MEK_TLSLDM: case MEK_TPREL_HI: case MEK_TPREL_LO: return false; case MEK_LO: case MEK_CALL_LO16: AbsVal = SignExtend64<16>(AbsVal); break; case MEK_CALL_HI16: case MEK_HI: AbsVal = SignExtend64<16>((AbsVal + 0x8000) >> 16); break; case MEK_HIGHER: AbsVal = SignExtend64<16>((AbsVal + 0x80008000LL) >> 32); break; case MEK_HIGHEST: AbsVal = SignExtend64<16>((AbsVal + 0x800080008000LL) >> 48); break; case MEK_NEG: AbsVal = -AbsVal; break; } Res = MCValue::get(AbsVal); return true; } // We want to defer it for relocatable expressions since the constant is // applied to the whole symbol value. // // The value of getKind() that is given to MCValue is only intended to aid // debugging when inspecting MCValue objects. It shouldn't be relied upon // for decision making. Res = MCValue::get(Res.getSymA(), Res.getSymB(), Res.getConstant(), getKind()); return true; } void MipsMCExpr::visitUsedExpr(MCStreamer &Streamer) const { Streamer.visitUsedExpr(*getSubExpr()); } static void fixELFSymbolsInTLSFixupsImpl(const MCExpr *Expr, MCAssembler &Asm) { switch (Expr->getKind()) { case MCExpr::Target: fixELFSymbolsInTLSFixupsImpl(cast(Expr)->getSubExpr(), Asm); break; case MCExpr::Constant: break; case MCExpr::Binary: { const MCBinaryExpr *BE = cast(Expr); fixELFSymbolsInTLSFixupsImpl(BE->getLHS(), Asm); fixELFSymbolsInTLSFixupsImpl(BE->getRHS(), Asm); break; } case MCExpr::SymbolRef: { // We're known to be under a TLS fixup, so any symbol should be // modified. There should be only one. const MCSymbolRefExpr &SymRef = *cast(Expr); cast(SymRef.getSymbol()).setType(ELF::STT_TLS); break; } case MCExpr::Unary: fixELFSymbolsInTLSFixupsImpl(cast(Expr)->getSubExpr(), Asm); break; } } void MipsMCExpr::fixELFSymbolsInTLSFixups(MCAssembler &Asm) const { switch (getKind()) { case MEK_None: case MEK_Special: llvm_unreachable("MEK_None and MEK_Special are invalid"); break; - case MEK_DTPREL: - llvm_unreachable("MEK_DTPREL is used for TLS DIEExpr only"); - break; case MEK_CALL_HI16: case MEK_CALL_LO16: case MEK_GOT: case MEK_GOT_CALL: case MEK_GOT_DISP: case MEK_GOT_HI16: case MEK_GOT_LO16: case MEK_GOT_OFST: case MEK_GOT_PAGE: case MEK_GPREL: case MEK_HI: case MEK_HIGHER: case MEK_HIGHEST: case MEK_LO: case MEK_NEG: case MEK_PCREL_HI16: case MEK_PCREL_LO16: // If we do have nested target-specific expressions, they will be in // a consecutive chain. if (const MipsMCExpr *E = dyn_cast(getSubExpr())) E->fixELFSymbolsInTLSFixups(Asm); break; + case MEK_DTPREL: case MEK_DTPREL_HI: case MEK_DTPREL_LO: case MEK_TLSLDM: case MEK_TLSGD: case MEK_GOTTPREL: case MEK_TPREL_HI: case MEK_TPREL_LO: fixELFSymbolsInTLSFixupsImpl(getSubExpr(), Asm); break; } } bool MipsMCExpr::isGpOff(MipsExprKind &Kind) const { if (getKind() == MEK_HI || getKind() == MEK_LO) { if (const MipsMCExpr *S1 = dyn_cast(getSubExpr())) { if (const MipsMCExpr *S2 = dyn_cast(S1->getSubExpr())) { if (S1->getKind() == MEK_NEG && S2->getKind() == MEK_GPREL) { Kind = getKind(); return true; } } } } return false; } Index: vendor/llvm/dist-release_80/lib/Target/Mips/MicroMips32r6InstrInfo.td =================================================================== --- vendor/llvm/dist-release_80/lib/Target/Mips/MicroMips32r6InstrInfo.td (revision 343793) +++ vendor/llvm/dist-release_80/lib/Target/Mips/MicroMips32r6InstrInfo.td (revision 343794) @@ -1,1817 +1,1818 @@ //=- MicroMips32r6InstrInfo.td - MicroMips r6 Instruction Information -*- tablegen -*-=// // // The LLVM Compiler Infrastructure // // This file is distributed under the University of Illinois Open Source // License. See LICENSE.TXT for details. // //===----------------------------------------------------------------------===// // // This file describes microMIPSr6 instructions. // //===----------------------------------------------------------------------===// def brtarget21_mm : Operand { let EncoderMethod = "getBranchTarget21OpValueMM"; let OperandType = "OPERAND_PCREL"; let DecoderMethod = "DecodeBranchTarget21MM"; let ParserMatchClass = MipsJumpTargetAsmOperand; } def brtarget26_mm : Operand { let EncoderMethod = "getBranchTarget26OpValueMM"; let OperandType = "OPERAND_PCREL"; let DecoderMethod = "DecodeBranchTarget26MM"; let ParserMatchClass = MipsJumpTargetAsmOperand; } def brtargetr6 : Operand { let EncoderMethod = "getBranchTargetOpValueMMR6"; let OperandType = "OPERAND_PCREL"; let DecoderMethod = "DecodeBranchTargetMM"; let ParserMatchClass = MipsJumpTargetAsmOperand; } def brtarget_lsl2_mm : Operand { let EncoderMethod = "getBranchTargetOpValueLsl2MMR6"; let OperandType = "OPERAND_PCREL"; // Instructions that use this operand have their decoder method // set with DecodeDisambiguates let DecoderMethod = ""; let ParserMatchClass = MipsJumpTargetAsmOperand; } //===----------------------------------------------------------------------===// // // Instruction Encodings // //===----------------------------------------------------------------------===// class ADD_MMR6_ENC : ARITH_FM_MMR6<"add", 0x110>; class ADDIU_MMR6_ENC : ADDI_FM_MMR6<"addiu", 0xc>; class ADDU_MMR6_ENC : ARITH_FM_MMR6<"addu", 0x150>; class ADDIUPC_MMR6_ENC : PCREL19_FM_MMR6<0b00>; class ALUIPC_MMR6_ENC : PCREL16_FM_MMR6<0b11111>; class AND_MMR6_ENC : ARITH_FM_MMR6<"and", 0x250>; class ANDI_MMR6_ENC : ADDI_FM_MMR6<"andi", 0x34>; class AUIPC_MMR6_ENC : PCREL16_FM_MMR6<0b11110>; class ALIGN_MMR6_ENC : POOL32A_ALIGN_FM_MMR6<0b011111>; class AUI_MMR6_ENC : AUI_FM_MMR6; class BALC_MMR6_ENC : BRANCH_OFF26_FM<0b101101>; class BC_MMR6_ENC : BRANCH_OFF26_FM<0b100101>; class BC16_MMR6_ENC : BC16_FM_MM16R6; class BEQZC16_MMR6_ENC : BEQZC_BNEZC_FM_MM16R6<0x23>; class BNEZC16_MMR6_ENC : BEQZC_BNEZC_FM_MM16R6<0x2b>; class BITSWAP_MMR6_ENC : POOL32A_BITSWAP_FM_MMR6<0b101100>; class BRK_MMR6_ENC : BREAK_MMR6_ENC<"break">; class BEQZC_MMR6_ENC : CMP_BRANCH_OFF21_FM_MMR6<"beqzc", 0b100000>; class BNEZC_MMR6_ENC : CMP_BRANCH_OFF21_FM_MMR6<"bnezc", 0b101000>; class BGEC_MMR6_ENC : CMP_BRANCH_2R_OFF16_FM_MMR6<"bgec", 0b111101>, DecodeDisambiguates<"POP75GroupBranchMMR6">; class BGEUC_MMR6_ENC : CMP_BRANCH_2R_OFF16_FM_MMR6<"bgeuc", 0b110000>, DecodeDisambiguates<"BlezGroupBranchMMR6">; class BLTC_MMR6_ENC : CMP_BRANCH_2R_OFF16_FM_MMR6<"bltc", 0b110101>, DecodeDisambiguates<"POP65GroupBranchMMR6">; class BLTUC_MMR6_ENC : CMP_BRANCH_2R_OFF16_FM_MMR6<"bltuc", 0b111000>, DecodeDisambiguates<"BgtzGroupBranchMMR6">; class BEQC_MMR6_ENC : CMP_BRANCH_2R_OFF16_FM_MMR6<"beqc", 0b011101>; class BNEC_MMR6_ENC : CMP_BRANCH_2R_OFF16_FM_MMR6<"bnec", 0b011111>; class BLTZC_MMR6_ENC : CMP_BRANCH_1R_BOTH_OFF16_FM_MMR6<"bltzc", 0b110101>, DecodeDisambiguates<"POP65GroupBranchMMR6">; class BLEZC_MMR6_ENC : CMP_BRANCH_1R_RT_OFF16_FM_MMR6<"blezc", 0b111101>, DecodeDisambiguates<"POP75GroupBranchMMR6">; class BGEZC_MMR6_ENC : CMP_BRANCH_1R_BOTH_OFF16_FM_MMR6<"bgezc", 0b111101>, DecodeDisambiguates<"POP75GroupBranchMMR6">; class BGTZC_MMR6_ENC : CMP_BRANCH_1R_RT_OFF16_FM_MMR6<"bgtzc", 0b110101>, DecodeDisambiguates<"POP65GroupBranchMMR6">; class BEQZALC_MMR6_ENC : CMP_BRANCH_1R_RT_OFF16_FM_MMR6<"beqzalc", 0b011101>, DecodeDisambiguates<"POP35GroupBranchMMR6">; class BNEZALC_MMR6_ENC : CMP_BRANCH_1R_RT_OFF16_FM_MMR6<"bnezalc", 0b011111>, DecodeDisambiguates<"POP37GroupBranchMMR6">; class BGTZALC_MMR6_ENC : CMP_BRANCH_1R_RT_OFF16_FM_MMR6<"bgtzalc", 0b111000>, MMDecodeDisambiguatedBy<"BgtzGroupBranchMMR6">; class BLTZALC_MMR6_ENC : CMP_BRANCH_1R_BOTH_OFF16_FM_MMR6<"bltzalc", 0b111000>, MMDecodeDisambiguatedBy<"BgtzGroupBranchMMR6">; class BGEZALC_MMR6_ENC : CMP_BRANCH_1R_BOTH_OFF16_FM_MMR6<"bgezalc", 0b110000>, MMDecodeDisambiguatedBy<"BlezGroupBranchMMR6">; class BLEZALC_MMR6_ENC : CMP_BRANCH_1R_RT_OFF16_FM_MMR6<"blezalc", 0b110000>, MMDecodeDisambiguatedBy<"BlezGroupBranchMMR6">; class CACHE_MMR6_ENC : CACHE_PREF_FM_MMR6<0b001000, 0b0110>; class CLO_MMR6_ENC : POOL32A_2R_FM_MMR6<0b0100101100>; class CLZ_MMR6_ENC : SPECIAL_2R_FM_MMR6<0b010000>; class DIV_MMR6_ENC : ARITH_FM_MMR6<"div", 0x118>; class DIVU_MMR6_ENC : ARITH_FM_MMR6<"divu", 0x198>; class EHB_MMR6_ENC : BARRIER_MMR6_ENC<"ehb", 0x3>; class EI_MMR6_ENC : POOL32A_EIDI_MMR6_ENC<"ei", 0x15d>; class DI_MMR6_ENC : POOL32A_EIDI_MMR6_ENC<"di", 0b0100011101>; class ERET_MMR6_ENC : POOL32A_ERET_FM_MMR6<"eret", 0x3cd>; class DERET_MMR6_ENC : POOL32A_ERET_FM_MMR6<"eret", 0b1110001101>; class ERETNC_MMR6_ENC : ERETNC_FM_MMR6<"eretnc">; class GINVI_MMR6_ENC : POOL32A_GINV_FM_MMR6<"ginvi", 0b00>; class GINVT_MMR6_ENC : POOL32A_GINV_FM_MMR6<"ginvt", 0b10>; class JALRC16_MMR6_ENC : POOL16C_JALRC_FM_MM16R6<0xb>; class JIALC_MMR6_ENC : JMP_IDX_COMPACT_FM<0b100000>; class JIC_MMR6_ENC : JMP_IDX_COMPACT_FM<0b101000>; class JRC16_MMR6_ENC: POOL16C_JALRC_FM_MM16R6<0x3>; class JRCADDIUSP_MMR6_ENC : POOL16C_JRCADDIUSP_FM_MM16R6<0x13>; class LSA_MMR6_ENC : POOL32A_LSA_FM<0b001111>; class LWPC_MMR6_ENC : PCREL19_FM_MMR6<0b01>; class LWM16_MMR6_ENC : POOL16C_LWM_SWM_FM_MM16R6<0x2>; class MFC0_MMR6_ENC : POOL32A_MFTC0_FM_MMR6<"mfc0", 0b00011, 0b111100>; class MFC1_MMR6_ENC : POOL32F_MFTC1_FM_MMR6<"mfc1", 0b10000000>; class MFC2_MMR6_ENC : POOL32A_MFTC2_FM_MMR6<"mfc2", 0b0100110100>; class MFHC0_MMR6_ENC : POOL32A_MFTC0_FM_MMR6<"mfhc0", 0b00011, 0b110100>; class MFHC2_MMR6_ENC : POOL32A_MFTC2_FM_MMR6<"mfhc2", 0b1000110100>; class MOD_MMR6_ENC : ARITH_FM_MMR6<"mod", 0x158>; class MODU_MMR6_ENC : ARITH_FM_MMR6<"modu", 0x1d8>; class MUL_MMR6_ENC : ARITH_FM_MMR6<"mul", 0x18>; class MUH_MMR6_ENC : ARITH_FM_MMR6<"muh", 0x58>; class MULU_MMR6_ENC : ARITH_FM_MMR6<"mulu", 0x98>; class MUHU_MMR6_ENC : ARITH_FM_MMR6<"muhu", 0xd8>; class MTC0_MMR6_ENC : POOL32A_MFTC0_FM_MMR6<"mtc0", 0b01011, 0b111100>; class MTC1_MMR6_ENC : POOL32F_MFTC1_FM_MMR6<"mtc1", 0b10100000>; class MTC2_MMR6_ENC : POOL32A_MFTC2_FM_MMR6<"mtc2", 0b0101110100>; class MTHC0_MMR6_ENC : POOL32A_MFTC0_FM_MMR6<"mthc0", 0b01011, 0b110100>; class MTHC2_MMR6_ENC : POOL32A_MFTC2_FM_MMR6<"mthc2", 0b1001110100>; class NOR_MMR6_ENC : ARITH_FM_MMR6<"nor", 0x2d0>; class OR_MMR6_ENC : ARITH_FM_MMR6<"or", 0x290>; class ORI_MMR6_ENC : ADDI_FM_MMR6<"ori", 0x14>; class PREF_MMR6_ENC : CACHE_PREF_FM_MMR6<0b011000, 0b0010>; class SB16_MMR6_ENC : LOAD_STORE_FM_MM16<0x22>; class SELEQZ_MMR6_ENC : POOL32A_FM_MMR6<0b0101000000>; class SELNEZ_MMR6_ENC : POOL32A_FM_MMR6<0b0110000000>; class SH16_MMR6_ENC : LOAD_STORE_FM_MM16<0x2a>; class SLL_MMR6_ENC : SHIFT_MMR6_ENC<"sll", 0x00, 0b0>; class SUB_MMR6_ENC : ARITH_FM_MMR6<"sub", 0x190>; class SUBU_MMR6_ENC : ARITH_FM_MMR6<"subu", 0x1d0>; class SW_MMR6_ENC : SW32_FM_MMR6<"sw", 0x3e>; class SW16_MMR6_ENC : LOAD_STORE_FM_MM16<0x3a>; class SWM16_MMR6_ENC : POOL16C_LWM_SWM_FM_MM16R6<0xa>; class SWSP_MMR6_ENC : LOAD_STORE_SP_FM_MM16<0x32>; class WRPGPR_MMR6_ENC : POOL32A_WRPGPR_WSBH_FM_MMR6<"wrpgpr", 0x3c5>; class WSBH_MMR6_ENC : POOL32A_WRPGPR_WSBH_FM_MMR6<"wsbh", 0x1ec>; class LB_MMR6_ENC : LB32_FM_MMR6; class LBU_MMR6_ENC : LBU32_FM_MMR6; class PAUSE_MMR6_ENC : POOL32A_PAUSE_FM_MMR6<"pause", 0b00101>; class RDHWR_MMR6_ENC : POOL32A_RDHWR_FM_MMR6; class WAIT_MMR6_ENC : WAIT_FM_MM, MMR6Arch<"wait">; class SSNOP_MMR6_ENC : BARRIER_FM_MM<0x1>, MMR6Arch<"ssnop">; class SYNC_MMR6_ENC : POOL32A_SYNC_FM_MMR6; class SYNCI_MMR6_ENC : POOL32I_SYNCI_FM_MMR6, MMR6Arch<"synci">; class RDPGPR_MMR6_ENC : POOL32A_RDPGPR_FM_MMR6<0b1110000101>; class SDBBP_MMR6_ENC : SDBBP_FM_MM, MMR6Arch<"sdbbp">; class SIGRIE_MMR6_ENC : SIGRIE_FM_MM, MMR6Arch<"sigrie">; class XOR_MMR6_ENC : ARITH_FM_MMR6<"xor", 0x310>; class XORI_MMR6_ENC : ADDI_FM_MMR6<"xori", 0x1c>; class ABS_S_MMR6_ENC : POOL32F_ABS_FM_MMR6<"abs.s", 0, 0b0001101>; class ABS_D_MMR6_ENC : POOL32F_ABS_FM_MMR6<"abs.d", 1, 0b0001101>; class FLOOR_L_S_MMR6_ENC : POOL32F_MATH_FM_MMR6<"floor.l.s", 0, 0b00001100>; class FLOOR_L_D_MMR6_ENC : POOL32F_MATH_FM_MMR6<"floor.l.d", 1, 0b00001100>; class FLOOR_W_S_MMR6_ENC : POOL32F_MATH_FM_MMR6<"floor.w.s", 0, 0b00101100>; class FLOOR_W_D_MMR6_ENC : POOL32F_MATH_FM_MMR6<"floor.w.d", 1, 0b00101100>; class CEIL_L_S_MMR6_ENC : POOL32F_MATH_FM_MMR6<"ceil.l.s", 0, 0b01001100>; class CEIL_L_D_MMR6_ENC : POOL32F_MATH_FM_MMR6<"ceil.l.d", 1, 0b01001100>; class CEIL_W_S_MMR6_ENC : POOL32F_MATH_FM_MMR6<"ceil.w.s", 0, 0b01101100>; class CEIL_W_D_MMR6_ENC : POOL32F_MATH_FM_MMR6<"ceil.w.d", 1, 0b01101100>; class TRUNC_L_S_MMR6_ENC : POOL32F_MATH_FM_MMR6<"trunc.l.s", 0, 0b10001100>; class TRUNC_L_D_MMR6_ENC : POOL32F_MATH_FM_MMR6<"trunc.l.d", 1, 0b10001100>; class TRUNC_W_S_MMR6_ENC : POOL32F_MATH_FM_MMR6<"trunc.w.s", 0, 0b10101100>; class TRUNC_W_D_MMR6_ENC : POOL32F_MATH_FM_MMR6<"trunc.w.d", 1, 0b10101100>; class SB_MMR6_ENC : SB32_SH32_STORE_FM_MMR6<0b000110>; class SH_MMR6_ENC : SB32_SH32_STORE_FM_MMR6<0b001110>; class LW_MMR6_ENC : LOAD_WORD_FM_MMR6; class LUI_MMR6_ENC : LOAD_UPPER_IMM_FM_MMR6; class JALRC_HB_MMR6_ENC : POOL32A_JALRC_FM_MMR6<"jalrc.hb", 0b0001111100>; class RINT_S_MMR6_ENC : POOL32F_RINT_FM_MMR6<"rint.s", 0>; class RINT_D_MMR6_ENC : POOL32F_RINT_FM_MMR6<"rint.d", 1>; class ROUND_L_S_MMR6_ENC : POOL32F_RECIP_ROUND_FM_MMR6<"round.l.s", 0, 0b11001100>; class ROUND_L_D_MMR6_ENC : POOL32F_RECIP_ROUND_FM_MMR6<"round.l.d", 1, 0b11001100>; class ROUND_W_S_MMR6_ENC : POOL32F_RECIP_ROUND_FM_MMR6<"round.w.s", 0, 0b11101100>; class ROUND_W_D_MMR6_ENC : POOL32F_RECIP_ROUND_FM_MMR6<"round.w.d", 1, 0b11101100>; class SEL_S_MMR6_ENC : POOL32F_SEL_FM_MMR6<"sel.s", 0, 0b010111000>; class SEL_D_MMR6_ENC : POOL32F_SEL_FM_MMR6<"sel.d", 1, 0b010111000>; class SELEQZ_S_MMR6_ENC : POOL32F_SEL_FM_MMR6<"seleqz.s", 0, 0b000111000>; class SELEQZ_D_MMR6_ENC : POOL32F_SEL_FM_MMR6<"seleqz.d", 1, 0b000111000>; class SELNEZ_S_MMR6_ENC : POOL32F_SEL_FM_MMR6<"selnez.s", 0, 0b001111000>; class SELNEZ_D_MMR6_ENC : POOL32F_SEL_FM_MMR6<"selnez.d", 1, 0b001111000>; class CLASS_S_MMR6_ENC : POOL32F_CLASS_FM_MMR6<"class.s", 0, 0b001100000>; class CLASS_D_MMR6_ENC : POOL32F_CLASS_FM_MMR6<"class.d", 1, 0b001100000>; class EXT_MMR6_ENC : POOL32A_EXT_INS_FM_MMR6<"ext", 0b101100>; class INS_MMR6_ENC : POOL32A_EXT_INS_FM_MMR6<"ins", 0b001100>; class JALRC_MMR6_ENC : POOL32A_JALRC_FM_MMR6<"jalrc", 0b0000111100>; class BOVC_MMR6_ENC : POP35_BOVC_FM_MMR6<"bovc">; class BNVC_MMR6_ENC : POP37_BNVC_FM_MMR6<"bnvc">; class ADDU16_MMR6_ENC : POOL16A_ADDU16_FM_MMR6; class AND16_MMR6_ENC : POOL16C_AND16_FM_MMR6; class ANDI16_MMR6_ENC : ANDI_FM_MM16<0b001011>; class NOT16_MMR6_ENC : POOL16C_NOT16_FM_MMR6; class OR16_MMR6_ENC : POOL16C_OR16_XOR16_FM_MMR6<0b1001>; class SLL16_MMR6_ENC : SHIFT_FM_MM16<0>; class SRL16_MMR6_ENC : SHIFT_FM_MM16<1>; class BREAK16_MMR6_ENC : POOL16C_BREAKPOINT_FM_MMR6<0b011011>; class LI16_MMR6_ENC : LI_FM_MM16; class MOVE16_MMR6_ENC : MOVE_FM_MM16<0b000011>; class MOVEP_MMR6_ENC : POOL16C_MOVEP16_FM_MMR6; class SDBBP16_MMR6_ENC : POOL16C_BREAKPOINT_FM_MMR6<0b111011>; class SUBU16_MMR6_ENC : POOL16A_SUBU16_FM_MMR6; class XOR16_MMR6_ENC : POOL16C_OR16_XOR16_FM_MMR6<0b1000>; class TLBINV_MMR6_ENC : POOL32A_TLBINV_FM_MMR6<"tlbinv", 0x10d>; class TLBINVF_MMR6_ENC : POOL32A_TLBINV_FM_MMR6<"tlbinvf", 0x14d>; class DVP_MMR6_ENC : POOL32A_DVPEVP_FM_MMR6<"dvp", 0b0001100101>; class EVP_MMR6_ENC : POOL32A_DVPEVP_FM_MMR6<"evp", 0b0011100101>; class BC1EQZC_MMR6_ENC : POOL32I_BRANCH_COP_1_2_FM_MMR6<"bc1eqzc", 0b01000>; class BC1NEZC_MMR6_ENC : POOL32I_BRANCH_COP_1_2_FM_MMR6<"bc1nezc", 0b01001>; class BC2EQZC_MMR6_ENC : POOL32I_BRANCH_COP_1_2_FM_MMR6<"bc2eqzc", 0b01010>; class BC2NEZC_MMR6_ENC : POOL32I_BRANCH_COP_1_2_FM_MMR6<"bc2nezc", 0b01011>; class LDC1_MMR6_ENC : LDWC1_SDWC1_FM_MMR6<"ldc1", 0b101111>; class SDC1_MMR6_ENC : LDWC1_SDWC1_FM_MMR6<"sdc1", 0b101110>; class LDC2_MMR6_ENC : POOL32B_LDWC2_SDWC2_FM_MMR6<"ldc2", 0b0010>; class SDC2_MMR6_ENC : POOL32B_LDWC2_SDWC2_FM_MMR6<"sdc2", 0b1010>; class LWC2_MMR6_ENC : POOL32B_LDWC2_SDWC2_FM_MMR6<"lwc2", 0b0000>; class SWC2_MMR6_ENC : POOL32B_LDWC2_SDWC2_FM_MMR6<"swc2", 0b1000>; class LL_MMR6_ENC : POOL32C_LL_E_SC_E_FM_MMR6<"ll", 0b0011, 0b000>; class SC_MMR6_ENC : POOL32C_LL_E_SC_E_FM_MMR6<"sc", 0b1011, 0b000>; /// Floating Point Instructions class FADD_S_MMR6_ENC : POOL32F_ARITH_FM_MMR6<"add.s", 0, 0b00110000>; class FSUB_S_MMR6_ENC : POOL32F_ARITH_FM_MMR6<"sub.s", 0, 0b01110000>; class FMUL_S_MMR6_ENC : POOL32F_ARITH_FM_MMR6<"mul.s", 0, 0b10110000>; class FDIV_S_MMR6_ENC : POOL32F_ARITH_FM_MMR6<"div.s", 0, 0b11110000>; class MADDF_S_MMR6_ENC : POOL32F_ARITHF_FM_MMR6<"maddf.s", 0, 0b110111000>; class MADDF_D_MMR6_ENC : POOL32F_ARITHF_FM_MMR6<"maddf.d", 1, 0b110111000>; class MSUBF_S_MMR6_ENC : POOL32F_ARITHF_FM_MMR6<"msubf.s", 0, 0b111111000>; class MSUBF_D_MMR6_ENC : POOL32F_ARITHF_FM_MMR6<"msubf.d", 1, 0b111111000>; class FMOV_S_MMR6_ENC : POOL32F_MOV_NEG_FM_MMR6<"mov.s", 0, 0b0000001>; class FNEG_S_MMR6_ENC : POOL32F_MOV_NEG_FM_MMR6<"neg.s", 0, 0b0101101>; class MAX_S_MMR6_ENC : POOL32F_MINMAX_FM<"max.s", 0, 0b000001011>; class MAX_D_MMR6_ENC : POOL32F_MINMAX_FM<"max.d", 1, 0b000001011>; class MAXA_S_MMR6_ENC : POOL32F_MINMAX_FM<"maxa.s", 0, 0b000101011>; class MAXA_D_MMR6_ENC : POOL32F_MINMAX_FM<"maxa.d", 1, 0b000101011>; class MIN_S_MMR6_ENC : POOL32F_MINMAX_FM<"min.s", 0, 0b000000011>; class MIN_D_MMR6_ENC : POOL32F_MINMAX_FM<"min.d", 1, 0b000000011>; class MINA_S_MMR6_ENC : POOL32F_MINMAX_FM<"mina.s", 0, 0b000100011>; class MINA_D_MMR6_ENC : POOL32F_MINMAX_FM<"mina.d", 1, 0b000100011>; class CVT_L_S_MMR6_ENC : POOL32F_CVT_LW_FM<"cvt.l.s", 0, 0b00000100>; class CVT_L_D_MMR6_ENC : POOL32F_CVT_LW_FM<"cvt.l.d", 1, 0b00000100>; class CVT_W_S_MMR6_ENC : POOL32F_CVT_LW_FM<"cvt.w.s", 0, 0b00100100>; class CVT_D_L_MMR6_ENC : POOL32F_CVT_DS_FM<"cvt.d.l", 2, 0b1001101>; class CVT_S_W_MMR6_ENC : POOL32F_CVT_DS_FM<"cvt.s.w", 1, 0b1101101>; class CVT_S_L_MMR6_ENC : POOL32F_CVT_DS_FM<"cvt.s.l", 2, 0b1101101>; //===----------------------------------------------------------------------===// // // Instruction Descriptions // //===----------------------------------------------------------------------===// class CMP_CBR_RT_Z_MMR6_DESC_BASE : BRANCH_DESC_BASE { dag InOperandList = (ins GPROpnd:$rt, opnd:$offset); dag OutOperandList = (outs); string AsmString = !strconcat(instr_asm, "\t$rt, $offset"); list Defs = [AT]; InstrItinClass Itinerary = II_BCCZC; } class BEQZALC_MMR6_DESC : CMP_CBR_RT_Z_MMR6_DESC_BASE<"beqzalc", brtarget_mm, GPR32Opnd> { list Defs = [RA]; } class BGEZALC_MMR6_DESC : CMP_CBR_RT_Z_MMR6_DESC_BASE<"bgezalc", brtarget_mm, GPR32Opnd> { list Defs = [RA]; } class BGTZALC_MMR6_DESC : CMP_CBR_RT_Z_MMR6_DESC_BASE<"bgtzalc", brtarget_mm, GPR32Opnd> { list Defs = [RA]; } class BLEZALC_MMR6_DESC : CMP_CBR_RT_Z_MMR6_DESC_BASE<"blezalc", brtarget_mm, GPR32Opnd> { list Defs = [RA]; } class BLTZALC_MMR6_DESC : CMP_CBR_RT_Z_MMR6_DESC_BASE<"bltzalc", brtarget_mm, GPR32Opnd> { list Defs = [RA]; } class BNEZALC_MMR6_DESC : CMP_CBR_RT_Z_MMR6_DESC_BASE<"bnezalc", brtarget_mm, GPR32Opnd> { list Defs = [RA]; } class BLTZC_MMR6_DESC : CMP_CBR_RT_Z_MMR6_DESC_BASE<"bltzc", brtarget_lsl2_mm, GPR32Opnd>; class BLEZC_MMR6_DESC : CMP_CBR_RT_Z_MMR6_DESC_BASE<"blezc", brtarget_lsl2_mm, GPR32Opnd>; class BGEZC_MMR6_DESC : CMP_CBR_RT_Z_MMR6_DESC_BASE<"bgezc", brtarget_lsl2_mm, GPR32Opnd>; class BGTZC_MMR6_DESC : CMP_CBR_RT_Z_MMR6_DESC_BASE<"bgtzc", brtarget_lsl2_mm, GPR32Opnd>; class CMP_CBR_2R_MMR6_DESC_BASE : BRANCH_DESC_BASE { dag InOperandList = (ins GPROpnd:$rs, GPROpnd:$rt, opnd:$offset); dag OutOperandList = (outs); string AsmString = !strconcat(instr_asm, "\t$rs, $rt, $offset"); list Defs = [AT]; InstrItinClass Itinerary = II_BCCC; } class BGEC_MMR6_DESC : CMP_CBR_2R_MMR6_DESC_BASE<"bgec", brtarget_lsl2_mm, GPR32Opnd>; class BGEUC_MMR6_DESC : CMP_CBR_2R_MMR6_DESC_BASE<"bgeuc", brtarget_lsl2_mm, GPR32Opnd>; class BLTC_MMR6_DESC : CMP_CBR_2R_MMR6_DESC_BASE<"bltc", brtarget_lsl2_mm, GPR32Opnd>; class BLTUC_MMR6_DESC : CMP_CBR_2R_MMR6_DESC_BASE<"bltuc", brtarget_lsl2_mm, GPR32Opnd>; class BEQC_MMR6_DESC : CMP_CBR_2R_MMR6_DESC_BASE<"beqc", brtarget_lsl2_mm, GPR32Opnd>; class BNEC_MMR6_DESC : CMP_CBR_2R_MMR6_DESC_BASE<"bnec", brtarget_lsl2_mm, GPR32Opnd>; class ADD_MMR6_DESC : ArithLogicR<"add", GPR32Opnd, 1, II_ADD>; class ADDIU_MMR6_DESC : ArithLogicI<"addiu", simm16, GPR32Opnd, II_ADDIU, immSExt16, add>; class ADDU_MMR6_DESC : ArithLogicR<"addu", GPR32Opnd, 1, II_ADDU>; class MUL_MMR6_DESC : ArithLogicR<"mul", GPR32Opnd, 1, II_MUL, mul>; class MUH_MMR6_DESC : ArithLogicR<"muh", GPR32Opnd, 1, II_MUH, mulhs>; class MULU_MMR6_DESC : ArithLogicR<"mulu", GPR32Opnd, 1, II_MULU>; class MUHU_MMR6_DESC : ArithLogicR<"muhu", GPR32Opnd, 1, II_MUHU, mulhu>; class BC_MMR6_DESC_BASE : BRANCH_DESC_BASE, MMR6Arch { dag InOperandList = (ins opnd:$offset); dag OutOperandList = (outs); string AsmString = !strconcat(instr_asm, "\t$offset"); bit isBarrier = 1; InstrItinClass Itinerary = Itin; } class BALC_MMR6_DESC : BC_MMR6_DESC_BASE<"balc", brtarget26_mm, II_BALC> { bit isCall = 1; list Defs = [RA]; } class BC_MMR6_DESC : BC_MMR6_DESC_BASE<"bc", brtarget26_mm, II_BC> { list Pattern = [(br bb:$offset)]; } class BC16_MMR6_DESC : MicroMipsInst16<(outs), (ins brtarget10_mm:$offset), !strconcat("bc16", "\t$offset"), [], II_BC, FrmI>, MMR6Arch<"bc16"> { let isBranch = 1; let isTerminator = 1; let isBarrier = 1; let hasDelaySlot = 0; let AdditionalPredicates = [RelocPIC]; let Defs = [AT]; } class BEQZC_BNEZC_MM16R6_DESC_BASE : CBranchZeroMM, MMR6Arch { let isBranch = 1; let isTerminator = 1; let hasDelaySlot = 0; let Defs = [AT]; } class BEQZC16_MMR6_DESC : BEQZC_BNEZC_MM16R6_DESC_BASE<"beqzc16">; class BNEZC16_MMR6_DESC : BEQZC_BNEZC_MM16R6_DESC_BASE<"bnezc16">; class SUB_MMR6_DESC : ArithLogicR<"sub", GPR32Opnd, 0, II_SUB>; class SUBU_MMR6_DESC : ArithLogicR<"subu", GPR32Opnd, 0,II_SUBU>; class BITSWAP_MMR6_DESC_BASE : MMR6Arch { dag OutOperandList = (outs GPROpnd:$rd); dag InOperandList = (ins GPROpnd:$rt); string AsmString = !strconcat(instr_asm, "\t$rd, $rt"); list Pattern = []; InstrItinClass Itinerary = II_BITSWAP; } class BITSWAP_MMR6_DESC : BITSWAP_MMR6_DESC_BASE<"bitswap", GPR32Opnd>; class BRK_MMR6_DESC : BRK_FT<"break">; class CACHE_HINT_MMR6_DESC : MMR6Arch { dag OutOperandList = (outs); dag InOperandList = (ins MemOpnd:$addr, uimm5:$hint); string AsmString = !strconcat(instr_asm, "\t$hint, $addr"); list Pattern = []; string DecoderMethod = "DecodeCacheOpMM"; InstrItinClass Itinerary = Itin; } class CACHE_MMR6_DESC : CACHE_HINT_MMR6_DESC<"cache", mem_mm_12, GPR32Opnd, II_CACHE>; class PREF_MMR6_DESC : CACHE_HINT_MMR6_DESC<"pref", mem_mm_12, GPR32Opnd, II_PREF>; class LB_LBU_MMR6_DESC_BASE : MMR6Arch { dag OutOperandList = (outs GPROpnd:$rt); dag InOperandList = (ins MemOpnd:$addr); string AsmString = !strconcat(instr_asm, "\t$rt, $addr"); string DecoderMethod = "DecodeLoadByte15"; bit mayLoad = 1; InstrItinClass Itinerary = Itin; } class LB_MMR6_DESC : LB_LBU_MMR6_DESC_BASE<"lb", mem_mm_16, GPR32Opnd, II_LB>; class LBU_MMR6_DESC : LB_LBU_MMR6_DESC_BASE<"lbu", mem_mm_16, GPR32Opnd, II_LBU>; class CLO_CLZ_MMR6_DESC_BASE : MMR6Arch { dag OutOperandList = (outs GPROpnd:$rt); dag InOperandList = (ins GPROpnd:$rs); string AsmString = !strconcat(instr_asm, "\t$rt, $rs"); InstrItinClass Itinerary = Itin; } class CLO_MMR6_DESC : CLO_CLZ_MMR6_DESC_BASE<"clo", GPR32Opnd, II_CLO>; class CLZ_MMR6_DESC : CLO_CLZ_MMR6_DESC_BASE<"clz", GPR32Opnd, II_CLZ>; class EHB_MMR6_DESC : Barrier<"ehb", II_EHB>; class EI_MMR6_DESC : DEI_FT<"ei", GPR32Opnd, II_EI>; class DI_MMR6_DESC : DEI_FT<"di", GPR32Opnd, II_DI>; class ERET_MMR6_DESC : ER_FT<"eret", II_ERET>; class DERET_MMR6_DESC : ER_FT<"deret", II_DERET>; class ERETNC_MMR6_DESC : ER_FT<"eretnc", II_ERETNC>; class JALRC16_MMR6_DESC_BASE : MicroMipsInst16<(outs), (ins RO:$rs), !strconcat(opstr, "\t$rs"), [(MipsJmpLink RO:$rs)], II_JALR, FrmR>, MMR6Arch { let isCall = 1; let hasDelaySlot = 0; let Defs = [RA]; + let hasPostISelHook = 1; } class JALRC16_MMR6_DESC : JALRC16_MMR6_DESC_BASE<"jalr", GPR32Opnd>; class JMP_MMR6_IDX_COMPACT_DESC_BASE : MMR6Arch { dag InOperandList = (ins GPROpnd:$rt, opnd:$offset); string AsmString = !strconcat(opstr, "\t$rt, $offset"); list Pattern = []; bit isTerminator = 1; bit hasDelaySlot = 0; InstrItinClass Itinerary = Itin; } class JIALC_MMR6_DESC : JMP_MMR6_IDX_COMPACT_DESC_BASE<"jialc", calloffset16, GPR32Opnd, II_JIALC> { bit isCall = 1; list Defs = [RA]; } class JIC_MMR6_DESC : JMP_MMR6_IDX_COMPACT_DESC_BASE<"jic", jmpoffset16, GPR32Opnd, II_JIC> { bit isBarrier = 1; list Defs = [AT]; } class JRC16_MMR6_DESC_BASE : MicroMipsInst16<(outs), (ins RO:$rs), !strconcat(opstr, "\t$rs"), [], II_JR, FrmR>, MMR6Arch { let hasDelaySlot = 0; let isBranch = 1; let isIndirectBranch = 1; } class JRC16_MMR6_DESC : JRC16_MMR6_DESC_BASE<"jrc16", GPR32Opnd>; class JRCADDIUSP_MMR6_DESC : MicroMipsInst16<(outs), (ins uimm5_lsl2:$imm), "jrcaddiusp\t$imm", [], II_JRADDIUSP, FrmR>, MMR6Arch<"jrcaddiusp"> { let hasDelaySlot = 0; let isTerminator = 1; let isBarrier = 1; let isBranch = 1; let isIndirectBranch = 1; } class ALIGN_MMR6_DESC_BASE : MMR6Arch { dag OutOperandList = (outs GPROpnd:$rd); dag InOperandList = (ins GPROpnd:$rs, GPROpnd:$rt, ImmOpnd:$bp); string AsmString = !strconcat(instr_asm, "\t$rd, $rs, $rt, $bp"); list Pattern = []; InstrItinClass Itinerary = Itin; } class ALIGN_MMR6_DESC : ALIGN_MMR6_DESC_BASE<"align", GPR32Opnd, uimm2, II_ALIGN>; class AUI_MMR6_DESC_BASE : MMR6Arch { dag OutOperandList = (outs GPROpnd:$rt); dag InOperandList = (ins GPROpnd:$rs, uimm16:$imm); string AsmString = !strconcat(instr_asm, "\t$rt, $rs, $imm"); list Pattern = []; InstrItinClass Itinerary = Itin; } class AUI_MMR6_DESC : AUI_MMR6_DESC_BASE<"aui", GPR32Opnd, II_AUI>; class ALUIPC_MMR6_DESC_BASE : MMR6Arch { dag OutOperandList = (outs GPROpnd:$rt); dag InOperandList = (ins simm16:$imm); string AsmString = !strconcat(instr_asm, "\t$rt, $imm"); list Pattern = []; InstrItinClass Itinerary = Itin; } class ALUIPC_MMR6_DESC : ALUIPC_MMR6_DESC_BASE<"aluipc", GPR32Opnd, II_ALUIPC>; class AUIPC_MMR6_DESC : ALUIPC_MMR6_DESC_BASE<"auipc", GPR32Opnd, II_AUIPC>; class LSA_MMR6_DESC_BASE : MMR6Arch { dag OutOperandList = (outs GPROpnd:$rd); dag InOperandList = (ins GPROpnd:$rs, GPROpnd:$rt, ImmOpnd:$imm2); string AsmString = !strconcat(instr_asm, "\t$rt, $rs, $rd, $imm2"); list Pattern = []; InstrItinClass Itinerary = Itin; } class LSA_MMR6_DESC : LSA_MMR6_DESC_BASE<"lsa", GPR32Opnd, uimm2_plus1, II_LSA>; class PCREL_MMR6_DESC_BASE : MMR6Arch { dag OutOperandList = (outs GPROpnd:$rt); dag InOperandList = (ins ImmOpnd:$imm); string AsmString = !strconcat(instr_asm, "\t$rt, $imm"); list Pattern = []; InstrItinClass Itinerary = Itin; } class ADDIUPC_MMR6_DESC : PCREL_MMR6_DESC_BASE<"addiupc", GPR32Opnd, simm19_lsl2, II_ADDIUPC>; class LWPC_MMR6_DESC: PCREL_MMR6_DESC_BASE<"lwpc", GPR32Opnd, simm19_lsl2, II_LWPC>; class SELEQNE_Z_MMR6_DESC_BASE : MMR6Arch { dag OutOperandList = (outs GPROpnd:$rd); dag InOperandList = (ins GPROpnd:$rs, GPROpnd:$rt); string AsmString = !strconcat(instr_asm, "\t$rd, $rs, $rt"); list Pattern = []; InstrItinClass Itinerary = Itin; } class SELEQZ_MMR6_DESC : SELEQNE_Z_MMR6_DESC_BASE<"seleqz", GPR32Opnd, II_SELCCZ>; class SELNEZ_MMR6_DESC : SELEQNE_Z_MMR6_DESC_BASE<"selnez", GPR32Opnd, II_SELCCZ>; class PAUSE_MMR6_DESC : Barrier<"pause", II_PAUSE>; class RDHWR_MMR6_DESC : MMR6Arch<"rdhwr">, MipsR6Inst { dag OutOperandList = (outs GPR32Opnd:$rt); dag InOperandList = (ins HWRegsOpnd:$rs, uimm3:$sel); string AsmString = !strconcat("rdhwr", "\t$rt, $rs, $sel"); list Pattern = []; InstrItinClass Itinerary = II_RDHWR; Format Form = FrmR; } class WAIT_MMR6_DESC : WaitMM<"wait">; // FIXME: ssnop should not be defined for R6. Per MD000582 microMIPS32 6.03: // Assemblers targeting specifically Release 6 should reject the SSNOP // instruction with an error. class SSNOP_MMR6_DESC : Barrier<"ssnop", II_SSNOP>; class SLL_MMR6_DESC : shift_rotate_imm<"sll", uimm5, GPR32Opnd, II_SLL>; class DIVMOD_MMR6_DESC_BASE : MipsR6Inst { dag OutOperandList = (outs GPROpnd:$rd); dag InOperandList = (ins GPROpnd:$rs, GPROpnd:$rt); string AsmString = !strconcat(opstr, "\t$rd, $rs, $rt"); list Pattern = [(set GPROpnd:$rd, (OpNode GPROpnd:$rs, GPROpnd:$rt))]; string BaseOpcode = opstr; Format f = FrmR; let isCommutable = 0; let isReMaterializable = 1; InstrItinClass Itinerary = Itin; // This instruction doesn't trap division by zero itself. We must insert // teq instructions as well. bit usesCustomInserter = 1; } class DIV_MMR6_DESC : DIVMOD_MMR6_DESC_BASE<"div", GPR32Opnd, II_DIV, sdiv>; class DIVU_MMR6_DESC : DIVMOD_MMR6_DESC_BASE<"divu", GPR32Opnd, II_DIVU, udiv>; class MOD_MMR6_DESC : DIVMOD_MMR6_DESC_BASE<"mod", GPR32Opnd, II_MOD, srem>; class MODU_MMR6_DESC : DIVMOD_MMR6_DESC_BASE<"modu", GPR32Opnd, II_MODU, urem>; class AND_MMR6_DESC : ArithLogicR<"and", GPR32Opnd, 1, II_AND, and>; class ANDI_MMR6_DESC : ArithLogicI<"andi", uimm16, GPR32Opnd, II_ANDI>; class NOR_MMR6_DESC : LogicNOR<"nor", GPR32Opnd>; class OR_MMR6_DESC : ArithLogicR<"or", GPR32Opnd, 1, II_OR, or>; class ORI_MMR6_DESC : ArithLogicI<"ori", uimm16, GPR32Opnd, II_ORI, immZExt16, or> { int AddedComplexity = 1; } class XOR_MMR6_DESC : ArithLogicR<"xor", GPR32Opnd, 1, II_XOR, xor>; class XORI_MMR6_DESC : ArithLogicI<"xori", uimm16, GPR32Opnd, II_XORI, immZExt16, xor>; class SW_MMR6_DESC : Store<"sw", GPR32Opnd> { InstrItinClass Itinerary = II_SW; } class WRPGPR_WSBH_MMR6_DESC_BASE { dag InOperandList = (ins RO:$rs); dag OutOperandList = (outs RO:$rt); string AsmString = !strconcat(instr_asm, "\t$rt, $rs"); list Pattern = []; Format f = FrmR; string BaseOpcode = instr_asm; bit hasSideEffects = 0; InstrItinClass Itinerary = Itin; } class WRPGPR_MMR6_DESC : WRPGPR_WSBH_MMR6_DESC_BASE<"wrpgpr", GPR32Opnd, II_WRPGPR>; class WSBH_MMR6_DESC : WRPGPR_WSBH_MMR6_DESC_BASE<"wsbh", GPR32Opnd, II_WSBH>; class MTC0_MMR6_DESC_BASE { dag InOperandList = (ins SrcRC:$rt, uimm3:$sel); dag OutOperandList = (outs DstRC:$rs); string AsmString = !strconcat(opstr, "\t$rt, $rs, $sel"); list Pattern = []; Format f = FrmFR; string BaseOpcode = opstr; InstrItinClass Itinerary = Itin; } class MTC1_MMR6_DESC_BASE< string opstr, RegisterOperand DstRC, RegisterOperand SrcRC, InstrItinClass Itin = NoItinerary, SDPatternOperator OpNode = null_frag> : MipsR6Inst { dag InOperandList = (ins SrcRC:$rt); dag OutOperandList = (outs DstRC:$fs); string AsmString = !strconcat(opstr, "\t$rt, $fs"); list Pattern = [(set DstRC:$fs, (OpNode SrcRC:$rt))]; Format f = FrmFR; InstrItinClass Itinerary = Itin; string BaseOpcode = opstr; } class MTC1_64_MMR6_DESC_BASE< string opstr, RegisterOperand DstRC, RegisterOperand SrcRC, InstrItinClass Itin = NoItinerary> : MipsR6Inst { dag InOperandList = (ins DstRC:$fs_in, SrcRC:$rt); dag OutOperandList = (outs DstRC:$fs); string AsmString = !strconcat(opstr, "\t$rt, $fs"); list Pattern = []; Format f = FrmFR; InstrItinClass Itinerary = Itin; string BaseOpcode = opstr; // $fs_in is part of a white lie to work around a widespread bug in the FPU // implementation. See expandBuildPairF64 for details. let Constraints = "$fs = $fs_in"; } class MTC2_MMR6_DESC_BASE { dag InOperandList = (ins SrcRC:$rt); dag OutOperandList = (outs DstRC:$impl); string AsmString = !strconcat(opstr, "\t$rt, $impl"); list Pattern = []; Format f = FrmFR; string BaseOpcode = opstr; InstrItinClass Itinerary = Itin; } class MTC0_MMR6_DESC : MTC0_MMR6_DESC_BASE<"mtc0", COP0Opnd, GPR32Opnd, II_MTC0>; class MTC1_MMR6_DESC : MTC1_MMR6_DESC_BASE<"mtc1", FGR32Opnd, GPR32Opnd, II_MTC1, bitconvert>, HARDFLOAT; class MTC2_MMR6_DESC : MTC2_MMR6_DESC_BASE<"mtc2", COP2Opnd, GPR32Opnd, II_MTC2>; class MTHC0_MMR6_DESC : MTC0_MMR6_DESC_BASE<"mthc0", COP0Opnd, GPR32Opnd, II_MTHC0>; class MTHC2_MMR6_DESC : MTC2_MMR6_DESC_BASE<"mthc2", COP2Opnd, GPR32Opnd, II_MTC2>; class MFC0_MMR6_DESC_BASE { dag InOperandList = (ins SrcRC:$rs, uimm3:$sel); dag OutOperandList = (outs DstRC:$rt); string AsmString = !strconcat(opstr, "\t$rt, $rs, $sel"); list Pattern = []; Format f = FrmFR; string BaseOpcode = opstr; InstrItinClass Itinerary = Itin; } class MFC1_MMR6_DESC_BASE : MipsR6Inst { dag InOperandList = (ins SrcRC:$fs); dag OutOperandList = (outs DstRC:$rt); string AsmString = !strconcat(opstr, "\t$rt, $fs"); list Pattern = [(set DstRC:$rt, (OpNode SrcRC:$fs))]; Format f = FrmFR; InstrItinClass Itinerary = Itin; string BaseOpcode = opstr; } class MFC2_MMR6_DESC_BASE { dag InOperandList = (ins SrcRC:$impl); dag OutOperandList = (outs DstRC:$rt); string AsmString = !strconcat(opstr, "\t$rt, $impl"); list Pattern = []; Format f = FrmFR; string BaseOpcode = opstr; InstrItinClass Itinerary = Itin; } class MFC0_MMR6_DESC : MFC0_MMR6_DESC_BASE<"mfc0", GPR32Opnd, COP0Opnd, II_MFC0>; class MFC1_MMR6_DESC : MFC1_MMR6_DESC_BASE<"mfc1", GPR32Opnd, FGR32Opnd, II_MFC1, bitconvert>, HARDFLOAT; class MFC2_MMR6_DESC : MFC2_MMR6_DESC_BASE<"mfc2", GPR32Opnd, COP2Opnd, II_MFC2>; class MFHC0_MMR6_DESC : MFC0_MMR6_DESC_BASE<"mfhc0", GPR32Opnd, COP0Opnd, II_MFHC0>; class MFHC2_MMR6_DESC : MFC2_MMR6_DESC_BASE<"mfhc2", GPR32Opnd, COP2Opnd, II_MFC2>; class LDC1_D64_MMR6_DESC : MipsR6Inst, HARDFLOAT, FGR_64 { dag InOperandList = (ins mem_mm_16:$addr); dag OutOperandList = (outs FGR64Opnd:$ft); string AsmString = !strconcat("ldc1", "\t$ft, $addr"); list Pattern = [(set FGR64Opnd:$ft, (load addrimm16:$addr))]; Format f = FrmFI; InstrItinClass Itinerary = II_LDC1; string BaseOpcode = "ldc1"; bit mayLoad = 1; let DecoderMethod = "DecodeFMemMMR2"; } class SDC1_D64_MMR6_DESC : MipsR6Inst, HARDFLOAT, FGR_64 { dag InOperandList = (ins FGR64Opnd:$ft, mem_mm_16:$addr); dag OutOperandList = (outs); string AsmString = !strconcat("sdc1", "\t$ft, $addr"); list Pattern = [(store FGR64Opnd:$ft, addrimm16:$addr)]; Format f = FrmFI; InstrItinClass Itinerary = II_SDC1; string BaseOpcode = "sdc1"; bit mayStore = 1; let DecoderMethod = "DecodeFMemMMR2"; } class LDC2_LWC2_MMR6_DESC_BASE { dag OutOperandList = (outs COP2Opnd:$rt); dag InOperandList = (ins mem_mm_11:$addr); string AsmString = !strconcat(opstr, "\t$rt, $addr"); list Pattern = [(set COP2Opnd:$rt, (load addrimm11:$addr))]; Format f = FrmFI; InstrItinClass Itinerary = itin; string BaseOpcode = opstr; bit mayLoad = 1; string DecoderMethod = "DecodeFMemCop2MMR6"; } class LDC2_MMR6_DESC : LDC2_LWC2_MMR6_DESC_BASE<"ldc2", II_LDC2>; class LWC2_MMR6_DESC : LDC2_LWC2_MMR6_DESC_BASE<"lwc2", II_LWC2>; class SDC2_SWC2_MMR6_DESC_BASE { dag OutOperandList = (outs); dag InOperandList = (ins COP2Opnd:$rt, mem_mm_11:$addr); string AsmString = !strconcat(opstr, "\t$rt, $addr"); list Pattern = [(store COP2Opnd:$rt, addrimm11:$addr)]; Format f = FrmFI; InstrItinClass Itinerary = itin; string BaseOpcode = opstr; bit mayStore = 1; string DecoderMethod = "DecodeFMemCop2MMR6"; } class SDC2_MMR6_DESC : SDC2_SWC2_MMR6_DESC_BASE<"sdc2", II_SDC2>; class SWC2_MMR6_DESC : SDC2_SWC2_MMR6_DESC_BASE<"swc2", II_SWC2>; class GINV_MMR6_DESC_BASE { dag InOperandList = (ins SrcRC:$rs, uimm2:$type); dag OutOperandList = (outs); string AsmString = !strconcat(opstr, "\t$rs, $type"); list Pattern = []; Format f = FrmFR; string BaseOpcode = opstr; InstrItinClass Itinerary = Itin; } class GINVI_MMR6_DESC : GINV_MMR6_DESC_BASE<"ginvi", GPR32Opnd, II_GINVI> { dag InOperandList = (ins GPR32Opnd:$rs); string AsmString = "ginvi\t$rs"; } class GINVT_MMR6_DESC : GINV_MMR6_DESC_BASE<"ginvt", GPR32Opnd, II_GINVT>; class SC_MMR6_DESC_BASE { dag OutOperandList = (outs GPR32Opnd:$dst); dag InOperandList = (ins GPR32Opnd:$rt, mem_mm_9:$addr); string AsmString = !strconcat(opstr, "\t$rt, $addr"); InstrItinClass Itinerary = itin; string BaseOpcode = opstr; bit mayStore = 1; string Constraints = "$rt = $dst"; string DecoderMethod = "DecodeMemMMImm9"; } class LL_MMR6_DESC_BASE { dag OutOperandList = (outs GPR32Opnd:$rt); dag InOperandList = (ins mem_mm_9:$addr); string AsmString = !strconcat(opstr, "\t$rt, $addr"); InstrItinClass Itinerary = itin; string BaseOpcode = opstr; bit mayLoad = 1; string DecoderMethod = "DecodeMemMMImm9"; } class SC_MMR6_DESC : SC_MMR6_DESC_BASE<"sc", II_SC>; class LL_MMR6_DESC : LL_MMR6_DESC_BASE<"ll", II_LL>; /// Floating Point Instructions class FARITH_MMR6_DESC_BASE : HARDFLOAT { dag OutOperandList = (outs RC:$fd); dag InOperandList = (ins RC:$ft, RC:$fs); string AsmString = !strconcat(instr_asm, "\t$fd, $fs, $ft"); list Pattern = [(set RC:$fd, (OpNode RC:$fs, RC:$ft))]; InstrItinClass Itinerary = Itin; bit isCommutable = isComm; } class FADD_S_MMR6_DESC : FARITH_MMR6_DESC_BASE<"add.s", FGR32Opnd, II_ADD_S, 1, fadd>; class FSUB_S_MMR6_DESC : FARITH_MMR6_DESC_BASE<"sub.s", FGR32Opnd, II_SUB_S, 0, fsub>; class FMUL_S_MMR6_DESC : FARITH_MMR6_DESC_BASE<"mul.s", FGR32Opnd, II_MUL_S, 1, fmul>; class FDIV_S_MMR6_DESC : FARITH_MMR6_DESC_BASE<"div.s", FGR32Opnd, II_DIV_S, 0, fdiv>; class MADDF_S_MMR6_DESC : COP1_4R_DESC_BASE<"maddf.s", FGR32Opnd, II_MADDF_S>, HARDFLOAT; class MADDF_D_MMR6_DESC : COP1_4R_DESC_BASE<"maddf.d", FGR64Opnd, II_MADDF_D>, HARDFLOAT; class MSUBF_S_MMR6_DESC : COP1_4R_DESC_BASE<"msubf.s", FGR32Opnd, II_MSUBF_S>, HARDFLOAT; class MSUBF_D_MMR6_DESC : COP1_4R_DESC_BASE<"msubf.d", FGR64Opnd, II_MSUBF_D>, HARDFLOAT; class FMOV_FNEG_MMR6_DESC_BASE : HARDFLOAT, NeverHasSideEffects { dag OutOperandList = (outs DstRC:$ft); dag InOperandList = (ins SrcRC:$fs); string AsmString = !strconcat(instr_asm, "\t$ft, $fs"); list Pattern = [(set DstRC:$ft, (OpNode SrcRC:$fs))]; InstrItinClass Itinerary = Itin; Format Form = FrmFR; } class FMOV_S_MMR6_DESC : FMOV_FNEG_MMR6_DESC_BASE<"mov.s", FGR32Opnd, FGR32Opnd, II_MOV_S>; class FNEG_S_MMR6_DESC : FMOV_FNEG_MMR6_DESC_BASE<"neg.s", FGR32Opnd, FGR32Opnd, II_NEG, fneg>; class MAX_S_MMR6_DESC : MAX_MIN_DESC_BASE<"max.s", FGR32Opnd, II_MAX_S>, HARDFLOAT; class MAX_D_MMR6_DESC : MAX_MIN_DESC_BASE<"max.d", FGR64Opnd, II_MAX_D>, HARDFLOAT; class MIN_S_MMR6_DESC : MAX_MIN_DESC_BASE<"min.s", FGR32Opnd, II_MIN_S>, HARDFLOAT; class MIN_D_MMR6_DESC : MAX_MIN_DESC_BASE<"min.d", FGR64Opnd, II_MIN_D>, HARDFLOAT; class MAXA_S_MMR6_DESC : MAX_MIN_DESC_BASE<"maxa.s", FGR32Opnd, II_MAXA_S>, HARDFLOAT; class MAXA_D_MMR6_DESC : MAX_MIN_DESC_BASE<"maxa.d", FGR64Opnd, II_MAXA_D>, HARDFLOAT; class MINA_S_MMR6_DESC : MAX_MIN_DESC_BASE<"mina.s", FGR32Opnd, II_MINA_S>, HARDFLOAT; class MINA_D_MMR6_DESC : MAX_MIN_DESC_BASE<"mina.d", FGR64Opnd, II_MINA_D>, HARDFLOAT; class CVT_MMR6_DESC_BASE< string instr_asm, RegisterOperand DstRC, RegisterOperand SrcRC, InstrItinClass Itin, SDPatternOperator OpNode = null_frag> : HARDFLOAT, NeverHasSideEffects { dag OutOperandList = (outs DstRC:$ft); dag InOperandList = (ins SrcRC:$fs); string AsmString = !strconcat(instr_asm, "\t$ft, $fs"); list Pattern = [(set DstRC:$ft, (OpNode SrcRC:$fs))]; InstrItinClass Itinerary = Itin; Format Form = FrmFR; } class CVT_L_S_MMR6_DESC : CVT_MMR6_DESC_BASE<"cvt.l.s", FGR64Opnd, FGR32Opnd, II_CVT>; class CVT_L_D_MMR6_DESC : CVT_MMR6_DESC_BASE<"cvt.l.d", FGR64Opnd, FGR64Opnd, II_CVT>; class CVT_W_S_MMR6_DESC : CVT_MMR6_DESC_BASE<"cvt.w.s", FGR32Opnd, FGR32Opnd, II_CVT>; class CVT_D_L_MMR6_DESC : CVT_MMR6_DESC_BASE<"cvt.d.l", FGR64Opnd, FGR64Opnd, II_CVT>, FGR_64; class CVT_S_W_MMR6_DESC : CVT_MMR6_DESC_BASE<"cvt.s.w", FGR32Opnd, FGR32Opnd, II_CVT>; class CVT_S_L_MMR6_DESC : CVT_MMR6_DESC_BASE<"cvt.s.l", FGR64Opnd, FGR32Opnd, II_CVT>, FGR_64; multiclass CMP_CC_MMR6 format, string Typestr, RegisterOperand FGROpnd, InstrItinClass Itin> { def CMP_AF_#NAME : R6MMR6Rel, POOL32F_CMP_FM< !strconcat("cmp.af.", Typestr), format, FIELD_CMP_COND_AF>, CMP_CONDN_DESC_BASE<"af", Typestr, FGROpnd, Itin>, HARDFLOAT, ISA_MICROMIPS32R6; def CMP_UN_#NAME : R6MMR6Rel, POOL32F_CMP_FM< !strconcat("cmp.un.", Typestr), format, FIELD_CMP_COND_UN>, CMP_CONDN_DESC_BASE<"un", Typestr, FGROpnd, Itin, setuo>, HARDFLOAT, ISA_MICROMIPS32R6; def CMP_EQ_#NAME : R6MMR6Rel, POOL32F_CMP_FM< !strconcat("cmp.eq.", Typestr), format, FIELD_CMP_COND_EQ>, CMP_CONDN_DESC_BASE<"eq", Typestr, FGROpnd, Itin, setoeq>, HARDFLOAT, ISA_MICROMIPS32R6; def CMP_UEQ_#NAME : R6MMR6Rel, POOL32F_CMP_FM< !strconcat("cmp.ueq.", Typestr), format, FIELD_CMP_COND_UEQ>, CMP_CONDN_DESC_BASE<"ueq", Typestr, FGROpnd, Itin, setueq>, HARDFLOAT, ISA_MICROMIPS32R6; def CMP_LT_#NAME : R6MMR6Rel, POOL32F_CMP_FM< !strconcat("cmp.lt.", Typestr), format, FIELD_CMP_COND_LT>, CMP_CONDN_DESC_BASE<"lt", Typestr, FGROpnd, Itin, setolt>, HARDFLOAT, ISA_MICROMIPS32R6; def CMP_ULT_#NAME : R6MMR6Rel, POOL32F_CMP_FM< !strconcat("cmp.ult.", Typestr), format, FIELD_CMP_COND_ULT>, CMP_CONDN_DESC_BASE<"ult", Typestr, FGROpnd, Itin, setult>, HARDFLOAT, ISA_MICROMIPS32R6; def CMP_LE_#NAME : R6MMR6Rel, POOL32F_CMP_FM< !strconcat("cmp.le.", Typestr), format, FIELD_CMP_COND_LE>, CMP_CONDN_DESC_BASE<"le", Typestr, FGROpnd, Itin, setole>, HARDFLOAT, ISA_MICROMIPS32R6; def CMP_ULE_#NAME : R6MMR6Rel, POOL32F_CMP_FM< !strconcat("cmp.ule.", Typestr), format, FIELD_CMP_COND_ULE>, CMP_CONDN_DESC_BASE<"ule", Typestr, FGROpnd, Itin, setule>, HARDFLOAT, ISA_MICROMIPS32R6; def CMP_SAF_#NAME : R6MMR6Rel, POOL32F_CMP_FM< !strconcat("cmp.saf.", Typestr), format, FIELD_CMP_COND_SAF>, CMP_CONDN_DESC_BASE<"saf", Typestr, FGROpnd, Itin>, HARDFLOAT, ISA_MICROMIPS32R6; def CMP_SUN_#NAME : R6MMR6Rel, POOL32F_CMP_FM< !strconcat("cmp.sun.", Typestr), format, FIELD_CMP_COND_SUN>, CMP_CONDN_DESC_BASE<"sun", Typestr, FGROpnd, Itin>, HARDFLOAT, ISA_MICROMIPS32R6; def CMP_SEQ_#NAME : R6MMR6Rel, POOL32F_CMP_FM< !strconcat("cmp.seq.", Typestr), format, FIELD_CMP_COND_SEQ>, CMP_CONDN_DESC_BASE<"seq", Typestr, FGROpnd, Itin>, HARDFLOAT, ISA_MICROMIPS32R6; def CMP_SUEQ_#NAME : R6MMR6Rel, POOL32F_CMP_FM< !strconcat("cmp.sueq.", Typestr), format, FIELD_CMP_COND_SUEQ>, CMP_CONDN_DESC_BASE<"sueq", Typestr, FGROpnd, Itin>, HARDFLOAT, ISA_MICROMIPS32R6; def CMP_SLT_#NAME : R6MMR6Rel, POOL32F_CMP_FM< !strconcat("cmp.slt.", Typestr), format, FIELD_CMP_COND_SLT>, CMP_CONDN_DESC_BASE<"slt", Typestr, FGROpnd, Itin>, HARDFLOAT, ISA_MICROMIPS32R6; def CMP_SULT_#NAME : R6MMR6Rel, POOL32F_CMP_FM< !strconcat("cmp.sult.", Typestr), format, FIELD_CMP_COND_SULT>, CMP_CONDN_DESC_BASE<"sult", Typestr, FGROpnd, Itin>, HARDFLOAT, ISA_MICROMIPS32R6; def CMP_SLE_#NAME : R6MMR6Rel, POOL32F_CMP_FM< !strconcat("cmp.sle.", Typestr), format, FIELD_CMP_COND_SLE>, CMP_CONDN_DESC_BASE<"sle", Typestr, FGROpnd, Itin>, HARDFLOAT, ISA_MICROMIPS32R6; def CMP_SULE_#NAME : R6MMR6Rel, POOL32F_CMP_FM< !strconcat("cmp.sule.", Typestr), format, FIELD_CMP_COND_SULE>, CMP_CONDN_DESC_BASE<"sule", Typestr, FGROpnd, Itin>, HARDFLOAT, ISA_MICROMIPS32R6; } class ABSS_FT_MMR6_DESC_BASE : HARDFLOAT, NeverHasSideEffects { dag OutOperandList = (outs DstRC:$ft); dag InOperandList = (ins SrcRC:$fs); string AsmString = !strconcat(instr_asm, "\t$ft, $fs"); list Pattern = [(set DstRC:$ft, (OpNode SrcRC:$fs))]; InstrItinClass Itinerary = Itin; Format Form = FrmFR; list EncodingPredicates = [HasStdEnc]; } class FLOOR_L_S_MMR6_DESC : ABSS_FT_MMR6_DESC_BASE<"floor.l.s", FGR64Opnd, FGR32Opnd, II_FLOOR>; class FLOOR_L_D_MMR6_DESC : ABSS_FT_MMR6_DESC_BASE<"floor.l.d", FGR64Opnd, FGR64Opnd, II_FLOOR>; class FLOOR_W_S_MMR6_DESC : ABSS_FT_MMR6_DESC_BASE<"floor.w.s", FGR32Opnd, FGR32Opnd, II_FLOOR>; class FLOOR_W_D_MMR6_DESC : ABSS_FT_MMR6_DESC_BASE<"floor.w.d", FGR32Opnd, AFGR64Opnd, II_FLOOR>; class CEIL_L_S_MMR6_DESC : ABSS_FT_MMR6_DESC_BASE<"ceil.l.s", FGR64Opnd, FGR32Opnd, II_CEIL>; class CEIL_L_D_MMR6_DESC : ABSS_FT_MMR6_DESC_BASE<"ceil.l.d", FGR64Opnd, FGR64Opnd, II_CEIL>; class CEIL_W_S_MMR6_DESC : ABSS_FT_MMR6_DESC_BASE<"ceil.w.s", FGR32Opnd, FGR32Opnd, II_CEIL>; class CEIL_W_D_MMR6_DESC : ABSS_FT_MMR6_DESC_BASE<"ceil.w.d", FGR32Opnd, AFGR64Opnd, II_CEIL>; class TRUNC_L_S_MMR6_DESC : ABSS_FT_MMR6_DESC_BASE<"trunc.l.s", FGR64Opnd, FGR32Opnd, II_TRUNC>; class TRUNC_L_D_MMR6_DESC : ABSS_FT_MMR6_DESC_BASE<"trunc.l.d", FGR64Opnd, FGR64Opnd, II_TRUNC>; class TRUNC_W_S_MMR6_DESC : ABSS_FT_MMR6_DESC_BASE<"trunc.w.s", FGR32Opnd, FGR32Opnd, II_TRUNC>; class TRUNC_W_D_MMR6_DESC : ABSS_FT_MMR6_DESC_BASE<"trunc.w.d", FGR32Opnd, AFGR64Opnd, II_TRUNC>; class SQRT_S_MMR6_DESC : ABSS_FT_MMR6_DESC_BASE<"sqrt.s", FGR32Opnd, FGR32Opnd, II_SQRT_S, fsqrt>; class SQRT_D_MMR6_DESC : ABSS_FT_MMR6_DESC_BASE<"sqrt.d", AFGR64Opnd, AFGR64Opnd, II_SQRT_D, fsqrt>; class ROUND_L_S_MMR6_DESC : ABSS_FT_MMR6_DESC_BASE<"round.l.s", FGR64Opnd, FGR32Opnd, II_ROUND>; class ROUND_L_D_MMR6_DESC : ABSS_FT_MMR6_DESC_BASE<"round.l.d", FGR64Opnd, FGR64Opnd, II_ROUND>; class ROUND_W_S_MMR6_DESC : ABSS_FT_MMR6_DESC_BASE<"round.w.s", FGR32Opnd, FGR32Opnd, II_ROUND>; class ROUND_W_D_MMR6_DESC : ABSS_FT_MMR6_DESC_BASE<"round.w.d", FGR64Opnd, FGR64Opnd, II_ROUND>; class SEL_S_MMR6_DESC : COP1_SEL_DESC_BASE<"sel.s", FGR32Opnd, II_SEL_S>; class SEL_D_MMR6_DESC : COP1_SEL_D_DESC_BASE<"sel.d", FGR64Opnd, II_SEL_D>; class SELEQZ_S_MMR6_DESC : SELEQNEZ_DESC_BASE<"seleqz.s", FGR32Opnd, II_SELCCZ_S>; class SELEQZ_D_MMR6_DESC : SELEQNEZ_DESC_BASE<"seleqz.d", FGR64Opnd, II_SELCCZ_D>; class SELNEZ_S_MMR6_DESC : SELEQNEZ_DESC_BASE<"selnez.s", FGR32Opnd, II_SELCCZ_S>; class SELNEZ_D_MMR6_DESC : SELEQNEZ_DESC_BASE<"selnez.d", FGR64Opnd, II_SELCCZ_D>; class RINT_S_MMR6_DESC : CLASS_RINT_DESC_BASE<"rint.s", FGR32Opnd, II_RINT_S>; class RINT_D_MMR6_DESC : CLASS_RINT_DESC_BASE<"rint.d", FGR64Opnd, II_RINT_S>; class CLASS_S_MMR6_DESC : CLASS_RINT_DESC_BASE<"class.s", FGR32Opnd, II_CLASS_S>; class CLASS_D_MMR6_DESC : CLASS_RINT_DESC_BASE<"class.d", FGR64Opnd, II_CLASS_S>; class STORE_MMR6_DESC_BASE : Store, MMR6Arch { let DecoderMethod = "DecodeMemMMImm16"; InstrItinClass Itinerary = Itin; } class SB_MMR6_DESC : STORE_MMR6_DESC_BASE<"sb", GPR32Opnd, II_SB>; class SH_MMR6_DESC : STORE_MMR6_DESC_BASE<"sh", GPR32Opnd, II_SH>; class ADDU16_MMR6_DESC : ArithRMM16<"addu16", GPRMM16Opnd, 1, II_ADDU, add>, MMR6Arch<"addu16"> { int AddedComplexity = 1; } class AND16_MMR6_DESC : LogicRMM16<"and16", GPRMM16Opnd, II_AND>, MMR6Arch<"and16">; class ANDI16_MMR6_DESC : AndImmMM16<"andi16", GPRMM16Opnd, II_AND>, MMR6Arch<"andi16">; class NOT16_MMR6_DESC : NotMM16<"not16", GPRMM16Opnd>, MMR6Arch<"not16"> { int AddedComplexity = 1; } class OR16_MMR6_DESC : LogicRMM16<"or16", GPRMM16Opnd, II_OR>, MMR6Arch<"or16">; class SLL16_MMR6_DESC : ShiftIMM16<"sll16", uimm3_shift, GPRMM16Opnd, II_SLL>, MMR6Arch<"sll16">; class SRL16_MMR6_DESC : ShiftIMM16<"srl16", uimm3_shift, GPRMM16Opnd, II_SRL>, MMR6Arch<"srl16">; class BREAK16_MMR6_DESC : BrkSdbbp16MM<"break16", II_BREAK>, MMR6Arch<"break16">; class LI16_MMR6_DESC : LoadImmMM16<"li16", li16_imm, GPRMM16Opnd>, MMR6Arch<"li16">, IsAsCheapAsAMove; class MOVE16_MMR6_DESC : MoveMM16<"move16", GPR32Opnd>, MMR6Arch<"move16">; class MOVEP_MMR6_DESC : MovePMM16<"movep", GPRMM16OpndMovePPairFirst, GPRMM16OpndMovePPairSecond, GPRMM16OpndMoveP>, MMR6Arch<"movep">; class SDBBP16_MMR6_DESC : BrkSdbbp16MM<"sdbbp16", II_SDBBP>, MMR6Arch<"sdbbp16">; class SUBU16_MMR6_DESC : ArithRMM16<"subu16", GPRMM16Opnd, 0, II_SUBU, sub>, MMR6Arch<"subu16"> { int AddedComplexity = 1; } class XOR16_MMR6_DESC : LogicRMM16<"xor16", GPRMM16Opnd, II_XOR>, MMR6Arch<"xor16">; class LW_MMR6_DESC : MMR6Arch<"lw">, MipsR6Inst { dag OutOperandList = (outs GPR32Opnd:$rt); dag InOperandList = (ins mem:$addr); string AsmString = "lw\t$rt, $addr"; let DecoderMethod = "DecodeMemMMImm16"; let canFoldAsLoad = 1; let mayLoad = 1; list Pattern = [(set GPR32Opnd:$rt, (load addrDefault:$addr))]; InstrItinClass Itinerary = II_LW; } class LUI_MMR6_DESC : IsAsCheapAsAMove, MMR6Arch<"lui">, MipsR6Inst{ dag OutOperandList = (outs GPR32Opnd:$rt); dag InOperandList = (ins uimm16:$imm16); string AsmString = "lui\t$rt, $imm16"; list Pattern = []; bit hasSideEffects = 0; bit isReMaterializable = 1; InstrItinClass Itinerary = II_LUI; Format Form = FrmI; } class SYNC_MMR6_DESC : MMR6Arch<"sync">, MipsR6Inst { dag OutOperandList = (outs); dag InOperandList = (ins uimm5:$stype); string AsmString = !strconcat("sync", "\t$stype"); list Pattern = [(MipsSync immZExt5:$stype)]; InstrItinClass Itinerary = II_SYNC; bit HasSideEffects = 1; } class SYNCI_MMR6_DESC : SYNCI_FT<"synci", mem_mm_16> { let DecoderMethod = "DecodeSynciR6"; } class RDPGPR_MMR6_DESC : MMR6Arch<"rdpgpr">, MipsR6Inst { dag OutOperandList = (outs GPR32Opnd:$rt); dag InOperandList = (ins GPR32Opnd:$rd); string AsmString = !strconcat("rdpgpr", "\t$rt, $rd"); InstrItinClass Itinerary = II_RDPGPR; } class SDBBP_MMR6_DESC : MipsR6Inst { dag OutOperandList = (outs); dag InOperandList = (ins uimm20:$code_); string AsmString = !strconcat("sdbbp", "\t$code_"); list Pattern = []; InstrItinClass Itinerary = II_SDBBP; } class SIGRIE_MMR6_DESC : MipsR6Inst { dag OutOperandList = (outs); dag InOperandList = (ins uimm16:$code_); string AsmString = !strconcat("sigrie", "\t$code_"); list Pattern = []; InstrItinClass Itinerary = II_SIGRIE; } class LWM16_MMR6_DESC : MicroMipsInst16<(outs reglist16:$rt), (ins mem_mm_4sp:$addr), !strconcat("lwm16", "\t$rt, $addr"), [], II_LWM, FrmI>, MMR6Arch<"lwm16"> { let DecoderMethod = "DecodeMemMMReglistImm4Lsl2"; let mayLoad = 1; ComplexPattern Addr = addr; } class SWM16_MMR6_DESC : MicroMipsInst16<(outs), (ins reglist16:$rt, mem_mm_4sp:$addr), !strconcat("swm16", "\t$rt, $addr"), [], II_SWM, FrmI>, MMR6Arch<"swm16"> { let DecoderMethod = "DecodeMemMMReglistImm4Lsl2"; let mayStore = 1; ComplexPattern Addr = addr; } class SB16_MMR6_DESC_BASE : MicroMipsInst16<(outs), (ins RTOpnd:$rt, MemOpnd:$addr), !strconcat(opstr, "\t$rt, $addr"), [], Itin, FrmI>, MMR6Arch { let DecoderMethod = "DecodeMemMMImm4"; let mayStore = 1; } class SB16_MMR6_DESC : SB16_MMR6_DESC_BASE<"sb16", GPRMM16OpndZero, GPRMM16Opnd, truncstorei8, II_SB, mem_mm_4>; class SH16_MMR6_DESC : SB16_MMR6_DESC_BASE<"sh16", GPRMM16OpndZero, GPRMM16Opnd, truncstorei16, II_SH, mem_mm_4_lsl1>; class SW16_MMR6_DESC : SB16_MMR6_DESC_BASE<"sw16", GPRMM16OpndZero, GPRMM16Opnd, store, II_SW, mem_mm_4_lsl2>; class SWSP_MMR6_DESC : MicroMipsInst16<(outs), (ins GPR32Opnd:$rt, mem_mm_sp_imm5_lsl2:$offset), !strconcat("sw", "\t$rt, $offset"), [], II_SW, FrmI>, MMR6Arch<"sw"> { let DecoderMethod = "DecodeMemMMSPImm5Lsl2"; let mayStore = 1; } class JALRC_HB_MMR6_DESC { dag OutOperandList = (outs GPR32Opnd:$rt); dag InOperandList = (ins GPR32Opnd:$rs); string AsmString = !strconcat("jalrc.hb", "\t$rt, $rs"); list Pattern = []; InstrItinClass Itinerary = II_JALR_HB; Format Form = FrmJ; bit isIndirectBranch = 1; bit hasDelaySlot = 0; } class TLBINV_MMR6_DESC_BASE { dag OutOperandList = (outs); dag InOperandList = (ins); string AsmString = opstr; list Pattern = []; InstrItinClass Itinerary = Itin; } class TLBINV_MMR6_DESC : TLBINV_MMR6_DESC_BASE<"tlbinv", II_TLBINV>; class TLBINVF_MMR6_DESC : TLBINV_MMR6_DESC_BASE<"tlbinvf", II_TLBINVF>; class DVPEVP_MMR6_DESC_BASE { dag OutOperandList = (outs GPR32Opnd:$rs); dag InOperandList = (ins); string AsmString = !strconcat(opstr, "\t$rs"); list Pattern = []; InstrItinClass Itinerary = Itin; bit hasUnModeledSideEffects = 1; } class DVP_MMR6_DESC : DVPEVP_MMR6_DESC_BASE<"dvp", II_DVP>; class EVP_MMR6_DESC : DVPEVP_MMR6_DESC_BASE<"evp", II_EVP>; class BEQZC_MMR6_DESC : CMP_CBR_EQNE_Z_DESC_BASE<"beqzc", brtarget21_mm, GPR32Opnd>, MMR6Arch<"beqzc">; class BNEZC_MMR6_DESC : CMP_CBR_EQNE_Z_DESC_BASE<"bnezc", brtarget21_mm, GPR32Opnd>, MMR6Arch<"bnezc">; class BRANCH_COP1_MMR6_DESC_BASE : InstSE<(outs), (ins FGR64Opnd:$rt, brtarget_mm:$offset), !strconcat(opstr, "\t$rt, $offset"), [], II_BC1CCZ, FrmI>, HARDFLOAT, BRANCH_DESC_BASE { list Defs = [AT]; } class BC1EQZC_MMR6_DESC : BRANCH_COP1_MMR6_DESC_BASE<"bc1eqzc">; class BC1NEZC_MMR6_DESC : BRANCH_COP1_MMR6_DESC_BASE<"bc1nezc">; class BRANCH_COP2_MMR6_DESC_BASE : BRANCH_DESC_BASE { dag InOperandList = (ins COP2Opnd:$rt, brtarget_mm:$offset); dag OutOperandList = (outs); string AsmString = !strconcat(opstr, "\t$rt, $offset"); list Defs = [AT]; InstrItinClass Itinerary = Itin; } class BC2EQZC_MMR6_DESC : BRANCH_COP2_MMR6_DESC_BASE<"bc2eqzc", II_BC2CCZ>; class BC2NEZC_MMR6_DESC : BRANCH_COP2_MMR6_DESC_BASE<"bc2nezc", II_BC2CCZ>; class EXT_MMR6_DESC { dag OutOperandList = (outs GPR32Opnd:$rt); dag InOperandList = (ins GPR32Opnd:$rs, uimm5:$pos, uimm5_plus1:$size); string AsmString = !strconcat("ext", "\t$rt, $rs, $pos, $size"); list Pattern = [(set GPR32Opnd:$rt, (MipsExt GPR32Opnd:$rs, imm:$pos, imm:$size))]; InstrItinClass Itinerary = II_EXT; Format Form = FrmR; string BaseOpcode = "ext"; } class INS_MMR6_DESC { dag OutOperandList = (outs GPR32Opnd:$rt); dag InOperandList = (ins GPR32Opnd:$rs, uimm5:$pos, uimm5_inssize_plus1:$size, GPR32Opnd:$src); string AsmString = !strconcat("ins", "\t$rt, $rs, $pos, $size"); list Pattern = [(set GPR32Opnd:$rt, (MipsIns GPR32Opnd:$rs, imm:$pos, imm:$size, GPR32Opnd:$src))]; InstrItinClass Itinerary = II_INS; Format Form = FrmR; string BaseOpcode = "ins"; string Constraints = "$src = $rt"; } class JALRC_MMR6_DESC { dag OutOperandList = (outs GPR32Opnd:$rt); dag InOperandList = (ins GPR32Opnd:$rs); string AsmString = !strconcat("jalrc", "\t$rt, $rs"); list Pattern = []; InstrItinClass Itinerary = II_JALRC; bit isCall = 1; bit hasDelaySlot = 0; list Defs = [RA]; } class BOVC_BNVC_MMR6_DESC_BASE : BRANCH_DESC_BASE { dag InOperandList = (ins GPROpnd:$rt, GPROpnd:$rs, opnd:$offset); dag OutOperandList = (outs); string AsmString = !strconcat(instr_asm, "\t$rt, $rs, $offset"); list Defs = [AT]; InstrItinClass Itinerary = II_BCCC; } class BOVC_MMR6_DESC : BOVC_BNVC_MMR6_DESC_BASE<"bovc", brtargetr6, GPR32Opnd>; class BNVC_MMR6_DESC : BOVC_BNVC_MMR6_DESC_BASE<"bnvc", brtargetr6, GPR32Opnd>; //===----------------------------------------------------------------------===// // // Instruction Definitions // //===----------------------------------------------------------------------===// let DecoderNamespace = "MicroMipsR6" in { def ADD_MMR6 : StdMMR6Rel, ADD_MMR6_DESC, ADD_MMR6_ENC, ISA_MICROMIPS32R6; def ADDIU_MMR6 : StdMMR6Rel, ADDIU_MMR6_DESC, ADDIU_MMR6_ENC, ISA_MICROMIPS32R6; def ADDU_MMR6 : StdMMR6Rel, ADDU_MMR6_DESC, ADDU_MMR6_ENC, ISA_MICROMIPS32R6; def ADDIUPC_MMR6 : R6MMR6Rel, ADDIUPC_MMR6_ENC, ADDIUPC_MMR6_DESC, ISA_MICROMIPS32R6; def ALUIPC_MMR6 : R6MMR6Rel, ALUIPC_MMR6_ENC, ALUIPC_MMR6_DESC, ISA_MICROMIPS32R6; def AND_MMR6 : StdMMR6Rel, AND_MMR6_DESC, AND_MMR6_ENC, ISA_MICROMIPS32R6; def ANDI_MMR6 : StdMMR6Rel, ANDI_MMR6_DESC, ANDI_MMR6_ENC, ISA_MICROMIPS32R6; def AUIPC_MMR6 : R6MMR6Rel, AUIPC_MMR6_ENC, AUIPC_MMR6_DESC, ISA_MICROMIPS32R6; def ALIGN_MMR6 : R6MMR6Rel, ALIGN_MMR6_ENC, ALIGN_MMR6_DESC, ISA_MICROMIPS32R6; def AUI_MMR6 : R6MMR6Rel, AUI_MMR6_ENC, AUI_MMR6_DESC, ISA_MICROMIPS32R6; def BALC_MMR6 : R6MMR6Rel, BALC_MMR6_ENC, BALC_MMR6_DESC, ISA_MICROMIPS32R6; def BC_MMR6 : R6MMR6Rel, BC_MMR6_ENC, BC_MMR6_DESC, ISA_MICROMIPS32R6; def BC16_MMR6 : StdMMR6Rel, BC16_MMR6_DESC, BC16_MMR6_ENC, ISA_MICROMIPS32R6; def BEQZC_MMR6 : R6MMR6Rel, BEQZC_MMR6_ENC, BEQZC_MMR6_DESC, ISA_MICROMIPS32R6; def BEQZC16_MMR6 : StdMMR6Rel, BEQZC16_MMR6_DESC, BEQZC16_MMR6_ENC, ISA_MICROMIPS32R6; def BNEZC_MMR6 : R6MMR6Rel, BNEZC_MMR6_ENC, BNEZC_MMR6_DESC, ISA_MICROMIPS32R6; def BNEZC16_MMR6 : StdMMR6Rel, BNEZC16_MMR6_DESC, BNEZC16_MMR6_ENC, ISA_MICROMIPS32R6; def BITSWAP_MMR6 : R6MMR6Rel, BITSWAP_MMR6_ENC, BITSWAP_MMR6_DESC, ISA_MICROMIPS32R6; def BEQZALC_MMR6 : R6MMR6Rel, BEQZALC_MMR6_ENC, BEQZALC_MMR6_DESC, ISA_MICROMIPS32R6; def BNEZALC_MMR6 : R6MMR6Rel, BNEZALC_MMR6_ENC, BNEZALC_MMR6_DESC, ISA_MICROMIPS32R6; def BREAK_MMR6 : StdMMR6Rel, BRK_MMR6_DESC, BRK_MMR6_ENC, ISA_MICROMIPS32R6; def CACHE_MMR6 : R6MMR6Rel, CACHE_MMR6_ENC, CACHE_MMR6_DESC, ISA_MICROMIPS32R6; def CLO_MMR6 : R6MMR6Rel, CLO_MMR6_ENC, CLO_MMR6_DESC, ISA_MICROMIPS32R6; def CLZ_MMR6 : R6MMR6Rel, CLZ_MMR6_ENC, CLZ_MMR6_DESC, ISA_MICROMIPS32R6; def DIV_MMR6 : R6MMR6Rel, DIV_MMR6_DESC, DIV_MMR6_ENC, ISA_MICROMIPS32R6; def DIVU_MMR6 : R6MMR6Rel, DIVU_MMR6_DESC, DIVU_MMR6_ENC, ISA_MICROMIPS32R6; def EHB_MMR6 : StdMMR6Rel, EHB_MMR6_DESC, EHB_MMR6_ENC, ISA_MICROMIPS32R6; def EI_MMR6 : StdMMR6Rel, EI_MMR6_DESC, EI_MMR6_ENC, ISA_MICROMIPS32R6; def DI_MMR6 : StdMMR6Rel, DI_MMR6_DESC, DI_MMR6_ENC, ISA_MICROMIPS32R6; def ERET_MMR6 : StdMMR6Rel, ERET_MMR6_DESC, ERET_MMR6_ENC, ISA_MICROMIPS32R6; def DERET_MMR6 : StdMMR6Rel, DERET_MMR6_DESC, DERET_MMR6_ENC, ISA_MICROMIPS32R6; def ERETNC_MMR6 : R6MMR6Rel, ERETNC_MMR6_DESC, ERETNC_MMR6_ENC, ISA_MICROMIPS32R6; def GINVI_MMR6 : R6MMR6Rel, GINVI_MMR6_ENC, GINVI_MMR6_DESC, ISA_MICROMIPS32R6, ASE_GINV; def GINVT_MMR6 : R6MMR6Rel, GINVT_MMR6_ENC, GINVT_MMR6_DESC, ISA_MICROMIPS32R6, ASE_GINV; let FastISelShouldIgnore = 1 in def JALRC16_MMR6 : R6MMR6Rel, JALRC16_MMR6_DESC, JALRC16_MMR6_ENC, ISA_MICROMIPS32R6; def JIALC_MMR6 : R6MMR6Rel, JIALC_MMR6_ENC, JIALC_MMR6_DESC, ISA_MICROMIPS32R6; def JIC_MMR6 : R6MMR6Rel, JIC_MMR6_ENC, JIC_MMR6_DESC, ISA_MICROMIPS32R6; def JRC16_MMR6 : R6MMR6Rel, JRC16_MMR6_DESC, JRC16_MMR6_ENC, ISA_MICROMIPS32R6; def JRCADDIUSP_MMR6 : R6MMR6Rel, JRCADDIUSP_MMR6_DESC, JRCADDIUSP_MMR6_ENC, ISA_MICROMIPS32R6; def LSA_MMR6 : R6MMR6Rel, LSA_MMR6_ENC, LSA_MMR6_DESC, ISA_MICROMIPS32R6; def LWPC_MMR6 : R6MMR6Rel, LWPC_MMR6_ENC, LWPC_MMR6_DESC, ISA_MICROMIPS32R6; def LWM16_MMR6 : StdMMR6Rel, LWM16_MMR6_DESC, LWM16_MMR6_ENC, ISA_MICROMIPS32R6; def MTC0_MMR6 : StdMMR6Rel, MTC0_MMR6_ENC, MTC0_MMR6_DESC, ISA_MICROMIPS32R6; def MTC1_MMR6 : StdMMR6Rel, MTC1_MMR6_DESC, MTC1_MMR6_ENC, ISA_MICROMIPS32R6; def MTC2_MMR6 : StdMMR6Rel, MTC2_MMR6_ENC, MTC2_MMR6_DESC, ISA_MICROMIPS32R6; def MTHC0_MMR6 : R6MMR6Rel, MTHC0_MMR6_ENC, MTHC0_MMR6_DESC, ISA_MICROMIPS32R6; def MTHC2_MMR6 : StdMMR6Rel, MTHC2_MMR6_ENC, MTHC2_MMR6_DESC, ISA_MICROMIPS32R6; def MFC0_MMR6 : StdMMR6Rel, MFC0_MMR6_ENC, MFC0_MMR6_DESC, ISA_MICROMIPS32R6; def MFC1_MMR6 : StdMMR6Rel, MFC1_MMR6_DESC, MFC1_MMR6_ENC, ISA_MICROMIPS32R6; def MFC2_MMR6 : StdMMR6Rel, MFC2_MMR6_ENC, MFC2_MMR6_DESC, ISA_MICROMIPS32R6; def MFHC0_MMR6 : R6MMR6Rel, MFHC0_MMR6_ENC, MFHC0_MMR6_DESC, ISA_MICROMIPS32R6; def MFHC2_MMR6 : StdMMR6Rel, MFHC2_MMR6_ENC, MFHC2_MMR6_DESC, ISA_MICROMIPS32R6; def MOD_MMR6 : R6MMR6Rel, MOD_MMR6_DESC, MOD_MMR6_ENC, ISA_MICROMIPS32R6; def MODU_MMR6 : R6MMR6Rel, MODU_MMR6_DESC, MODU_MMR6_ENC, ISA_MICROMIPS32R6; def MUL_MMR6 : R6MMR6Rel, MUL_MMR6_DESC, MUL_MMR6_ENC, ISA_MICROMIPS32R6; def MUH_MMR6 : R6MMR6Rel, MUH_MMR6_DESC, MUH_MMR6_ENC, ISA_MICROMIPS32R6; def MULU_MMR6 : R6MMR6Rel, MULU_MMR6_DESC, MULU_MMR6_ENC, ISA_MICROMIPS32R6; def MUHU_MMR6 : R6MMR6Rel, MUHU_MMR6_DESC, MUHU_MMR6_ENC, ISA_MICROMIPS32R6; def NOR_MMR6 : StdMMR6Rel, NOR_MMR6_DESC, NOR_MMR6_ENC, ISA_MICROMIPS32R6; def OR_MMR6 : StdMMR6Rel, OR_MMR6_DESC, OR_MMR6_ENC, ISA_MICROMIPS32R6; def ORI_MMR6 : StdMMR6Rel, ORI_MMR6_DESC, ORI_MMR6_ENC, ISA_MICROMIPS32R6; def PREF_MMR6 : R6MMR6Rel, PREF_MMR6_ENC, PREF_MMR6_DESC, ISA_MICROMIPS32R6; def SB16_MMR6 : StdMMR6Rel, SB16_MMR6_DESC, SB16_MMR6_ENC, ISA_MICROMIPS32R6; def SELEQZ_MMR6 : R6MMR6Rel, SELEQZ_MMR6_ENC, SELEQZ_MMR6_DESC, ISA_MICROMIPS32R6; def SELNEZ_MMR6 : R6MMR6Rel, SELNEZ_MMR6_ENC, SELNEZ_MMR6_DESC, ISA_MICROMIPS32R6; def SH16_MMR6 : StdMMR6Rel, SH16_MMR6_DESC, SH16_MMR6_ENC, ISA_MICROMIPS32R6; def SLL_MMR6 : StdMMR6Rel, SLL_MMR6_DESC, SLL_MMR6_ENC, ISA_MICROMIPS32R6; def SUB_MMR6 : StdMMR6Rel, SUB_MMR6_DESC, SUB_MMR6_ENC, ISA_MICROMIPS32R6; def SUBU_MMR6 : StdMMR6Rel, SUBU_MMR6_DESC, SUBU_MMR6_ENC, ISA_MICROMIPS32R6; def SW16_MMR6 : StdMMR6Rel, SW16_MMR6_DESC, SW16_MMR6_ENC, ISA_MICROMIPS32R6; def SWM16_MMR6 : StdMMR6Rel, SWM16_MMR6_DESC, SWM16_MMR6_ENC, ISA_MICROMIPS32R6; def SWSP_MMR6 : StdMMR6Rel, SWSP_MMR6_DESC, SWSP_MMR6_ENC, ISA_MICROMIPS32R6; def WRPGPR_MMR6 : StdMMR6Rel, WRPGPR_MMR6_ENC, WRPGPR_MMR6_DESC, ISA_MICROMIPS32R6; def WSBH_MMR6 : StdMMR6Rel, WSBH_MMR6_ENC, WSBH_MMR6_DESC, ISA_MICROMIPS32R6; def LB_MMR6 : R6MMR6Rel, LB_MMR6_ENC, LB_MMR6_DESC, ISA_MICROMIPS32R6; def LBU_MMR6 : R6MMR6Rel, LBU_MMR6_ENC, LBU_MMR6_DESC, ISA_MICROMIPS32R6; def PAUSE_MMR6 : StdMMR6Rel, PAUSE_MMR6_DESC, PAUSE_MMR6_ENC, ISA_MICROMIPS32R6; def RDHWR_MMR6 : R6MMR6Rel, RDHWR_MMR6_DESC, RDHWR_MMR6_ENC, ISA_MICROMIPS32R6; def WAIT_MMR6 : StdMMR6Rel, WAIT_MMR6_DESC, WAIT_MMR6_ENC, ISA_MICROMIPS32R6; def SSNOP_MMR6 : StdMMR6Rel, SSNOP_MMR6_DESC, SSNOP_MMR6_ENC, ISA_MICROMIPS32R6; def SYNC_MMR6 : StdMMR6Rel, SYNC_MMR6_DESC, SYNC_MMR6_ENC, ISA_MICROMIPS32R6; def SYNCI_MMR6 : StdMMR6Rel, SYNCI_MMR6_DESC, SYNCI_MMR6_ENC, ISA_MICROMIPS32R6; def RDPGPR_MMR6 : R6MMR6Rel, RDPGPR_MMR6_DESC, RDPGPR_MMR6_ENC, ISA_MICROMIPS32R6; def SDBBP_MMR6 : R6MMR6Rel, SDBBP_MMR6_DESC, SDBBP_MMR6_ENC, ISA_MICROMIPS32R6; def SIGRIE_MMR6 : R6MMR6Rel, SIGRIE_MMR6_DESC, SIGRIE_MMR6_ENC, ISA_MICROMIPS32R6; def XOR_MMR6 : StdMMR6Rel, XOR_MMR6_DESC, XOR_MMR6_ENC, ISA_MICROMIPS32R6; def XORI_MMR6 : StdMMR6Rel, XORI_MMR6_DESC, XORI_MMR6_ENC, ISA_MICROMIPS32R6; let DecoderMethod = "DecodeMemMMImm16" in { def SW_MMR6 : StdMMR6Rel, SW_MMR6_DESC, SW_MMR6_ENC, ISA_MICROMIPS32R6; } /// Floating Point Instructions def FADD_S_MMR6 : StdMMR6Rel, FADD_S_MMR6_ENC, FADD_S_MMR6_DESC, ISA_MICROMIPS32R6; def FSUB_S_MMR6 : StdMMR6Rel, FSUB_S_MMR6_ENC, FSUB_S_MMR6_DESC, ISA_MICROMIPS32R6; def FMUL_S_MMR6 : StdMMR6Rel, FMUL_S_MMR6_ENC, FMUL_S_MMR6_DESC, ISA_MICROMIPS32R6; def FDIV_S_MMR6 : StdMMR6Rel, FDIV_S_MMR6_ENC, FDIV_S_MMR6_DESC, ISA_MICROMIPS32R6; def MADDF_S_MMR6 : R6MMR6Rel, MADDF_S_MMR6_ENC, MADDF_S_MMR6_DESC, ISA_MICROMIPS32R6; def MADDF_D_MMR6 : R6MMR6Rel, MADDF_D_MMR6_ENC, MADDF_D_MMR6_DESC, ISA_MICROMIPS32R6; def MSUBF_S_MMR6 : R6MMR6Rel, MSUBF_S_MMR6_ENC, MSUBF_S_MMR6_DESC, ISA_MICROMIPS32R6; def MSUBF_D_MMR6 : R6MMR6Rel, MSUBF_D_MMR6_ENC, MSUBF_D_MMR6_DESC, ISA_MICROMIPS32R6; def FMOV_S_MMR6 : StdMMR6Rel, FMOV_S_MMR6_ENC, FMOV_S_MMR6_DESC, ISA_MICROMIPS32R6; def FNEG_S_MMR6 : StdMMR6Rel, FNEG_S_MMR6_ENC, FNEG_S_MMR6_DESC, ISA_MICROMIPS32R6; def MAX_S_MMR6 : R6MMR6Rel, MAX_S_MMR6_ENC, MAX_S_MMR6_DESC, ISA_MICROMIPS32R6; def MAX_D_MMR6 : R6MMR6Rel, MAX_D_MMR6_ENC, MAX_D_MMR6_DESC, ISA_MICROMIPS32R6; def MIN_S_MMR6 : R6MMR6Rel, MIN_S_MMR6_ENC, MIN_S_MMR6_DESC, ISA_MICROMIPS32R6; def MIN_D_MMR6 : R6MMR6Rel, MIN_D_MMR6_ENC, MIN_D_MMR6_DESC, ISA_MICROMIPS32R6; def MAXA_S_MMR6 : R6MMR6Rel, MAXA_S_MMR6_ENC, MAXA_S_MMR6_DESC, ISA_MICROMIPS32R6; def MAXA_D_MMR6 : R6MMR6Rel, MAXA_D_MMR6_ENC, MAXA_D_MMR6_DESC, ISA_MICROMIPS32R6; def MINA_S_MMR6 : R6MMR6Rel, MINA_S_MMR6_ENC, MINA_S_MMR6_DESC, ISA_MICROMIPS32R6; def MINA_D_MMR6 : R6MMR6Rel, MINA_D_MMR6_ENC, MINA_D_MMR6_DESC, ISA_MICROMIPS32R6; def CVT_L_S_MMR6 : StdMMR6Rel, CVT_L_S_MMR6_ENC, CVT_L_S_MMR6_DESC, ISA_MICROMIPS32R6; def CVT_L_D_MMR6 : StdMMR6Rel, CVT_L_D_MMR6_ENC, CVT_L_D_MMR6_DESC, ISA_MICROMIPS32R6; def CVT_W_S_MMR6 : StdMMR6Rel, CVT_W_S_MMR6_ENC, CVT_W_S_MMR6_DESC, ISA_MICROMIPS32R6; def CVT_D_L_MMR6 : StdMMR6Rel, CVT_D_L_MMR6_ENC, CVT_D_L_MMR6_DESC, ISA_MICROMIPS32R6; def CVT_S_W_MMR6 : StdMMR6Rel, CVT_S_W_MMR6_ENC, CVT_S_W_MMR6_DESC, ISA_MICROMIPS32R6; def CVT_S_L_MMR6 : StdMMR6Rel, CVT_S_L_MMR6_ENC, CVT_S_L_MMR6_DESC, ISA_MICROMIPS32R6; defm S_MMR6 : CMP_CC_MMR6<0b000101, "s", FGR32Opnd, II_CMP_CC_S>; defm D_MMR6 : CMP_CC_MMR6<0b010101, "d", FGR64Opnd, II_CMP_CC_D>; def FLOOR_L_S_MMR6 : StdMMR6Rel, FLOOR_L_S_MMR6_ENC, FLOOR_L_S_MMR6_DESC, ISA_MICROMIPS32R6; def FLOOR_L_D_MMR6 : StdMMR6Rel, FLOOR_L_D_MMR6_ENC, FLOOR_L_D_MMR6_DESC, ISA_MICROMIPS32R6; def FLOOR_W_S_MMR6 : StdMMR6Rel, FLOOR_W_S_MMR6_ENC, FLOOR_W_S_MMR6_DESC, ISA_MICROMIPS32R6; def FLOOR_W_D_MMR6 : StdMMR6Rel, FLOOR_W_D_MMR6_ENC, FLOOR_W_D_MMR6_DESC, ISA_MICROMIPS32R6; def CEIL_L_S_MMR6 : StdMMR6Rel, CEIL_L_S_MMR6_ENC, CEIL_L_S_MMR6_DESC, ISA_MICROMIPS32R6; def CEIL_L_D_MMR6 : StdMMR6Rel, CEIL_L_D_MMR6_ENC, CEIL_L_D_MMR6_DESC, ISA_MICROMIPS32R6; def CEIL_W_S_MMR6 : StdMMR6Rel, CEIL_W_S_MMR6_ENC, CEIL_W_S_MMR6_DESC, ISA_MICROMIPS32R6; def CEIL_W_D_MMR6 : StdMMR6Rel, CEIL_W_D_MMR6_ENC, CEIL_W_D_MMR6_DESC, ISA_MICROMIPS32R6; def TRUNC_L_S_MMR6 : StdMMR6Rel, TRUNC_L_S_MMR6_ENC, TRUNC_L_S_MMR6_DESC, ISA_MICROMIPS32R6; def TRUNC_L_D_MMR6 : StdMMR6Rel, TRUNC_L_D_MMR6_ENC, TRUNC_L_D_MMR6_DESC, ISA_MICROMIPS32R6; def TRUNC_W_S_MMR6 : StdMMR6Rel, TRUNC_W_S_MMR6_ENC, TRUNC_W_S_MMR6_DESC, ISA_MICROMIPS32R6; def TRUNC_W_D_MMR6 : StdMMR6Rel, TRUNC_W_D_MMR6_ENC, TRUNC_W_D_MMR6_DESC, ISA_MICROMIPS32R6; def SB_MMR6 : StdMMR6Rel, SB_MMR6_DESC, SB_MMR6_ENC, ISA_MICROMIPS32R6; def SH_MMR6 : StdMMR6Rel, SH_MMR6_DESC, SH_MMR6_ENC, ISA_MICROMIPS32R6; def LW_MMR6 : StdMMR6Rel, LW_MMR6_DESC, LW_MMR6_ENC, ISA_MICROMIPS32R6; def LUI_MMR6 : R6MMR6Rel, LUI_MMR6_DESC, LUI_MMR6_ENC, ISA_MICROMIPS32R6; def ADDU16_MMR6 : StdMMR6Rel, ADDU16_MMR6_DESC, ADDU16_MMR6_ENC, ISA_MICROMIPS32R6; def AND16_MMR6 : StdMMR6Rel, AND16_MMR6_DESC, AND16_MMR6_ENC, ISA_MICROMIPS32R6; def ANDI16_MMR6 : StdMMR6Rel, ANDI16_MMR6_DESC, ANDI16_MMR6_ENC, ISA_MICROMIPS32R6; def NOT16_MMR6 : StdMMR6Rel, NOT16_MMR6_DESC, NOT16_MMR6_ENC, ISA_MICROMIPS32R6; def OR16_MMR6 : StdMMR6Rel, OR16_MMR6_DESC, OR16_MMR6_ENC, ISA_MICROMIPS32R6; def SLL16_MMR6 : StdMMR6Rel, SLL16_MMR6_DESC, SLL16_MMR6_ENC, ISA_MICROMIPS32R6; def SRL16_MMR6 : StdMMR6Rel, SRL16_MMR6_DESC, SRL16_MMR6_ENC, ISA_MICROMIPS32R6; def BREAK16_MMR6 : StdMMR6Rel, BREAK16_MMR6_DESC, BREAK16_MMR6_ENC, ISA_MICROMIPS32R6; def LI16_MMR6 : StdMMR6Rel, LI16_MMR6_DESC, LI16_MMR6_ENC, ISA_MICROMIPS32R6; def MOVE16_MMR6 : StdMMR6Rel, MOVE16_MMR6_DESC, MOVE16_MMR6_ENC, ISA_MICROMIPS32R6; def MOVEP_MMR6 : StdMMR6Rel, MOVEP_MMR6_DESC, MOVEP_MMR6_ENC, ISA_MICROMIPS32R6; def SDBBP16_MMR6 : StdMMR6Rel, SDBBP16_MMR6_DESC, SDBBP16_MMR6_ENC, ISA_MICROMIPS32R6; def SUBU16_MMR6 : StdMMR6Rel, SUBU16_MMR6_DESC, SUBU16_MMR6_ENC, ISA_MICROMIPS32R6; def XOR16_MMR6 : StdMMR6Rel, XOR16_MMR6_DESC, XOR16_MMR6_ENC, ISA_MICROMIPS32R6; def JALRC_HB_MMR6 : R6MMR6Rel, JALRC_HB_MMR6_ENC, JALRC_HB_MMR6_DESC, ISA_MICROMIPS32R6; def EXT_MMR6 : StdMMR6Rel, EXT_MMR6_ENC, EXT_MMR6_DESC, ISA_MICROMIPS32R6; def INS_MMR6 : StdMMR6Rel, INS_MMR6_ENC, INS_MMR6_DESC, ISA_MICROMIPS32R6; def JALRC_MMR6 : R6MMR6Rel, JALRC_MMR6_ENC, JALRC_MMR6_DESC, ISA_MICROMIPS32R6; def RINT_S_MMR6 : StdMMR6Rel, RINT_S_MMR6_ENC, RINT_S_MMR6_DESC, ISA_MICROMIPS32R6; def RINT_D_MMR6 : StdMMR6Rel, RINT_D_MMR6_ENC, RINT_D_MMR6_DESC, ISA_MICROMIPS32R6; def ROUND_L_S_MMR6 : StdMMR6Rel, ROUND_L_S_MMR6_ENC, ROUND_L_S_MMR6_DESC, ISA_MICROMIPS32R6; def ROUND_L_D_MMR6 : StdMMR6Rel, ROUND_L_D_MMR6_ENC, ROUND_L_D_MMR6_DESC, ISA_MICROMIPS32R6; def ROUND_W_S_MMR6 : StdMMR6Rel, ROUND_W_S_MMR6_ENC, ROUND_W_S_MMR6_DESC, ISA_MICROMIPS32R6; def ROUND_W_D_MMR6 : StdMMR6Rel, ROUND_W_D_MMR6_ENC, ROUND_W_D_MMR6_DESC, ISA_MICROMIPS32R6; def SEL_S_MMR6 : R6MMR6Rel, SEL_S_MMR6_ENC, SEL_S_MMR6_DESC, ISA_MICROMIPS32R6; def SEL_D_MMR6 : R6MMR6Rel, SEL_D_MMR6_ENC, SEL_D_MMR6_DESC, ISA_MICROMIPS32R6; def SELEQZ_S_MMR6 : R6MMR6Rel, SELEQZ_S_MMR6_ENC, SELEQZ_S_MMR6_DESC, ISA_MICROMIPS32R6; def SELEQZ_D_MMR6 : R6MMR6Rel, SELEQZ_D_MMR6_ENC, SELEQZ_D_MMR6_DESC, ISA_MICROMIPS32R6; def SELNEZ_S_MMR6 : R6MMR6Rel, SELNEZ_S_MMR6_ENC, SELNEZ_S_MMR6_DESC, ISA_MICROMIPS32R6; def SELNEZ_D_MMR6 : R6MMR6Rel, SELNEZ_D_MMR6_ENC, SELNEZ_D_MMR6_DESC, ISA_MICROMIPS32R6; def CLASS_S_MMR6 : StdMMR6Rel, CLASS_S_MMR6_ENC, CLASS_S_MMR6_DESC, ISA_MICROMIPS32R6; def CLASS_D_MMR6 : StdMMR6Rel, CLASS_D_MMR6_ENC, CLASS_D_MMR6_DESC, ISA_MICROMIPS32R6; def TLBINV_MMR6 : StdMMR6Rel, TLBINV_MMR6_ENC, TLBINV_MMR6_DESC, ISA_MICROMIPS32R6; def TLBINVF_MMR6 : StdMMR6Rel, TLBINVF_MMR6_ENC, TLBINVF_MMR6_DESC, ISA_MICROMIPS32R6; def DVP_MMR6 : R6MMR6Rel, DVP_MMR6_ENC, DVP_MMR6_DESC, ISA_MICROMIPS32R6; def EVP_MMR6 : R6MMR6Rel, EVP_MMR6_ENC, EVP_MMR6_DESC, ISA_MICROMIPS32R6; def BC1EQZC_MMR6 : R6MMR6Rel, BC1EQZC_MMR6_DESC, BC1EQZC_MMR6_ENC, ISA_MICROMIPS32R6; def BC1NEZC_MMR6 : R6MMR6Rel, BC1NEZC_MMR6_DESC, BC1NEZC_MMR6_ENC, ISA_MICROMIPS32R6; def BC2EQZC_MMR6 : R6MMR6Rel, MipsR6Inst, BC2EQZC_MMR6_ENC, BC2EQZC_MMR6_DESC, ISA_MICROMIPS32R6; def BC2NEZC_MMR6 : R6MMR6Rel, MipsR6Inst, BC2NEZC_MMR6_ENC, BC2NEZC_MMR6_DESC, ISA_MICROMIPS32R6; let DecoderNamespace = "MicroMipsFP64" in { def LDC1_D64_MMR6 : StdMMR6Rel, LDC1_D64_MMR6_DESC, LDC1_MMR6_ENC, ISA_MICROMIPS32R6 { let BaseOpcode = "LDC164"; } def SDC1_D64_MMR6 : StdMMR6Rel, SDC1_D64_MMR6_DESC, SDC1_MMR6_ENC, ISA_MICROMIPS32R6; } def LDC2_MMR6 : StdMMR6Rel, LDC2_MMR6_ENC, LDC2_MMR6_DESC, ISA_MICROMIPS32R6; def SDC2_MMR6 : StdMMR6Rel, SDC2_MMR6_ENC, SDC2_MMR6_DESC, ISA_MICROMIPS32R6; def LWC2_MMR6 : StdMMR6Rel, LWC2_MMR6_ENC, LWC2_MMR6_DESC, ISA_MICROMIPS32R6; def SWC2_MMR6 : StdMMR6Rel, SWC2_MMR6_ENC, SWC2_MMR6_DESC, ISA_MICROMIPS32R6; def LL_MMR6 : R6MMR6Rel, LL_MMR6_ENC, LL_MMR6_DESC, ISA_MICROMIPS32R6; def SC_MMR6 : R6MMR6Rel, SC_MMR6_ENC, SC_MMR6_DESC, ISA_MICROMIPS32R6; } def BOVC_MMR6 : R6MMR6Rel, BOVC_MMR6_ENC, BOVC_MMR6_DESC, ISA_MICROMIPS32R6, MMDecodeDisambiguatedBy<"POP35GroupBranchMMR6">; def BNVC_MMR6 : R6MMR6Rel, BNVC_MMR6_ENC, BNVC_MMR6_DESC, ISA_MICROMIPS32R6, MMDecodeDisambiguatedBy<"POP37GroupBranchMMR6">; def BGEC_MMR6 : R6MMR6Rel, BGEC_MMR6_ENC, BGEC_MMR6_DESC, ISA_MICROMIPS32R6; def BGEUC_MMR6 : R6MMR6Rel, BGEUC_MMR6_ENC, BGEUC_MMR6_DESC, ISA_MICROMIPS32R6; def BLTC_MMR6 : R6MMR6Rel, BLTC_MMR6_ENC, BLTC_MMR6_DESC, ISA_MICROMIPS32R6; def BLTUC_MMR6 : R6MMR6Rel, BLTUC_MMR6_ENC, BLTUC_MMR6_DESC, ISA_MICROMIPS32R6; def BEQC_MMR6 : R6MMR6Rel, BEQC_MMR6_ENC, BEQC_MMR6_DESC, ISA_MICROMIPS32R6, DecodeDisambiguates<"POP35GroupBranchMMR6">; def BNEC_MMR6 : R6MMR6Rel, BNEC_MMR6_ENC, BNEC_MMR6_DESC, ISA_MICROMIPS32R6, DecodeDisambiguates<"POP37GroupBranchMMR6">; def BLTZC_MMR6 : R6MMR6Rel, BLTZC_MMR6_ENC, BLTZC_MMR6_DESC, ISA_MICROMIPS32R6; def BLEZC_MMR6 : R6MMR6Rel, BLEZC_MMR6_ENC, BLEZC_MMR6_DESC, ISA_MICROMIPS32R6; def BGEZC_MMR6 : R6MMR6Rel, BGEZC_MMR6_ENC, BGEZC_MMR6_DESC, ISA_MICROMIPS32R6; def BGTZC_MMR6 : R6MMR6Rel, BGTZC_MMR6_ENC, BGTZC_MMR6_DESC, ISA_MICROMIPS32R6; def BGEZALC_MMR6 : R6MMR6Rel, BGEZALC_MMR6_ENC, BGEZALC_MMR6_DESC, ISA_MICROMIPS32R6; def BGTZALC_MMR6 : R6MMR6Rel, BGTZALC_MMR6_ENC, BGTZALC_MMR6_DESC, ISA_MICROMIPS32R6; def BLEZALC_MMR6 : R6MMR6Rel, BLEZALC_MMR6_ENC, BLEZALC_MMR6_DESC, ISA_MICROMIPS32R6; def BLTZALC_MMR6 : R6MMR6Rel, BLTZALC_MMR6_ENC, BLTZALC_MMR6_DESC, ISA_MICROMIPS32R6; //===----------------------------------------------------------------------===// // // MicroMips instruction aliases // //===----------------------------------------------------------------------===// def : MipsInstAlias<"ei", (EI_MMR6 ZERO), 1>, ISA_MICROMIPS32R6; def : MipsInstAlias<"di", (DI_MMR6 ZERO), 1>, ISA_MICROMIPS32R6; def : MipsInstAlias<"nop", (SLL_MMR6 ZERO, ZERO, 0), 1>, ISA_MICROMIPS32R6; def B_MMR6_Pseudo : MipsAsmPseudoInst<(outs), (ins brtarget_mm:$offset), !strconcat("b", "\t$offset")> { string DecoderNamespace = "MicroMipsR6"; } def : MipsInstAlias<"sync", (SYNC_MMR6 0), 1>, ISA_MICROMIPS32R6; def : MipsInstAlias<"sdbbp", (SDBBP_MMR6 0), 1>, ISA_MICROMIPS32R6; def : MipsInstAlias<"sigrie", (SIGRIE_MMR6 0), 1>, ISA_MICROMIPS32R6; def : MipsInstAlias<"rdhwr $rt, $rs", (RDHWR_MMR6 GPR32Opnd:$rt, HWRegsOpnd:$rs, 0), 1>, ISA_MICROMIPS32R6; def : MipsInstAlias<"mtc0 $rt, $rs", (MTC0_MMR6 COP0Opnd:$rs, GPR32Opnd:$rt, 0), 0>, ISA_MICROMIPS32R6; def : MipsInstAlias<"mthc0 $rt, $rs", (MTHC0_MMR6 COP0Opnd:$rs, GPR32Opnd:$rt, 0), 0>, ISA_MICROMIPS32R6; def : MipsInstAlias<"mfc0 $rt, $rs", (MFC0_MMR6 GPR32Opnd:$rt, COP0Opnd:$rs, 0), 0>, ISA_MICROMIPS32R6; def : MipsInstAlias<"mfhc0 $rt, $rs", (MFHC0_MMR6 GPR32Opnd:$rt, COP0Opnd:$rs, 0), 0>, ISA_MICROMIPS32R6; def : MipsInstAlias<"jalrc.hb $rs", (JALRC_HB_MMR6 RA, GPR32Opnd:$rs), 1>, ISA_MICROMIPS32R6; def : MipsInstAlias<"jal $offset", (BALC_MMR6 brtarget26_mm:$offset), 0>, ISA_MICROMIPS32R6; def : MipsInstAlias<"dvp", (DVP_MMR6 ZERO), 0>, ISA_MICROMIPS32R6; def : MipsInstAlias<"evp", (EVP_MMR6 ZERO), 0>, ISA_MICROMIPS32R6; def : MipsInstAlias<"jalrc $rs", (JALRC_MMR6 RA, GPR32Opnd:$rs), 1>, ISA_MICROMIPS32R6; def : MipsInstAlias<"and $rs, $rt, $imm", (ANDI_MMR6 GPR32Opnd:$rs, GPR32Opnd:$rt, uimm16:$imm), 0>, ISA_MICROMIPS32R6; def : MipsInstAlias<"and $rs, $imm", (ANDI_MMR6 GPR32Opnd:$rs, GPR32Opnd:$rs, uimm16:$imm), 0>, ISA_MICROMIPS32R6; def : MipsInstAlias<"or $rs, $rt, $imm", (ORI_MMR6 GPR32Opnd:$rs, GPR32Opnd:$rt, uimm16:$imm), 0>, ISA_MICROMIPS32R6; def : MipsInstAlias<"or $rs, $imm", (ORI_MMR6 GPR32Opnd:$rs, GPR32Opnd:$rs, uimm16:$imm), 0>, ISA_MICROMIPS32R6; def : MipsInstAlias<"xor $rs, $rt, $imm", (XORI_MMR6 GPR32Opnd:$rs, GPR32Opnd:$rt, uimm16:$imm), 0>, ISA_MICROMIPS32R6; def : MipsInstAlias<"xor $rs, $imm", (XORI_MMR6 GPR32Opnd:$rs, GPR32Opnd:$rs, uimm16:$imm), 0>, ISA_MICROMIPS32R6; def : MipsInstAlias<"not $rt, $rs", (NOR_MMR6 GPR32Opnd:$rt, GPR32Opnd:$rs, ZERO), 0>, ISA_MICROMIPS32R6; def : MipsInstAlias<"not $rt", (NOR_MMR6 GPR32Opnd:$rt, GPR32Opnd:$rt, ZERO), 0>, ISA_MICROMIPS32R6; def : MipsInstAlias<"lapc $rd, $imm", (ADDIUPC_MMR6 GPR32Opnd:$rd, simm19_lsl2:$imm)>, ISA_MICROMIPS32R6; def : MipsInstAlias<"neg $rt, $rs", (SUB_MMR6 GPR32Opnd:$rt, ZERO, GPR32Opnd:$rs), 1>, ISA_MICROMIPS32R6; def : MipsInstAlias<"neg $rt", (SUB_MMR6 GPR32Opnd:$rt, ZERO, GPR32Opnd:$rt), 1>, ISA_MICROMIPS32R6; def : MipsInstAlias<"negu $rt, $rs", (SUBU_MMR6 GPR32Opnd:$rt, ZERO, GPR32Opnd:$rs), 1>, ISA_MICROMIPS32R6; def : MipsInstAlias<"negu $rt", (SUBU_MMR6 GPR32Opnd:$rt, ZERO, GPR32Opnd:$rt), 1>, ISA_MICROMIPS32R6; def : MipsInstAlias<"beqz16 $rs, $offset", (BEQZC16_MMR6 GPRMM16Opnd:$rs, brtarget7_mm:$offset), 0>, ISA_MICROMIPS32R6; def : MipsInstAlias<"bnez16 $rs, $offset", (BNEZC16_MMR6 GPRMM16Opnd:$rs, brtarget7_mm:$offset), 0>, ISA_MICROMIPS32R6; def : MipsInstAlias<"b16 $offset", (BC16_MMR6 brtarget10_mm:$offset), 0>, ISA_MICROMIPS32R6; //===----------------------------------------------------------------------===// // // MicroMips arbitrary patterns that map to one or more instructions // //===----------------------------------------------------------------------===// def : MipsPat<(store GPRMM16:$src, addrimm4lsl2:$addr), (SW16_MMR6 GPRMM16:$src, addrimm4lsl2:$addr)>, ISA_MICROMIPS32R6; def : MipsPat<(subc GPR32:$lhs, GPR32:$rhs), (SUBU_MMR6 GPR32:$lhs, GPR32:$rhs)>, ISA_MICROMIPS32R6; def : MipsPat<(select i32:$cond, i32:$t, i32:$f), (OR_MM (SELNEZ_MMR6 i32:$t, i32:$cond), (SELEQZ_MMR6 i32:$f, i32:$cond))>, ISA_MICROMIPS32R6; def : MipsPat<(select i32:$cond, i32:$t, immz), (SELNEZ_MMR6 i32:$t, i32:$cond)>, ISA_MICROMIPS32R6; def : MipsPat<(select i32:$cond, immz, i32:$f), (SELEQZ_MMR6 i32:$f, i32:$cond)>, ISA_MICROMIPS32R6; defm : SelectInt_Pats, ISA_MICROMIPS32R6; defm S_MMR6 : Cmp_Pats, ISA_MICROMIPS32R6; defm D_MMR6 : Cmp_Pats, ISA_MICROMIPS32R6; def : MipsPat<(f32 fpimm0), (MTC1_MMR6 ZERO)>, ISA_MICROMIPS32R6; def : MipsPat<(f32 fpimm0neg), (FNEG_S_MMR6 (MTC1_MMR6 ZERO))>, ISA_MICROMIPS32R6; def : MipsPat<(MipsTruncIntFP FGR64Opnd:$src), (TRUNC_W_D_MMR6 FGR64Opnd:$src)>, ISA_MICROMIPS32R6; def : MipsPat<(and GPRMM16:$src, immZExtAndi16:$imm), (ANDI16_MMR6 GPRMM16:$src, immZExtAndi16:$imm)>, ISA_MICROMIPS32R6; def : MipsPat<(and GPR32:$src, immZExt16:$imm), (ANDI_MMR6 GPR32:$src, immZExt16:$imm)>, ISA_MICROMIPS32R6; def : MipsPat<(i32 immZExt16:$imm), (XORI_MMR6 ZERO, immZExt16:$imm)>, ISA_MICROMIPS32R6; def : MipsPat<(not GPRMM16:$in), (NOT16_MMR6 GPRMM16:$in)>, ISA_MICROMIPS32R6; def : MipsPat<(not GPR32:$in), (NOR_MMR6 GPR32Opnd:$in, ZERO)>, ISA_MICROMIPS32R6; // Patterns for load with a reg+imm operand. let AddedComplexity = 41 in { def : LoadRegImmPat, FGR_64, ISA_MICROMIPS32R6; def : StoreRegImmPat, FGR_64, ISA_MICROMIPS32R6; } def TAILCALL_MMR6 : TailCall, ISA_MICROMIPS32R6; def TAILCALLREG_MMR6 : TailCallReg, ISA_MICROMIPS32R6; def PseudoIndirectBranch_MMR6 : PseudoIndirectBranchBase, ISA_MICROMIPS32R6; def : MipsPat<(MipsTailCall (iPTR tglobaladdr:$dst)), (TAILCALL_MMR6 tglobaladdr:$dst)>, ISA_MICROMIPS32R6; def : MipsPat<(MipsTailCall (iPTR texternalsym:$dst)), (TAILCALL_MMR6 texternalsym:$dst)>, ISA_MICROMIPS32R6; def : MipsPat<(brcond (i32 (setne GPR32:$lhs, 0)), bb:$dst), (BNEZC_MMR6 GPR32:$lhs, bb:$dst)>, ISA_MICROMIPS32R6; def : MipsPat<(brcond (i32 (seteq GPR32:$lhs, 0)), bb:$dst), (BEQZC_MMR6 GPR32:$lhs, bb:$dst)>, ISA_MICROMIPS32R6; def : MipsPat<(brcond (i32 (setge GPR32:$lhs, GPR32:$rhs)), bb:$dst), (BEQZC_MMR6 (SLT_MM GPR32:$lhs, GPR32:$rhs), bb:$dst)>, ISA_MICROMIPS32R6; def : MipsPat<(brcond (i32 (setuge GPR32:$lhs, GPR32:$rhs)), bb:$dst), (BEQZC_MMR6 (SLTu_MM GPR32:$lhs, GPR32:$rhs), bb:$dst)>, ISA_MICROMIPS32R6; def : MipsPat<(brcond (i32 (setge GPR32:$lhs, immSExt16:$rhs)), bb:$dst), (BEQZC_MMR6 (SLTi_MM GPR32:$lhs, immSExt16:$rhs), bb:$dst)>, ISA_MICROMIPS32R6; def : MipsPat<(brcond (i32 (setuge GPR32:$lhs, immSExt16:$rhs)), bb:$dst), (BEQZC_MMR6 (SLTiu_MM GPR32:$lhs, immSExt16:$rhs), bb:$dst)>, ISA_MICROMIPS32R6; def : MipsPat<(brcond (i32 (setgt GPR32:$lhs, immSExt16Plus1:$rhs)), bb:$dst), (BEQZC_MMR6 (SLTi_MM GPR32:$lhs, (Plus1 imm:$rhs)), bb:$dst)>, ISA_MICROMIPS32R6; def : MipsPat<(brcond (i32 (setugt GPR32:$lhs, immSExt16Plus1:$rhs)), bb:$dst), (BEQZC_MMR6 (SLTiu_MM GPR32:$lhs, (Plus1 imm:$rhs)), bb:$dst)>, ISA_MICROMIPS32R6; def : MipsPat<(brcond (i32 (setle GPR32:$lhs, GPR32:$rhs)), bb:$dst), (BEQZC_MMR6 (SLT_MM GPR32:$rhs, GPR32:$lhs), bb:$dst)>, ISA_MICROMIPS32R6; def : MipsPat<(brcond (i32 (setule GPR32:$lhs, GPR32:$rhs)), bb:$dst), (BEQZC_MMR6 (SLTu_MM GPR32:$rhs, GPR32:$lhs), bb:$dst)>, ISA_MICROMIPS32R6; def : MipsPat<(brcond GPR32:$cond, bb:$dst), (BNEZC_MMR6 GPR32:$cond, bb:$dst)>, ISA_MICROMIPS32R6; Index: vendor/llvm/dist-release_80/lib/Target/Mips/MicroMipsInstrInfo.td =================================================================== --- vendor/llvm/dist-release_80/lib/Target/Mips/MicroMipsInstrInfo.td (revision 343793) +++ vendor/llvm/dist-release_80/lib/Target/Mips/MicroMipsInstrInfo.td (revision 343794) @@ -1,1455 +1,1456 @@ //===--- MicroMipsInstrFormats.td - microMIPS Inst Defs -*- tablegen -*----===// // // The LLVM Compiler Infrastructure // // This file is distributed under the University of Illinois Open Source // License. See LICENSE.TXT for details. // //===----------------------------------------------------------------------===// // // This files describes the defintions of the microMIPSr3 instructions. // //===----------------------------------------------------------------------===// def addrimm11 : ComplexPattern; def addrimm12 : ComplexPattern; def addrimm16 : ComplexPattern; def addrimm4lsl2 : ComplexPattern; def simm9_addiusp : Operand { let EncoderMethod = "getSImm9AddiuspValue"; let DecoderMethod = "DecodeSimm9SP"; } def uimm3_shift : Operand { let EncoderMethod = "getUImm3Mod8Encoding"; let DecoderMethod = "DecodePOOL16BEncodedField"; } def simm3_lsa2 : Operand { let EncoderMethod = "getSImm3Lsa2Value"; let DecoderMethod = "DecodeAddiur2Simm7"; } def uimm4_andi : Operand { let EncoderMethod = "getUImm4AndValue"; let DecoderMethod = "DecodeANDI16Imm"; } def immSExtAddiur2 : ImmLeaf 0);}]>; def immSExtAddius5 : ImmLeaf= -8 && Imm <= 7;}]>; def immZExtAndi16 : ImmLeaf= 1 && Imm <= 4) || Imm == 7 || Imm == 8 || Imm == 15 || Imm == 16 || Imm == 31 || Imm == 32 || Imm == 63 || Imm == 64 || Imm == 255 || Imm == 32768 || Imm == 65535 );}]>; def immZExt2Shift : ImmLeaf= 1 && Imm <= 8;}]>; def immLi16 : ImmLeaf= -1 && Imm <= 126;}]>; def MicroMipsMemGPRMM16AsmOperand : AsmOperandClass { let Name = "MicroMipsMem"; let RenderMethod = "addMicroMipsMemOperands"; let ParserMethod = "parseMemOperand"; let PredicateMethod = "isMemWithGRPMM16Base"; } // Define the classes of pointers used by microMIPS. // The numbers must match those in MipsRegisterInfo::MipsPtrClass. def ptr_gpr16mm_rc : PointerLikeRegClass<1>; def ptr_sp_rc : PointerLikeRegClass<2>; def ptr_gp_rc : PointerLikeRegClass<3>; class mem_mm_4_generic : Operand { let PrintMethod = "printMemOperand"; let MIOperandInfo = (ops ptr_gpr16mm_rc, simm4); let OperandType = "OPERAND_MEMORY"; let ParserMatchClass = MicroMipsMemGPRMM16AsmOperand; } def mem_mm_4 : mem_mm_4_generic { let EncoderMethod = "getMemEncodingMMImm4"; } def mem_mm_4_lsl1 : mem_mm_4_generic { let EncoderMethod = "getMemEncodingMMImm4Lsl1"; } def mem_mm_4_lsl2 : mem_mm_4_generic { let EncoderMethod = "getMemEncodingMMImm4Lsl2"; } def MicroMipsMemSPAsmOperand : AsmOperandClass { let Name = "MicroMipsMemSP"; let RenderMethod = "addMemOperands"; let ParserMethod = "parseMemOperand"; let PredicateMethod = "isMemWithUimmWordAlignedOffsetSP<7>"; } def MicroMipsMemGPAsmOperand : AsmOperandClass { let Name = "MicroMipsMemGP"; let RenderMethod = "addMemOperands"; let ParserMethod = "parseMemOperand"; let PredicateMethod = "isMemWithSimmWordAlignedOffsetGP<9>"; } def mem_mm_sp_imm5_lsl2 : Operand { let PrintMethod = "printMemOperand"; let MIOperandInfo = (ops ptr_sp_rc:$base, simm5:$offset); let OperandType = "OPERAND_MEMORY"; let ParserMatchClass = MicroMipsMemSPAsmOperand; let EncoderMethod = "getMemEncodingMMSPImm5Lsl2"; } def mem_mm_gp_simm7_lsl2 : Operand { let PrintMethod = "printMemOperand"; let MIOperandInfo = (ops ptr_gp_rc:$base, simm7_lsl2:$offset); let OperandType = "OPERAND_MEMORY"; let ParserMatchClass = MicroMipsMemGPAsmOperand; let EncoderMethod = "getMemEncodingMMGPImm7Lsl2"; } def mem_mm_9 : Operand { let PrintMethod = "printMemOperand"; let MIOperandInfo = (ops ptr_rc, simm9); let EncoderMethod = "getMemEncodingMMImm9"; let ParserMatchClass = MipsMemSimm9AsmOperand; let OperandType = "OPERAND_MEMORY"; } def mem_mm_11 : Operand { let PrintMethod = "printMemOperand"; let MIOperandInfo = (ops GPR32, simm11); let EncoderMethod = "getMemEncodingMMImm11"; let ParserMatchClass = MipsMemSimm11AsmOperand; let OperandType = "OPERAND_MEMORY"; } def mem_mm_12 : Operand { let PrintMethod = "printMemOperand"; let MIOperandInfo = (ops ptr_rc, simm12); let EncoderMethod = "getMemEncodingMMImm12"; let ParserMatchClass = MipsMemAsmOperand; let OperandType = "OPERAND_MEMORY"; } def mem_mm_16 : Operand { let PrintMethod = "printMemOperand"; let MIOperandInfo = (ops ptr_rc, simm16); let EncoderMethod = "getMemEncodingMMImm16"; let DecoderMethod = "DecodeMemMMImm16"; let ParserMatchClass = MipsMemSimm16AsmOperand; let OperandType = "OPERAND_MEMORY"; } def MipsMemUimm4AsmOperand : AsmOperandClass { let Name = "MemOffsetUimm4"; let SuperClasses = [MipsMemAsmOperand]; let RenderMethod = "addMemOperands"; let ParserMethod = "parseMemOperand"; let PredicateMethod = "isMemWithUimmOffsetSP<6>"; } def mem_mm_4sp : Operand { let PrintMethod = "printMemOperand"; let MIOperandInfo = (ops ptr_sp_rc, uimm8); let EncoderMethod = "getMemEncodingMMImm4sp"; let ParserMatchClass = MipsMemUimm4AsmOperand; let OperandType = "OPERAND_MEMORY"; } def jmptarget_mm : Operand { let EncoderMethod = "getJumpTargetOpValueMM"; } def calltarget_mm : Operand { let EncoderMethod = "getJumpTargetOpValueMM"; } def brtarget7_mm : Operand { let EncoderMethod = "getBranchTarget7OpValueMM"; let OperandType = "OPERAND_PCREL"; let DecoderMethod = "DecodeBranchTarget7MM"; let ParserMatchClass = MipsJumpTargetAsmOperand; } def brtarget10_mm : Operand { let EncoderMethod = "getBranchTargetOpValueMMPC10"; let OperandType = "OPERAND_PCREL"; let DecoderMethod = "DecodeBranchTarget10MM"; let ParserMatchClass = MipsJumpTargetAsmOperand; } def brtarget_mm : Operand { let EncoderMethod = "getBranchTargetOpValueMM"; let OperandType = "OPERAND_PCREL"; let DecoderMethod = "DecodeBranchTargetMM"; let ParserMatchClass = MipsJumpTargetAsmOperand; } def simm23_lsl2 : Operand { let EncoderMethod = "getSimm23Lsl2Encoding"; let DecoderMethod = "DecodeSimm23Lsl2"; } class CompactBranchMM : InstSE<(outs), (ins RO:$rs, opnd:$offset), !strconcat(opstr, "\t$rs, $offset"), [], II_BCCZC, FrmI> { let isBranch = 1; let isTerminator = 1; let hasDelaySlot = 0; let Defs = [AT]; } let canFoldAsLoad = 1 in class LoadLeftRightMM : InstSE<(outs RO:$rt), (ins MemOpnd:$addr, RO:$src), !strconcat(opstr, "\t$rt, $addr"), [(set RO:$rt, (OpNode addrimm12:$addr, RO:$src))], Itin, FrmI> { let DecoderMethod = "DecodeMemMMImm12"; string Constraints = "$src = $rt"; let BaseOpcode = opstr; bit mayLoad = 1; bit mayStore = 0; } class StoreLeftRightMM: InstSE<(outs), (ins RO:$rt, MemOpnd:$addr), !strconcat(opstr, "\t$rt, $addr"), [(OpNode RO:$rt, addrimm12:$addr)], Itin, FrmI> { let DecoderMethod = "DecodeMemMMImm12"; let BaseOpcode = opstr; bit mayLoad = 0; bit mayStore = 1; } class MovePMM16 : MicroMipsInst16<(outs RO1:$rd1, RO2:$rd2), (ins RO3:$rs, RO3:$rt), !strconcat(opstr, "\t$rd1, $rd2, $rs, $rt"), [], NoItinerary, FrmR> { let isReMaterializable = 1; let isMoveReg = 1; let DecoderMethod = "DecodeMovePOperands"; } class StorePairMM : InstSE<(outs), (ins GPR32Opnd:$rt, GPR32Opnd:$rt2, mem_simm12:$addr), !strconcat(opstr, "\t$rt, $addr"), [], II_SWP, FrmI, opstr> { let DecoderMethod = "DecodeMemMMImm12"; let mayStore = 1; let AsmMatchConverter = "ConvertXWPOperands"; } class LoadPairMM : InstSE<(outs GPR32Opnd:$rt, GPR32Opnd:$rt2), (ins mem_simm12:$addr), !strconcat(opstr, "\t$rt, $addr"), [], II_LWP, FrmI, opstr> { let DecoderMethod = "DecodeMemMMImm12"; let mayLoad = 1; let AsmMatchConverter = "ConvertXWPOperands"; } class LLBaseMM : InstSE<(outs RO:$rt), (ins mem_mm_12:$addr), !strconcat(opstr, "\t$rt, $addr"), [], II_LL, FrmI> { let DecoderMethod = "DecodeMemMMImm12"; let mayLoad = 1; } class LLEBaseMM : InstSE<(outs RO:$rt), (ins mem_simm9:$addr), !strconcat(opstr, "\t$rt, $addr"), [], II_LLE, FrmI> { let DecoderMethod = "DecodeMemMMImm9"; string BaseOpcode = opstr; let mayLoad = 1; } class SCBaseMM : InstSE<(outs RO:$dst), (ins RO:$rt, mem_mm_12:$addr), !strconcat(opstr, "\t$rt, $addr"), [], II_SC, FrmI> { let DecoderMethod = "DecodeMemMMImm12"; let mayStore = 1; let Constraints = "$rt = $dst"; } class SCEBaseMM : InstSE<(outs RO:$dst), (ins RO:$rt, mem_simm9:$addr), !strconcat(opstr, "\t$rt, $addr"), [], II_SCE, FrmI> { let DecoderMethod = "DecodeMemMMImm9"; string BaseOpcode = opstr; let mayStore = 1; let Constraints = "$rt = $dst"; } class LoadMM : InstSE<(outs RO:$rt), (ins MO:$addr), !strconcat(opstr, "\t$rt, $addr"), [(set RO:$rt, (OpNode addrimm12:$addr))], Itin, FrmI, opstr> { let DecoderMethod = "DecodeMemMMImm12"; let canFoldAsLoad = 1; let mayLoad = 1; } class ArithRMM16 : MicroMipsInst16<(outs RO:$rd), (ins RO:$rs, RO:$rt), !strconcat(opstr, "\t$rd, $rs, $rt"), [(set RO:$rd, (OpNode RO:$rs, RO:$rt))], Itin, FrmR> { let isCommutable = isComm; } class AndImmMM16 : MicroMipsInst16<(outs RO:$rd), (ins RO:$rs, uimm4_andi:$imm), !strconcat(opstr, "\t$rd, $rs, $imm"), [], Itin, FrmI>; class LogicRMM16 : MicroMipsInst16<(outs RO:$dst), (ins RO:$rs, RO:$rt), !strconcat(opstr, "\t$rt, $rs"), [(set RO:$dst, (OpNode RO:$rs, RO:$rt))], Itin, FrmR> { let isCommutable = 1; let Constraints = "$rt = $dst"; } class NotMM16 : MicroMipsInst16<(outs RO:$rt), (ins RO:$rs), !strconcat(opstr, "\t$rt, $rs"), [(set RO:$rt, (not RO:$rs))], II_NOT, FrmR>; class ShiftIMM16 : MicroMipsInst16<(outs RO:$rd), (ins RO:$rt, ImmOpnd:$shamt), !strconcat(opstr, "\t$rd, $rt, $shamt"), [], Itin, FrmR>; class LoadMM16 : MicroMipsInst16<(outs RO:$rt), (ins MemOpnd:$addr), !strconcat(opstr, "\t$rt, $addr"), [], Itin, FrmI> { let DecoderMethod = "DecodeMemMMImm4"; let canFoldAsLoad = 1; let mayLoad = 1; } class StoreMM16 : MicroMipsInst16<(outs), (ins RTOpnd:$rt, MemOpnd:$addr), !strconcat(opstr, "\t$rt, $addr"), [], Itin, FrmI> { let DecoderMethod = "DecodeMemMMImm4"; let mayStore = 1; } class LoadSPMM16 : MicroMipsInst16<(outs RO:$rt), (ins MemOpnd:$offset), !strconcat(opstr, "\t$rt, $offset"), [], Itin, FrmI> { let DecoderMethod = "DecodeMemMMSPImm5Lsl2"; let canFoldAsLoad = 1; let mayLoad = 1; } class StoreSPMM16 : MicroMipsInst16<(outs), (ins RO:$rt, MemOpnd:$offset), !strconcat(opstr, "\t$rt, $offset"), [], Itin, FrmI> { let DecoderMethod = "DecodeMemMMSPImm5Lsl2"; let mayStore = 1; } class LoadGPMM16 : MicroMipsInst16<(outs RO:$rt), (ins MemOpnd:$offset), !strconcat(opstr, "\t$rt, $offset"), [], Itin, FrmI> { let DecoderMethod = "DecodeMemMMGPImm7Lsl2"; let canFoldAsLoad = 1; let mayLoad = 1; } class AddImmUR2 : MicroMipsInst16<(outs RO:$rd), (ins RO:$rs, simm3_lsa2:$imm), !strconcat(opstr, "\t$rd, $rs, $imm"), [], II_ADDIU, FrmR> { let isCommutable = 1; } class AddImmUS5 : MicroMipsInst16<(outs RO:$dst), (ins RO:$rd, simm4:$imm), !strconcat(opstr, "\t$rd, $imm"), [], II_ADDIU, FrmR> { let Constraints = "$rd = $dst"; } class AddImmUR1SP : MicroMipsInst16<(outs RO:$rd), (ins uimm6_lsl2:$imm), !strconcat(opstr, "\t$rd, $imm"), [], II_ADDIU, FrmR>; class AddImmUSP : MicroMipsInst16<(outs), (ins simm9_addiusp:$imm), !strconcat(opstr, "\t$imm"), [], II_ADDIU, FrmI>; class MoveFromHILOMM : MicroMipsInst16<(outs RO:$rd), (ins), !strconcat(opstr, "\t$rd"), [], II_MFHI_MFLO, FrmR> { let Uses = [UseReg]; let hasSideEffects = 0; let isMoveReg = 1; } class MoveMM16 : MicroMipsInst16<(outs RO:$rd), (ins RO:$rs), !strconcat(opstr, "\t$rd, $rs"), [], II_MOVE, FrmR> { let isReMaterializable = 1; let isMoveReg = 1; } class LoadImmMM16 : MicroMipsInst16<(outs RO:$rd), (ins Od:$imm), !strconcat(opstr, "\t$rd, $imm"), [], II_LI, FrmI> { let isReMaterializable = 1; } // 16-bit Jump and Link (Call) class JumpLinkRegMM16 : MicroMipsInst16<(outs), (ins RO:$rs), !strconcat(opstr, "\t$rs"), [(MipsJmpLink RO:$rs)], II_JALR, FrmR> { let isCall = 1; let hasDelaySlot = 1; let Defs = [RA]; + let hasPostISelHook = 1; } // 16-bit Jump Reg class JumpRegMM16 : MicroMipsInst16<(outs), (ins RO:$rs), !strconcat(opstr, "\t$rs"), [], II_JR, FrmR> { let hasDelaySlot = 1; let isBranch = 1; let isIndirectBranch = 1; } // Base class for JRADDIUSP instruction. class JumpRAddiuStackMM16 : MicroMipsInst16<(outs), (ins uimm5_lsl2:$imm), "jraddiusp\t$imm", [], II_JRADDIUSP, FrmR> { let isTerminator = 1; let isBarrier = 1; let isBranch = 1; let isIndirectBranch = 1; } // 16-bit Jump and Link (Call) - Short Delay Slot class JumpLinkRegSMM16 : MicroMipsInst16<(outs), (ins RO:$rs), !strconcat(opstr, "\t$rs"), [], II_JALRS, FrmR> { let isCall = 1; let hasDelaySlot = 1; let Defs = [RA]; } // 16-bit Jump Register Compact - No delay slot class JumpRegCMM16 : MicroMipsInst16<(outs), (ins RO:$rs), !strconcat(opstr, "\t$rs"), [], II_JRC, FrmR> { let isTerminator = 1; let isBarrier = 1; let isBranch = 1; let isIndirectBranch = 1; } // Break16 and Sdbbp16 class BrkSdbbp16MM : MicroMipsInst16<(outs), (ins uimm4:$code_), !strconcat(opstr, "\t$code_"), [], Itin, FrmOther>; class CBranchZeroMM : MicroMipsInst16<(outs), (ins RO:$rs, opnd:$offset), !strconcat(opstr, "\t$rs, $offset"), [], II_BCCZ, FrmI> { let isBranch = 1; let isTerminator = 1; let hasDelaySlot = 1; let Defs = [AT]; } // MicroMIPS Jump and Link (Call) - Short Delay Slot let isCall = 1, hasDelaySlot = 1, Defs = [RA] in { class JumpLinkMM : InstSE<(outs), (ins opnd:$target), !strconcat(opstr, "\t$target"), [], II_JALS, FrmJ, opstr> { let DecoderMethod = "DecodeJumpTargetMM"; } class JumpLinkRegMM: InstSE<(outs RO:$rd), (ins RO:$rs), !strconcat(opstr, "\t$rd, $rs"), [], II_JALRS, FrmR>; class BranchCompareToZeroLinkMM : InstSE<(outs), (ins RO:$rs, opnd:$offset), !strconcat(opstr, "\t$rs, $offset"), [], II_BCCZALS, FrmI, opstr>; } class LoadWordIndexedScaledMM : InstSE<(outs RO:$rd), (ins PtrRC:$base, PtrRC:$index), !strconcat(opstr, "\t$rd, ${index}(${base})"), [], II_LWXS, FrmFI>; class PrefetchIndexed : InstSE<(outs), (ins PtrRC:$base, PtrRC:$index, uimm5:$hint), !strconcat(opstr, "\t$hint, ${index}(${base})"), [], II_PREF, FrmOther>; class AddImmUPC : InstSE<(outs RO:$rs), (ins simm23_lsl2:$imm), !strconcat(opstr, "\t$rs, $imm"), [], II_ADDIU, FrmR>; /// A list of registers used by load/store multiple instructions. def RegListAsmOperand : AsmOperandClass { let Name = "RegList"; let ParserMethod = "parseRegisterList"; } def reglist : Operand { let EncoderMethod = "getRegisterListOpValue"; let ParserMatchClass = RegListAsmOperand; let PrintMethod = "printRegisterList"; let DecoderMethod = "DecodeRegListOperand"; } def RegList16AsmOperand : AsmOperandClass { let Name = "RegList16"; let ParserMethod = "parseRegisterList"; let PredicateMethod = "isRegList16"; let RenderMethod = "addRegListOperands"; } def reglist16 : Operand { let EncoderMethod = "getRegisterListOpValue16"; let DecoderMethod = "DecodeRegListOperand16"; let PrintMethod = "printRegisterList"; let ParserMatchClass = RegList16AsmOperand; } class StoreMultMM : InstSE<(outs), (ins reglist:$rt, mem_mm_12:$addr), !strconcat(opstr, "\t$rt, $addr"), [], Itin, FrmI, opstr> { let DecoderMethod = "DecodeMemMMImm12"; let mayStore = 1; } class LoadMultMM : InstSE<(outs reglist:$rt), (ins mem_mm_12:$addr), !strconcat(opstr, "\t$rt, $addr"), [], Itin, FrmI, opstr> { let DecoderMethod = "DecodeMemMMImm12"; let mayLoad = 1; } class StoreMultMM16 : MicroMipsInst16<(outs), (ins reglist16:$rt, mem_mm_4sp:$addr), !strconcat(opstr, "\t$rt, $addr"), [], Itin, FrmI> { let DecoderMethod = "DecodeMemMMReglistImm4Lsl2"; let mayStore = 1; } class LoadMultMM16 : MicroMipsInst16<(outs reglist16:$rt), (ins mem_mm_4sp:$addr), !strconcat(opstr, "\t$rt, $addr"), [], Itin, FrmI> { let DecoderMethod = "DecodeMemMMReglistImm4Lsl2"; let mayLoad = 1; } class UncondBranchMM16 : MicroMipsInst16<(outs), (ins brtarget10_mm:$offset), !strconcat(opstr, "\t$offset"), [], II_B, FrmI> { let isBranch = 1; let isTerminator = 1; let isBarrier = 1; let hasDelaySlot = 1; let Predicates = [RelocPIC, InMicroMips]; let Defs = [AT]; } class HypcallMM : InstSE<(outs), (ins uimm10:$code_), !strconcat(opstr, "\t$code_"), [], II_HYPCALL, FrmOther> { let BaseOpcode = opstr; } class TLBINVMM : InstSE<(outs), (ins), opstr, [], Itin, FrmOther> { let BaseOpcode = opstr; } class MfCop0MM : InstSE<(outs DstRC:$rt), (ins SrcRC:$rs, uimm3:$sel), !strconcat(opstr, "\t$rt, $rs, $sel"), [], Itin, FrmR> { let BaseOpcode = opstr; } class MtCop0MM : InstSE<(outs DstRC:$rs), (ins SrcRC:$rt, uimm3:$sel), !strconcat(opstr, "\t$rt, $rs, $sel"), [], Itin, FrmR> { let BaseOpcode = opstr; } let FastISelShouldIgnore = 1 in { def ADDU16_MM : ArithRMM16<"addu16", GPRMM16Opnd, 1, II_ADDU, add>, ARITH_FM_MM16<0>, ISA_MICROMIPS32_NOT_MIPS32R6; def AND16_MM : LogicRMM16<"and16", GPRMM16Opnd, II_AND, and>, LOGIC_FM_MM16<0x2>, ISA_MICROMIPS32_NOT_MIPS32R6; } def ANDI16_MM : AndImmMM16<"andi16", GPRMM16Opnd, II_AND>, ANDI_FM_MM16<0x0b>, ISA_MICROMIPS32_NOT_MIPS32R6; def NOT16_MM : NotMM16<"not16", GPRMM16Opnd>, LOGIC_FM_MM16<0x0>, ISA_MICROMIPS32_NOT_MIPS32R6; let FastISelShouldIgnore = 1 in def OR16_MM : LogicRMM16<"or16", GPRMM16Opnd, II_OR, or>, LOGIC_FM_MM16<0x3>, ISA_MICROMIPS32_NOT_MIPS32R6; def SLL16_MM : ShiftIMM16<"sll16", uimm3_shift, GPRMM16Opnd, II_SLL>, SHIFT_FM_MM16<0>, ISA_MICROMIPS32_NOT_MIPS32R6; def SRL16_MM : ShiftIMM16<"srl16", uimm3_shift, GPRMM16Opnd, II_SRL>, SHIFT_FM_MM16<1>, ISA_MICROMIPS32_NOT_MIPS32R6; let FastISelShouldIgnore = 1 in { def SUBU16_MM : ArithRMM16<"subu16", GPRMM16Opnd, 0, II_SUBU, sub>, ARITH_FM_MM16<1>, ISA_MICROMIPS32_NOT_MIPS32R6; def XOR16_MM : LogicRMM16<"xor16", GPRMM16Opnd, II_XOR, xor>, LOGIC_FM_MM16<0x1>, ISA_MICROMIPS32_NOT_MIPS32R6; } def LBU16_MM : LoadMM16<"lbu16", GPRMM16Opnd, zextloadi8, II_LBU, mem_mm_4>, LOAD_STORE_FM_MM16<0x02>, ISA_MICROMIPS; def LHU16_MM : LoadMM16<"lhu16", GPRMM16Opnd, zextloadi16, II_LHU, mem_mm_4_lsl1>, LOAD_STORE_FM_MM16<0x0a>, ISA_MICROMIPS; def LW16_MM : LoadMM16<"lw16", GPRMM16Opnd, load, II_LW, mem_mm_4_lsl2>, LOAD_STORE_FM_MM16<0x1a>, ISA_MICROMIPS; def SB16_MM : StoreMM16<"sb16", GPRMM16OpndZero, GPRMM16Opnd, truncstorei8, II_SB, mem_mm_4>, LOAD_STORE_FM_MM16<0x22>, ISA_MICROMIPS32_NOT_MIPS32R6; def SH16_MM : StoreMM16<"sh16", GPRMM16OpndZero, GPRMM16Opnd, truncstorei16, II_SH, mem_mm_4_lsl1>, LOAD_STORE_FM_MM16<0x2a>, ISA_MICROMIPS32_NOT_MIPS32R6; def SW16_MM : StoreMM16<"sw16", GPRMM16OpndZero, GPRMM16Opnd, store, II_SW, mem_mm_4_lsl2>, LOAD_STORE_FM_MM16<0x3a>, ISA_MICROMIPS32_NOT_MIPS32R6; def LWGP_MM : LoadGPMM16<"lw", GPRMM16Opnd, II_LW, mem_mm_gp_simm7_lsl2>, LOAD_GP_FM_MM16<0x19>, ISA_MICROMIPS; def LWSP_MM : LoadSPMM16<"lw", GPR32Opnd, II_LW, mem_mm_sp_imm5_lsl2>, LOAD_STORE_SP_FM_MM16<0x12>, ISA_MICROMIPS; def SWSP_MM : StoreSPMM16<"sw", GPR32Opnd, II_SW, mem_mm_sp_imm5_lsl2>, LOAD_STORE_SP_FM_MM16<0x32>, ISA_MICROMIPS32_NOT_MIPS32R6; def ADDIUR1SP_MM : AddImmUR1SP<"addiur1sp", GPRMM16Opnd>, ADDIUR1SP_FM_MM16, ISA_MICROMIPS; def ADDIUR2_MM : AddImmUR2<"addiur2", GPRMM16Opnd>, ADDIUR2_FM_MM16, ISA_MICROMIPS; def ADDIUS5_MM : AddImmUS5<"addius5", GPR32Opnd>, ADDIUS5_FM_MM16, ISA_MICROMIPS; def ADDIUSP_MM : AddImmUSP<"addiusp">, ADDIUSP_FM_MM16, ISA_MICROMIPS; def MFHI16_MM : MoveFromHILOMM<"mfhi16", GPR32Opnd, AC0>, MFHILO_FM_MM16<0x10>, ISA_MICROMIPS32_NOT_MIPS32R6; def MFLO16_MM : MoveFromHILOMM<"mflo16", GPR32Opnd, AC0>, MFHILO_FM_MM16<0x12>, ISA_MICROMIPS32_NOT_MIPS32R6; def MOVE16_MM : MoveMM16<"move", GPR32Opnd>, MOVE_FM_MM16<0x03>, ISA_MICROMIPS32_NOT_MIPS32R6; def MOVEP_MM : MovePMM16<"movep", GPRMM16OpndMovePPairFirst, GPRMM16OpndMovePPairSecond, GPRMM16OpndMoveP>, MOVEP_FM_MM16, ISA_MICROMIPS32_NOT_MIPS32R6; def LI16_MM : LoadImmMM16<"li16", li16_imm, GPRMM16Opnd>, LI_FM_MM16, IsAsCheapAsAMove, ISA_MICROMIPS32_NOT_MIPS32R6; def JALR16_MM : JumpLinkRegMM16<"jalr", GPR32Opnd>, JALR_FM_MM16<0x0e>, ISA_MICROMIPS32_NOT_MIPS32R6; def JALRS16_MM : JumpLinkRegSMM16<"jalrs16", GPR32Opnd>, JALR_FM_MM16<0x0f>, ISA_MICROMIPS32_NOT_MIPS32R6; def JRC16_MM : JumpRegCMM16<"jrc", GPR32Opnd>, JALR_FM_MM16<0x0d>, ISA_MICROMIPS32_NOT_MIPS32R6; def JRADDIUSP : JumpRAddiuStackMM16, JRADDIUSP_FM_MM16<0x18>, ISA_MICROMIPS32_NOT_MIPS32R6; def JR16_MM : JumpRegMM16<"jr16", GPR32Opnd>, JALR_FM_MM16<0x0c>, ISA_MICROMIPS32_NOT_MIPS32R6; def BEQZ16_MM : CBranchZeroMM<"beqz16", brtarget7_mm, GPRMM16Opnd>, BEQNEZ_FM_MM16<0x23>, ISA_MICROMIPS32_NOT_MIPS32R6; def BNEZ16_MM : CBranchZeroMM<"bnez16", brtarget7_mm, GPRMM16Opnd>, BEQNEZ_FM_MM16<0x2b>, ISA_MICROMIPS32_NOT_MIPS32R6; def B16_MM : UncondBranchMM16<"b16">, B16_FM, ISA_MICROMIPS32_NOT_MIPS32R6; def BREAK16_MM : BrkSdbbp16MM<"break16", II_BREAK>, BRKSDBBP16_FM_MM<0x28>, ISA_MICROMIPS32_NOT_MIPS32R6; def SDBBP16_MM : BrkSdbbp16MM<"sdbbp16", II_SDBBP>, BRKSDBBP16_FM_MM<0x2C>, ISA_MICROMIPS32_NOT_MIPS32R6; let DecoderNamespace = "MicroMips" in { /// Load and Store Instructions - multiple def SWM16_MM : StoreMultMM16<"swm16", II_SWM>, LWM_FM_MM16<0x5>, ISA_MICROMIPS32_NOT_MIPS32R6; def LWM16_MM : LoadMultMM16<"lwm16", II_LWM>, LWM_FM_MM16<0x4>, ISA_MICROMIPS32_NOT_MIPS32R6; def CFC2_MM : InstSE<(outs GPR32Opnd:$rt), (ins COP2Opnd:$impl), "cfc2\t$rt, $impl", [], II_CFC2, FrmFR, "cfc2">, POOL32A_CFTC2_FM_MM<0b1100110100>, ISA_MICROMIPS; def CTC2_MM : InstSE<(outs COP2Opnd:$impl), (ins GPR32Opnd:$rt), "ctc2\t$rt, $impl", [], II_CTC2, FrmFR, "ctc2">, POOL32A_CFTC2_FM_MM<0b1101110100>, ISA_MICROMIPS; } class WaitMM : InstSE<(outs), (ins uimm10:$code_), !strconcat(opstr, "\t$code_"), [], II_WAIT, FrmOther, opstr>; let DecoderNamespace = "MicroMips" in { /// Compact Branch Instructions def BEQZC_MM : CompactBranchMM<"beqzc", brtarget_mm, seteq, GPR32Opnd>, COMPACT_BRANCH_FM_MM<0x7>, ISA_MICROMIPS32_NOT_MIPS32R6; def BNEZC_MM : CompactBranchMM<"bnezc", brtarget_mm, setne, GPR32Opnd>, COMPACT_BRANCH_FM_MM<0x5>, ISA_MICROMIPS32_NOT_MIPS32R6; /// Arithmetic Instructions (ALU Immediate) def ADDiu_MM : MMRel, ArithLogicI<"addiu", simm16, GPR32Opnd, II_ADDIU>, ADDI_FM_MM<0xc>, ISA_MICROMIPS32_NOT_MIPS32R6; def ADDi_MM : MMRel, ArithLogicI<"addi", simm16, GPR32Opnd, II_ADDI>, ADDI_FM_MM<0x4>, ISA_MICROMIPS32_NOT_MIPS32R6; def SLTi_MM : MMRel, SetCC_I<"slti", setlt, simm16, immSExt16, GPR32Opnd>, SLTI_FM_MM<0x24>, ISA_MICROMIPS; def SLTiu_MM : MMRel, SetCC_I<"sltiu", setult, simm16, immSExt16, GPR32Opnd>, SLTI_FM_MM<0x2c>, ISA_MICROMIPS; def ANDi_MM : MMRel, ArithLogicI<"andi", uimm16, GPR32Opnd, II_ANDI>, ADDI_FM_MM<0x34>, ISA_MICROMIPS32_NOT_MIPS32R6; def ORi_MM : MMRel, ArithLogicI<"ori", uimm16, GPR32Opnd, II_ORI, immZExt16, or>, ADDI_FM_MM<0x14>, ISA_MICROMIPS32_NOT_MIPS32R6; def XORi_MM : MMRel, ArithLogicI<"xori", uimm16, GPR32Opnd, II_XORI, immZExt16, xor>, ADDI_FM_MM<0x1c>, ISA_MICROMIPS32_NOT_MIPS32R6; def LUi_MM : MMRel, LoadUpper<"lui", GPR32Opnd, uimm16_relaxed>, LUI_FM_MM, ISA_MICROMIPS32_NOT_MIPS32R6; def LEA_ADDiu_MM : MMRel, EffectiveAddress<"addiu", GPR32Opnd>, LW_FM_MM<0xc>, ISA_MICROMIPS; /// Arithmetic Instructions (3-Operand, R-Type) def ADDu_MM : MMRel, ArithLogicR<"addu", GPR32Opnd, 1, II_ADDU, add>, ADD_FM_MM<0, 0x150>, ISA_MICROMIPS32_NOT_MIPS32R6; def SUBu_MM : MMRel, ArithLogicR<"subu", GPR32Opnd, 0, II_SUBU, sub>, ADD_FM_MM<0, 0x1d0>, ISA_MICROMIPS32_NOT_MIPS32R6; let Defs = [HI0, LO0] in def MUL_MM : MMRel, ArithLogicR<"mul", GPR32Opnd, 1, II_MUL, mul>, ADD_FM_MM<0, 0x210>, ISA_MICROMIPS32_NOT_MIPS32R6; def ADD_MM : MMRel, ArithLogicR<"add", GPR32Opnd, 1, II_ADD>, ADD_FM_MM<0, 0x110>, ISA_MICROMIPS32_NOT_MIPS32R6; def SUB_MM : MMRel, ArithLogicR<"sub", GPR32Opnd, 0, II_SUB>, ADD_FM_MM<0, 0x190>, ISA_MICROMIPS32_NOT_MIPS32R6; def SLT_MM : MMRel, SetCC_R<"slt", setlt, GPR32Opnd>, ADD_FM_MM<0, 0x350>, ISA_MICROMIPS; def SLTu_MM : MMRel, SetCC_R<"sltu", setult, GPR32Opnd>, ADD_FM_MM<0, 0x390>, ISA_MICROMIPS; def AND_MM : MMRel, ArithLogicR<"and", GPR32Opnd, 1, II_AND, and>, ADD_FM_MM<0, 0x250>, ISA_MICROMIPS32_NOT_MIPS32R6; def OR_MM : MMRel, ArithLogicR<"or", GPR32Opnd, 1, II_OR, or>, ADD_FM_MM<0, 0x290>, ISA_MICROMIPS32_NOT_MIPS32R6; def XOR_MM : MMRel, ArithLogicR<"xor", GPR32Opnd, 1, II_XOR, xor>, ADD_FM_MM<0, 0x310>, ISA_MICROMIPS32_NOT_MIPS32R6; def NOR_MM : MMRel, LogicNOR<"nor", GPR32Opnd>, ADD_FM_MM<0, 0x2d0>, ISA_MICROMIPS32_NOT_MIPS32R6; def MULT_MM : MMRel, Mult<"mult", II_MULT, GPR32Opnd, [HI0, LO0]>, MULT_FM_MM<0x22c>, ISA_MICROMIPS32_NOT_MIPS32R6; def MULTu_MM : MMRel, Mult<"multu", II_MULTU, GPR32Opnd, [HI0, LO0]>, MULT_FM_MM<0x26c>, ISA_MICROMIPS32_NOT_MIPS32R6; def SDIV_MM : MMRel, Div<"div", II_DIV, GPR32Opnd, [HI0, LO0]>, MULT_FM_MM<0x2ac>, ISA_MICROMIPS32_NOT_MIPS32R6; def UDIV_MM : MMRel, Div<"divu", II_DIVU, GPR32Opnd, [HI0, LO0]>, MULT_FM_MM<0x2ec>, ISA_MICROMIPS32_NOT_MIPS32R6; /// Arithmetic Instructions with PC and Immediate def ADDIUPC_MM : AddImmUPC<"addiupc", GPRMM16Opnd>, ADDIUPC_FM_MM, ISA_MICROMIPS32_NOT_MIPS32R6; /// Shift Instructions def SLL_MM : MMRel, shift_rotate_imm<"sll", uimm5, GPR32Opnd, II_SLL>, SRA_FM_MM<0, 0>, ISA_MICROMIPS; def SRL_MM : MMRel, shift_rotate_imm<"srl", uimm5, GPR32Opnd, II_SRL>, SRA_FM_MM<0x40, 0>, ISA_MICROMIPS; def SRA_MM : MMRel, shift_rotate_imm<"sra", uimm5, GPR32Opnd, II_SRA>, SRA_FM_MM<0x80, 0>, ISA_MICROMIPS; def SLLV_MM : MMRel, shift_rotate_reg<"sllv", GPR32Opnd, II_SLLV>, SRLV_FM_MM<0x10, 0>, ISA_MICROMIPS; def SRLV_MM : MMRel, shift_rotate_reg<"srlv", GPR32Opnd, II_SRLV>, SRLV_FM_MM<0x50, 0>, ISA_MICROMIPS; def SRAV_MM : MMRel, shift_rotate_reg<"srav", GPR32Opnd, II_SRAV>, SRLV_FM_MM<0x90, 0>, ISA_MICROMIPS; def ROTR_MM : MMRel, shift_rotate_imm<"rotr", uimm5, GPR32Opnd, II_ROTR>, SRA_FM_MM<0xc0, 0>, ISA_MICROMIPS { list Pattern = [(set GPR32Opnd:$rd, (rotr GPR32Opnd:$rt, immZExt5:$shamt))]; } def ROTRV_MM : MMRel, shift_rotate_reg<"rotrv", GPR32Opnd, II_ROTRV>, SRLV_FM_MM<0xd0, 0>, ISA_MICROMIPS { list Pattern = [(set GPR32Opnd:$rd, (rotr GPR32Opnd:$rt, GPR32Opnd:$rs))]; } /// Load and Store Instructions - aligned let DecoderMethod = "DecodeMemMMImm16" in { def LB_MM : LoadMemory<"lb", GPR32Opnd, mem_mm_16, sextloadi8, II_LB>, MMRel, LW_FM_MM<0x7>, ISA_MICROMIPS; def LBu_MM : LoadMemory<"lbu", GPR32Opnd, mem_mm_16, zextloadi8, II_LBU>, MMRel, LW_FM_MM<0x5>, ISA_MICROMIPS; def LH_MM : LoadMemory<"lh", GPR32Opnd, mem_simmptr, sextloadi16, II_LH, addrDefault>, MMRel, LW_FM_MM<0xf>, ISA_MICROMIPS; def LHu_MM : LoadMemory<"lhu", GPR32Opnd, mem_simmptr, zextloadi16, II_LHU>, MMRel, LW_FM_MM<0xd>, ISA_MICROMIPS; def LW_MM : Load<"lw", GPR32Opnd, null_frag, II_LW>, MMRel, LW_FM_MM<0x3f>, ISA_MICROMIPS; def SB_MM : Store<"sb", GPR32Opnd, truncstorei8, II_SB>, MMRel, LW_FM_MM<0x6>, ISA_MICROMIPS; def SH_MM : Store<"sh", GPR32Opnd, truncstorei16, II_SH>, MMRel, LW_FM_MM<0xe>, ISA_MICROMIPS; def SW_MM : Store<"sw", GPR32Opnd, null_frag, II_SW>, MMRel, LW_FM_MM<0x3e>, ISA_MICROMIPS; } } let DecoderNamespace = "MicroMips" in { let DecoderMethod = "DecodeMemMMImm9" in { def LBE_MM : MMRel, Load<"lbe", GPR32Opnd, null_frag, II_LBE>, POOL32C_LHUE_FM_MM<0x18, 0x6, 0x4>, ISA_MICROMIPS, ASE_EVA; def LBuE_MM : MMRel, Load<"lbue", GPR32Opnd, null_frag, II_LBUE>, POOL32C_LHUE_FM_MM<0x18, 0x6, 0x0>, ISA_MICROMIPS, ASE_EVA; def LHE_MM : MMRel, LoadMemory<"lhe", GPR32Opnd, mem_simm9, null_frag, II_LHE>, POOL32C_LHUE_FM_MM<0x18, 0x6, 0x5>, ISA_MICROMIPS, ASE_EVA; def LHuE_MM : MMRel, LoadMemory<"lhue", GPR32Opnd, mem_simm9, null_frag, II_LHUE>, POOL32C_LHUE_FM_MM<0x18, 0x6, 0x1>, ISA_MICROMIPS, ASE_EVA; def LWE_MM : MMRel, LoadMemory<"lwe", GPR32Opnd, mem_simm9, null_frag, II_LWE>, POOL32C_LHUE_FM_MM<0x18, 0x6, 0x7>, ISA_MICROMIPS, ASE_EVA; def SBE_MM : MMRel, StoreMemory<"sbe", GPR32Opnd, mem_simm9, null_frag, II_SBE>, POOL32C_LHUE_FM_MM<0x18, 0xa, 0x4>, ISA_MICROMIPS, ASE_EVA; def SHE_MM : MMRel, StoreMemory<"she", GPR32Opnd, mem_simm9, null_frag, II_SHE>, POOL32C_LHUE_FM_MM<0x18, 0xa, 0x5>, ISA_MICROMIPS, ASE_EVA; def SWE_MM : MMRel, StoreMemory<"swe", GPR32Opnd, mem_simm9, null_frag, II_SWE>, POOL32C_LHUE_FM_MM<0x18, 0xa, 0x7>, ISA_MICROMIPS, ASE_EVA; def LWLE_MM : MMRel, LoadLeftRightMM<"lwle", MipsLWL, GPR32Opnd, mem_mm_9, II_LWLE>, POOL32C_STEVA_LDEVA_FM_MM<0x6, 0x2>, ISA_MICROMIPS32_NOT_MIPS32R6, ASE_EVA; def LWRE_MM : MMRel, LoadLeftRightMM<"lwre", MipsLWR, GPR32Opnd, mem_mm_9, II_LWRE>, POOL32C_STEVA_LDEVA_FM_MM<0x6, 0x3>, ISA_MICROMIPS32_NOT_MIPS32R6, ASE_EVA; def SWLE_MM : MMRel, StoreLeftRightMM<"swle", MipsSWL, GPR32Opnd, mem_mm_9, II_SWLE>, POOL32C_STEVA_LDEVA_FM_MM<0xa, 0x0>, ISA_MICROMIPS32_NOT_MIPS32R6, ASE_EVA; def SWRE_MM : MMRel, StoreLeftRightMM<"swre", MipsSWR, GPR32Opnd, mem_mm_9, II_SWRE>, POOL32C_STEVA_LDEVA_FM_MM<0xa, 0x1>, ISA_MICROMIPS32_NOT_MIPS32R6, ASE_EVA; } def LWXS_MM : LoadWordIndexedScaledMM<"lwxs", GPR32Opnd>, LWXS_FM_MM<0x118>, ISA_MICROMIPS; /// Load and Store Instructions - unaligned def LWL_MM : MMRel, LoadLeftRightMM<"lwl", MipsLWL, GPR32Opnd, mem_mm_12, II_LWL>, LWL_FM_MM<0x0>, ISA_MICROMIPS32_NOT_MIPS32R6; def LWR_MM : MMRel, LoadLeftRightMM<"lwr", MipsLWR, GPR32Opnd, mem_mm_12, II_LWR>, LWL_FM_MM<0x1>, ISA_MICROMIPS32_NOT_MIPS32R6; def SWL_MM : MMRel, StoreLeftRightMM<"swl", MipsSWL, GPR32Opnd, mem_mm_12, II_SWL>, LWL_FM_MM<0x8>, ISA_MICROMIPS32_NOT_MIPS32R6; def SWR_MM : MMRel, StoreLeftRightMM<"swr", MipsSWR, GPR32Opnd, mem_mm_12, II_SWR>, LWL_FM_MM<0x9>, ISA_MICROMIPS32_NOT_MIPS32R6; } let DecoderNamespace = "MicroMips" in { /// Load and Store Instructions - multiple def SWM32_MM : StoreMultMM<"swm32", II_SWM>, LWM_FM_MM<0xd>, ISA_MICROMIPS; def LWM32_MM : LoadMultMM<"lwm32", II_LWM>, LWM_FM_MM<0x5>, ISA_MICROMIPS; /// Load and Store Pair Instructions def SWP_MM : StorePairMM<"swp">, LWM_FM_MM<0x9>, ISA_MICROMIPS; def LWP_MM : LoadPairMM<"lwp">, LWM_FM_MM<0x1>, ISA_MICROMIPS; /// Load and Store multiple pseudo Instructions class LoadWordMultMM : MipsAsmPseudoInst<(outs reglist:$rt), (ins mem_mm_12:$addr), !strconcat(instr_asm, "\t$rt, $addr")> ; class StoreWordMultMM : MipsAsmPseudoInst<(outs), (ins reglist:$rt, mem_mm_12:$addr), !strconcat(instr_asm, "\t$rt, $addr")> ; def SWM_MM : StoreWordMultMM<"swm">, ISA_MICROMIPS; def LWM_MM : LoadWordMultMM<"lwm">, ISA_MICROMIPS; /// Move Conditional def MOVZ_I_MM : MMRel, CMov_I_I_FT<"movz", GPR32Opnd, GPR32Opnd, II_MOVZ>, ADD_FM_MM<0, 0x58>, ISA_MICROMIPS32_NOT_MIPS32R6; def MOVN_I_MM : MMRel, CMov_I_I_FT<"movn", GPR32Opnd, GPR32Opnd, II_MOVN>, ADD_FM_MM<0, 0x18>, ISA_MICROMIPS32_NOT_MIPS32R6; def MOVT_I_MM : MMRel, CMov_F_I_FT<"movt", GPR32Opnd, II_MOVT, MipsCMovFP_T>, CMov_F_I_FM_MM<0x25>, ISA_MICROMIPS32_NOT_MIPS32R6; def MOVF_I_MM : MMRel, CMov_F_I_FT<"movf", GPR32Opnd, II_MOVF, MipsCMovFP_F>, CMov_F_I_FM_MM<0x5>, ISA_MICROMIPS32_NOT_MIPS32R6; /// Move to/from HI/LO def MTHI_MM : MMRel, MoveToLOHI<"mthi", GPR32Opnd, [HI0]>, MTLO_FM_MM<0x0b5>, ISA_MICROMIPS32_NOT_MIPS32R6; def MTLO_MM : MMRel, MoveToLOHI<"mtlo", GPR32Opnd, [LO0]>, MTLO_FM_MM<0x0f5>, ISA_MICROMIPS32_NOT_MIPS32R6; def MFHI_MM : MMRel, MoveFromLOHI<"mfhi", GPR32Opnd, AC0>, MFLO_FM_MM<0x035>, ISA_MICROMIPS32_NOT_MIPS32R6; def MFLO_MM : MMRel, MoveFromLOHI<"mflo", GPR32Opnd, AC0>, MFLO_FM_MM<0x075>, ISA_MICROMIPS32_NOT_MIPS32R6; /// Multiply Add/Sub Instructions def MADD_MM : MMRel, MArithR<"madd", II_MADD, 1>, MULT_FM_MM<0x32c>, ISA_MICROMIPS32_NOT_MIPS32R6; def MADDU_MM : MMRel, MArithR<"maddu", II_MADDU, 1>, MULT_FM_MM<0x36c>, ISA_MICROMIPS32_NOT_MIPS32R6; def MSUB_MM : MMRel, MArithR<"msub", II_MSUB>, MULT_FM_MM<0x3ac>, ISA_MICROMIPS32_NOT_MIPS32R6; def MSUBU_MM : MMRel, MArithR<"msubu", II_MSUBU>, MULT_FM_MM<0x3ec>, ISA_MICROMIPS32_NOT_MIPS32R6; /// Count Leading def CLZ_MM : MMRel, CountLeading0<"clz", GPR32Opnd, II_CLZ>, CLO_FM_MM<0x16c>, ISA_MICROMIPS; def CLO_MM : MMRel, CountLeading1<"clo", GPR32Opnd, II_CLO>, CLO_FM_MM<0x12c>, ISA_MICROMIPS; /// Sign Ext In Register Instructions. def SEB_MM : MMRel, SignExtInReg<"seb", i8, GPR32Opnd, II_SEB>, SEB_FM_MM<0x0ac>, ISA_MICROMIPS; def SEH_MM : MMRel, SignExtInReg<"seh", i16, GPR32Opnd, II_SEH>, SEB_FM_MM<0x0ec>, ISA_MICROMIPS; /// Word Swap Bytes Within Halfwords def WSBH_MM : MMRel, SubwordSwap<"wsbh", GPR32Opnd, II_WSBH>, SEB_FM_MM<0x1ec>, ISA_MICROMIPS; // TODO: Add '0 < pos+size <= 32' constraint check to ext instruction def EXT_MM : MMRel, ExtBase<"ext", GPR32Opnd, uimm5, uimm5_plus1, immZExt5, immZExt5Plus1, MipsExt>, EXT_FM_MM<0x2c>, ISA_MICROMIPS32_NOT_MIPS32R6; def INS_MM : MMRel, InsBase<"ins", GPR32Opnd, uimm5, uimm5_inssize_plus1, immZExt5, immZExt5Plus1>, EXT_FM_MM<0x0c>, ISA_MICROMIPS32_NOT_MIPS32R6; /// Jump Instructions let DecoderMethod = "DecodeJumpTargetMM" in def J_MM : MMRel, JumpFJ, J_FM_MM<0x35>, AdditionalRequires<[RelocNotPIC]>, IsBranch, ISA_MICROMIPS32_NOT_MIPS32R6; let DecoderMethod = "DecodeJumpTargetMM" in { def JAL_MM : MMRel, JumpLink<"jal", calltarget_mm>, J_FM_MM<0x3d>, ISA_MICROMIPS32_NOT_MIPS32R6; def JALX_MM : MMRel, JumpLink<"jalx", calltarget>, J_FM_MM<0x3c>, ISA_MICROMIPS32_NOT_MIPS32R6; } def JR_MM : MMRel, IndirectBranch<"jr", GPR32Opnd>, JR_FM_MM<0x3c>, ISA_MICROMIPS32_NOT_MIPS32R6; def JALR_MM : JumpLinkReg<"jalr", GPR32Opnd>, JALR_FM_MM<0x03c>, ISA_MICROMIPS32_NOT_MIPS32R6; /// Jump Instructions - Short Delay Slot def JALS_MM : JumpLinkMM<"jals", calltarget_mm>, J_FM_MM<0x1d>, ISA_MICROMIPS32_NOT_MIPS32R6; def JALRS_MM : JumpLinkRegMM<"jalrs", GPR32Opnd>, JALR_FM_MM<0x13c>, ISA_MICROMIPS32_NOT_MIPS32R6; /// Branch Instructions def BEQ_MM : MMRel, CBranch<"beq", brtarget_mm, seteq, GPR32Opnd>, BEQ_FM_MM<0x25>, ISA_MICROMIPS32_NOT_MIPS32R6; def BNE_MM : MMRel, CBranch<"bne", brtarget_mm, setne, GPR32Opnd>, BEQ_FM_MM<0x2d>, ISA_MICROMIPS32_NOT_MIPS32R6; def BGEZ_MM : MMRel, CBranchZero<"bgez", brtarget_mm, setge, GPR32Opnd>, BGEZ_FM_MM<0x2>, ISA_MICROMIPS32_NOT_MIPS32R6; def BGTZ_MM : MMRel, CBranchZero<"bgtz", brtarget_mm, setgt, GPR32Opnd>, BGEZ_FM_MM<0x6>, ISA_MICROMIPS32_NOT_MIPS32R6; def BLEZ_MM : MMRel, CBranchZero<"blez", brtarget_mm, setle, GPR32Opnd>, BGEZ_FM_MM<0x4>, ISA_MICROMIPS32_NOT_MIPS32R6; def BLTZ_MM : MMRel, CBranchZero<"bltz", brtarget_mm, setlt, GPR32Opnd>, BGEZ_FM_MM<0x0>, ISA_MICROMIPS32_NOT_MIPS32R6; def BGEZAL_MM : MMRel, BGEZAL_FT<"bgezal", brtarget_mm, GPR32Opnd>, BGEZAL_FM_MM<0x03>, ISA_MICROMIPS32_NOT_MIPS32R6; def BLTZAL_MM : MMRel, BGEZAL_FT<"bltzal", brtarget_mm, GPR32Opnd>, BGEZAL_FM_MM<0x01>, ISA_MICROMIPS32_NOT_MIPS32R6; def BAL_BR_MM : BAL_BR_Pseudo, ISA_MICROMIPS32_NOT_MIPS32R6; /// Branch Instructions - Short Delay Slot def BGEZALS_MM : BranchCompareToZeroLinkMM<"bgezals", brtarget_mm, GPR32Opnd>, BGEZAL_FM_MM<0x13>, ISA_MICROMIPS32_NOT_MIPS32R6; def BLTZALS_MM : BranchCompareToZeroLinkMM<"bltzals", brtarget_mm, GPR32Opnd>, BGEZAL_FM_MM<0x11>, ISA_MICROMIPS32_NOT_MIPS32R6; def B_MM : UncondBranch, IsBranch, ISA_MICROMIPS32_NOT_MIPS32R6; /// Control Instructions def SYNC_MM : MMRel, SYNC_FT<"sync">, SYNC_FM_MM, ISA_MICROMIPS; let DecoderMethod = "DecodeSyncI_MM" in def SYNCI_MM : MMRel, SYNCI_FT<"synci", mem_mm_16>, SYNCI_FM_MM, ISA_MICROMIPS32_NOT_MIPS32R6; def BREAK_MM : MMRel, BRK_FT<"break">, BRK_FM_MM, ISA_MICROMIPS; def SYSCALL_MM : MMRel, SYS_FT<"syscall", uimm10, II_SYSCALL>, SYS_FM_MM, ISA_MICROMIPS; def WAIT_MM : MMRel, WaitMM<"wait">, WAIT_FM_MM, ISA_MICROMIPS; def ERET_MM : MMRel, ER_FT<"eret", II_ERET>, ER_FM_MM<0x3cd>, ISA_MICROMIPS; def DERET_MM : MMRel, ER_FT<"deret", II_DERET>, ER_FM_MM<0x38d>, ISA_MICROMIPS; def EI_MM : MMRel, DEI_FT<"ei", GPR32Opnd, II_EI>, EI_FM_MM<0x15d>, ISA_MICROMIPS; def DI_MM : MMRel, DEI_FT<"di", GPR32Opnd, II_DI>, EI_FM_MM<0x11d>, ISA_MICROMIPS; def TRAP_MM : TrapBase, ISA_MICROMIPS; /// Trap Instructions def TEQ_MM : MMRel, TEQ_FT<"teq", GPR32Opnd, uimm4, II_TEQ>, TEQ_FM_MM<0x0>, ISA_MICROMIPS; def TGE_MM : MMRel, TEQ_FT<"tge", GPR32Opnd, uimm4, II_TGE>, TEQ_FM_MM<0x08>, ISA_MICROMIPS; def TGEU_MM : MMRel, TEQ_FT<"tgeu", GPR32Opnd, uimm4, II_TGEU>, TEQ_FM_MM<0x10>, ISA_MICROMIPS; def TLT_MM : MMRel, TEQ_FT<"tlt", GPR32Opnd, uimm4, II_TLT>, TEQ_FM_MM<0x20>, ISA_MICROMIPS; def TLTU_MM : MMRel, TEQ_FT<"tltu", GPR32Opnd, uimm4, II_TLTU>, TEQ_FM_MM<0x28>, ISA_MICROMIPS; def TNE_MM : MMRel, TEQ_FT<"tne", GPR32Opnd, uimm4, II_TNE>, TEQ_FM_MM<0x30>, ISA_MICROMIPS; def TEQI_MM : MMRel, TEQI_FT<"teqi", GPR32Opnd, II_TEQI>, TEQI_FM_MM<0x0e>, ISA_MICROMIPS32_NOT_MIPS32R6; def TGEI_MM : MMRel, TEQI_FT<"tgei", GPR32Opnd, II_TGEI>, TEQI_FM_MM<0x09>, ISA_MICROMIPS32_NOT_MIPS32R6; def TGEIU_MM : MMRel, TEQI_FT<"tgeiu", GPR32Opnd, II_TGEIU>, TEQI_FM_MM<0x0b>, ISA_MICROMIPS32_NOT_MIPS32R6; def TLTI_MM : MMRel, TEQI_FT<"tlti", GPR32Opnd, II_TLTI>, TEQI_FM_MM<0x08>, ISA_MICROMIPS32_NOT_MIPS32R6; def TLTIU_MM : MMRel, TEQI_FT<"tltiu", GPR32Opnd, II_TTLTIU>, TEQI_FM_MM<0x0a>, ISA_MICROMIPS32_NOT_MIPS32R6; def TNEI_MM : MMRel, TEQI_FT<"tnei", GPR32Opnd, II_TNEI>, TEQI_FM_MM<0x0c>, ISA_MICROMIPS32_NOT_MIPS32R6; /// Load-linked, Store-conditional def LL_MM : LLBaseMM<"ll", GPR32Opnd>, LL_FM_MM<0x3>, ISA_MICROMIPS32_NOT_MIPS32R6; def SC_MM : SCBaseMM<"sc", GPR32Opnd>, LL_FM_MM<0xb>, ISA_MICROMIPS32_NOT_MIPS32R6; def LLE_MM : MMRel, LLEBaseMM<"lle", GPR32Opnd>, LLE_FM_MM<0x6>, ISA_MICROMIPS, ASE_EVA; def SCE_MM : MMRel, SCEBaseMM<"sce", GPR32Opnd>, LLE_FM_MM<0xA>, ISA_MICROMIPS, ASE_EVA; let DecoderMethod = "DecodeCacheOpMM" in { def CACHE_MM : MMRel, CacheOp<"cache", mem_mm_12, II_CACHE>, CACHE_PREF_FM_MM<0x08, 0x6>, ISA_MICROMIPS32_NOT_MIPS32R6; def PREF_MM : MMRel, CacheOp<"pref", mem_mm_12, II_PREF>, CACHE_PREF_FM_MM<0x18, 0x2>, ISA_MICROMIPS32_NOT_MIPS32R6; } let DecoderMethod = "DecodePrefeOpMM" in { def PREFE_MM : MMRel, CacheOp<"prefe", mem_mm_9, II_PREFE>, CACHE_PREFE_FM_MM<0x18, 0x2>, ISA_MICROMIPS, ASE_EVA; def CACHEE_MM : MMRel, CacheOp<"cachee", mem_mm_9, II_CACHEE>, CACHE_PREFE_FM_MM<0x18, 0x3>, ISA_MICROMIPS, ASE_EVA; } def SSNOP_MM : MMRel, Barrier<"ssnop", II_SSNOP>, BARRIER_FM_MM<0x1>, ISA_MICROMIPS; def EHB_MM : MMRel, Barrier<"ehb", II_EHB>, BARRIER_FM_MM<0x3>, ISA_MICROMIPS; def PAUSE_MM : MMRel, Barrier<"pause", II_PAUSE>, BARRIER_FM_MM<0x5>, ISA_MICROMIPS; def TLBP_MM : MMRel, TLB<"tlbp", II_TLBP>, COP0_TLB_FM_MM<0x0d>, ISA_MICROMIPS; def TLBR_MM : MMRel, TLB<"tlbr", II_TLBR>, COP0_TLB_FM_MM<0x4d>, ISA_MICROMIPS; def TLBWI_MM : MMRel, TLB<"tlbwi", II_TLBWI>, COP0_TLB_FM_MM<0x8d>, ISA_MICROMIPS; def TLBWR_MM : MMRel, TLB<"tlbwr", II_TLBWR>, COP0_TLB_FM_MM<0xcd>, ISA_MICROMIPS; def SDBBP_MM : MMRel, SYS_FT<"sdbbp", uimm10, II_SDBBP>, SDBBP_FM_MM, ISA_MICROMIPS; def PREFX_MM : PrefetchIndexed<"prefx">, POOL32F_PREFX_FM_MM<0x15, 0x1A0>, ISA_MICROMIPS32_NOT_MIPS32R6; } let AdditionalPredicates = [NotDSP] in { def PseudoMULT_MM : MultDivPseudo, ISA_MICROMIPS32_NOT_MIPS32R6; def PseudoMULTu_MM : MultDivPseudo, ISA_MICROMIPS32_NOT_MIPS32R6; def PseudoMFHI_MM : PseudoMFLOHI, ISA_MICROMIPS32_NOT_MIPS32R6; def PseudoMFLO_MM : PseudoMFLOHI, ISA_MICROMIPS32_NOT_MIPS32R6; def PseudoMTLOHI_MM : PseudoMTLOHI, ISA_MICROMIPS32_NOT_MIPS32R6; def PseudoMADD_MM : MAddSubPseudo, ISA_MICROMIPS32_NOT_MIPS32R6; def PseudoMADDU_MM : MAddSubPseudo, ISA_MICROMIPS32_NOT_MIPS32R6; def PseudoMSUB_MM : MAddSubPseudo, ISA_MICROMIPS32_NOT_MIPS32R6; def PseudoMSUBU_MM : MAddSubPseudo, ISA_MICROMIPS32_NOT_MIPS32R6; } def TAILCALL_MM : TailCall, ISA_MIPS1_NOT_32R6_64R6; def TAILCALLREG_MM : TailCallReg, ISA_MICROMIPS32_NOT_MIPS32R6; def PseudoIndirectBranch_MM : PseudoIndirectBranchBase, ISA_MICROMIPS32_NOT_MIPS32R6; let DecoderNamespace = "MicroMips" in { def RDHWR_MM : MMRel, R6MMR6Rel, ReadHardware, RDHWR_FM_MM, ISA_MICROMIPS32_NOT_MIPS32R6; def LWU_MM : MMRel, LoadMM<"lwu", GPR32Opnd, zextloadi32, II_LWU, mem_simm12>, LL_FM_MM<0xe>, ISA_MICROMIPS32_NOT_MIPS32R6; } let DecoderNamespace = "MicroMips" in { def MFGC0_MM : MMRel, MfCop0MM<"mfgc0", GPR32Opnd, COP0Opnd, II_MFGC0>, POOL32A_MFTC0_FM_MM<0b10011, 0b111100>, ISA_MICROMIPS32R5, ASE_VIRT; def MFHGC0_MM : MMRel, MfCop0MM<"mfhgc0", GPR32Opnd, COP0Opnd, II_MFHGC0>, POOL32A_MFTC0_FM_MM<0b10011, 0b110100>, ISA_MICROMIPS32R5, ASE_VIRT; def MTGC0_MM : MMRel, MtCop0MM<"mtgc0", COP0Opnd, GPR32Opnd, II_MTGC0>, POOL32A_MFTC0_FM_MM<0b11011, 0b111100>, ISA_MICROMIPS32R5, ASE_VIRT; def MTHGC0_MM : MMRel, MtCop0MM<"mthgc0", COP0Opnd, GPR32Opnd, II_MTHGC0>, POOL32A_MFTC0_FM_MM<0b11011, 0b110100>, ISA_MICROMIPS32R5, ASE_VIRT; def HYPCALL_MM : MMRel, HypcallMM<"hypcall">, POOL32A_HYPCALL_FM_MM, ISA_MICROMIPS32R5, ASE_VIRT; def TLBGINV_MM : MMRel, TLBINVMM<"tlbginv", II_TLBGINV>, POOL32A_TLBINV_FM_MM<0x105>, ISA_MICROMIPS32R5, ASE_VIRT; def TLBGINVF_MM : MMRel, TLBINVMM<"tlbginvf", II_TLBGINVF>, POOL32A_TLBINV_FM_MM<0x145>, ISA_MICROMIPS32R5, ASE_VIRT; def TLBGP_MM : MMRel, TLBINVMM<"tlbgp", II_TLBGP>, POOL32A_TLBINV_FM_MM<0x5>, ISA_MICROMIPS32R5, ASE_VIRT; def TLBGR_MM : MMRel, TLBINVMM<"tlbgr", II_TLBGR>, POOL32A_TLBINV_FM_MM<0x45>, ISA_MICROMIPS32R5, ASE_VIRT; def TLBGWI_MM : MMRel, TLBINVMM<"tlbgwi", II_TLBGWI>, POOL32A_TLBINV_FM_MM<0x85>, ISA_MICROMIPS32R5, ASE_VIRT; def TLBGWR_MM : MMRel, TLBINVMM<"tlbgwr", II_TLBGWR>, POOL32A_TLBINV_FM_MM<0xc5>, ISA_MICROMIPS32R5, ASE_VIRT; } //===----------------------------------------------------------------------===// // MicroMips arbitrary patterns that map to one or more instructions //===----------------------------------------------------------------------===// defm : MipsHiLoRelocs, ISA_MICROMIPS; def : MipsPat<(MipsGotHi tglobaladdr:$in), (LUi_MM tglobaladdr:$in)>, ISA_MICROMIPS; def : MipsPat<(MipsGotHi texternalsym:$in), (LUi_MM texternalsym:$in)>, ISA_MICROMIPS; def : MipsPat<(MipsTlsHi tglobaltlsaddr:$in), (LUi_MM tglobaltlsaddr:$in)>, ISA_MICROMIPS; // gp_rel relocs def : MipsPat<(add GPR32:$gp, (MipsGPRel tglobaladdr:$in)), (ADDiu_MM GPR32:$gp, tglobaladdr:$in)>, ISA_MICROMIPS; def : MipsPat<(add GPR32:$gp, (MipsGPRel tconstpool:$in)), (ADDiu_MM GPR32:$gp, tconstpool:$in)>, ISA_MICROMIPS; def : WrapperPat, ISA_MICROMIPS; def : WrapperPat, ISA_MICROMIPS; def : WrapperPat, ISA_MICROMIPS; def : WrapperPat, ISA_MICROMIPS; def : WrapperPat, ISA_MICROMIPS; def : WrapperPat, ISA_MICROMIPS; def : MipsPat<(atomic_load_8 addr:$a), (LB_MM addr:$a)>, ISA_MICROMIPS; def : MipsPat<(atomic_load_16 addr:$a), (LH_MM addr:$a)>, ISA_MICROMIPS; def : MipsPat<(atomic_load_32 addr:$a), (LW_MM addr:$a)>, ISA_MICROMIPS; def : MipsPat<(i32 immLi16:$imm), (LI16_MM immLi16:$imm)>, ISA_MICROMIPS; defm : MaterializeImms, ISA_MICROMIPS; def : MipsPat<(not GPRMM16:$in), (NOT16_MM GPRMM16:$in)>, ISA_MICROMIPS; def : MipsPat<(not GPR32:$in), (NOR_MM GPR32Opnd:$in, ZERO)>, ISA_MICROMIPS; def : MipsPat<(add GPRMM16:$src, immSExtAddiur2:$imm), (ADDIUR2_MM GPRMM16:$src, immSExtAddiur2:$imm)>, ISA_MICROMIPS; def : MipsPat<(add GPR32:$src, immSExtAddius5:$imm), (ADDIUS5_MM GPR32:$src, immSExtAddius5:$imm)>, ISA_MICROMIPS; def : MipsPat<(add GPR32:$src, immSExt16:$imm), (ADDiu_MM GPR32:$src, immSExt16:$imm)>, ISA_MICROMIPS; def : MipsPat<(and GPRMM16:$src, immZExtAndi16:$imm), (ANDI16_MM GPRMM16:$src, immZExtAndi16:$imm)>, ISA_MICROMIPS; def : MipsPat<(and GPR32:$src, immZExt16:$imm), (ANDi_MM GPR32:$src, immZExt16:$imm)>, ISA_MICROMIPS; def : MipsPat<(shl GPRMM16:$src, immZExt2Shift:$imm), (SLL16_MM GPRMM16:$src, immZExt2Shift:$imm)>, ISA_MICROMIPS; def : MipsPat<(shl GPR32:$src, immZExt5:$imm), (SLL_MM GPR32:$src, immZExt5:$imm)>, ISA_MICROMIPS; def : MipsPat<(shl GPR32:$lhs, GPR32:$rhs), (SLLV_MM GPR32:$lhs, GPR32:$rhs)>, ISA_MICROMIPS; def : MipsPat<(srl GPRMM16:$src, immZExt2Shift:$imm), (SRL16_MM GPRMM16:$src, immZExt2Shift:$imm)>, ISA_MICROMIPS; def : MipsPat<(srl GPR32:$src, immZExt5:$imm), (SRL_MM GPR32:$src, immZExt5:$imm)>, ISA_MICROMIPS; def : MipsPat<(srl GPR32:$lhs, GPR32:$rhs), (SRLV_MM GPR32:$lhs, GPR32:$rhs)>, ISA_MICROMIPS; def : MipsPat<(sra GPR32:$src, immZExt5:$imm), (SRA_MM GPR32:$src, immZExt5:$imm)>, ISA_MICROMIPS; def : MipsPat<(sra GPR32:$lhs, GPR32:$rhs), (SRAV_MM GPR32:$lhs, GPR32:$rhs)>, ISA_MICROMIPS; def : MipsPat<(store GPRMM16:$src, addrimm4lsl2:$addr), (SW16_MM GPRMM16:$src, addrimm4lsl2:$addr)>, ISA_MICROMIPS; def : MipsPat<(store GPR32:$src, addr:$addr), (SW_MM GPR32:$src, addr:$addr)>, ISA_MICROMIPS; def : MipsPat<(load addrimm4lsl2:$addr), (LW16_MM addrimm4lsl2:$addr)>, ISA_MICROMIPS; def : MipsPat<(load addr:$addr), (LW_MM addr:$addr)>, ISA_MICROMIPS; def : MipsPat<(subc GPR32:$lhs, GPR32:$rhs), (SUBu_MM GPR32:$lhs, GPR32:$rhs)>, ISA_MICROMIPS; def : MipsPat<(i32 (extloadi1 addr:$src)), (LBu_MM addr:$src)>, ISA_MICROMIPS; def : MipsPat<(i32 (extloadi8 addr:$src)), (LBu_MM addr:$src)>, ISA_MICROMIPS; def : MipsPat<(i32 (extloadi16 addr:$src)), (LHu_MM addr:$src)>, ISA_MICROMIPS; let AddedComplexity = 40 in def : MipsPat<(i32 (sextloadi16 addrRegImm:$a)), (LH_MM addrRegImm:$a)>, ISA_MICROMIPS; def : MipsPat<(bswap GPR32:$rt), (ROTR_MM (WSBH_MM GPR32:$rt), 16)>, ISA_MICROMIPS; def : MipsPat<(MipsJmpLink (i32 texternalsym:$dst)), (JAL_MM texternalsym:$dst)>, ISA_MICROMIPS32_NOT_MIPS32R6; def : MipsPat<(MipsTailCall (iPTR tglobaladdr:$dst)), (TAILCALL_MM tglobaladdr:$dst)>, ISA_MICROMIPS32_NOT_MIPS32R6; def : MipsPat<(MipsTailCall (iPTR texternalsym:$dst)), (TAILCALL_MM texternalsym:$dst)>, ISA_MICROMIPS32_NOT_MIPS32R6; defm : BrcondPats, ISA_MICROMIPS32_NOT_MIPS32R6; def : MipsPat<(brcond (i32 (setlt i32:$lhs, 1)), bb:$dst), (BLEZ_MM i32:$lhs, bb:$dst)>, ISA_MICROMIPS32_NOT_MIPS32R6; def : MipsPat<(brcond (i32 (setgt i32:$lhs, -1)), bb:$dst), (BGEZ_MM i32:$lhs, bb:$dst)>, ISA_MICROMIPS32_NOT_MIPS32R6; defm : SeteqPats, ISA_MICROMIPS; defm : SetlePats, ISA_MICROMIPS; defm : SetgtPats, ISA_MICROMIPS; defm : SetgePats, ISA_MICROMIPS; defm : SetgeImmPats, ISA_MICROMIPS; // Select patterns // Instantiation of conditional move patterns. defm : MovzPats0, ISA_MICROMIPS32_NOT_MIPS32R6; defm : MovzPats1, ISA_MICROMIPS32_NOT_MIPS32R6; defm : MovzPats2, ISA_MICROMIPS32_NOT_MIPS32R6; defm : MovnPats, INSN_MIPS4_32_NOT_32R6_64R6; // Instantiation of conditional move patterns. defm : MovzPats0, ISA_MICROMIPS32_NOT_MIPS32R6; defm : MovzPats1, ISA_MICROMIPS32_NOT_MIPS32R6; defm : MovzPats2, ISA_MICROMIPS32_NOT_MIPS32R6; defm : MovnPats, ISA_MICROMIPS32_NOT_MIPS32R6; //===----------------------------------------------------------------------===// // MicroMips instruction aliases //===----------------------------------------------------------------------===// class UncondBranchMMPseudo : MipsAsmPseudoInst<(outs), (ins brtarget_mm:$offset), !strconcat(opstr, "\t$offset")>; def B_MM_Pseudo : UncondBranchMMPseudo<"b">, ISA_MICROMIPS; let EncodingPredicates = [InMicroMips] in { def SDIV_MM_Pseudo : MultDivPseudo, ISA_MIPS1_NOT_32R6_64R6; def UDIV_MM_Pseudo : MultDivPseudo, ISA_MIPS1_NOT_32R6_64R6; def : MipsInstAlias<"wait", (WAIT_MM 0x0), 1>, ISA_MICROMIPS; def : MipsInstAlias<"nop", (SLL_MM ZERO, ZERO, 0), 1>, ISA_MICROMIPS; def : MipsInstAlias<"nop", (MOVE16_MM ZERO, ZERO), 1>, ISA_MICROMIPS; def : MipsInstAlias<"ei", (EI_MM ZERO), 1>, ISA_MICROMIPS; def : MipsInstAlias<"di", (DI_MM ZERO), 1>, ISA_MICROMIPS; def : MipsInstAlias<"neg $rt, $rs", (SUB_MM GPR32Opnd:$rt, ZERO, GPR32Opnd:$rs), 1>, ISA_MICROMIPS32_NOT_MIPS32R6; def : MipsInstAlias<"neg $rt", (SUB_MM GPR32Opnd:$rt, ZERO, GPR32Opnd:$rt), 1>, ISA_MICROMIPS32_NOT_MIPS32R6; def : MipsInstAlias<"negu $rt, $rs", (SUBu_MM GPR32Opnd:$rt, ZERO, GPR32Opnd:$rs), 1>, ISA_MICROMIPS32_NOT_MIPS32R6; def : MipsInstAlias<"negu $rt", (SUBu_MM GPR32Opnd:$rt, ZERO, GPR32Opnd:$rt), 1>, ISA_MICROMIPS32_NOT_MIPS32R6; def : MipsInstAlias<"teq $rs, $rt", (TEQ_MM GPR32Opnd:$rs, GPR32Opnd:$rt, 0), 1>; def : MipsInstAlias<"tge $rs, $rt", (TGE_MM GPR32Opnd:$rs, GPR32Opnd:$rt, 0), 1>; def : MipsInstAlias<"tgeu $rs, $rt", (TGEU_MM GPR32Opnd:$rs, GPR32Opnd:$rt, 0), 1>; def : MipsInstAlias<"tlt $rs, $rt", (TLT_MM GPR32Opnd:$rs, GPR32Opnd:$rt, 0), 1>; def : MipsInstAlias<"tltu $rs, $rt", (TLTU_MM GPR32Opnd:$rs, GPR32Opnd:$rt, 0), 1>; def : MipsInstAlias<"tne $rs, $rt", (TNE_MM GPR32Opnd:$rs, GPR32Opnd:$rt, 0), 1>; def : MipsInstAlias< "sgt $rd, $rs, $rt", (SLT_MM GPR32Opnd:$rd, GPR32Opnd:$rt, GPR32Opnd:$rs), 0>; def : MipsInstAlias< "sgt $rs, $rt", (SLT_MM GPR32Opnd:$rs, GPR32Opnd:$rt, GPR32Opnd:$rs), 0>; def : MipsInstAlias< "sgtu $rd, $rs, $rt", (SLTu_MM GPR32Opnd:$rd, GPR32Opnd:$rt, GPR32Opnd:$rs), 0>; def : MipsInstAlias< "sgtu $rs, $rt", (SLTu_MM GPR32Opnd:$rs, GPR32Opnd:$rt, GPR32Opnd:$rs), 0>; def : MipsInstAlias<"sll $rd, $rt, $rs", (SLLV_MM GPR32Opnd:$rd, GPR32Opnd:$rt, GPR32Opnd:$rs), 0>; def : MipsInstAlias<"sra $rd, $rt, $rs", (SRAV_MM GPR32Opnd:$rd, GPR32Opnd:$rt, GPR32Opnd:$rs), 0>; def : MipsInstAlias<"srl $rd, $rt, $rs", (SRLV_MM GPR32Opnd:$rd, GPR32Opnd:$rt, GPR32Opnd:$rs), 0>; def : MipsInstAlias<"sll $rd, $rt", (SLLV_MM GPR32Opnd:$rd, GPR32Opnd:$rd, GPR32Opnd:$rt), 0>; def : MipsInstAlias<"sra $rd, $rt", (SRAV_MM GPR32Opnd:$rd, GPR32Opnd:$rd, GPR32Opnd:$rt), 0>; def : MipsInstAlias<"srl $rd, $rt", (SRLV_MM GPR32Opnd:$rd, GPR32Opnd:$rd, GPR32Opnd:$rt), 0>; def : MipsInstAlias<"sll $rd, $shamt", (SLL_MM GPR32Opnd:$rd, GPR32Opnd:$rd, uimm5:$shamt), 0>; def : MipsInstAlias<"sra $rd, $shamt", (SRA_MM GPR32Opnd:$rd, GPR32Opnd:$rd, uimm5:$shamt), 0>; def : MipsInstAlias<"srl $rd, $shamt", (SRL_MM GPR32Opnd:$rd, GPR32Opnd:$rd, uimm5:$shamt), 0>; def : MipsInstAlias<"rotr $rt, $imm", (ROTR_MM GPR32Opnd:$rt, GPR32Opnd:$rt, uimm5:$imm), 0>; def : MipsInstAlias<"syscall", (SYSCALL_MM 0), 1>, ISA_MICROMIPS; def : MipsInstAlias<"sync", (SYNC_MM 0), 1>, ISA_MICROMIPS; defm : OneOrTwoOperandMacroImmediateAlias<"add", ADDi_MM>, ISA_MICROMIPS; defm : OneOrTwoOperandMacroImmediateAlias<"addu", ADDiu_MM>, ISA_MICROMIPS; defm : OneOrTwoOperandMacroImmediateAlias<"and", ANDi_MM>, ISA_MICROMIPS; defm : OneOrTwoOperandMacroImmediateAlias<"or", ORi_MM>, ISA_MICROMIPS; defm : OneOrTwoOperandMacroImmediateAlias<"xor", XORi_MM>, ISA_MICROMIPS; defm : OneOrTwoOperandMacroImmediateAlias<"slt", SLTi_MM>, ISA_MICROMIPS; defm : OneOrTwoOperandMacroImmediateAlias<"sltu", SLTiu_MM>, ISA_MICROMIPS; def : MipsInstAlias<"not $rt, $rs", (NOR_MM GPR32Opnd:$rt, GPR32Opnd:$rs, ZERO), 0>, ISA_MICROMIPS32_NOT_MIPS32R6; def : MipsInstAlias<"not $rt", (NOR_MM GPR32Opnd:$rt, GPR32Opnd:$rt, ZERO), 0>, ISA_MICROMIPS32_NOT_MIPS32R6; def : MipsInstAlias<"bnez $rs,$offset", (BNE_MM GPR32Opnd:$rs, ZERO, brtarget:$offset), 0>, ISA_MICROMIPS; def : MipsInstAlias<"beqz $rs,$offset", (BEQ_MM GPR32Opnd:$rs, ZERO, brtarget:$offset), 0>, ISA_MICROMIPS; def : MipsInstAlias<"seh $rd", (SEH_MM GPR32Opnd:$rd, GPR32Opnd:$rd), 0>, ISA_MICROMIPS; def : MipsInstAlias<"seb $rd", (SEB_MM GPR32Opnd:$rd, GPR32Opnd:$rd), 0>, ISA_MICROMIPS; def : MipsInstAlias<"break", (BREAK_MM 0, 0), 1>, ISA_MICROMIPS; def : MipsInstAlias<"break $imm", (BREAK_MM uimm10:$imm, 0), 1>, ISA_MICROMIPS; def : MipsInstAlias<"bal $offset", (BGEZAL_MM ZERO, brtarget_mm:$offset), 1>, ISA_MICROMIPS32_NOT_MIPS32R6; def : MipsInstAlias<"j $rs", (JR_MM GPR32Opnd:$rs), 0>, ISA_MICROMIPS32_NOT_MIPS32R6; } def : MipsInstAlias<"rdhwr $rt, $rs", (RDHWR_MM GPR32Opnd:$rt, HWRegsOpnd:$rs, 0), 1>, ISA_MICROMIPS32_NOT_MIPS32R6; def : MipsInstAlias<"hypcall", (HYPCALL_MM 0), 1>, ISA_MICROMIPS32R5, ASE_VIRT; def : MipsInstAlias<"mfgc0 $rt, $rs", (MFGC0_MM GPR32Opnd:$rt, COP0Opnd:$rs, 0), 0>, ISA_MICROMIPS32R5, ASE_VIRT; def : MipsInstAlias<"mfhgc0 $rt, $rs", (MFHGC0_MM GPR32Opnd:$rt, COP0Opnd:$rs, 0), 0>, ISA_MICROMIPS32R5, ASE_VIRT; def : MipsInstAlias<"mtgc0 $rt, $rs", (MTGC0_MM COP0Opnd:$rs, GPR32Opnd:$rt, 0), 0>, ISA_MICROMIPS32R5, ASE_VIRT; def : MipsInstAlias<"mthgc0 $rt, $rs", (MTHGC0_MM COP0Opnd:$rs, GPR32Opnd:$rt, 0), 0>, ISA_MICROMIPS32R5, ASE_VIRT; Index: vendor/llvm/dist-release_80/lib/Target/Mips/Mips32r6InstrInfo.td =================================================================== --- vendor/llvm/dist-release_80/lib/Target/Mips/Mips32r6InstrInfo.td (revision 343793) +++ vendor/llvm/dist-release_80/lib/Target/Mips/Mips32r6InstrInfo.td (revision 343794) @@ -1,1143 +1,1143 @@ //=- Mips32r6InstrInfo.td - Mips32r6 Instruction Information -*- tablegen -*-=// // // The LLVM Compiler Infrastructure // // This file is distributed under the University of Illinois Open Source // License. See LICENSE.TXT for details. // //===----------------------------------------------------------------------===// // // This file describes Mips32r6 instructions. // //===----------------------------------------------------------------------===// include "Mips32r6InstrFormats.td" //===----------------------------------------------------------------------===// // // Mips profiles and nodes // //===----------------------------------------------------------------------===// def SDT_MipsFSelect : SDTypeProfile<1, 3, [SDTCisFP<1>, SDTCisSameAs<0,2>, SDTCisSameAs<2,3>]>; def MipsFSelect : SDNode<"MipsISD::FSELECT", SDT_MipsFSelect>; //===----------------------------------------------------------------------===// // // Mips Operands // //===----------------------------------------------------------------------===// // Notes about removals/changes from MIPS32r6: // Reencoded: jr -> jalr // Reencoded: jr.hb -> jalr.hb def brtarget21 : Operand { let EncoderMethod = "getBranchTarget21OpValue"; let OperandType = "OPERAND_PCREL"; let DecoderMethod = "DecodeBranchTarget21"; let ParserMatchClass = MipsJumpTargetAsmOperand; } def brtarget26 : Operand { let EncoderMethod = "getBranchTarget26OpValue"; let OperandType = "OPERAND_PCREL"; let DecoderMethod = "DecodeBranchTarget26"; let ParserMatchClass = MipsJumpTargetAsmOperand; } def jmpoffset16 : Operand { let EncoderMethod = "getJumpOffset16OpValue"; let ParserMatchClass = MipsJumpTargetAsmOperand; } def calloffset16 : Operand { let EncoderMethod = "getJumpOffset16OpValue"; let ParserMatchClass = MipsJumpTargetAsmOperand; } //===----------------------------------------------------------------------===// // // Instruction Encodings // //===----------------------------------------------------------------------===// class ADDIUPC_ENC : PCREL19_FM; class ALIGN_ENC : SPECIAL3_ALIGN_FM; class ALUIPC_ENC : PCREL16_FM; class AUI_ENC : AUI_FM; class AUIPC_ENC : PCREL16_FM; class BAL_ENC : BAL_FM; class BALC_ENC : BRANCH_OFF26_FM<0b111010>; class BC_ENC : BRANCH_OFF26_FM<0b110010>; class BEQC_ENC : CMP_BRANCH_2R_OFF16_FM, DecodeDisambiguates<"AddiGroupBranch">; class BEQZALC_ENC : CMP_BRANCH_1R_RT_OFF16_FM, DecodeDisambiguatedBy<"DaddiGroupBranch">; class BNEC_ENC : CMP_BRANCH_2R_OFF16_FM, DecodeDisambiguates<"DaddiGroupBranch">; class BNEZALC_ENC : CMP_BRANCH_1R_RT_OFF16_FM, DecodeDisambiguatedBy<"DaddiGroupBranch">; class BLTZC_ENC : CMP_BRANCH_1R_BOTH_OFF16_FM, DecodeDisambiguates<"BgtzlGroupBranch">; class BGEC_ENC : CMP_BRANCH_2R_OFF16_FM, DecodeDisambiguatedBy<"BlezlGroupBranch">; class BGEUC_ENC : CMP_BRANCH_2R_OFF16_FM, DecodeDisambiguatedBy<"BlezGroupBranch">; class BGEZC_ENC : CMP_BRANCH_1R_BOTH_OFF16_FM, DecodeDisambiguates<"BlezlGroupBranch">; class BGTZALC_ENC : CMP_BRANCH_1R_RT_OFF16_FM, DecodeDisambiguatedBy<"BgtzGroupBranch">; class BLTC_ENC : CMP_BRANCH_2R_OFF16_FM, DecodeDisambiguatedBy<"BgtzlGroupBranch">; class BLTUC_ENC : CMP_BRANCH_2R_OFF16_FM, DecodeDisambiguatedBy<"BgtzGroupBranch">; class BLEZC_ENC : CMP_BRANCH_1R_RT_OFF16_FM, DecodeDisambiguatedBy<"BlezlGroupBranch">; class BLTZALC_ENC : CMP_BRANCH_1R_BOTH_OFF16_FM, DecodeDisambiguates<"BgtzGroupBranch">; class BGTZC_ENC : CMP_BRANCH_1R_RT_OFF16_FM, DecodeDisambiguatedBy<"BgtzlGroupBranch">; class BEQZC_ENC : CMP_BRANCH_OFF21_FM<0b110110>; class BGEZALC_ENC : CMP_BRANCH_1R_BOTH_OFF16_FM, DecodeDisambiguates<"BlezGroupBranch">; class BNEZC_ENC : CMP_BRANCH_OFF21_FM<0b111110>; class BC1EQZ_ENC : COP1_BCCZ_FM; class BC1NEZ_ENC : COP1_BCCZ_FM; class BC2EQZ_ENC : COP2_BCCZ_FM; class BC2NEZ_ENC : COP2_BCCZ_FM; class DVP_ENC : COP0_EVP_DVP_FM<0b1>; class EVP_ENC : COP0_EVP_DVP_FM<0b0>; class JIALC_ENC : JMP_IDX_COMPACT_FM<0b111110>; class JIC_ENC : JMP_IDX_COMPACT_FM<0b110110>; class JR_HB_R6_ENC : JR_HB_R6_FM; class BITSWAP_ENC : SPECIAL3_2R_FM; class BLEZALC_ENC : CMP_BRANCH_1R_RT_OFF16_FM, DecodeDisambiguatedBy<"BlezGroupBranch">; class BNVC_ENC : CMP_BRANCH_2R_OFF16_FM, DecodeDisambiguatedBy<"DaddiGroupBranch">; class BOVC_ENC : CMP_BRANCH_2R_OFF16_FM, DecodeDisambiguatedBy<"AddiGroupBranch">; class DIV_ENC : SPECIAL_3R_FM<0b00010, 0b011010>; class DIVU_ENC : SPECIAL_3R_FM<0b00010, 0b011011>; class MOD_ENC : SPECIAL_3R_FM<0b00011, 0b011010>; class MODU_ENC : SPECIAL_3R_FM<0b00011, 0b011011>; class MUH_ENC : SPECIAL_3R_FM<0b00011, 0b011000>; class MUHU_ENC : SPECIAL_3R_FM<0b00011, 0b011001>; class MUL_R6_ENC : SPECIAL_3R_FM<0b00010, 0b011000>; class MULU_ENC : SPECIAL_3R_FM<0b00010, 0b011001>; class MADDF_S_ENC : COP1_3R_FM<0b011000, FIELD_FMT_S>; class MADDF_D_ENC : COP1_3R_FM<0b011000, FIELD_FMT_D>; class MSUBF_S_ENC : COP1_3R_FM<0b011001, FIELD_FMT_S>; class MSUBF_D_ENC : COP1_3R_FM<0b011001, FIELD_FMT_D>; class SEL_D_ENC : COP1_3R_FM<0b010000, FIELD_FMT_D>; class SEL_S_ENC : COP1_3R_FM<0b010000, FIELD_FMT_S>; class SELEQZ_ENC : SPECIAL_3R_FM<0b00000, 0b110101>; class SELNEZ_ENC : SPECIAL_3R_FM<0b00000, 0b110111>; class LWPC_ENC : PCREL19_FM; class LWUPC_ENC : PCREL19_FM; class MAX_S_ENC : COP1_3R_FM<0b011101, FIELD_FMT_S>; class MAX_D_ENC : COP1_3R_FM<0b011101, FIELD_FMT_D>; class MIN_S_ENC : COP1_3R_FM<0b011100, FIELD_FMT_S>; class MIN_D_ENC : COP1_3R_FM<0b011100, FIELD_FMT_D>; class MAXA_S_ENC : COP1_3R_FM<0b011111, FIELD_FMT_S>; class MAXA_D_ENC : COP1_3R_FM<0b011111, FIELD_FMT_D>; class MINA_S_ENC : COP1_3R_FM<0b011110, FIELD_FMT_S>; class MINA_D_ENC : COP1_3R_FM<0b011110, FIELD_FMT_D>; class SELEQZ_S_ENC : COP1_3R_FM<0b010100, FIELD_FMT_S>; class SELEQZ_D_ENC : COP1_3R_FM<0b010100, FIELD_FMT_D>; class SELNEZ_S_ENC : COP1_3R_FM<0b010111, FIELD_FMT_S>; class SELNEZ_D_ENC : COP1_3R_FM<0b010111, FIELD_FMT_D>; class RINT_S_ENC : COP1_2R_FM<0b011010, FIELD_FMT_S>; class RINT_D_ENC : COP1_2R_FM<0b011010, FIELD_FMT_D>; class CLASS_S_ENC : COP1_2R_FM<0b011011, FIELD_FMT_S>; class CLASS_D_ENC : COP1_2R_FM<0b011011, FIELD_FMT_D>; class CACHE_ENC : SPECIAL3_MEM_FM; class PREF_ENC : SPECIAL3_MEM_FM; class LDC2_R6_ENC : COP2LDST_FM; class LWC2_R6_ENC : COP2LDST_FM; class SDC2_R6_ENC : COP2LDST_FM; class SWC2_R6_ENC : COP2LDST_FM; class LSA_R6_ENC : SPECIAL_LSA_FM; class LL_R6_ENC : SPECIAL3_LL_SC_FM; class SC_R6_ENC : SPECIAL3_LL_SC_FM; class CLO_R6_ENC : SPECIAL_2R_FM; class CLZ_R6_ENC : SPECIAL_2R_FM; class SDBBP_R6_ENC : SPECIAL_SDBBP_FM; class CRC32B_ENC : SPECIAL3_2R_SZ_CRC<0,0>; class CRC32H_ENC : SPECIAL3_2R_SZ_CRC<1,0>; class CRC32W_ENC : SPECIAL3_2R_SZ_CRC<2,0>; class CRC32CB_ENC : SPECIAL3_2R_SZ_CRC<0,1>; class CRC32CH_ENC : SPECIAL3_2R_SZ_CRC<1,1>; class CRC32CW_ENC : SPECIAL3_2R_SZ_CRC<2,1>; class GINVI_ENC : SPECIAL3_GINV<0>; class GINVT_ENC : SPECIAL3_GINV<2>; class SIGRIE_ENC : SIGRIE_FM; //===----------------------------------------------------------------------===// // // Instruction Multiclasses // //===----------------------------------------------------------------------===// class CMP_CONDN_DESC_BASE { dag OutOperandList = (outs FGRCCOpnd:$fd); dag InOperandList = (ins FGROpnd:$fs, FGROpnd:$ft); string AsmString = !strconcat("cmp.", CondStr, ".", Typestr, "\t$fd, $fs, $ft"); list Pattern = [(set FGRCCOpnd:$fd, (Op FGROpnd:$fs, FGROpnd:$ft))]; bit isCTI = 1; InstrItinClass Itinerary = Itin; } multiclass CMP_CC_M { let AdditionalPredicates = [NotInMicroMips] in { def CMP_F_#NAME : R6MMR6Rel, COP1_CMP_CONDN_FM, CMP_CONDN_DESC_BASE<"af", Typestr, FGROpnd, Itin>, MipsR6Arch, ISA_MIPS32R6, HARDFLOAT; def CMP_UN_#NAME : R6MMR6Rel, COP1_CMP_CONDN_FM, CMP_CONDN_DESC_BASE<"un", Typestr, FGROpnd, Itin, setuo>, MipsR6Arch, ISA_MIPS32R6, HARDFLOAT; def CMP_EQ_#NAME : R6MMR6Rel, COP1_CMP_CONDN_FM, CMP_CONDN_DESC_BASE<"eq", Typestr, FGROpnd, Itin, setoeq>, MipsR6Arch, ISA_MIPS32R6, HARDFLOAT; def CMP_UEQ_#NAME : R6MMR6Rel, COP1_CMP_CONDN_FM, CMP_CONDN_DESC_BASE<"ueq", Typestr, FGROpnd, Itin, setueq>, MipsR6Arch, ISA_MIPS32R6, HARDFLOAT; def CMP_LT_#NAME : R6MMR6Rel, COP1_CMP_CONDN_FM, CMP_CONDN_DESC_BASE<"lt", Typestr, FGROpnd, Itin, setolt>, MipsR6Arch, ISA_MIPS32R6, HARDFLOAT; def CMP_ULT_#NAME : R6MMR6Rel, COP1_CMP_CONDN_FM, CMP_CONDN_DESC_BASE<"ult", Typestr, FGROpnd, Itin, setult>, MipsR6Arch, ISA_MIPS32R6, HARDFLOAT; def CMP_LE_#NAME : R6MMR6Rel, COP1_CMP_CONDN_FM, CMP_CONDN_DESC_BASE<"le", Typestr, FGROpnd, Itin, setole>, MipsR6Arch, ISA_MIPS32R6, HARDFLOAT; def CMP_ULE_#NAME : R6MMR6Rel, COP1_CMP_CONDN_FM, CMP_CONDN_DESC_BASE<"ule", Typestr, FGROpnd, Itin, setule>, MipsR6Arch, ISA_MIPS32R6, HARDFLOAT; def CMP_SAF_#NAME : R6MMR6Rel, COP1_CMP_CONDN_FM, CMP_CONDN_DESC_BASE<"saf", Typestr, FGROpnd, Itin>, MipsR6Arch, ISA_MIPS32R6, HARDFLOAT; def CMP_SUN_#NAME : R6MMR6Rel, COP1_CMP_CONDN_FM, CMP_CONDN_DESC_BASE<"sun", Typestr, FGROpnd, Itin>, MipsR6Arch, ISA_MIPS32R6, HARDFLOAT; def CMP_SEQ_#NAME : R6MMR6Rel, COP1_CMP_CONDN_FM, CMP_CONDN_DESC_BASE<"seq", Typestr, FGROpnd, Itin>, MipsR6Arch, ISA_MIPS32R6, HARDFLOAT; def CMP_SUEQ_#NAME : R6MMR6Rel, COP1_CMP_CONDN_FM, CMP_CONDN_DESC_BASE<"sueq", Typestr, FGROpnd, Itin>, MipsR6Arch, ISA_MIPS32R6, HARDFLOAT; def CMP_SLT_#NAME : R6MMR6Rel, COP1_CMP_CONDN_FM, CMP_CONDN_DESC_BASE<"slt", Typestr, FGROpnd, Itin>, MipsR6Arch, ISA_MIPS32R6, HARDFLOAT; def CMP_SULT_#NAME : R6MMR6Rel, COP1_CMP_CONDN_FM, CMP_CONDN_DESC_BASE<"sult", Typestr, FGROpnd, Itin>, MipsR6Arch, ISA_MIPS32R6, HARDFLOAT; def CMP_SLE_#NAME : R6MMR6Rel, COP1_CMP_CONDN_FM, CMP_CONDN_DESC_BASE<"sle", Typestr, FGROpnd, Itin>, MipsR6Arch, ISA_MIPS32R6, HARDFLOAT; def CMP_SULE_#NAME : R6MMR6Rel, COP1_CMP_CONDN_FM, CMP_CONDN_DESC_BASE<"sule", Typestr, FGROpnd, Itin>, MipsR6Arch, ISA_MIPS32R6, HARDFLOAT; } } //===----------------------------------------------------------------------===// // // Instruction Descriptions // //===----------------------------------------------------------------------===// class PCREL_DESC_BASE : MipsR6Arch { dag OutOperandList = (outs GPROpnd:$rs); dag InOperandList = (ins ImmOpnd:$imm); string AsmString = !strconcat(instr_asm, "\t$rs, $imm"); list Pattern = []; InstrItinClass Itinerary = itin; } class ADDIUPC_DESC : PCREL_DESC_BASE<"addiupc", GPR32Opnd, simm19_lsl2, II_ADDIUPC>; class LWPC_DESC: PCREL_DESC_BASE<"lwpc", GPR32Opnd, simm19_lsl2, II_LWPC>; class LWUPC_DESC: PCREL_DESC_BASE<"lwupc", GPR32Opnd, simm19_lsl2, II_LWUPC>; class ALIGN_DESC_BASE : MipsR6Arch { dag OutOperandList = (outs GPROpnd:$rd); dag InOperandList = (ins GPROpnd:$rs, GPROpnd:$rt, ImmOpnd:$bp); string AsmString = !strconcat(instr_asm, "\t$rd, $rs, $rt, $bp"); list Pattern = []; InstrItinClass Itinerary = itin; } class ALIGN_DESC : ALIGN_DESC_BASE<"align", GPR32Opnd, uimm2, II_ALIGN>; class ALUIPC_DESC_BASE : MipsR6Arch { dag OutOperandList = (outs GPROpnd:$rs); dag InOperandList = (ins simm16:$imm); string AsmString = !strconcat(instr_asm, "\t$rs, $imm"); list Pattern = []; InstrItinClass Itinerary = itin; } class ALUIPC_DESC : ALUIPC_DESC_BASE<"aluipc", GPR32Opnd, II_ALUIPC>; class AUIPC_DESC : ALUIPC_DESC_BASE<"auipc", GPR32Opnd, II_AUIPC>; class AUI_DESC_BASE : MipsR6Arch { dag OutOperandList = (outs GPROpnd:$rt); dag InOperandList = (ins GPROpnd:$rs, uimm16:$imm); string AsmString = !strconcat(instr_asm, "\t$rt, $rs, $imm"); list Pattern = []; InstrItinClass Itinerary = itin; } class AUI_DESC : AUI_DESC_BASE<"aui", GPR32Opnd, II_AUI>; class BRANCH_DESC_BASE { bit isBranch = 1; bit isTerminator = 1; bit hasDelaySlot = 0; bit isCTI = 1; } class BC_DESC_BASE : BRANCH_DESC_BASE, MipsR6Arch { dag InOperandList = (ins opnd:$offset); dag OutOperandList = (outs); string AsmString = !strconcat(instr_asm, "\t$offset"); bit isBarrier = 1; InstrItinClass Itinerary = II_BC; bit isCTI = 1; } class CMP_BC_DESC_BASE : BRANCH_DESC_BASE, MipsR6Arch { dag InOperandList = (ins GPROpnd:$rs, GPROpnd:$rt, opnd:$offset); dag OutOperandList = (outs); string AsmString = !strconcat(instr_asm, "\t$rs, $rt, $offset"); list Defs = [AT]; InstrItinClass Itinerary = II_BCCC; bit hasForbiddenSlot = 1; bit isCTI = 1; } class CMP_CBR_EQNE_Z_DESC_BASE : BRANCH_DESC_BASE, MipsR6Arch { dag InOperandList = (ins GPROpnd:$rs, opnd:$offset); dag OutOperandList = (outs); string AsmString = !strconcat(instr_asm, "\t$rs, $offset"); list Defs = [AT]; InstrItinClass Itinerary = II_BCCZC; bit hasForbiddenSlot = 1; bit isCTI = 1; } class CMP_CBR_RT_Z_DESC_BASE : BRANCH_DESC_BASE, MipsR6Arch { dag InOperandList = (ins GPROpnd:$rt, opnd:$offset); dag OutOperandList = (outs); string AsmString = !strconcat(instr_asm, "\t$rt, $offset"); list Defs = [AT]; InstrItinClass Itinerary = II_BCCZC; bit hasForbiddenSlot = 1; bit isCTI = 1; } class BAL_DESC : BC_DESC_BASE<"bal", brtarget> { bit isCall = 1; bit hasDelaySlot = 1; list Defs = [RA]; bit isCTI = 1; } class BALC_DESC : BC_DESC_BASE<"balc", brtarget26> { bit isCall = 1; list Defs = [RA]; InstrItinClass Itinerary = II_BALC; bit isCTI = 1; } class BC_DESC : BC_DESC_BASE<"bc", brtarget26>; class BGEC_DESC : CMP_BC_DESC_BASE<"bgec", brtarget, GPR32Opnd>; class BGEUC_DESC : CMP_BC_DESC_BASE<"bgeuc", brtarget, GPR32Opnd>; class BEQC_DESC : CMP_BC_DESC_BASE<"beqc", brtarget, GPR32Opnd>; class BNEC_DESC : CMP_BC_DESC_BASE<"bnec", brtarget, GPR32Opnd>; class BLTC_DESC : CMP_BC_DESC_BASE<"bltc", brtarget, GPR32Opnd>; class BLTUC_DESC : CMP_BC_DESC_BASE<"bltuc", brtarget, GPR32Opnd>; class BLTZC_DESC : CMP_CBR_RT_Z_DESC_BASE<"bltzc", brtarget, GPR32Opnd>; class BGEZC_DESC : CMP_CBR_RT_Z_DESC_BASE<"bgezc", brtarget, GPR32Opnd>; class BLEZC_DESC : CMP_CBR_RT_Z_DESC_BASE<"blezc", brtarget, GPR32Opnd>; class BGTZC_DESC : CMP_CBR_RT_Z_DESC_BASE<"bgtzc", brtarget, GPR32Opnd>; class BEQZC_DESC : CMP_CBR_EQNE_Z_DESC_BASE<"beqzc", brtarget21, GPR32Opnd>; class BNEZC_DESC : CMP_CBR_EQNE_Z_DESC_BASE<"bnezc", brtarget21, GPR32Opnd>; class COP1_BCCZ_DESC_BASE : BRANCH_DESC_BASE { dag InOperandList = (ins FGR64Opnd:$ft, brtarget:$offset); dag OutOperandList = (outs); string AsmString = instr_asm; bit hasDelaySlot = 1; InstrItinClass Itinerary = II_BC1CCZ; } class BC1EQZ_DESC : COP1_BCCZ_DESC_BASE<"bc1eqz $ft, $offset">; class BC1NEZ_DESC : COP1_BCCZ_DESC_BASE<"bc1nez $ft, $offset">; class COP2_BCCZ_DESC_BASE : BRANCH_DESC_BASE { dag InOperandList = (ins COP2Opnd:$ct, brtarget:$offset); dag OutOperandList = (outs); string AsmString = instr_asm; bit hasDelaySlot = 1; bit isCTI = 1; InstrItinClass Itinerary = II_BC2CCZ; } class BC2EQZ_DESC : COP2_BCCZ_DESC_BASE<"bc2eqz $ct, $offset">; class BC2NEZ_DESC : COP2_BCCZ_DESC_BASE<"bc2nez $ct, $offset">; class BOVC_DESC : CMP_BC_DESC_BASE<"bovc", brtarget, GPR32Opnd>; class BNVC_DESC : CMP_BC_DESC_BASE<"bnvc", brtarget, GPR32Opnd>; class JMP_IDX_COMPACT_DESC_BASE : MipsR6Arch { dag InOperandList = (ins GPROpnd:$rt, opnd:$offset); string AsmString = !strconcat(opstr, "\t$rt, $offset"); list Pattern = []; bit hasDelaySlot = 0; InstrItinClass Itinerary = itin; bit isCTI = 1; bit isBranch = 1; bit isIndirectBranch = 1; } class JIALC_DESC : JMP_IDX_COMPACT_DESC_BASE<"jialc", calloffset16, GPR32Opnd, II_JIALC> { bit isCall = 1; list Defs = [RA]; } class JIC_DESC : JMP_IDX_COMPACT_DESC_BASE<"jic", jmpoffset16, GPR32Opnd, II_JIALC> { bit isBarrier = 1; bit isTerminator = 1; list Defs = [AT]; } class JR_HB_R6_DESC : JR_HB_DESC_BASE<"jr.hb", GPR32Opnd> { bit isBranch = 1; bit isIndirectBranch = 1; bit hasDelaySlot = 1; bit isTerminator=1; bit isBarrier=1; bit isCTI = 1; InstrItinClass Itinerary = II_JR_HB; } class BITSWAP_DESC_BASE : MipsR6Arch { dag OutOperandList = (outs GPROpnd:$rd); dag InOperandList = (ins GPROpnd:$rt); string AsmString = !strconcat(instr_asm, "\t$rd, $rt"); list Pattern = []; InstrItinClass Itinerary = itin; } class BITSWAP_DESC : BITSWAP_DESC_BASE<"bitswap", GPR32Opnd, II_BITSWAP>; class DIVMOD_DESC_BASE : MipsR6Arch { dag OutOperandList = (outs GPROpnd:$rd); dag InOperandList = (ins GPROpnd:$rs, GPROpnd:$rt); string AsmString = !strconcat(instr_asm, "\t$rd, $rs, $rt"); list Pattern = [(set GPROpnd:$rd, (Op GPROpnd:$rs, GPROpnd:$rt))]; InstrItinClass Itinerary = itin; // This instruction doesn't trap division by zero itself. We must insert // teq instructions as well. bit usesCustomInserter = 1; } class DVPEVP_DESC_BASE : MipsR6Arch { dag OutOperandList = (outs GPR32Opnd:$rt); dag InOperandList = (ins); string AsmString = !strconcat(instr_asm, "\t$rt"); list Pattern = []; InstrItinClass Itinerary = Itin; bit hasUnModeledSideEffects = 1; } class DVP_DESC : DVPEVP_DESC_BASE<"dvp", II_DVP>; class EVP_DESC : DVPEVP_DESC_BASE<"evp", II_EVP>; class DIV_DESC : DIVMOD_DESC_BASE<"div", GPR32Opnd, II_DIV, sdiv>; class DIVU_DESC : DIVMOD_DESC_BASE<"divu", GPR32Opnd, II_DIVU, udiv>; class MOD_DESC : DIVMOD_DESC_BASE<"mod", GPR32Opnd, II_MOD, srem>; class MODU_DESC : DIVMOD_DESC_BASE<"modu", GPR32Opnd, II_MODU, urem>; class BEQZALC_DESC : CMP_CBR_RT_Z_DESC_BASE<"beqzalc", brtarget, GPR32Opnd> { list Defs = [RA]; } class BGEZALC_DESC : CMP_CBR_RT_Z_DESC_BASE<"bgezalc", brtarget, GPR32Opnd> { list Defs = [RA]; } class BGTZALC_DESC : CMP_CBR_RT_Z_DESC_BASE<"bgtzalc", brtarget, GPR32Opnd> { list Defs = [RA]; } class BLEZALC_DESC : CMP_CBR_RT_Z_DESC_BASE<"blezalc", brtarget, GPR32Opnd> { list Defs = [RA]; } class BLTZALC_DESC : CMP_CBR_RT_Z_DESC_BASE<"bltzalc", brtarget, GPR32Opnd> { list Defs = [RA]; } class BNEZALC_DESC : CMP_CBR_RT_Z_DESC_BASE<"bnezalc", brtarget, GPR32Opnd> { list Defs = [RA]; } class MUL_R6_DESC_BASE : MipsR6Arch { dag OutOperandList = (outs GPROpnd:$rd); dag InOperandList = (ins GPROpnd:$rs, GPROpnd:$rt); string AsmString = !strconcat(instr_asm, "\t$rd, $rs, $rt"); list Pattern = [(set GPROpnd:$rd, (Op GPROpnd:$rs, GPROpnd:$rt))]; InstrItinClass Itinerary = itin; } class MUH_DESC : MUL_R6_DESC_BASE<"muh", GPR32Opnd, II_MUH, mulhs>; class MUHU_DESC : MUL_R6_DESC_BASE<"muhu", GPR32Opnd, II_MUHU, mulhu>; class MUL_R6_DESC : MUL_R6_DESC_BASE<"mul", GPR32Opnd, II_MUL, mul>; class MULU_DESC : MUL_R6_DESC_BASE<"mulu", GPR32Opnd, II_MULU>; class COP1_SEL_DESC_BASE { dag OutOperandList = (outs FGROpnd:$fd); dag InOperandList = (ins FGRCCOpnd:$fd_in, FGROpnd:$fs, FGROpnd:$ft); string AsmString = !strconcat(instr_asm, "\t$fd, $fs, $ft"); list Pattern = [(set FGROpnd:$fd, (select FGRCCOpnd:$fd_in, FGROpnd:$ft, FGROpnd:$fs))]; string Constraints = "$fd_in = $fd"; InstrItinClass Itinerary = itin; } class COP1_SEL_D_DESC_BASE { dag OutOperandList = (outs FGROpnd:$fd); dag InOperandList = (ins FGROpnd:$fd_in, FGROpnd:$fs, FGROpnd:$ft); string AsmString = !strconcat(instr_asm, "\t$fd, $fs, $ft"); list Pattern = [(set FGROpnd:$fd, (MipsFSelect FGROpnd:$fd_in, FGROpnd:$ft, FGROpnd:$fs))]; string Constraints = "$fd_in = $fd"; InstrItinClass Itinerary = itin; } class SEL_D_DESC : COP1_SEL_D_DESC_BASE<"sel.d", FGR64Opnd, II_SEL_D>, MipsR6Arch<"sel.d">; class SEL_S_DESC : COP1_SEL_DESC_BASE<"sel.s", FGR32Opnd, II_SEL_S>, MipsR6Arch<"sel.s">; class SELEQNE_Z_DESC_BASE : MipsR6Arch { dag OutOperandList = (outs GPROpnd:$rd); dag InOperandList = (ins GPROpnd:$rs, GPROpnd:$rt); string AsmString = !strconcat(instr_asm, "\t$rd, $rs, $rt"); list Pattern = []; InstrItinClass Itinerary = II_SELCCZ; } class SELEQZ_DESC : SELEQNE_Z_DESC_BASE<"seleqz", GPR32Opnd>; class SELNEZ_DESC : SELEQNE_Z_DESC_BASE<"selnez", GPR32Opnd>; class COP1_4R_DESC_BASE { dag OutOperandList = (outs FGROpnd:$fd); dag InOperandList = (ins FGROpnd:$fd_in, FGROpnd:$fs, FGROpnd:$ft); string AsmString = !strconcat(instr_asm, "\t$fd, $fs, $ft"); list Pattern = []; string Constraints = "$fd_in = $fd"; InstrItinClass Itinerary = itin; } class MADDF_S_DESC : COP1_4R_DESC_BASE<"maddf.s", FGR32Opnd, II_MADDF_S>; class MADDF_D_DESC : COP1_4R_DESC_BASE<"maddf.d", FGR64Opnd, II_MADDF_D>; class MSUBF_S_DESC : COP1_4R_DESC_BASE<"msubf.s", FGR32Opnd, II_MSUBF_S>; class MSUBF_D_DESC : COP1_4R_DESC_BASE<"msubf.d", FGR64Opnd, II_MSUBF_D>; class MAX_MIN_DESC_BASE { dag OutOperandList = (outs FGROpnd:$fd); dag InOperandList = (ins FGROpnd:$fs, FGROpnd:$ft); string AsmString = !strconcat(instr_asm, "\t$fd, $fs, $ft"); list Pattern = []; InstrItinClass Itinerary = itin; } class MAX_S_DESC : MAX_MIN_DESC_BASE<"max.s", FGR32Opnd, II_MAX_S>; class MAX_D_DESC : MAX_MIN_DESC_BASE<"max.d", FGR64Opnd, II_MAX_D>; class MIN_S_DESC : MAX_MIN_DESC_BASE<"min.s", FGR32Opnd, II_MIN_S>; class MIN_D_DESC : MAX_MIN_DESC_BASE<"min.d", FGR64Opnd, II_MIN_D>; class MAXA_S_DESC : MAX_MIN_DESC_BASE<"maxa.s", FGR32Opnd, II_MAX_S>; class MAXA_D_DESC : MAX_MIN_DESC_BASE<"maxa.d", FGR64Opnd, II_MAX_D>; class MINA_S_DESC : MAX_MIN_DESC_BASE<"mina.s", FGR32Opnd, II_MIN_D>; class MINA_D_DESC : MAX_MIN_DESC_BASE<"mina.d", FGR64Opnd, II_MIN_S>; class SELEQNEZ_DESC_BASE { dag OutOperandList = (outs FGROpnd:$fd); dag InOperandList = (ins FGROpnd:$fs, FGROpnd:$ft); string AsmString = !strconcat(instr_asm, "\t$fd, $fs, $ft"); list Pattern = []; InstrItinClass Itinerary = itin; } class SELEQZ_S_DESC : SELEQNEZ_DESC_BASE<"seleqz.s", FGR32Opnd, II_SELCCZ_S>, MipsR6Arch<"seleqz.s">; class SELEQZ_D_DESC : SELEQNEZ_DESC_BASE<"seleqz.d", FGR64Opnd, II_SELCCZ_D>, MipsR6Arch<"seleqz.d">; class SELNEZ_S_DESC : SELEQNEZ_DESC_BASE<"selnez.s", FGR32Opnd, II_SELCCZ_S>, MipsR6Arch<"selnez.s">; class SELNEZ_D_DESC : SELEQNEZ_DESC_BASE<"selnez.d", FGR64Opnd, II_SELCCZ_D>, MipsR6Arch<"selnez.d">; class CLASS_RINT_DESC_BASE { dag OutOperandList = (outs FGROpnd:$fd); dag InOperandList = (ins FGROpnd:$fs); string AsmString = !strconcat(instr_asm, "\t$fd, $fs"); list Pattern = []; InstrItinClass Itinerary = itin; } class RINT_S_DESC : CLASS_RINT_DESC_BASE<"rint.s", FGR32Opnd, II_RINT_S>; class RINT_D_DESC : CLASS_RINT_DESC_BASE<"rint.d", FGR64Opnd, II_RINT_D>; class CLASS_S_DESC : CLASS_RINT_DESC_BASE<"class.s", FGR32Opnd, II_CLASS_S>; class CLASS_D_DESC : CLASS_RINT_DESC_BASE<"class.d", FGR64Opnd, II_CLASS_D>; class CACHE_HINT_DESC : MipsR6Arch { dag OutOperandList = (outs); dag InOperandList = (ins MemOpnd:$addr, uimm5:$hint); string AsmString = !strconcat(instr_asm, "\t$hint, $addr"); list Pattern = []; string DecoderMethod = "DecodeCacheeOp_CacheOpR6"; InstrItinClass Itinerary = itin; } class CACHE_DESC : CACHE_HINT_DESC<"cache", mem_simm9, GPR32Opnd, II_CACHE>; class PREF_DESC : CACHE_HINT_DESC<"pref", mem_simm9, GPR32Opnd, II_PREF>; class COP2LD_DESC_BASE { dag OutOperandList = (outs COPOpnd:$rt); dag InOperandList = (ins mem_simm11:$addr); string AsmString = !strconcat(instr_asm, "\t$rt, $addr"); list Pattern = []; bit mayLoad = 1; string DecoderMethod = "DecodeFMemCop2R6"; InstrItinClass Itinerary = itin; } class LDC2_R6_DESC : COP2LD_DESC_BASE<"ldc2", COP2Opnd, II_LDC2>; class LWC2_R6_DESC : COP2LD_DESC_BASE<"lwc2", COP2Opnd, II_LWC2>; class COP2ST_DESC_BASE { dag OutOperandList = (outs); dag InOperandList = (ins COPOpnd:$rt, mem_simm11:$addr); string AsmString = !strconcat(instr_asm, "\t$rt, $addr"); list Pattern = []; bit mayStore = 1; string DecoderMethod = "DecodeFMemCop2R6"; InstrItinClass Itinerary = itin; } class SDC2_R6_DESC : COP2ST_DESC_BASE<"sdc2", COP2Opnd, II_SDC2>; class SWC2_R6_DESC : COP2ST_DESC_BASE<"swc2", COP2Opnd, II_SWC2>; class LSA_R6_DESC_BASE : MipsR6Arch { dag OutOperandList = (outs GPROpnd:$rd); dag InOperandList = (ins GPROpnd:$rs, GPROpnd:$rt, ImmOpnd:$imm2); string AsmString = !strconcat(instr_asm, "\t$rd, $rs, $rt, $imm2"); list Pattern = []; InstrItinClass Itinerary = itin; } class LSA_R6_DESC : LSA_R6_DESC_BASE<"lsa", GPR32Opnd, uimm2_plus1, II_LSA>; class LL_R6_DESC_BASE : MipsR6Arch { dag OutOperandList = (outs GPROpnd:$rt); dag InOperandList = (ins MemOpnd:$addr); string AsmString = !strconcat(instr_asm, "\t$rt, $addr"); list Pattern = []; bit mayLoad = 1; InstrItinClass Itinerary = itin; } class LL_R6_DESC : LL_R6_DESC_BASE<"ll", GPR32Opnd, mem_simm9, II_LL>; class SC_R6_DESC_BASE { dag OutOperandList = (outs GPROpnd:$dst); dag InOperandList = (ins GPROpnd:$rt, mem_simm9:$addr); string AsmString = !strconcat(instr_asm, "\t$rt, $addr"); list Pattern = []; bit mayStore = 1; string Constraints = "$rt = $dst"; InstrItinClass Itinerary = itin; } class SC_R6_DESC : SC_R6_DESC_BASE<"sc", GPR32Opnd, II_SC>; class CLO_CLZ_R6_DESC_BASE : MipsR6Arch { dag OutOperandList = (outs GPROpnd:$rd); dag InOperandList = (ins GPROpnd:$rs); string AsmString = !strconcat(instr_asm, "\t$rd, $rs"); InstrItinClass Itinerary = itin; } class CLO_R6_DESC_BASE : CLO_CLZ_R6_DESC_BASE { list Pattern = [(set GPROpnd:$rd, (ctlz (not GPROpnd:$rs)))]; } class CLZ_R6_DESC_BASE : CLO_CLZ_R6_DESC_BASE { list Pattern = [(set GPROpnd:$rd, (ctlz GPROpnd:$rs))]; } class CLO_R6_DESC : CLO_R6_DESC_BASE<"clo", GPR32Opnd, II_CLO>; class CLZ_R6_DESC : CLZ_R6_DESC_BASE<"clz", GPR32Opnd, II_CLZ>; class SDBBP_R6_DESC { dag OutOperandList = (outs); dag InOperandList = (ins uimm20:$code_); string AsmString = "sdbbp\t$code_"; list Pattern = []; bit isCTI = 1; InstrItinClass Itinerary = II_SDBBP; } class CRC_DESC_BASE : MipsR6Arch { dag OutOperandList = (outs GPROpnd:$rd); dag InOperandList = (ins GPROpnd:$rs, GPROpnd:$rt); string AsmString = !strconcat(instr_asm, "\t$rd, $rs, $rt"); list Pattern = []; InstrItinClass Itinerary = itin; } class CRC32B_DESC : CRC_DESC_BASE<"crc32b", GPR32Opnd, II_CRC32B>; class CRC32H_DESC : CRC_DESC_BASE<"crc32h", GPR32Opnd, II_CRC32H>; class CRC32W_DESC : CRC_DESC_BASE<"crc32w", GPR32Opnd, II_CRC32W>; class CRC32CB_DESC : CRC_DESC_BASE<"crc32cb", GPR32Opnd, II_CRC32CB>; class CRC32CH_DESC : CRC_DESC_BASE<"crc32ch", GPR32Opnd, II_CRC32CH>; class CRC32CW_DESC : CRC_DESC_BASE<"crc32cw", GPR32Opnd, II_CRC32CW>; class GINV_DESC_BASE : MipsR6Arch { dag OutOperandList = (outs); dag InOperandList = (ins GPROpnd:$rs, uimm2:$type_); string AsmString = !strconcat(instr_asm, "\t$rs, $type_"); list Pattern = []; InstrItinClass Itinerary = itin; bit hasSideEffects = 1; } class GINVI_DESC : GINV_DESC_BASE<"ginvi", GPR32Opnd, II_GINVI> { dag InOperandList = (ins GPR32Opnd:$rs); string AsmString = "ginvi\t$rs"; } class GINVT_DESC : GINV_DESC_BASE<"ginvt", GPR32Opnd, II_GINVT>; class SIGRIE_DESC { dag OutOperandList = (outs); dag InOperandList = (ins uimm16:$code_); string AsmString = "sigrie\t$code_"; list Pattern = []; InstrItinClass Itinerary = II_SIGRIE; } //===----------------------------------------------------------------------===// // // Instruction Definitions // //===----------------------------------------------------------------------===// def ADDIUPC : R6MMR6Rel, ADDIUPC_ENC, ADDIUPC_DESC, ISA_MIPS32R6; def ALIGN : R6MMR6Rel, ALIGN_ENC, ALIGN_DESC, ISA_MIPS32R6; def ALUIPC : R6MMR6Rel, ALUIPC_ENC, ALUIPC_DESC, ISA_MIPS32R6; def AUI : R6MMR6Rel, AUI_ENC, AUI_DESC, ISA_MIPS32R6; def AUIPC : R6MMR6Rel, AUIPC_ENC, AUIPC_DESC, ISA_MIPS32R6; def BAL : BAL_ENC, BAL_DESC, ISA_MIPS32R6; def BALC : R6MMR6Rel, BALC_ENC, BALC_DESC, ISA_MIPS32R6; let AdditionalPredicates = [NotInMicroMips] in { def BC1EQZ : BC1EQZ_ENC, BC1EQZ_DESC, ISA_MIPS32R6, HARDFLOAT; def BC1NEZ : BC1NEZ_ENC, BC1NEZ_DESC, ISA_MIPS32R6, HARDFLOAT; def BC2EQZ : BC2EQZ_ENC, BC2EQZ_DESC, ISA_MIPS32R6; def BC2NEZ : BC2NEZ_ENC, BC2NEZ_DESC, ISA_MIPS32R6; def BC : R6MMR6Rel, BC_ENC, BC_DESC, ISA_MIPS32R6; def BEQC : R6MMR6Rel, BEQC_ENC, BEQC_DESC, ISA_MIPS32R6; def BEQZALC : R6MMR6Rel, BEQZALC_ENC, BEQZALC_DESC, ISA_MIPS32R6; def BEQZC : R6MMR6Rel, BEQZC_ENC, BEQZC_DESC, ISA_MIPS32R6; def BGEC : R6MMR6Rel, BGEC_ENC, BGEC_DESC, ISA_MIPS32R6; def BGEUC : R6MMR6Rel, BGEUC_ENC, BGEUC_DESC, ISA_MIPS32R6; def BGEZALC : R6MMR6Rel, BGEZALC_ENC, BGEZALC_DESC, ISA_MIPS32R6; def BGEZC : R6MMR6Rel, BGEZC_ENC, BGEZC_DESC, ISA_MIPS32R6; def BGTZALC : R6MMR6Rel, BGTZALC_ENC, BGTZALC_DESC, ISA_MIPS32R6; def BGTZC : R6MMR6Rel, BGTZC_ENC, BGTZC_DESC, ISA_MIPS32R6; } def BITSWAP : R6MMR6Rel, BITSWAP_ENC, BITSWAP_DESC, ISA_MIPS32R6; let AdditionalPredicates = [NotInMicroMips] in { def BLEZALC : R6MMR6Rel, BLEZALC_ENC, BLEZALC_DESC, ISA_MIPS32R6; def BLEZC : R6MMR6Rel, BLEZC_ENC, BLEZC_DESC, ISA_MIPS32R6; def BLTC : R6MMR6Rel, BLTC_ENC, BLTC_DESC, ISA_MIPS32R6; def BLTUC : R6MMR6Rel, BLTUC_ENC, BLTUC_DESC, ISA_MIPS32R6; def BLTZALC : R6MMR6Rel, BLTZALC_ENC, BLTZALC_DESC, ISA_MIPS32R6; def BLTZC : R6MMR6Rel, BLTZC_ENC, BLTZC_DESC, ISA_MIPS32R6; def BNEC : R6MMR6Rel, BNEC_ENC, BNEC_DESC, ISA_MIPS32R6; def BNEZALC : R6MMR6Rel, BNEZALC_ENC, BNEZALC_DESC, ISA_MIPS32R6; def BNEZC : R6MMR6Rel, BNEZC_ENC, BNEZC_DESC, ISA_MIPS32R6; def BNVC : R6MMR6Rel, BNVC_ENC, BNVC_DESC, ISA_MIPS32R6; def BOVC : R6MMR6Rel, BOVC_ENC, BOVC_DESC, ISA_MIPS32R6; def CACHE_R6 : R6MMR6Rel, CACHE_ENC, CACHE_DESC, ISA_MIPS32R6; def CLASS_D : CLASS_D_ENC, CLASS_D_DESC, ISA_MIPS32R6, HARDFLOAT; def CLASS_S : CLASS_S_ENC, CLASS_S_DESC, ISA_MIPS32R6, HARDFLOAT; } def CLO_R6 : R6MMR6Rel, CLO_R6_ENC, CLO_R6_DESC, ISA_MIPS32R6; def CLZ_R6 : R6MMR6Rel, CLZ_R6_ENC, CLZ_R6_DESC, ISA_MIPS32R6; defm S : CMP_CC_M; defm D : CMP_CC_M; let AdditionalPredicates = [NotInMicroMips] in { def DIV : R6MMR6Rel, DIV_ENC, DIV_DESC, ISA_MIPS32R6; def DIVU : R6MMR6Rel, DIVU_ENC, DIVU_DESC, ISA_MIPS32R6; } def DVP : R6MMR6Rel, DVP_ENC, DVP_DESC, ISA_MIPS32R6; def EVP : R6MMR6Rel, EVP_ENC, EVP_DESC, ISA_MIPS32R6; def JIALC : R6MMR6Rel, JIALC_ENC, JIALC_DESC, ISA_MIPS32R6; def JIC : R6MMR6Rel, JIC_ENC, JIC_DESC, ISA_MIPS32R6; def JR_HB_R6 : JR_HB_R6_ENC, JR_HB_R6_DESC, ISA_MIPS32R6; let AdditionalPredicates = [NotInMicroMips] in { def LDC2_R6 : LDC2_R6_ENC, LDC2_R6_DESC, ISA_MIPS32R6; def LL_R6 : LL_R6_ENC, LL_R6_DESC, PTR_32, ISA_MIPS32R6; } def LSA_R6 : R6MMR6Rel, LSA_R6_ENC, LSA_R6_DESC, ISA_MIPS32R6; let AdditionalPredicates = [NotInMicroMips] in { def LWC2_R6 : LWC2_R6_ENC, LWC2_R6_DESC, ISA_MIPS32R6; } def LWPC : R6MMR6Rel, LWPC_ENC, LWPC_DESC, ISA_MIPS32R6; let AdditionalPredicates = [NotInMicroMips] in { def LWUPC : R6MMR6Rel, LWUPC_ENC, LWUPC_DESC, ISA_MIPS32R6; def MADDF_S : MADDF_S_ENC, MADDF_S_DESC, ISA_MIPS32R6, HARDFLOAT; def MADDF_D : MADDF_D_ENC, MADDF_D_DESC, ISA_MIPS32R6, HARDFLOAT; def MAXA_D : MAXA_D_ENC, MAXA_D_DESC, ISA_MIPS32R6, HARDFLOAT; def MAXA_S : MAXA_S_ENC, MAXA_S_DESC, ISA_MIPS32R6, HARDFLOAT; def MAX_D : MAX_D_ENC, MAX_D_DESC, ISA_MIPS32R6, HARDFLOAT; def MAX_S : MAX_S_ENC, MAX_S_DESC, ISA_MIPS32R6, HARDFLOAT; def MINA_D : MINA_D_ENC, MINA_D_DESC, ISA_MIPS32R6, HARDFLOAT; def MINA_S : MINA_S_ENC, MINA_S_DESC, ISA_MIPS32R6, HARDFLOAT; def MIN_D : MIN_D_ENC, MIN_D_DESC, ISA_MIPS32R6, HARDFLOAT; def MIN_S : MIN_S_ENC, MIN_S_DESC, ISA_MIPS32R6, HARDFLOAT; def MOD : R6MMR6Rel, MOD_ENC, MOD_DESC, ISA_MIPS32R6; def MODU : R6MMR6Rel, MODU_ENC, MODU_DESC, ISA_MIPS32R6; def MSUBF_S : MSUBF_S_ENC, MSUBF_S_DESC, ISA_MIPS32R6, HARDFLOAT; def MSUBF_D : MSUBF_D_ENC, MSUBF_D_DESC, ISA_MIPS32R6, HARDFLOAT; def MUH : R6MMR6Rel, MUH_ENC, MUH_DESC, ISA_MIPS32R6; def MUHU : R6MMR6Rel, MUHU_ENC, MUHU_DESC, ISA_MIPS32R6; def MUL_R6 : R6MMR6Rel, MUL_R6_ENC, MUL_R6_DESC, ISA_MIPS32R6; def MULU : R6MMR6Rel, MULU_ENC, MULU_DESC, ISA_MIPS32R6; } def NAL; // BAL with rd=0 let AdditionalPredicates = [NotInMicroMips] in { def PREF_R6 : R6MMR6Rel, PREF_ENC, PREF_DESC, ISA_MIPS32R6; def RINT_D : RINT_D_ENC, RINT_D_DESC, ISA_MIPS32R6, HARDFLOAT; def RINT_S : RINT_S_ENC, RINT_S_DESC, ISA_MIPS32R6, HARDFLOAT; def SC_R6 : SC_R6_ENC, SC_R6_DESC, PTR_32, ISA_MIPS32R6; def SDBBP_R6 : SDBBP_R6_ENC, SDBBP_R6_DESC, ISA_MIPS32R6; def SELEQZ : R6MMR6Rel, SELEQZ_ENC, SELEQZ_DESC, ISA_MIPS32R6, GPR_32; def SELNEZ : R6MMR6Rel, SELNEZ_ENC, SELNEZ_DESC, ISA_MIPS32R6, GPR_32; def SELEQZ_D : R6MMR6Rel, SELEQZ_D_ENC, SELEQZ_D_DESC, ISA_MIPS32R6, HARDFLOAT; def SELEQZ_S : R6MMR6Rel, SELEQZ_S_ENC, SELEQZ_S_DESC, ISA_MIPS32R6, HARDFLOAT; def SELNEZ_D : R6MMR6Rel, SELNEZ_D_ENC, SELNEZ_D_DESC, ISA_MIPS32R6, HARDFLOAT; def SELNEZ_S : R6MMR6Rel, SELNEZ_S_ENC, SELNEZ_S_DESC, ISA_MIPS32R6, HARDFLOAT; def SEL_D : R6MMR6Rel, SEL_D_ENC, SEL_D_DESC, ISA_MIPS32R6, HARDFLOAT; def SEL_S : R6MMR6Rel, SEL_S_ENC, SEL_S_DESC, ISA_MIPS32R6, HARDFLOAT; def SDC2_R6 : SDC2_R6_ENC, SDC2_R6_DESC, ISA_MIPS32R6; def SWC2_R6 : SWC2_R6_ENC, SWC2_R6_DESC, ISA_MIPS32R6; def SIGRIE : SIGRIE_ENC, SIGRIE_DESC, ISA_MIPS32R6; } let AdditionalPredicates = [NotInMicroMips] in { def CRC32B : R6MMR6Rel, CRC32B_ENC, CRC32B_DESC, ISA_MIPS32R6, ASE_CRC; def CRC32H : R6MMR6Rel, CRC32H_ENC, CRC32H_DESC, ISA_MIPS32R6, ASE_CRC; def CRC32W : R6MMR6Rel, CRC32W_ENC, CRC32W_DESC, ISA_MIPS32R6, ASE_CRC; def CRC32CB : R6MMR6Rel, CRC32CB_ENC, CRC32CB_DESC, ISA_MIPS32R6, ASE_CRC; def CRC32CH : R6MMR6Rel, CRC32CH_ENC, CRC32CH_DESC, ISA_MIPS32R6, ASE_CRC; def CRC32CW : R6MMR6Rel, CRC32CW_ENC, CRC32CW_DESC, ISA_MIPS32R6, ASE_CRC; } let AdditionalPredicates = [NotInMicroMips] in { def GINVI : R6MMR6Rel, GINVI_ENC, GINVI_DESC, ISA_MIPS32R6, ASE_GINV; def GINVT : R6MMR6Rel, GINVT_ENC, GINVT_DESC, ISA_MIPS32R6, ASE_GINV; } //===----------------------------------------------------------------------===// // // Instruction Aliases // //===----------------------------------------------------------------------===// def : MipsInstAlias<"dvp", (DVP ZERO), 0>, ISA_MIPS32R6; def : MipsInstAlias<"evp", (EVP ZERO), 0>, ISA_MIPS32R6; let AdditionalPredicates = [NotInMicroMips] in { def : MipsInstAlias<"sdbbp", (SDBBP_R6 0)>, ISA_MIPS32R6; def : MipsInstAlias<"sigrie", (SIGRIE 0)>, ISA_MIPS32R6; def : MipsInstAlias<"jr $rs", (JALR ZERO, GPR32Opnd:$rs), 1>, ISA_MIPS32R6, GPR_32; } def : MipsInstAlias<"jrc $rs", (JIC GPR32Opnd:$rs, 0), 1>, ISA_MIPS32R6, GPR_32; let AdditionalPredicates = [NotInMicroMips] in { def : MipsInstAlias<"jalrc $rs", (JIALC GPR32Opnd:$rs, 0), 1>, ISA_MIPS32R6, GPR_32; } def : MipsInstAlias<"div $rs, $rt", (DIV GPR32Opnd:$rs, GPR32Opnd:$rs, GPR32Opnd:$rt)>, ISA_MIPS32R6; def : MipsInstAlias<"divu $rs, $rt", (DIVU GPR32Opnd:$rs, GPR32Opnd:$rs, GPR32Opnd:$rt)>, ISA_MIPS32R6; def : MipsInstAlias<"lapc $rd, $imm", (ADDIUPC GPR32Opnd:$rd, simm19_lsl2:$imm)>, ISA_MIPS32R6; //===----------------------------------------------------------------------===// // // Patterns and Pseudo Instructions // //===----------------------------------------------------------------------===// // comparisons supported via another comparison multiclass Cmp_Pats { def : MipsPat<(setone VT:$lhs, VT:$rhs), (NOROp (!cast("CMP_UEQ_"#NAME) VT:$lhs, VT:$rhs), ZEROReg)>; def : MipsPat<(seto VT:$lhs, VT:$rhs), (NOROp (!cast("CMP_UN_"#NAME) VT:$lhs, VT:$rhs), ZEROReg)>; def : MipsPat<(setune VT:$lhs, VT:$rhs), (NOROp (!cast("CMP_EQ_"#NAME) VT:$lhs, VT:$rhs), ZEROReg)>; def : MipsPat<(seteq VT:$lhs, VT:$rhs), (!cast("CMP_EQ_"#NAME) VT:$lhs, VT:$rhs)>; def : MipsPat<(setgt VT:$lhs, VT:$rhs), (!cast("CMP_LE_"#NAME) VT:$rhs, VT:$lhs)>; def : MipsPat<(setge VT:$lhs, VT:$rhs), (!cast("CMP_LT_"#NAME) VT:$rhs, VT:$lhs)>; def : MipsPat<(setlt VT:$lhs, VT:$rhs), (!cast("CMP_LT_"#NAME) VT:$lhs, VT:$rhs)>; def : MipsPat<(setle VT:$lhs, VT:$rhs), (!cast("CMP_LE_"#NAME) VT:$lhs, VT:$rhs)>; def : MipsPat<(setne VT:$lhs, VT:$rhs), (NOROp (!cast("CMP_EQ_"#NAME) VT:$lhs, VT:$rhs), ZEROReg)>; } let AdditionalPredicates = [NotInMicroMips] in { defm S : Cmp_Pats, ISA_MIPS32R6; defm D : Cmp_Pats, ISA_MIPS32R6; } // i32 selects multiclass SelectInt_Pats { // reg, immz def : MipsPat<(select (Opg (seteq RC:$cond, immz)), RC:$t, RC:$f), (OROp (SELEQZOp RC:$t, RC:$cond), (SELNEZOp RC:$f, RC:$cond))>; def : MipsPat<(select (Opg (setne RC:$cond, immz)), RC:$t, RC:$f), (OROp (SELNEZOp RC:$t, RC:$cond), (SELEQZOp RC:$f, RC:$cond))>; // reg, immZExt16[_64] def : MipsPat<(select (Opg (seteq RC:$cond, imm_type:$imm)), RC:$t, RC:$f), (OROp (SELEQZOp RC:$t, (XORiOp RC:$cond, imm_type:$imm)), (SELNEZOp RC:$f, (XORiOp RC:$cond, imm_type:$imm)))>; def : MipsPat<(select (Opg (setne RC:$cond, imm_type:$imm)), RC:$t, RC:$f), (OROp (SELNEZOp RC:$t, (XORiOp RC:$cond, imm_type:$imm)), (SELEQZOp RC:$f, (XORiOp RC:$cond, imm_type:$imm)))>; // reg, immSExt16Plus1 def : MipsPat<(select (Opg (setgt RC:$cond, immSExt16Plus1:$imm)), RC:$t, RC:$f), (OROp (SELEQZOp RC:$t, (SLTiOp RC:$cond, (Plus1 imm:$imm))), (SELNEZOp RC:$f, (SLTiOp RC:$cond, (Plus1 imm:$imm))))>; def : MipsPat<(select (Opg (setugt RC:$cond, immSExt16Plus1:$imm)), RC:$t, RC:$f), (OROp (SELEQZOp RC:$t, (SLTiuOp RC:$cond, (Plus1 imm:$imm))), (SELNEZOp RC:$f, (SLTiuOp RC:$cond, (Plus1 imm:$imm))))>; def : MipsPat<(select (Opg (seteq RC:$cond, immz)), RC:$t, immz), (SELEQZOp RC:$t, RC:$cond)>; def : MipsPat<(select (Opg (setne RC:$cond, immz)), RC:$t, immz), (SELNEZOp RC:$t, RC:$cond)>; def : MipsPat<(select (Opg (seteq RC:$cond, immz)), immz, RC:$f), (SELNEZOp RC:$f, RC:$cond)>; def : MipsPat<(select (Opg (setne RC:$cond, immz)), immz, RC:$f), (SELEQZOp RC:$f, RC:$cond)>; } let AdditionalPredicates = [NotInMicroMips] in { defm : SelectInt_Pats, ISA_MIPS32R6; def : MipsPat<(select i32:$cond, i32:$t, i32:$f), (OR (SELNEZ i32:$t, i32:$cond), (SELEQZ i32:$f, i32:$cond))>, ISA_MIPS32R6; def : MipsPat<(select i32:$cond, i32:$t, immz), (SELNEZ i32:$t, i32:$cond)>, ISA_MIPS32R6; def : MipsPat<(select i32:$cond, immz, i32:$f), (SELEQZ i32:$f, i32:$cond)>, ISA_MIPS32R6; } // Pseudo instructions let isCall = 1, isTerminator = 1, isReturn = 1, isBarrier = 1, hasDelaySlot = 1, - hasExtraSrcRegAllocReq = 1, isCTI = 1, Defs = [AT] in { + hasExtraSrcRegAllocReq = 1, isCTI = 1, Defs = [AT], hasPostISelHook = 1 in { class TailCallRegR6 : PseudoSE<(outs), (ins RO:$rs), [(MipsTailCall RO:$rs)], II_JR>, PseudoInstExpansion<(JumpInst RT:$rt, RO:$rs)>; } class PseudoIndirectBranchBaseR6 : MipsPseudo<(outs), (ins RO:$rs), [(brind RO:$rs)], II_IndirectBranchPseudo>, PseudoInstExpansion<(JumpInst RT:$rt, RO:$rs)> { let isTerminator=1; let isBarrier=1; let hasDelaySlot = 1; let isBranch = 1; let isIndirectBranch = 1; bit isCTI = 1; } let AdditionalPredicates = [NotInMips16Mode, NotInMicroMips, NoIndirectJumpGuards] in { def TAILCALLR6REG : TailCallRegR6, ISA_MIPS32R6; def PseudoIndirectBranchR6 : PseudoIndirectBranchBaseR6, ISA_MIPS32R6; } let AdditionalPredicates = [NotInMips16Mode, NotInMicroMips, UseIndirectJumpsHazard] in { def TAILCALLHBR6REG : TailCallReg, ISA_MIPS32R6; def PseudoIndrectHazardBranchR6 : PseudoIndirectBranchBase, ISA_MIPS32R6; } Index: vendor/llvm/dist-release_80/lib/Target/Mips/MipsAsmPrinter.cpp =================================================================== --- vendor/llvm/dist-release_80/lib/Target/Mips/MipsAsmPrinter.cpp (revision 343793) +++ vendor/llvm/dist-release_80/lib/Target/Mips/MipsAsmPrinter.cpp (revision 343794) @@ -1,1263 +1,1304 @@ //===- MipsAsmPrinter.cpp - Mips LLVM Assembly Printer --------------------===// // // The LLVM Compiler Infrastructure // // This file is distributed under the University of Illinois Open Source // License. See LICENSE.TXT for details. // //===----------------------------------------------------------------------===// // // This file contains a printer that converts from our internal representation // of machine-dependent LLVM code to GAS-format MIPS assembly language. // //===----------------------------------------------------------------------===// #include "MipsAsmPrinter.h" #include "InstPrinter/MipsInstPrinter.h" #include "MCTargetDesc/MipsABIInfo.h" #include "MCTargetDesc/MipsBaseInfo.h" #include "MCTargetDesc/MipsMCNaCl.h" #include "MCTargetDesc/MipsMCTargetDesc.h" #include "Mips.h" #include "MipsMCInstLower.h" #include "MipsMachineFunction.h" #include "MipsSubtarget.h" #include "MipsTargetMachine.h" #include "MipsTargetStreamer.h" #include "llvm/ADT/SmallString.h" #include "llvm/ADT/StringRef.h" #include "llvm/ADT/Triple.h" #include "llvm/ADT/Twine.h" #include "llvm/BinaryFormat/ELF.h" #include "llvm/CodeGen/MachineBasicBlock.h" #include "llvm/CodeGen/MachineConstantPool.h" #include "llvm/CodeGen/MachineFrameInfo.h" #include "llvm/CodeGen/MachineFunction.h" #include "llvm/CodeGen/MachineInstr.h" #include "llvm/CodeGen/MachineJumpTableInfo.h" #include "llvm/CodeGen/MachineOperand.h" #include "llvm/CodeGen/TargetRegisterInfo.h" #include "llvm/CodeGen/TargetSubtargetInfo.h" #include "llvm/IR/Attributes.h" #include "llvm/IR/BasicBlock.h" #include "llvm/IR/DataLayout.h" #include "llvm/IR/Function.h" #include "llvm/IR/InlineAsm.h" #include "llvm/IR/Instructions.h" #include "llvm/MC/MCContext.h" #include "llvm/MC/MCExpr.h" #include "llvm/MC/MCInst.h" #include "llvm/MC/MCInstBuilder.h" #include "llvm/MC/MCObjectFileInfo.h" #include "llvm/MC/MCSectionELF.h" #include "llvm/MC/MCSymbol.h" #include "llvm/MC/MCSymbolELF.h" #include "llvm/Support/Casting.h" #include "llvm/Support/ErrorHandling.h" #include "llvm/Support/TargetRegistry.h" #include "llvm/Support/raw_ostream.h" #include "llvm/Target/TargetMachine.h" #include #include #include #include #include #include using namespace llvm; #define DEBUG_TYPE "mips-asm-printer" +extern cl::opt EmitJalrReloc; + MipsTargetStreamer &MipsAsmPrinter::getTargetStreamer() const { return static_cast(*OutStreamer->getTargetStreamer()); } bool MipsAsmPrinter::runOnMachineFunction(MachineFunction &MF) { Subtarget = &MF.getSubtarget(); MipsFI = MF.getInfo(); if (Subtarget->inMips16Mode()) for (std::map< const char *, const Mips16HardFloatInfo::FuncSignature *>::const_iterator it = MipsFI->StubsNeeded.begin(); it != MipsFI->StubsNeeded.end(); ++it) { const char *Symbol = it->first; const Mips16HardFloatInfo::FuncSignature *Signature = it->second; if (StubsNeeded.find(Symbol) == StubsNeeded.end()) StubsNeeded[Symbol] = Signature; } MCP = MF.getConstantPool(); // In NaCl, all indirect jump targets must be aligned to bundle size. if (Subtarget->isTargetNaCl()) NaClAlignIndirectJumpTargets(MF); AsmPrinter::runOnMachineFunction(MF); emitXRayTable(); return true; } bool MipsAsmPrinter::lowerOperand(const MachineOperand &MO, MCOperand &MCOp) { MCOp = MCInstLowering.LowerOperand(MO); return MCOp.isValid(); } #include "MipsGenMCPseudoLowering.inc" // Lower PseudoReturn/PseudoIndirectBranch/PseudoIndirectBranch64 to JR, JR_MM, // JALR, or JALR64 as appropriate for the target. void MipsAsmPrinter::emitPseudoIndirectBranch(MCStreamer &OutStreamer, const MachineInstr *MI) { bool HasLinkReg = false; bool InMicroMipsMode = Subtarget->inMicroMipsMode(); MCInst TmpInst0; if (Subtarget->hasMips64r6()) { // MIPS64r6 should use (JALR64 ZERO_64, $rs) TmpInst0.setOpcode(Mips::JALR64); HasLinkReg = true; } else if (Subtarget->hasMips32r6()) { // MIPS32r6 should use (JALR ZERO, $rs) if (InMicroMipsMode) TmpInst0.setOpcode(Mips::JRC16_MMR6); else { TmpInst0.setOpcode(Mips::JALR); HasLinkReg = true; } } else if (Subtarget->inMicroMipsMode()) // microMIPS should use (JR_MM $rs) TmpInst0.setOpcode(Mips::JR_MM); else { // Everything else should use (JR $rs) TmpInst0.setOpcode(Mips::JR); } MCOperand MCOp; if (HasLinkReg) { unsigned ZeroReg = Subtarget->isGP64bit() ? Mips::ZERO_64 : Mips::ZERO; TmpInst0.addOperand(MCOperand::createReg(ZeroReg)); } lowerOperand(MI->getOperand(0), MCOp); TmpInst0.addOperand(MCOp); EmitToStreamer(OutStreamer, TmpInst0); } +// If there is an MO_JALR operand, insert: +// +// .reloc tmplabel, R_{MICRO}MIPS_JALR, symbol +// tmplabel: +// +// This is an optimization hint for the linker which may then replace +// an indirect call with a direct branch. +static void emitDirectiveRelocJalr(const MachineInstr &MI, + MCContext &OutContext, + TargetMachine &TM, + MCStreamer &OutStreamer, + const MipsSubtarget &Subtarget) { + for (unsigned int I = MI.getDesc().getNumOperands(), E = MI.getNumOperands(); + I < E; ++I) { + MachineOperand MO = MI.getOperand(I); + if (MO.isMCSymbol() && (MO.getTargetFlags() & MipsII::MO_JALR)) { + MCSymbol *Callee = MO.getMCSymbol(); + if (Callee && !Callee->getName().empty()) { + MCSymbol *OffsetLabel = OutContext.createTempSymbol(); + const MCExpr *OffsetExpr = + MCSymbolRefExpr::create(OffsetLabel, OutContext); + const MCExpr *CaleeExpr = + MCSymbolRefExpr::create(Callee, OutContext); + OutStreamer.EmitRelocDirective + (*OffsetExpr, + Subtarget.inMicroMipsMode() ? "R_MICROMIPS_JALR" : "R_MIPS_JALR", + CaleeExpr, SMLoc(), *TM.getMCSubtargetInfo()); + OutStreamer.EmitLabel(OffsetLabel); + return; + } + } + } +} + void MipsAsmPrinter::EmitInstruction(const MachineInstr *MI) { MipsTargetStreamer &TS = getTargetStreamer(); unsigned Opc = MI->getOpcode(); TS.forbidModuleDirective(); if (MI->isDebugValue()) { SmallString<128> Str; raw_svector_ostream OS(Str); PrintDebugValueComment(MI, OS); return; } if (MI->isDebugLabel()) return; // If we just ended a constant pool, mark it as such. if (InConstantPool && Opc != Mips::CONSTPOOL_ENTRY) { OutStreamer->EmitDataRegion(MCDR_DataRegionEnd); InConstantPool = false; } if (Opc == Mips::CONSTPOOL_ENTRY) { // CONSTPOOL_ENTRY - This instruction represents a floating // constant pool in the function. The first operand is the ID# // for this instruction, the second is the index into the // MachineConstantPool that this is, the third is the size in // bytes of this constant pool entry. // The required alignment is specified on the basic block holding this MI. // unsigned LabelId = (unsigned)MI->getOperand(0).getImm(); unsigned CPIdx = (unsigned)MI->getOperand(1).getIndex(); // If this is the first entry of the pool, mark it. if (!InConstantPool) { OutStreamer->EmitDataRegion(MCDR_DataRegion); InConstantPool = true; } OutStreamer->EmitLabel(GetCPISymbol(LabelId)); const MachineConstantPoolEntry &MCPE = MCP->getConstants()[CPIdx]; if (MCPE.isMachineConstantPoolEntry()) EmitMachineConstantPoolValue(MCPE.Val.MachineCPVal); else EmitGlobalConstant(MF->getDataLayout(), MCPE.Val.ConstVal); return; } switch (Opc) { case Mips::PATCHABLE_FUNCTION_ENTER: LowerPATCHABLE_FUNCTION_ENTER(*MI); return; case Mips::PATCHABLE_FUNCTION_EXIT: LowerPATCHABLE_FUNCTION_EXIT(*MI); return; case Mips::PATCHABLE_TAIL_CALL: LowerPATCHABLE_TAIL_CALL(*MI); return; + } + + if (EmitJalrReloc && + (MI->isReturn() || MI->isCall() || MI->isIndirectBranch())) { + emitDirectiveRelocJalr(*MI, OutContext, TM, *OutStreamer, *Subtarget); } MachineBasicBlock::const_instr_iterator I = MI->getIterator(); MachineBasicBlock::const_instr_iterator E = MI->getParent()->instr_end(); do { // Do any auto-generated pseudo lowerings. if (emitPseudoExpansionLowering(*OutStreamer, &*I)) continue; if (I->getOpcode() == Mips::PseudoReturn || I->getOpcode() == Mips::PseudoReturn64 || I->getOpcode() == Mips::PseudoIndirectBranch || I->getOpcode() == Mips::PseudoIndirectBranch64 || I->getOpcode() == Mips::TAILCALLREG || I->getOpcode() == Mips::TAILCALLREG64) { emitPseudoIndirectBranch(*OutStreamer, &*I); continue; } // The inMips16Mode() test is not permanent. // Some instructions are marked as pseudo right now which // would make the test fail for the wrong reason but // that will be fixed soon. We need this here because we are // removing another test for this situation downstream in the // callchain. // if (I->isPseudo() && !Subtarget->inMips16Mode() && !isLongBranchPseudo(I->getOpcode())) llvm_unreachable("Pseudo opcode found in EmitInstruction()"); MCInst TmpInst0; MCInstLowering.Lower(&*I, TmpInst0); EmitToStreamer(*OutStreamer, TmpInst0); } while ((++I != E) && I->isInsideBundle()); // Delay slot check } //===----------------------------------------------------------------------===// // // Mips Asm Directives // // -- Frame directive "frame Stackpointer, Stacksize, RARegister" // Describe the stack frame. // // -- Mask directives "(f)mask bitmask, offset" // Tells the assembler which registers are saved and where. // bitmask - contain a little endian bitset indicating which registers are // saved on function prologue (e.g. with a 0x80000000 mask, the // assembler knows the register 31 (RA) is saved at prologue. // offset - the position before stack pointer subtraction indicating where // the first saved register on prologue is located. (e.g. with a // // Consider the following function prologue: // // .frame $fp,48,$ra // .mask 0xc0000000,-8 // addiu $sp, $sp, -48 // sw $ra, 40($sp) // sw $fp, 36($sp) // // With a 0xc0000000 mask, the assembler knows the register 31 (RA) and // 30 (FP) are saved at prologue. As the save order on prologue is from // left to right, RA is saved first. A -8 offset means that after the // stack pointer subtration, the first register in the mask (RA) will be // saved at address 48-8=40. // //===----------------------------------------------------------------------===// //===----------------------------------------------------------------------===// // Mask directives //===----------------------------------------------------------------------===// // Create a bitmask with all callee saved registers for CPU or Floating Point // registers. For CPU registers consider RA, GP and FP for saving if necessary. void MipsAsmPrinter::printSavedRegsBitmask() { // CPU and FPU Saved Registers Bitmasks unsigned CPUBitmask = 0, FPUBitmask = 0; int CPUTopSavedRegOff, FPUTopSavedRegOff; // Set the CPU and FPU Bitmasks const MachineFrameInfo &MFI = MF->getFrameInfo(); const TargetRegisterInfo *TRI = MF->getSubtarget().getRegisterInfo(); const std::vector &CSI = MFI.getCalleeSavedInfo(); // size of stack area to which FP callee-saved regs are saved. unsigned CPURegSize = TRI->getRegSizeInBits(Mips::GPR32RegClass) / 8; unsigned FGR32RegSize = TRI->getRegSizeInBits(Mips::FGR32RegClass) / 8; unsigned AFGR64RegSize = TRI->getRegSizeInBits(Mips::AFGR64RegClass) / 8; bool HasAFGR64Reg = false; unsigned CSFPRegsSize = 0; for (const auto &I : CSI) { unsigned Reg = I.getReg(); unsigned RegNum = TRI->getEncodingValue(Reg); // If it's a floating point register, set the FPU Bitmask. // If it's a general purpose register, set the CPU Bitmask. if (Mips::FGR32RegClass.contains(Reg)) { FPUBitmask |= (1 << RegNum); CSFPRegsSize += FGR32RegSize; } else if (Mips::AFGR64RegClass.contains(Reg)) { FPUBitmask |= (3 << RegNum); CSFPRegsSize += AFGR64RegSize; HasAFGR64Reg = true; } else if (Mips::GPR32RegClass.contains(Reg)) CPUBitmask |= (1 << RegNum); } // FP Regs are saved right below where the virtual frame pointer points to. FPUTopSavedRegOff = FPUBitmask ? (HasAFGR64Reg ? -AFGR64RegSize : -FGR32RegSize) : 0; // CPU Regs are saved below FP Regs. CPUTopSavedRegOff = CPUBitmask ? -CSFPRegsSize - CPURegSize : 0; MipsTargetStreamer &TS = getTargetStreamer(); // Print CPUBitmask TS.emitMask(CPUBitmask, CPUTopSavedRegOff); // Print FPUBitmask TS.emitFMask(FPUBitmask, FPUTopSavedRegOff); } //===----------------------------------------------------------------------===// // Frame and Set directives //===----------------------------------------------------------------------===// /// Frame Directive void MipsAsmPrinter::emitFrameDirective() { const TargetRegisterInfo &RI = *MF->getSubtarget().getRegisterInfo(); unsigned stackReg = RI.getFrameRegister(*MF); unsigned returnReg = RI.getRARegister(); unsigned stackSize = MF->getFrameInfo().getStackSize(); getTargetStreamer().emitFrame(stackReg, stackSize, returnReg); } /// Emit Set directives. const char *MipsAsmPrinter::getCurrentABIString() const { switch (static_cast(TM).getABI().GetEnumValue()) { case MipsABIInfo::ABI::O32: return "abi32"; case MipsABIInfo::ABI::N32: return "abiN32"; case MipsABIInfo::ABI::N64: return "abi64"; default: llvm_unreachable("Unknown Mips ABI"); } } void MipsAsmPrinter::EmitFunctionEntryLabel() { MipsTargetStreamer &TS = getTargetStreamer(); // NaCl sandboxing requires that indirect call instructions are masked. // This means that function entry points should be bundle-aligned. if (Subtarget->isTargetNaCl()) EmitAlignment(std::max(MF->getAlignment(), MIPS_NACL_BUNDLE_ALIGN)); if (Subtarget->inMicroMipsMode()) { TS.emitDirectiveSetMicroMips(); TS.setUsesMicroMips(); TS.updateABIInfo(*Subtarget); } else TS.emitDirectiveSetNoMicroMips(); if (Subtarget->inMips16Mode()) TS.emitDirectiveSetMips16(); else TS.emitDirectiveSetNoMips16(); TS.emitDirectiveEnt(*CurrentFnSym); OutStreamer->EmitLabel(CurrentFnSym); } /// EmitFunctionBodyStart - Targets can override this to emit stuff before /// the first basic block in the function. void MipsAsmPrinter::EmitFunctionBodyStart() { MipsTargetStreamer &TS = getTargetStreamer(); MCInstLowering.Initialize(&MF->getContext()); bool IsNakedFunction = MF->getFunction().hasFnAttribute(Attribute::Naked); if (!IsNakedFunction) emitFrameDirective(); if (!IsNakedFunction) printSavedRegsBitmask(); if (!Subtarget->inMips16Mode()) { TS.emitDirectiveSetNoReorder(); TS.emitDirectiveSetNoMacro(); TS.emitDirectiveSetNoAt(); } } /// EmitFunctionBodyEnd - Targets can override this to emit stuff after /// the last basic block in the function. void MipsAsmPrinter::EmitFunctionBodyEnd() { MipsTargetStreamer &TS = getTargetStreamer(); // There are instruction for this macros, but they must // always be at the function end, and we can't emit and // break with BB logic. if (!Subtarget->inMips16Mode()) { TS.emitDirectiveSetAt(); TS.emitDirectiveSetMacro(); TS.emitDirectiveSetReorder(); } TS.emitDirectiveEnd(CurrentFnSym->getName()); // Make sure to terminate any constant pools that were at the end // of the function. if (!InConstantPool) return; InConstantPool = false; OutStreamer->EmitDataRegion(MCDR_DataRegionEnd); } void MipsAsmPrinter::EmitBasicBlockEnd(const MachineBasicBlock &MBB) { AsmPrinter::EmitBasicBlockEnd(MBB); MipsTargetStreamer &TS = getTargetStreamer(); if (MBB.empty()) TS.emitDirectiveInsn(); } /// isBlockOnlyReachableByFallthough - Return true if the basic block has /// exactly one predecessor and the control transfer mechanism between /// the predecessor and this block is a fall-through. bool MipsAsmPrinter::isBlockOnlyReachableByFallthrough(const MachineBasicBlock* MBB) const { // The predecessor has to be immediately before this block. const MachineBasicBlock *Pred = *MBB->pred_begin(); // If the predecessor is a switch statement, assume a jump table // implementation, so it is not a fall through. if (const BasicBlock *bb = Pred->getBasicBlock()) if (isa(bb->getTerminator())) return false; // If this is a landing pad, it isn't a fall through. If it has no preds, // then nothing falls through to it. if (MBB->isEHPad() || MBB->pred_empty()) return false; // If there isn't exactly one predecessor, it can't be a fall through. MachineBasicBlock::const_pred_iterator PI = MBB->pred_begin(), PI2 = PI; ++PI2; if (PI2 != MBB->pred_end()) return false; // The predecessor has to be immediately before this block. if (!Pred->isLayoutSuccessor(MBB)) return false; // If the block is completely empty, then it definitely does fall through. if (Pred->empty()) return true; // Otherwise, check the last instruction. // Check if the last terminator is an unconditional branch. MachineBasicBlock::const_iterator I = Pred->end(); while (I != Pred->begin() && !(--I)->isTerminator()) ; return !I->isBarrier(); } // Print out an operand for an inline asm expression. bool MipsAsmPrinter::PrintAsmOperand(const MachineInstr *MI, unsigned OpNum, unsigned AsmVariant, const char *ExtraCode, raw_ostream &O) { // Does this asm operand have a single letter operand modifier? if (ExtraCode && ExtraCode[0]) { if (ExtraCode[1] != 0) return true; // Unknown modifier. const MachineOperand &MO = MI->getOperand(OpNum); switch (ExtraCode[0]) { default: // See if this is a generic print operand return AsmPrinter::PrintAsmOperand(MI,OpNum,AsmVariant,ExtraCode,O); case 'X': // hex const int if ((MO.getType()) != MachineOperand::MO_Immediate) return true; O << "0x" << Twine::utohexstr(MO.getImm()); return false; case 'x': // hex const int (low 16 bits) if ((MO.getType()) != MachineOperand::MO_Immediate) return true; O << "0x" << Twine::utohexstr(MO.getImm() & 0xffff); return false; case 'd': // decimal const int if ((MO.getType()) != MachineOperand::MO_Immediate) return true; O << MO.getImm(); return false; case 'm': // decimal const int minus 1 if ((MO.getType()) != MachineOperand::MO_Immediate) return true; O << MO.getImm() - 1; return false; case 'y': // exact log2 if ((MO.getType()) != MachineOperand::MO_Immediate) return true; if (!isPowerOf2_64(MO.getImm())) return true; O << Log2_64(MO.getImm()); return false; case 'z': // $0 if zero, regular printing otherwise if (MO.getType() == MachineOperand::MO_Immediate && MO.getImm() == 0) { O << "$0"; return false; } // If not, call printOperand as normal. break; case 'D': // Second part of a double word register operand case 'L': // Low order register of a double word register operand case 'M': // High order register of a double word register operand { if (OpNum == 0) return true; const MachineOperand &FlagsOP = MI->getOperand(OpNum - 1); if (!FlagsOP.isImm()) return true; unsigned Flags = FlagsOP.getImm(); unsigned NumVals = InlineAsm::getNumOperandRegisters(Flags); // Number of registers represented by this operand. We are looking // for 2 for 32 bit mode and 1 for 64 bit mode. if (NumVals != 2) { if (Subtarget->isGP64bit() && NumVals == 1 && MO.isReg()) { unsigned Reg = MO.getReg(); O << '$' << MipsInstPrinter::getRegisterName(Reg); return false; } return true; } unsigned RegOp = OpNum; if (!Subtarget->isGP64bit()){ // Endianness reverses which register holds the high or low value // between M and L. switch(ExtraCode[0]) { case 'M': RegOp = (Subtarget->isLittle()) ? OpNum + 1 : OpNum; break; case 'L': RegOp = (Subtarget->isLittle()) ? OpNum : OpNum + 1; break; case 'D': // Always the second part RegOp = OpNum + 1; } if (RegOp >= MI->getNumOperands()) return true; const MachineOperand &MO = MI->getOperand(RegOp); if (!MO.isReg()) return true; unsigned Reg = MO.getReg(); O << '$' << MipsInstPrinter::getRegisterName(Reg); return false; } break; } case 'w': // Print MSA registers for the 'f' constraint // In LLVM, the 'w' modifier doesn't need to do anything. // We can just call printOperand as normal. break; } } printOperand(MI, OpNum, O); return false; } bool MipsAsmPrinter::PrintAsmMemoryOperand(const MachineInstr *MI, unsigned OpNum, unsigned AsmVariant, const char *ExtraCode, raw_ostream &O) { assert(OpNum + 1 < MI->getNumOperands() && "Insufficient operands"); const MachineOperand &BaseMO = MI->getOperand(OpNum); const MachineOperand &OffsetMO = MI->getOperand(OpNum + 1); assert(BaseMO.isReg() && "Unexpected base pointer for inline asm memory operand."); assert(OffsetMO.isImm() && "Unexpected offset for inline asm memory operand."); int Offset = OffsetMO.getImm(); // Currently we are expecting either no ExtraCode or 'D','M','L'. if (ExtraCode) { switch (ExtraCode[0]) { case 'D': Offset += 4; break; case 'M': if (Subtarget->isLittle()) Offset += 4; break; case 'L': if (!Subtarget->isLittle()) Offset += 4; break; default: return true; // Unknown modifier. } } O << Offset << "($" << MipsInstPrinter::getRegisterName(BaseMO.getReg()) << ")"; return false; } void MipsAsmPrinter::printOperand(const MachineInstr *MI, int opNum, raw_ostream &O) { const MachineOperand &MO = MI->getOperand(opNum); bool closeP = false; if (MO.getTargetFlags()) closeP = true; switch(MO.getTargetFlags()) { case MipsII::MO_GPREL: O << "%gp_rel("; break; case MipsII::MO_GOT_CALL: O << "%call16("; break; case MipsII::MO_GOT: O << "%got("; break; case MipsII::MO_ABS_HI: O << "%hi("; break; case MipsII::MO_ABS_LO: O << "%lo("; break; case MipsII::MO_HIGHER: O << "%higher("; break; case MipsII::MO_HIGHEST: O << "%highest(("; break; case MipsII::MO_TLSGD: O << "%tlsgd("; break; case MipsII::MO_GOTTPREL: O << "%gottprel("; break; case MipsII::MO_TPREL_HI: O << "%tprel_hi("; break; case MipsII::MO_TPREL_LO: O << "%tprel_lo("; break; case MipsII::MO_GPOFF_HI: O << "%hi(%neg(%gp_rel("; break; case MipsII::MO_GPOFF_LO: O << "%lo(%neg(%gp_rel("; break; case MipsII::MO_GOT_DISP: O << "%got_disp("; break; case MipsII::MO_GOT_PAGE: O << "%got_page("; break; case MipsII::MO_GOT_OFST: O << "%got_ofst("; break; } switch (MO.getType()) { case MachineOperand::MO_Register: O << '$' << StringRef(MipsInstPrinter::getRegisterName(MO.getReg())).lower(); break; case MachineOperand::MO_Immediate: O << MO.getImm(); break; case MachineOperand::MO_MachineBasicBlock: MO.getMBB()->getSymbol()->print(O, MAI); return; case MachineOperand::MO_GlobalAddress: getSymbol(MO.getGlobal())->print(O, MAI); break; case MachineOperand::MO_BlockAddress: { MCSymbol *BA = GetBlockAddressSymbol(MO.getBlockAddress()); O << BA->getName(); break; } case MachineOperand::MO_ConstantPoolIndex: O << getDataLayout().getPrivateGlobalPrefix() << "CPI" << getFunctionNumber() << "_" << MO.getIndex(); if (MO.getOffset()) O << "+" << MO.getOffset(); break; default: llvm_unreachable(""); } if (closeP) O << ")"; } void MipsAsmPrinter:: printMemOperand(const MachineInstr *MI, int opNum, raw_ostream &O) { // Load/Store memory operands -- imm($reg) // If PIC target the target is loaded as the // pattern lw $25,%call16($28) // opNum can be invalid if instruction has reglist as operand. // MemOperand is always last operand of instruction (base + offset). switch (MI->getOpcode()) { default: break; case Mips::SWM32_MM: case Mips::LWM32_MM: opNum = MI->getNumOperands() - 2; break; } printOperand(MI, opNum+1, O); O << "("; printOperand(MI, opNum, O); O << ")"; } void MipsAsmPrinter:: printMemOperandEA(const MachineInstr *MI, int opNum, raw_ostream &O) { // when using stack locations for not load/store instructions // print the same way as all normal 3 operand instructions. printOperand(MI, opNum, O); O << ", "; printOperand(MI, opNum+1, O); } void MipsAsmPrinter:: printFCCOperand(const MachineInstr *MI, int opNum, raw_ostream &O, const char *Modifier) { const MachineOperand &MO = MI->getOperand(opNum); O << Mips::MipsFCCToString((Mips::CondCode)MO.getImm()); } void MipsAsmPrinter:: printRegisterList(const MachineInstr *MI, int opNum, raw_ostream &O) { for (int i = opNum, e = MI->getNumOperands(); i != e; ++i) { if (i != opNum) O << ", "; printOperand(MI, i, O); } } void MipsAsmPrinter::EmitStartOfAsmFile(Module &M) { MipsTargetStreamer &TS = getTargetStreamer(); // MipsTargetStreamer has an initialization order problem when emitting an // object file directly (see MipsTargetELFStreamer for full details). Work // around it by re-initializing the PIC state here. TS.setPic(OutContext.getObjectFileInfo()->isPositionIndependent()); // Compute MIPS architecture attributes based on the default subtarget // that we'd have constructed. Module level directives aren't LTO // clean anyhow. // FIXME: For ifunc related functions we could iterate over and look // for a feature string that doesn't match the default one. const Triple &TT = TM.getTargetTriple(); StringRef CPU = MIPS_MC::selectMipsCPU(TT, TM.getTargetCPU()); StringRef FS = TM.getTargetFeatureString(); const MipsTargetMachine &MTM = static_cast(TM); const MipsSubtarget STI(TT, CPU, FS, MTM.isLittleEndian(), MTM, 0); bool IsABICalls = STI.isABICalls(); const MipsABIInfo &ABI = MTM.getABI(); if (IsABICalls) { TS.emitDirectiveAbiCalls(); // FIXME: This condition should be a lot more complicated that it is here. // Ideally it should test for properties of the ABI and not the ABI // itself. // For the moment, I'm only correcting enough to make MIPS-IV work. if (!isPositionIndependent() && STI.hasSym32()) TS.emitDirectiveOptionPic0(); } // Tell the assembler which ABI we are using std::string SectionName = std::string(".mdebug.") + getCurrentABIString(); OutStreamer->SwitchSection( OutContext.getELFSection(SectionName, ELF::SHT_PROGBITS, 0)); // NaN: At the moment we only support: // 1. .nan legacy (default) // 2. .nan 2008 STI.isNaN2008() ? TS.emitDirectiveNaN2008() : TS.emitDirectiveNaNLegacy(); // TODO: handle O64 ABI TS.updateABIInfo(STI); // We should always emit a '.module fp=...' but binutils 2.24 does not accept // it. We therefore emit it when it contradicts the ABI defaults (-mfpxx or // -mfp64) and omit it otherwise. if (ABI.IsO32() && (STI.isABI_FPXX() || STI.isFP64bit())) TS.emitDirectiveModuleFP(); // We should always emit a '.module [no]oddspreg' but binutils 2.24 does not // accept it. We therefore emit it when it contradicts the default or an // option has changed the default (i.e. FPXX) and omit it otherwise. if (ABI.IsO32() && (!STI.useOddSPReg() || STI.isABI_FPXX())) TS.emitDirectiveModuleOddSPReg(); } void MipsAsmPrinter::emitInlineAsmStart() const { MipsTargetStreamer &TS = getTargetStreamer(); // GCC's choice of assembler options for inline assembly code ('at', 'macro' // and 'reorder') is different from LLVM's choice for generated code ('noat', // 'nomacro' and 'noreorder'). // In order to maintain compatibility with inline assembly code which depends // on GCC's assembler options being used, we have to switch to those options // for the duration of the inline assembly block and then switch back. TS.emitDirectiveSetPush(); TS.emitDirectiveSetAt(); TS.emitDirectiveSetMacro(); TS.emitDirectiveSetReorder(); OutStreamer->AddBlankLine(); } void MipsAsmPrinter::emitInlineAsmEnd(const MCSubtargetInfo &StartInfo, const MCSubtargetInfo *EndInfo) const { OutStreamer->AddBlankLine(); getTargetStreamer().emitDirectiveSetPop(); } void MipsAsmPrinter::EmitJal(const MCSubtargetInfo &STI, MCSymbol *Symbol) { MCInst I; I.setOpcode(Mips::JAL); I.addOperand( MCOperand::createExpr(MCSymbolRefExpr::create(Symbol, OutContext))); OutStreamer->EmitInstruction(I, STI); } void MipsAsmPrinter::EmitInstrReg(const MCSubtargetInfo &STI, unsigned Opcode, unsigned Reg) { MCInst I; I.setOpcode(Opcode); I.addOperand(MCOperand::createReg(Reg)); OutStreamer->EmitInstruction(I, STI); } void MipsAsmPrinter::EmitInstrRegReg(const MCSubtargetInfo &STI, unsigned Opcode, unsigned Reg1, unsigned Reg2) { MCInst I; // // Because of the current td files for Mips32, the operands for MTC1 // appear backwards from their normal assembly order. It's not a trivial // change to fix this in the td file so we adjust for it here. // if (Opcode == Mips::MTC1) { unsigned Temp = Reg1; Reg1 = Reg2; Reg2 = Temp; } I.setOpcode(Opcode); I.addOperand(MCOperand::createReg(Reg1)); I.addOperand(MCOperand::createReg(Reg2)); OutStreamer->EmitInstruction(I, STI); } void MipsAsmPrinter::EmitInstrRegRegReg(const MCSubtargetInfo &STI, unsigned Opcode, unsigned Reg1, unsigned Reg2, unsigned Reg3) { MCInst I; I.setOpcode(Opcode); I.addOperand(MCOperand::createReg(Reg1)); I.addOperand(MCOperand::createReg(Reg2)); I.addOperand(MCOperand::createReg(Reg3)); OutStreamer->EmitInstruction(I, STI); } void MipsAsmPrinter::EmitMovFPIntPair(const MCSubtargetInfo &STI, unsigned MovOpc, unsigned Reg1, unsigned Reg2, unsigned FPReg1, unsigned FPReg2, bool LE) { if (!LE) { unsigned temp = Reg1; Reg1 = Reg2; Reg2 = temp; } EmitInstrRegReg(STI, MovOpc, Reg1, FPReg1); EmitInstrRegReg(STI, MovOpc, Reg2, FPReg2); } void MipsAsmPrinter::EmitSwapFPIntParams(const MCSubtargetInfo &STI, Mips16HardFloatInfo::FPParamVariant PV, bool LE, bool ToFP) { using namespace Mips16HardFloatInfo; unsigned MovOpc = ToFP ? Mips::MTC1 : Mips::MFC1; switch (PV) { case FSig: EmitInstrRegReg(STI, MovOpc, Mips::A0, Mips::F12); break; case FFSig: EmitMovFPIntPair(STI, MovOpc, Mips::A0, Mips::A1, Mips::F12, Mips::F14, LE); break; case FDSig: EmitInstrRegReg(STI, MovOpc, Mips::A0, Mips::F12); EmitMovFPIntPair(STI, MovOpc, Mips::A2, Mips::A3, Mips::F14, Mips::F15, LE); break; case DSig: EmitMovFPIntPair(STI, MovOpc, Mips::A0, Mips::A1, Mips::F12, Mips::F13, LE); break; case DDSig: EmitMovFPIntPair(STI, MovOpc, Mips::A0, Mips::A1, Mips::F12, Mips::F13, LE); EmitMovFPIntPair(STI, MovOpc, Mips::A2, Mips::A3, Mips::F14, Mips::F15, LE); break; case DFSig: EmitMovFPIntPair(STI, MovOpc, Mips::A0, Mips::A1, Mips::F12, Mips::F13, LE); EmitInstrRegReg(STI, MovOpc, Mips::A2, Mips::F14); break; case NoSig: return; } } void MipsAsmPrinter::EmitSwapFPIntRetval( const MCSubtargetInfo &STI, Mips16HardFloatInfo::FPReturnVariant RV, bool LE) { using namespace Mips16HardFloatInfo; unsigned MovOpc = Mips::MFC1; switch (RV) { case FRet: EmitInstrRegReg(STI, MovOpc, Mips::V0, Mips::F0); break; case DRet: EmitMovFPIntPair(STI, MovOpc, Mips::V0, Mips::V1, Mips::F0, Mips::F1, LE); break; case CFRet: EmitMovFPIntPair(STI, MovOpc, Mips::V0, Mips::V1, Mips::F0, Mips::F1, LE); break; case CDRet: EmitMovFPIntPair(STI, MovOpc, Mips::V0, Mips::V1, Mips::F0, Mips::F1, LE); EmitMovFPIntPair(STI, MovOpc, Mips::A0, Mips::A1, Mips::F2, Mips::F3, LE); break; case NoFPRet: break; } } void MipsAsmPrinter::EmitFPCallStub( const char *Symbol, const Mips16HardFloatInfo::FuncSignature *Signature) { using namespace Mips16HardFloatInfo; MCSymbol *MSymbol = OutContext.getOrCreateSymbol(StringRef(Symbol)); bool LE = getDataLayout().isLittleEndian(); // Construct a local MCSubtargetInfo here. // This is because the MachineFunction won't exist (but have not yet been // freed) and since we're at the global level we can use the default // constructed subtarget. std::unique_ptr STI(TM.getTarget().createMCSubtargetInfo( TM.getTargetTriple().str(), TM.getTargetCPU(), TM.getTargetFeatureString())); // // .global xxxx // OutStreamer->EmitSymbolAttribute(MSymbol, MCSA_Global); const char *RetType; // // make the comment field identifying the return and parameter // types of the floating point stub // # Stub function to call rettype xxxx (params) // switch (Signature->RetSig) { case FRet: RetType = "float"; break; case DRet: RetType = "double"; break; case CFRet: RetType = "complex"; break; case CDRet: RetType = "double complex"; break; case NoFPRet: RetType = ""; break; } const char *Parms; switch (Signature->ParamSig) { case FSig: Parms = "float"; break; case FFSig: Parms = "float, float"; break; case FDSig: Parms = "float, double"; break; case DSig: Parms = "double"; break; case DDSig: Parms = "double, double"; break; case DFSig: Parms = "double, float"; break; case NoSig: Parms = ""; break; } OutStreamer->AddComment("\t# Stub function to call " + Twine(RetType) + " " + Twine(Symbol) + " (" + Twine(Parms) + ")"); // // probably not necessary but we save and restore the current section state // OutStreamer->PushSection(); // // .section mips16.call.fpxxxx,"ax",@progbits // MCSectionELF *M = OutContext.getELFSection( ".mips16.call.fp." + std::string(Symbol), ELF::SHT_PROGBITS, ELF::SHF_ALLOC | ELF::SHF_EXECINSTR); OutStreamer->SwitchSection(M, nullptr); // // .align 2 // OutStreamer->EmitValueToAlignment(4); MipsTargetStreamer &TS = getTargetStreamer(); // // .set nomips16 // .set nomicromips // TS.emitDirectiveSetNoMips16(); TS.emitDirectiveSetNoMicroMips(); // // .ent __call_stub_fp_xxxx // .type __call_stub_fp_xxxx,@function // __call_stub_fp_xxxx: // std::string x = "__call_stub_fp_" + std::string(Symbol); MCSymbolELF *Stub = cast(OutContext.getOrCreateSymbol(StringRef(x))); TS.emitDirectiveEnt(*Stub); MCSymbol *MType = OutContext.getOrCreateSymbol("__call_stub_fp_" + Twine(Symbol)); OutStreamer->EmitSymbolAttribute(MType, MCSA_ELF_TypeFunction); OutStreamer->EmitLabel(Stub); // Only handle non-pic for now. assert(!isPositionIndependent() && "should not be here if we are compiling pic"); TS.emitDirectiveSetReorder(); // // We need to add a MipsMCExpr class to MCTargetDesc to fully implement // stubs without raw text but this current patch is for compiler generated // functions and they all return some value. // The calling sequence for non pic is different in that case and we need // to implement %lo and %hi in order to handle the case of no return value // See the corresponding method in Mips16HardFloat for details. // // mov the return address to S2. // we have no stack space to store it and we are about to make another call. // We need to make sure that the enclosing function knows to save S2 // This should have already been handled. // // Mov $18, $31 EmitInstrRegRegReg(*STI, Mips::OR, Mips::S2, Mips::RA, Mips::ZERO); EmitSwapFPIntParams(*STI, Signature->ParamSig, LE, true); // Jal xxxx // EmitJal(*STI, MSymbol); // fix return values EmitSwapFPIntRetval(*STI, Signature->RetSig, LE); // // do the return // if (Signature->RetSig == NoFPRet) // llvm_unreachable("should not be any stubs here with no return value"); // else EmitInstrReg(*STI, Mips::JR, Mips::S2); MCSymbol *Tmp = OutContext.createTempSymbol(); OutStreamer->EmitLabel(Tmp); const MCSymbolRefExpr *E = MCSymbolRefExpr::create(Stub, OutContext); const MCSymbolRefExpr *T = MCSymbolRefExpr::create(Tmp, OutContext); const MCExpr *T_min_E = MCBinaryExpr::createSub(T, E, OutContext); OutStreamer->emitELFSize(Stub, T_min_E); TS.emitDirectiveEnd(x); OutStreamer->PopSection(); } void MipsAsmPrinter::EmitEndOfAsmFile(Module &M) { // Emit needed stubs // for (std::map< const char *, const Mips16HardFloatInfo::FuncSignature *>::const_iterator it = StubsNeeded.begin(); it != StubsNeeded.end(); ++it) { const char *Symbol = it->first; const Mips16HardFloatInfo::FuncSignature *Signature = it->second; EmitFPCallStub(Symbol, Signature); } // return to the text section OutStreamer->SwitchSection(OutContext.getObjectFileInfo()->getTextSection()); } void MipsAsmPrinter::EmitSled(const MachineInstr &MI, SledKind Kind) { const uint8_t NoopsInSledCount = Subtarget->isGP64bit() ? 15 : 11; // For mips32 we want to emit the following pattern: // // .Lxray_sled_N: // ALIGN // B .tmpN // 11 NOP instructions (44 bytes) // ADDIU T9, T9, 52 // .tmpN // // We need the 44 bytes (11 instructions) because at runtime, we'd // be patching over the full 48 bytes (12 instructions) with the following // pattern: // // ADDIU SP, SP, -8 // NOP // SW RA, 4(SP) // SW T9, 0(SP) // LUI T9, %hi(__xray_FunctionEntry/Exit) // ORI T9, T9, %lo(__xray_FunctionEntry/Exit) // LUI T0, %hi(function_id) // JALR T9 // ORI T0, T0, %lo(function_id) // LW T9, 0(SP) // LW RA, 4(SP) // ADDIU SP, SP, 8 // // We add 52 bytes to t9 because we want to adjust the function pointer to // the actual start of function i.e. the address just after the noop sled. // We do this because gp displacement relocation is emitted at the start of // of the function i.e after the nop sled and to correctly calculate the // global offset table address, t9 must hold the address of the instruction // containing the gp displacement relocation. // FIXME: Is this correct for the static relocation model? // // For mips64 we want to emit the following pattern: // // .Lxray_sled_N: // ALIGN // B .tmpN // 15 NOP instructions (60 bytes) // .tmpN // // We need the 60 bytes (15 instructions) because at runtime, we'd // be patching over the full 64 bytes (16 instructions) with the following // pattern: // // DADDIU SP, SP, -16 // NOP // SD RA, 8(SP) // SD T9, 0(SP) // LUI T9, %highest(__xray_FunctionEntry/Exit) // ORI T9, T9, %higher(__xray_FunctionEntry/Exit) // DSLL T9, T9, 16 // ORI T9, T9, %hi(__xray_FunctionEntry/Exit) // DSLL T9, T9, 16 // ORI T9, T9, %lo(__xray_FunctionEntry/Exit) // LUI T0, %hi(function_id) // JALR T9 // ADDIU T0, T0, %lo(function_id) // LD T9, 0(SP) // LD RA, 8(SP) // DADDIU SP, SP, 16 // OutStreamer->EmitCodeAlignment(4); auto CurSled = OutContext.createTempSymbol("xray_sled_", true); OutStreamer->EmitLabel(CurSled); auto Target = OutContext.createTempSymbol(); // Emit "B .tmpN" instruction, which jumps over the nop sled to the actual // start of function const MCExpr *TargetExpr = MCSymbolRefExpr::create( Target, MCSymbolRefExpr::VariantKind::VK_None, OutContext); EmitToStreamer(*OutStreamer, MCInstBuilder(Mips::BEQ) .addReg(Mips::ZERO) .addReg(Mips::ZERO) .addExpr(TargetExpr)); for (int8_t I = 0; I < NoopsInSledCount; I++) EmitToStreamer(*OutStreamer, MCInstBuilder(Mips::SLL) .addReg(Mips::ZERO) .addReg(Mips::ZERO) .addImm(0)); OutStreamer->EmitLabel(Target); if (!Subtarget->isGP64bit()) { EmitToStreamer(*OutStreamer, MCInstBuilder(Mips::ADDiu) .addReg(Mips::T9) .addReg(Mips::T9) .addImm(0x34)); } recordSled(CurSled, MI, Kind); } void MipsAsmPrinter::LowerPATCHABLE_FUNCTION_ENTER(const MachineInstr &MI) { EmitSled(MI, SledKind::FUNCTION_ENTER); } void MipsAsmPrinter::LowerPATCHABLE_FUNCTION_EXIT(const MachineInstr &MI) { EmitSled(MI, SledKind::FUNCTION_EXIT); } void MipsAsmPrinter::LowerPATCHABLE_TAIL_CALL(const MachineInstr &MI) { EmitSled(MI, SledKind::TAIL_CALL); } void MipsAsmPrinter::PrintDebugValueComment(const MachineInstr *MI, raw_ostream &OS) { // TODO: implement } // Emit .dtprelword or .dtpreldword directive // and value for debug thread local expression. void MipsAsmPrinter::EmitDebugValue(const MCExpr *Value, unsigned Size) const { if (auto *MipsExpr = dyn_cast(Value)) { if (MipsExpr && MipsExpr->getKind() == MipsMCExpr::MEK_DTPREL) { switch (Size) { case 4: OutStreamer->EmitDTPRel32Value(MipsExpr->getSubExpr()); break; case 8: OutStreamer->EmitDTPRel64Value(MipsExpr->getSubExpr()); break; default: llvm_unreachable("Unexpected size of expression value."); } return; } } AsmPrinter::EmitDebugValue(Value, Size); } // Align all targets of indirect branches on bundle size. Used only if target // is NaCl. void MipsAsmPrinter::NaClAlignIndirectJumpTargets(MachineFunction &MF) { // Align all blocks that are jumped to through jump table. if (MachineJumpTableInfo *JtInfo = MF.getJumpTableInfo()) { const std::vector &JT = JtInfo->getJumpTables(); for (unsigned I = 0; I < JT.size(); ++I) { const std::vector &MBBs = JT[I].MBBs; for (unsigned J = 0; J < MBBs.size(); ++J) MBBs[J]->setAlignment(MIPS_NACL_BUNDLE_ALIGN); } } // If basic block address is taken, block can be target of indirect branch. for (auto &MBB : MF) { if (MBB.hasAddressTaken()) MBB.setAlignment(MIPS_NACL_BUNDLE_ALIGN); } } bool MipsAsmPrinter::isLongBranchPseudo(int Opcode) const { return (Opcode == Mips::LONG_BRANCH_LUi || Opcode == Mips::LONG_BRANCH_LUi2Op || Opcode == Mips::LONG_BRANCH_LUi2Op_64 || Opcode == Mips::LONG_BRANCH_ADDiu || Opcode == Mips::LONG_BRANCH_ADDiu2Op || Opcode == Mips::LONG_BRANCH_DADDiu || Opcode == Mips::LONG_BRANCH_DADDiu2Op); } // Force static initialization. extern "C" void LLVMInitializeMipsAsmPrinter() { RegisterAsmPrinter X(getTheMipsTarget()); RegisterAsmPrinter Y(getTheMipselTarget()); RegisterAsmPrinter A(getTheMips64Target()); RegisterAsmPrinter B(getTheMips64elTarget()); } Index: vendor/llvm/dist-release_80/lib/Target/Mips/MipsFastISel.cpp =================================================================== --- vendor/llvm/dist-release_80/lib/Target/Mips/MipsFastISel.cpp (revision 343793) +++ vendor/llvm/dist-release_80/lib/Target/Mips/MipsFastISel.cpp (revision 343794) @@ -1,2128 +1,2141 @@ //===- MipsFastISel.cpp - Mips FastISel implementation --------------------===// // // The LLVM Compiler Infrastructure // // This file is distributed under the University of Illinois Open Source // License. See LICENSE.TXT for details. // //===----------------------------------------------------------------------===// /// /// \file /// This file defines the MIPS-specific support for the FastISel class. /// Some of the target-specific code is generated by tablegen in the file /// MipsGenFastISel.inc, which is #included here. /// //===----------------------------------------------------------------------===// #include "MCTargetDesc/MipsABIInfo.h" #include "MCTargetDesc/MipsBaseInfo.h" #include "MipsCCState.h" #include "MipsISelLowering.h" #include "MipsInstrInfo.h" #include "MipsMachineFunction.h" #include "MipsSubtarget.h" #include "MipsTargetMachine.h" #include "llvm/ADT/APInt.h" #include "llvm/ADT/ArrayRef.h" #include "llvm/ADT/DenseMap.h" #include "llvm/ADT/SmallVector.h" #include "llvm/Analysis/TargetLibraryInfo.h" #include "llvm/CodeGen/CallingConvLower.h" #include "llvm/CodeGen/FastISel.h" #include "llvm/CodeGen/FunctionLoweringInfo.h" #include "llvm/CodeGen/ISDOpcodes.h" #include "llvm/CodeGen/MachineBasicBlock.h" #include "llvm/CodeGen/MachineFrameInfo.h" #include "llvm/CodeGen/MachineInstrBuilder.h" #include "llvm/CodeGen/MachineMemOperand.h" #include "llvm/CodeGen/MachineRegisterInfo.h" #include "llvm/CodeGen/TargetInstrInfo.h" #include "llvm/CodeGen/TargetLowering.h" #include "llvm/CodeGen/ValueTypes.h" #include "llvm/IR/Attributes.h" #include "llvm/IR/CallingConv.h" #include "llvm/IR/Constant.h" #include "llvm/IR/Constants.h" #include "llvm/IR/DataLayout.h" #include "llvm/IR/Function.h" #include "llvm/IR/GetElementPtrTypeIterator.h" #include "llvm/IR/GlobalValue.h" #include "llvm/IR/GlobalVariable.h" #include "llvm/IR/InstrTypes.h" #include "llvm/IR/Instruction.h" #include "llvm/IR/Instructions.h" #include "llvm/IR/IntrinsicInst.h" #include "llvm/IR/Operator.h" #include "llvm/IR/Type.h" #include "llvm/IR/User.h" #include "llvm/IR/Value.h" +#include "llvm/MC/MCContext.h" #include "llvm/MC/MCInstrDesc.h" #include "llvm/MC/MCRegisterInfo.h" #include "llvm/MC/MCSymbol.h" #include "llvm/Support/Casting.h" #include "llvm/Support/Compiler.h" #include "llvm/Support/Debug.h" #include "llvm/Support/ErrorHandling.h" #include "llvm/Support/MachineValueType.h" #include "llvm/Support/MathExtras.h" #include "llvm/Support/raw_ostream.h" #include #include #include #include #define DEBUG_TYPE "mips-fastisel" using namespace llvm; +extern cl::opt EmitJalrReloc; + namespace { class MipsFastISel final : public FastISel { // All possible address modes. class Address { public: using BaseKind = enum { RegBase, FrameIndexBase }; private: BaseKind Kind = RegBase; union { unsigned Reg; int FI; } Base; int64_t Offset = 0; const GlobalValue *GV = nullptr; public: // Innocuous defaults for our address. Address() { Base.Reg = 0; } void setKind(BaseKind K) { Kind = K; } BaseKind getKind() const { return Kind; } bool isRegBase() const { return Kind == RegBase; } bool isFIBase() const { return Kind == FrameIndexBase; } void setReg(unsigned Reg) { assert(isRegBase() && "Invalid base register access!"); Base.Reg = Reg; } unsigned getReg() const { assert(isRegBase() && "Invalid base register access!"); return Base.Reg; } void setFI(unsigned FI) { assert(isFIBase() && "Invalid base frame index access!"); Base.FI = FI; } unsigned getFI() const { assert(isFIBase() && "Invalid base frame index access!"); return Base.FI; } void setOffset(int64_t Offset_) { Offset = Offset_; } int64_t getOffset() const { return Offset; } void setGlobalValue(const GlobalValue *G) { GV = G; } const GlobalValue *getGlobalValue() { return GV; } }; /// Subtarget - Keep a pointer to the MipsSubtarget around so that we can /// make the right decision when generating code for different targets. const TargetMachine &TM; const MipsSubtarget *Subtarget; const TargetInstrInfo &TII; const TargetLowering &TLI; MipsFunctionInfo *MFI; // Convenience variables to avoid some queries. LLVMContext *Context; bool fastLowerArguments() override; bool fastLowerCall(CallLoweringInfo &CLI) override; bool fastLowerIntrinsicCall(const IntrinsicInst *II) override; bool UnsupportedFPMode; // To allow fast-isel to proceed and just not handle // floating point but not reject doing fast-isel in other // situations private: // Selection routines. bool selectLogicalOp(const Instruction *I); bool selectLoad(const Instruction *I); bool selectStore(const Instruction *I); bool selectBranch(const Instruction *I); bool selectSelect(const Instruction *I); bool selectCmp(const Instruction *I); bool selectFPExt(const Instruction *I); bool selectFPTrunc(const Instruction *I); bool selectFPToInt(const Instruction *I, bool IsSigned); bool selectRet(const Instruction *I); bool selectTrunc(const Instruction *I); bool selectIntExt(const Instruction *I); bool selectShift(const Instruction *I); bool selectDivRem(const Instruction *I, unsigned ISDOpcode); // Utility helper routines. bool isTypeLegal(Type *Ty, MVT &VT); bool isTypeSupported(Type *Ty, MVT &VT); bool isLoadTypeLegal(Type *Ty, MVT &VT); bool computeAddress(const Value *Obj, Address &Addr); bool computeCallAddress(const Value *V, Address &Addr); void simplifyAddress(Address &Addr); // Emit helper routines. bool emitCmp(unsigned DestReg, const CmpInst *CI); bool emitLoad(MVT VT, unsigned &ResultReg, Address &Addr, unsigned Alignment = 0); bool emitStore(MVT VT, unsigned SrcReg, Address Addr, MachineMemOperand *MMO = nullptr); bool emitStore(MVT VT, unsigned SrcReg, Address &Addr, unsigned Alignment = 0); unsigned emitIntExt(MVT SrcVT, unsigned SrcReg, MVT DestVT, bool isZExt); bool emitIntExt(MVT SrcVT, unsigned SrcReg, MVT DestVT, unsigned DestReg, bool IsZExt); bool emitIntZExt(MVT SrcVT, unsigned SrcReg, MVT DestVT, unsigned DestReg); bool emitIntSExt(MVT SrcVT, unsigned SrcReg, MVT DestVT, unsigned DestReg); bool emitIntSExt32r1(MVT SrcVT, unsigned SrcReg, MVT DestVT, unsigned DestReg); bool emitIntSExt32r2(MVT SrcVT, unsigned SrcReg, MVT DestVT, unsigned DestReg); unsigned getRegEnsuringSimpleIntegerWidening(const Value *, bool IsUnsigned); unsigned emitLogicalOp(unsigned ISDOpc, MVT RetVT, const Value *LHS, const Value *RHS); unsigned materializeFP(const ConstantFP *CFP, MVT VT); unsigned materializeGV(const GlobalValue *GV, MVT VT); unsigned materializeInt(const Constant *C, MVT VT); unsigned materialize32BitInt(int64_t Imm, const TargetRegisterClass *RC); unsigned materializeExternalCallSym(MCSymbol *Syn); MachineInstrBuilder emitInst(unsigned Opc) { return BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, DbgLoc, TII.get(Opc)); } MachineInstrBuilder emitInst(unsigned Opc, unsigned DstReg) { return BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, DbgLoc, TII.get(Opc), DstReg); } MachineInstrBuilder emitInstStore(unsigned Opc, unsigned SrcReg, unsigned MemReg, int64_t MemOffset) { return emitInst(Opc).addReg(SrcReg).addReg(MemReg).addImm(MemOffset); } MachineInstrBuilder emitInstLoad(unsigned Opc, unsigned DstReg, unsigned MemReg, int64_t MemOffset) { return emitInst(Opc, DstReg).addReg(MemReg).addImm(MemOffset); } unsigned fastEmitInst_rr(unsigned MachineInstOpcode, const TargetRegisterClass *RC, unsigned Op0, bool Op0IsKill, unsigned Op1, bool Op1IsKill); // for some reason, this default is not generated by tablegen // so we explicitly generate it here. unsigned fastEmitInst_riir(uint64_t inst, const TargetRegisterClass *RC, unsigned Op0, bool Op0IsKill, uint64_t imm1, uint64_t imm2, unsigned Op3, bool Op3IsKill) { return 0; } // Call handling routines. private: CCAssignFn *CCAssignFnForCall(CallingConv::ID CC) const; bool processCallArgs(CallLoweringInfo &CLI, SmallVectorImpl &ArgVTs, unsigned &NumBytes); bool finishCall(CallLoweringInfo &CLI, MVT RetVT, unsigned NumBytes); const MipsABIInfo &getABI() const { return static_cast(TM).getABI(); } public: // Backend specific FastISel code. explicit MipsFastISel(FunctionLoweringInfo &funcInfo, const TargetLibraryInfo *libInfo) : FastISel(funcInfo, libInfo), TM(funcInfo.MF->getTarget()), Subtarget(&funcInfo.MF->getSubtarget()), TII(*Subtarget->getInstrInfo()), TLI(*Subtarget->getTargetLowering()) { MFI = funcInfo.MF->getInfo(); Context = &funcInfo.Fn->getContext(); UnsupportedFPMode = Subtarget->isFP64bit() || Subtarget->useSoftFloat(); } unsigned fastMaterializeAlloca(const AllocaInst *AI) override; unsigned fastMaterializeConstant(const Constant *C) override; bool fastSelectInstruction(const Instruction *I) override; #include "MipsGenFastISel.inc" }; } // end anonymous namespace static bool CC_Mips(unsigned ValNo, MVT ValVT, MVT LocVT, CCValAssign::LocInfo LocInfo, ISD::ArgFlagsTy ArgFlags, CCState &State) LLVM_ATTRIBUTE_UNUSED; static bool CC_MipsO32_FP32(unsigned ValNo, MVT ValVT, MVT LocVT, CCValAssign::LocInfo LocInfo, ISD::ArgFlagsTy ArgFlags, CCState &State) { llvm_unreachable("should not be called"); } static bool CC_MipsO32_FP64(unsigned ValNo, MVT ValVT, MVT LocVT, CCValAssign::LocInfo LocInfo, ISD::ArgFlagsTy ArgFlags, CCState &State) { llvm_unreachable("should not be called"); } #include "MipsGenCallingConv.inc" CCAssignFn *MipsFastISel::CCAssignFnForCall(CallingConv::ID CC) const { return CC_MipsO32; } unsigned MipsFastISel::emitLogicalOp(unsigned ISDOpc, MVT RetVT, const Value *LHS, const Value *RHS) { // Canonicalize immediates to the RHS first. if (isa(LHS) && !isa(RHS)) std::swap(LHS, RHS); unsigned Opc; switch (ISDOpc) { case ISD::AND: Opc = Mips::AND; break; case ISD::OR: Opc = Mips::OR; break; case ISD::XOR: Opc = Mips::XOR; break; default: llvm_unreachable("unexpected opcode"); } unsigned LHSReg = getRegForValue(LHS); if (!LHSReg) return 0; unsigned RHSReg; if (const auto *C = dyn_cast(RHS)) RHSReg = materializeInt(C, MVT::i32); else RHSReg = getRegForValue(RHS); if (!RHSReg) return 0; unsigned ResultReg = createResultReg(&Mips::GPR32RegClass); if (!ResultReg) return 0; emitInst(Opc, ResultReg).addReg(LHSReg).addReg(RHSReg); return ResultReg; } unsigned MipsFastISel::fastMaterializeAlloca(const AllocaInst *AI) { assert(TLI.getValueType(DL, AI->getType(), true) == MVT::i32 && "Alloca should always return a pointer."); DenseMap::iterator SI = FuncInfo.StaticAllocaMap.find(AI); if (SI != FuncInfo.StaticAllocaMap.end()) { unsigned ResultReg = createResultReg(&Mips::GPR32RegClass); BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, DbgLoc, TII.get(Mips::LEA_ADDiu), ResultReg) .addFrameIndex(SI->second) .addImm(0); return ResultReg; } return 0; } unsigned MipsFastISel::materializeInt(const Constant *C, MVT VT) { if (VT != MVT::i32 && VT != MVT::i16 && VT != MVT::i8 && VT != MVT::i1) return 0; const TargetRegisterClass *RC = &Mips::GPR32RegClass; const ConstantInt *CI = cast(C); return materialize32BitInt(CI->getZExtValue(), RC); } unsigned MipsFastISel::materialize32BitInt(int64_t Imm, const TargetRegisterClass *RC) { unsigned ResultReg = createResultReg(RC); if (isInt<16>(Imm)) { unsigned Opc = Mips::ADDiu; emitInst(Opc, ResultReg).addReg(Mips::ZERO).addImm(Imm); return ResultReg; } else if (isUInt<16>(Imm)) { emitInst(Mips::ORi, ResultReg).addReg(Mips::ZERO).addImm(Imm); return ResultReg; } unsigned Lo = Imm & 0xFFFF; unsigned Hi = (Imm >> 16) & 0xFFFF; if (Lo) { // Both Lo and Hi have nonzero bits. unsigned TmpReg = createResultReg(RC); emitInst(Mips::LUi, TmpReg).addImm(Hi); emitInst(Mips::ORi, ResultReg).addReg(TmpReg).addImm(Lo); } else { emitInst(Mips::LUi, ResultReg).addImm(Hi); } return ResultReg; } unsigned MipsFastISel::materializeFP(const ConstantFP *CFP, MVT VT) { if (UnsupportedFPMode) return 0; int64_t Imm = CFP->getValueAPF().bitcastToAPInt().getZExtValue(); if (VT == MVT::f32) { const TargetRegisterClass *RC = &Mips::FGR32RegClass; unsigned DestReg = createResultReg(RC); unsigned TempReg = materialize32BitInt(Imm, &Mips::GPR32RegClass); emitInst(Mips::MTC1, DestReg).addReg(TempReg); return DestReg; } else if (VT == MVT::f64) { const TargetRegisterClass *RC = &Mips::AFGR64RegClass; unsigned DestReg = createResultReg(RC); unsigned TempReg1 = materialize32BitInt(Imm >> 32, &Mips::GPR32RegClass); unsigned TempReg2 = materialize32BitInt(Imm & 0xFFFFFFFF, &Mips::GPR32RegClass); emitInst(Mips::BuildPairF64, DestReg).addReg(TempReg2).addReg(TempReg1); return DestReg; } return 0; } unsigned MipsFastISel::materializeGV(const GlobalValue *GV, MVT VT) { // For now 32-bit only. if (VT != MVT::i32) return 0; const TargetRegisterClass *RC = &Mips::GPR32RegClass; unsigned DestReg = createResultReg(RC); const GlobalVariable *GVar = dyn_cast(GV); bool IsThreadLocal = GVar && GVar->isThreadLocal(); // TLS not supported at this time. if (IsThreadLocal) return 0; emitInst(Mips::LW, DestReg) .addReg(MFI->getGlobalBaseReg()) .addGlobalAddress(GV, 0, MipsII::MO_GOT); if ((GV->hasInternalLinkage() || (GV->hasLocalLinkage() && !isa(GV)))) { unsigned TempReg = createResultReg(RC); emitInst(Mips::ADDiu, TempReg) .addReg(DestReg) .addGlobalAddress(GV, 0, MipsII::MO_ABS_LO); DestReg = TempReg; } return DestReg; } unsigned MipsFastISel::materializeExternalCallSym(MCSymbol *Sym) { const TargetRegisterClass *RC = &Mips::GPR32RegClass; unsigned DestReg = createResultReg(RC); emitInst(Mips::LW, DestReg) .addReg(MFI->getGlobalBaseReg()) .addSym(Sym, MipsII::MO_GOT); return DestReg; } // Materialize a constant into a register, and return the register // number (or zero if we failed to handle it). unsigned MipsFastISel::fastMaterializeConstant(const Constant *C) { EVT CEVT = TLI.getValueType(DL, C->getType(), true); // Only handle simple types. if (!CEVT.isSimple()) return 0; MVT VT = CEVT.getSimpleVT(); if (const ConstantFP *CFP = dyn_cast(C)) return (UnsupportedFPMode) ? 0 : materializeFP(CFP, VT); else if (const GlobalValue *GV = dyn_cast(C)) return materializeGV(GV, VT); else if (isa(C)) return materializeInt(C, VT); return 0; } bool MipsFastISel::computeAddress(const Value *Obj, Address &Addr) { const User *U = nullptr; unsigned Opcode = Instruction::UserOp1; if (const Instruction *I = dyn_cast(Obj)) { // Don't walk into other basic blocks unless the object is an alloca from // another block, otherwise it may not have a virtual register assigned. if (FuncInfo.StaticAllocaMap.count(static_cast(Obj)) || FuncInfo.MBBMap[I->getParent()] == FuncInfo.MBB) { Opcode = I->getOpcode(); U = I; } } else if (const ConstantExpr *C = dyn_cast(Obj)) { Opcode = C->getOpcode(); U = C; } switch (Opcode) { default: break; case Instruction::BitCast: // Look through bitcasts. return computeAddress(U->getOperand(0), Addr); case Instruction::GetElementPtr: { Address SavedAddr = Addr; int64_t TmpOffset = Addr.getOffset(); // Iterate through the GEP folding the constants into offsets where // we can. gep_type_iterator GTI = gep_type_begin(U); for (User::const_op_iterator i = U->op_begin() + 1, e = U->op_end(); i != e; ++i, ++GTI) { const Value *Op = *i; if (StructType *STy = GTI.getStructTypeOrNull()) { const StructLayout *SL = DL.getStructLayout(STy); unsigned Idx = cast(Op)->getZExtValue(); TmpOffset += SL->getElementOffset(Idx); } else { uint64_t S = DL.getTypeAllocSize(GTI.getIndexedType()); while (true) { if (const ConstantInt *CI = dyn_cast(Op)) { // Constant-offset addressing. TmpOffset += CI->getSExtValue() * S; break; } if (canFoldAddIntoGEP(U, Op)) { // A compatible add with a constant operand. Fold the constant. ConstantInt *CI = cast(cast(Op)->getOperand(1)); TmpOffset += CI->getSExtValue() * S; // Iterate on the other operand. Op = cast(Op)->getOperand(0); continue; } // Unsupported goto unsupported_gep; } } } // Try to grab the base operand now. Addr.setOffset(TmpOffset); if (computeAddress(U->getOperand(0), Addr)) return true; // We failed, restore everything and try the other options. Addr = SavedAddr; unsupported_gep: break; } case Instruction::Alloca: { const AllocaInst *AI = cast(Obj); DenseMap::iterator SI = FuncInfo.StaticAllocaMap.find(AI); if (SI != FuncInfo.StaticAllocaMap.end()) { Addr.setKind(Address::FrameIndexBase); Addr.setFI(SI->second); return true; } break; } } Addr.setReg(getRegForValue(Obj)); return Addr.getReg() != 0; } bool MipsFastISel::computeCallAddress(const Value *V, Address &Addr) { const User *U = nullptr; unsigned Opcode = Instruction::UserOp1; if (const auto *I = dyn_cast(V)) { // Check if the value is defined in the same basic block. This information // is crucial to know whether or not folding an operand is valid. if (I->getParent() == FuncInfo.MBB->getBasicBlock()) { Opcode = I->getOpcode(); U = I; } } else if (const auto *C = dyn_cast(V)) { Opcode = C->getOpcode(); U = C; } switch (Opcode) { default: break; case Instruction::BitCast: // Look past bitcasts if its operand is in the same BB. return computeCallAddress(U->getOperand(0), Addr); break; case Instruction::IntToPtr: // Look past no-op inttoptrs if its operand is in the same BB. if (TLI.getValueType(DL, U->getOperand(0)->getType()) == TLI.getPointerTy(DL)) return computeCallAddress(U->getOperand(0), Addr); break; case Instruction::PtrToInt: // Look past no-op ptrtoints if its operand is in the same BB. if (TLI.getValueType(DL, U->getType()) == TLI.getPointerTy(DL)) return computeCallAddress(U->getOperand(0), Addr); break; } if (const GlobalValue *GV = dyn_cast(V)) { Addr.setGlobalValue(GV); return true; } // If all else fails, try to materialize the value in a register. if (!Addr.getGlobalValue()) { Addr.setReg(getRegForValue(V)); return Addr.getReg() != 0; } return false; } bool MipsFastISel::isTypeLegal(Type *Ty, MVT &VT) { EVT evt = TLI.getValueType(DL, Ty, true); // Only handle simple types. if (evt == MVT::Other || !evt.isSimple()) return false; VT = evt.getSimpleVT(); // Handle all legal types, i.e. a register that will directly hold this // value. return TLI.isTypeLegal(VT); } bool MipsFastISel::isTypeSupported(Type *Ty, MVT &VT) { if (Ty->isVectorTy()) return false; if (isTypeLegal(Ty, VT)) return true; // If this is a type than can be sign or zero-extended to a basic operation // go ahead and accept it now. if (VT == MVT::i1 || VT == MVT::i8 || VT == MVT::i16) return true; return false; } bool MipsFastISel::isLoadTypeLegal(Type *Ty, MVT &VT) { if (isTypeLegal(Ty, VT)) return true; // We will extend this in a later patch: // If this is a type than can be sign or zero-extended to a basic operation // go ahead and accept it now. if (VT == MVT::i8 || VT == MVT::i16) return true; return false; } // Because of how EmitCmp is called with fast-isel, you can // end up with redundant "andi" instructions after the sequences emitted below. // We should try and solve this issue in the future. // bool MipsFastISel::emitCmp(unsigned ResultReg, const CmpInst *CI) { const Value *Left = CI->getOperand(0), *Right = CI->getOperand(1); bool IsUnsigned = CI->isUnsigned(); unsigned LeftReg = getRegEnsuringSimpleIntegerWidening(Left, IsUnsigned); if (LeftReg == 0) return false; unsigned RightReg = getRegEnsuringSimpleIntegerWidening(Right, IsUnsigned); if (RightReg == 0) return false; CmpInst::Predicate P = CI->getPredicate(); switch (P) { default: return false; case CmpInst::ICMP_EQ: { unsigned TempReg = createResultReg(&Mips::GPR32RegClass); emitInst(Mips::XOR, TempReg).addReg(LeftReg).addReg(RightReg); emitInst(Mips::SLTiu, ResultReg).addReg(TempReg).addImm(1); break; } case CmpInst::ICMP_NE: { unsigned TempReg = createResultReg(&Mips::GPR32RegClass); emitInst(Mips::XOR, TempReg).addReg(LeftReg).addReg(RightReg); emitInst(Mips::SLTu, ResultReg).addReg(Mips::ZERO).addReg(TempReg); break; } case CmpInst::ICMP_UGT: emitInst(Mips::SLTu, ResultReg).addReg(RightReg).addReg(LeftReg); break; case CmpInst::ICMP_ULT: emitInst(Mips::SLTu, ResultReg).addReg(LeftReg).addReg(RightReg); break; case CmpInst::ICMP_UGE: { unsigned TempReg = createResultReg(&Mips::GPR32RegClass); emitInst(Mips::SLTu, TempReg).addReg(LeftReg).addReg(RightReg); emitInst(Mips::XORi, ResultReg).addReg(TempReg).addImm(1); break; } case CmpInst::ICMP_ULE: { unsigned TempReg = createResultReg(&Mips::GPR32RegClass); emitInst(Mips::SLTu, TempReg).addReg(RightReg).addReg(LeftReg); emitInst(Mips::XORi, ResultReg).addReg(TempReg).addImm(1); break; } case CmpInst::ICMP_SGT: emitInst(Mips::SLT, ResultReg).addReg(RightReg).addReg(LeftReg); break; case CmpInst::ICMP_SLT: emitInst(Mips::SLT, ResultReg).addReg(LeftReg).addReg(RightReg); break; case CmpInst::ICMP_SGE: { unsigned TempReg = createResultReg(&Mips::GPR32RegClass); emitInst(Mips::SLT, TempReg).addReg(LeftReg).addReg(RightReg); emitInst(Mips::XORi, ResultReg).addReg(TempReg).addImm(1); break; } case CmpInst::ICMP_SLE: { unsigned TempReg = createResultReg(&Mips::GPR32RegClass); emitInst(Mips::SLT, TempReg).addReg(RightReg).addReg(LeftReg); emitInst(Mips::XORi, ResultReg).addReg(TempReg).addImm(1); break; } case CmpInst::FCMP_OEQ: case CmpInst::FCMP_UNE: case CmpInst::FCMP_OLT: case CmpInst::FCMP_OLE: case CmpInst::FCMP_OGT: case CmpInst::FCMP_OGE: { if (UnsupportedFPMode) return false; bool IsFloat = Left->getType()->isFloatTy(); bool IsDouble = Left->getType()->isDoubleTy(); if (!IsFloat && !IsDouble) return false; unsigned Opc, CondMovOpc; switch (P) { case CmpInst::FCMP_OEQ: Opc = IsFloat ? Mips::C_EQ_S : Mips::C_EQ_D32; CondMovOpc = Mips::MOVT_I; break; case CmpInst::FCMP_UNE: Opc = IsFloat ? Mips::C_EQ_S : Mips::C_EQ_D32; CondMovOpc = Mips::MOVF_I; break; case CmpInst::FCMP_OLT: Opc = IsFloat ? Mips::C_OLT_S : Mips::C_OLT_D32; CondMovOpc = Mips::MOVT_I; break; case CmpInst::FCMP_OLE: Opc = IsFloat ? Mips::C_OLE_S : Mips::C_OLE_D32; CondMovOpc = Mips::MOVT_I; break; case CmpInst::FCMP_OGT: Opc = IsFloat ? Mips::C_ULE_S : Mips::C_ULE_D32; CondMovOpc = Mips::MOVF_I; break; case CmpInst::FCMP_OGE: Opc = IsFloat ? Mips::C_ULT_S : Mips::C_ULT_D32; CondMovOpc = Mips::MOVF_I; break; default: llvm_unreachable("Only switching of a subset of CCs."); } unsigned RegWithZero = createResultReg(&Mips::GPR32RegClass); unsigned RegWithOne = createResultReg(&Mips::GPR32RegClass); emitInst(Mips::ADDiu, RegWithZero).addReg(Mips::ZERO).addImm(0); emitInst(Mips::ADDiu, RegWithOne).addReg(Mips::ZERO).addImm(1); emitInst(Opc).addReg(Mips::FCC0, RegState::Define).addReg(LeftReg) .addReg(RightReg); emitInst(CondMovOpc, ResultReg) .addReg(RegWithOne) .addReg(Mips::FCC0) .addReg(RegWithZero); break; } } return true; } bool MipsFastISel::emitLoad(MVT VT, unsigned &ResultReg, Address &Addr, unsigned Alignment) { // // more cases will be handled here in following patches. // unsigned Opc; switch (VT.SimpleTy) { case MVT::i32: ResultReg = createResultReg(&Mips::GPR32RegClass); Opc = Mips::LW; break; case MVT::i16: ResultReg = createResultReg(&Mips::GPR32RegClass); Opc = Mips::LHu; break; case MVT::i8: ResultReg = createResultReg(&Mips::GPR32RegClass); Opc = Mips::LBu; break; case MVT::f32: if (UnsupportedFPMode) return false; ResultReg = createResultReg(&Mips::FGR32RegClass); Opc = Mips::LWC1; break; case MVT::f64: if (UnsupportedFPMode) return false; ResultReg = createResultReg(&Mips::AFGR64RegClass); Opc = Mips::LDC1; break; default: return false; } if (Addr.isRegBase()) { simplifyAddress(Addr); emitInstLoad(Opc, ResultReg, Addr.getReg(), Addr.getOffset()); return true; } if (Addr.isFIBase()) { unsigned FI = Addr.getFI(); unsigned Align = 4; int64_t Offset = Addr.getOffset(); MachineFrameInfo &MFI = MF->getFrameInfo(); MachineMemOperand *MMO = MF->getMachineMemOperand( MachinePointerInfo::getFixedStack(*MF, FI), MachineMemOperand::MOLoad, MFI.getObjectSize(FI), Align); BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, DbgLoc, TII.get(Opc), ResultReg) .addFrameIndex(FI) .addImm(Offset) .addMemOperand(MMO); return true; } return false; } bool MipsFastISel::emitStore(MVT VT, unsigned SrcReg, Address &Addr, unsigned Alignment) { // // more cases will be handled here in following patches. // unsigned Opc; switch (VT.SimpleTy) { case MVT::i8: Opc = Mips::SB; break; case MVT::i16: Opc = Mips::SH; break; case MVT::i32: Opc = Mips::SW; break; case MVT::f32: if (UnsupportedFPMode) return false; Opc = Mips::SWC1; break; case MVT::f64: if (UnsupportedFPMode) return false; Opc = Mips::SDC1; break; default: return false; } if (Addr.isRegBase()) { simplifyAddress(Addr); emitInstStore(Opc, SrcReg, Addr.getReg(), Addr.getOffset()); return true; } if (Addr.isFIBase()) { unsigned FI = Addr.getFI(); unsigned Align = 4; int64_t Offset = Addr.getOffset(); MachineFrameInfo &MFI = MF->getFrameInfo(); MachineMemOperand *MMO = MF->getMachineMemOperand( MachinePointerInfo::getFixedStack(*MF, FI), MachineMemOperand::MOStore, MFI.getObjectSize(FI), Align); BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, DbgLoc, TII.get(Opc)) .addReg(SrcReg) .addFrameIndex(FI) .addImm(Offset) .addMemOperand(MMO); return true; } return false; } bool MipsFastISel::selectLogicalOp(const Instruction *I) { MVT VT; if (!isTypeSupported(I->getType(), VT)) return false; unsigned ResultReg; switch (I->getOpcode()) { default: llvm_unreachable("Unexpected instruction."); case Instruction::And: ResultReg = emitLogicalOp(ISD::AND, VT, I->getOperand(0), I->getOperand(1)); break; case Instruction::Or: ResultReg = emitLogicalOp(ISD::OR, VT, I->getOperand(0), I->getOperand(1)); break; case Instruction::Xor: ResultReg = emitLogicalOp(ISD::XOR, VT, I->getOperand(0), I->getOperand(1)); break; } if (!ResultReg) return false; updateValueMap(I, ResultReg); return true; } bool MipsFastISel::selectLoad(const Instruction *I) { // Atomic loads need special handling. if (cast(I)->isAtomic()) return false; // Verify we have a legal type before going any further. MVT VT; if (!isLoadTypeLegal(I->getType(), VT)) return false; // See if we can handle this address. Address Addr; if (!computeAddress(I->getOperand(0), Addr)) return false; unsigned ResultReg; if (!emitLoad(VT, ResultReg, Addr, cast(I)->getAlignment())) return false; updateValueMap(I, ResultReg); return true; } bool MipsFastISel::selectStore(const Instruction *I) { Value *Op0 = I->getOperand(0); unsigned SrcReg = 0; // Atomic stores need special handling. if (cast(I)->isAtomic()) return false; // Verify we have a legal type before going any further. MVT VT; if (!isLoadTypeLegal(I->getOperand(0)->getType(), VT)) return false; // Get the value to be stored into a register. SrcReg = getRegForValue(Op0); if (SrcReg == 0) return false; // See if we can handle this address. Address Addr; if (!computeAddress(I->getOperand(1), Addr)) return false; if (!emitStore(VT, SrcReg, Addr, cast(I)->getAlignment())) return false; return true; } // This can cause a redundant sltiu to be generated. // FIXME: try and eliminate this in a future patch. bool MipsFastISel::selectBranch(const Instruction *I) { const BranchInst *BI = cast(I); MachineBasicBlock *BrBB = FuncInfo.MBB; // // TBB is the basic block for the case where the comparison is true. // FBB is the basic block for the case where the comparison is false. // if (cond) goto TBB // goto FBB // TBB: // MachineBasicBlock *TBB = FuncInfo.MBBMap[BI->getSuccessor(0)]; MachineBasicBlock *FBB = FuncInfo.MBBMap[BI->getSuccessor(1)]; // For now, just try the simplest case where it's fed by a compare. if (const CmpInst *CI = dyn_cast(BI->getCondition())) { MVT CIMVT = TLI.getValueType(DL, CI->getOperand(0)->getType(), true).getSimpleVT(); if (CIMVT == MVT::i1) return false; unsigned CondReg = getRegForValue(CI); BuildMI(*BrBB, FuncInfo.InsertPt, DbgLoc, TII.get(Mips::BGTZ)) .addReg(CondReg) .addMBB(TBB); finishCondBranch(BI->getParent(), TBB, FBB); return true; } return false; } bool MipsFastISel::selectCmp(const Instruction *I) { const CmpInst *CI = cast(I); unsigned ResultReg = createResultReg(&Mips::GPR32RegClass); if (!emitCmp(ResultReg, CI)) return false; updateValueMap(I, ResultReg); return true; } // Attempt to fast-select a floating-point extend instruction. bool MipsFastISel::selectFPExt(const Instruction *I) { if (UnsupportedFPMode) return false; Value *Src = I->getOperand(0); EVT SrcVT = TLI.getValueType(DL, Src->getType(), true); EVT DestVT = TLI.getValueType(DL, I->getType(), true); if (SrcVT != MVT::f32 || DestVT != MVT::f64) return false; unsigned SrcReg = getRegForValue(Src); // this must be a 32bit floating point register class // maybe we should handle this differently if (!SrcReg) return false; unsigned DestReg = createResultReg(&Mips::AFGR64RegClass); emitInst(Mips::CVT_D32_S, DestReg).addReg(SrcReg); updateValueMap(I, DestReg); return true; } bool MipsFastISel::selectSelect(const Instruction *I) { assert(isa(I) && "Expected a select instruction."); LLVM_DEBUG(dbgs() << "selectSelect\n"); MVT VT; if (!isTypeSupported(I->getType(), VT) || UnsupportedFPMode) { LLVM_DEBUG( dbgs() << ".. .. gave up (!isTypeSupported || UnsupportedFPMode)\n"); return false; } unsigned CondMovOpc; const TargetRegisterClass *RC; if (VT.isInteger() && !VT.isVector() && VT.getSizeInBits() <= 32) { CondMovOpc = Mips::MOVN_I_I; RC = &Mips::GPR32RegClass; } else if (VT == MVT::f32) { CondMovOpc = Mips::MOVN_I_S; RC = &Mips::FGR32RegClass; } else if (VT == MVT::f64) { CondMovOpc = Mips::MOVN_I_D32; RC = &Mips::AFGR64RegClass; } else return false; const SelectInst *SI = cast(I); const Value *Cond = SI->getCondition(); unsigned Src1Reg = getRegForValue(SI->getTrueValue()); unsigned Src2Reg = getRegForValue(SI->getFalseValue()); unsigned CondReg = getRegForValue(Cond); if (!Src1Reg || !Src2Reg || !CondReg) return false; unsigned ZExtCondReg = createResultReg(&Mips::GPR32RegClass); if (!ZExtCondReg) return false; if (!emitIntExt(MVT::i1, CondReg, MVT::i32, ZExtCondReg, true)) return false; unsigned ResultReg = createResultReg(RC); unsigned TempReg = createResultReg(RC); if (!ResultReg || !TempReg) return false; emitInst(TargetOpcode::COPY, TempReg).addReg(Src2Reg); emitInst(CondMovOpc, ResultReg) .addReg(Src1Reg).addReg(ZExtCondReg).addReg(TempReg); updateValueMap(I, ResultReg); return true; } // Attempt to fast-select a floating-point truncate instruction. bool MipsFastISel::selectFPTrunc(const Instruction *I) { if (UnsupportedFPMode) return false; Value *Src = I->getOperand(0); EVT SrcVT = TLI.getValueType(DL, Src->getType(), true); EVT DestVT = TLI.getValueType(DL, I->getType(), true); if (SrcVT != MVT::f64 || DestVT != MVT::f32) return false; unsigned SrcReg = getRegForValue(Src); if (!SrcReg) return false; unsigned DestReg = createResultReg(&Mips::FGR32RegClass); if (!DestReg) return false; emitInst(Mips::CVT_S_D32, DestReg).addReg(SrcReg); updateValueMap(I, DestReg); return true; } // Attempt to fast-select a floating-point-to-integer conversion. bool MipsFastISel::selectFPToInt(const Instruction *I, bool IsSigned) { if (UnsupportedFPMode) return false; MVT DstVT, SrcVT; if (!IsSigned) return false; // We don't handle this case yet. There is no native // instruction for this but it can be synthesized. Type *DstTy = I->getType(); if (!isTypeLegal(DstTy, DstVT)) return false; if (DstVT != MVT::i32) return false; Value *Src = I->getOperand(0); Type *SrcTy = Src->getType(); if (!isTypeLegal(SrcTy, SrcVT)) return false; if (SrcVT != MVT::f32 && SrcVT != MVT::f64) return false; unsigned SrcReg = getRegForValue(Src); if (SrcReg == 0) return false; // Determine the opcode for the conversion, which takes place // entirely within FPRs. unsigned DestReg = createResultReg(&Mips::GPR32RegClass); unsigned TempReg = createResultReg(&Mips::FGR32RegClass); unsigned Opc = (SrcVT == MVT::f32) ? Mips::TRUNC_W_S : Mips::TRUNC_W_D32; // Generate the convert. emitInst(Opc, TempReg).addReg(SrcReg); emitInst(Mips::MFC1, DestReg).addReg(TempReg); updateValueMap(I, DestReg); return true; } bool MipsFastISel::processCallArgs(CallLoweringInfo &CLI, SmallVectorImpl &OutVTs, unsigned &NumBytes) { CallingConv::ID CC = CLI.CallConv; SmallVector ArgLocs; CCState CCInfo(CC, false, *FuncInfo.MF, ArgLocs, *Context); CCInfo.AnalyzeCallOperands(OutVTs, CLI.OutFlags, CCAssignFnForCall(CC)); // Get a count of how many bytes are to be pushed on the stack. NumBytes = CCInfo.getNextStackOffset(); // This is the minimum argument area used for A0-A3. if (NumBytes < 16) NumBytes = 16; emitInst(Mips::ADJCALLSTACKDOWN).addImm(16).addImm(0); // Process the args. MVT firstMVT; for (unsigned i = 0, e = ArgLocs.size(); i != e; ++i) { CCValAssign &VA = ArgLocs[i]; const Value *ArgVal = CLI.OutVals[VA.getValNo()]; MVT ArgVT = OutVTs[VA.getValNo()]; if (i == 0) { firstMVT = ArgVT; if (ArgVT == MVT::f32) { VA.convertToReg(Mips::F12); } else if (ArgVT == MVT::f64) { VA.convertToReg(Mips::D6); } } else if (i == 1) { if ((firstMVT == MVT::f32) || (firstMVT == MVT::f64)) { if (ArgVT == MVT::f32) { VA.convertToReg(Mips::F14); } else if (ArgVT == MVT::f64) { VA.convertToReg(Mips::D7); } } } if (((ArgVT == MVT::i32) || (ArgVT == MVT::f32) || (ArgVT == MVT::i16) || (ArgVT == MVT::i8)) && VA.isMemLoc()) { switch (VA.getLocMemOffset()) { case 0: VA.convertToReg(Mips::A0); break; case 4: VA.convertToReg(Mips::A1); break; case 8: VA.convertToReg(Mips::A2); break; case 12: VA.convertToReg(Mips::A3); break; default: break; } } unsigned ArgReg = getRegForValue(ArgVal); if (!ArgReg) return false; // Handle arg promotion: SExt, ZExt, AExt. switch (VA.getLocInfo()) { case CCValAssign::Full: break; case CCValAssign::AExt: case CCValAssign::SExt: { MVT DestVT = VA.getLocVT(); MVT SrcVT = ArgVT; ArgReg = emitIntExt(SrcVT, ArgReg, DestVT, /*isZExt=*/false); if (!ArgReg) return false; break; } case CCValAssign::ZExt: { MVT DestVT = VA.getLocVT(); MVT SrcVT = ArgVT; ArgReg = emitIntExt(SrcVT, ArgReg, DestVT, /*isZExt=*/true); if (!ArgReg) return false; break; } default: llvm_unreachable("Unknown arg promotion!"); } // Now copy/store arg to correct locations. if (VA.isRegLoc() && !VA.needsCustom()) { BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, DbgLoc, TII.get(TargetOpcode::COPY), VA.getLocReg()).addReg(ArgReg); CLI.OutRegs.push_back(VA.getLocReg()); } else if (VA.needsCustom()) { llvm_unreachable("Mips does not use custom args."); return false; } else { // // FIXME: This path will currently return false. It was copied // from the AArch64 port and should be essentially fine for Mips too. // The work to finish up this path will be done in a follow-on patch. // assert(VA.isMemLoc() && "Assuming store on stack."); // Don't emit stores for undef values. if (isa(ArgVal)) continue; // Need to store on the stack. // FIXME: This alignment is incorrect but this path is disabled // for now (will return false). We need to determine the right alignment // based on the normal alignment for the underlying machine type. // unsigned ArgSize = alignTo(ArgVT.getSizeInBits(), 4); unsigned BEAlign = 0; if (ArgSize < 8 && !Subtarget->isLittle()) BEAlign = 8 - ArgSize; Address Addr; Addr.setKind(Address::RegBase); Addr.setReg(Mips::SP); Addr.setOffset(VA.getLocMemOffset() + BEAlign); unsigned Alignment = DL.getABITypeAlignment(ArgVal->getType()); MachineMemOperand *MMO = FuncInfo.MF->getMachineMemOperand( MachinePointerInfo::getStack(*FuncInfo.MF, Addr.getOffset()), MachineMemOperand::MOStore, ArgVT.getStoreSize(), Alignment); (void)(MMO); // if (!emitStore(ArgVT, ArgReg, Addr, MMO)) return false; // can't store on the stack yet. } } return true; } bool MipsFastISel::finishCall(CallLoweringInfo &CLI, MVT RetVT, unsigned NumBytes) { CallingConv::ID CC = CLI.CallConv; emitInst(Mips::ADJCALLSTACKUP).addImm(16).addImm(0); if (RetVT != MVT::isVoid) { SmallVector RVLocs; MipsCCState CCInfo(CC, false, *FuncInfo.MF, RVLocs, *Context); CCInfo.AnalyzeCallResult(CLI.Ins, RetCC_Mips, CLI.RetTy, CLI.Symbol ? CLI.Symbol->getName().data() : nullptr); // Only handle a single return value. if (RVLocs.size() != 1) return false; // Copy all of the result registers out of their specified physreg. MVT CopyVT = RVLocs[0].getValVT(); // Special handling for extended integers. if (RetVT == MVT::i1 || RetVT == MVT::i8 || RetVT == MVT::i16) CopyVT = MVT::i32; unsigned ResultReg = createResultReg(TLI.getRegClassFor(CopyVT)); if (!ResultReg) return false; BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, DbgLoc, TII.get(TargetOpcode::COPY), ResultReg).addReg(RVLocs[0].getLocReg()); CLI.InRegs.push_back(RVLocs[0].getLocReg()); CLI.ResultReg = ResultReg; CLI.NumResultRegs = 1; } return true; } bool MipsFastISel::fastLowerArguments() { LLVM_DEBUG(dbgs() << "fastLowerArguments\n"); if (!FuncInfo.CanLowerReturn) { LLVM_DEBUG(dbgs() << ".. gave up (!CanLowerReturn)\n"); return false; } const Function *F = FuncInfo.Fn; if (F->isVarArg()) { LLVM_DEBUG(dbgs() << ".. gave up (varargs)\n"); return false; } CallingConv::ID CC = F->getCallingConv(); if (CC != CallingConv::C) { LLVM_DEBUG(dbgs() << ".. gave up (calling convention is not C)\n"); return false; } std::array GPR32ArgRegs = {{Mips::A0, Mips::A1, Mips::A2, Mips::A3}}; std::array FGR32ArgRegs = {{Mips::F12, Mips::F14}}; std::array AFGR64ArgRegs = {{Mips::D6, Mips::D7}}; auto NextGPR32 = GPR32ArgRegs.begin(); auto NextFGR32 = FGR32ArgRegs.begin(); auto NextAFGR64 = AFGR64ArgRegs.begin(); struct AllocatedReg { const TargetRegisterClass *RC; unsigned Reg; AllocatedReg(const TargetRegisterClass *RC, unsigned Reg) : RC(RC), Reg(Reg) {} }; // Only handle simple cases. i.e. All arguments are directly mapped to // registers of the appropriate type. SmallVector Allocation; for (const auto &FormalArg : F->args()) { if (FormalArg.hasAttribute(Attribute::InReg) || FormalArg.hasAttribute(Attribute::StructRet) || FormalArg.hasAttribute(Attribute::ByVal)) { LLVM_DEBUG(dbgs() << ".. gave up (inreg, structret, byval)\n"); return false; } Type *ArgTy = FormalArg.getType(); if (ArgTy->isStructTy() || ArgTy->isArrayTy() || ArgTy->isVectorTy()) { LLVM_DEBUG(dbgs() << ".. gave up (struct, array, or vector)\n"); return false; } EVT ArgVT = TLI.getValueType(DL, ArgTy); LLVM_DEBUG(dbgs() << ".. " << FormalArg.getArgNo() << ": " << ArgVT.getEVTString() << "\n"); if (!ArgVT.isSimple()) { LLVM_DEBUG(dbgs() << ".. .. gave up (not a simple type)\n"); return false; } switch (ArgVT.getSimpleVT().SimpleTy) { case MVT::i1: case MVT::i8: case MVT::i16: if (!FormalArg.hasAttribute(Attribute::SExt) && !FormalArg.hasAttribute(Attribute::ZExt)) { // It must be any extend, this shouldn't happen for clang-generated IR // so just fall back on SelectionDAG. LLVM_DEBUG(dbgs() << ".. .. gave up (i8/i16 arg is not extended)\n"); return false; } if (NextGPR32 == GPR32ArgRegs.end()) { LLVM_DEBUG(dbgs() << ".. .. gave up (ran out of GPR32 arguments)\n"); return false; } LLVM_DEBUG(dbgs() << ".. .. GPR32(" << *NextGPR32 << ")\n"); Allocation.emplace_back(&Mips::GPR32RegClass, *NextGPR32++); // Allocating any GPR32 prohibits further use of floating point arguments. NextFGR32 = FGR32ArgRegs.end(); NextAFGR64 = AFGR64ArgRegs.end(); break; case MVT::i32: if (FormalArg.hasAttribute(Attribute::ZExt)) { // The O32 ABI does not permit a zero-extended i32. LLVM_DEBUG(dbgs() << ".. .. gave up (i32 arg is zero extended)\n"); return false; } if (NextGPR32 == GPR32ArgRegs.end()) { LLVM_DEBUG(dbgs() << ".. .. gave up (ran out of GPR32 arguments)\n"); return false; } LLVM_DEBUG(dbgs() << ".. .. GPR32(" << *NextGPR32 << ")\n"); Allocation.emplace_back(&Mips::GPR32RegClass, *NextGPR32++); // Allocating any GPR32 prohibits further use of floating point arguments. NextFGR32 = FGR32ArgRegs.end(); NextAFGR64 = AFGR64ArgRegs.end(); break; case MVT::f32: if (UnsupportedFPMode) { LLVM_DEBUG(dbgs() << ".. .. gave up (UnsupportedFPMode)\n"); return false; } if (NextFGR32 == FGR32ArgRegs.end()) { LLVM_DEBUG(dbgs() << ".. .. gave up (ran out of FGR32 arguments)\n"); return false; } LLVM_DEBUG(dbgs() << ".. .. FGR32(" << *NextFGR32 << ")\n"); Allocation.emplace_back(&Mips::FGR32RegClass, *NextFGR32++); // Allocating an FGR32 also allocates the super-register AFGR64, and // ABI rules require us to skip the corresponding GPR32. if (NextGPR32 != GPR32ArgRegs.end()) NextGPR32++; if (NextAFGR64 != AFGR64ArgRegs.end()) NextAFGR64++; break; case MVT::f64: if (UnsupportedFPMode) { LLVM_DEBUG(dbgs() << ".. .. gave up (UnsupportedFPMode)\n"); return false; } if (NextAFGR64 == AFGR64ArgRegs.end()) { LLVM_DEBUG(dbgs() << ".. .. gave up (ran out of AFGR64 arguments)\n"); return false; } LLVM_DEBUG(dbgs() << ".. .. AFGR64(" << *NextAFGR64 << ")\n"); Allocation.emplace_back(&Mips::AFGR64RegClass, *NextAFGR64++); // Allocating an FGR32 also allocates the super-register AFGR64, and // ABI rules require us to skip the corresponding GPR32 pair. if (NextGPR32 != GPR32ArgRegs.end()) NextGPR32++; if (NextGPR32 != GPR32ArgRegs.end()) NextGPR32++; if (NextFGR32 != FGR32ArgRegs.end()) NextFGR32++; break; default: LLVM_DEBUG(dbgs() << ".. .. gave up (unknown type)\n"); return false; } } for (const auto &FormalArg : F->args()) { unsigned ArgNo = FormalArg.getArgNo(); unsigned SrcReg = Allocation[ArgNo].Reg; unsigned DstReg = FuncInfo.MF->addLiveIn(SrcReg, Allocation[ArgNo].RC); // FIXME: Unfortunately it's necessary to emit a copy from the livein copy. // Without this, EmitLiveInCopies may eliminate the livein if its only // use is a bitcast (which isn't turned into an instruction). unsigned ResultReg = createResultReg(Allocation[ArgNo].RC); BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, DbgLoc, TII.get(TargetOpcode::COPY), ResultReg) .addReg(DstReg, getKillRegState(true)); updateValueMap(&FormalArg, ResultReg); } // Calculate the size of the incoming arguments area. // We currently reject all the cases where this would be non-zero. unsigned IncomingArgSizeInBytes = 0; // Account for the reserved argument area on ABI's that have one (O32). // It seems strange to do this on the caller side but it's necessary in // SelectionDAG's implementation. IncomingArgSizeInBytes = std::min(getABI().GetCalleeAllocdArgSizeInBytes(CC), IncomingArgSizeInBytes); MF->getInfo()->setFormalArgInfo(IncomingArgSizeInBytes, false); return true; } bool MipsFastISel::fastLowerCall(CallLoweringInfo &CLI) { CallingConv::ID CC = CLI.CallConv; bool IsTailCall = CLI.IsTailCall; bool IsVarArg = CLI.IsVarArg; const Value *Callee = CLI.Callee; MCSymbol *Symbol = CLI.Symbol; // Do not handle FastCC. if (CC == CallingConv::Fast) return false; // Allow SelectionDAG isel to handle tail calls. if (IsTailCall) return false; // Let SDISel handle vararg functions. if (IsVarArg) return false; // FIXME: Only handle *simple* calls for now. MVT RetVT; if (CLI.RetTy->isVoidTy()) RetVT = MVT::isVoid; else if (!isTypeSupported(CLI.RetTy, RetVT)) return false; for (auto Flag : CLI.OutFlags) if (Flag.isInReg() || Flag.isSRet() || Flag.isNest() || Flag.isByVal()) return false; // Set up the argument vectors. SmallVector OutVTs; OutVTs.reserve(CLI.OutVals.size()); for (auto *Val : CLI.OutVals) { MVT VT; if (!isTypeLegal(Val->getType(), VT) && !(VT == MVT::i1 || VT == MVT::i8 || VT == MVT::i16)) return false; // We don't handle vector parameters yet. if (VT.isVector() || VT.getSizeInBits() > 64) return false; OutVTs.push_back(VT); } Address Addr; if (!computeCallAddress(Callee, Addr)) return false; // Handle the arguments now that we've gotten them. unsigned NumBytes; if (!processCallArgs(CLI, OutVTs, NumBytes)) return false; if (!Addr.getGlobalValue()) return false; // Issue the call. unsigned DestAddress; if (Symbol) DestAddress = materializeExternalCallSym(Symbol); else DestAddress = materializeGV(Addr.getGlobalValue(), MVT::i32); emitInst(TargetOpcode::COPY, Mips::T9).addReg(DestAddress); MachineInstrBuilder MIB = BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, DbgLoc, TII.get(Mips::JALR), Mips::RA).addReg(Mips::T9); // Add implicit physical register uses to the call. for (auto Reg : CLI.OutRegs) MIB.addReg(Reg, RegState::Implicit); // Add a register mask with the call-preserved registers. // Proper defs for return values will be added by setPhysRegsDeadExcept(). MIB.addRegMask(TRI.getCallPreservedMask(*FuncInfo.MF, CC)); CLI.Call = MIB; + + if (EmitJalrReloc && !Subtarget->inMips16Mode()) { + // Attach callee address to the instruction, let asm printer emit + // .reloc R_MIPS_JALR. + if (Symbol) + MIB.addSym(Symbol, MipsII::MO_JALR); + else + MIB.addSym(FuncInfo.MF->getContext().getOrCreateSymbol( + Addr.getGlobalValue()->getName()), MipsII::MO_JALR); + } // Finish off the call including any return values. return finishCall(CLI, RetVT, NumBytes); } bool MipsFastISel::fastLowerIntrinsicCall(const IntrinsicInst *II) { switch (II->getIntrinsicID()) { default: return false; case Intrinsic::bswap: { Type *RetTy = II->getCalledFunction()->getReturnType(); MVT VT; if (!isTypeSupported(RetTy, VT)) return false; unsigned SrcReg = getRegForValue(II->getOperand(0)); if (SrcReg == 0) return false; unsigned DestReg = createResultReg(&Mips::GPR32RegClass); if (DestReg == 0) return false; if (VT == MVT::i16) { if (Subtarget->hasMips32r2()) { emitInst(Mips::WSBH, DestReg).addReg(SrcReg); updateValueMap(II, DestReg); return true; } else { unsigned TempReg[3]; for (int i = 0; i < 3; i++) { TempReg[i] = createResultReg(&Mips::GPR32RegClass); if (TempReg[i] == 0) return false; } emitInst(Mips::SLL, TempReg[0]).addReg(SrcReg).addImm(8); emitInst(Mips::SRL, TempReg[1]).addReg(SrcReg).addImm(8); emitInst(Mips::OR, TempReg[2]).addReg(TempReg[0]).addReg(TempReg[1]); emitInst(Mips::ANDi, DestReg).addReg(TempReg[2]).addImm(0xFFFF); updateValueMap(II, DestReg); return true; } } else if (VT == MVT::i32) { if (Subtarget->hasMips32r2()) { unsigned TempReg = createResultReg(&Mips::GPR32RegClass); emitInst(Mips::WSBH, TempReg).addReg(SrcReg); emitInst(Mips::ROTR, DestReg).addReg(TempReg).addImm(16); updateValueMap(II, DestReg); return true; } else { unsigned TempReg[8]; for (int i = 0; i < 8; i++) { TempReg[i] = createResultReg(&Mips::GPR32RegClass); if (TempReg[i] == 0) return false; } emitInst(Mips::SRL, TempReg[0]).addReg(SrcReg).addImm(8); emitInst(Mips::SRL, TempReg[1]).addReg(SrcReg).addImm(24); emitInst(Mips::ANDi, TempReg[2]).addReg(TempReg[0]).addImm(0xFF00); emitInst(Mips::OR, TempReg[3]).addReg(TempReg[1]).addReg(TempReg[2]); emitInst(Mips::ANDi, TempReg[4]).addReg(SrcReg).addImm(0xFF00); emitInst(Mips::SLL, TempReg[5]).addReg(TempReg[4]).addImm(8); emitInst(Mips::SLL, TempReg[6]).addReg(SrcReg).addImm(24); emitInst(Mips::OR, TempReg[7]).addReg(TempReg[3]).addReg(TempReg[5]); emitInst(Mips::OR, DestReg).addReg(TempReg[6]).addReg(TempReg[7]); updateValueMap(II, DestReg); return true; } } return false; } case Intrinsic::memcpy: case Intrinsic::memmove: { const auto *MTI = cast(II); // Don't handle volatile. if (MTI->isVolatile()) return false; if (!MTI->getLength()->getType()->isIntegerTy(32)) return false; const char *IntrMemName = isa(II) ? "memcpy" : "memmove"; return lowerCallTo(II, IntrMemName, II->getNumArgOperands() - 1); } case Intrinsic::memset: { const MemSetInst *MSI = cast(II); // Don't handle volatile. if (MSI->isVolatile()) return false; if (!MSI->getLength()->getType()->isIntegerTy(32)) return false; return lowerCallTo(II, "memset", II->getNumArgOperands() - 1); } } return false; } bool MipsFastISel::selectRet(const Instruction *I) { const Function &F = *I->getParent()->getParent(); const ReturnInst *Ret = cast(I); LLVM_DEBUG(dbgs() << "selectRet\n"); if (!FuncInfo.CanLowerReturn) return false; // Build a list of return value registers. SmallVector RetRegs; if (Ret->getNumOperands() > 0) { CallingConv::ID CC = F.getCallingConv(); // Do not handle FastCC. if (CC == CallingConv::Fast) return false; SmallVector Outs; GetReturnInfo(CC, F.getReturnType(), F.getAttributes(), Outs, TLI, DL); // Analyze operands of the call, assigning locations to each operand. SmallVector ValLocs; MipsCCState CCInfo(CC, F.isVarArg(), *FuncInfo.MF, ValLocs, I->getContext()); CCAssignFn *RetCC = RetCC_Mips; CCInfo.AnalyzeReturn(Outs, RetCC); // Only handle a single return value for now. if (ValLocs.size() != 1) return false; CCValAssign &VA = ValLocs[0]; const Value *RV = Ret->getOperand(0); // Don't bother handling odd stuff for now. if ((VA.getLocInfo() != CCValAssign::Full) && (VA.getLocInfo() != CCValAssign::BCvt)) return false; // Only handle register returns for now. if (!VA.isRegLoc()) return false; unsigned Reg = getRegForValue(RV); if (Reg == 0) return false; unsigned SrcReg = Reg + VA.getValNo(); unsigned DestReg = VA.getLocReg(); // Avoid a cross-class copy. This is very unlikely. if (!MRI.getRegClass(SrcReg)->contains(DestReg)) return false; EVT RVEVT = TLI.getValueType(DL, RV->getType()); if (!RVEVT.isSimple()) return false; if (RVEVT.isVector()) return false; MVT RVVT = RVEVT.getSimpleVT(); if (RVVT == MVT::f128) return false; // Do not handle FGR64 returns for now. if (RVVT == MVT::f64 && UnsupportedFPMode) { LLVM_DEBUG(dbgs() << ".. .. gave up (UnsupportedFPMode\n"); return false; } MVT DestVT = VA.getValVT(); // Special handling for extended integers. if (RVVT != DestVT) { if (RVVT != MVT::i1 && RVVT != MVT::i8 && RVVT != MVT::i16) return false; if (Outs[0].Flags.isZExt() || Outs[0].Flags.isSExt()) { bool IsZExt = Outs[0].Flags.isZExt(); SrcReg = emitIntExt(RVVT, SrcReg, DestVT, IsZExt); if (SrcReg == 0) return false; } } // Make the copy. BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, DbgLoc, TII.get(TargetOpcode::COPY), DestReg).addReg(SrcReg); // Add register to return instruction. RetRegs.push_back(VA.getLocReg()); } MachineInstrBuilder MIB = emitInst(Mips::RetRA); for (unsigned i = 0, e = RetRegs.size(); i != e; ++i) MIB.addReg(RetRegs[i], RegState::Implicit); return true; } bool MipsFastISel::selectTrunc(const Instruction *I) { // The high bits for a type smaller than the register size are assumed to be // undefined. Value *Op = I->getOperand(0); EVT SrcVT, DestVT; SrcVT = TLI.getValueType(DL, Op->getType(), true); DestVT = TLI.getValueType(DL, I->getType(), true); if (SrcVT != MVT::i32 && SrcVT != MVT::i16 && SrcVT != MVT::i8) return false; if (DestVT != MVT::i16 && DestVT != MVT::i8 && DestVT != MVT::i1) return false; unsigned SrcReg = getRegForValue(Op); if (!SrcReg) return false; // Because the high bits are undefined, a truncate doesn't generate // any code. updateValueMap(I, SrcReg); return true; } bool MipsFastISel::selectIntExt(const Instruction *I) { Type *DestTy = I->getType(); Value *Src = I->getOperand(0); Type *SrcTy = Src->getType(); bool isZExt = isa(I); unsigned SrcReg = getRegForValue(Src); if (!SrcReg) return false; EVT SrcEVT, DestEVT; SrcEVT = TLI.getValueType(DL, SrcTy, true); DestEVT = TLI.getValueType(DL, DestTy, true); if (!SrcEVT.isSimple()) return false; if (!DestEVT.isSimple()) return false; MVT SrcVT = SrcEVT.getSimpleVT(); MVT DestVT = DestEVT.getSimpleVT(); unsigned ResultReg = createResultReg(&Mips::GPR32RegClass); if (!emitIntExt(SrcVT, SrcReg, DestVT, ResultReg, isZExt)) return false; updateValueMap(I, ResultReg); return true; } bool MipsFastISel::emitIntSExt32r1(MVT SrcVT, unsigned SrcReg, MVT DestVT, unsigned DestReg) { unsigned ShiftAmt; switch (SrcVT.SimpleTy) { default: return false; case MVT::i8: ShiftAmt = 24; break; case MVT::i16: ShiftAmt = 16; break; } unsigned TempReg = createResultReg(&Mips::GPR32RegClass); emitInst(Mips::SLL, TempReg).addReg(SrcReg).addImm(ShiftAmt); emitInst(Mips::SRA, DestReg).addReg(TempReg).addImm(ShiftAmt); return true; } bool MipsFastISel::emitIntSExt32r2(MVT SrcVT, unsigned SrcReg, MVT DestVT, unsigned DestReg) { switch (SrcVT.SimpleTy) { default: return false; case MVT::i8: emitInst(Mips::SEB, DestReg).addReg(SrcReg); break; case MVT::i16: emitInst(Mips::SEH, DestReg).addReg(SrcReg); break; } return true; } bool MipsFastISel::emitIntSExt(MVT SrcVT, unsigned SrcReg, MVT DestVT, unsigned DestReg) { if ((DestVT != MVT::i32) && (DestVT != MVT::i16)) return false; if (Subtarget->hasMips32r2()) return emitIntSExt32r2(SrcVT, SrcReg, DestVT, DestReg); return emitIntSExt32r1(SrcVT, SrcReg, DestVT, DestReg); } bool MipsFastISel::emitIntZExt(MVT SrcVT, unsigned SrcReg, MVT DestVT, unsigned DestReg) { int64_t Imm; switch (SrcVT.SimpleTy) { default: return false; case MVT::i1: Imm = 1; break; case MVT::i8: Imm = 0xff; break; case MVT::i16: Imm = 0xffff; break; } emitInst(Mips::ANDi, DestReg).addReg(SrcReg).addImm(Imm); return true; } bool MipsFastISel::emitIntExt(MVT SrcVT, unsigned SrcReg, MVT DestVT, unsigned DestReg, bool IsZExt) { // FastISel does not have plumbing to deal with extensions where the SrcVT or // DestVT are odd things, so test to make sure that they are both types we can // handle (i1/i8/i16/i32 for SrcVT and i8/i16/i32/i64 for DestVT), otherwise // bail out to SelectionDAG. if (((DestVT != MVT::i8) && (DestVT != MVT::i16) && (DestVT != MVT::i32)) || ((SrcVT != MVT::i1) && (SrcVT != MVT::i8) && (SrcVT != MVT::i16))) return false; if (IsZExt) return emitIntZExt(SrcVT, SrcReg, DestVT, DestReg); return emitIntSExt(SrcVT, SrcReg, DestVT, DestReg); } unsigned MipsFastISel::emitIntExt(MVT SrcVT, unsigned SrcReg, MVT DestVT, bool isZExt) { unsigned DestReg = createResultReg(&Mips::GPR32RegClass); bool Success = emitIntExt(SrcVT, SrcReg, DestVT, DestReg, isZExt); return Success ? DestReg : 0; } bool MipsFastISel::selectDivRem(const Instruction *I, unsigned ISDOpcode) { EVT DestEVT = TLI.getValueType(DL, I->getType(), true); if (!DestEVT.isSimple()) return false; MVT DestVT = DestEVT.getSimpleVT(); if (DestVT != MVT::i32) return false; unsigned DivOpc; switch (ISDOpcode) { default: return false; case ISD::SDIV: case ISD::SREM: DivOpc = Mips::SDIV; break; case ISD::UDIV: case ISD::UREM: DivOpc = Mips::UDIV; break; } unsigned Src0Reg = getRegForValue(I->getOperand(0)); unsigned Src1Reg = getRegForValue(I->getOperand(1)); if (!Src0Reg || !Src1Reg) return false; emitInst(DivOpc).addReg(Src0Reg).addReg(Src1Reg); emitInst(Mips::TEQ).addReg(Src1Reg).addReg(Mips::ZERO).addImm(7); unsigned ResultReg = createResultReg(&Mips::GPR32RegClass); if (!ResultReg) return false; unsigned MFOpc = (ISDOpcode == ISD::SREM || ISDOpcode == ISD::UREM) ? Mips::MFHI : Mips::MFLO; emitInst(MFOpc, ResultReg); updateValueMap(I, ResultReg); return true; } bool MipsFastISel::selectShift(const Instruction *I) { MVT RetVT; if (!isTypeSupported(I->getType(), RetVT)) return false; unsigned ResultReg = createResultReg(&Mips::GPR32RegClass); if (!ResultReg) return false; unsigned Opcode = I->getOpcode(); const Value *Op0 = I->getOperand(0); unsigned Op0Reg = getRegForValue(Op0); if (!Op0Reg) return false; // If AShr or LShr, then we need to make sure the operand0 is sign extended. if (Opcode == Instruction::AShr || Opcode == Instruction::LShr) { unsigned TempReg = createResultReg(&Mips::GPR32RegClass); if (!TempReg) return false; MVT Op0MVT = TLI.getValueType(DL, Op0->getType(), true).getSimpleVT(); bool IsZExt = Opcode == Instruction::LShr; if (!emitIntExt(Op0MVT, Op0Reg, MVT::i32, TempReg, IsZExt)) return false; Op0Reg = TempReg; } if (const auto *C = dyn_cast(I->getOperand(1))) { uint64_t ShiftVal = C->getZExtValue(); switch (Opcode) { default: llvm_unreachable("Unexpected instruction."); case Instruction::Shl: Opcode = Mips::SLL; break; case Instruction::AShr: Opcode = Mips::SRA; break; case Instruction::LShr: Opcode = Mips::SRL; break; } emitInst(Opcode, ResultReg).addReg(Op0Reg).addImm(ShiftVal); updateValueMap(I, ResultReg); return true; } unsigned Op1Reg = getRegForValue(I->getOperand(1)); if (!Op1Reg) return false; switch (Opcode) { default: llvm_unreachable("Unexpected instruction."); case Instruction::Shl: Opcode = Mips::SLLV; break; case Instruction::AShr: Opcode = Mips::SRAV; break; case Instruction::LShr: Opcode = Mips::SRLV; break; } emitInst(Opcode, ResultReg).addReg(Op0Reg).addReg(Op1Reg); updateValueMap(I, ResultReg); return true; } bool MipsFastISel::fastSelectInstruction(const Instruction *I) { switch (I->getOpcode()) { default: break; case Instruction::Load: return selectLoad(I); case Instruction::Store: return selectStore(I); case Instruction::SDiv: if (!selectBinaryOp(I, ISD::SDIV)) return selectDivRem(I, ISD::SDIV); return true; case Instruction::UDiv: if (!selectBinaryOp(I, ISD::UDIV)) return selectDivRem(I, ISD::UDIV); return true; case Instruction::SRem: if (!selectBinaryOp(I, ISD::SREM)) return selectDivRem(I, ISD::SREM); return true; case Instruction::URem: if (!selectBinaryOp(I, ISD::UREM)) return selectDivRem(I, ISD::UREM); return true; case Instruction::Shl: case Instruction::LShr: case Instruction::AShr: return selectShift(I); case Instruction::And: case Instruction::Or: case Instruction::Xor: return selectLogicalOp(I); case Instruction::Br: return selectBranch(I); case Instruction::Ret: return selectRet(I); case Instruction::Trunc: return selectTrunc(I); case Instruction::ZExt: case Instruction::SExt: return selectIntExt(I); case Instruction::FPTrunc: return selectFPTrunc(I); case Instruction::FPExt: return selectFPExt(I); case Instruction::FPToSI: return selectFPToInt(I, /*isSigned*/ true); case Instruction::FPToUI: return selectFPToInt(I, /*isSigned*/ false); case Instruction::ICmp: case Instruction::FCmp: return selectCmp(I); case Instruction::Select: return selectSelect(I); } return false; } unsigned MipsFastISel::getRegEnsuringSimpleIntegerWidening(const Value *V, bool IsUnsigned) { unsigned VReg = getRegForValue(V); if (VReg == 0) return 0; MVT VMVT = TLI.getValueType(DL, V->getType(), true).getSimpleVT(); if (VMVT == MVT::i1) return 0; if ((VMVT == MVT::i8) || (VMVT == MVT::i16)) { unsigned TempReg = createResultReg(&Mips::GPR32RegClass); if (!emitIntExt(VMVT, VReg, MVT::i32, TempReg, IsUnsigned)) return 0; VReg = TempReg; } return VReg; } void MipsFastISel::simplifyAddress(Address &Addr) { if (!isInt<16>(Addr.getOffset())) { unsigned TempReg = materialize32BitInt(Addr.getOffset(), &Mips::GPR32RegClass); unsigned DestReg = createResultReg(&Mips::GPR32RegClass); emitInst(Mips::ADDu, DestReg).addReg(TempReg).addReg(Addr.getReg()); Addr.setReg(DestReg); Addr.setOffset(0); } } unsigned MipsFastISel::fastEmitInst_rr(unsigned MachineInstOpcode, const TargetRegisterClass *RC, unsigned Op0, bool Op0IsKill, unsigned Op1, bool Op1IsKill) { // We treat the MUL instruction in a special way because it clobbers // the HI0 & LO0 registers. The TableGen definition of this instruction can // mark these registers only as implicitly defined. As a result, the // register allocator runs out of registers when this instruction is // followed by another instruction that defines the same registers too. // We can fix this by explicitly marking those registers as dead. if (MachineInstOpcode == Mips::MUL) { unsigned ResultReg = createResultReg(RC); const MCInstrDesc &II = TII.get(MachineInstOpcode); Op0 = constrainOperandRegClass(II, Op0, II.getNumDefs()); Op1 = constrainOperandRegClass(II, Op1, II.getNumDefs() + 1); BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, DbgLoc, II, ResultReg) .addReg(Op0, getKillRegState(Op0IsKill)) .addReg(Op1, getKillRegState(Op1IsKill)) .addReg(Mips::HI0, RegState::ImplicitDefine | RegState::Dead) .addReg(Mips::LO0, RegState::ImplicitDefine | RegState::Dead); return ResultReg; } return FastISel::fastEmitInst_rr(MachineInstOpcode, RC, Op0, Op0IsKill, Op1, Op1IsKill); } namespace llvm { FastISel *Mips::createFastISel(FunctionLoweringInfo &funcInfo, const TargetLibraryInfo *libInfo) { return new MipsFastISel(funcInfo, libInfo); } } // end namespace llvm Index: vendor/llvm/dist-release_80/lib/Target/Mips/MipsISelLowering.cpp =================================================================== --- vendor/llvm/dist-release_80/lib/Target/Mips/MipsISelLowering.cpp (revision 343793) +++ vendor/llvm/dist-release_80/lib/Target/Mips/MipsISelLowering.cpp (revision 343794) @@ -1,4456 +1,4507 @@ //===- MipsISelLowering.cpp - Mips DAG Lowering Implementation ------------===// // // The LLVM Compiler Infrastructure // // This file is distributed under the University of Illinois Open Source // License. See LICENSE.TXT for details. // //===----------------------------------------------------------------------===// // // This file defines the interfaces that Mips uses to lower LLVM code into a // selection DAG. // //===----------------------------------------------------------------------===// #include "MipsISelLowering.h" #include "InstPrinter/MipsInstPrinter.h" #include "MCTargetDesc/MipsBaseInfo.h" #include "MCTargetDesc/MipsMCTargetDesc.h" #include "MipsCCState.h" #include "MipsInstrInfo.h" #include "MipsMachineFunction.h" #include "MipsRegisterInfo.h" #include "MipsSubtarget.h" #include "MipsTargetMachine.h" #include "MipsTargetObjectFile.h" #include "llvm/ADT/APFloat.h" #include "llvm/ADT/ArrayRef.h" #include "llvm/ADT/SmallVector.h" #include "llvm/ADT/Statistic.h" #include "llvm/ADT/StringRef.h" #include "llvm/ADT/StringSwitch.h" #include "llvm/CodeGen/CallingConvLower.h" #include "llvm/CodeGen/FunctionLoweringInfo.h" #include "llvm/CodeGen/ISDOpcodes.h" #include "llvm/CodeGen/MachineBasicBlock.h" #include "llvm/CodeGen/MachineFrameInfo.h" #include "llvm/CodeGen/MachineFunction.h" #include "llvm/CodeGen/MachineInstr.h" #include "llvm/CodeGen/MachineInstrBuilder.h" #include "llvm/CodeGen/MachineJumpTableInfo.h" #include "llvm/CodeGen/MachineMemOperand.h" #include "llvm/CodeGen/MachineOperand.h" #include "llvm/CodeGen/MachineRegisterInfo.h" #include "llvm/CodeGen/RuntimeLibcalls.h" #include "llvm/CodeGen/SelectionDAG.h" #include "llvm/CodeGen/SelectionDAGNodes.h" #include "llvm/CodeGen/TargetFrameLowering.h" #include "llvm/CodeGen/TargetInstrInfo.h" #include "llvm/CodeGen/TargetRegisterInfo.h" #include "llvm/CodeGen/ValueTypes.h" #include "llvm/IR/CallingConv.h" #include "llvm/IR/Constants.h" #include "llvm/IR/DataLayout.h" #include "llvm/IR/DebugLoc.h" #include "llvm/IR/DerivedTypes.h" #include "llvm/IR/Function.h" #include "llvm/IR/GlobalValue.h" #include "llvm/IR/Type.h" #include "llvm/IR/Value.h" +#include "llvm/MC/MCContext.h" #include "llvm/MC/MCRegisterInfo.h" #include "llvm/Support/Casting.h" #include "llvm/Support/CodeGen.h" #include "llvm/Support/CommandLine.h" #include "llvm/Support/Compiler.h" #include "llvm/Support/ErrorHandling.h" #include "llvm/Support/MachineValueType.h" #include "llvm/Support/MathExtras.h" #include "llvm/Target/TargetMachine.h" #include "llvm/Target/TargetOptions.h" #include #include #include #include #include #include #include #include using namespace llvm; #define DEBUG_TYPE "mips-lower" STATISTIC(NumTailCalls, "Number of tail calls"); static cl::opt LargeGOT("mxgot", cl::Hidden, cl::desc("MIPS: Enable GOT larger than 64k."), cl::init(false)); static cl::opt NoZeroDivCheck("mno-check-zero-division", cl::Hidden, cl::desc("MIPS: Don't trap on integer division by zero."), cl::init(false)); +extern cl::opt EmitJalrReloc; + static const MCPhysReg Mips64DPRegs[8] = { Mips::D12_64, Mips::D13_64, Mips::D14_64, Mips::D15_64, Mips::D16_64, Mips::D17_64, Mips::D18_64, Mips::D19_64 }; // If I is a shifted mask, set the size (Size) and the first bit of the // mask (Pos), and return true. // For example, if I is 0x003ff800, (Pos, Size) = (11, 11). static bool isShiftedMask(uint64_t I, uint64_t &Pos, uint64_t &Size) { if (!isShiftedMask_64(I)) return false; Size = countPopulation(I); Pos = countTrailingZeros(I); return true; } // The MIPS MSA ABI passes vector arguments in the integer register set. // The number of integer registers used is dependant on the ABI used. MVT MipsTargetLowering::getRegisterTypeForCallingConv(LLVMContext &Context, CallingConv::ID CC, EVT VT) const { if (VT.isVector()) { if (Subtarget.isABI_O32()) { return MVT::i32; } else { return (VT.getSizeInBits() == 32) ? MVT::i32 : MVT::i64; } } return MipsTargetLowering::getRegisterType(Context, VT); } unsigned MipsTargetLowering::getNumRegistersForCallingConv(LLVMContext &Context, CallingConv::ID CC, EVT VT) const { if (VT.isVector()) return std::max((VT.getSizeInBits() / (Subtarget.isABI_O32() ? 32 : 64)), 1U); return MipsTargetLowering::getNumRegisters(Context, VT); } unsigned MipsTargetLowering::getVectorTypeBreakdownForCallingConv( LLVMContext &Context, CallingConv::ID CC, EVT VT, EVT &IntermediateVT, unsigned &NumIntermediates, MVT &RegisterVT) const { // Break down vector types to either 2 i64s or 4 i32s. RegisterVT = getRegisterTypeForCallingConv(Context, CC, VT); IntermediateVT = RegisterVT; NumIntermediates = VT.getSizeInBits() < RegisterVT.getSizeInBits() ? VT.getVectorNumElements() : VT.getSizeInBits() / RegisterVT.getSizeInBits(); return NumIntermediates; } SDValue MipsTargetLowering::getGlobalReg(SelectionDAG &DAG, EVT Ty) const { MipsFunctionInfo *FI = DAG.getMachineFunction().getInfo(); return DAG.getRegister(FI->getGlobalBaseReg(), Ty); } SDValue MipsTargetLowering::getTargetNode(GlobalAddressSDNode *N, EVT Ty, SelectionDAG &DAG, unsigned Flag) const { return DAG.getTargetGlobalAddress(N->getGlobal(), SDLoc(N), Ty, 0, Flag); } SDValue MipsTargetLowering::getTargetNode(ExternalSymbolSDNode *N, EVT Ty, SelectionDAG &DAG, unsigned Flag) const { return DAG.getTargetExternalSymbol(N->getSymbol(), Ty, Flag); } SDValue MipsTargetLowering::getTargetNode(BlockAddressSDNode *N, EVT Ty, SelectionDAG &DAG, unsigned Flag) const { return DAG.getTargetBlockAddress(N->getBlockAddress(), Ty, 0, Flag); } SDValue MipsTargetLowering::getTargetNode(JumpTableSDNode *N, EVT Ty, SelectionDAG &DAG, unsigned Flag) const { return DAG.getTargetJumpTable(N->getIndex(), Ty, Flag); } SDValue MipsTargetLowering::getTargetNode(ConstantPoolSDNode *N, EVT Ty, SelectionDAG &DAG, unsigned Flag) const { return DAG.getTargetConstantPool(N->getConstVal(), Ty, N->getAlignment(), N->getOffset(), Flag); } const char *MipsTargetLowering::getTargetNodeName(unsigned Opcode) const { switch ((MipsISD::NodeType)Opcode) { case MipsISD::FIRST_NUMBER: break; case MipsISD::JmpLink: return "MipsISD::JmpLink"; case MipsISD::TailCall: return "MipsISD::TailCall"; case MipsISD::Highest: return "MipsISD::Highest"; case MipsISD::Higher: return "MipsISD::Higher"; case MipsISD::Hi: return "MipsISD::Hi"; case MipsISD::Lo: return "MipsISD::Lo"; case MipsISD::GotHi: return "MipsISD::GotHi"; case MipsISD::TlsHi: return "MipsISD::TlsHi"; case MipsISD::GPRel: return "MipsISD::GPRel"; case MipsISD::ThreadPointer: return "MipsISD::ThreadPointer"; case MipsISD::Ret: return "MipsISD::Ret"; case MipsISD::ERet: return "MipsISD::ERet"; case MipsISD::EH_RETURN: return "MipsISD::EH_RETURN"; case MipsISD::FMS: return "MipsISD::FMS"; case MipsISD::FPBrcond: return "MipsISD::FPBrcond"; case MipsISD::FPCmp: return "MipsISD::FPCmp"; case MipsISD::FSELECT: return "MipsISD::FSELECT"; case MipsISD::MTC1_D64: return "MipsISD::MTC1_D64"; case MipsISD::CMovFP_T: return "MipsISD::CMovFP_T"; case MipsISD::CMovFP_F: return "MipsISD::CMovFP_F"; case MipsISD::TruncIntFP: return "MipsISD::TruncIntFP"; case MipsISD::MFHI: return "MipsISD::MFHI"; case MipsISD::MFLO: return "MipsISD::MFLO"; case MipsISD::MTLOHI: return "MipsISD::MTLOHI"; case MipsISD::Mult: return "MipsISD::Mult"; case MipsISD::Multu: return "MipsISD::Multu"; case MipsISD::MAdd: return "MipsISD::MAdd"; case MipsISD::MAddu: return "MipsISD::MAddu"; case MipsISD::MSub: return "MipsISD::MSub"; case MipsISD::MSubu: return "MipsISD::MSubu"; case MipsISD::DivRem: return "MipsISD::DivRem"; case MipsISD::DivRemU: return "MipsISD::DivRemU"; case MipsISD::DivRem16: return "MipsISD::DivRem16"; case MipsISD::DivRemU16: return "MipsISD::DivRemU16"; case MipsISD::BuildPairF64: return "MipsISD::BuildPairF64"; case MipsISD::ExtractElementF64: return "MipsISD::ExtractElementF64"; case MipsISD::Wrapper: return "MipsISD::Wrapper"; case MipsISD::DynAlloc: return "MipsISD::DynAlloc"; case MipsISD::Sync: return "MipsISD::Sync"; case MipsISD::Ext: return "MipsISD::Ext"; case MipsISD::Ins: return "MipsISD::Ins"; case MipsISD::CIns: return "MipsISD::CIns"; case MipsISD::LWL: return "MipsISD::LWL"; case MipsISD::LWR: return "MipsISD::LWR"; case MipsISD::SWL: return "MipsISD::SWL"; case MipsISD::SWR: return "MipsISD::SWR"; case MipsISD::LDL: return "MipsISD::LDL"; case MipsISD::LDR: return "MipsISD::LDR"; case MipsISD::SDL: return "MipsISD::SDL"; case MipsISD::SDR: return "MipsISD::SDR"; case MipsISD::EXTP: return "MipsISD::EXTP"; case MipsISD::EXTPDP: return "MipsISD::EXTPDP"; case MipsISD::EXTR_S_H: return "MipsISD::EXTR_S_H"; case MipsISD::EXTR_W: return "MipsISD::EXTR_W"; case MipsISD::EXTR_R_W: return "MipsISD::EXTR_R_W"; case MipsISD::EXTR_RS_W: return "MipsISD::EXTR_RS_W"; case MipsISD::SHILO: return "MipsISD::SHILO"; case MipsISD::MTHLIP: return "MipsISD::MTHLIP"; case MipsISD::MULSAQ_S_W_PH: return "MipsISD::MULSAQ_S_W_PH"; case MipsISD::MAQ_S_W_PHL: return "MipsISD::MAQ_S_W_PHL"; case MipsISD::MAQ_S_W_PHR: return "MipsISD::MAQ_S_W_PHR"; case MipsISD::MAQ_SA_W_PHL: return "MipsISD::MAQ_SA_W_PHL"; case MipsISD::MAQ_SA_W_PHR: return "MipsISD::MAQ_SA_W_PHR"; case MipsISD::DPAU_H_QBL: return "MipsISD::DPAU_H_QBL"; case MipsISD::DPAU_H_QBR: return "MipsISD::DPAU_H_QBR"; case MipsISD::DPSU_H_QBL: return "MipsISD::DPSU_H_QBL"; case MipsISD::DPSU_H_QBR: return "MipsISD::DPSU_H_QBR"; case MipsISD::DPAQ_S_W_PH: return "MipsISD::DPAQ_S_W_PH"; case MipsISD::DPSQ_S_W_PH: return "MipsISD::DPSQ_S_W_PH"; case MipsISD::DPAQ_SA_L_W: return "MipsISD::DPAQ_SA_L_W"; case MipsISD::DPSQ_SA_L_W: return "MipsISD::DPSQ_SA_L_W"; case MipsISD::DPA_W_PH: return "MipsISD::DPA_W_PH"; case MipsISD::DPS_W_PH: return "MipsISD::DPS_W_PH"; case MipsISD::DPAQX_S_W_PH: return "MipsISD::DPAQX_S_W_PH"; case MipsISD::DPAQX_SA_W_PH: return "MipsISD::DPAQX_SA_W_PH"; case MipsISD::DPAX_W_PH: return "MipsISD::DPAX_W_PH"; case MipsISD::DPSX_W_PH: return "MipsISD::DPSX_W_PH"; case MipsISD::DPSQX_S_W_PH: return "MipsISD::DPSQX_S_W_PH"; case MipsISD::DPSQX_SA_W_PH: return "MipsISD::DPSQX_SA_W_PH"; case MipsISD::MULSA_W_PH: return "MipsISD::MULSA_W_PH"; case MipsISD::MULT: return "MipsISD::MULT"; case MipsISD::MULTU: return "MipsISD::MULTU"; case MipsISD::MADD_DSP: return "MipsISD::MADD_DSP"; case MipsISD::MADDU_DSP: return "MipsISD::MADDU_DSP"; case MipsISD::MSUB_DSP: return "MipsISD::MSUB_DSP"; case MipsISD::MSUBU_DSP: return "MipsISD::MSUBU_DSP"; case MipsISD::SHLL_DSP: return "MipsISD::SHLL_DSP"; case MipsISD::SHRA_DSP: return "MipsISD::SHRA_DSP"; case MipsISD::SHRL_DSP: return "MipsISD::SHRL_DSP"; case MipsISD::SETCC_DSP: return "MipsISD::SETCC_DSP"; case MipsISD::SELECT_CC_DSP: return "MipsISD::SELECT_CC_DSP"; case MipsISD::VALL_ZERO: return "MipsISD::VALL_ZERO"; case MipsISD::VANY_ZERO: return "MipsISD::VANY_ZERO"; case MipsISD::VALL_NONZERO: return "MipsISD::VALL_NONZERO"; case MipsISD::VANY_NONZERO: return "MipsISD::VANY_NONZERO"; case MipsISD::VCEQ: return "MipsISD::VCEQ"; case MipsISD::VCLE_S: return "MipsISD::VCLE_S"; case MipsISD::VCLE_U: return "MipsISD::VCLE_U"; case MipsISD::VCLT_S: return "MipsISD::VCLT_S"; case MipsISD::VCLT_U: return "MipsISD::VCLT_U"; case MipsISD::VEXTRACT_SEXT_ELT: return "MipsISD::VEXTRACT_SEXT_ELT"; case MipsISD::VEXTRACT_ZEXT_ELT: return "MipsISD::VEXTRACT_ZEXT_ELT"; case MipsISD::VNOR: return "MipsISD::VNOR"; case MipsISD::VSHF: return "MipsISD::VSHF"; case MipsISD::SHF: return "MipsISD::SHF"; case MipsISD::ILVEV: return "MipsISD::ILVEV"; case MipsISD::ILVOD: return "MipsISD::ILVOD"; case MipsISD::ILVL: return "MipsISD::ILVL"; case MipsISD::ILVR: return "MipsISD::ILVR"; case MipsISD::PCKEV: return "MipsISD::PCKEV"; case MipsISD::PCKOD: return "MipsISD::PCKOD"; case MipsISD::INSVE: return "MipsISD::INSVE"; } return nullptr; } MipsTargetLowering::MipsTargetLowering(const MipsTargetMachine &TM, const MipsSubtarget &STI) : TargetLowering(TM), Subtarget(STI), ABI(TM.getABI()) { // Mips does not have i1 type, so use i32 for // setcc operations results (slt, sgt, ...). setBooleanContents(ZeroOrOneBooleanContent); setBooleanVectorContents(ZeroOrNegativeOneBooleanContent); // The cmp.cond.fmt instruction in MIPS32r6/MIPS64r6 uses 0 and -1 like MSA // does. Integer booleans still use 0 and 1. if (Subtarget.hasMips32r6()) setBooleanContents(ZeroOrOneBooleanContent, ZeroOrNegativeOneBooleanContent); // Load extented operations for i1 types must be promoted for (MVT VT : MVT::integer_valuetypes()) { setLoadExtAction(ISD::EXTLOAD, VT, MVT::i1, Promote); setLoadExtAction(ISD::ZEXTLOAD, VT, MVT::i1, Promote); setLoadExtAction(ISD::SEXTLOAD, VT, MVT::i1, Promote); } // MIPS doesn't have extending float->double load/store. Set LoadExtAction // for f32, f16 for (MVT VT : MVT::fp_valuetypes()) { setLoadExtAction(ISD::EXTLOAD, VT, MVT::f32, Expand); setLoadExtAction(ISD::EXTLOAD, VT, MVT::f16, Expand); } // Set LoadExtAction for f16 vectors to Expand for (MVT VT : MVT::fp_vector_valuetypes()) { MVT F16VT = MVT::getVectorVT(MVT::f16, VT.getVectorNumElements()); if (F16VT.isValid()) setLoadExtAction(ISD::EXTLOAD, VT, F16VT, Expand); } setTruncStoreAction(MVT::f32, MVT::f16, Expand); setTruncStoreAction(MVT::f64, MVT::f16, Expand); setTruncStoreAction(MVT::f64, MVT::f32, Expand); // Used by legalize types to correctly generate the setcc result. // Without this, every float setcc comes with a AND/OR with the result, // we don't want this, since the fpcmp result goes to a flag register, // which is used implicitly by brcond and select operations. AddPromotedToType(ISD::SETCC, MVT::i1, MVT::i32); // Mips Custom Operations setOperationAction(ISD::BR_JT, MVT::Other, Expand); setOperationAction(ISD::GlobalAddress, MVT::i32, Custom); setOperationAction(ISD::BlockAddress, MVT::i32, Custom); setOperationAction(ISD::GlobalTLSAddress, MVT::i32, Custom); setOperationAction(ISD::JumpTable, MVT::i32, Custom); setOperationAction(ISD::ConstantPool, MVT::i32, Custom); setOperationAction(ISD::SELECT, MVT::f32, Custom); setOperationAction(ISD::SELECT, MVT::f64, Custom); setOperationAction(ISD::SELECT, MVT::i32, Custom); setOperationAction(ISD::SETCC, MVT::f32, Custom); setOperationAction(ISD::SETCC, MVT::f64, Custom); setOperationAction(ISD::BRCOND, MVT::Other, Custom); setOperationAction(ISD::FCOPYSIGN, MVT::f32, Custom); setOperationAction(ISD::FCOPYSIGN, MVT::f64, Custom); setOperationAction(ISD::FP_TO_SINT, MVT::i32, Custom); if (Subtarget.isGP64bit()) { setOperationAction(ISD::GlobalAddress, MVT::i64, Custom); setOperationAction(ISD::BlockAddress, MVT::i64, Custom); setOperationAction(ISD::GlobalTLSAddress, MVT::i64, Custom); setOperationAction(ISD::JumpTable, MVT::i64, Custom); setOperationAction(ISD::ConstantPool, MVT::i64, Custom); setOperationAction(ISD::SELECT, MVT::i64, Custom); setOperationAction(ISD::LOAD, MVT::i64, Custom); setOperationAction(ISD::STORE, MVT::i64, Custom); setOperationAction(ISD::FP_TO_SINT, MVT::i64, Custom); setOperationAction(ISD::SHL_PARTS, MVT::i64, Custom); setOperationAction(ISD::SRA_PARTS, MVT::i64, Custom); setOperationAction(ISD::SRL_PARTS, MVT::i64, Custom); } if (!Subtarget.isGP64bit()) { setOperationAction(ISD::SHL_PARTS, MVT::i32, Custom); setOperationAction(ISD::SRA_PARTS, MVT::i32, Custom); setOperationAction(ISD::SRL_PARTS, MVT::i32, Custom); } setOperationAction(ISD::EH_DWARF_CFA, MVT::i32, Custom); if (Subtarget.isGP64bit()) setOperationAction(ISD::EH_DWARF_CFA, MVT::i64, Custom); setOperationAction(ISD::SDIV, MVT::i32, Expand); setOperationAction(ISD::SREM, MVT::i32, Expand); setOperationAction(ISD::UDIV, MVT::i32, Expand); setOperationAction(ISD::UREM, MVT::i32, Expand); setOperationAction(ISD::SDIV, MVT::i64, Expand); setOperationAction(ISD::SREM, MVT::i64, Expand); setOperationAction(ISD::UDIV, MVT::i64, Expand); setOperationAction(ISD::UREM, MVT::i64, Expand); // Operations not directly supported by Mips. setOperationAction(ISD::BR_CC, MVT::f32, Expand); setOperationAction(ISD::BR_CC, MVT::f64, Expand); setOperationAction(ISD::BR_CC, MVT::i32, Expand); setOperationAction(ISD::BR_CC, MVT::i64, Expand); setOperationAction(ISD::SELECT_CC, MVT::i32, Expand); setOperationAction(ISD::SELECT_CC, MVT::i64, Expand); setOperationAction(ISD::SELECT_CC, MVT::f32, Expand); setOperationAction(ISD::SELECT_CC, MVT::f64, Expand); setOperationAction(ISD::UINT_TO_FP, MVT::i32, Expand); setOperationAction(ISD::UINT_TO_FP, MVT::i64, Expand); setOperationAction(ISD::FP_TO_UINT, MVT::i32, Expand); setOperationAction(ISD::FP_TO_UINT, MVT::i64, Expand); setOperationAction(ISD::SIGN_EXTEND_INREG, MVT::i1, Expand); if (Subtarget.hasCnMips()) { setOperationAction(ISD::CTPOP, MVT::i32, Legal); setOperationAction(ISD::CTPOP, MVT::i64, Legal); } else { setOperationAction(ISD::CTPOP, MVT::i32, Expand); setOperationAction(ISD::CTPOP, MVT::i64, Expand); } setOperationAction(ISD::CTTZ, MVT::i32, Expand); setOperationAction(ISD::CTTZ, MVT::i64, Expand); setOperationAction(ISD::ROTL, MVT::i32, Expand); setOperationAction(ISD::ROTL, MVT::i64, Expand); setOperationAction(ISD::DYNAMIC_STACKALLOC, MVT::i32, Expand); setOperationAction(ISD::DYNAMIC_STACKALLOC, MVT::i64, Expand); if (!Subtarget.hasMips32r2()) setOperationAction(ISD::ROTR, MVT::i32, Expand); if (!Subtarget.hasMips64r2()) setOperationAction(ISD::ROTR, MVT::i64, Expand); setOperationAction(ISD::FSIN, MVT::f32, Expand); setOperationAction(ISD::FSIN, MVT::f64, Expand); setOperationAction(ISD::FCOS, MVT::f32, Expand); setOperationAction(ISD::FCOS, MVT::f64, Expand); setOperationAction(ISD::FSINCOS, MVT::f32, Expand); setOperationAction(ISD::FSINCOS, MVT::f64, Expand); setOperationAction(ISD::FPOW, MVT::f32, Expand); setOperationAction(ISD::FPOW, MVT::f64, Expand); setOperationAction(ISD::FLOG, MVT::f32, Expand); setOperationAction(ISD::FLOG2, MVT::f32, Expand); setOperationAction(ISD::FLOG10, MVT::f32, Expand); setOperationAction(ISD::FEXP, MVT::f32, Expand); setOperationAction(ISD::FMA, MVT::f32, Expand); setOperationAction(ISD::FMA, MVT::f64, Expand); setOperationAction(ISD::FREM, MVT::f32, Expand); setOperationAction(ISD::FREM, MVT::f64, Expand); // Lower f16 conversion operations into library calls setOperationAction(ISD::FP16_TO_FP, MVT::f32, Expand); setOperationAction(ISD::FP_TO_FP16, MVT::f32, Expand); setOperationAction(ISD::FP16_TO_FP, MVT::f64, Expand); setOperationAction(ISD::FP_TO_FP16, MVT::f64, Expand); setOperationAction(ISD::EH_RETURN, MVT::Other, Custom); setOperationAction(ISD::VASTART, MVT::Other, Custom); setOperationAction(ISD::VAARG, MVT::Other, Custom); setOperationAction(ISD::VACOPY, MVT::Other, Expand); setOperationAction(ISD::VAEND, MVT::Other, Expand); // Use the default for now setOperationAction(ISD::STACKSAVE, MVT::Other, Expand); setOperationAction(ISD::STACKRESTORE, MVT::Other, Expand); if (!Subtarget.isGP64bit()) { setOperationAction(ISD::ATOMIC_LOAD, MVT::i64, Expand); setOperationAction(ISD::ATOMIC_STORE, MVT::i64, Expand); } if (!Subtarget.hasMips32r2()) { setOperationAction(ISD::SIGN_EXTEND_INREG, MVT::i8, Expand); setOperationAction(ISD::SIGN_EXTEND_INREG, MVT::i16, Expand); } // MIPS16 lacks MIPS32's clz and clo instructions. if (!Subtarget.hasMips32() || Subtarget.inMips16Mode()) setOperationAction(ISD::CTLZ, MVT::i32, Expand); if (!Subtarget.hasMips64()) setOperationAction(ISD::CTLZ, MVT::i64, Expand); if (!Subtarget.hasMips32r2()) setOperationAction(ISD::BSWAP, MVT::i32, Expand); if (!Subtarget.hasMips64r2()) setOperationAction(ISD::BSWAP, MVT::i64, Expand); if (Subtarget.isGP64bit()) { setLoadExtAction(ISD::SEXTLOAD, MVT::i64, MVT::i32, Custom); setLoadExtAction(ISD::ZEXTLOAD, MVT::i64, MVT::i32, Custom); setLoadExtAction(ISD::EXTLOAD, MVT::i64, MVT::i32, Custom); setTruncStoreAction(MVT::i64, MVT::i32, Custom); } setOperationAction(ISD::TRAP, MVT::Other, Legal); setTargetDAGCombine(ISD::SDIVREM); setTargetDAGCombine(ISD::UDIVREM); setTargetDAGCombine(ISD::SELECT); setTargetDAGCombine(ISD::AND); setTargetDAGCombine(ISD::OR); setTargetDAGCombine(ISD::ADD); setTargetDAGCombine(ISD::SUB); setTargetDAGCombine(ISD::AssertZext); setTargetDAGCombine(ISD::SHL); if (ABI.IsO32()) { // These libcalls are not available in 32-bit. setLibcallName(RTLIB::SHL_I128, nullptr); setLibcallName(RTLIB::SRL_I128, nullptr); setLibcallName(RTLIB::SRA_I128, nullptr); } setMinFunctionAlignment(Subtarget.isGP64bit() ? 3 : 2); // The arguments on the stack are defined in terms of 4-byte slots on O32 // and 8-byte slots on N32/N64. setMinStackArgumentAlignment((ABI.IsN32() || ABI.IsN64()) ? 8 : 4); setStackPointerRegisterToSaveRestore(ABI.IsN64() ? Mips::SP_64 : Mips::SP); MaxStoresPerMemcpy = 16; isMicroMips = Subtarget.inMicroMipsMode(); } const MipsTargetLowering *MipsTargetLowering::create(const MipsTargetMachine &TM, const MipsSubtarget &STI) { if (STI.inMips16Mode()) return createMips16TargetLowering(TM, STI); return createMipsSETargetLowering(TM, STI); } // Create a fast isel object. FastISel * MipsTargetLowering::createFastISel(FunctionLoweringInfo &funcInfo, const TargetLibraryInfo *libInfo) const { const MipsTargetMachine &TM = static_cast(funcInfo.MF->getTarget()); // We support only the standard encoding [MIPS32,MIPS32R5] ISAs. bool UseFastISel = TM.Options.EnableFastISel && Subtarget.hasMips32() && !Subtarget.hasMips32r6() && !Subtarget.inMips16Mode() && !Subtarget.inMicroMipsMode(); // Disable if either of the following is true: // We do not generate PIC, the ABI is not O32, LargeGOT is being used. if (!TM.isPositionIndependent() || !TM.getABI().IsO32() || LargeGOT) UseFastISel = false; return UseFastISel ? Mips::createFastISel(funcInfo, libInfo) : nullptr; } EVT MipsTargetLowering::getSetCCResultType(const DataLayout &, LLVMContext &, EVT VT) const { if (!VT.isVector()) return MVT::i32; return VT.changeVectorElementTypeToInteger(); } static SDValue performDivRemCombine(SDNode *N, SelectionDAG &DAG, TargetLowering::DAGCombinerInfo &DCI, const MipsSubtarget &Subtarget) { if (DCI.isBeforeLegalizeOps()) return SDValue(); EVT Ty = N->getValueType(0); unsigned LO = (Ty == MVT::i32) ? Mips::LO0 : Mips::LO0_64; unsigned HI = (Ty == MVT::i32) ? Mips::HI0 : Mips::HI0_64; unsigned Opc = N->getOpcode() == ISD::SDIVREM ? MipsISD::DivRem16 : MipsISD::DivRemU16; SDLoc DL(N); SDValue DivRem = DAG.getNode(Opc, DL, MVT::Glue, N->getOperand(0), N->getOperand(1)); SDValue InChain = DAG.getEntryNode(); SDValue InGlue = DivRem; // insert MFLO if (N->hasAnyUseOfValue(0)) { SDValue CopyFromLo = DAG.getCopyFromReg(InChain, DL, LO, Ty, InGlue); DAG.ReplaceAllUsesOfValueWith(SDValue(N, 0), CopyFromLo); InChain = CopyFromLo.getValue(1); InGlue = CopyFromLo.getValue(2); } // insert MFHI if (N->hasAnyUseOfValue(1)) { SDValue CopyFromHi = DAG.getCopyFromReg(InChain, DL, HI, Ty, InGlue); DAG.ReplaceAllUsesOfValueWith(SDValue(N, 1), CopyFromHi); } return SDValue(); } static Mips::CondCode condCodeToFCC(ISD::CondCode CC) { switch (CC) { default: llvm_unreachable("Unknown fp condition code!"); case ISD::SETEQ: case ISD::SETOEQ: return Mips::FCOND_OEQ; case ISD::SETUNE: return Mips::FCOND_UNE; case ISD::SETLT: case ISD::SETOLT: return Mips::FCOND_OLT; case ISD::SETGT: case ISD::SETOGT: return Mips::FCOND_OGT; case ISD::SETLE: case ISD::SETOLE: return Mips::FCOND_OLE; case ISD::SETGE: case ISD::SETOGE: return Mips::FCOND_OGE; case ISD::SETULT: return Mips::FCOND_ULT; case ISD::SETULE: return Mips::FCOND_ULE; case ISD::SETUGT: return Mips::FCOND_UGT; case ISD::SETUGE: return Mips::FCOND_UGE; case ISD::SETUO: return Mips::FCOND_UN; case ISD::SETO: return Mips::FCOND_OR; case ISD::SETNE: case ISD::SETONE: return Mips::FCOND_ONE; case ISD::SETUEQ: return Mips::FCOND_UEQ; } } /// This function returns true if the floating point conditional branches and /// conditional moves which use condition code CC should be inverted. static bool invertFPCondCodeUser(Mips::CondCode CC) { if (CC >= Mips::FCOND_F && CC <= Mips::FCOND_NGT) return false; assert((CC >= Mips::FCOND_T && CC <= Mips::FCOND_GT) && "Illegal Condition Code"); return true; } // Creates and returns an FPCmp node from a setcc node. // Returns Op if setcc is not a floating point comparison. static SDValue createFPCmp(SelectionDAG &DAG, const SDValue &Op) { // must be a SETCC node if (Op.getOpcode() != ISD::SETCC) return Op; SDValue LHS = Op.getOperand(0); if (!LHS.getValueType().isFloatingPoint()) return Op; SDValue RHS = Op.getOperand(1); SDLoc DL(Op); // Assume the 3rd operand is a CondCodeSDNode. Add code to check the type of // node if necessary. ISD::CondCode CC = cast(Op.getOperand(2))->get(); return DAG.getNode(MipsISD::FPCmp, DL, MVT::Glue, LHS, RHS, DAG.getConstant(condCodeToFCC(CC), DL, MVT::i32)); } // Creates and returns a CMovFPT/F node. static SDValue createCMovFP(SelectionDAG &DAG, SDValue Cond, SDValue True, SDValue False, const SDLoc &DL) { ConstantSDNode *CC = cast(Cond.getOperand(2)); bool invert = invertFPCondCodeUser((Mips::CondCode)CC->getSExtValue()); SDValue FCC0 = DAG.getRegister(Mips::FCC0, MVT::i32); return DAG.getNode((invert ? MipsISD::CMovFP_F : MipsISD::CMovFP_T), DL, True.getValueType(), True, FCC0, False, Cond); } static SDValue performSELECTCombine(SDNode *N, SelectionDAG &DAG, TargetLowering::DAGCombinerInfo &DCI, const MipsSubtarget &Subtarget) { if (DCI.isBeforeLegalizeOps()) return SDValue(); SDValue SetCC = N->getOperand(0); if ((SetCC.getOpcode() != ISD::SETCC) || !SetCC.getOperand(0).getValueType().isInteger()) return SDValue(); SDValue False = N->getOperand(2); EVT FalseTy = False.getValueType(); if (!FalseTy.isInteger()) return SDValue(); ConstantSDNode *FalseC = dyn_cast(False); // If the RHS (False) is 0, we swap the order of the operands // of ISD::SELECT (obviously also inverting the condition) so that we can // take advantage of conditional moves using the $0 register. // Example: // return (a != 0) ? x : 0; // load $reg, x // movz $reg, $0, a if (!FalseC) return SDValue(); const SDLoc DL(N); if (!FalseC->getZExtValue()) { ISD::CondCode CC = cast(SetCC.getOperand(2))->get(); SDValue True = N->getOperand(1); SetCC = DAG.getSetCC(DL, SetCC.getValueType(), SetCC.getOperand(0), SetCC.getOperand(1), ISD::getSetCCInverse(CC, true)); return DAG.getNode(ISD::SELECT, DL, FalseTy, SetCC, False, True); } // If both operands are integer constants there's a possibility that we // can do some interesting optimizations. SDValue True = N->getOperand(1); ConstantSDNode *TrueC = dyn_cast(True); if (!TrueC || !True.getValueType().isInteger()) return SDValue(); // We'll also ignore MVT::i64 operands as this optimizations proves // to be ineffective because of the required sign extensions as the result // of a SETCC operator is always MVT::i32 for non-vector types. if (True.getValueType() == MVT::i64) return SDValue(); int64_t Diff = TrueC->getSExtValue() - FalseC->getSExtValue(); // 1) (a < x) ? y : y-1 // slti $reg1, a, x // addiu $reg2, $reg1, y-1 if (Diff == 1) return DAG.getNode(ISD::ADD, DL, SetCC.getValueType(), SetCC, False); // 2) (a < x) ? y-1 : y // slti $reg1, a, x // xor $reg1, $reg1, 1 // addiu $reg2, $reg1, y-1 if (Diff == -1) { ISD::CondCode CC = cast(SetCC.getOperand(2))->get(); SetCC = DAG.getSetCC(DL, SetCC.getValueType(), SetCC.getOperand(0), SetCC.getOperand(1), ISD::getSetCCInverse(CC, true)); return DAG.getNode(ISD::ADD, DL, SetCC.getValueType(), SetCC, True); } // Could not optimize. return SDValue(); } static SDValue performCMovFPCombine(SDNode *N, SelectionDAG &DAG, TargetLowering::DAGCombinerInfo &DCI, const MipsSubtarget &Subtarget) { if (DCI.isBeforeLegalizeOps()) return SDValue(); SDValue ValueIfTrue = N->getOperand(0), ValueIfFalse = N->getOperand(2); ConstantSDNode *FalseC = dyn_cast(ValueIfFalse); if (!FalseC || FalseC->getZExtValue()) return SDValue(); // Since RHS (False) is 0, we swap the order of the True/False operands // (obviously also inverting the condition) so that we can // take advantage of conditional moves using the $0 register. // Example: // return (a != 0) ? x : 0; // load $reg, x // movz $reg, $0, a unsigned Opc = (N->getOpcode() == MipsISD::CMovFP_T) ? MipsISD::CMovFP_F : MipsISD::CMovFP_T; SDValue FCC = N->getOperand(1), Glue = N->getOperand(3); return DAG.getNode(Opc, SDLoc(N), ValueIfFalse.getValueType(), ValueIfFalse, FCC, ValueIfTrue, Glue); } static SDValue performANDCombine(SDNode *N, SelectionDAG &DAG, TargetLowering::DAGCombinerInfo &DCI, const MipsSubtarget &Subtarget) { if (DCI.isBeforeLegalizeOps() || !Subtarget.hasExtractInsert()) return SDValue(); SDValue FirstOperand = N->getOperand(0); unsigned FirstOperandOpc = FirstOperand.getOpcode(); SDValue Mask = N->getOperand(1); EVT ValTy = N->getValueType(0); SDLoc DL(N); uint64_t Pos = 0, SMPos, SMSize; ConstantSDNode *CN; SDValue NewOperand; unsigned Opc; // Op's second operand must be a shifted mask. if (!(CN = dyn_cast(Mask)) || !isShiftedMask(CN->getZExtValue(), SMPos, SMSize)) return SDValue(); if (FirstOperandOpc == ISD::SRA || FirstOperandOpc == ISD::SRL) { // Pattern match EXT. // $dst = and ((sra or srl) $src , pos), (2**size - 1) // => ext $dst, $src, pos, size // The second operand of the shift must be an immediate. if (!(CN = dyn_cast(FirstOperand.getOperand(1)))) return SDValue(); Pos = CN->getZExtValue(); // Return if the shifted mask does not start at bit 0 or the sum of its size // and Pos exceeds the word's size. if (SMPos != 0 || Pos + SMSize > ValTy.getSizeInBits()) return SDValue(); Opc = MipsISD::Ext; NewOperand = FirstOperand.getOperand(0); } else if (FirstOperandOpc == ISD::SHL && Subtarget.hasCnMips()) { // Pattern match CINS. // $dst = and (shl $src , pos), mask // => cins $dst, $src, pos, size // mask is a shifted mask with consecutive 1's, pos = shift amount, // size = population count. // The second operand of the shift must be an immediate. if (!(CN = dyn_cast(FirstOperand.getOperand(1)))) return SDValue(); Pos = CN->getZExtValue(); if (SMPos != Pos || Pos >= ValTy.getSizeInBits() || SMSize >= 32 || Pos + SMSize > ValTy.getSizeInBits()) return SDValue(); NewOperand = FirstOperand.getOperand(0); // SMSize is 'location' (position) in this case, not size. SMSize--; Opc = MipsISD::CIns; } else { // Pattern match EXT. // $dst = and $src, (2**size - 1) , if size > 16 // => ext $dst, $src, pos, size , pos = 0 // If the mask is <= 0xffff, andi can be used instead. if (CN->getZExtValue() <= 0xffff) return SDValue(); // Return if the mask doesn't start at position 0. if (SMPos) return SDValue(); Opc = MipsISD::Ext; NewOperand = FirstOperand; } return DAG.getNode(Opc, DL, ValTy, NewOperand, DAG.getConstant(Pos, DL, MVT::i32), DAG.getConstant(SMSize, DL, MVT::i32)); } static SDValue performORCombine(SDNode *N, SelectionDAG &DAG, TargetLowering::DAGCombinerInfo &DCI, const MipsSubtarget &Subtarget) { // Pattern match INS. // $dst = or (and $src1 , mask0), (and (shl $src, pos), mask1), // where mask1 = (2**size - 1) << pos, mask0 = ~mask1 // => ins $dst, $src, size, pos, $src1 if (DCI.isBeforeLegalizeOps() || !Subtarget.hasExtractInsert()) return SDValue(); SDValue And0 = N->getOperand(0), And1 = N->getOperand(1); uint64_t SMPos0, SMSize0, SMPos1, SMSize1; ConstantSDNode *CN, *CN1; // See if Op's first operand matches (and $src1 , mask0). if (And0.getOpcode() != ISD::AND) return SDValue(); if (!(CN = dyn_cast(And0.getOperand(1))) || !isShiftedMask(~CN->getSExtValue(), SMPos0, SMSize0)) return SDValue(); // See if Op's second operand matches (and (shl $src, pos), mask1). if (And1.getOpcode() == ISD::AND && And1.getOperand(0).getOpcode() == ISD::SHL) { if (!(CN = dyn_cast(And1.getOperand(1))) || !isShiftedMask(CN->getZExtValue(), SMPos1, SMSize1)) return SDValue(); // The shift masks must have the same position and size. if (SMPos0 != SMPos1 || SMSize0 != SMSize1) return SDValue(); SDValue Shl = And1.getOperand(0); if (!(CN = dyn_cast(Shl.getOperand(1)))) return SDValue(); unsigned Shamt = CN->getZExtValue(); // Return if the shift amount and the first bit position of mask are not the // same. EVT ValTy = N->getValueType(0); if ((Shamt != SMPos0) || (SMPos0 + SMSize0 > ValTy.getSizeInBits())) return SDValue(); SDLoc DL(N); return DAG.getNode(MipsISD::Ins, DL, ValTy, Shl.getOperand(0), DAG.getConstant(SMPos0, DL, MVT::i32), DAG.getConstant(SMSize0, DL, MVT::i32), And0.getOperand(0)); } else { // Pattern match DINS. // $dst = or (and $src, mask0), mask1 // where mask0 = ((1 << SMSize0) -1) << SMPos0 // => dins $dst, $src, pos, size if (~CN->getSExtValue() == ((((int64_t)1 << SMSize0) - 1) << SMPos0) && ((SMSize0 + SMPos0 <= 64 && Subtarget.hasMips64r2()) || (SMSize0 + SMPos0 <= 32))) { // Check if AND instruction has constant as argument bool isConstCase = And1.getOpcode() != ISD::AND; if (And1.getOpcode() == ISD::AND) { if (!(CN1 = dyn_cast(And1->getOperand(1)))) return SDValue(); } else { if (!(CN1 = dyn_cast(N->getOperand(1)))) return SDValue(); } // Don't generate INS if constant OR operand doesn't fit into bits // cleared by constant AND operand. if (CN->getSExtValue() & CN1->getSExtValue()) return SDValue(); SDLoc DL(N); EVT ValTy = N->getOperand(0)->getValueType(0); SDValue Const1; SDValue SrlX; if (!isConstCase) { Const1 = DAG.getConstant(SMPos0, DL, MVT::i32); SrlX = DAG.getNode(ISD::SRL, DL, And1->getValueType(0), And1, Const1); } return DAG.getNode( MipsISD::Ins, DL, N->getValueType(0), isConstCase ? DAG.getConstant(CN1->getSExtValue() >> SMPos0, DL, ValTy) : SrlX, DAG.getConstant(SMPos0, DL, MVT::i32), DAG.getConstant(ValTy.getSizeInBits() / 8 < 8 ? SMSize0 & 31 : SMSize0, DL, MVT::i32), And0->getOperand(0)); } return SDValue(); } } static SDValue performMADD_MSUBCombine(SDNode *ROOTNode, SelectionDAG &CurDAG, const MipsSubtarget &Subtarget) { // ROOTNode must have a multiplication as an operand for the match to be // successful. if (ROOTNode->getOperand(0).getOpcode() != ISD::MUL && ROOTNode->getOperand(1).getOpcode() != ISD::MUL) return SDValue(); // We don't handle vector types here. if (ROOTNode->getValueType(0).isVector()) return SDValue(); // For MIPS64, madd / msub instructions are inefficent to use with 64 bit // arithmetic. E.g. // (add (mul a b) c) => // let res = (madd (mthi (drotr c 32))x(mtlo c) a b) in // MIPS64: (or (dsll (mfhi res) 32) (dsrl (dsll (mflo res) 32) 32) // or // MIPS64R2: (dins (mflo res) (mfhi res) 32 32) // // The overhead of setting up the Hi/Lo registers and reassembling the // result makes this a dubious optimzation for MIPS64. The core of the // problem is that Hi/Lo contain the upper and lower 32 bits of the // operand and result. // // It requires a chain of 4 add/mul for MIPS64R2 to get better code // density than doing it naively, 5 for MIPS64. Additionally, using // madd/msub on MIPS64 requires the operands actually be 32 bit sign // extended operands, not true 64 bit values. // // FIXME: For the moment, disable this completely for MIPS64. if (Subtarget.hasMips64()) return SDValue(); SDValue Mult = ROOTNode->getOperand(0).getOpcode() == ISD::MUL ? ROOTNode->getOperand(0) : ROOTNode->getOperand(1); SDValue AddOperand = ROOTNode->getOperand(0).getOpcode() == ISD::MUL ? ROOTNode->getOperand(1) : ROOTNode->getOperand(0); // Transform this to a MADD only if the user of this node is the add. // If there are other users of the mul, this function returns here. if (!Mult.hasOneUse()) return SDValue(); // maddu and madd are unusual instructions in that on MIPS64 bits 63..31 // must be in canonical form, i.e. sign extended. For MIPS32, the operands // of the multiply must have 32 or more sign bits, otherwise we cannot // perform this optimization. We have to check this here as we're performing // this optimization pre-legalization. SDValue MultLHS = Mult->getOperand(0); SDValue MultRHS = Mult->getOperand(1); bool IsSigned = MultLHS->getOpcode() == ISD::SIGN_EXTEND && MultRHS->getOpcode() == ISD::SIGN_EXTEND; bool IsUnsigned = MultLHS->getOpcode() == ISD::ZERO_EXTEND && MultRHS->getOpcode() == ISD::ZERO_EXTEND; if (!IsSigned && !IsUnsigned) return SDValue(); // Initialize accumulator. SDLoc DL(ROOTNode); SDValue TopHalf; SDValue BottomHalf; BottomHalf = CurDAG.getNode(ISD::EXTRACT_ELEMENT, DL, MVT::i32, AddOperand, CurDAG.getIntPtrConstant(0, DL)); TopHalf = CurDAG.getNode(ISD::EXTRACT_ELEMENT, DL, MVT::i32, AddOperand, CurDAG.getIntPtrConstant(1, DL)); SDValue ACCIn = CurDAG.getNode(MipsISD::MTLOHI, DL, MVT::Untyped, BottomHalf, TopHalf); // Create MipsMAdd(u) / MipsMSub(u) node. bool IsAdd = ROOTNode->getOpcode() == ISD::ADD; unsigned Opcode = IsAdd ? (IsUnsigned ? MipsISD::MAddu : MipsISD::MAdd) : (IsUnsigned ? MipsISD::MSubu : MipsISD::MSub); SDValue MAddOps[3] = { CurDAG.getNode(ISD::TRUNCATE, DL, MVT::i32, Mult->getOperand(0)), CurDAG.getNode(ISD::TRUNCATE, DL, MVT::i32, Mult->getOperand(1)), ACCIn}; EVT VTs[2] = {MVT::i32, MVT::i32}; SDValue MAdd = CurDAG.getNode(Opcode, DL, VTs, MAddOps); SDValue ResLo = CurDAG.getNode(MipsISD::MFLO, DL, MVT::i32, MAdd); SDValue ResHi = CurDAG.getNode(MipsISD::MFHI, DL, MVT::i32, MAdd); SDValue Combined = CurDAG.getNode(ISD::BUILD_PAIR, DL, MVT::i64, ResLo, ResHi); return Combined; } static SDValue performSUBCombine(SDNode *N, SelectionDAG &DAG, TargetLowering::DAGCombinerInfo &DCI, const MipsSubtarget &Subtarget) { // (sub v0 (mul v1, v2)) => (msub v1, v2, v0) if (DCI.isBeforeLegalizeOps()) { if (Subtarget.hasMips32() && !Subtarget.hasMips32r6() && !Subtarget.inMips16Mode() && N->getValueType(0) == MVT::i64) return performMADD_MSUBCombine(N, DAG, Subtarget); return SDValue(); } return SDValue(); } static SDValue performADDCombine(SDNode *N, SelectionDAG &DAG, TargetLowering::DAGCombinerInfo &DCI, const MipsSubtarget &Subtarget) { // (add v0 (mul v1, v2)) => (madd v1, v2, v0) if (DCI.isBeforeLegalizeOps()) { if (Subtarget.hasMips32() && !Subtarget.hasMips32r6() && !Subtarget.inMips16Mode() && N->getValueType(0) == MVT::i64) return performMADD_MSUBCombine(N, DAG, Subtarget); return SDValue(); } // (add v0, (add v1, abs_lo(tjt))) => (add (add v0, v1), abs_lo(tjt)) SDValue Add = N->getOperand(1); if (Add.getOpcode() != ISD::ADD) return SDValue(); SDValue Lo = Add.getOperand(1); if ((Lo.getOpcode() != MipsISD::Lo) || (Lo.getOperand(0).getOpcode() != ISD::TargetJumpTable)) return SDValue(); EVT ValTy = N->getValueType(0); SDLoc DL(N); SDValue Add1 = DAG.getNode(ISD::ADD, DL, ValTy, N->getOperand(0), Add.getOperand(0)); return DAG.getNode(ISD::ADD, DL, ValTy, Add1, Lo); } static SDValue performSHLCombine(SDNode *N, SelectionDAG &DAG, TargetLowering::DAGCombinerInfo &DCI, const MipsSubtarget &Subtarget) { // Pattern match CINS. // $dst = shl (and $src , imm), pos // => cins $dst, $src, pos, size if (DCI.isBeforeLegalizeOps() || !Subtarget.hasCnMips()) return SDValue(); SDValue FirstOperand = N->getOperand(0); unsigned FirstOperandOpc = FirstOperand.getOpcode(); SDValue SecondOperand = N->getOperand(1); EVT ValTy = N->getValueType(0); SDLoc DL(N); uint64_t Pos = 0, SMPos, SMSize; ConstantSDNode *CN; SDValue NewOperand; // The second operand of the shift must be an immediate. if (!(CN = dyn_cast(SecondOperand))) return SDValue(); Pos = CN->getZExtValue(); if (Pos >= ValTy.getSizeInBits()) return SDValue(); if (FirstOperandOpc != ISD::AND) return SDValue(); // AND's second operand must be a shifted mask. if (!(CN = dyn_cast(FirstOperand.getOperand(1))) || !isShiftedMask(CN->getZExtValue(), SMPos, SMSize)) return SDValue(); // Return if the shifted mask does not start at bit 0 or the sum of its size // and Pos exceeds the word's size. if (SMPos != 0 || SMSize > 32 || Pos + SMSize > ValTy.getSizeInBits()) return SDValue(); NewOperand = FirstOperand.getOperand(0); // SMSize is 'location' (position) in this case, not size. SMSize--; return DAG.getNode(MipsISD::CIns, DL, ValTy, NewOperand, DAG.getConstant(Pos, DL, MVT::i32), DAG.getConstant(SMSize, DL, MVT::i32)); } SDValue MipsTargetLowering::PerformDAGCombine(SDNode *N, DAGCombinerInfo &DCI) const { SelectionDAG &DAG = DCI.DAG; unsigned Opc = N->getOpcode(); switch (Opc) { default: break; case ISD::SDIVREM: case ISD::UDIVREM: return performDivRemCombine(N, DAG, DCI, Subtarget); case ISD::SELECT: return performSELECTCombine(N, DAG, DCI, Subtarget); case MipsISD::CMovFP_F: case MipsISD::CMovFP_T: return performCMovFPCombine(N, DAG, DCI, Subtarget); case ISD::AND: return performANDCombine(N, DAG, DCI, Subtarget); case ISD::OR: return performORCombine(N, DAG, DCI, Subtarget); case ISD::ADD: return performADDCombine(N, DAG, DCI, Subtarget); case ISD::SHL: return performSHLCombine(N, DAG, DCI, Subtarget); case ISD::SUB: return performSUBCombine(N, DAG, DCI, Subtarget); } return SDValue(); } bool MipsTargetLowering::isCheapToSpeculateCttz() const { return Subtarget.hasMips32(); } bool MipsTargetLowering::isCheapToSpeculateCtlz() const { return Subtarget.hasMips32(); } void MipsTargetLowering::LowerOperationWrapper(SDNode *N, SmallVectorImpl &Results, SelectionDAG &DAG) const { SDValue Res = LowerOperation(SDValue(N, 0), DAG); for (unsigned I = 0, E = Res->getNumValues(); I != E; ++I) Results.push_back(Res.getValue(I)); } void MipsTargetLowering::ReplaceNodeResults(SDNode *N, SmallVectorImpl &Results, SelectionDAG &DAG) const { return LowerOperationWrapper(N, Results, DAG); } SDValue MipsTargetLowering:: LowerOperation(SDValue Op, SelectionDAG &DAG) const { switch (Op.getOpcode()) { case ISD::BRCOND: return lowerBRCOND(Op, DAG); case ISD::ConstantPool: return lowerConstantPool(Op, DAG); case ISD::GlobalAddress: return lowerGlobalAddress(Op, DAG); case ISD::BlockAddress: return lowerBlockAddress(Op, DAG); case ISD::GlobalTLSAddress: return lowerGlobalTLSAddress(Op, DAG); case ISD::JumpTable: return lowerJumpTable(Op, DAG); case ISD::SELECT: return lowerSELECT(Op, DAG); case ISD::SETCC: return lowerSETCC(Op, DAG); case ISD::VASTART: return lowerVASTART(Op, DAG); case ISD::VAARG: return lowerVAARG(Op, DAG); case ISD::FCOPYSIGN: return lowerFCOPYSIGN(Op, DAG); case ISD::FRAMEADDR: return lowerFRAMEADDR(Op, DAG); case ISD::RETURNADDR: return lowerRETURNADDR(Op, DAG); case ISD::EH_RETURN: return lowerEH_RETURN(Op, DAG); case ISD::ATOMIC_FENCE: return lowerATOMIC_FENCE(Op, DAG); case ISD::SHL_PARTS: return lowerShiftLeftParts(Op, DAG); case ISD::SRA_PARTS: return lowerShiftRightParts(Op, DAG, true); case ISD::SRL_PARTS: return lowerShiftRightParts(Op, DAG, false); case ISD::LOAD: return lowerLOAD(Op, DAG); case ISD::STORE: return lowerSTORE(Op, DAG); case ISD::EH_DWARF_CFA: return lowerEH_DWARF_CFA(Op, DAG); case ISD::FP_TO_SINT: return lowerFP_TO_SINT(Op, DAG); } return SDValue(); } //===----------------------------------------------------------------------===// // Lower helper functions //===----------------------------------------------------------------------===// // addLiveIn - This helper function adds the specified physical register to the // MachineFunction as a live in value. It also creates a corresponding // virtual register for it. static unsigned addLiveIn(MachineFunction &MF, unsigned PReg, const TargetRegisterClass *RC) { unsigned VReg = MF.getRegInfo().createVirtualRegister(RC); MF.getRegInfo().addLiveIn(PReg, VReg); return VReg; } static MachineBasicBlock *insertDivByZeroTrap(MachineInstr &MI, MachineBasicBlock &MBB, const TargetInstrInfo &TII, bool Is64Bit, bool IsMicroMips) { if (NoZeroDivCheck) return &MBB; // Insert instruction "teq $divisor_reg, $zero, 7". MachineBasicBlock::iterator I(MI); MachineInstrBuilder MIB; MachineOperand &Divisor = MI.getOperand(2); MIB = BuildMI(MBB, std::next(I), MI.getDebugLoc(), TII.get(IsMicroMips ? Mips::TEQ_MM : Mips::TEQ)) .addReg(Divisor.getReg(), getKillRegState(Divisor.isKill())) .addReg(Mips::ZERO) .addImm(7); // Use the 32-bit sub-register if this is a 64-bit division. if (Is64Bit) MIB->getOperand(0).setSubReg(Mips::sub_32); // Clear Divisor's kill flag. Divisor.setIsKill(false); // We would normally delete the original instruction here but in this case // we only needed to inject an additional instruction rather than replace it. return &MBB; } MachineBasicBlock * MipsTargetLowering::EmitInstrWithCustomInserter(MachineInstr &MI, MachineBasicBlock *BB) const { switch (MI.getOpcode()) { default: llvm_unreachable("Unexpected instr type to insert"); case Mips::ATOMIC_LOAD_ADD_I8: return emitAtomicBinaryPartword(MI, BB, 1); case Mips::ATOMIC_LOAD_ADD_I16: return emitAtomicBinaryPartword(MI, BB, 2); case Mips::ATOMIC_LOAD_ADD_I32: return emitAtomicBinary(MI, BB); case Mips::ATOMIC_LOAD_ADD_I64: return emitAtomicBinary(MI, BB); case Mips::ATOMIC_LOAD_AND_I8: return emitAtomicBinaryPartword(MI, BB, 1); case Mips::ATOMIC_LOAD_AND_I16: return emitAtomicBinaryPartword(MI, BB, 2); case Mips::ATOMIC_LOAD_AND_I32: return emitAtomicBinary(MI, BB); case Mips::ATOMIC_LOAD_AND_I64: return emitAtomicBinary(MI, BB); case Mips::ATOMIC_LOAD_OR_I8: return emitAtomicBinaryPartword(MI, BB, 1); case Mips::ATOMIC_LOAD_OR_I16: return emitAtomicBinaryPartword(MI, BB, 2); case Mips::ATOMIC_LOAD_OR_I32: return emitAtomicBinary(MI, BB); case Mips::ATOMIC_LOAD_OR_I64: return emitAtomicBinary(MI, BB); case Mips::ATOMIC_LOAD_XOR_I8: return emitAtomicBinaryPartword(MI, BB, 1); case Mips::ATOMIC_LOAD_XOR_I16: return emitAtomicBinaryPartword(MI, BB, 2); case Mips::ATOMIC_LOAD_XOR_I32: return emitAtomicBinary(MI, BB); case Mips::ATOMIC_LOAD_XOR_I64: return emitAtomicBinary(MI, BB); case Mips::ATOMIC_LOAD_NAND_I8: return emitAtomicBinaryPartword(MI, BB, 1); case Mips::ATOMIC_LOAD_NAND_I16: return emitAtomicBinaryPartword(MI, BB, 2); case Mips::ATOMIC_LOAD_NAND_I32: return emitAtomicBinary(MI, BB); case Mips::ATOMIC_LOAD_NAND_I64: return emitAtomicBinary(MI, BB); case Mips::ATOMIC_LOAD_SUB_I8: return emitAtomicBinaryPartword(MI, BB, 1); case Mips::ATOMIC_LOAD_SUB_I16: return emitAtomicBinaryPartword(MI, BB, 2); case Mips::ATOMIC_LOAD_SUB_I32: return emitAtomicBinary(MI, BB); case Mips::ATOMIC_LOAD_SUB_I64: return emitAtomicBinary(MI, BB); case Mips::ATOMIC_SWAP_I8: return emitAtomicBinaryPartword(MI, BB, 1); case Mips::ATOMIC_SWAP_I16: return emitAtomicBinaryPartword(MI, BB, 2); case Mips::ATOMIC_SWAP_I32: return emitAtomicBinary(MI, BB); case Mips::ATOMIC_SWAP_I64: return emitAtomicBinary(MI, BB); case Mips::ATOMIC_CMP_SWAP_I8: return emitAtomicCmpSwapPartword(MI, BB, 1); case Mips::ATOMIC_CMP_SWAP_I16: return emitAtomicCmpSwapPartword(MI, BB, 2); case Mips::ATOMIC_CMP_SWAP_I32: return emitAtomicCmpSwap(MI, BB); case Mips::ATOMIC_CMP_SWAP_I64: return emitAtomicCmpSwap(MI, BB); case Mips::PseudoSDIV: case Mips::PseudoUDIV: case Mips::DIV: case Mips::DIVU: case Mips::MOD: case Mips::MODU: return insertDivByZeroTrap(MI, *BB, *Subtarget.getInstrInfo(), false, false); case Mips::SDIV_MM_Pseudo: case Mips::UDIV_MM_Pseudo: case Mips::SDIV_MM: case Mips::UDIV_MM: case Mips::DIV_MMR6: case Mips::DIVU_MMR6: case Mips::MOD_MMR6: case Mips::MODU_MMR6: return insertDivByZeroTrap(MI, *BB, *Subtarget.getInstrInfo(), false, true); case Mips::PseudoDSDIV: case Mips::PseudoDUDIV: case Mips::DDIV: case Mips::DDIVU: case Mips::DMOD: case Mips::DMODU: return insertDivByZeroTrap(MI, *BB, *Subtarget.getInstrInfo(), true, false); case Mips::PseudoSELECT_I: case Mips::PseudoSELECT_I64: case Mips::PseudoSELECT_S: case Mips::PseudoSELECT_D32: case Mips::PseudoSELECT_D64: return emitPseudoSELECT(MI, BB, false, Mips::BNE); case Mips::PseudoSELECTFP_F_I: case Mips::PseudoSELECTFP_F_I64: case Mips::PseudoSELECTFP_F_S: case Mips::PseudoSELECTFP_F_D32: case Mips::PseudoSELECTFP_F_D64: return emitPseudoSELECT(MI, BB, true, Mips::BC1F); case Mips::PseudoSELECTFP_T_I: case Mips::PseudoSELECTFP_T_I64: case Mips::PseudoSELECTFP_T_S: case Mips::PseudoSELECTFP_T_D32: case Mips::PseudoSELECTFP_T_D64: return emitPseudoSELECT(MI, BB, true, Mips::BC1T); case Mips::PseudoD_SELECT_I: case Mips::PseudoD_SELECT_I64: return emitPseudoD_SELECT(MI, BB); } } // This function also handles Mips::ATOMIC_SWAP_I32 (when BinOpcode == 0), and // Mips::ATOMIC_LOAD_NAND_I32 (when Nand == true) MachineBasicBlock * MipsTargetLowering::emitAtomicBinary(MachineInstr &MI, MachineBasicBlock *BB) const { MachineFunction *MF = BB->getParent(); MachineRegisterInfo &RegInfo = MF->getRegInfo(); const TargetInstrInfo *TII = Subtarget.getInstrInfo(); DebugLoc DL = MI.getDebugLoc(); unsigned AtomicOp; switch (MI.getOpcode()) { case Mips::ATOMIC_LOAD_ADD_I32: AtomicOp = Mips::ATOMIC_LOAD_ADD_I32_POSTRA; break; case Mips::ATOMIC_LOAD_SUB_I32: AtomicOp = Mips::ATOMIC_LOAD_SUB_I32_POSTRA; break; case Mips::ATOMIC_LOAD_AND_I32: AtomicOp = Mips::ATOMIC_LOAD_AND_I32_POSTRA; break; case Mips::ATOMIC_LOAD_OR_I32: AtomicOp = Mips::ATOMIC_LOAD_OR_I32_POSTRA; break; case Mips::ATOMIC_LOAD_XOR_I32: AtomicOp = Mips::ATOMIC_LOAD_XOR_I32_POSTRA; break; case Mips::ATOMIC_LOAD_NAND_I32: AtomicOp = Mips::ATOMIC_LOAD_NAND_I32_POSTRA; break; case Mips::ATOMIC_SWAP_I32: AtomicOp = Mips::ATOMIC_SWAP_I32_POSTRA; break; case Mips::ATOMIC_LOAD_ADD_I64: AtomicOp = Mips::ATOMIC_LOAD_ADD_I64_POSTRA; break; case Mips::ATOMIC_LOAD_SUB_I64: AtomicOp = Mips::ATOMIC_LOAD_SUB_I64_POSTRA; break; case Mips::ATOMIC_LOAD_AND_I64: AtomicOp = Mips::ATOMIC_LOAD_AND_I64_POSTRA; break; case Mips::ATOMIC_LOAD_OR_I64: AtomicOp = Mips::ATOMIC_LOAD_OR_I64_POSTRA; break; case Mips::ATOMIC_LOAD_XOR_I64: AtomicOp = Mips::ATOMIC_LOAD_XOR_I64_POSTRA; break; case Mips::ATOMIC_LOAD_NAND_I64: AtomicOp = Mips::ATOMIC_LOAD_NAND_I64_POSTRA; break; case Mips::ATOMIC_SWAP_I64: AtomicOp = Mips::ATOMIC_SWAP_I64_POSTRA; break; default: llvm_unreachable("Unknown pseudo atomic for replacement!"); } unsigned OldVal = MI.getOperand(0).getReg(); unsigned Ptr = MI.getOperand(1).getReg(); unsigned Incr = MI.getOperand(2).getReg(); unsigned Scratch = RegInfo.createVirtualRegister(RegInfo.getRegClass(OldVal)); MachineBasicBlock::iterator II(MI); // The scratch registers here with the EarlyClobber | Define | Implicit // flags is used to persuade the register allocator and the machine // verifier to accept the usage of this register. This has to be a real // register which has an UNDEF value but is dead after the instruction which // is unique among the registers chosen for the instruction. // The EarlyClobber flag has the semantic properties that the operand it is // attached to is clobbered before the rest of the inputs are read. Hence it // must be unique among the operands to the instruction. // The Define flag is needed to coerce the machine verifier that an Undef // value isn't a problem. // The Dead flag is needed as the value in scratch isn't used by any other // instruction. Kill isn't used as Dead is more precise. // The implicit flag is here due to the interaction between the other flags // and the machine verifier. // For correctness purpose, a new pseudo is introduced here. We need this // new pseudo, so that FastRegisterAllocator does not see an ll/sc sequence // that is spread over >1 basic blocks. A register allocator which // introduces (or any codegen infact) a store, can violate the expectations // of the hardware. // // An atomic read-modify-write sequence starts with a linked load // instruction and ends with a store conditional instruction. The atomic // read-modify-write sequence fails if any of the following conditions // occur between the execution of ll and sc: // * A coherent store is completed by another process or coherent I/O // module into the block of synchronizable physical memory containing // the word. The size and alignment of the block is // implementation-dependent. // * A coherent store is executed between an LL and SC sequence on the // same processor to the block of synchornizable physical memory // containing the word. // unsigned PtrCopy = RegInfo.createVirtualRegister(RegInfo.getRegClass(Ptr)); unsigned IncrCopy = RegInfo.createVirtualRegister(RegInfo.getRegClass(Incr)); BuildMI(*BB, II, DL, TII->get(Mips::COPY), IncrCopy).addReg(Incr); BuildMI(*BB, II, DL, TII->get(Mips::COPY), PtrCopy).addReg(Ptr); BuildMI(*BB, II, DL, TII->get(AtomicOp)) .addReg(OldVal, RegState::Define | RegState::EarlyClobber) .addReg(PtrCopy) .addReg(IncrCopy) .addReg(Scratch, RegState::Define | RegState::EarlyClobber | RegState::Implicit | RegState::Dead); MI.eraseFromParent(); return BB; } MachineBasicBlock *MipsTargetLowering::emitSignExtendToI32InReg( MachineInstr &MI, MachineBasicBlock *BB, unsigned Size, unsigned DstReg, unsigned SrcReg) const { const TargetInstrInfo *TII = Subtarget.getInstrInfo(); const DebugLoc &DL = MI.getDebugLoc(); if (Subtarget.hasMips32r2() && Size == 1) { BuildMI(BB, DL, TII->get(Mips::SEB), DstReg).addReg(SrcReg); return BB; } if (Subtarget.hasMips32r2() && Size == 2) { BuildMI(BB, DL, TII->get(Mips::SEH), DstReg).addReg(SrcReg); return BB; } MachineFunction *MF = BB->getParent(); MachineRegisterInfo &RegInfo = MF->getRegInfo(); const TargetRegisterClass *RC = getRegClassFor(MVT::i32); unsigned ScrReg = RegInfo.createVirtualRegister(RC); assert(Size < 32); int64_t ShiftImm = 32 - (Size * 8); BuildMI(BB, DL, TII->get(Mips::SLL), ScrReg).addReg(SrcReg).addImm(ShiftImm); BuildMI(BB, DL, TII->get(Mips::SRA), DstReg).addReg(ScrReg).addImm(ShiftImm); return BB; } MachineBasicBlock *MipsTargetLowering::emitAtomicBinaryPartword( MachineInstr &MI, MachineBasicBlock *BB, unsigned Size) const { assert((Size == 1 || Size == 2) && "Unsupported size for EmitAtomicBinaryPartial."); MachineFunction *MF = BB->getParent(); MachineRegisterInfo &RegInfo = MF->getRegInfo(); const TargetRegisterClass *RC = getRegClassFor(MVT::i32); const bool ArePtrs64bit = ABI.ArePtrs64bit(); const TargetRegisterClass *RCp = getRegClassFor(ArePtrs64bit ? MVT::i64 : MVT::i32); const TargetInstrInfo *TII = Subtarget.getInstrInfo(); DebugLoc DL = MI.getDebugLoc(); unsigned Dest = MI.getOperand(0).getReg(); unsigned Ptr = MI.getOperand(1).getReg(); unsigned Incr = MI.getOperand(2).getReg(); unsigned AlignedAddr = RegInfo.createVirtualRegister(RCp); unsigned ShiftAmt = RegInfo.createVirtualRegister(RC); unsigned Mask = RegInfo.createVirtualRegister(RC); unsigned Mask2 = RegInfo.createVirtualRegister(RC); unsigned Incr2 = RegInfo.createVirtualRegister(RC); unsigned MaskLSB2 = RegInfo.createVirtualRegister(RCp); unsigned PtrLSB2 = RegInfo.createVirtualRegister(RC); unsigned MaskUpper = RegInfo.createVirtualRegister(RC); unsigned Scratch = RegInfo.createVirtualRegister(RC); unsigned Scratch2 = RegInfo.createVirtualRegister(RC); unsigned Scratch3 = RegInfo.createVirtualRegister(RC); unsigned AtomicOp = 0; switch (MI.getOpcode()) { case Mips::ATOMIC_LOAD_NAND_I8: AtomicOp = Mips::ATOMIC_LOAD_NAND_I8_POSTRA; break; case Mips::ATOMIC_LOAD_NAND_I16: AtomicOp = Mips::ATOMIC_LOAD_NAND_I16_POSTRA; break; case Mips::ATOMIC_SWAP_I8: AtomicOp = Mips::ATOMIC_SWAP_I8_POSTRA; break; case Mips::ATOMIC_SWAP_I16: AtomicOp = Mips::ATOMIC_SWAP_I16_POSTRA; break; case Mips::ATOMIC_LOAD_ADD_I8: AtomicOp = Mips::ATOMIC_LOAD_ADD_I8_POSTRA; break; case Mips::ATOMIC_LOAD_ADD_I16: AtomicOp = Mips::ATOMIC_LOAD_ADD_I16_POSTRA; break; case Mips::ATOMIC_LOAD_SUB_I8: AtomicOp = Mips::ATOMIC_LOAD_SUB_I8_POSTRA; break; case Mips::ATOMIC_LOAD_SUB_I16: AtomicOp = Mips::ATOMIC_LOAD_SUB_I16_POSTRA; break; case Mips::ATOMIC_LOAD_AND_I8: AtomicOp = Mips::ATOMIC_LOAD_AND_I8_POSTRA; break; case Mips::ATOMIC_LOAD_AND_I16: AtomicOp = Mips::ATOMIC_LOAD_AND_I16_POSTRA; break; case Mips::ATOMIC_LOAD_OR_I8: AtomicOp = Mips::ATOMIC_LOAD_OR_I8_POSTRA; break; case Mips::ATOMIC_LOAD_OR_I16: AtomicOp = Mips::ATOMIC_LOAD_OR_I16_POSTRA; break; case Mips::ATOMIC_LOAD_XOR_I8: AtomicOp = Mips::ATOMIC_LOAD_XOR_I8_POSTRA; break; case Mips::ATOMIC_LOAD_XOR_I16: AtomicOp = Mips::ATOMIC_LOAD_XOR_I16_POSTRA; break; default: llvm_unreachable("Unknown subword atomic pseudo for expansion!"); } // insert new blocks after the current block const BasicBlock *LLVM_BB = BB->getBasicBlock(); MachineBasicBlock *exitMBB = MF->CreateMachineBasicBlock(LLVM_BB); MachineFunction::iterator It = ++BB->getIterator(); MF->insert(It, exitMBB); // Transfer the remainder of BB and its successor edges to exitMBB. exitMBB->splice(exitMBB->begin(), BB, std::next(MachineBasicBlock::iterator(MI)), BB->end()); exitMBB->transferSuccessorsAndUpdatePHIs(BB); BB->addSuccessor(exitMBB, BranchProbability::getOne()); // thisMBB: // addiu masklsb2,$0,-4 # 0xfffffffc // and alignedaddr,ptr,masklsb2 // andi ptrlsb2,ptr,3 // sll shiftamt,ptrlsb2,3 // ori maskupper,$0,255 # 0xff // sll mask,maskupper,shiftamt // nor mask2,$0,mask // sll incr2,incr,shiftamt int64_t MaskImm = (Size == 1) ? 255 : 65535; BuildMI(BB, DL, TII->get(ABI.GetPtrAddiuOp()), MaskLSB2) .addReg(ABI.GetNullPtr()).addImm(-4); BuildMI(BB, DL, TII->get(ABI.GetPtrAndOp()), AlignedAddr) .addReg(Ptr).addReg(MaskLSB2); BuildMI(BB, DL, TII->get(Mips::ANDi), PtrLSB2) .addReg(Ptr, 0, ArePtrs64bit ? Mips::sub_32 : 0).addImm(3); if (Subtarget.isLittle()) { BuildMI(BB, DL, TII->get(Mips::SLL), ShiftAmt).addReg(PtrLSB2).addImm(3); } else { unsigned Off = RegInfo.createVirtualRegister(RC); BuildMI(BB, DL, TII->get(Mips::XORi), Off) .addReg(PtrLSB2).addImm((Size == 1) ? 3 : 2); BuildMI(BB, DL, TII->get(Mips::SLL), ShiftAmt).addReg(Off).addImm(3); } BuildMI(BB, DL, TII->get(Mips::ORi), MaskUpper) .addReg(Mips::ZERO).addImm(MaskImm); BuildMI(BB, DL, TII->get(Mips::SLLV), Mask) .addReg(MaskUpper).addReg(ShiftAmt); BuildMI(BB, DL, TII->get(Mips::NOR), Mask2).addReg(Mips::ZERO).addReg(Mask); BuildMI(BB, DL, TII->get(Mips::SLLV), Incr2).addReg(Incr).addReg(ShiftAmt); // The purposes of the flags on the scratch registers is explained in // emitAtomicBinary. In summary, we need a scratch register which is going to // be undef, that is unique among registers chosen for the instruction. BuildMI(BB, DL, TII->get(AtomicOp)) .addReg(Dest, RegState::Define | RegState::EarlyClobber) .addReg(AlignedAddr) .addReg(Incr2) .addReg(Mask) .addReg(Mask2) .addReg(ShiftAmt) .addReg(Scratch, RegState::EarlyClobber | RegState::Define | RegState::Dead | RegState::Implicit) .addReg(Scratch2, RegState::EarlyClobber | RegState::Define | RegState::Dead | RegState::Implicit) .addReg(Scratch3, RegState::EarlyClobber | RegState::Define | RegState::Dead | RegState::Implicit); MI.eraseFromParent(); // The instruction is gone now. return exitMBB; } // Lower atomic compare and swap to a pseudo instruction, taking care to // define a scratch register for the pseudo instruction's expansion. The // instruction is expanded after the register allocator as to prevent // the insertion of stores between the linked load and the store conditional. MachineBasicBlock * MipsTargetLowering::emitAtomicCmpSwap(MachineInstr &MI, MachineBasicBlock *BB) const { assert((MI.getOpcode() == Mips::ATOMIC_CMP_SWAP_I32 || MI.getOpcode() == Mips::ATOMIC_CMP_SWAP_I64) && "Unsupported atomic psseudo for EmitAtomicCmpSwap."); const unsigned Size = MI.getOpcode() == Mips::ATOMIC_CMP_SWAP_I32 ? 4 : 8; MachineFunction *MF = BB->getParent(); MachineRegisterInfo &MRI = MF->getRegInfo(); const TargetRegisterClass *RC = getRegClassFor(MVT::getIntegerVT(Size * 8)); const TargetInstrInfo *TII = Subtarget.getInstrInfo(); DebugLoc DL = MI.getDebugLoc(); unsigned AtomicOp = MI.getOpcode() == Mips::ATOMIC_CMP_SWAP_I32 ? Mips::ATOMIC_CMP_SWAP_I32_POSTRA : Mips::ATOMIC_CMP_SWAP_I64_POSTRA; unsigned Dest = MI.getOperand(0).getReg(); unsigned Ptr = MI.getOperand(1).getReg(); unsigned OldVal = MI.getOperand(2).getReg(); unsigned NewVal = MI.getOperand(3).getReg(); unsigned Scratch = MRI.createVirtualRegister(RC); MachineBasicBlock::iterator II(MI); // We need to create copies of the various registers and kill them at the // atomic pseudo. If the copies are not made, when the atomic is expanded // after fast register allocation, the spills will end up outside of the // blocks that their values are defined in, causing livein errors. unsigned DestCopy = MRI.createVirtualRegister(MRI.getRegClass(Dest)); unsigned PtrCopy = MRI.createVirtualRegister(MRI.getRegClass(Ptr)); unsigned OldValCopy = MRI.createVirtualRegister(MRI.getRegClass(OldVal)); unsigned NewValCopy = MRI.createVirtualRegister(MRI.getRegClass(NewVal)); BuildMI(*BB, II, DL, TII->get(Mips::COPY), DestCopy).addReg(Dest); BuildMI(*BB, II, DL, TII->get(Mips::COPY), PtrCopy).addReg(Ptr); BuildMI(*BB, II, DL, TII->get(Mips::COPY), OldValCopy).addReg(OldVal); BuildMI(*BB, II, DL, TII->get(Mips::COPY), NewValCopy).addReg(NewVal); // The purposes of the flags on the scratch registers is explained in // emitAtomicBinary. In summary, we need a scratch register which is going to // be undef, that is unique among registers chosen for the instruction. BuildMI(*BB, II, DL, TII->get(AtomicOp)) .addReg(Dest, RegState::Define | RegState::EarlyClobber) .addReg(PtrCopy, RegState::Kill) .addReg(OldValCopy, RegState::Kill) .addReg(NewValCopy, RegState::Kill) .addReg(Scratch, RegState::EarlyClobber | RegState::Define | RegState::Dead | RegState::Implicit); MI.eraseFromParent(); // The instruction is gone now. return BB; } MachineBasicBlock *MipsTargetLowering::emitAtomicCmpSwapPartword( MachineInstr &MI, MachineBasicBlock *BB, unsigned Size) const { assert((Size == 1 || Size == 2) && "Unsupported size for EmitAtomicCmpSwapPartial."); MachineFunction *MF = BB->getParent(); MachineRegisterInfo &RegInfo = MF->getRegInfo(); const TargetRegisterClass *RC = getRegClassFor(MVT::i32); const bool ArePtrs64bit = ABI.ArePtrs64bit(); const TargetRegisterClass *RCp = getRegClassFor(ArePtrs64bit ? MVT::i64 : MVT::i32); const TargetInstrInfo *TII = Subtarget.getInstrInfo(); DebugLoc DL = MI.getDebugLoc(); unsigned Dest = MI.getOperand(0).getReg(); unsigned Ptr = MI.getOperand(1).getReg(); unsigned CmpVal = MI.getOperand(2).getReg(); unsigned NewVal = MI.getOperand(3).getReg(); unsigned AlignedAddr = RegInfo.createVirtualRegister(RCp); unsigned ShiftAmt = RegInfo.createVirtualRegister(RC); unsigned Mask = RegInfo.createVirtualRegister(RC); unsigned Mask2 = RegInfo.createVirtualRegister(RC); unsigned ShiftedCmpVal = RegInfo.createVirtualRegister(RC); unsigned ShiftedNewVal = RegInfo.createVirtualRegister(RC); unsigned MaskLSB2 = RegInfo.createVirtualRegister(RCp); unsigned PtrLSB2 = RegInfo.createVirtualRegister(RC); unsigned MaskUpper = RegInfo.createVirtualRegister(RC); unsigned MaskedCmpVal = RegInfo.createVirtualRegister(RC); unsigned MaskedNewVal = RegInfo.createVirtualRegister(RC); unsigned AtomicOp = MI.getOpcode() == Mips::ATOMIC_CMP_SWAP_I8 ? Mips::ATOMIC_CMP_SWAP_I8_POSTRA : Mips::ATOMIC_CMP_SWAP_I16_POSTRA; // The scratch registers here with the EarlyClobber | Define | Dead | Implicit // flags are used to coerce the register allocator and the machine verifier to // accept the usage of these registers. // The EarlyClobber flag has the semantic properties that the operand it is // attached to is clobbered before the rest of the inputs are read. Hence it // must be unique among the operands to the instruction. // The Define flag is needed to coerce the machine verifier that an Undef // value isn't a problem. // The Dead flag is needed as the value in scratch isn't used by any other // instruction. Kill isn't used as Dead is more precise. unsigned Scratch = RegInfo.createVirtualRegister(RC); unsigned Scratch2 = RegInfo.createVirtualRegister(RC); // insert new blocks after the current block const BasicBlock *LLVM_BB = BB->getBasicBlock(); MachineBasicBlock *exitMBB = MF->CreateMachineBasicBlock(LLVM_BB); MachineFunction::iterator It = ++BB->getIterator(); MF->insert(It, exitMBB); // Transfer the remainder of BB and its successor edges to exitMBB. exitMBB->splice(exitMBB->begin(), BB, std::next(MachineBasicBlock::iterator(MI)), BB->end()); exitMBB->transferSuccessorsAndUpdatePHIs(BB); BB->addSuccessor(exitMBB, BranchProbability::getOne()); // thisMBB: // addiu masklsb2,$0,-4 # 0xfffffffc // and alignedaddr,ptr,masklsb2 // andi ptrlsb2,ptr,3 // xori ptrlsb2,ptrlsb2,3 # Only for BE // sll shiftamt,ptrlsb2,3 // ori maskupper,$0,255 # 0xff // sll mask,maskupper,shiftamt // nor mask2,$0,mask // andi maskedcmpval,cmpval,255 // sll shiftedcmpval,maskedcmpval,shiftamt // andi maskednewval,newval,255 // sll shiftednewval,maskednewval,shiftamt int64_t MaskImm = (Size == 1) ? 255 : 65535; BuildMI(BB, DL, TII->get(ArePtrs64bit ? Mips::DADDiu : Mips::ADDiu), MaskLSB2) .addReg(ABI.GetNullPtr()).addImm(-4); BuildMI(BB, DL, TII->get(ArePtrs64bit ? Mips::AND64 : Mips::AND), AlignedAddr) .addReg(Ptr).addReg(MaskLSB2); BuildMI(BB, DL, TII->get(Mips::ANDi), PtrLSB2) .addReg(Ptr, 0, ArePtrs64bit ? Mips::sub_32 : 0).addImm(3); if (Subtarget.isLittle()) { BuildMI(BB, DL, TII->get(Mips::SLL), ShiftAmt).addReg(PtrLSB2).addImm(3); } else { unsigned Off = RegInfo.createVirtualRegister(RC); BuildMI(BB, DL, TII->get(Mips::XORi), Off) .addReg(PtrLSB2).addImm((Size == 1) ? 3 : 2); BuildMI(BB, DL, TII->get(Mips::SLL), ShiftAmt).addReg(Off).addImm(3); } BuildMI(BB, DL, TII->get(Mips::ORi), MaskUpper) .addReg(Mips::ZERO).addImm(MaskImm); BuildMI(BB, DL, TII->get(Mips::SLLV), Mask) .addReg(MaskUpper).addReg(ShiftAmt); BuildMI(BB, DL, TII->get(Mips::NOR), Mask2).addReg(Mips::ZERO).addReg(Mask); BuildMI(BB, DL, TII->get(Mips::ANDi), MaskedCmpVal) .addReg(CmpVal).addImm(MaskImm); BuildMI(BB, DL, TII->get(Mips::SLLV), ShiftedCmpVal) .addReg(MaskedCmpVal).addReg(ShiftAmt); BuildMI(BB, DL, TII->get(Mips::ANDi), MaskedNewVal) .addReg(NewVal).addImm(MaskImm); BuildMI(BB, DL, TII->get(Mips::SLLV), ShiftedNewVal) .addReg(MaskedNewVal).addReg(ShiftAmt); // The purposes of the flags on the scratch registers are explained in // emitAtomicBinary. In summary, we need a scratch register which is going to // be undef, that is unique among the register chosen for the instruction. BuildMI(BB, DL, TII->get(AtomicOp)) .addReg(Dest, RegState::Define | RegState::EarlyClobber) .addReg(AlignedAddr) .addReg(Mask) .addReg(ShiftedCmpVal) .addReg(Mask2) .addReg(ShiftedNewVal) .addReg(ShiftAmt) .addReg(Scratch, RegState::EarlyClobber | RegState::Define | RegState::Dead | RegState::Implicit) .addReg(Scratch2, RegState::EarlyClobber | RegState::Define | RegState::Dead | RegState::Implicit); MI.eraseFromParent(); // The instruction is gone now. return exitMBB; } SDValue MipsTargetLowering::lowerBRCOND(SDValue Op, SelectionDAG &DAG) const { // The first operand is the chain, the second is the condition, the third is // the block to branch to if the condition is true. SDValue Chain = Op.getOperand(0); SDValue Dest = Op.getOperand(2); SDLoc DL(Op); assert(!Subtarget.hasMips32r6() && !Subtarget.hasMips64r6()); SDValue CondRes = createFPCmp(DAG, Op.getOperand(1)); // Return if flag is not set by a floating point comparison. if (CondRes.getOpcode() != MipsISD::FPCmp) return Op; SDValue CCNode = CondRes.getOperand(2); Mips::CondCode CC = (Mips::CondCode)cast(CCNode)->getZExtValue(); unsigned Opc = invertFPCondCodeUser(CC) ? Mips::BRANCH_F : Mips::BRANCH_T; SDValue BrCode = DAG.getConstant(Opc, DL, MVT::i32); SDValue FCC0 = DAG.getRegister(Mips::FCC0, MVT::i32); return DAG.getNode(MipsISD::FPBrcond, DL, Op.getValueType(), Chain, BrCode, FCC0, Dest, CondRes); } SDValue MipsTargetLowering:: lowerSELECT(SDValue Op, SelectionDAG &DAG) const { assert(!Subtarget.hasMips32r6() && !Subtarget.hasMips64r6()); SDValue Cond = createFPCmp(DAG, Op.getOperand(0)); // Return if flag is not set by a floating point comparison. if (Cond.getOpcode() != MipsISD::FPCmp) return Op; return createCMovFP(DAG, Cond, Op.getOperand(1), Op.getOperand(2), SDLoc(Op)); } SDValue MipsTargetLowering::lowerSETCC(SDValue Op, SelectionDAG &DAG) const { assert(!Subtarget.hasMips32r6() && !Subtarget.hasMips64r6()); SDValue Cond = createFPCmp(DAG, Op); assert(Cond.getOpcode() == MipsISD::FPCmp && "Floating point operand expected."); SDLoc DL(Op); SDValue True = DAG.getConstant(1, DL, MVT::i32); SDValue False = DAG.getConstant(0, DL, MVT::i32); return createCMovFP(DAG, Cond, True, False, DL); } SDValue MipsTargetLowering::lowerGlobalAddress(SDValue Op, SelectionDAG &DAG) const { EVT Ty = Op.getValueType(); GlobalAddressSDNode *N = cast(Op); const GlobalValue *GV = N->getGlobal(); if (!isPositionIndependent()) { const MipsTargetObjectFile *TLOF = static_cast( getTargetMachine().getObjFileLowering()); const GlobalObject *GO = GV->getBaseObject(); if (GO && TLOF->IsGlobalInSmallSection(GO, getTargetMachine())) // %gp_rel relocation return getAddrGPRel(N, SDLoc(N), Ty, DAG, ABI.IsN64()); // %hi/%lo relocation return Subtarget.hasSym32() ? getAddrNonPIC(N, SDLoc(N), Ty, DAG) // %highest/%higher/%hi/%lo relocation : getAddrNonPICSym64(N, SDLoc(N), Ty, DAG); } // Every other architecture would use shouldAssumeDSOLocal in here, but // mips is special. // * In PIC code mips requires got loads even for local statics! // * To save on got entries, for local statics the got entry contains the // page and an additional add instruction takes care of the low bits. // * It is legal to access a hidden symbol with a non hidden undefined, // so one cannot guarantee that all access to a hidden symbol will know // it is hidden. // * Mips linkers don't support creating a page and a full got entry for // the same symbol. // * Given all that, we have to use a full got entry for hidden symbols :-( if (GV->hasLocalLinkage()) return getAddrLocal(N, SDLoc(N), Ty, DAG, ABI.IsN32() || ABI.IsN64()); if (LargeGOT) return getAddrGlobalLargeGOT( N, SDLoc(N), Ty, DAG, MipsII::MO_GOT_HI16, MipsII::MO_GOT_LO16, DAG.getEntryNode(), MachinePointerInfo::getGOT(DAG.getMachineFunction())); return getAddrGlobal( N, SDLoc(N), Ty, DAG, (ABI.IsN32() || ABI.IsN64()) ? MipsII::MO_GOT_DISP : MipsII::MO_GOT, DAG.getEntryNode(), MachinePointerInfo::getGOT(DAG.getMachineFunction())); } SDValue MipsTargetLowering::lowerBlockAddress(SDValue Op, SelectionDAG &DAG) const { BlockAddressSDNode *N = cast(Op); EVT Ty = Op.getValueType(); if (!isPositionIndependent()) return Subtarget.hasSym32() ? getAddrNonPIC(N, SDLoc(N), Ty, DAG) : getAddrNonPICSym64(N, SDLoc(N), Ty, DAG); return getAddrLocal(N, SDLoc(N), Ty, DAG, ABI.IsN32() || ABI.IsN64()); } SDValue MipsTargetLowering:: lowerGlobalTLSAddress(SDValue Op, SelectionDAG &DAG) const { // If the relocation model is PIC, use the General Dynamic TLS Model or // Local Dynamic TLS model, otherwise use the Initial Exec or // Local Exec TLS Model. GlobalAddressSDNode *GA = cast(Op); if (DAG.getTarget().useEmulatedTLS()) return LowerToTLSEmulatedModel(GA, DAG); SDLoc DL(GA); const GlobalValue *GV = GA->getGlobal(); EVT PtrVT = getPointerTy(DAG.getDataLayout()); TLSModel::Model model = getTargetMachine().getTLSModel(GV); if (model == TLSModel::GeneralDynamic || model == TLSModel::LocalDynamic) { // General Dynamic and Local Dynamic TLS Model. unsigned Flag = (model == TLSModel::LocalDynamic) ? MipsII::MO_TLSLDM : MipsII::MO_TLSGD; SDValue TGA = DAG.getTargetGlobalAddress(GV, DL, PtrVT, 0, Flag); SDValue Argument = DAG.getNode(MipsISD::Wrapper, DL, PtrVT, getGlobalReg(DAG, PtrVT), TGA); unsigned PtrSize = PtrVT.getSizeInBits(); IntegerType *PtrTy = Type::getIntNTy(*DAG.getContext(), PtrSize); SDValue TlsGetAddr = DAG.getExternalSymbol("__tls_get_addr", PtrVT); ArgListTy Args; ArgListEntry Entry; Entry.Node = Argument; Entry.Ty = PtrTy; Args.push_back(Entry); TargetLowering::CallLoweringInfo CLI(DAG); CLI.setDebugLoc(DL) .setChain(DAG.getEntryNode()) .setLibCallee(CallingConv::C, PtrTy, TlsGetAddr, std::move(Args)); std::pair CallResult = LowerCallTo(CLI); SDValue Ret = CallResult.first; if (model != TLSModel::LocalDynamic) return Ret; SDValue TGAHi = DAG.getTargetGlobalAddress(GV, DL, PtrVT, 0, MipsII::MO_DTPREL_HI); SDValue Hi = DAG.getNode(MipsISD::TlsHi, DL, PtrVT, TGAHi); SDValue TGALo = DAG.getTargetGlobalAddress(GV, DL, PtrVT, 0, MipsII::MO_DTPREL_LO); SDValue Lo = DAG.getNode(MipsISD::Lo, DL, PtrVT, TGALo); SDValue Add = DAG.getNode(ISD::ADD, DL, PtrVT, Hi, Ret); return DAG.getNode(ISD::ADD, DL, PtrVT, Add, Lo); } SDValue Offset; if (model == TLSModel::InitialExec) { // Initial Exec TLS Model SDValue TGA = DAG.getTargetGlobalAddress(GV, DL, PtrVT, 0, MipsII::MO_GOTTPREL); TGA = DAG.getNode(MipsISD::Wrapper, DL, PtrVT, getGlobalReg(DAG, PtrVT), TGA); Offset = DAG.getLoad(PtrVT, DL, DAG.getEntryNode(), TGA, MachinePointerInfo()); } else { // Local Exec TLS Model assert(model == TLSModel::LocalExec); SDValue TGAHi = DAG.getTargetGlobalAddress(GV, DL, PtrVT, 0, MipsII::MO_TPREL_HI); SDValue TGALo = DAG.getTargetGlobalAddress(GV, DL, PtrVT, 0, MipsII::MO_TPREL_LO); SDValue Hi = DAG.getNode(MipsISD::TlsHi, DL, PtrVT, TGAHi); SDValue Lo = DAG.getNode(MipsISD::Lo, DL, PtrVT, TGALo); Offset = DAG.getNode(ISD::ADD, DL, PtrVT, Hi, Lo); } SDValue ThreadPointer = DAG.getNode(MipsISD::ThreadPointer, DL, PtrVT); return DAG.getNode(ISD::ADD, DL, PtrVT, ThreadPointer, Offset); } SDValue MipsTargetLowering:: lowerJumpTable(SDValue Op, SelectionDAG &DAG) const { JumpTableSDNode *N = cast(Op); EVT Ty = Op.getValueType(); if (!isPositionIndependent()) return Subtarget.hasSym32() ? getAddrNonPIC(N, SDLoc(N), Ty, DAG) : getAddrNonPICSym64(N, SDLoc(N), Ty, DAG); return getAddrLocal(N, SDLoc(N), Ty, DAG, ABI.IsN32() || ABI.IsN64()); } SDValue MipsTargetLowering:: lowerConstantPool(SDValue Op, SelectionDAG &DAG) const { ConstantPoolSDNode *N = cast(Op); EVT Ty = Op.getValueType(); if (!isPositionIndependent()) { const MipsTargetObjectFile *TLOF = static_cast( getTargetMachine().getObjFileLowering()); if (TLOF->IsConstantInSmallSection(DAG.getDataLayout(), N->getConstVal(), getTargetMachine())) // %gp_rel relocation return getAddrGPRel(N, SDLoc(N), Ty, DAG, ABI.IsN64()); return Subtarget.hasSym32() ? getAddrNonPIC(N, SDLoc(N), Ty, DAG) : getAddrNonPICSym64(N, SDLoc(N), Ty, DAG); } return getAddrLocal(N, SDLoc(N), Ty, DAG, ABI.IsN32() || ABI.IsN64()); } SDValue MipsTargetLowering::lowerVASTART(SDValue Op, SelectionDAG &DAG) const { MachineFunction &MF = DAG.getMachineFunction(); MipsFunctionInfo *FuncInfo = MF.getInfo(); SDLoc DL(Op); SDValue FI = DAG.getFrameIndex(FuncInfo->getVarArgsFrameIndex(), getPointerTy(MF.getDataLayout())); // vastart just stores the address of the VarArgsFrameIndex slot into the // memory location argument. const Value *SV = cast(Op.getOperand(2))->getValue(); return DAG.getStore(Op.getOperand(0), DL, FI, Op.getOperand(1), MachinePointerInfo(SV)); } SDValue MipsTargetLowering::lowerVAARG(SDValue Op, SelectionDAG &DAG) const { SDNode *Node = Op.getNode(); EVT VT = Node->getValueType(0); SDValue Chain = Node->getOperand(0); SDValue VAListPtr = Node->getOperand(1); unsigned Align = Node->getConstantOperandVal(3); const Value *SV = cast(Node->getOperand(2))->getValue(); SDLoc DL(Node); unsigned ArgSlotSizeInBytes = (ABI.IsN32() || ABI.IsN64()) ? 8 : 4; SDValue VAListLoad = DAG.getLoad(getPointerTy(DAG.getDataLayout()), DL, Chain, VAListPtr, MachinePointerInfo(SV)); SDValue VAList = VAListLoad; // Re-align the pointer if necessary. // It should only ever be necessary for 64-bit types on O32 since the minimum // argument alignment is the same as the maximum type alignment for N32/N64. // // FIXME: We currently align too often. The code generator doesn't notice // when the pointer is still aligned from the last va_arg (or pair of // va_args for the i64 on O32 case). if (Align > getMinStackArgumentAlignment()) { assert(((Align & (Align-1)) == 0) && "Expected Align to be a power of 2"); VAList = DAG.getNode(ISD::ADD, DL, VAList.getValueType(), VAList, DAG.getConstant(Align - 1, DL, VAList.getValueType())); VAList = DAG.getNode(ISD::AND, DL, VAList.getValueType(), VAList, DAG.getConstant(-(int64_t)Align, DL, VAList.getValueType())); } // Increment the pointer, VAList, to the next vaarg. auto &TD = DAG.getDataLayout(); unsigned ArgSizeInBytes = TD.getTypeAllocSize(VT.getTypeForEVT(*DAG.getContext())); SDValue Tmp3 = DAG.getNode(ISD::ADD, DL, VAList.getValueType(), VAList, DAG.getConstant(alignTo(ArgSizeInBytes, ArgSlotSizeInBytes), DL, VAList.getValueType())); // Store the incremented VAList to the legalized pointer Chain = DAG.getStore(VAListLoad.getValue(1), DL, Tmp3, VAListPtr, MachinePointerInfo(SV)); // In big-endian mode we must adjust the pointer when the load size is smaller // than the argument slot size. We must also reduce the known alignment to // match. For example in the N64 ABI, we must add 4 bytes to the offset to get // the correct half of the slot, and reduce the alignment from 8 (slot // alignment) down to 4 (type alignment). if (!Subtarget.isLittle() && ArgSizeInBytes < ArgSlotSizeInBytes) { unsigned Adjustment = ArgSlotSizeInBytes - ArgSizeInBytes; VAList = DAG.getNode(ISD::ADD, DL, VAListPtr.getValueType(), VAList, DAG.getIntPtrConstant(Adjustment, DL)); } // Load the actual argument out of the pointer VAList return DAG.getLoad(VT, DL, Chain, VAList, MachinePointerInfo()); } static SDValue lowerFCOPYSIGN32(SDValue Op, SelectionDAG &DAG, bool HasExtractInsert) { EVT TyX = Op.getOperand(0).getValueType(); EVT TyY = Op.getOperand(1).getValueType(); SDLoc DL(Op); SDValue Const1 = DAG.getConstant(1, DL, MVT::i32); SDValue Const31 = DAG.getConstant(31, DL, MVT::i32); SDValue Res; // If operand is of type f64, extract the upper 32-bit. Otherwise, bitcast it // to i32. SDValue X = (TyX == MVT::f32) ? DAG.getNode(ISD::BITCAST, DL, MVT::i32, Op.getOperand(0)) : DAG.getNode(MipsISD::ExtractElementF64, DL, MVT::i32, Op.getOperand(0), Const1); SDValue Y = (TyY == MVT::f32) ? DAG.getNode(ISD::BITCAST, DL, MVT::i32, Op.getOperand(1)) : DAG.getNode(MipsISD::ExtractElementF64, DL, MVT::i32, Op.getOperand(1), Const1); if (HasExtractInsert) { // ext E, Y, 31, 1 ; extract bit31 of Y // ins X, E, 31, 1 ; insert extracted bit at bit31 of X SDValue E = DAG.getNode(MipsISD::Ext, DL, MVT::i32, Y, Const31, Const1); Res = DAG.getNode(MipsISD::Ins, DL, MVT::i32, E, Const31, Const1, X); } else { // sll SllX, X, 1 // srl SrlX, SllX, 1 // srl SrlY, Y, 31 // sll SllY, SrlX, 31 // or Or, SrlX, SllY SDValue SllX = DAG.getNode(ISD::SHL, DL, MVT::i32, X, Const1); SDValue SrlX = DAG.getNode(ISD::SRL, DL, MVT::i32, SllX, Const1); SDValue SrlY = DAG.getNode(ISD::SRL, DL, MVT::i32, Y, Const31); SDValue SllY = DAG.getNode(ISD::SHL, DL, MVT::i32, SrlY, Const31); Res = DAG.getNode(ISD::OR, DL, MVT::i32, SrlX, SllY); } if (TyX == MVT::f32) return DAG.getNode(ISD::BITCAST, DL, Op.getOperand(0).getValueType(), Res); SDValue LowX = DAG.getNode(MipsISD::ExtractElementF64, DL, MVT::i32, Op.getOperand(0), DAG.getConstant(0, DL, MVT::i32)); return DAG.getNode(MipsISD::BuildPairF64, DL, MVT::f64, LowX, Res); } static SDValue lowerFCOPYSIGN64(SDValue Op, SelectionDAG &DAG, bool HasExtractInsert) { unsigned WidthX = Op.getOperand(0).getValueSizeInBits(); unsigned WidthY = Op.getOperand(1).getValueSizeInBits(); EVT TyX = MVT::getIntegerVT(WidthX), TyY = MVT::getIntegerVT(WidthY); SDLoc DL(Op); SDValue Const1 = DAG.getConstant(1, DL, MVT::i32); // Bitcast to integer nodes. SDValue X = DAG.getNode(ISD::BITCAST, DL, TyX, Op.getOperand(0)); SDValue Y = DAG.getNode(ISD::BITCAST, DL, TyY, Op.getOperand(1)); if (HasExtractInsert) { // ext E, Y, width(Y) - 1, 1 ; extract bit width(Y)-1 of Y // ins X, E, width(X) - 1, 1 ; insert extracted bit at bit width(X)-1 of X SDValue E = DAG.getNode(MipsISD::Ext, DL, TyY, Y, DAG.getConstant(WidthY - 1, DL, MVT::i32), Const1); if (WidthX > WidthY) E = DAG.getNode(ISD::ZERO_EXTEND, DL, TyX, E); else if (WidthY > WidthX) E = DAG.getNode(ISD::TRUNCATE, DL, TyX, E); SDValue I = DAG.getNode(MipsISD::Ins, DL, TyX, E, DAG.getConstant(WidthX - 1, DL, MVT::i32), Const1, X); return DAG.getNode(ISD::BITCAST, DL, Op.getOperand(0).getValueType(), I); } // (d)sll SllX, X, 1 // (d)srl SrlX, SllX, 1 // (d)srl SrlY, Y, width(Y)-1 // (d)sll SllY, SrlX, width(Y)-1 // or Or, SrlX, SllY SDValue SllX = DAG.getNode(ISD::SHL, DL, TyX, X, Const1); SDValue SrlX = DAG.getNode(ISD::SRL, DL, TyX, SllX, Const1); SDValue SrlY = DAG.getNode(ISD::SRL, DL, TyY, Y, DAG.getConstant(WidthY - 1, DL, MVT::i32)); if (WidthX > WidthY) SrlY = DAG.getNode(ISD::ZERO_EXTEND, DL, TyX, SrlY); else if (WidthY > WidthX) SrlY = DAG.getNode(ISD::TRUNCATE, DL, TyX, SrlY); SDValue SllY = DAG.getNode(ISD::SHL, DL, TyX, SrlY, DAG.getConstant(WidthX - 1, DL, MVT::i32)); SDValue Or = DAG.getNode(ISD::OR, DL, TyX, SrlX, SllY); return DAG.getNode(ISD::BITCAST, DL, Op.getOperand(0).getValueType(), Or); } SDValue MipsTargetLowering::lowerFCOPYSIGN(SDValue Op, SelectionDAG &DAG) const { if (Subtarget.isGP64bit()) return lowerFCOPYSIGN64(Op, DAG, Subtarget.hasExtractInsert()); return lowerFCOPYSIGN32(Op, DAG, Subtarget.hasExtractInsert()); } SDValue MipsTargetLowering:: lowerFRAMEADDR(SDValue Op, SelectionDAG &DAG) const { // check the depth assert((cast(Op.getOperand(0))->getZExtValue() == 0) && "Frame address can only be determined for current frame."); MachineFrameInfo &MFI = DAG.getMachineFunction().getFrameInfo(); MFI.setFrameAddressIsTaken(true); EVT VT = Op.getValueType(); SDLoc DL(Op); SDValue FrameAddr = DAG.getCopyFromReg( DAG.getEntryNode(), DL, ABI.IsN64() ? Mips::FP_64 : Mips::FP, VT); return FrameAddr; } SDValue MipsTargetLowering::lowerRETURNADDR(SDValue Op, SelectionDAG &DAG) const { if (verifyReturnAddressArgumentIsConstant(Op, DAG)) return SDValue(); // check the depth assert((cast(Op.getOperand(0))->getZExtValue() == 0) && "Return address can be determined only for current frame."); MachineFunction &MF = DAG.getMachineFunction(); MachineFrameInfo &MFI = MF.getFrameInfo(); MVT VT = Op.getSimpleValueType(); unsigned RA = ABI.IsN64() ? Mips::RA_64 : Mips::RA; MFI.setReturnAddressIsTaken(true); // Return RA, which contains the return address. Mark it an implicit live-in. unsigned Reg = MF.addLiveIn(RA, getRegClassFor(VT)); return DAG.getCopyFromReg(DAG.getEntryNode(), SDLoc(Op), Reg, VT); } // An EH_RETURN is the result of lowering llvm.eh.return which in turn is // generated from __builtin_eh_return (offset, handler) // The effect of this is to adjust the stack pointer by "offset" // and then branch to "handler". SDValue MipsTargetLowering::lowerEH_RETURN(SDValue Op, SelectionDAG &DAG) const { MachineFunction &MF = DAG.getMachineFunction(); MipsFunctionInfo *MipsFI = MF.getInfo(); MipsFI->setCallsEhReturn(); SDValue Chain = Op.getOperand(0); SDValue Offset = Op.getOperand(1); SDValue Handler = Op.getOperand(2); SDLoc DL(Op); EVT Ty = ABI.IsN64() ? MVT::i64 : MVT::i32; // Store stack offset in V1, store jump target in V0. Glue CopyToReg and // EH_RETURN nodes, so that instructions are emitted back-to-back. unsigned OffsetReg = ABI.IsN64() ? Mips::V1_64 : Mips::V1; unsigned AddrReg = ABI.IsN64() ? Mips::V0_64 : Mips::V0; Chain = DAG.getCopyToReg(Chain, DL, OffsetReg, Offset, SDValue()); Chain = DAG.getCopyToReg(Chain, DL, AddrReg, Handler, Chain.getValue(1)); return DAG.getNode(MipsISD::EH_RETURN, DL, MVT::Other, Chain, DAG.getRegister(OffsetReg, Ty), DAG.getRegister(AddrReg, getPointerTy(MF.getDataLayout())), Chain.getValue(1)); } SDValue MipsTargetLowering::lowerATOMIC_FENCE(SDValue Op, SelectionDAG &DAG) const { // FIXME: Need pseudo-fence for 'singlethread' fences // FIXME: Set SType for weaker fences where supported/appropriate. unsigned SType = 0; SDLoc DL(Op); return DAG.getNode(MipsISD::Sync, DL, MVT::Other, Op.getOperand(0), DAG.getConstant(SType, DL, MVT::i32)); } SDValue MipsTargetLowering::lowerShiftLeftParts(SDValue Op, SelectionDAG &DAG) const { SDLoc DL(Op); MVT VT = Subtarget.isGP64bit() ? MVT::i64 : MVT::i32; SDValue Lo = Op.getOperand(0), Hi = Op.getOperand(1); SDValue Shamt = Op.getOperand(2); // if shamt < (VT.bits): // lo = (shl lo, shamt) // hi = (or (shl hi, shamt) (srl (srl lo, 1), ~shamt)) // else: // lo = 0 // hi = (shl lo, shamt[4:0]) SDValue Not = DAG.getNode(ISD::XOR, DL, MVT::i32, Shamt, DAG.getConstant(-1, DL, MVT::i32)); SDValue ShiftRight1Lo = DAG.getNode(ISD::SRL, DL, VT, Lo, DAG.getConstant(1, DL, VT)); SDValue ShiftRightLo = DAG.getNode(ISD::SRL, DL, VT, ShiftRight1Lo, Not); SDValue ShiftLeftHi = DAG.getNode(ISD::SHL, DL, VT, Hi, Shamt); SDValue Or = DAG.getNode(ISD::OR, DL, VT, ShiftLeftHi, ShiftRightLo); SDValue ShiftLeftLo = DAG.getNode(ISD::SHL, DL, VT, Lo, Shamt); SDValue Cond = DAG.getNode(ISD::AND, DL, MVT::i32, Shamt, DAG.getConstant(VT.getSizeInBits(), DL, MVT::i32)); Lo = DAG.getNode(ISD::SELECT, DL, VT, Cond, DAG.getConstant(0, DL, VT), ShiftLeftLo); Hi = DAG.getNode(ISD::SELECT, DL, VT, Cond, ShiftLeftLo, Or); SDValue Ops[2] = {Lo, Hi}; return DAG.getMergeValues(Ops, DL); } SDValue MipsTargetLowering::lowerShiftRightParts(SDValue Op, SelectionDAG &DAG, bool IsSRA) const { SDLoc DL(Op); SDValue Lo = Op.getOperand(0), Hi = Op.getOperand(1); SDValue Shamt = Op.getOperand(2); MVT VT = Subtarget.isGP64bit() ? MVT::i64 : MVT::i32; // if shamt < (VT.bits): // lo = (or (shl (shl hi, 1), ~shamt) (srl lo, shamt)) // if isSRA: // hi = (sra hi, shamt) // else: // hi = (srl hi, shamt) // else: // if isSRA: // lo = (sra hi, shamt[4:0]) // hi = (sra hi, 31) // else: // lo = (srl hi, shamt[4:0]) // hi = 0 SDValue Not = DAG.getNode(ISD::XOR, DL, MVT::i32, Shamt, DAG.getConstant(-1, DL, MVT::i32)); SDValue ShiftLeft1Hi = DAG.getNode(ISD::SHL, DL, VT, Hi, DAG.getConstant(1, DL, VT)); SDValue ShiftLeftHi = DAG.getNode(ISD::SHL, DL, VT, ShiftLeft1Hi, Not); SDValue ShiftRightLo = DAG.getNode(ISD::SRL, DL, VT, Lo, Shamt); SDValue Or = DAG.getNode(ISD::OR, DL, VT, ShiftLeftHi, ShiftRightLo); SDValue ShiftRightHi = DAG.getNode(IsSRA ? ISD::SRA : ISD::SRL, DL, VT, Hi, Shamt); SDValue Cond = DAG.getNode(ISD::AND, DL, MVT::i32, Shamt, DAG.getConstant(VT.getSizeInBits(), DL, MVT::i32)); SDValue Ext = DAG.getNode(ISD::SRA, DL, VT, Hi, DAG.getConstant(VT.getSizeInBits() - 1, DL, VT)); if (!(Subtarget.hasMips4() || Subtarget.hasMips32())) { SDVTList VTList = DAG.getVTList(VT, VT); return DAG.getNode(Subtarget.isGP64bit() ? Mips::PseudoD_SELECT_I64 : Mips::PseudoD_SELECT_I, DL, VTList, Cond, ShiftRightHi, IsSRA ? Ext : DAG.getConstant(0, DL, VT), Or, ShiftRightHi); } Lo = DAG.getNode(ISD::SELECT, DL, VT, Cond, ShiftRightHi, Or); Hi = DAG.getNode(ISD::SELECT, DL, VT, Cond, IsSRA ? Ext : DAG.getConstant(0, DL, VT), ShiftRightHi); SDValue Ops[2] = {Lo, Hi}; return DAG.getMergeValues(Ops, DL); } static SDValue createLoadLR(unsigned Opc, SelectionDAG &DAG, LoadSDNode *LD, SDValue Chain, SDValue Src, unsigned Offset) { SDValue Ptr = LD->getBasePtr(); EVT VT = LD->getValueType(0), MemVT = LD->getMemoryVT(); EVT BasePtrVT = Ptr.getValueType(); SDLoc DL(LD); SDVTList VTList = DAG.getVTList(VT, MVT::Other); if (Offset) Ptr = DAG.getNode(ISD::ADD, DL, BasePtrVT, Ptr, DAG.getConstant(Offset, DL, BasePtrVT)); SDValue Ops[] = { Chain, Ptr, Src }; return DAG.getMemIntrinsicNode(Opc, DL, VTList, Ops, MemVT, LD->getMemOperand()); } // Expand an unaligned 32 or 64-bit integer load node. SDValue MipsTargetLowering::lowerLOAD(SDValue Op, SelectionDAG &DAG) const { LoadSDNode *LD = cast(Op); EVT MemVT = LD->getMemoryVT(); if (Subtarget.systemSupportsUnalignedAccess()) return Op; // Return if load is aligned or if MemVT is neither i32 nor i64. if ((LD->getAlignment() >= MemVT.getSizeInBits() / 8) || ((MemVT != MVT::i32) && (MemVT != MVT::i64))) return SDValue(); bool IsLittle = Subtarget.isLittle(); EVT VT = Op.getValueType(); ISD::LoadExtType ExtType = LD->getExtensionType(); SDValue Chain = LD->getChain(), Undef = DAG.getUNDEF(VT); assert((VT == MVT::i32) || (VT == MVT::i64)); // Expand // (set dst, (i64 (load baseptr))) // to // (set tmp, (ldl (add baseptr, 7), undef)) // (set dst, (ldr baseptr, tmp)) if ((VT == MVT::i64) && (ExtType == ISD::NON_EXTLOAD)) { SDValue LDL = createLoadLR(MipsISD::LDL, DAG, LD, Chain, Undef, IsLittle ? 7 : 0); return createLoadLR(MipsISD::LDR, DAG, LD, LDL.getValue(1), LDL, IsLittle ? 0 : 7); } SDValue LWL = createLoadLR(MipsISD::LWL, DAG, LD, Chain, Undef, IsLittle ? 3 : 0); SDValue LWR = createLoadLR(MipsISD::LWR, DAG, LD, LWL.getValue(1), LWL, IsLittle ? 0 : 3); // Expand // (set dst, (i32 (load baseptr))) or // (set dst, (i64 (sextload baseptr))) or // (set dst, (i64 (extload baseptr))) // to // (set tmp, (lwl (add baseptr, 3), undef)) // (set dst, (lwr baseptr, tmp)) if ((VT == MVT::i32) || (ExtType == ISD::SEXTLOAD) || (ExtType == ISD::EXTLOAD)) return LWR; assert((VT == MVT::i64) && (ExtType == ISD::ZEXTLOAD)); // Expand // (set dst, (i64 (zextload baseptr))) // to // (set tmp0, (lwl (add baseptr, 3), undef)) // (set tmp1, (lwr baseptr, tmp0)) // (set tmp2, (shl tmp1, 32)) // (set dst, (srl tmp2, 32)) SDLoc DL(LD); SDValue Const32 = DAG.getConstant(32, DL, MVT::i32); SDValue SLL = DAG.getNode(ISD::SHL, DL, MVT::i64, LWR, Const32); SDValue SRL = DAG.getNode(ISD::SRL, DL, MVT::i64, SLL, Const32); SDValue Ops[] = { SRL, LWR.getValue(1) }; return DAG.getMergeValues(Ops, DL); } static SDValue createStoreLR(unsigned Opc, SelectionDAG &DAG, StoreSDNode *SD, SDValue Chain, unsigned Offset) { SDValue Ptr = SD->getBasePtr(), Value = SD->getValue(); EVT MemVT = SD->getMemoryVT(), BasePtrVT = Ptr.getValueType(); SDLoc DL(SD); SDVTList VTList = DAG.getVTList(MVT::Other); if (Offset) Ptr = DAG.getNode(ISD::ADD, DL, BasePtrVT, Ptr, DAG.getConstant(Offset, DL, BasePtrVT)); SDValue Ops[] = { Chain, Value, Ptr }; return DAG.getMemIntrinsicNode(Opc, DL, VTList, Ops, MemVT, SD->getMemOperand()); } // Expand an unaligned 32 or 64-bit integer store node. static SDValue lowerUnalignedIntStore(StoreSDNode *SD, SelectionDAG &DAG, bool IsLittle) { SDValue Value = SD->getValue(), Chain = SD->getChain(); EVT VT = Value.getValueType(); // Expand // (store val, baseptr) or // (truncstore val, baseptr) // to // (swl val, (add baseptr, 3)) // (swr val, baseptr) if ((VT == MVT::i32) || SD->isTruncatingStore()) { SDValue SWL = createStoreLR(MipsISD::SWL, DAG, SD, Chain, IsLittle ? 3 : 0); return createStoreLR(MipsISD::SWR, DAG, SD, SWL, IsLittle ? 0 : 3); } assert(VT == MVT::i64); // Expand // (store val, baseptr) // to // (sdl val, (add baseptr, 7)) // (sdr val, baseptr) SDValue SDL = createStoreLR(MipsISD::SDL, DAG, SD, Chain, IsLittle ? 7 : 0); return createStoreLR(MipsISD::SDR, DAG, SD, SDL, IsLittle ? 0 : 7); } // Lower (store (fp_to_sint $fp) $ptr) to (store (TruncIntFP $fp), $ptr). static SDValue lowerFP_TO_SINT_STORE(StoreSDNode *SD, SelectionDAG &DAG, bool SingleFloat) { SDValue Val = SD->getValue(); if (Val.getOpcode() != ISD::FP_TO_SINT || (Val.getValueSizeInBits() > 32 && SingleFloat)) return SDValue(); EVT FPTy = EVT::getFloatingPointVT(Val.getValueSizeInBits()); SDValue Tr = DAG.getNode(MipsISD::TruncIntFP, SDLoc(Val), FPTy, Val.getOperand(0)); return DAG.getStore(SD->getChain(), SDLoc(SD), Tr, SD->getBasePtr(), SD->getPointerInfo(), SD->getAlignment(), SD->getMemOperand()->getFlags()); } SDValue MipsTargetLowering::lowerSTORE(SDValue Op, SelectionDAG &DAG) const { StoreSDNode *SD = cast(Op); EVT MemVT = SD->getMemoryVT(); // Lower unaligned integer stores. if (!Subtarget.systemSupportsUnalignedAccess() && (SD->getAlignment() < MemVT.getSizeInBits() / 8) && ((MemVT == MVT::i32) || (MemVT == MVT::i64))) return lowerUnalignedIntStore(SD, DAG, Subtarget.isLittle()); return lowerFP_TO_SINT_STORE(SD, DAG, Subtarget.isSingleFloat()); } SDValue MipsTargetLowering::lowerEH_DWARF_CFA(SDValue Op, SelectionDAG &DAG) const { // Return a fixed StackObject with offset 0 which points to the old stack // pointer. MachineFrameInfo &MFI = DAG.getMachineFunction().getFrameInfo(); EVT ValTy = Op->getValueType(0); int FI = MFI.CreateFixedObject(Op.getValueSizeInBits() / 8, 0, false); return DAG.getFrameIndex(FI, ValTy); } SDValue MipsTargetLowering::lowerFP_TO_SINT(SDValue Op, SelectionDAG &DAG) const { if (Op.getValueSizeInBits() > 32 && Subtarget.isSingleFloat()) return SDValue(); EVT FPTy = EVT::getFloatingPointVT(Op.getValueSizeInBits()); SDValue Trunc = DAG.getNode(MipsISD::TruncIntFP, SDLoc(Op), FPTy, Op.getOperand(0)); return DAG.getNode(ISD::BITCAST, SDLoc(Op), Op.getValueType(), Trunc); } //===----------------------------------------------------------------------===// // Calling Convention Implementation //===----------------------------------------------------------------------===// //===----------------------------------------------------------------------===// // TODO: Implement a generic logic using tblgen that can support this. // Mips O32 ABI rules: // --- // i32 - Passed in A0, A1, A2, A3 and stack // f32 - Only passed in f32 registers if no int reg has been used yet to hold // an argument. Otherwise, passed in A1, A2, A3 and stack. // f64 - Only passed in two aliased f32 registers if no int reg has been used // yet to hold an argument. Otherwise, use A2, A3 and stack. If A1 is // not used, it must be shadowed. If only A3 is available, shadow it and // go to stack. // vXiX - Received as scalarized i32s, passed in A0 - A3 and the stack. // vXf32 - Passed in either a pair of registers {A0, A1}, {A2, A3} or {A0 - A3} // with the remainder spilled to the stack. // vXf64 - Passed in either {A0, A1, A2, A3} or {A2, A3} and in both cases // spilling the remainder to the stack. // // For vararg functions, all arguments are passed in A0, A1, A2, A3 and stack. //===----------------------------------------------------------------------===// static bool CC_MipsO32(unsigned ValNo, MVT ValVT, MVT LocVT, CCValAssign::LocInfo LocInfo, ISD::ArgFlagsTy ArgFlags, CCState &State, ArrayRef F64Regs) { const MipsSubtarget &Subtarget = static_cast( State.getMachineFunction().getSubtarget()); static const MCPhysReg IntRegs[] = { Mips::A0, Mips::A1, Mips::A2, Mips::A3 }; const MipsCCState * MipsState = static_cast(&State); static const MCPhysReg F32Regs[] = { Mips::F12, Mips::F14 }; static const MCPhysReg FloatVectorIntRegs[] = { Mips::A0, Mips::A2 }; // Do not process byval args here. if (ArgFlags.isByVal()) return true; // Promote i8 and i16 if (ArgFlags.isInReg() && !Subtarget.isLittle()) { if (LocVT == MVT::i8 || LocVT == MVT::i16 || LocVT == MVT::i32) { LocVT = MVT::i32; if (ArgFlags.isSExt()) LocInfo = CCValAssign::SExtUpper; else if (ArgFlags.isZExt()) LocInfo = CCValAssign::ZExtUpper; else LocInfo = CCValAssign::AExtUpper; } } // Promote i8 and i16 if (LocVT == MVT::i8 || LocVT == MVT::i16) { LocVT = MVT::i32; if (ArgFlags.isSExt()) LocInfo = CCValAssign::SExt; else if (ArgFlags.isZExt()) LocInfo = CCValAssign::ZExt; else LocInfo = CCValAssign::AExt; } unsigned Reg; // f32 and f64 are allocated in A0, A1, A2, A3 when either of the following // is true: function is vararg, argument is 3rd or higher, there is previous // argument which is not f32 or f64. bool AllocateFloatsInIntReg = State.isVarArg() || ValNo > 1 || State.getFirstUnallocated(F32Regs) != ValNo; unsigned OrigAlign = ArgFlags.getOrigAlign(); bool isI64 = (ValVT == MVT::i32 && OrigAlign == 8); bool isVectorFloat = MipsState->WasOriginalArgVectorFloat(ValNo); // The MIPS vector ABI for floats passes them in a pair of registers if (ValVT == MVT::i32 && isVectorFloat) { // This is the start of an vector that was scalarized into an unknown number // of components. It doesn't matter how many there are. Allocate one of the // notional 8 byte aligned registers which map onto the argument stack, and // shadow the register lost to alignment requirements. if (ArgFlags.isSplit()) { Reg = State.AllocateReg(FloatVectorIntRegs); if (Reg == Mips::A2) State.AllocateReg(Mips::A1); else if (Reg == 0) State.AllocateReg(Mips::A3); } else { // If we're an intermediate component of the split, we can just attempt to // allocate a register directly. Reg = State.AllocateReg(IntRegs); } } else if (ValVT == MVT::i32 || (ValVT == MVT::f32 && AllocateFloatsInIntReg)) { Reg = State.AllocateReg(IntRegs); // If this is the first part of an i64 arg, // the allocated register must be either A0 or A2. if (isI64 && (Reg == Mips::A1 || Reg == Mips::A3)) Reg = State.AllocateReg(IntRegs); LocVT = MVT::i32; } else if (ValVT == MVT::f64 && AllocateFloatsInIntReg) { // Allocate int register and shadow next int register. If first // available register is Mips::A1 or Mips::A3, shadow it too. Reg = State.AllocateReg(IntRegs); if (Reg == Mips::A1 || Reg == Mips::A3) Reg = State.AllocateReg(IntRegs); State.AllocateReg(IntRegs); LocVT = MVT::i32; } else if (ValVT.isFloatingPoint() && !AllocateFloatsInIntReg) { // we are guaranteed to find an available float register if (ValVT == MVT::f32) { Reg = State.AllocateReg(F32Regs); // Shadow int register State.AllocateReg(IntRegs); } else { Reg = State.AllocateReg(F64Regs); // Shadow int registers unsigned Reg2 = State.AllocateReg(IntRegs); if (Reg2 == Mips::A1 || Reg2 == Mips::A3) State.AllocateReg(IntRegs); State.AllocateReg(IntRegs); } } else llvm_unreachable("Cannot handle this ValVT."); if (!Reg) { unsigned Offset = State.AllocateStack(ValVT.getStoreSize(), OrigAlign); State.addLoc(CCValAssign::getMem(ValNo, ValVT, Offset, LocVT, LocInfo)); } else State.addLoc(CCValAssign::getReg(ValNo, ValVT, Reg, LocVT, LocInfo)); return false; } static bool CC_MipsO32_FP32(unsigned ValNo, MVT ValVT, MVT LocVT, CCValAssign::LocInfo LocInfo, ISD::ArgFlagsTy ArgFlags, CCState &State) { static const MCPhysReg F64Regs[] = { Mips::D6, Mips::D7 }; return CC_MipsO32(ValNo, ValVT, LocVT, LocInfo, ArgFlags, State, F64Regs); } static bool CC_MipsO32_FP64(unsigned ValNo, MVT ValVT, MVT LocVT, CCValAssign::LocInfo LocInfo, ISD::ArgFlagsTy ArgFlags, CCState &State) { static const MCPhysReg F64Regs[] = { Mips::D12_64, Mips::D14_64 }; return CC_MipsO32(ValNo, ValVT, LocVT, LocInfo, ArgFlags, State, F64Regs); } static bool CC_MipsO32(unsigned ValNo, MVT ValVT, MVT LocVT, CCValAssign::LocInfo LocInfo, ISD::ArgFlagsTy ArgFlags, CCState &State) LLVM_ATTRIBUTE_UNUSED; #include "MipsGenCallingConv.inc" CCAssignFn *MipsTargetLowering::CCAssignFnForCall() const{ return CC_Mips; } CCAssignFn *MipsTargetLowering::CCAssignFnForReturn() const{ return RetCC_Mips; } //===----------------------------------------------------------------------===// // Call Calling Convention Implementation //===----------------------------------------------------------------------===// // Return next O32 integer argument register. static unsigned getNextIntArgReg(unsigned Reg) { assert((Reg == Mips::A0) || (Reg == Mips::A2)); return (Reg == Mips::A0) ? Mips::A1 : Mips::A3; } SDValue MipsTargetLowering::passArgOnStack(SDValue StackPtr, unsigned Offset, SDValue Chain, SDValue Arg, const SDLoc &DL, bool IsTailCall, SelectionDAG &DAG) const { if (!IsTailCall) { SDValue PtrOff = DAG.getNode(ISD::ADD, DL, getPointerTy(DAG.getDataLayout()), StackPtr, DAG.getIntPtrConstant(Offset, DL)); return DAG.getStore(Chain, DL, Arg, PtrOff, MachinePointerInfo()); } MachineFrameInfo &MFI = DAG.getMachineFunction().getFrameInfo(); int FI = MFI.CreateFixedObject(Arg.getValueSizeInBits() / 8, Offset, false); SDValue FIN = DAG.getFrameIndex(FI, getPointerTy(DAG.getDataLayout())); return DAG.getStore(Chain, DL, Arg, FIN, MachinePointerInfo(), /* Alignment = */ 0, MachineMemOperand::MOVolatile); } void MipsTargetLowering:: getOpndList(SmallVectorImpl &Ops, std::deque> &RegsToPass, bool IsPICCall, bool GlobalOrExternal, bool InternalLinkage, bool IsCallReloc, CallLoweringInfo &CLI, SDValue Callee, SDValue Chain) const { // Insert node "GP copy globalreg" before call to function. // // R_MIPS_CALL* operators (emitted when non-internal functions are called // in PIC mode) allow symbols to be resolved via lazy binding. // The lazy binding stub requires GP to point to the GOT. // Note that we don't need GP to point to the GOT for indirect calls // (when R_MIPS_CALL* is not used for the call) because Mips linker generates // lazy binding stub for a function only when R_MIPS_CALL* are the only relocs // used for the function (that is, Mips linker doesn't generate lazy binding // stub for a function whose address is taken in the program). if (IsPICCall && !InternalLinkage && IsCallReloc) { unsigned GPReg = ABI.IsN64() ? Mips::GP_64 : Mips::GP; EVT Ty = ABI.IsN64() ? MVT::i64 : MVT::i32; RegsToPass.push_back(std::make_pair(GPReg, getGlobalReg(CLI.DAG, Ty))); } // Build a sequence of copy-to-reg nodes chained together with token // chain and flag operands which copy the outgoing args into registers. // The InFlag in necessary since all emitted instructions must be // stuck together. SDValue InFlag; for (unsigned i = 0, e = RegsToPass.size(); i != e; ++i) { Chain = CLI.DAG.getCopyToReg(Chain, CLI.DL, RegsToPass[i].first, RegsToPass[i].second, InFlag); InFlag = Chain.getValue(1); } // Add argument registers to the end of the list so that they are // known live into the call. for (unsigned i = 0, e = RegsToPass.size(); i != e; ++i) Ops.push_back(CLI.DAG.getRegister(RegsToPass[i].first, RegsToPass[i].second.getValueType())); // Add a register mask operand representing the call-preserved registers. const TargetRegisterInfo *TRI = Subtarget.getRegisterInfo(); const uint32_t *Mask = TRI->getCallPreservedMask(CLI.DAG.getMachineFunction(), CLI.CallConv); assert(Mask && "Missing call preserved mask for calling convention"); if (Subtarget.inMips16HardFloat()) { if (GlobalAddressSDNode *G = dyn_cast(CLI.Callee)) { StringRef Sym = G->getGlobal()->getName(); Function *F = G->getGlobal()->getParent()->getFunction(Sym); if (F && F->hasFnAttribute("__Mips16RetHelper")) { Mask = MipsRegisterInfo::getMips16RetHelperMask(); } } } Ops.push_back(CLI.DAG.getRegisterMask(Mask)); if (InFlag.getNode()) Ops.push_back(InFlag); } +void MipsTargetLowering::AdjustInstrPostInstrSelection(MachineInstr &MI, + SDNode *Node) const { + switch (MI.getOpcode()) { + default: + return; + case Mips::JALR: + case Mips::JALRPseudo: + case Mips::JALR64: + case Mips::JALR64Pseudo: + case Mips::JALR16_MM: + case Mips::JALRC16_MMR6: + case Mips::TAILCALLREG: + case Mips::TAILCALLREG64: + case Mips::TAILCALLR6REG: + case Mips::TAILCALL64R6REG: + case Mips::TAILCALLREG_MM: + case Mips::TAILCALLREG_MMR6: { + if (!EmitJalrReloc || + Subtarget.inMips16Mode() || + !isPositionIndependent() || + Node->getNumOperands() < 1 || + Node->getOperand(0).getNumOperands() < 2) { + return; + } + // We are after the callee address, set by LowerCall(). + // If added to MI, asm printer will emit .reloc R_MIPS_JALR for the + // symbol. + const SDValue TargetAddr = Node->getOperand(0).getOperand(1); + StringRef Sym; + if (const GlobalAddressSDNode *G = + dyn_cast_or_null(TargetAddr)) { + Sym = G->getGlobal()->getName(); + } + else if (const ExternalSymbolSDNode *ES = + dyn_cast_or_null(TargetAddr)) { + Sym = ES->getSymbol(); + } + + if (Sym.empty()) + return; + + MachineFunction *MF = MI.getParent()->getParent(); + MCSymbol *S = MF->getContext().getOrCreateSymbol(Sym); + MI.addOperand(MachineOperand::CreateMCSymbol(S, MipsII::MO_JALR)); + } + } +} + /// LowerCall - functions arguments are copied from virtual regs to /// (physical regs)/(stack frame), CALLSEQ_START and CALLSEQ_END are emitted. SDValue MipsTargetLowering::LowerCall(TargetLowering::CallLoweringInfo &CLI, SmallVectorImpl &InVals) const { SelectionDAG &DAG = CLI.DAG; SDLoc DL = CLI.DL; SmallVectorImpl &Outs = CLI.Outs; SmallVectorImpl &OutVals = CLI.OutVals; SmallVectorImpl &Ins = CLI.Ins; SDValue Chain = CLI.Chain; SDValue Callee = CLI.Callee; bool &IsTailCall = CLI.IsTailCall; CallingConv::ID CallConv = CLI.CallConv; bool IsVarArg = CLI.IsVarArg; MachineFunction &MF = DAG.getMachineFunction(); MachineFrameInfo &MFI = MF.getFrameInfo(); const TargetFrameLowering *TFL = Subtarget.getFrameLowering(); MipsFunctionInfo *FuncInfo = MF.getInfo(); bool IsPIC = isPositionIndependent(); // Analyze operands of the call, assigning locations to each operand. SmallVector ArgLocs; MipsCCState CCInfo( CallConv, IsVarArg, DAG.getMachineFunction(), ArgLocs, *DAG.getContext(), MipsCCState::getSpecialCallingConvForCallee(Callee.getNode(), Subtarget)); const ExternalSymbolSDNode *ES = dyn_cast_or_null(Callee.getNode()); // There is one case where CALLSEQ_START..CALLSEQ_END can be nested, which // is during the lowering of a call with a byval argument which produces // a call to memcpy. For the O32 case, this causes the caller to allocate // stack space for the reserved argument area for the callee, then recursively // again for the memcpy call. In the NEWABI case, this doesn't occur as those // ABIs mandate that the callee allocates the reserved argument area. We do // still produce nested CALLSEQ_START..CALLSEQ_END with zero space though. // // If the callee has a byval argument and memcpy is used, we are mandated // to already have produced a reserved argument area for the callee for O32. // Therefore, the reserved argument area can be reused for both calls. // // Other cases of calling memcpy cannot have a chain with a CALLSEQ_START // present, as we have yet to hook that node onto the chain. // // Hence, the CALLSEQ_START and CALLSEQ_END nodes can be eliminated in this // case. GCC does a similar trick, in that wherever possible, it calculates // the maximum out going argument area (including the reserved area), and // preallocates the stack space on entrance to the caller. // - // FIXME: We should do the same for efficency and space. + // FIXME: We should do the same for efficiency and space. // Note: The check on the calling convention below must match // MipsABIInfo::GetCalleeAllocdArgSizeInBytes(). bool MemcpyInByVal = ES && StringRef(ES->getSymbol()) == StringRef("memcpy") && CallConv != CallingConv::Fast && Chain.getOpcode() == ISD::CALLSEQ_START; // Allocate the reserved argument area. It seems strange to do this from the // caller side but removing it breaks the frame size calculation. unsigned ReservedArgArea = MemcpyInByVal ? 0 : ABI.GetCalleeAllocdArgSizeInBytes(CallConv); CCInfo.AllocateStack(ReservedArgArea, 1); CCInfo.AnalyzeCallOperands(Outs, CC_Mips, CLI.getArgs(), ES ? ES->getSymbol() : nullptr); // Get a count of how many bytes are to be pushed on the stack. unsigned NextStackOffset = CCInfo.getNextStackOffset(); // Check if it's really possible to do a tail call. Restrict it to functions // that are part of this compilation unit. bool InternalLinkage = false; if (IsTailCall) { IsTailCall = isEligibleForTailCallOptimization( CCInfo, NextStackOffset, *MF.getInfo()); if (GlobalAddressSDNode *G = dyn_cast(Callee)) { InternalLinkage = G->getGlobal()->hasInternalLinkage(); IsTailCall &= (InternalLinkage || G->getGlobal()->hasLocalLinkage() || G->getGlobal()->hasPrivateLinkage() || G->getGlobal()->hasHiddenVisibility() || G->getGlobal()->hasProtectedVisibility()); } } if (!IsTailCall && CLI.CS && CLI.CS.isMustTailCall()) report_fatal_error("failed to perform tail call elimination on a call " "site marked musttail"); if (IsTailCall) ++NumTailCalls; // Chain is the output chain of the last Load/Store or CopyToReg node. // ByValChain is the output chain of the last Memcpy node created for copying // byval arguments to the stack. unsigned StackAlignment = TFL->getStackAlignment(); NextStackOffset = alignTo(NextStackOffset, StackAlignment); SDValue NextStackOffsetVal = DAG.getIntPtrConstant(NextStackOffset, DL, true); if (!(IsTailCall || MemcpyInByVal)) Chain = DAG.getCALLSEQ_START(Chain, NextStackOffset, 0, DL); SDValue StackPtr = DAG.getCopyFromReg(Chain, DL, ABI.IsN64() ? Mips::SP_64 : Mips::SP, getPointerTy(DAG.getDataLayout())); std::deque> RegsToPass; SmallVector MemOpChains; CCInfo.rewindByValRegsInfo(); // Walk the register/memloc assignments, inserting copies/loads. for (unsigned i = 0, e = ArgLocs.size(); i != e; ++i) { SDValue Arg = OutVals[i]; CCValAssign &VA = ArgLocs[i]; MVT ValVT = VA.getValVT(), LocVT = VA.getLocVT(); ISD::ArgFlagsTy Flags = Outs[i].Flags; bool UseUpperBits = false; // ByVal Arg. if (Flags.isByVal()) { unsigned FirstByValReg, LastByValReg; unsigned ByValIdx = CCInfo.getInRegsParamsProcessed(); CCInfo.getInRegsParamInfo(ByValIdx, FirstByValReg, LastByValReg); assert(Flags.getByValSize() && "ByVal args of size 0 should have been ignored by front-end."); assert(ByValIdx < CCInfo.getInRegsParamsCount()); assert(!IsTailCall && "Do not tail-call optimize if there is a byval argument."); passByValArg(Chain, DL, RegsToPass, MemOpChains, StackPtr, MFI, DAG, Arg, FirstByValReg, LastByValReg, Flags, Subtarget.isLittle(), VA); CCInfo.nextInRegsParam(); continue; } // Promote the value if needed. switch (VA.getLocInfo()) { default: llvm_unreachable("Unknown loc info!"); case CCValAssign::Full: if (VA.isRegLoc()) { if ((ValVT == MVT::f32 && LocVT == MVT::i32) || (ValVT == MVT::f64 && LocVT == MVT::i64) || (ValVT == MVT::i64 && LocVT == MVT::f64)) Arg = DAG.getNode(ISD::BITCAST, DL, LocVT, Arg); else if (ValVT == MVT::f64 && LocVT == MVT::i32) { SDValue Lo = DAG.getNode(MipsISD::ExtractElementF64, DL, MVT::i32, Arg, DAG.getConstant(0, DL, MVT::i32)); SDValue Hi = DAG.getNode(MipsISD::ExtractElementF64, DL, MVT::i32, Arg, DAG.getConstant(1, DL, MVT::i32)); if (!Subtarget.isLittle()) std::swap(Lo, Hi); unsigned LocRegLo = VA.getLocReg(); unsigned LocRegHigh = getNextIntArgReg(LocRegLo); RegsToPass.push_back(std::make_pair(LocRegLo, Lo)); RegsToPass.push_back(std::make_pair(LocRegHigh, Hi)); continue; } } break; case CCValAssign::BCvt: Arg = DAG.getNode(ISD::BITCAST, DL, LocVT, Arg); break; case CCValAssign::SExtUpper: UseUpperBits = true; LLVM_FALLTHROUGH; case CCValAssign::SExt: Arg = DAG.getNode(ISD::SIGN_EXTEND, DL, LocVT, Arg); break; case CCValAssign::ZExtUpper: UseUpperBits = true; LLVM_FALLTHROUGH; case CCValAssign::ZExt: Arg = DAG.getNode(ISD::ZERO_EXTEND, DL, LocVT, Arg); break; case CCValAssign::AExtUpper: UseUpperBits = true; LLVM_FALLTHROUGH; case CCValAssign::AExt: Arg = DAG.getNode(ISD::ANY_EXTEND, DL, LocVT, Arg); break; } if (UseUpperBits) { unsigned ValSizeInBits = Outs[i].ArgVT.getSizeInBits(); unsigned LocSizeInBits = VA.getLocVT().getSizeInBits(); Arg = DAG.getNode( ISD::SHL, DL, VA.getLocVT(), Arg, DAG.getConstant(LocSizeInBits - ValSizeInBits, DL, VA.getLocVT())); } // Arguments that can be passed on register must be kept at // RegsToPass vector if (VA.isRegLoc()) { RegsToPass.push_back(std::make_pair(VA.getLocReg(), Arg)); continue; } // Register can't get to this point... assert(VA.isMemLoc()); // emit ISD::STORE whichs stores the // parameter value to a stack Location MemOpChains.push_back(passArgOnStack(StackPtr, VA.getLocMemOffset(), Chain, Arg, DL, IsTailCall, DAG)); } // Transform all store nodes into one single node because all store // nodes are independent of each other. if (!MemOpChains.empty()) Chain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, MemOpChains); // If the callee is a GlobalAddress/ExternalSymbol node (quite common, every // direct call is) turn it into a TargetGlobalAddress/TargetExternalSymbol // node so that legalize doesn't hack it. EVT Ty = Callee.getValueType(); bool GlobalOrExternal = false, IsCallReloc = false; // The long-calls feature is ignored in case of PIC. // While we do not support -mshared / -mno-shared properly, // ignore long-calls in case of -mabicalls too. if (!Subtarget.isABICalls() && !IsPIC) { // If the function should be called using "long call", // get its address into a register to prevent using // of the `jal` instruction for the direct call. if (auto *N = dyn_cast(Callee)) { if (Subtarget.useLongCalls()) Callee = Subtarget.hasSym32() ? getAddrNonPIC(N, SDLoc(N), Ty, DAG) : getAddrNonPICSym64(N, SDLoc(N), Ty, DAG); } else if (auto *N = dyn_cast(Callee)) { bool UseLongCalls = Subtarget.useLongCalls(); // If the function has long-call/far/near attribute // it overrides command line switch pased to the backend. if (auto *F = dyn_cast(N->getGlobal())) { if (F->hasFnAttribute("long-call")) UseLongCalls = true; else if (F->hasFnAttribute("short-call")) UseLongCalls = false; } if (UseLongCalls) Callee = Subtarget.hasSym32() ? getAddrNonPIC(N, SDLoc(N), Ty, DAG) : getAddrNonPICSym64(N, SDLoc(N), Ty, DAG); } } if (GlobalAddressSDNode *G = dyn_cast(Callee)) { if (IsPIC) { const GlobalValue *Val = G->getGlobal(); InternalLinkage = Val->hasInternalLinkage(); if (InternalLinkage) Callee = getAddrLocal(G, DL, Ty, DAG, ABI.IsN32() || ABI.IsN64()); else if (LargeGOT) { Callee = getAddrGlobalLargeGOT(G, DL, Ty, DAG, MipsII::MO_CALL_HI16, MipsII::MO_CALL_LO16, Chain, FuncInfo->callPtrInfo(Val)); IsCallReloc = true; } else { Callee = getAddrGlobal(G, DL, Ty, DAG, MipsII::MO_GOT_CALL, Chain, FuncInfo->callPtrInfo(Val)); IsCallReloc = true; } } else Callee = DAG.getTargetGlobalAddress(G->getGlobal(), DL, getPointerTy(DAG.getDataLayout()), 0, MipsII::MO_NO_FLAG); GlobalOrExternal = true; } else if (ExternalSymbolSDNode *S = dyn_cast(Callee)) { const char *Sym = S->getSymbol(); if (!IsPIC) // static Callee = DAG.getTargetExternalSymbol( Sym, getPointerTy(DAG.getDataLayout()), MipsII::MO_NO_FLAG); else if (LargeGOT) { Callee = getAddrGlobalLargeGOT(S, DL, Ty, DAG, MipsII::MO_CALL_HI16, MipsII::MO_CALL_LO16, Chain, FuncInfo->callPtrInfo(Sym)); IsCallReloc = true; } else { // PIC Callee = getAddrGlobal(S, DL, Ty, DAG, MipsII::MO_GOT_CALL, Chain, FuncInfo->callPtrInfo(Sym)); IsCallReloc = true; } GlobalOrExternal = true; } SmallVector Ops(1, Chain); SDVTList NodeTys = DAG.getVTList(MVT::Other, MVT::Glue); getOpndList(Ops, RegsToPass, IsPIC, GlobalOrExternal, InternalLinkage, IsCallReloc, CLI, Callee, Chain); if (IsTailCall) { MF.getFrameInfo().setHasTailCall(); return DAG.getNode(MipsISD::TailCall, DL, MVT::Other, Ops); } Chain = DAG.getNode(MipsISD::JmpLink, DL, NodeTys, Ops); SDValue InFlag = Chain.getValue(1); // Create the CALLSEQ_END node in the case of where it is not a call to // memcpy. if (!(MemcpyInByVal)) { Chain = DAG.getCALLSEQ_END(Chain, NextStackOffsetVal, DAG.getIntPtrConstant(0, DL, true), InFlag, DL); InFlag = Chain.getValue(1); } // Handle result values, copying them out of physregs into vregs that we // return. return LowerCallResult(Chain, InFlag, CallConv, IsVarArg, Ins, DL, DAG, InVals, CLI); } /// LowerCallResult - Lower the result values of a call into the /// appropriate copies out of appropriate physical registers. SDValue MipsTargetLowering::LowerCallResult( SDValue Chain, SDValue InFlag, CallingConv::ID CallConv, bool IsVarArg, const SmallVectorImpl &Ins, const SDLoc &DL, SelectionDAG &DAG, SmallVectorImpl &InVals, TargetLowering::CallLoweringInfo &CLI) const { // Assign locations to each value returned by this call. SmallVector RVLocs; MipsCCState CCInfo(CallConv, IsVarArg, DAG.getMachineFunction(), RVLocs, *DAG.getContext()); const ExternalSymbolSDNode *ES = dyn_cast_or_null(CLI.Callee.getNode()); CCInfo.AnalyzeCallResult(Ins, RetCC_Mips, CLI.RetTy, ES ? ES->getSymbol() : nullptr); // Copy all of the result registers out of their specified physreg. for (unsigned i = 0; i != RVLocs.size(); ++i) { CCValAssign &VA = RVLocs[i]; assert(VA.isRegLoc() && "Can only return in registers!"); SDValue Val = DAG.getCopyFromReg(Chain, DL, RVLocs[i].getLocReg(), RVLocs[i].getLocVT(), InFlag); Chain = Val.getValue(1); InFlag = Val.getValue(2); if (VA.isUpperBitsInLoc()) { unsigned ValSizeInBits = Ins[i].ArgVT.getSizeInBits(); unsigned LocSizeInBits = VA.getLocVT().getSizeInBits(); unsigned Shift = VA.getLocInfo() == CCValAssign::ZExtUpper ? ISD::SRL : ISD::SRA; Val = DAG.getNode( Shift, DL, VA.getLocVT(), Val, DAG.getConstant(LocSizeInBits - ValSizeInBits, DL, VA.getLocVT())); } switch (VA.getLocInfo()) { default: llvm_unreachable("Unknown loc info!"); case CCValAssign::Full: break; case CCValAssign::BCvt: Val = DAG.getNode(ISD::BITCAST, DL, VA.getValVT(), Val); break; case CCValAssign::AExt: case CCValAssign::AExtUpper: Val = DAG.getNode(ISD::TRUNCATE, DL, VA.getValVT(), Val); break; case CCValAssign::ZExt: case CCValAssign::ZExtUpper: Val = DAG.getNode(ISD::AssertZext, DL, VA.getLocVT(), Val, DAG.getValueType(VA.getValVT())); Val = DAG.getNode(ISD::TRUNCATE, DL, VA.getValVT(), Val); break; case CCValAssign::SExt: case CCValAssign::SExtUpper: Val = DAG.getNode(ISD::AssertSext, DL, VA.getLocVT(), Val, DAG.getValueType(VA.getValVT())); Val = DAG.getNode(ISD::TRUNCATE, DL, VA.getValVT(), Val); break; } InVals.push_back(Val); } return Chain; } static SDValue UnpackFromArgumentSlot(SDValue Val, const CCValAssign &VA, EVT ArgVT, const SDLoc &DL, SelectionDAG &DAG) { MVT LocVT = VA.getLocVT(); EVT ValVT = VA.getValVT(); // Shift into the upper bits if necessary. switch (VA.getLocInfo()) { default: break; case CCValAssign::AExtUpper: case CCValAssign::SExtUpper: case CCValAssign::ZExtUpper: { unsigned ValSizeInBits = ArgVT.getSizeInBits(); unsigned LocSizeInBits = VA.getLocVT().getSizeInBits(); unsigned Opcode = VA.getLocInfo() == CCValAssign::ZExtUpper ? ISD::SRL : ISD::SRA; Val = DAG.getNode( Opcode, DL, VA.getLocVT(), Val, DAG.getConstant(LocSizeInBits - ValSizeInBits, DL, VA.getLocVT())); break; } } // If this is an value smaller than the argument slot size (32-bit for O32, // 64-bit for N32/N64), it has been promoted in some way to the argument slot // size. Extract the value and insert any appropriate assertions regarding // sign/zero extension. switch (VA.getLocInfo()) { default: llvm_unreachable("Unknown loc info!"); case CCValAssign::Full: break; case CCValAssign::AExtUpper: case CCValAssign::AExt: Val = DAG.getNode(ISD::TRUNCATE, DL, ValVT, Val); break; case CCValAssign::SExtUpper: case CCValAssign::SExt: Val = DAG.getNode(ISD::AssertSext, DL, LocVT, Val, DAG.getValueType(ValVT)); Val = DAG.getNode(ISD::TRUNCATE, DL, ValVT, Val); break; case CCValAssign::ZExtUpper: case CCValAssign::ZExt: Val = DAG.getNode(ISD::AssertZext, DL, LocVT, Val, DAG.getValueType(ValVT)); Val = DAG.getNode(ISD::TRUNCATE, DL, ValVT, Val); break; case CCValAssign::BCvt: Val = DAG.getNode(ISD::BITCAST, DL, ValVT, Val); break; } return Val; } //===----------------------------------------------------------------------===// // Formal Arguments Calling Convention Implementation //===----------------------------------------------------------------------===// /// LowerFormalArguments - transform physical registers into virtual registers /// and generate load operations for arguments places on the stack. SDValue MipsTargetLowering::LowerFormalArguments( SDValue Chain, CallingConv::ID CallConv, bool IsVarArg, const SmallVectorImpl &Ins, const SDLoc &DL, SelectionDAG &DAG, SmallVectorImpl &InVals) const { MachineFunction &MF = DAG.getMachineFunction(); MachineFrameInfo &MFI = MF.getFrameInfo(); MipsFunctionInfo *MipsFI = MF.getInfo(); MipsFI->setVarArgsFrameIndex(0); // Used with vargs to acumulate store chains. std::vector OutChains; // Assign locations to all of the incoming arguments. SmallVector ArgLocs; MipsCCState CCInfo(CallConv, IsVarArg, DAG.getMachineFunction(), ArgLocs, *DAG.getContext()); CCInfo.AllocateStack(ABI.GetCalleeAllocdArgSizeInBytes(CallConv), 1); const Function &Func = DAG.getMachineFunction().getFunction(); Function::const_arg_iterator FuncArg = Func.arg_begin(); if (Func.hasFnAttribute("interrupt") && !Func.arg_empty()) report_fatal_error( "Functions with the interrupt attribute cannot have arguments!"); CCInfo.AnalyzeFormalArguments(Ins, CC_Mips_FixedArg); MipsFI->setFormalArgInfo(CCInfo.getNextStackOffset(), CCInfo.getInRegsParamsCount() > 0); unsigned CurArgIdx = 0; CCInfo.rewindByValRegsInfo(); for (unsigned i = 0, e = ArgLocs.size(); i != e; ++i) { CCValAssign &VA = ArgLocs[i]; if (Ins[i].isOrigArg()) { std::advance(FuncArg, Ins[i].getOrigArgIndex() - CurArgIdx); CurArgIdx = Ins[i].getOrigArgIndex(); } EVT ValVT = VA.getValVT(); ISD::ArgFlagsTy Flags = Ins[i].Flags; bool IsRegLoc = VA.isRegLoc(); if (Flags.isByVal()) { assert(Ins[i].isOrigArg() && "Byval arguments cannot be implicit"); unsigned FirstByValReg, LastByValReg; unsigned ByValIdx = CCInfo.getInRegsParamsProcessed(); CCInfo.getInRegsParamInfo(ByValIdx, FirstByValReg, LastByValReg); assert(Flags.getByValSize() && "ByVal args of size 0 should have been ignored by front-end."); assert(ByValIdx < CCInfo.getInRegsParamsCount()); copyByValRegs(Chain, DL, OutChains, DAG, Flags, InVals, &*FuncArg, FirstByValReg, LastByValReg, VA, CCInfo); CCInfo.nextInRegsParam(); continue; } // Arguments stored on registers if (IsRegLoc) { MVT RegVT = VA.getLocVT(); unsigned ArgReg = VA.getLocReg(); const TargetRegisterClass *RC = getRegClassFor(RegVT); // Transform the arguments stored on // physical registers into virtual ones unsigned Reg = addLiveIn(DAG.getMachineFunction(), ArgReg, RC); SDValue ArgValue = DAG.getCopyFromReg(Chain, DL, Reg, RegVT); ArgValue = UnpackFromArgumentSlot(ArgValue, VA, Ins[i].ArgVT, DL, DAG); // Handle floating point arguments passed in integer registers and // long double arguments passed in floating point registers. if ((RegVT == MVT::i32 && ValVT == MVT::f32) || (RegVT == MVT::i64 && ValVT == MVT::f64) || (RegVT == MVT::f64 && ValVT == MVT::i64)) ArgValue = DAG.getNode(ISD::BITCAST, DL, ValVT, ArgValue); else if (ABI.IsO32() && RegVT == MVT::i32 && ValVT == MVT::f64) { unsigned Reg2 = addLiveIn(DAG.getMachineFunction(), getNextIntArgReg(ArgReg), RC); SDValue ArgValue2 = DAG.getCopyFromReg(Chain, DL, Reg2, RegVT); if (!Subtarget.isLittle()) std::swap(ArgValue, ArgValue2); ArgValue = DAG.getNode(MipsISD::BuildPairF64, DL, MVT::f64, ArgValue, ArgValue2); } InVals.push_back(ArgValue); } else { // VA.isRegLoc() MVT LocVT = VA.getLocVT(); if (ABI.IsO32()) { // We ought to be able to use LocVT directly but O32 sets it to i32 // when allocating floating point values to integer registers. // This shouldn't influence how we load the value into registers unless // we are targeting softfloat. if (VA.getValVT().isFloatingPoint() && !Subtarget.useSoftFloat()) LocVT = VA.getValVT(); } // sanity check assert(VA.isMemLoc()); // The stack pointer offset is relative to the caller stack frame. int FI = MFI.CreateFixedObject(LocVT.getSizeInBits() / 8, VA.getLocMemOffset(), true); // Create load nodes to retrieve arguments from the stack SDValue FIN = DAG.getFrameIndex(FI, getPointerTy(DAG.getDataLayout())); SDValue ArgValue = DAG.getLoad( LocVT, DL, Chain, FIN, MachinePointerInfo::getFixedStack(DAG.getMachineFunction(), FI)); OutChains.push_back(ArgValue.getValue(1)); ArgValue = UnpackFromArgumentSlot(ArgValue, VA, Ins[i].ArgVT, DL, DAG); InVals.push_back(ArgValue); } } for (unsigned i = 0, e = ArgLocs.size(); i != e; ++i) { // The mips ABIs for returning structs by value requires that we copy // the sret argument into $v0 for the return. Save the argument into // a virtual register so that we can access it from the return points. if (Ins[i].Flags.isSRet()) { unsigned Reg = MipsFI->getSRetReturnReg(); if (!Reg) { Reg = MF.getRegInfo().createVirtualRegister( getRegClassFor(ABI.IsN64() ? MVT::i64 : MVT::i32)); MipsFI->setSRetReturnReg(Reg); } SDValue Copy = DAG.getCopyToReg(DAG.getEntryNode(), DL, Reg, InVals[i]); Chain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, Copy, Chain); break; } } if (IsVarArg) writeVarArgRegs(OutChains, Chain, DL, DAG, CCInfo); // All stores are grouped in one node to allow the matching between // the size of Ins and InVals. This only happens when on varg functions if (!OutChains.empty()) { OutChains.push_back(Chain); Chain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, OutChains); } return Chain; } //===----------------------------------------------------------------------===// // Return Value Calling Convention Implementation //===----------------------------------------------------------------------===// bool MipsTargetLowering::CanLowerReturn(CallingConv::ID CallConv, MachineFunction &MF, bool IsVarArg, const SmallVectorImpl &Outs, LLVMContext &Context) const { SmallVector RVLocs; MipsCCState CCInfo(CallConv, IsVarArg, MF, RVLocs, Context); return CCInfo.CheckReturn(Outs, RetCC_Mips); } bool MipsTargetLowering::shouldSignExtendTypeInLibCall(EVT Type, bool IsSigned) const { if ((ABI.IsN32() || ABI.IsN64()) && Type == MVT::i32) return true; return IsSigned; } SDValue MipsTargetLowering::LowerInterruptReturn(SmallVectorImpl &RetOps, const SDLoc &DL, SelectionDAG &DAG) const { MachineFunction &MF = DAG.getMachineFunction(); MipsFunctionInfo *MipsFI = MF.getInfo(); MipsFI->setISR(); return DAG.getNode(MipsISD::ERet, DL, MVT::Other, RetOps); } SDValue MipsTargetLowering::LowerReturn(SDValue Chain, CallingConv::ID CallConv, bool IsVarArg, const SmallVectorImpl &Outs, const SmallVectorImpl &OutVals, const SDLoc &DL, SelectionDAG &DAG) const { // CCValAssign - represent the assignment of // the return value to a location SmallVector RVLocs; MachineFunction &MF = DAG.getMachineFunction(); // CCState - Info about the registers and stack slot. MipsCCState CCInfo(CallConv, IsVarArg, MF, RVLocs, *DAG.getContext()); // Analyze return values. CCInfo.AnalyzeReturn(Outs, RetCC_Mips); SDValue Flag; SmallVector RetOps(1, Chain); // Copy the result values into the output registers. for (unsigned i = 0; i != RVLocs.size(); ++i) { SDValue Val = OutVals[i]; CCValAssign &VA = RVLocs[i]; assert(VA.isRegLoc() && "Can only return in registers!"); bool UseUpperBits = false; switch (VA.getLocInfo()) { default: llvm_unreachable("Unknown loc info!"); case CCValAssign::Full: break; case CCValAssign::BCvt: Val = DAG.getNode(ISD::BITCAST, DL, VA.getLocVT(), Val); break; case CCValAssign::AExtUpper: UseUpperBits = true; LLVM_FALLTHROUGH; case CCValAssign::AExt: Val = DAG.getNode(ISD::ANY_EXTEND, DL, VA.getLocVT(), Val); break; case CCValAssign::ZExtUpper: UseUpperBits = true; LLVM_FALLTHROUGH; case CCValAssign::ZExt: Val = DAG.getNode(ISD::ZERO_EXTEND, DL, VA.getLocVT(), Val); break; case CCValAssign::SExtUpper: UseUpperBits = true; LLVM_FALLTHROUGH; case CCValAssign::SExt: Val = DAG.getNode(ISD::SIGN_EXTEND, DL, VA.getLocVT(), Val); break; } if (UseUpperBits) { unsigned ValSizeInBits = Outs[i].ArgVT.getSizeInBits(); unsigned LocSizeInBits = VA.getLocVT().getSizeInBits(); Val = DAG.getNode( ISD::SHL, DL, VA.getLocVT(), Val, DAG.getConstant(LocSizeInBits - ValSizeInBits, DL, VA.getLocVT())); } Chain = DAG.getCopyToReg(Chain, DL, VA.getLocReg(), Val, Flag); // Guarantee that all emitted copies are stuck together with flags. Flag = Chain.getValue(1); RetOps.push_back(DAG.getRegister(VA.getLocReg(), VA.getLocVT())); } // The mips ABIs for returning structs by value requires that we copy // the sret argument into $v0 for the return. We saved the argument into // a virtual register in the entry block, so now we copy the value out // and into $v0. if (MF.getFunction().hasStructRetAttr()) { MipsFunctionInfo *MipsFI = MF.getInfo(); unsigned Reg = MipsFI->getSRetReturnReg(); if (!Reg) llvm_unreachable("sret virtual register not created in the entry block"); SDValue Val = DAG.getCopyFromReg(Chain, DL, Reg, getPointerTy(DAG.getDataLayout())); unsigned V0 = ABI.IsN64() ? Mips::V0_64 : Mips::V0; Chain = DAG.getCopyToReg(Chain, DL, V0, Val, Flag); Flag = Chain.getValue(1); RetOps.push_back(DAG.getRegister(V0, getPointerTy(DAG.getDataLayout()))); } RetOps[0] = Chain; // Update chain. // Add the flag if we have it. if (Flag.getNode()) RetOps.push_back(Flag); // ISRs must use "eret". if (DAG.getMachineFunction().getFunction().hasFnAttribute("interrupt")) return LowerInterruptReturn(RetOps, DL, DAG); // Standard return on Mips is a "jr $ra" return DAG.getNode(MipsISD::Ret, DL, MVT::Other, RetOps); } //===----------------------------------------------------------------------===// // Mips Inline Assembly Support //===----------------------------------------------------------------------===// /// getConstraintType - Given a constraint letter, return the type of /// constraint it is for this target. MipsTargetLowering::ConstraintType MipsTargetLowering::getConstraintType(StringRef Constraint) const { // Mips specific constraints // GCC config/mips/constraints.md // // 'd' : An address register. Equivalent to r // unless generating MIPS16 code. // 'y' : Equivalent to r; retained for // backwards compatibility. // 'c' : A register suitable for use in an indirect // jump. This will always be $25 for -mabicalls. // 'l' : The lo register. 1 word storage. // 'x' : The hilo register pair. Double word storage. if (Constraint.size() == 1) { switch (Constraint[0]) { default : break; case 'd': case 'y': case 'f': case 'c': case 'l': case 'x': return C_RegisterClass; case 'R': return C_Memory; } } if (Constraint == "ZC") return C_Memory; return TargetLowering::getConstraintType(Constraint); } /// Examine constraint type and operand type and determine a weight value. /// This object must already have been set up with the operand type /// and the current alternative constraint selected. TargetLowering::ConstraintWeight MipsTargetLowering::getSingleConstraintMatchWeight( AsmOperandInfo &info, const char *constraint) const { ConstraintWeight weight = CW_Invalid; Value *CallOperandVal = info.CallOperandVal; // If we don't have a value, we can't do a match, // but allow it at the lowest weight. if (!CallOperandVal) return CW_Default; Type *type = CallOperandVal->getType(); // Look at the constraint type. switch (*constraint) { default: weight = TargetLowering::getSingleConstraintMatchWeight(info, constraint); break; case 'd': case 'y': if (type->isIntegerTy()) weight = CW_Register; break; case 'f': // FPU or MSA register if (Subtarget.hasMSA() && type->isVectorTy() && cast(type)->getBitWidth() == 128) weight = CW_Register; else if (type->isFloatTy()) weight = CW_Register; break; case 'c': // $25 for indirect jumps case 'l': // lo register case 'x': // hilo register pair if (type->isIntegerTy()) weight = CW_SpecificReg; break; case 'I': // signed 16 bit immediate case 'J': // integer zero case 'K': // unsigned 16 bit immediate case 'L': // signed 32 bit immediate where lower 16 bits are 0 case 'N': // immediate in the range of -65535 to -1 (inclusive) case 'O': // signed 15 bit immediate (+- 16383) case 'P': // immediate in the range of 65535 to 1 (inclusive) if (isa(CallOperandVal)) weight = CW_Constant; break; case 'R': weight = CW_Memory; break; } return weight; } /// This is a helper function to parse a physical register string and split it /// into non-numeric and numeric parts (Prefix and Reg). The first boolean flag /// that is returned indicates whether parsing was successful. The second flag /// is true if the numeric part exists. static std::pair parsePhysicalReg(StringRef C, StringRef &Prefix, unsigned long long &Reg) { if (C.front() != '{' || C.back() != '}') return std::make_pair(false, false); // Search for the first numeric character. StringRef::const_iterator I, B = C.begin() + 1, E = C.end() - 1; I = std::find_if(B, E, isdigit); Prefix = StringRef(B, I - B); // The second flag is set to false if no numeric characters were found. if (I == E) return std::make_pair(true, false); // Parse the numeric characters. return std::make_pair(!getAsUnsignedInteger(StringRef(I, E - I), 10, Reg), true); } EVT MipsTargetLowering::getTypeForExtReturn(LLVMContext &Context, EVT VT, ISD::NodeType) const { bool Cond = !Subtarget.isABI_O32() && VT.getSizeInBits() == 32; EVT MinVT = getRegisterType(Context, Cond ? MVT::i64 : MVT::i32); return VT.bitsLT(MinVT) ? MinVT : VT; } std::pair MipsTargetLowering:: parseRegForInlineAsmConstraint(StringRef C, MVT VT) const { const TargetRegisterInfo *TRI = Subtarget.getRegisterInfo(); const TargetRegisterClass *RC; StringRef Prefix; unsigned long long Reg; std::pair R = parsePhysicalReg(C, Prefix, Reg); if (!R.first) return std::make_pair(0U, nullptr); if ((Prefix == "hi" || Prefix == "lo")) { // Parse hi/lo. // No numeric characters follow "hi" or "lo". if (R.second) return std::make_pair(0U, nullptr); RC = TRI->getRegClass(Prefix == "hi" ? Mips::HI32RegClassID : Mips::LO32RegClassID); return std::make_pair(*(RC->begin()), RC); } else if (Prefix.startswith("$msa")) { // Parse $msa(ir|csr|access|save|modify|request|map|unmap) // No numeric characters follow the name. if (R.second) return std::make_pair(0U, nullptr); Reg = StringSwitch(Prefix) .Case("$msair", Mips::MSAIR) .Case("$msacsr", Mips::MSACSR) .Case("$msaaccess", Mips::MSAAccess) .Case("$msasave", Mips::MSASave) .Case("$msamodify", Mips::MSAModify) .Case("$msarequest", Mips::MSARequest) .Case("$msamap", Mips::MSAMap) .Case("$msaunmap", Mips::MSAUnmap) .Default(0); if (!Reg) return std::make_pair(0U, nullptr); RC = TRI->getRegClass(Mips::MSACtrlRegClassID); return std::make_pair(Reg, RC); } if (!R.second) return std::make_pair(0U, nullptr); if (Prefix == "$f") { // Parse $f0-$f31. // If the size of FP registers is 64-bit or Reg is an even number, select // the 64-bit register class. Otherwise, select the 32-bit register class. if (VT == MVT::Other) VT = (Subtarget.isFP64bit() || !(Reg % 2)) ? MVT::f64 : MVT::f32; RC = getRegClassFor(VT); if (RC == &Mips::AFGR64RegClass) { assert(Reg % 2 == 0); Reg >>= 1; } } else if (Prefix == "$fcc") // Parse $fcc0-$fcc7. RC = TRI->getRegClass(Mips::FCCRegClassID); else if (Prefix == "$w") { // Parse $w0-$w31. RC = getRegClassFor((VT == MVT::Other) ? MVT::v16i8 : VT); } else { // Parse $0-$31. assert(Prefix == "$"); RC = getRegClassFor((VT == MVT::Other) ? MVT::i32 : VT); } assert(Reg < RC->getNumRegs()); return std::make_pair(*(RC->begin() + Reg), RC); } /// Given a register class constraint, like 'r', if this corresponds directly /// to an LLVM register class, return a register of 0 and the register class /// pointer. std::pair MipsTargetLowering::getRegForInlineAsmConstraint(const TargetRegisterInfo *TRI, StringRef Constraint, MVT VT) const { if (Constraint.size() == 1) { switch (Constraint[0]) { case 'd': // Address register. Same as 'r' unless generating MIPS16 code. case 'y': // Same as 'r'. Exists for compatibility. case 'r': if (VT == MVT::i32 || VT == MVT::i16 || VT == MVT::i8) { if (Subtarget.inMips16Mode()) return std::make_pair(0U, &Mips::CPU16RegsRegClass); return std::make_pair(0U, &Mips::GPR32RegClass); } if (VT == MVT::i64 && !Subtarget.isGP64bit()) return std::make_pair(0U, &Mips::GPR32RegClass); if (VT == MVT::i64 && Subtarget.isGP64bit()) return std::make_pair(0U, &Mips::GPR64RegClass); // This will generate an error message return std::make_pair(0U, nullptr); case 'f': // FPU or MSA register if (VT == MVT::v16i8) return std::make_pair(0U, &Mips::MSA128BRegClass); else if (VT == MVT::v8i16 || VT == MVT::v8f16) return std::make_pair(0U, &Mips::MSA128HRegClass); else if (VT == MVT::v4i32 || VT == MVT::v4f32) return std::make_pair(0U, &Mips::MSA128WRegClass); else if (VT == MVT::v2i64 || VT == MVT::v2f64) return std::make_pair(0U, &Mips::MSA128DRegClass); else if (VT == MVT::f32) return std::make_pair(0U, &Mips::FGR32RegClass); else if ((VT == MVT::f64) && (!Subtarget.isSingleFloat())) { if (Subtarget.isFP64bit()) return std::make_pair(0U, &Mips::FGR64RegClass); return std::make_pair(0U, &Mips::AFGR64RegClass); } break; case 'c': // register suitable for indirect jump if (VT == MVT::i32) return std::make_pair((unsigned)Mips::T9, &Mips::GPR32RegClass); if (VT == MVT::i64) return std::make_pair((unsigned)Mips::T9_64, &Mips::GPR64RegClass); // This will generate an error message return std::make_pair(0U, nullptr); case 'l': // use the `lo` register to store values // that are no bigger than a word if (VT == MVT::i32 || VT == MVT::i16 || VT == MVT::i8) return std::make_pair((unsigned)Mips::LO0, &Mips::LO32RegClass); return std::make_pair((unsigned)Mips::LO0_64, &Mips::LO64RegClass); case 'x': // use the concatenated `hi` and `lo` registers // to store doubleword values // Fixme: Not triggering the use of both hi and low // This will generate an error message return std::make_pair(0U, nullptr); } } std::pair R; R = parseRegForInlineAsmConstraint(Constraint, VT); if (R.second) return R; return TargetLowering::getRegForInlineAsmConstraint(TRI, Constraint, VT); } /// LowerAsmOperandForConstraint - Lower the specified operand into the Ops /// vector. If it is invalid, don't add anything to Ops. void MipsTargetLowering::LowerAsmOperandForConstraint(SDValue Op, std::string &Constraint, std::vector&Ops, SelectionDAG &DAG) const { SDLoc DL(Op); SDValue Result; // Only support length 1 constraints for now. if (Constraint.length() > 1) return; char ConstraintLetter = Constraint[0]; switch (ConstraintLetter) { default: break; // This will fall through to the generic implementation case 'I': // Signed 16 bit constant // If this fails, the parent routine will give an error if (ConstantSDNode *C = dyn_cast(Op)) { EVT Type = Op.getValueType(); int64_t Val = C->getSExtValue(); if (isInt<16>(Val)) { Result = DAG.getTargetConstant(Val, DL, Type); break; } } return; case 'J': // integer zero if (ConstantSDNode *C = dyn_cast(Op)) { EVT Type = Op.getValueType(); int64_t Val = C->getZExtValue(); if (Val == 0) { Result = DAG.getTargetConstant(0, DL, Type); break; } } return; case 'K': // unsigned 16 bit immediate if (ConstantSDNode *C = dyn_cast(Op)) { EVT Type = Op.getValueType(); uint64_t Val = (uint64_t)C->getZExtValue(); if (isUInt<16>(Val)) { Result = DAG.getTargetConstant(Val, DL, Type); break; } } return; case 'L': // signed 32 bit immediate where lower 16 bits are 0 if (ConstantSDNode *C = dyn_cast(Op)) { EVT Type = Op.getValueType(); int64_t Val = C->getSExtValue(); if ((isInt<32>(Val)) && ((Val & 0xffff) == 0)){ Result = DAG.getTargetConstant(Val, DL, Type); break; } } return; case 'N': // immediate in the range of -65535 to -1 (inclusive) if (ConstantSDNode *C = dyn_cast(Op)) { EVT Type = Op.getValueType(); int64_t Val = C->getSExtValue(); if ((Val >= -65535) && (Val <= -1)) { Result = DAG.getTargetConstant(Val, DL, Type); break; } } return; case 'O': // signed 15 bit immediate if (ConstantSDNode *C = dyn_cast(Op)) { EVT Type = Op.getValueType(); int64_t Val = C->getSExtValue(); if ((isInt<15>(Val))) { Result = DAG.getTargetConstant(Val, DL, Type); break; } } return; case 'P': // immediate in the range of 1 to 65535 (inclusive) if (ConstantSDNode *C = dyn_cast(Op)) { EVT Type = Op.getValueType(); int64_t Val = C->getSExtValue(); if ((Val <= 65535) && (Val >= 1)) { Result = DAG.getTargetConstant(Val, DL, Type); break; } } return; } if (Result.getNode()) { Ops.push_back(Result); return; } TargetLowering::LowerAsmOperandForConstraint(Op, Constraint, Ops, DAG); } bool MipsTargetLowering::isLegalAddressingMode(const DataLayout &DL, const AddrMode &AM, Type *Ty, unsigned AS, Instruction *I) const { // No global is ever allowed as a base. if (AM.BaseGV) return false; switch (AM.Scale) { case 0: // "r+i" or just "i", depending on HasBaseReg. break; case 1: if (!AM.HasBaseReg) // allow "r+i". break; return false; // disallow "r+r" or "r+r+i". default: return false; } return true; } bool MipsTargetLowering::isOffsetFoldingLegal(const GlobalAddressSDNode *GA) const { // The Mips target isn't yet aware of offsets. return false; } EVT MipsTargetLowering::getOptimalMemOpType(uint64_t Size, unsigned DstAlign, unsigned SrcAlign, bool IsMemset, bool ZeroMemset, bool MemcpyStrSrc, MachineFunction &MF) const { if (Subtarget.hasMips64()) return MVT::i64; return MVT::i32; } bool MipsTargetLowering::isFPImmLegal(const APFloat &Imm, EVT VT) const { if (VT != MVT::f32 && VT != MVT::f64) return false; if (Imm.isNegZero()) return false; return Imm.isZero(); } unsigned MipsTargetLowering::getJumpTableEncoding() const { // FIXME: For space reasons this should be: EK_GPRel32BlockAddress. if (ABI.IsN64() && isPositionIndependent()) return MachineJumpTableInfo::EK_GPRel64BlockAddress; return TargetLowering::getJumpTableEncoding(); } bool MipsTargetLowering::useSoftFloat() const { return Subtarget.useSoftFloat(); } void MipsTargetLowering::copyByValRegs( SDValue Chain, const SDLoc &DL, std::vector &OutChains, SelectionDAG &DAG, const ISD::ArgFlagsTy &Flags, SmallVectorImpl &InVals, const Argument *FuncArg, unsigned FirstReg, unsigned LastReg, const CCValAssign &VA, MipsCCState &State) const { MachineFunction &MF = DAG.getMachineFunction(); MachineFrameInfo &MFI = MF.getFrameInfo(); unsigned GPRSizeInBytes = Subtarget.getGPRSizeInBytes(); unsigned NumRegs = LastReg - FirstReg; unsigned RegAreaSize = NumRegs * GPRSizeInBytes; unsigned FrameObjSize = std::max(Flags.getByValSize(), RegAreaSize); int FrameObjOffset; ArrayRef ByValArgRegs = ABI.GetByValArgRegs(); if (RegAreaSize) FrameObjOffset = (int)ABI.GetCalleeAllocdArgSizeInBytes(State.getCallingConv()) - (int)((ByValArgRegs.size() - FirstReg) * GPRSizeInBytes); else FrameObjOffset = VA.getLocMemOffset(); // Create frame object. EVT PtrTy = getPointerTy(DAG.getDataLayout()); // Make the fixed object stored to mutable so that the load instructions // referencing it have their memory dependencies added. // Set the frame object as isAliased which clears the underlying objects // vector in ScheduleDAGInstrs::buildSchedGraph() resulting in addition of all // stores as dependencies for loads referencing this fixed object. int FI = MFI.CreateFixedObject(FrameObjSize, FrameObjOffset, false, true); SDValue FIN = DAG.getFrameIndex(FI, PtrTy); InVals.push_back(FIN); if (!NumRegs) return; // Copy arg registers. MVT RegTy = MVT::getIntegerVT(GPRSizeInBytes * 8); const TargetRegisterClass *RC = getRegClassFor(RegTy); for (unsigned I = 0; I < NumRegs; ++I) { unsigned ArgReg = ByValArgRegs[FirstReg + I]; unsigned VReg = addLiveIn(MF, ArgReg, RC); unsigned Offset = I * GPRSizeInBytes; SDValue StorePtr = DAG.getNode(ISD::ADD, DL, PtrTy, FIN, DAG.getConstant(Offset, DL, PtrTy)); SDValue Store = DAG.getStore(Chain, DL, DAG.getRegister(VReg, RegTy), StorePtr, MachinePointerInfo(FuncArg, Offset)); OutChains.push_back(Store); } } // Copy byVal arg to registers and stack. void MipsTargetLowering::passByValArg( SDValue Chain, const SDLoc &DL, std::deque> &RegsToPass, SmallVectorImpl &MemOpChains, SDValue StackPtr, MachineFrameInfo &MFI, SelectionDAG &DAG, SDValue Arg, unsigned FirstReg, unsigned LastReg, const ISD::ArgFlagsTy &Flags, bool isLittle, const CCValAssign &VA) const { unsigned ByValSizeInBytes = Flags.getByValSize(); unsigned OffsetInBytes = 0; // From beginning of struct unsigned RegSizeInBytes = Subtarget.getGPRSizeInBytes(); unsigned Alignment = std::min(Flags.getByValAlign(), RegSizeInBytes); EVT PtrTy = getPointerTy(DAG.getDataLayout()), RegTy = MVT::getIntegerVT(RegSizeInBytes * 8); unsigned NumRegs = LastReg - FirstReg; if (NumRegs) { ArrayRef ArgRegs = ABI.GetByValArgRegs(); bool LeftoverBytes = (NumRegs * RegSizeInBytes > ByValSizeInBytes); unsigned I = 0; // Copy words to registers. for (; I < NumRegs - LeftoverBytes; ++I, OffsetInBytes += RegSizeInBytes) { SDValue LoadPtr = DAG.getNode(ISD::ADD, DL, PtrTy, Arg, DAG.getConstant(OffsetInBytes, DL, PtrTy)); SDValue LoadVal = DAG.getLoad(RegTy, DL, Chain, LoadPtr, MachinePointerInfo(), Alignment); MemOpChains.push_back(LoadVal.getValue(1)); unsigned ArgReg = ArgRegs[FirstReg + I]; RegsToPass.push_back(std::make_pair(ArgReg, LoadVal)); } // Return if the struct has been fully copied. if (ByValSizeInBytes == OffsetInBytes) return; // Copy the remainder of the byval argument with sub-word loads and shifts. if (LeftoverBytes) { SDValue Val; for (unsigned LoadSizeInBytes = RegSizeInBytes / 2, TotalBytesLoaded = 0; OffsetInBytes < ByValSizeInBytes; LoadSizeInBytes /= 2) { unsigned RemainingSizeInBytes = ByValSizeInBytes - OffsetInBytes; if (RemainingSizeInBytes < LoadSizeInBytes) continue; // Load subword. SDValue LoadPtr = DAG.getNode(ISD::ADD, DL, PtrTy, Arg, DAG.getConstant(OffsetInBytes, DL, PtrTy)); SDValue LoadVal = DAG.getExtLoad( ISD::ZEXTLOAD, DL, RegTy, Chain, LoadPtr, MachinePointerInfo(), MVT::getIntegerVT(LoadSizeInBytes * 8), Alignment); MemOpChains.push_back(LoadVal.getValue(1)); // Shift the loaded value. unsigned Shamt; if (isLittle) Shamt = TotalBytesLoaded * 8; else Shamt = (RegSizeInBytes - (TotalBytesLoaded + LoadSizeInBytes)) * 8; SDValue Shift = DAG.getNode(ISD::SHL, DL, RegTy, LoadVal, DAG.getConstant(Shamt, DL, MVT::i32)); if (Val.getNode()) Val = DAG.getNode(ISD::OR, DL, RegTy, Val, Shift); else Val = Shift; OffsetInBytes += LoadSizeInBytes; TotalBytesLoaded += LoadSizeInBytes; Alignment = std::min(Alignment, LoadSizeInBytes); } unsigned ArgReg = ArgRegs[FirstReg + I]; RegsToPass.push_back(std::make_pair(ArgReg, Val)); return; } } // Copy remainder of byval arg to it with memcpy. unsigned MemCpySize = ByValSizeInBytes - OffsetInBytes; SDValue Src = DAG.getNode(ISD::ADD, DL, PtrTy, Arg, DAG.getConstant(OffsetInBytes, DL, PtrTy)); SDValue Dst = DAG.getNode(ISD::ADD, DL, PtrTy, StackPtr, DAG.getIntPtrConstant(VA.getLocMemOffset(), DL)); Chain = DAG.getMemcpy(Chain, DL, Dst, Src, DAG.getConstant(MemCpySize, DL, PtrTy), Alignment, /*isVolatile=*/false, /*AlwaysInline=*/false, /*isTailCall=*/false, MachinePointerInfo(), MachinePointerInfo()); MemOpChains.push_back(Chain); } void MipsTargetLowering::writeVarArgRegs(std::vector &OutChains, SDValue Chain, const SDLoc &DL, SelectionDAG &DAG, CCState &State) const { ArrayRef ArgRegs = ABI.GetVarArgRegs(); unsigned Idx = State.getFirstUnallocated(ArgRegs); unsigned RegSizeInBytes = Subtarget.getGPRSizeInBytes(); MVT RegTy = MVT::getIntegerVT(RegSizeInBytes * 8); const TargetRegisterClass *RC = getRegClassFor(RegTy); MachineFunction &MF = DAG.getMachineFunction(); MachineFrameInfo &MFI = MF.getFrameInfo(); MipsFunctionInfo *MipsFI = MF.getInfo(); // Offset of the first variable argument from stack pointer. int VaArgOffset; if (ArgRegs.size() == Idx) VaArgOffset = alignTo(State.getNextStackOffset(), RegSizeInBytes); else { VaArgOffset = (int)ABI.GetCalleeAllocdArgSizeInBytes(State.getCallingConv()) - (int)(RegSizeInBytes * (ArgRegs.size() - Idx)); } // Record the frame index of the first variable argument // which is a value necessary to VASTART. int FI = MFI.CreateFixedObject(RegSizeInBytes, VaArgOffset, true); MipsFI->setVarArgsFrameIndex(FI); // Copy the integer registers that have not been used for argument passing // to the argument register save area. For O32, the save area is allocated // in the caller's stack frame, while for N32/64, it is allocated in the // callee's stack frame. for (unsigned I = Idx; I < ArgRegs.size(); ++I, VaArgOffset += RegSizeInBytes) { unsigned Reg = addLiveIn(MF, ArgRegs[I], RC); SDValue ArgValue = DAG.getCopyFromReg(Chain, DL, Reg, RegTy); FI = MFI.CreateFixedObject(RegSizeInBytes, VaArgOffset, true); SDValue PtrOff = DAG.getFrameIndex(FI, getPointerTy(DAG.getDataLayout())); SDValue Store = DAG.getStore(Chain, DL, ArgValue, PtrOff, MachinePointerInfo()); cast(Store.getNode())->getMemOperand()->setValue( (Value *)nullptr); OutChains.push_back(Store); } } void MipsTargetLowering::HandleByVal(CCState *State, unsigned &Size, unsigned Align) const { const TargetFrameLowering *TFL = Subtarget.getFrameLowering(); assert(Size && "Byval argument's size shouldn't be 0."); Align = std::min(Align, TFL->getStackAlignment()); unsigned FirstReg = 0; unsigned NumRegs = 0; if (State->getCallingConv() != CallingConv::Fast) { unsigned RegSizeInBytes = Subtarget.getGPRSizeInBytes(); ArrayRef IntArgRegs = ABI.GetByValArgRegs(); // FIXME: The O32 case actually describes no shadow registers. const MCPhysReg *ShadowRegs = ABI.IsO32() ? IntArgRegs.data() : Mips64DPRegs; // We used to check the size as well but we can't do that anymore since // CCState::HandleByVal() rounds up the size after calling this function. assert(!(Align % RegSizeInBytes) && "Byval argument's alignment should be a multiple of" "RegSizeInBytes."); FirstReg = State->getFirstUnallocated(IntArgRegs); // If Align > RegSizeInBytes, the first arg register must be even. // FIXME: This condition happens to do the right thing but it's not the // right way to test it. We want to check that the stack frame offset // of the register is aligned. if ((Align > RegSizeInBytes) && (FirstReg % 2)) { State->AllocateReg(IntArgRegs[FirstReg], ShadowRegs[FirstReg]); ++FirstReg; } // Mark the registers allocated. Size = alignTo(Size, RegSizeInBytes); for (unsigned I = FirstReg; Size > 0 && (I < IntArgRegs.size()); Size -= RegSizeInBytes, ++I, ++NumRegs) State->AllocateReg(IntArgRegs[I], ShadowRegs[I]); } State->addInRegsParamInfo(FirstReg, FirstReg + NumRegs); } MachineBasicBlock *MipsTargetLowering::emitPseudoSELECT(MachineInstr &MI, MachineBasicBlock *BB, bool isFPCmp, unsigned Opc) const { assert(!(Subtarget.hasMips4() || Subtarget.hasMips32()) && "Subtarget already supports SELECT nodes with the use of" "conditional-move instructions."); const TargetInstrInfo *TII = Subtarget.getInstrInfo(); DebugLoc DL = MI.getDebugLoc(); // To "insert" a SELECT instruction, we actually have to insert the // diamond control-flow pattern. The incoming instruction knows the // destination vreg to set, the condition code register to branch on, the // true/false values to select between, and a branch opcode to use. const BasicBlock *LLVM_BB = BB->getBasicBlock(); MachineFunction::iterator It = ++BB->getIterator(); // thisMBB: // ... // TrueVal = ... // setcc r1, r2, r3 // bNE r1, r0, copy1MBB // fallthrough --> copy0MBB MachineBasicBlock *thisMBB = BB; MachineFunction *F = BB->getParent(); MachineBasicBlock *copy0MBB = F->CreateMachineBasicBlock(LLVM_BB); MachineBasicBlock *sinkMBB = F->CreateMachineBasicBlock(LLVM_BB); F->insert(It, copy0MBB); F->insert(It, sinkMBB); // Transfer the remainder of BB and its successor edges to sinkMBB. sinkMBB->splice(sinkMBB->begin(), BB, std::next(MachineBasicBlock::iterator(MI)), BB->end()); sinkMBB->transferSuccessorsAndUpdatePHIs(BB); // Next, add the true and fallthrough blocks as its successors. BB->addSuccessor(copy0MBB); BB->addSuccessor(sinkMBB); if (isFPCmp) { // bc1[tf] cc, sinkMBB BuildMI(BB, DL, TII->get(Opc)) .addReg(MI.getOperand(1).getReg()) .addMBB(sinkMBB); } else { // bne rs, $0, sinkMBB BuildMI(BB, DL, TII->get(Opc)) .addReg(MI.getOperand(1).getReg()) .addReg(Mips::ZERO) .addMBB(sinkMBB); } // copy0MBB: // %FalseValue = ... // # fallthrough to sinkMBB BB = copy0MBB; // Update machine-CFG edges BB->addSuccessor(sinkMBB); // sinkMBB: // %Result = phi [ %TrueValue, thisMBB ], [ %FalseValue, copy0MBB ] // ... BB = sinkMBB; BuildMI(*BB, BB->begin(), DL, TII->get(Mips::PHI), MI.getOperand(0).getReg()) .addReg(MI.getOperand(2).getReg()) .addMBB(thisMBB) .addReg(MI.getOperand(3).getReg()) .addMBB(copy0MBB); MI.eraseFromParent(); // The pseudo instruction is gone now. return BB; } MachineBasicBlock *MipsTargetLowering::emitPseudoD_SELECT(MachineInstr &MI, MachineBasicBlock *BB) const { assert(!(Subtarget.hasMips4() || Subtarget.hasMips32()) && "Subtarget already supports SELECT nodes with the use of" "conditional-move instructions."); const TargetInstrInfo *TII = Subtarget.getInstrInfo(); DebugLoc DL = MI.getDebugLoc(); // D_SELECT substitutes two SELECT nodes that goes one after another and // have the same condition operand. On machines which don't have // conditional-move instruction, it reduces unnecessary branch instructions // which are result of using two diamond patterns that are result of two // SELECT pseudo instructions. const BasicBlock *LLVM_BB = BB->getBasicBlock(); MachineFunction::iterator It = ++BB->getIterator(); // thisMBB: // ... // TrueVal = ... // setcc r1, r2, r3 // bNE r1, r0, copy1MBB // fallthrough --> copy0MBB MachineBasicBlock *thisMBB = BB; MachineFunction *F = BB->getParent(); MachineBasicBlock *copy0MBB = F->CreateMachineBasicBlock(LLVM_BB); MachineBasicBlock *sinkMBB = F->CreateMachineBasicBlock(LLVM_BB); F->insert(It, copy0MBB); F->insert(It, sinkMBB); // Transfer the remainder of BB and its successor edges to sinkMBB. sinkMBB->splice(sinkMBB->begin(), BB, std::next(MachineBasicBlock::iterator(MI)), BB->end()); sinkMBB->transferSuccessorsAndUpdatePHIs(BB); // Next, add the true and fallthrough blocks as its successors. BB->addSuccessor(copy0MBB); BB->addSuccessor(sinkMBB); // bne rs, $0, sinkMBB BuildMI(BB, DL, TII->get(Mips::BNE)) .addReg(MI.getOperand(2).getReg()) .addReg(Mips::ZERO) .addMBB(sinkMBB); // copy0MBB: // %FalseValue = ... // # fallthrough to sinkMBB BB = copy0MBB; // Update machine-CFG edges BB->addSuccessor(sinkMBB); // sinkMBB: // %Result = phi [ %TrueValue, thisMBB ], [ %FalseValue, copy0MBB ] // ... BB = sinkMBB; // Use two PHI nodes to select two reults BuildMI(*BB, BB->begin(), DL, TII->get(Mips::PHI), MI.getOperand(0).getReg()) .addReg(MI.getOperand(3).getReg()) .addMBB(thisMBB) .addReg(MI.getOperand(5).getReg()) .addMBB(copy0MBB); BuildMI(*BB, BB->begin(), DL, TII->get(Mips::PHI), MI.getOperand(1).getReg()) .addReg(MI.getOperand(4).getReg()) .addMBB(thisMBB) .addReg(MI.getOperand(6).getReg()) .addMBB(copy0MBB); MI.eraseFromParent(); // The pseudo instruction is gone now. return BB; } // FIXME? Maybe this could be a TableGen attribute on some registers and // this table could be generated automatically from RegInfo. unsigned MipsTargetLowering::getRegisterByName(const char* RegName, EVT VT, SelectionDAG &DAG) const { // Named registers is expected to be fairly rare. For now, just support $28 // since the linux kernel uses it. if (Subtarget.isGP64bit()) { unsigned Reg = StringSwitch(RegName) .Case("$28", Mips::GP_64) .Default(0); if (Reg) return Reg; } else { unsigned Reg = StringSwitch(RegName) .Case("$28", Mips::GP) .Default(0); if (Reg) return Reg; } report_fatal_error("Invalid register name global variable"); } Index: vendor/llvm/dist-release_80/lib/Target/Mips/MipsISelLowering.h =================================================================== --- vendor/llvm/dist-release_80/lib/Target/Mips/MipsISelLowering.h (revision 343793) +++ vendor/llvm/dist-release_80/lib/Target/Mips/MipsISelLowering.h (revision 343794) @@ -1,723 +1,726 @@ //===- MipsISelLowering.h - Mips DAG Lowering Interface ---------*- C++ -*-===// // // The LLVM Compiler Infrastructure // // This file is distributed under the University of Illinois Open Source // License. See LICENSE.TXT for details. // //===----------------------------------------------------------------------===// // // This file defines the interfaces that Mips uses to lower LLVM code into a // selection DAG. // //===----------------------------------------------------------------------===// #ifndef LLVM_LIB_TARGET_MIPS_MIPSISELLOWERING_H #define LLVM_LIB_TARGET_MIPS_MIPSISELLOWERING_H #include "MCTargetDesc/MipsABIInfo.h" #include "MCTargetDesc/MipsBaseInfo.h" #include "MCTargetDesc/MipsMCTargetDesc.h" #include "Mips.h" #include "llvm/CodeGen/CallingConvLower.h" #include "llvm/CodeGen/ISDOpcodes.h" #include "llvm/CodeGen/MachineMemOperand.h" #include "llvm/CodeGen/SelectionDAG.h" #include "llvm/CodeGen/SelectionDAGNodes.h" #include "llvm/CodeGen/TargetLowering.h" #include "llvm/CodeGen/ValueTypes.h" #include "llvm/IR/CallingConv.h" #include "llvm/IR/InlineAsm.h" #include "llvm/IR/Type.h" #include "llvm/Support/MachineValueType.h" #include "llvm/Target/TargetMachine.h" #include #include #include #include #include #include namespace llvm { class Argument; class CCState; class CCValAssign; class FastISel; class FunctionLoweringInfo; class MachineBasicBlock; class MachineFrameInfo; class MachineInstr; class MipsCCState; class MipsFunctionInfo; class MipsSubtarget; class MipsTargetMachine; class TargetLibraryInfo; class TargetRegisterClass; namespace MipsISD { enum NodeType : unsigned { // Start the numbering from where ISD NodeType finishes. FIRST_NUMBER = ISD::BUILTIN_OP_END, // Jump and link (call) JmpLink, // Tail call TailCall, // Get the Highest (63-48) 16 bits from a 64-bit immediate Highest, // Get the Higher (47-32) 16 bits from a 64-bit immediate Higher, // Get the High 16 bits from a 32/64-bit immediate // No relation with Mips Hi register Hi, // Get the Lower 16 bits from a 32/64-bit immediate // No relation with Mips Lo register Lo, // Get the High 16 bits from a 32 bit immediate for accessing the GOT. GotHi, // Get the High 16 bits from a 32-bit immediate for accessing TLS. TlsHi, // Handle gp_rel (small data/bss sections) relocation. GPRel, // Thread Pointer ThreadPointer, // Vector Floating Point Multiply and Subtract FMS, // Floating Point Branch Conditional FPBrcond, // Floating Point Compare FPCmp, // Floating point select FSELECT, // Node used to generate an MTC1 i32 to f64 instruction MTC1_D64, // Floating Point Conditional Moves CMovFP_T, CMovFP_F, // FP-to-int truncation node. TruncIntFP, // Return Ret, // Interrupt, exception, error trap Return ERet, // Software Exception Return. EH_RETURN, // Node used to extract integer from accumulator. MFHI, MFLO, // Node used to insert integers to accumulator. MTLOHI, // Mult nodes. Mult, Multu, // MAdd/Sub nodes MAdd, MAddu, MSub, MSubu, // DivRem(u) DivRem, DivRemU, DivRem16, DivRemU16, BuildPairF64, ExtractElementF64, Wrapper, DynAlloc, Sync, Ext, Ins, CIns, // EXTR.W instrinsic nodes. EXTP, EXTPDP, EXTR_S_H, EXTR_W, EXTR_R_W, EXTR_RS_W, SHILO, MTHLIP, // DPA.W intrinsic nodes. MULSAQ_S_W_PH, MAQ_S_W_PHL, MAQ_S_W_PHR, MAQ_SA_W_PHL, MAQ_SA_W_PHR, DPAU_H_QBL, DPAU_H_QBR, DPSU_H_QBL, DPSU_H_QBR, DPAQ_S_W_PH, DPSQ_S_W_PH, DPAQ_SA_L_W, DPSQ_SA_L_W, DPA_W_PH, DPS_W_PH, DPAQX_S_W_PH, DPAQX_SA_W_PH, DPAX_W_PH, DPSX_W_PH, DPSQX_S_W_PH, DPSQX_SA_W_PH, MULSA_W_PH, MULT, MULTU, MADD_DSP, MADDU_DSP, MSUB_DSP, MSUBU_DSP, // DSP shift nodes. SHLL_DSP, SHRA_DSP, SHRL_DSP, // DSP setcc and select_cc nodes. SETCC_DSP, SELECT_CC_DSP, // Vector comparisons. // These take a vector and return a boolean. VALL_ZERO, VANY_ZERO, VALL_NONZERO, VANY_NONZERO, // These take a vector and return a vector bitmask. VCEQ, VCLE_S, VCLE_U, VCLT_S, VCLT_U, // Vector Shuffle with mask as an operand VSHF, // Generic shuffle SHF, // 4-element set shuffle. ILVEV, // Interleave even elements ILVOD, // Interleave odd elements ILVL, // Interleave left elements ILVR, // Interleave right elements PCKEV, // Pack even elements PCKOD, // Pack odd elements // Vector Lane Copy INSVE, // Copy element from one vector to another // Combined (XOR (OR $a, $b), -1) VNOR, // Extended vector element extraction VEXTRACT_SEXT_ELT, VEXTRACT_ZEXT_ELT, // Load/Store Left/Right nodes. LWL = ISD::FIRST_TARGET_MEMORY_OPCODE, LWR, SWL, SWR, LDL, LDR, SDL, SDR }; } // ene namespace MipsISD //===--------------------------------------------------------------------===// // TargetLowering Implementation //===--------------------------------------------------------------------===// class MipsTargetLowering : public TargetLowering { bool isMicroMips; public: explicit MipsTargetLowering(const MipsTargetMachine &TM, const MipsSubtarget &STI); static const MipsTargetLowering *create(const MipsTargetMachine &TM, const MipsSubtarget &STI); /// createFastISel - This method returns a target specific FastISel object, /// or null if the target does not support "fast" ISel. FastISel *createFastISel(FunctionLoweringInfo &funcInfo, const TargetLibraryInfo *libInfo) const override; MVT getScalarShiftAmountTy(const DataLayout &, EVT) const override { return MVT::i32; } EVT getTypeForExtReturn(LLVMContext &Context, EVT VT, ISD::NodeType) const override; bool isCheapToSpeculateCttz() const override; bool isCheapToSpeculateCtlz() const override; /// Return the register type for a given MVT, ensuring vectors are treated /// as a series of gpr sized integers. MVT getRegisterTypeForCallingConv(LLVMContext &Context, CallingConv::ID CC, EVT VT) const override; /// Return the number of registers for a given MVT, ensuring vectors are /// treated as a series of gpr sized integers. unsigned getNumRegistersForCallingConv(LLVMContext &Context, CallingConv::ID CC, EVT VT) const override; /// Break down vectors to the correct number of gpr sized integers. unsigned getVectorTypeBreakdownForCallingConv( LLVMContext &Context, CallingConv::ID CC, EVT VT, EVT &IntermediateVT, unsigned &NumIntermediates, MVT &RegisterVT) const override; /// Return the correct alignment for the current calling convention. unsigned getABIAlignmentForCallingConv(Type *ArgTy, DataLayout DL) const override { if (ArgTy->isVectorTy()) return std::min(DL.getABITypeAlignment(ArgTy), 8U); return DL.getABITypeAlignment(ArgTy); } ISD::NodeType getExtendForAtomicOps() const override { return ISD::SIGN_EXTEND; } void LowerOperationWrapper(SDNode *N, SmallVectorImpl &Results, SelectionDAG &DAG) const override; /// LowerOperation - Provide custom lowering hooks for some operations. SDValue LowerOperation(SDValue Op, SelectionDAG &DAG) const override; /// ReplaceNodeResults - Replace the results of node with an illegal result /// type with new values built out of custom code. /// void ReplaceNodeResults(SDNode *N, SmallVectorImpl&Results, SelectionDAG &DAG) const override; /// getTargetNodeName - This method returns the name of a target specific // DAG node. const char *getTargetNodeName(unsigned Opcode) const override; /// getSetCCResultType - get the ISD::SETCC result ValueType EVT getSetCCResultType(const DataLayout &DL, LLVMContext &Context, EVT VT) const override; SDValue PerformDAGCombine(SDNode *N, DAGCombinerInfo &DCI) const override; MachineBasicBlock * EmitInstrWithCustomInserter(MachineInstr &MI, MachineBasicBlock *MBB) const override; + void AdjustInstrPostInstrSelection(MachineInstr &MI, + SDNode *Node) const override; + void HandleByVal(CCState *, unsigned &, unsigned) const override; unsigned getRegisterByName(const char* RegName, EVT VT, SelectionDAG &DAG) const override; /// If a physical register, this returns the register that receives the /// exception address on entry to an EH pad. unsigned getExceptionPointerRegister(const Constant *PersonalityFn) const override { return ABI.IsN64() ? Mips::A0_64 : Mips::A0; } /// If a physical register, this returns the register that receives the /// exception typeid on entry to a landing pad. unsigned getExceptionSelectorRegister(const Constant *PersonalityFn) const override { return ABI.IsN64() ? Mips::A1_64 : Mips::A1; } /// Returns true if a cast between SrcAS and DestAS is a noop. bool isNoopAddrSpaceCast(unsigned SrcAS, unsigned DestAS) const override { // Mips doesn't have any special address spaces so we just reserve // the first 256 for software use (e.g. OpenCL) and treat casts // between them as noops. return SrcAS < 256 && DestAS < 256; } bool isJumpTableRelative() const override { return getTargetMachine().isPositionIndependent(); } CCAssignFn *CCAssignFnForCall() const; CCAssignFn *CCAssignFnForReturn() const; protected: SDValue getGlobalReg(SelectionDAG &DAG, EVT Ty) const; // This method creates the following nodes, which are necessary for // computing a local symbol's address: // // (add (load (wrapper $gp, %got(sym)), %lo(sym)) template SDValue getAddrLocal(NodeTy *N, const SDLoc &DL, EVT Ty, SelectionDAG &DAG, bool IsN32OrN64) const { unsigned GOTFlag = IsN32OrN64 ? MipsII::MO_GOT_PAGE : MipsII::MO_GOT; SDValue GOT = DAG.getNode(MipsISD::Wrapper, DL, Ty, getGlobalReg(DAG, Ty), getTargetNode(N, Ty, DAG, GOTFlag)); SDValue Load = DAG.getLoad(Ty, DL, DAG.getEntryNode(), GOT, MachinePointerInfo::getGOT(DAG.getMachineFunction())); unsigned LoFlag = IsN32OrN64 ? MipsII::MO_GOT_OFST : MipsII::MO_ABS_LO; SDValue Lo = DAG.getNode(MipsISD::Lo, DL, Ty, getTargetNode(N, Ty, DAG, LoFlag)); return DAG.getNode(ISD::ADD, DL, Ty, Load, Lo); } // This method creates the following nodes, which are necessary for // computing a global symbol's address: // // (load (wrapper $gp, %got(sym))) template SDValue getAddrGlobal(NodeTy *N, const SDLoc &DL, EVT Ty, SelectionDAG &DAG, unsigned Flag, SDValue Chain, const MachinePointerInfo &PtrInfo) const { SDValue Tgt = DAG.getNode(MipsISD::Wrapper, DL, Ty, getGlobalReg(DAG, Ty), getTargetNode(N, Ty, DAG, Flag)); return DAG.getLoad(Ty, DL, Chain, Tgt, PtrInfo); } // This method creates the following nodes, which are necessary for // computing a global symbol's address in large-GOT mode: // // (load (wrapper (add %hi(sym), $gp), %lo(sym))) template SDValue getAddrGlobalLargeGOT(NodeTy *N, const SDLoc &DL, EVT Ty, SelectionDAG &DAG, unsigned HiFlag, unsigned LoFlag, SDValue Chain, const MachinePointerInfo &PtrInfo) const { SDValue Hi = DAG.getNode(MipsISD::GotHi, DL, Ty, getTargetNode(N, Ty, DAG, HiFlag)); Hi = DAG.getNode(ISD::ADD, DL, Ty, Hi, getGlobalReg(DAG, Ty)); SDValue Wrapper = DAG.getNode(MipsISD::Wrapper, DL, Ty, Hi, getTargetNode(N, Ty, DAG, LoFlag)); return DAG.getLoad(Ty, DL, Chain, Wrapper, PtrInfo); } // This method creates the following nodes, which are necessary for // computing a symbol's address in non-PIC mode: // // (add %hi(sym), %lo(sym)) // // This method covers O32, N32 and N64 in sym32 mode. template SDValue getAddrNonPIC(NodeTy *N, const SDLoc &DL, EVT Ty, SelectionDAG &DAG) const { SDValue Hi = getTargetNode(N, Ty, DAG, MipsII::MO_ABS_HI); SDValue Lo = getTargetNode(N, Ty, DAG, MipsII::MO_ABS_LO); return DAG.getNode(ISD::ADD, DL, Ty, DAG.getNode(MipsISD::Hi, DL, Ty, Hi), DAG.getNode(MipsISD::Lo, DL, Ty, Lo)); } // This method creates the following nodes, which are necessary for // computing a symbol's address in non-PIC mode for N64. // // (add (shl (add (shl (add %highest(sym), %higher(sim)), 16), %high(sym)), // 16), %lo(%sym)) // // FIXME: This method is not efficent for (micro)MIPS64R6. template SDValue getAddrNonPICSym64(NodeTy *N, const SDLoc &DL, EVT Ty, SelectionDAG &DAG) const { SDValue Hi = getTargetNode(N, Ty, DAG, MipsII::MO_ABS_HI); SDValue Lo = getTargetNode(N, Ty, DAG, MipsII::MO_ABS_LO); SDValue Highest = DAG.getNode(MipsISD::Highest, DL, Ty, getTargetNode(N, Ty, DAG, MipsII::MO_HIGHEST)); SDValue Higher = getTargetNode(N, Ty, DAG, MipsII::MO_HIGHER); SDValue HigherPart = DAG.getNode(ISD::ADD, DL, Ty, Highest, DAG.getNode(MipsISD::Higher, DL, Ty, Higher)); SDValue Cst = DAG.getConstant(16, DL, MVT::i32); SDValue Shift = DAG.getNode(ISD::SHL, DL, Ty, HigherPart, Cst); SDValue Add = DAG.getNode(ISD::ADD, DL, Ty, Shift, DAG.getNode(MipsISD::Hi, DL, Ty, Hi)); SDValue Shift2 = DAG.getNode(ISD::SHL, DL, Ty, Add, Cst); return DAG.getNode(ISD::ADD, DL, Ty, Shift2, DAG.getNode(MipsISD::Lo, DL, Ty, Lo)); } // This method creates the following nodes, which are necessary for // computing a symbol's address using gp-relative addressing: // // (add $gp, %gp_rel(sym)) template SDValue getAddrGPRel(NodeTy *N, const SDLoc &DL, EVT Ty, SelectionDAG &DAG, bool IsN64) const { SDValue GPRel = getTargetNode(N, Ty, DAG, MipsII::MO_GPREL); return DAG.getNode( ISD::ADD, DL, Ty, DAG.getRegister(IsN64 ? Mips::GP_64 : Mips::GP, Ty), DAG.getNode(MipsISD::GPRel, DL, DAG.getVTList(Ty), GPRel)); } /// This function fills Ops, which is the list of operands that will later /// be used when a function call node is created. It also generates /// copyToReg nodes to set up argument registers. virtual void getOpndList(SmallVectorImpl &Ops, std::deque> &RegsToPass, bool IsPICCall, bool GlobalOrExternal, bool InternalLinkage, bool IsCallReloc, CallLoweringInfo &CLI, SDValue Callee, SDValue Chain) const; protected: SDValue lowerLOAD(SDValue Op, SelectionDAG &DAG) const; SDValue lowerSTORE(SDValue Op, SelectionDAG &DAG) const; // Subtarget Info const MipsSubtarget &Subtarget; // Cache the ABI from the TargetMachine, we use it everywhere. const MipsABIInfo &ABI; private: // Create a TargetGlobalAddress node. SDValue getTargetNode(GlobalAddressSDNode *N, EVT Ty, SelectionDAG &DAG, unsigned Flag) const; // Create a TargetExternalSymbol node. SDValue getTargetNode(ExternalSymbolSDNode *N, EVT Ty, SelectionDAG &DAG, unsigned Flag) const; // Create a TargetBlockAddress node. SDValue getTargetNode(BlockAddressSDNode *N, EVT Ty, SelectionDAG &DAG, unsigned Flag) const; // Create a TargetJumpTable node. SDValue getTargetNode(JumpTableSDNode *N, EVT Ty, SelectionDAG &DAG, unsigned Flag) const; // Create a TargetConstantPool node. SDValue getTargetNode(ConstantPoolSDNode *N, EVT Ty, SelectionDAG &DAG, unsigned Flag) const; // Lower Operand helpers SDValue LowerCallResult(SDValue Chain, SDValue InFlag, CallingConv::ID CallConv, bool isVarArg, const SmallVectorImpl &Ins, const SDLoc &dl, SelectionDAG &DAG, SmallVectorImpl &InVals, TargetLowering::CallLoweringInfo &CLI) const; // Lower Operand specifics SDValue lowerBRCOND(SDValue Op, SelectionDAG &DAG) const; SDValue lowerConstantPool(SDValue Op, SelectionDAG &DAG) const; SDValue lowerGlobalAddress(SDValue Op, SelectionDAG &DAG) const; SDValue lowerBlockAddress(SDValue Op, SelectionDAG &DAG) const; SDValue lowerGlobalTLSAddress(SDValue Op, SelectionDAG &DAG) const; SDValue lowerJumpTable(SDValue Op, SelectionDAG &DAG) const; SDValue lowerSELECT(SDValue Op, SelectionDAG &DAG) const; SDValue lowerSETCC(SDValue Op, SelectionDAG &DAG) const; SDValue lowerVASTART(SDValue Op, SelectionDAG &DAG) const; SDValue lowerVAARG(SDValue Op, SelectionDAG &DAG) const; SDValue lowerFCOPYSIGN(SDValue Op, SelectionDAG &DAG) const; SDValue lowerFABS(SDValue Op, SelectionDAG &DAG) const; SDValue lowerFRAMEADDR(SDValue Op, SelectionDAG &DAG) const; SDValue lowerRETURNADDR(SDValue Op, SelectionDAG &DAG) const; SDValue lowerEH_RETURN(SDValue Op, SelectionDAG &DAG) const; SDValue lowerATOMIC_FENCE(SDValue Op, SelectionDAG& DAG) const; SDValue lowerShiftLeftParts(SDValue Op, SelectionDAG& DAG) const; SDValue lowerShiftRightParts(SDValue Op, SelectionDAG& DAG, bool IsSRA) const; SDValue lowerEH_DWARF_CFA(SDValue Op, SelectionDAG &DAG) const; SDValue lowerFP_TO_SINT(SDValue Op, SelectionDAG &DAG) const; /// isEligibleForTailCallOptimization - Check whether the call is eligible /// for tail call optimization. virtual bool isEligibleForTailCallOptimization(const CCState &CCInfo, unsigned NextStackOffset, const MipsFunctionInfo &FI) const = 0; /// copyByValArg - Copy argument registers which were used to pass a byval /// argument to the stack. Create a stack frame object for the byval /// argument. void copyByValRegs(SDValue Chain, const SDLoc &DL, std::vector &OutChains, SelectionDAG &DAG, const ISD::ArgFlagsTy &Flags, SmallVectorImpl &InVals, const Argument *FuncArg, unsigned FirstReg, unsigned LastReg, const CCValAssign &VA, MipsCCState &State) const; /// passByValArg - Pass a byval argument in registers or on stack. void passByValArg(SDValue Chain, const SDLoc &DL, std::deque> &RegsToPass, SmallVectorImpl &MemOpChains, SDValue StackPtr, MachineFrameInfo &MFI, SelectionDAG &DAG, SDValue Arg, unsigned FirstReg, unsigned LastReg, const ISD::ArgFlagsTy &Flags, bool isLittle, const CCValAssign &VA) const; /// writeVarArgRegs - Write variable function arguments passed in registers /// to the stack. Also create a stack frame object for the first variable /// argument. void writeVarArgRegs(std::vector &OutChains, SDValue Chain, const SDLoc &DL, SelectionDAG &DAG, CCState &State) const; SDValue LowerFormalArguments(SDValue Chain, CallingConv::ID CallConv, bool isVarArg, const SmallVectorImpl &Ins, const SDLoc &dl, SelectionDAG &DAG, SmallVectorImpl &InVals) const override; SDValue passArgOnStack(SDValue StackPtr, unsigned Offset, SDValue Chain, SDValue Arg, const SDLoc &DL, bool IsTailCall, SelectionDAG &DAG) const; SDValue LowerCall(TargetLowering::CallLoweringInfo &CLI, SmallVectorImpl &InVals) const override; bool CanLowerReturn(CallingConv::ID CallConv, MachineFunction &MF, bool isVarArg, const SmallVectorImpl &Outs, LLVMContext &Context) const override; SDValue LowerReturn(SDValue Chain, CallingConv::ID CallConv, bool isVarArg, const SmallVectorImpl &Outs, const SmallVectorImpl &OutVals, const SDLoc &dl, SelectionDAG &DAG) const override; SDValue LowerInterruptReturn(SmallVectorImpl &RetOps, const SDLoc &DL, SelectionDAG &DAG) const; bool shouldSignExtendTypeInLibCall(EVT Type, bool IsSigned) const override; // Inline asm support ConstraintType getConstraintType(StringRef Constraint) const override; /// Examine constraint string and operand type and determine a weight value. /// The operand object must already have been set up with the operand type. ConstraintWeight getSingleConstraintMatchWeight( AsmOperandInfo &info, const char *constraint) const override; /// This function parses registers that appear in inline-asm constraints. /// It returns pair (0, 0) on failure. std::pair parseRegForInlineAsmConstraint(StringRef C, MVT VT) const; std::pair getRegForInlineAsmConstraint(const TargetRegisterInfo *TRI, StringRef Constraint, MVT VT) const override; /// LowerAsmOperandForConstraint - Lower the specified operand into the Ops /// vector. If it is invalid, don't add anything to Ops. If hasMemory is /// true it means one of the asm constraint of the inline asm instruction /// being processed is 'm'. void LowerAsmOperandForConstraint(SDValue Op, std::string &Constraint, std::vector &Ops, SelectionDAG &DAG) const override; unsigned getInlineAsmMemConstraint(StringRef ConstraintCode) const override { if (ConstraintCode == "R") return InlineAsm::Constraint_R; else if (ConstraintCode == "ZC") return InlineAsm::Constraint_ZC; return TargetLowering::getInlineAsmMemConstraint(ConstraintCode); } bool isLegalAddressingMode(const DataLayout &DL, const AddrMode &AM, Type *Ty, unsigned AS, Instruction *I = nullptr) const override; bool isOffsetFoldingLegal(const GlobalAddressSDNode *GA) const override; EVT getOptimalMemOpType(uint64_t Size, unsigned DstAlign, unsigned SrcAlign, bool IsMemset, bool ZeroMemset, bool MemcpyStrSrc, MachineFunction &MF) const override; /// isFPImmLegal - Returns true if the target can instruction select the /// specified FP immediate natively. If false, the legalizer will /// materialize the FP immediate as a load from a constant pool. bool isFPImmLegal(const APFloat &Imm, EVT VT) const override; unsigned getJumpTableEncoding() const override; bool useSoftFloat() const override; bool shouldInsertFencesForAtomic(const Instruction *I) const override { return true; } /// Emit a sign-extension using sll/sra, seb, or seh appropriately. MachineBasicBlock *emitSignExtendToI32InReg(MachineInstr &MI, MachineBasicBlock *BB, unsigned Size, unsigned DstReg, unsigned SrcRec) const; MachineBasicBlock *emitAtomicBinary(MachineInstr &MI, MachineBasicBlock *BB) const; MachineBasicBlock *emitAtomicBinaryPartword(MachineInstr &MI, MachineBasicBlock *BB, unsigned Size) const; MachineBasicBlock *emitAtomicCmpSwap(MachineInstr &MI, MachineBasicBlock *BB) const; MachineBasicBlock *emitAtomicCmpSwapPartword(MachineInstr &MI, MachineBasicBlock *BB, unsigned Size) const; MachineBasicBlock *emitSEL_D(MachineInstr &MI, MachineBasicBlock *BB) const; MachineBasicBlock *emitPseudoSELECT(MachineInstr &MI, MachineBasicBlock *BB, bool isFPCmp, unsigned Opc) const; MachineBasicBlock *emitPseudoD_SELECT(MachineInstr &MI, MachineBasicBlock *BB) const; }; /// Create MipsTargetLowering objects. const MipsTargetLowering * createMips16TargetLowering(const MipsTargetMachine &TM, const MipsSubtarget &STI); const MipsTargetLowering * createMipsSETargetLowering(const MipsTargetMachine &TM, const MipsSubtarget &STI); namespace Mips { FastISel *createFastISel(FunctionLoweringInfo &funcInfo, const TargetLibraryInfo *libInfo); } // end namespace Mips } // end namespace llvm #endif // LLVM_LIB_TARGET_MIPS_MIPSISELLOWERING_H Index: vendor/llvm/dist-release_80/lib/Target/Mips/MipsInstrInfo.cpp =================================================================== --- vendor/llvm/dist-release_80/lib/Target/Mips/MipsInstrInfo.cpp (revision 343793) +++ vendor/llvm/dist-release_80/lib/Target/Mips/MipsInstrInfo.cpp (revision 343794) @@ -1,831 +1,842 @@ //===- MipsInstrInfo.cpp - Mips Instruction Information -------------------===// // // The LLVM Compiler Infrastructure // // This file is distributed under the University of Illinois Open Source // License. See LICENSE.TXT for details. // //===----------------------------------------------------------------------===// // // This file contains the Mips implementation of the TargetInstrInfo class. // //===----------------------------------------------------------------------===// #include "MipsInstrInfo.h" #include "MCTargetDesc/MipsBaseInfo.h" #include "MCTargetDesc/MipsMCTargetDesc.h" #include "MipsSubtarget.h" #include "llvm/ADT/SmallVector.h" #include "llvm/CodeGen/MachineBasicBlock.h" #include "llvm/CodeGen/MachineFrameInfo.h" #include "llvm/CodeGen/MachineFunction.h" #include "llvm/CodeGen/MachineInstr.h" #include "llvm/CodeGen/MachineInstrBuilder.h" #include "llvm/CodeGen/MachineOperand.h" #include "llvm/CodeGen/TargetOpcodes.h" #include "llvm/CodeGen/TargetSubtargetInfo.h" #include "llvm/IR/DebugLoc.h" #include "llvm/MC/MCInstrDesc.h" #include "llvm/Target/TargetMachine.h" #include using namespace llvm; #define GET_INSTRINFO_CTOR_DTOR #include "MipsGenInstrInfo.inc" // Pin the vtable to this file. void MipsInstrInfo::anchor() {} MipsInstrInfo::MipsInstrInfo(const MipsSubtarget &STI, unsigned UncondBr) : MipsGenInstrInfo(Mips::ADJCALLSTACKDOWN, Mips::ADJCALLSTACKUP), Subtarget(STI), UncondBrOpc(UncondBr) {} const MipsInstrInfo *MipsInstrInfo::create(MipsSubtarget &STI) { if (STI.inMips16Mode()) return createMips16InstrInfo(STI); return createMipsSEInstrInfo(STI); } bool MipsInstrInfo::isZeroImm(const MachineOperand &op) const { return op.isImm() && op.getImm() == 0; } /// insertNoop - If data hazard condition is found insert the target nop /// instruction. // FIXME: This appears to be dead code. void MipsInstrInfo:: insertNoop(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI) const { DebugLoc DL; BuildMI(MBB, MI, DL, get(Mips::NOP)); } MachineMemOperand * MipsInstrInfo::GetMemOperand(MachineBasicBlock &MBB, int FI, MachineMemOperand::Flags Flags) const { MachineFunction &MF = *MBB.getParent(); MachineFrameInfo &MFI = MF.getFrameInfo(); unsigned Align = MFI.getObjectAlignment(FI); return MF.getMachineMemOperand(MachinePointerInfo::getFixedStack(MF, FI), Flags, MFI.getObjectSize(FI), Align); } //===----------------------------------------------------------------------===// // Branch Analysis //===----------------------------------------------------------------------===// void MipsInstrInfo::AnalyzeCondBr(const MachineInstr *Inst, unsigned Opc, MachineBasicBlock *&BB, SmallVectorImpl &Cond) const { assert(getAnalyzableBrOpc(Opc) && "Not an analyzable branch"); int NumOp = Inst->getNumExplicitOperands(); // for both int and fp branches, the last explicit operand is the // MBB. BB = Inst->getOperand(NumOp-1).getMBB(); Cond.push_back(MachineOperand::CreateImm(Opc)); for (int i = 0; i < NumOp-1; i++) Cond.push_back(Inst->getOperand(i)); } bool MipsInstrInfo::analyzeBranch(MachineBasicBlock &MBB, MachineBasicBlock *&TBB, MachineBasicBlock *&FBB, SmallVectorImpl &Cond, bool AllowModify) const { SmallVector BranchInstrs; BranchType BT = analyzeBranch(MBB, TBB, FBB, Cond, AllowModify, BranchInstrs); return (BT == BT_None) || (BT == BT_Indirect); } void MipsInstrInfo::BuildCondBr(MachineBasicBlock &MBB, MachineBasicBlock *TBB, const DebugLoc &DL, ArrayRef Cond) const { unsigned Opc = Cond[0].getImm(); const MCInstrDesc &MCID = get(Opc); MachineInstrBuilder MIB = BuildMI(&MBB, DL, MCID); for (unsigned i = 1; i < Cond.size(); ++i) { assert((Cond[i].isImm() || Cond[i].isReg()) && "Cannot copy operand for conditional branch!"); MIB.add(Cond[i]); } MIB.addMBB(TBB); } unsigned MipsInstrInfo::insertBranch(MachineBasicBlock &MBB, MachineBasicBlock *TBB, MachineBasicBlock *FBB, ArrayRef Cond, const DebugLoc &DL, int *BytesAdded) const { // Shouldn't be a fall through. assert(TBB && "insertBranch must not be told to insert a fallthrough"); assert(!BytesAdded && "code size not handled"); // # of condition operands: // Unconditional branches: 0 // Floating point branches: 1 (opc) // Int BranchZero: 2 (opc, reg) // Int Branch: 3 (opc, reg0, reg1) assert((Cond.size() <= 3) && "# of Mips branch conditions must be <= 3!"); // Two-way Conditional branch. if (FBB) { BuildCondBr(MBB, TBB, DL, Cond); BuildMI(&MBB, DL, get(UncondBrOpc)).addMBB(FBB); return 2; } // One way branch. // Unconditional branch. if (Cond.empty()) BuildMI(&MBB, DL, get(UncondBrOpc)).addMBB(TBB); else // Conditional branch. BuildCondBr(MBB, TBB, DL, Cond); return 1; } unsigned MipsInstrInfo::removeBranch(MachineBasicBlock &MBB, int *BytesRemoved) const { assert(!BytesRemoved && "code size not handled"); MachineBasicBlock::reverse_iterator I = MBB.rbegin(), REnd = MBB.rend(); unsigned removed = 0; // Up to 2 branches are removed. // Note that indirect branches are not removed. while (I != REnd && removed < 2) { // Skip past debug instructions. if (I->isDebugInstr()) { ++I; continue; } if (!getAnalyzableBrOpc(I->getOpcode())) break; // Remove the branch. I->eraseFromParent(); I = MBB.rbegin(); ++removed; } return removed; } /// reverseBranchCondition - Return the inverse opcode of the /// specified Branch instruction. bool MipsInstrInfo::reverseBranchCondition( SmallVectorImpl &Cond) const { assert( (Cond.size() && Cond.size() <= 3) && "Invalid Mips branch condition!"); Cond[0].setImm(getOppositeBranchOpc(Cond[0].getImm())); return false; } MipsInstrInfo::BranchType MipsInstrInfo::analyzeBranch( MachineBasicBlock &MBB, MachineBasicBlock *&TBB, MachineBasicBlock *&FBB, SmallVectorImpl &Cond, bool AllowModify, SmallVectorImpl &BranchInstrs) const { MachineBasicBlock::reverse_iterator I = MBB.rbegin(), REnd = MBB.rend(); // Skip all the debug instructions. while (I != REnd && I->isDebugInstr()) ++I; if (I == REnd || !isUnpredicatedTerminator(*I)) { // This block ends with no branches (it just falls through to its succ). // Leave TBB/FBB null. TBB = FBB = nullptr; return BT_NoBranch; } MachineInstr *LastInst = &*I; unsigned LastOpc = LastInst->getOpcode(); BranchInstrs.push_back(LastInst); // Not an analyzable branch (e.g., indirect jump). if (!getAnalyzableBrOpc(LastOpc)) return LastInst->isIndirectBranch() ? BT_Indirect : BT_None; // Get the second to last instruction in the block. unsigned SecondLastOpc = 0; MachineInstr *SecondLastInst = nullptr; // Skip past any debug instruction to see if the second last actual // is a branch. ++I; while (I != REnd && I->isDebugInstr()) ++I; if (I != REnd) { SecondLastInst = &*I; SecondLastOpc = getAnalyzableBrOpc(SecondLastInst->getOpcode()); // Not an analyzable branch (must be an indirect jump). if (isUnpredicatedTerminator(*SecondLastInst) && !SecondLastOpc) return BT_None; } // If there is only one terminator instruction, process it. if (!SecondLastOpc) { // Unconditional branch. if (LastInst->isUnconditionalBranch()) { TBB = LastInst->getOperand(0).getMBB(); return BT_Uncond; } // Conditional branch AnalyzeCondBr(LastInst, LastOpc, TBB, Cond); return BT_Cond; } // If we reached here, there are two branches. // If there are three terminators, we don't know what sort of block this is. if (++I != REnd && isUnpredicatedTerminator(*I)) return BT_None; BranchInstrs.insert(BranchInstrs.begin(), SecondLastInst); // If second to last instruction is an unconditional branch, // analyze it and remove the last instruction. if (SecondLastInst->isUnconditionalBranch()) { // Return if the last instruction cannot be removed. if (!AllowModify) return BT_None; TBB = SecondLastInst->getOperand(0).getMBB(); LastInst->eraseFromParent(); BranchInstrs.pop_back(); return BT_Uncond; } // Conditional branch followed by an unconditional branch. // The last one must be unconditional. if (!LastInst->isUnconditionalBranch()) return BT_None; AnalyzeCondBr(SecondLastInst, SecondLastOpc, TBB, Cond); FBB = LastInst->getOperand(0).getMBB(); return BT_CondUncond; } bool MipsInstrInfo::isBranchOffsetInRange(unsigned BranchOpc, int64_t BrOffset) const { switch (BranchOpc) { case Mips::B: case Mips::BAL: case Mips::BAL_BR: case Mips::BAL_BR_MM: case Mips::BC1F: case Mips::BC1FL: case Mips::BC1T: case Mips::BC1TL: case Mips::BEQ: case Mips::BEQ64: case Mips::BEQL: case Mips::BGEZ: case Mips::BGEZ64: case Mips::BGEZL: case Mips::BGEZAL: case Mips::BGEZALL: case Mips::BGTZ: case Mips::BGTZ64: case Mips::BGTZL: case Mips::BLEZ: case Mips::BLEZ64: case Mips::BLEZL: case Mips::BLTZ: case Mips::BLTZ64: case Mips::BLTZL: case Mips::BLTZAL: case Mips::BLTZALL: case Mips::BNE: case Mips::BNE64: case Mips::BNEL: return isInt<18>(BrOffset); // microMIPSr3 branches case Mips::B_MM: case Mips::BC1F_MM: case Mips::BC1T_MM: case Mips::BEQ_MM: case Mips::BGEZ_MM: case Mips::BGEZAL_MM: case Mips::BGTZ_MM: case Mips::BLEZ_MM: case Mips::BLTZ_MM: case Mips::BLTZAL_MM: case Mips::BNE_MM: case Mips::BEQZC_MM: case Mips::BNEZC_MM: return isInt<17>(BrOffset); // microMIPSR3 short branches. case Mips::B16_MM: return isInt<11>(BrOffset); case Mips::BEQZ16_MM: case Mips::BNEZ16_MM: return isInt<8>(BrOffset); // MIPSR6 branches. case Mips::BALC: case Mips::BC: return isInt<28>(BrOffset); case Mips::BC1EQZ: case Mips::BC1NEZ: case Mips::BC2EQZ: case Mips::BC2NEZ: case Mips::BEQC: case Mips::BEQC64: case Mips::BNEC: case Mips::BNEC64: case Mips::BGEC: case Mips::BGEC64: case Mips::BGEUC: case Mips::BGEUC64: case Mips::BGEZC: case Mips::BGEZC64: case Mips::BGTZC: case Mips::BGTZC64: case Mips::BLEZC: case Mips::BLEZC64: case Mips::BLTC: case Mips::BLTC64: case Mips::BLTUC: case Mips::BLTUC64: case Mips::BLTZC: case Mips::BLTZC64: case Mips::BNVC: case Mips::BOVC: case Mips::BGEZALC: case Mips::BEQZALC: case Mips::BGTZALC: case Mips::BLEZALC: case Mips::BLTZALC: case Mips::BNEZALC: return isInt<18>(BrOffset); case Mips::BEQZC: case Mips::BEQZC64: case Mips::BNEZC: case Mips::BNEZC64: return isInt<23>(BrOffset); // microMIPSR6 branches case Mips::BC16_MMR6: return isInt<11>(BrOffset); case Mips::BEQZC16_MMR6: case Mips::BNEZC16_MMR6: return isInt<8>(BrOffset); case Mips::BALC_MMR6: case Mips::BC_MMR6: return isInt<27>(BrOffset); case Mips::BC1EQZC_MMR6: case Mips::BC1NEZC_MMR6: case Mips::BC2EQZC_MMR6: case Mips::BC2NEZC_MMR6: case Mips::BGEZALC_MMR6: case Mips::BEQZALC_MMR6: case Mips::BGTZALC_MMR6: case Mips::BLEZALC_MMR6: case Mips::BLTZALC_MMR6: case Mips::BNEZALC_MMR6: case Mips::BNVC_MMR6: case Mips::BOVC_MMR6: return isInt<17>(BrOffset); case Mips::BEQC_MMR6: case Mips::BNEC_MMR6: case Mips::BGEC_MMR6: case Mips::BGEUC_MMR6: case Mips::BGEZC_MMR6: case Mips::BGTZC_MMR6: case Mips::BLEZC_MMR6: case Mips::BLTC_MMR6: case Mips::BLTUC_MMR6: case Mips::BLTZC_MMR6: return isInt<18>(BrOffset); case Mips::BEQZC_MMR6: case Mips::BNEZC_MMR6: return isInt<23>(BrOffset); // DSP branches. case Mips::BPOSGE32: return isInt<18>(BrOffset); case Mips::BPOSGE32_MM: case Mips::BPOSGE32C_MMR3: return isInt<17>(BrOffset); // cnMIPS branches. case Mips::BBIT0: case Mips::BBIT032: case Mips::BBIT1: case Mips::BBIT132: return isInt<18>(BrOffset); // MSA branches. case Mips::BZ_B: case Mips::BZ_H: case Mips::BZ_W: case Mips::BZ_D: case Mips::BZ_V: case Mips::BNZ_B: case Mips::BNZ_H: case Mips::BNZ_W: case Mips::BNZ_D: case Mips::BNZ_V: return isInt<18>(BrOffset); } llvm_unreachable("Unknown branch instruction!"); } /// Return the corresponding compact (no delay slot) form of a branch. unsigned MipsInstrInfo::getEquivalentCompactForm( const MachineBasicBlock::iterator I) const { unsigned Opcode = I->getOpcode(); bool canUseShortMicroMipsCTI = false; if (Subtarget.inMicroMipsMode()) { switch (Opcode) { case Mips::BNE: case Mips::BNE_MM: case Mips::BEQ: case Mips::BEQ_MM: // microMIPS has NE,EQ branches that do not have delay slots provided one // of the operands is zero. if (I->getOperand(1).getReg() == Subtarget.getABI().GetZeroReg()) canUseShortMicroMipsCTI = true; break; // For microMIPS the PseudoReturn and PseudoIndirectBranch are always // expanded to JR_MM, so they can be replaced with JRC16_MM. case Mips::JR: case Mips::PseudoReturn: case Mips::PseudoIndirectBranch: canUseShortMicroMipsCTI = true; break; } } // MIPSR6 forbids both operands being the zero register. if (Subtarget.hasMips32r6() && (I->getNumOperands() > 1) && (I->getOperand(0).isReg() && (I->getOperand(0).getReg() == Mips::ZERO || I->getOperand(0).getReg() == Mips::ZERO_64)) && (I->getOperand(1).isReg() && (I->getOperand(1).getReg() == Mips::ZERO || I->getOperand(1).getReg() == Mips::ZERO_64))) return 0; if (Subtarget.hasMips32r6() || canUseShortMicroMipsCTI) { switch (Opcode) { case Mips::B: return Mips::BC; case Mips::BAL: return Mips::BALC; case Mips::BEQ: case Mips::BEQ_MM: if (canUseShortMicroMipsCTI) return Mips::BEQZC_MM; else if (I->getOperand(0).getReg() == I->getOperand(1).getReg()) return 0; return Mips::BEQC; case Mips::BNE: case Mips::BNE_MM: if (canUseShortMicroMipsCTI) return Mips::BNEZC_MM; else if (I->getOperand(0).getReg() == I->getOperand(1).getReg()) return 0; return Mips::BNEC; case Mips::BGE: if (I->getOperand(0).getReg() == I->getOperand(1).getReg()) return 0; return Mips::BGEC; case Mips::BGEU: if (I->getOperand(0).getReg() == I->getOperand(1).getReg()) return 0; return Mips::BGEUC; case Mips::BGEZ: return Mips::BGEZC; case Mips::BGTZ: return Mips::BGTZC; case Mips::BLEZ: return Mips::BLEZC; case Mips::BLT: if (I->getOperand(0).getReg() == I->getOperand(1).getReg()) return 0; return Mips::BLTC; case Mips::BLTU: if (I->getOperand(0).getReg() == I->getOperand(1).getReg()) return 0; return Mips::BLTUC; case Mips::BLTZ: return Mips::BLTZC; case Mips::BEQ64: if (I->getOperand(0).getReg() == I->getOperand(1).getReg()) return 0; return Mips::BEQC64; case Mips::BNE64: if (I->getOperand(0).getReg() == I->getOperand(1).getReg()) return 0; return Mips::BNEC64; case Mips::BGTZ64: return Mips::BGTZC64; case Mips::BGEZ64: return Mips::BGEZC64; case Mips::BLTZ64: return Mips::BLTZC64; case Mips::BLEZ64: return Mips::BLEZC64; // For MIPSR6, the instruction 'jic' can be used for these cases. Some // tools will accept 'jrc reg' as an alias for 'jic 0, $reg'. case Mips::JR: case Mips::PseudoIndirectBranchR6: case Mips::PseudoReturn: case Mips::TAILCALLR6REG: if (canUseShortMicroMipsCTI) return Mips::JRC16_MM; return Mips::JIC; case Mips::JALRPseudo: return Mips::JIALC; case Mips::JR64: case Mips::PseudoIndirectBranch64R6: case Mips::PseudoReturn64: case Mips::TAILCALL64R6REG: return Mips::JIC64; case Mips::JALR64Pseudo: return Mips::JIALC64; default: return 0; } } return 0; } /// Predicate for distingushing between control transfer instructions and all /// other instructions for handling forbidden slots. Consider inline assembly /// as unsafe as well. bool MipsInstrInfo::SafeInForbiddenSlot(const MachineInstr &MI) const { if (MI.isInlineAsm()) return false; return (MI.getDesc().TSFlags & MipsII::IsCTI) == 0; } /// Predicate for distingushing instructions that have forbidden slots. bool MipsInstrInfo::HasForbiddenSlot(const MachineInstr &MI) const { return (MI.getDesc().TSFlags & MipsII::HasForbiddenSlot) != 0; } /// Return the number of bytes of code the specified instruction may be. unsigned MipsInstrInfo::getInstSizeInBytes(const MachineInstr &MI) const { switch (MI.getOpcode()) { default: return MI.getDesc().getSize(); case TargetOpcode::INLINEASM: { // Inline Asm: Variable size. const MachineFunction *MF = MI.getParent()->getParent(); const char *AsmStr = MI.getOperand(0).getSymbolName(); return getInlineAsmLength(AsmStr, *MF->getTarget().getMCAsmInfo()); } case Mips::CONSTPOOL_ENTRY: // If this machine instr is a constant pool entry, its size is recorded as // operand #2. return MI.getOperand(2).getImm(); } } MachineInstrBuilder MipsInstrInfo::genInstrWithNewOpc(unsigned NewOpc, MachineBasicBlock::iterator I) const { MachineInstrBuilder MIB; // Certain branches have two forms: e.g beq $1, $zero, dest vs beqz $1, dest // Pick the zero form of the branch for readable assembly and for greater // branch distance in non-microMIPS mode. // Additional MIPSR6 does not permit the use of register $zero for compact // branches. // FIXME: Certain atomic sequences on mips64 generate 32bit references to // Mips::ZERO, which is incorrect. This test should be updated to use // Subtarget.getABI().GetZeroReg() when those atomic sequences and others // are fixed. int ZeroOperandPosition = -1; bool BranchWithZeroOperand = false; if (I->isBranch() && !I->isPseudo()) { auto TRI = I->getParent()->getParent()->getSubtarget().getRegisterInfo(); ZeroOperandPosition = I->findRegisterUseOperandIdx(Mips::ZERO, false, TRI); BranchWithZeroOperand = ZeroOperandPosition != -1; } if (BranchWithZeroOperand) { switch (NewOpc) { case Mips::BEQC: NewOpc = Mips::BEQZC; break; case Mips::BNEC: NewOpc = Mips::BNEZC; break; case Mips::BGEC: NewOpc = Mips::BGEZC; break; case Mips::BLTC: NewOpc = Mips::BLTZC; break; case Mips::BEQC64: NewOpc = Mips::BEQZC64; break; case Mips::BNEC64: NewOpc = Mips::BNEZC64; break; } } MIB = BuildMI(*I->getParent(), I, I->getDebugLoc(), get(NewOpc)); // For MIPSR6 JI*C requires an immediate 0 as an operand, JIALC(64) an // immediate 0 as an operand and requires the removal of it's implicit-def %ra // implicit operand as copying the implicit operations of the instructio we're // looking at will give us the correct flags. if (NewOpc == Mips::JIC || NewOpc == Mips::JIALC || NewOpc == Mips::JIC64 || NewOpc == Mips::JIALC64) { if (NewOpc == Mips::JIALC || NewOpc == Mips::JIALC64) MIB->RemoveOperand(0); for (unsigned J = 0, E = I->getDesc().getNumOperands(); J < E; ++J) { MIB.add(I->getOperand(J)); } MIB.addImm(0); + // If I has an MCSymbol operand (used by asm printer, to emit R_MIPS_JALR), + // add it to the new instruction. + for (unsigned J = I->getDesc().getNumOperands(), E = I->getNumOperands(); + J < E; ++J) { + const MachineOperand &MO = I->getOperand(J); + if (MO.isMCSymbol() && (MO.getTargetFlags() & MipsII::MO_JALR)) + MIB.addSym(MO.getMCSymbol(), MipsII::MO_JALR); + } + + } else { for (unsigned J = 0, E = I->getDesc().getNumOperands(); J < E; ++J) { if (BranchWithZeroOperand && (unsigned)ZeroOperandPosition == J) continue; MIB.add(I->getOperand(J)); } } MIB.copyImplicitOps(*I); MIB.cloneMemRefs(*I); return MIB; } bool MipsInstrInfo::findCommutedOpIndices(MachineInstr &MI, unsigned &SrcOpIdx1, unsigned &SrcOpIdx2) const { assert(!MI.isBundle() && "TargetInstrInfo::findCommutedOpIndices() can't handle bundles"); const MCInstrDesc &MCID = MI.getDesc(); if (!MCID.isCommutable()) return false; switch (MI.getOpcode()) { case Mips::DPADD_U_H: case Mips::DPADD_U_W: case Mips::DPADD_U_D: case Mips::DPADD_S_H: case Mips::DPADD_S_W: case Mips::DPADD_S_D: // The first operand is both input and output, so it should not commute if (!fixCommutedOpIndices(SrcOpIdx1, SrcOpIdx2, 2, 3)) return false; if (!MI.getOperand(SrcOpIdx1).isReg() || !MI.getOperand(SrcOpIdx2).isReg()) return false; return true; } return TargetInstrInfo::findCommutedOpIndices(MI, SrcOpIdx1, SrcOpIdx2); } // ins, ext, dext*, dins have the following constraints: // X <= pos < Y // X < size <= Y // X < pos+size <= Y // // dinsm and dinsu have the following constraints: // X <= pos < Y // X <= size <= Y // X < pos+size <= Y // // The callee of verifyInsExtInstruction however gives the bounds of // dins[um] like the other (d)ins (d)ext(um) instructions, so that this // function doesn't have to vary it's behaviour based on the instruction // being checked. static bool verifyInsExtInstruction(const MachineInstr &MI, StringRef &ErrInfo, const int64_t PosLow, const int64_t PosHigh, const int64_t SizeLow, const int64_t SizeHigh, const int64_t BothLow, const int64_t BothHigh) { MachineOperand MOPos = MI.getOperand(2); if (!MOPos.isImm()) { ErrInfo = "Position is not an immediate!"; return false; } int64_t Pos = MOPos.getImm(); if (!((PosLow <= Pos) && (Pos < PosHigh))) { ErrInfo = "Position operand is out of range!"; return false; } MachineOperand MOSize = MI.getOperand(3); if (!MOSize.isImm()) { ErrInfo = "Size operand is not an immediate!"; return false; } int64_t Size = MOSize.getImm(); if (!((SizeLow < Size) && (Size <= SizeHigh))) { ErrInfo = "Size operand is out of range!"; return false; } if (!((BothLow < (Pos + Size)) && ((Pos + Size) <= BothHigh))) { ErrInfo = "Position + Size is out of range!"; return false; } return true; } // Perform target specific instruction verification. bool MipsInstrInfo::verifyInstruction(const MachineInstr &MI, StringRef &ErrInfo) const { // Verify that ins and ext instructions are well formed. switch (MI.getOpcode()) { case Mips::EXT: case Mips::EXT_MM: case Mips::INS: case Mips::INS_MM: case Mips::DINS: return verifyInsExtInstruction(MI, ErrInfo, 0, 32, 0, 32, 0, 32); case Mips::DINSM: // The ISA spec has a subtle difference between dinsm and dextm // in that it says: // 2 <= size <= 64 for 'dinsm' but 'dextm' has 32 < size <= 64. // To make the bounds checks similar, the range 1 < size <= 64 is checked // for 'dinsm'. return verifyInsExtInstruction(MI, ErrInfo, 0, 32, 1, 64, 32, 64); case Mips::DINSU: // The ISA spec has a subtle difference between dinsu and dextu in that // the size range of dinsu is specified as 1 <= size <= 32 whereas size // for dextu is 0 < size <= 32. The range checked for dinsu here is // 0 < size <= 32, which is equivalent and similar to dextu. return verifyInsExtInstruction(MI, ErrInfo, 32, 64, 0, 32, 32, 64); case Mips::DEXT: return verifyInsExtInstruction(MI, ErrInfo, 0, 32, 0, 32, 0, 63); case Mips::DEXTM: return verifyInsExtInstruction(MI, ErrInfo, 0, 32, 32, 64, 32, 64); case Mips::DEXTU: return verifyInsExtInstruction(MI, ErrInfo, 32, 64, 0, 32, 32, 64); case Mips::TAILCALLREG: case Mips::PseudoIndirectBranch: case Mips::JR: case Mips::JR64: case Mips::JALR: case Mips::JALR64: case Mips::JALRPseudo: if (!Subtarget.useIndirectJumpsHazard()) return true; ErrInfo = "invalid instruction when using jump guards!"; return false; default: return true; } return true; } std::pair MipsInstrInfo::decomposeMachineOperandsTargetFlags(unsigned TF) const { return std::make_pair(TF, 0u); } ArrayRef> MipsInstrInfo::getSerializableDirectMachineOperandTargetFlags() const { using namespace MipsII; static const std::pair Flags[] = { {MO_GOT, "mips-got"}, {MO_GOT_CALL, "mips-got-call"}, {MO_GPREL, "mips-gprel"}, {MO_ABS_HI, "mips-abs-hi"}, {MO_ABS_LO, "mips-abs-lo"}, {MO_TLSGD, "mips-tlsgd"}, {MO_TLSLDM, "mips-tlsldm"}, {MO_DTPREL_HI, "mips-dtprel-hi"}, {MO_DTPREL_LO, "mips-dtprel-lo"}, {MO_GOTTPREL, "mips-gottprel"}, {MO_TPREL_HI, "mips-tprel-hi"}, {MO_TPREL_LO, "mips-tprel-lo"}, {MO_GPOFF_HI, "mips-gpoff-hi"}, {MO_GPOFF_LO, "mips-gpoff-lo"}, {MO_GOT_DISP, "mips-got-disp"}, {MO_GOT_PAGE, "mips-got-page"}, {MO_GOT_OFST, "mips-got-ofst"}, {MO_HIGHER, "mips-higher"}, {MO_HIGHEST, "mips-highest"}, {MO_GOT_HI16, "mips-got-hi16"}, {MO_GOT_LO16, "mips-got-lo16"}, {MO_CALL_HI16, "mips-call-hi16"}, - {MO_CALL_LO16, "mips-call-lo16"} + {MO_CALL_LO16, "mips-call-lo16"}, + {MO_JALR, "mips-jalr"} }; return makeArrayRef(Flags); } Index: vendor/llvm/dist-release_80/lib/Target/Mips/MipsInstrInfo.td =================================================================== --- vendor/llvm/dist-release_80/lib/Target/Mips/MipsInstrInfo.td (revision 343793) +++ vendor/llvm/dist-release_80/lib/Target/Mips/MipsInstrInfo.td (revision 343794) @@ -1,3307 +1,3313 @@ //===- MipsInstrInfo.td - Target Description for Mips Target -*- tablegen -*-=// // // The LLVM Compiler Infrastructure // // This file is distributed under the University of Illinois Open Source // License. See LICENSE.TXT for details. // //===----------------------------------------------------------------------===// // // This file contains the Mips implementation of the TargetInstrInfo class. // //===----------------------------------------------------------------------===// //===----------------------------------------------------------------------===// // Mips profiles and nodes //===----------------------------------------------------------------------===// def SDT_MipsJmpLink : SDTypeProfile<0, 1, [SDTCisVT<0, iPTR>]>; def SDT_MipsCMov : SDTypeProfile<1, 4, [SDTCisSameAs<0, 1>, SDTCisSameAs<1, 2>, SDTCisSameAs<3, 4>, SDTCisInt<4>]>; def SDT_MipsCallSeqStart : SDCallSeqStart<[SDTCisVT<0, i32>, SDTCisVT<1, i32>]>; def SDT_MipsCallSeqEnd : SDCallSeqEnd<[SDTCisVT<0, i32>, SDTCisVT<1, i32>]>; def SDT_MFLOHI : SDTypeProfile<1, 1, [SDTCisInt<0>, SDTCisVT<1, untyped>]>; def SDT_MTLOHI : SDTypeProfile<1, 2, [SDTCisVT<0, untyped>, SDTCisInt<1>, SDTCisSameAs<1, 2>]>; def SDT_MipsMultDiv : SDTypeProfile<1, 2, [SDTCisVT<0, untyped>, SDTCisInt<1>, SDTCisSameAs<1, 2>]>; def SDT_MipsMAddMSub : SDTypeProfile<1, 3, [SDTCisVT<0, untyped>, SDTCisSameAs<0, 3>, SDTCisVT<1, i32>, SDTCisSameAs<1, 2>]>; def SDT_MipsDivRem16 : SDTypeProfile<0, 2, [SDTCisInt<0>, SDTCisSameAs<0, 1>]>; def SDT_MipsThreadPointer : SDTypeProfile<1, 0, [SDTCisPtrTy<0>]>; def SDT_Sync : SDTypeProfile<0, 1, [SDTCisVT<0, i32>]>; def SDT_Ext : SDTypeProfile<1, 3, [SDTCisInt<0>, SDTCisSameAs<0, 1>, SDTCisVT<2, i32>, SDTCisSameAs<2, 3>]>; def SDT_Ins : SDTypeProfile<1, 4, [SDTCisInt<0>, SDTCisSameAs<0, 1>, SDTCisVT<2, i32>, SDTCisSameAs<2, 3>, SDTCisSameAs<0, 4>]>; def SDTMipsLoadLR : SDTypeProfile<1, 2, [SDTCisInt<0>, SDTCisPtrTy<1>, SDTCisSameAs<0, 2>]>; // Call def MipsJmpLink : SDNode<"MipsISD::JmpLink",SDT_MipsJmpLink, [SDNPHasChain, SDNPOutGlue, SDNPOptInGlue, SDNPVariadic]>; // Tail call def MipsTailCall : SDNode<"MipsISD::TailCall", SDT_MipsJmpLink, [SDNPHasChain, SDNPOptInGlue, SDNPVariadic]>; // Hi and Lo nodes are used to handle global addresses. Used on // MipsISelLowering to lower stuff like GlobalAddress, ExternalSymbol // static model. (nothing to do with Mips Registers Hi and Lo) // Hi is the odd node out, on MIPS64 it can expand to either daddiu when // using static relocations with 64 bit symbols, or lui when using 32 bit // symbols. def MipsHigher : SDNode<"MipsISD::Higher", SDTIntUnaryOp>; def MipsHighest : SDNode<"MipsISD::Highest", SDTIntUnaryOp>; def MipsHi : SDNode<"MipsISD::Hi", SDTIntUnaryOp>; def MipsLo : SDNode<"MipsISD::Lo", SDTIntUnaryOp>; def MipsGPRel : SDNode<"MipsISD::GPRel", SDTIntUnaryOp>; // Hi node for accessing the GOT. def MipsGotHi : SDNode<"MipsISD::GotHi", SDTIntUnaryOp>; // Hi node for handling TLS offsets def MipsTlsHi : SDNode<"MipsISD::TlsHi", SDTIntUnaryOp>; // Thread pointer def MipsThreadPointer: SDNode<"MipsISD::ThreadPointer", SDT_MipsThreadPointer>; // Return def MipsRet : SDNode<"MipsISD::Ret", SDTNone, [SDNPHasChain, SDNPOptInGlue, SDNPVariadic]>; def MipsERet : SDNode<"MipsISD::ERet", SDTNone, [SDNPHasChain, SDNPOptInGlue, SDNPSideEffect]>; // These are target-independent nodes, but have target-specific formats. def callseq_start : SDNode<"ISD::CALLSEQ_START", SDT_MipsCallSeqStart, [SDNPHasChain, SDNPSideEffect, SDNPOutGlue]>; def callseq_end : SDNode<"ISD::CALLSEQ_END", SDT_MipsCallSeqEnd, [SDNPHasChain, SDNPSideEffect, SDNPOptInGlue, SDNPOutGlue]>; // Nodes used to extract LO/HI registers. def MipsMFHI : SDNode<"MipsISD::MFHI", SDT_MFLOHI>; def MipsMFLO : SDNode<"MipsISD::MFLO", SDT_MFLOHI>; // Node used to insert 32-bit integers to LOHI register pair. def MipsMTLOHI : SDNode<"MipsISD::MTLOHI", SDT_MTLOHI>; // Mult nodes. def MipsMult : SDNode<"MipsISD::Mult", SDT_MipsMultDiv>; def MipsMultu : SDNode<"MipsISD::Multu", SDT_MipsMultDiv>; // MAdd*/MSub* nodes def MipsMAdd : SDNode<"MipsISD::MAdd", SDT_MipsMAddMSub>; def MipsMAddu : SDNode<"MipsISD::MAddu", SDT_MipsMAddMSub>; def MipsMSub : SDNode<"MipsISD::MSub", SDT_MipsMAddMSub>; def MipsMSubu : SDNode<"MipsISD::MSubu", SDT_MipsMAddMSub>; // DivRem(u) nodes def MipsDivRem : SDNode<"MipsISD::DivRem", SDT_MipsMultDiv>; def MipsDivRemU : SDNode<"MipsISD::DivRemU", SDT_MipsMultDiv>; def MipsDivRem16 : SDNode<"MipsISD::DivRem16", SDT_MipsDivRem16, [SDNPOutGlue]>; def MipsDivRemU16 : SDNode<"MipsISD::DivRemU16", SDT_MipsDivRem16, [SDNPOutGlue]>; // Target constant nodes that are not part of any isel patterns and remain // unchanged can cause instructions with illegal operands to be emitted. // Wrapper node patterns give the instruction selector a chance to replace // target constant nodes that would otherwise remain unchanged with ADDiu // nodes. Without these wrapper node patterns, the following conditional move // instruction is emitted when function cmov2 in test/CodeGen/Mips/cmov.ll is // compiled: // movn %got(d)($gp), %got(c)($gp), $4 // This instruction is illegal since movn can take only register operands. def MipsWrapper : SDNode<"MipsISD::Wrapper", SDTIntBinOp>; def MipsSync : SDNode<"MipsISD::Sync", SDT_Sync, [SDNPHasChain,SDNPSideEffect]>; def MipsExt : SDNode<"MipsISD::Ext", SDT_Ext>; def MipsIns : SDNode<"MipsISD::Ins", SDT_Ins>; def MipsCIns : SDNode<"MipsISD::CIns", SDT_Ext>; def MipsLWL : SDNode<"MipsISD::LWL", SDTMipsLoadLR, [SDNPHasChain, SDNPMayLoad, SDNPMemOperand]>; def MipsLWR : SDNode<"MipsISD::LWR", SDTMipsLoadLR, [SDNPHasChain, SDNPMayLoad, SDNPMemOperand]>; def MipsSWL : SDNode<"MipsISD::SWL", SDTStore, [SDNPHasChain, SDNPMayStore, SDNPMemOperand]>; def MipsSWR : SDNode<"MipsISD::SWR", SDTStore, [SDNPHasChain, SDNPMayStore, SDNPMemOperand]>; def MipsLDL : SDNode<"MipsISD::LDL", SDTMipsLoadLR, [SDNPHasChain, SDNPMayLoad, SDNPMemOperand]>; def MipsLDR : SDNode<"MipsISD::LDR", SDTMipsLoadLR, [SDNPHasChain, SDNPMayLoad, SDNPMemOperand]>; def MipsSDL : SDNode<"MipsISD::SDL", SDTStore, [SDNPHasChain, SDNPMayStore, SDNPMemOperand]>; def MipsSDR : SDNode<"MipsISD::SDR", SDTStore, [SDNPHasChain, SDNPMayStore, SDNPMemOperand]>; //===----------------------------------------------------------------------===// // Mips Instruction Predicate Definitions. //===----------------------------------------------------------------------===// def HasMips2 : Predicate<"Subtarget->hasMips2()">, AssemblerPredicate<"FeatureMips2">; def HasMips3_32 : Predicate<"Subtarget->hasMips3_32()">, AssemblerPredicate<"FeatureMips3_32">; def HasMips3_32r2 : Predicate<"Subtarget->hasMips3_32r2()">, AssemblerPredicate<"FeatureMips3_32r2">; def HasMips3 : Predicate<"Subtarget->hasMips3()">, AssemblerPredicate<"FeatureMips3">; def NotMips3 : Predicate<"!Subtarget->hasMips3()">, AssemblerPredicate<"!FeatureMips3">; def HasMips4_32 : Predicate<"Subtarget->hasMips4_32()">, AssemblerPredicate<"FeatureMips4_32">; def NotMips4_32 : Predicate<"!Subtarget->hasMips4_32()">, AssemblerPredicate<"!FeatureMips4_32">; def HasMips4_32r2 : Predicate<"Subtarget->hasMips4_32r2()">, AssemblerPredicate<"FeatureMips4_32r2">; def HasMips5_32r2 : Predicate<"Subtarget->hasMips5_32r2()">, AssemblerPredicate<"FeatureMips5_32r2">; def HasMips32 : Predicate<"Subtarget->hasMips32()">, AssemblerPredicate<"FeatureMips32">; def HasMips32r2 : Predicate<"Subtarget->hasMips32r2()">, AssemblerPredicate<"FeatureMips32r2">; def HasMips32r5 : Predicate<"Subtarget->hasMips32r5()">, AssemblerPredicate<"FeatureMips32r5">; def HasMips32r6 : Predicate<"Subtarget->hasMips32r6()">, AssemblerPredicate<"FeatureMips32r6">; def NotMips32r6 : Predicate<"!Subtarget->hasMips32r6()">, AssemblerPredicate<"!FeatureMips32r6">; def IsGP64bit : Predicate<"Subtarget->isGP64bit()">, AssemblerPredicate<"FeatureGP64Bit">; def IsGP32bit : Predicate<"!Subtarget->isGP64bit()">, AssemblerPredicate<"!FeatureGP64Bit">; def IsPTR64bit : Predicate<"Subtarget->isABI_N64()">, AssemblerPredicate<"FeaturePTR64Bit">; def IsPTR32bit : Predicate<"!Subtarget->isABI_N64()">, AssemblerPredicate<"!FeaturePTR64Bit">; def HasMips64 : Predicate<"Subtarget->hasMips64()">, AssemblerPredicate<"FeatureMips64">; def NotMips64 : Predicate<"!Subtarget->hasMips64()">, AssemblerPredicate<"!FeatureMips64">; def HasMips64r2 : Predicate<"Subtarget->hasMips64r2()">, AssemblerPredicate<"FeatureMips64r2">; def HasMips64r5 : Predicate<"Subtarget->hasMips64r5()">, AssemblerPredicate<"FeatureMips64r5">; def HasMips64r6 : Predicate<"Subtarget->hasMips64r6()">, AssemblerPredicate<"FeatureMips64r6">; def NotMips64r6 : Predicate<"!Subtarget->hasMips64r6()">, AssemblerPredicate<"!FeatureMips64r6">; def InMips16Mode : Predicate<"Subtarget->inMips16Mode()">, AssemblerPredicate<"FeatureMips16">; def NotInMips16Mode : Predicate<"!Subtarget->inMips16Mode()">, AssemblerPredicate<"!FeatureMips16">; def HasCnMips : Predicate<"Subtarget->hasCnMips()">, AssemblerPredicate<"FeatureCnMips">; def NotCnMips : Predicate<"!Subtarget->hasCnMips()">, AssemblerPredicate<"!FeatureCnMips">; def IsSym32 : Predicate<"Subtarget->HasSym32()">, AssemblerPredicate<"FeatureSym32">; def IsSym64 : Predicate<"!Subtarget->HasSym32()">, AssemblerPredicate<"!FeatureSym32">; def IsN64 : Predicate<"Subtarget->isABI_N64()">; def IsNotN64 : Predicate<"!Subtarget->isABI_N64()">; def RelocNotPIC : Predicate<"!TM.isPositionIndependent()">; def RelocPIC : Predicate<"TM.isPositionIndependent()">; def NoNaNsFPMath : Predicate<"TM.Options.NoNaNsFPMath">; def HasStdEnc : Predicate<"Subtarget->hasStandardEncoding()">, AssemblerPredicate<"!FeatureMips16">; def NotDSP : Predicate<"!Subtarget->hasDSP()">; def InMicroMips : Predicate<"Subtarget->inMicroMipsMode()">, AssemblerPredicate<"FeatureMicroMips">; def NotInMicroMips : Predicate<"!Subtarget->inMicroMipsMode()">, AssemblerPredicate<"!FeatureMicroMips">; def IsLE : Predicate<"Subtarget->isLittle()">; def IsBE : Predicate<"!Subtarget->isLittle()">; def IsNotNaCl : Predicate<"!Subtarget->isTargetNaCl()">; def UseTCCInDIV : AssemblerPredicate<"FeatureUseTCCInDIV">; def HasEVA : Predicate<"Subtarget->hasEVA()">, AssemblerPredicate<"FeatureEVA">; def HasMSA : Predicate<"Subtarget->hasMSA()">, AssemblerPredicate<"FeatureMSA">; def HasMadd4 : Predicate<"!Subtarget->disableMadd4()">, AssemblerPredicate<"!FeatureMadd4">; def HasMT : Predicate<"Subtarget->hasMT()">, AssemblerPredicate<"FeatureMT">; def UseIndirectJumpsHazard : Predicate<"Subtarget->useIndirectJumpsHazard()">, AssemblerPredicate<"FeatureUseIndirectJumpsHazard">; def NoIndirectJumpGuards : Predicate<"!Subtarget->useIndirectJumpsHazard()">, AssemblerPredicate<"!FeatureUseIndirectJumpsHazard">; def HasCRC : Predicate<"Subtarget->hasCRC()">, AssemblerPredicate<"FeatureCRC">; def HasVirt : Predicate<"Subtarget->hasVirt()">, AssemblerPredicate<"FeatureVirt">; def HasGINV : Predicate<"Subtarget->hasGINV()">, AssemblerPredicate<"FeatureGINV">; // TODO: Add support for FPOpFusion::Standard def AllowFPOpFusion : Predicate<"TM.Options.AllowFPOpFusion ==" " FPOpFusion::Fast">; //===----------------------------------------------------------------------===// // Mips GPR size adjectives. // They are mutually exclusive. //===----------------------------------------------------------------------===// class GPR_32 { list GPRPredicates = [IsGP32bit]; } class GPR_64 { list GPRPredicates = [IsGP64bit]; } class PTR_32 { list PTRPredicates = [IsPTR32bit]; } class PTR_64 { list PTRPredicates = [IsPTR64bit]; } //===----------------------------------------------------------------------===// // Mips Symbol size adjectives. // They are mutally exculsive. //===----------------------------------------------------------------------===// class SYM_32 { list SYMPredicates = [IsSym32]; } class SYM_64 { list SYMPredicates = [IsSym64]; } //===----------------------------------------------------------------------===// // Mips ISA/ASE membership and instruction group membership adjectives. // They are mutually exclusive. //===----------------------------------------------------------------------===// // FIXME: I'd prefer to use additive predicates to build the instruction sets // but we are short on assembler feature bits at the moment. Using a // subtractive predicate will hopefully keep us under the 32 predicate // limit long enough to develop an alternative way to handle P1||P2 // predicates. class ISA_MIPS1 { list EncodingPredicates = [HasStdEnc]; } class ISA_MIPS1_NOT_MIPS3 { list InsnPredicates = [NotMips3]; list EncodingPredicates = [HasStdEnc]; } class ISA_MIPS1_NOT_4_32 { list InsnPredicates = [NotMips4_32]; list EncodingPredicates = [HasStdEnc]; } class ISA_MIPS1_NOT_32R6_64R6 { list InsnPredicates = [NotMips32r6, NotMips64r6]; list EncodingPredicates = [HasStdEnc]; } class ISA_MIPS2 { list InsnPredicates = [HasMips2]; list EncodingPredicates = [HasStdEnc]; } class ISA_MIPS2_NOT_32R6_64R6 { list InsnPredicates = [HasMips2, NotMips32r6, NotMips64r6]; list EncodingPredicates = [HasStdEnc]; } class ISA_MIPS3 { list InsnPredicates = [HasMips3]; list EncodingPredicates = [HasStdEnc]; } class ISA_MIPS3_NOT_32R6_64R6 { list InsnPredicates = [HasMips3, NotMips32r6, NotMips64r6]; list EncodingPredicates = [HasStdEnc]; } class ISA_MIPS32 { list InsnPredicates = [HasMips32]; list EncodingPredicates = [HasStdEnc]; } class ISA_MIPS32_NOT_32R6_64R6 { list InsnPredicates = [HasMips32, NotMips32r6, NotMips64r6]; list EncodingPredicates = [HasStdEnc]; } class ISA_MIPS32R2 { list InsnPredicates = [HasMips32r2]; list EncodingPredicates = [HasStdEnc]; } class ISA_MIPS32R2_NOT_32R6_64R6 { list InsnPredicates = [HasMips32r2, NotMips32r6, NotMips64r6]; list EncodingPredicates = [HasStdEnc]; } class ISA_MIPS32R5 { list InsnPredicates = [HasMips32r5]; list EncodingPredicates = [HasStdEnc]; } class ISA_MIPS64 { list InsnPredicates = [HasMips64]; list EncodingPredicates = [HasStdEnc]; } class ISA_MIPS64_NOT_64R6 { list InsnPredicates = [HasMips64, NotMips64r6]; list EncodingPredicates = [HasStdEnc]; } class ISA_MIPS64R2 { list InsnPredicates = [HasMips64r2]; list EncodingPredicates = [HasStdEnc]; } class ISA_MIPS64R5 { list InsnPredicates = [HasMips64r5]; list EncodingPredicates = [HasStdEnc]; } class ISA_MIPS32R6 { list InsnPredicates = [HasMips32r6]; list EncodingPredicates = [HasStdEnc]; } class ISA_MIPS64R6 { list InsnPredicates = [HasMips64r6]; list EncodingPredicates = [HasStdEnc]; } class ISA_MICROMIPS { list EncodingPredicates = [InMicroMips]; } class ISA_MICROMIPS32R5 { list InsnPredicates = [HasMips32r5]; list EncodingPredicates = [InMicroMips]; } class ISA_MICROMIPS32R6 { list InsnPredicates = [HasMips32r6]; list EncodingPredicates = [InMicroMips]; } class ISA_MICROMIPS64R6 { list InsnPredicates = [HasMips64r6]; list EncodingPredicates = [InMicroMips]; } class ISA_MICROMIPS32_NOT_MIPS32R6 { list InsnPredicates = [NotMips32r6]; list EncodingPredicates = [InMicroMips]; } class ASE_EVA { list ASEPredicate = [HasEVA]; } // The portions of MIPS-III that were also added to MIPS32 class INSN_MIPS3_32 { list InsnPredicates = [HasMips3_32]; list EncodingPredicates = [HasStdEnc]; } // The portions of MIPS-III that were also added to MIPS32 but were removed in // MIPS32r6 and MIPS64r6. class INSN_MIPS3_32_NOT_32R6_64R6 { list InsnPredicates = [HasMips3_32, NotMips32r6, NotMips64r6]; list EncodingPredicates = [HasStdEnc]; } // The portions of MIPS-III that were also added to MIPS32 class INSN_MIPS3_32R2 { list InsnPredicates = [HasMips3_32r2]; list EncodingPredicates = [HasStdEnc]; } // The portions of MIPS-IV that were also added to MIPS32. class INSN_MIPS4_32 { list InsnPredicates = [HasMips4_32]; list EncodingPredicates = [HasStdEnc]; } // The portions of MIPS-IV that were also added to MIPS32 but were removed in // MIPS32r6 and MIPS64r6. class INSN_MIPS4_32_NOT_32R6_64R6 { list InsnPredicates = [HasMips4_32, NotMips32r6, NotMips64r6]; list EncodingPredicates = [HasStdEnc]; } // The portions of MIPS-IV that were also added to MIPS32r2 but were removed in // MIPS32r6 and MIPS64r6. class INSN_MIPS4_32R2_NOT_32R6_64R6 { list InsnPredicates = [HasMips4_32r2, NotMips32r6, NotMips64r6]; list EncodingPredicates = [HasStdEnc]; } // The portions of MIPS-IV that were also added to MIPS32r2. class INSN_MIPS4_32R2 { list InsnPredicates = [HasMips4_32r2]; list EncodingPredicates = [HasStdEnc]; } // The portions of MIPS-V that were also added to MIPS32r2 but were removed in // MIPS32r6 and MIPS64r6. class INSN_MIPS5_32R2_NOT_32R6_64R6 { list InsnPredicates = [HasMips5_32r2, NotMips32r6, NotMips64r6]; list EncodingPredicates = [HasStdEnc]; } class ASE_CNMIPS { list ASEPredicate = [HasCnMips]; } class NOT_ASE_CNMIPS { list ASEPredicate = [NotCnMips]; } class ASE_MIPS64_CNMIPS { list ASEPredicate = [HasMips64, HasCnMips]; } class ASE_MSA { list ASEPredicate = [HasMSA]; } class ASE_MSA_NOT_MSA64 { list ASEPredicate = [HasMSA, NotMips64]; } class ASE_MSA64 { list ASEPredicate = [HasMSA, HasMips64]; } class ASE_MT { list ASEPredicate = [HasMT]; } class ASE_CRC { list ASEPredicate = [HasCRC]; } class ASE_VIRT { list ASEPredicate = [HasVirt]; } class ASE_GINV { list ASEPredicate = [HasGINV]; } // Class used for separating microMIPSr6 and microMIPS (r3) instruction. // It can be used only on instructions that doesn't inherit PredicateControl. class ISA_MICROMIPS_NOT_32R6 : PredicateControl { let InsnPredicates = [NotMips32r6]; let EncodingPredicates = [InMicroMips]; } class ASE_NOT_DSP { list ASEPredicate = [NotDSP]; } class MADD4 { list AdditionalPredicates = [HasMadd4]; } // Classses used for separating expansions that differ based on the ABI in // use. class ABI_N64 { list AdditionalPredicates = [IsN64]; } class ABI_NOT_N64 { list AdditionalPredicates = [IsNotN64]; } class FPOP_FUSION_FAST { list AdditionalPredicates = [AllowFPOpFusion]; } //===----------------------------------------------------------------------===// class MipsPat : Pat, PredicateControl; class MipsInstAlias : InstAlias, PredicateControl; class IsCommutable { bit isCommutable = 1; } class IsBranch { bit isBranch = 1; bit isCTI = 1; } class IsReturn { bit isReturn = 1; bit isCTI = 1; } class IsCall { bit isCall = 1; bit isCTI = 1; } class IsTailCall { bit isCall = 1; bit isTerminator = 1; bit isReturn = 1; bit isBarrier = 1; bit hasExtraSrcRegAllocReq = 1; bit isCodeGenOnly = 1; bit isCTI = 1; } class IsAsCheapAsAMove { bit isAsCheapAsAMove = 1; } class NeverHasSideEffects { bit hasSideEffects = 0; } //===----------------------------------------------------------------------===// // Instruction format superclass //===----------------------------------------------------------------------===// include "MipsInstrFormats.td" //===----------------------------------------------------------------------===// // Mips Operand, Complex Patterns and Transformations Definitions. //===----------------------------------------------------------------------===// class ConstantSImmAsmOperandClass Supers = [], int Offset = 0> : AsmOperandClass { let Name = "ConstantSImm" # Bits # "_" # Offset; let RenderMethod = "addConstantSImmOperands<" # Bits # ", " # Offset # ">"; let PredicateMethod = "isConstantSImm<" # Bits # ", " # Offset # ">"; let SuperClasses = Supers; let DiagnosticType = "SImm" # Bits # "_" # Offset; } class SimmLslAsmOperandClass Supers = [], int Shift = 0> : AsmOperandClass { let Name = "Simm" # Bits # "_Lsl" # Shift; let RenderMethod = "addImmOperands"; let PredicateMethod = "isScaledSImm<" # Bits # ", " # Shift # ">"; let SuperClasses = Supers; let DiagnosticType = "SImm" # Bits # "_Lsl" # Shift; } class ConstantUImmAsmOperandClass Supers = [], int Offset = 0> : AsmOperandClass { let Name = "ConstantUImm" # Bits # "_" # Offset; let RenderMethod = "addConstantUImmOperands<" # Bits # ", " # Offset # ">"; let PredicateMethod = "isConstantUImm<" # Bits # ", " # Offset # ">"; let SuperClasses = Supers; let DiagnosticType = "UImm" # Bits # "_" # Offset; } class ConstantUImmRangeAsmOperandClass Supers = []> : AsmOperandClass { let Name = "ConstantUImmRange" # Bottom # "_" # Top; let RenderMethod = "addImmOperands"; let PredicateMethod = "isConstantUImmRange<" # Bottom # ", " # Top # ">"; let SuperClasses = Supers; let DiagnosticType = "UImmRange" # Bottom # "_" # Top; } class SImmAsmOperandClass Supers = []> : AsmOperandClass { let Name = "SImm" # Bits; let RenderMethod = "addSImmOperands<" # Bits # ">"; let PredicateMethod = "isSImm<" # Bits # ">"; let SuperClasses = Supers; let DiagnosticType = "SImm" # Bits; } class UImmAsmOperandClass Supers = []> : AsmOperandClass { let Name = "UImm" # Bits; let RenderMethod = "addUImmOperands<" # Bits # ">"; let PredicateMethod = "isUImm<" # Bits # ">"; let SuperClasses = Supers; let DiagnosticType = "UImm" # Bits; } // Generic case - only to support certain assembly pseudo instructions. class UImmAnyAsmOperandClass Supers = []> : AsmOperandClass { let Name = "ImmAny"; let RenderMethod = "addConstantUImmOperands<32>"; let PredicateMethod = "isSImm<" # Bits # ">"; let SuperClasses = Supers; let DiagnosticType = "ImmAny"; } // AsmOperandClasses require a strict ordering which is difficult to manage // as a hierarchy. Instead, we use a linear ordering and impose an order that // is in some places arbitrary. // // Here the rules that are in use: // * Wider immediates are a superset of narrower immediates: // uimm4 < uimm5 < uimm6 // * For the same bit-width, unsigned immediates are a superset of signed // immediates:: // simm4 < uimm4 < simm5 < uimm5 // * For the same upper-bound, signed immediates are a superset of unsigned // immediates: // uimm3 < simm4 < uimm4 < simm4 // * Modified immediates are a superset of ordinary immediates: // uimm5 < uimm5_plus1 (1..32) < uimm5_plus32 (32..63) < uimm6 // The term 'superset' starts to break down here since the uimm5_plus* classes // are not true supersets of uimm5 (but they are still subsets of uimm6). // * 'Relaxed' immediates are supersets of the corresponding unsigned immediate. // uimm16 < uimm16_relaxed // * The codeGen pattern type is arbitrarily ordered. // uimm5 < uimm5_64, and uimm5 < vsplat_uimm5 // This is entirely arbitrary. We need an ordering and what we pick is // unimportant since only one is possible for a given mnemonic. def UImm32CoercedAsmOperandClass : UImmAnyAsmOperandClass<33, []> { let Name = "UImm32_Coerced"; let DiagnosticType = "UImm32_Coerced"; } def SImm32RelaxedAsmOperandClass : SImmAsmOperandClass<32, [UImm32CoercedAsmOperandClass]> { let Name = "SImm32_Relaxed"; let PredicateMethod = "isAnyImm<33>"; let DiagnosticType = "SImm32_Relaxed"; } def SImm32AsmOperandClass : SImmAsmOperandClass<32, [SImm32RelaxedAsmOperandClass]>; def ConstantUImm26AsmOperandClass : ConstantUImmAsmOperandClass<26, [SImm32AsmOperandClass]>; def ConstantUImm20AsmOperandClass : ConstantUImmAsmOperandClass<20, [ConstantUImm26AsmOperandClass]>; def ConstantSImm19Lsl2AsmOperandClass : AsmOperandClass { let Name = "SImm19Lsl2"; let RenderMethod = "addImmOperands"; let PredicateMethod = "isScaledSImm<19, 2>"; let SuperClasses = [ConstantUImm20AsmOperandClass]; let DiagnosticType = "SImm19_Lsl2"; } def UImm16RelaxedAsmOperandClass : UImmAsmOperandClass<16, [ConstantUImm20AsmOperandClass]> { let Name = "UImm16_Relaxed"; let PredicateMethod = "isAnyImm<16>"; let DiagnosticType = "UImm16_Relaxed"; } // Similar to the relaxed classes which take an SImm and render it as // an UImm, this takes a UImm and renders it as an SImm. def UImm16AltRelaxedAsmOperandClass : SImmAsmOperandClass<16, [UImm16RelaxedAsmOperandClass]> { let Name = "UImm16_AltRelaxed"; let PredicateMethod = "isUImm<16>"; let DiagnosticType = "UImm16_AltRelaxed"; } // FIXME: One of these should probably have UImm16AsmOperandClass as the // superclass instead of UImm16RelaxedasmOPerandClass. def UImm16AsmOperandClass : UImmAsmOperandClass<16, [UImm16RelaxedAsmOperandClass]>; def SImm16RelaxedAsmOperandClass : SImmAsmOperandClass<16, [UImm16RelaxedAsmOperandClass]> { let Name = "SImm16_Relaxed"; let PredicateMethod = "isAnyImm<16>"; let DiagnosticType = "SImm16_Relaxed"; } def SImm16AsmOperandClass : SImmAsmOperandClass<16, [SImm16RelaxedAsmOperandClass]>; def ConstantSImm10Lsl3AsmOperandClass : AsmOperandClass { let Name = "SImm10Lsl3"; let RenderMethod = "addImmOperands"; let PredicateMethod = "isScaledSImm<10, 3>"; let SuperClasses = [SImm16AsmOperandClass]; let DiagnosticType = "SImm10_Lsl3"; } def ConstantSImm10Lsl2AsmOperandClass : AsmOperandClass { let Name = "SImm10Lsl2"; let RenderMethod = "addImmOperands"; let PredicateMethod = "isScaledSImm<10, 2>"; let SuperClasses = [ConstantSImm10Lsl3AsmOperandClass]; let DiagnosticType = "SImm10_Lsl2"; } def ConstantSImm11AsmOperandClass : ConstantSImmAsmOperandClass<11, [ConstantSImm10Lsl2AsmOperandClass]>; def ConstantSImm10Lsl1AsmOperandClass : AsmOperandClass { let Name = "SImm10Lsl1"; let RenderMethod = "addImmOperands"; let PredicateMethod = "isScaledSImm<10, 1>"; let SuperClasses = [ConstantSImm11AsmOperandClass]; let DiagnosticType = "SImm10_Lsl1"; } def ConstantUImm10AsmOperandClass : ConstantUImmAsmOperandClass<10, [ConstantSImm10Lsl1AsmOperandClass]>; def ConstantSImm10AsmOperandClass : ConstantSImmAsmOperandClass<10, [ConstantUImm10AsmOperandClass]>; def ConstantSImm9AsmOperandClass : ConstantSImmAsmOperandClass<9, [ConstantSImm10AsmOperandClass]>; def ConstantSImm7Lsl2AsmOperandClass : AsmOperandClass { let Name = "SImm7Lsl2"; let RenderMethod = "addImmOperands"; let PredicateMethod = "isScaledSImm<7, 2>"; let SuperClasses = [ConstantSImm9AsmOperandClass]; let DiagnosticType = "SImm7_Lsl2"; } def ConstantUImm8AsmOperandClass : ConstantUImmAsmOperandClass<8, [ConstantSImm7Lsl2AsmOperandClass]>; def ConstantUImm7Sub1AsmOperandClass : ConstantUImmAsmOperandClass<7, [ConstantUImm8AsmOperandClass], -1> { // Specify the names since the -1 offset causes invalid identifiers otherwise. let Name = "UImm7_N1"; let DiagnosticType = "UImm7_N1"; } def ConstantUImm7AsmOperandClass : ConstantUImmAsmOperandClass<7, [ConstantUImm7Sub1AsmOperandClass]>; def ConstantUImm6Lsl2AsmOperandClass : AsmOperandClass { let Name = "UImm6Lsl2"; let RenderMethod = "addImmOperands"; let PredicateMethod = "isScaledUImm<6, 2>"; let SuperClasses = [ConstantUImm7AsmOperandClass]; let DiagnosticType = "UImm6_Lsl2"; } def ConstantUImm6AsmOperandClass : ConstantUImmAsmOperandClass<6, [ConstantUImm6Lsl2AsmOperandClass]>; def ConstantSImm6AsmOperandClass : ConstantSImmAsmOperandClass<6, [ConstantUImm6AsmOperandClass]>; def ConstantUImm5Lsl2AsmOperandClass : AsmOperandClass { let Name = "UImm5Lsl2"; let RenderMethod = "addImmOperands"; let PredicateMethod = "isScaledUImm<5, 2>"; let SuperClasses = [ConstantSImm6AsmOperandClass]; let DiagnosticType = "UImm5_Lsl2"; } def ConstantUImm5_Range2_64AsmOperandClass : ConstantUImmRangeAsmOperandClass<2, 64, [ConstantUImm5Lsl2AsmOperandClass]>; def ConstantUImm5Plus33AsmOperandClass : ConstantUImmAsmOperandClass<5, [ConstantUImm5_Range2_64AsmOperandClass], 33>; def ConstantUImm5ReportUImm6AsmOperandClass : ConstantUImmAsmOperandClass<5, [ConstantUImm5Plus33AsmOperandClass]> { let Name = "ConstantUImm5_0_Report_UImm6"; let DiagnosticType = "UImm5_0_Report_UImm6"; } def ConstantUImm5Plus32AsmOperandClass : ConstantUImmAsmOperandClass< 5, [ConstantUImm5ReportUImm6AsmOperandClass], 32>; def ConstantUImm5Plus32NormalizeAsmOperandClass : ConstantUImmAsmOperandClass<5, [ConstantUImm5Plus32AsmOperandClass], 32> { let Name = "ConstantUImm5_32_Norm"; // We must also subtract 32 when we render the operand. let RenderMethod = "addConstantUImmOperands<5, 32, -32>"; } def ConstantUImm5Plus1ReportUImm6AsmOperandClass : ConstantUImmAsmOperandClass< 5, [ConstantUImm5Plus32NormalizeAsmOperandClass], 1>{ let Name = "ConstantUImm5_Plus1_Report_UImm6"; } def ConstantUImm5Plus1AsmOperandClass : ConstantUImmAsmOperandClass< 5, [ConstantUImm5Plus1ReportUImm6AsmOperandClass], 1>; def ConstantUImm5AsmOperandClass : ConstantUImmAsmOperandClass<5, [ConstantUImm5Plus1AsmOperandClass]>; def ConstantSImm5AsmOperandClass : ConstantSImmAsmOperandClass<5, [ConstantUImm5AsmOperandClass]>; def ConstantUImm4AsmOperandClass : ConstantUImmAsmOperandClass<4, [ConstantSImm5AsmOperandClass]>; def ConstantSImm4AsmOperandClass : ConstantSImmAsmOperandClass<4, [ConstantUImm4AsmOperandClass]>; def ConstantUImm3AsmOperandClass : ConstantUImmAsmOperandClass<3, [ConstantSImm4AsmOperandClass]>; def ConstantUImm2Plus1AsmOperandClass : ConstantUImmAsmOperandClass<2, [ConstantUImm3AsmOperandClass], 1>; def ConstantUImm2AsmOperandClass : ConstantUImmAsmOperandClass<2, [ConstantUImm3AsmOperandClass]>; def ConstantUImm1AsmOperandClass : ConstantUImmAsmOperandClass<1, [ConstantUImm2AsmOperandClass]>; def ConstantImmzAsmOperandClass : AsmOperandClass { let Name = "ConstantImmz"; let RenderMethod = "addConstantUImmOperands<1>"; let PredicateMethod = "isConstantImmz"; let SuperClasses = [ConstantUImm1AsmOperandClass]; let DiagnosticType = "Immz"; } def Simm19Lsl2AsmOperand : SimmLslAsmOperandClass<19, [], 2>; def MipsJumpTargetAsmOperand : AsmOperandClass { let Name = "JumpTarget"; let ParserMethod = "parseJumpTarget"; let PredicateMethod = "isImm"; let RenderMethod = "addImmOperands"; } // Instruction operand types def jmptarget : Operand { let EncoderMethod = "getJumpTargetOpValue"; let ParserMatchClass = MipsJumpTargetAsmOperand; } def brtarget : Operand { let EncoderMethod = "getBranchTargetOpValue"; let OperandType = "OPERAND_PCREL"; let DecoderMethod = "DecodeBranchTarget"; let ParserMatchClass = MipsJumpTargetAsmOperand; } def brtarget1SImm16 : Operand { let EncoderMethod = "getBranchTargetOpValue1SImm16"; let OperandType = "OPERAND_PCREL"; let DecoderMethod = "DecodeBranchTarget1SImm16"; let ParserMatchClass = MipsJumpTargetAsmOperand; } def calltarget : Operand { let EncoderMethod = "getJumpTargetOpValue"; let ParserMatchClass = MipsJumpTargetAsmOperand; } def imm64: Operand; def simm19_lsl2 : Operand { let EncoderMethod = "getSimm19Lsl2Encoding"; let DecoderMethod = "DecodeSimm19Lsl2"; let ParserMatchClass = Simm19Lsl2AsmOperand; } def simm18_lsl3 : Operand { let EncoderMethod = "getSimm18Lsl3Encoding"; let DecoderMethod = "DecodeSimm18Lsl3"; let ParserMatchClass = MipsJumpTargetAsmOperand; } // Zero def uimmz : Operand { let PrintMethod = "printUImm<0>"; let ParserMatchClass = ConstantImmzAsmOperandClass; } // size operand of ins instruction def uimm_range_2_64 : Operand { let PrintMethod = "printUImm<6, 2>"; let EncoderMethod = "getSizeInsEncoding"; let DecoderMethod = "DecodeInsSize"; let ParserMatchClass = ConstantUImm5_Range2_64AsmOperandClass; } // Unsigned Operands foreach I = {1, 2, 3, 4, 5, 6, 7, 8, 10, 20, 26} in def uimm # I : Operand { let PrintMethod = "printUImm<" # I # ">"; let ParserMatchClass = !cast("ConstantUImm" # I # "AsmOperandClass"); } def uimm2_plus1 : Operand { let PrintMethod = "printUImm<2, 1>"; let EncoderMethod = "getUImmWithOffsetEncoding<2, 1>"; let DecoderMethod = "DecodeUImmWithOffset<2, 1>"; let ParserMatchClass = ConstantUImm2Plus1AsmOperandClass; } def uimm5_plus1 : Operand { let PrintMethod = "printUImm<5, 1>"; let EncoderMethod = "getUImmWithOffsetEncoding<5, 1>"; let DecoderMethod = "DecodeUImmWithOffset<5, 1>"; let ParserMatchClass = ConstantUImm5Plus1AsmOperandClass; } def uimm5_plus1_report_uimm6 : Operand { let PrintMethod = "printUImm<6, 1>"; let EncoderMethod = "getUImmWithOffsetEncoding<5, 1>"; let DecoderMethod = "DecodeUImmWithOffset<5, 1>"; let ParserMatchClass = ConstantUImm5Plus1ReportUImm6AsmOperandClass; } def uimm5_plus32 : Operand { let PrintMethod = "printUImm<5, 32>"; let ParserMatchClass = ConstantUImm5Plus32AsmOperandClass; } def uimm5_plus33 : Operand { let PrintMethod = "printUImm<5, 33>"; let EncoderMethod = "getUImmWithOffsetEncoding<5, 1>"; let DecoderMethod = "DecodeUImmWithOffset<5, 1>"; let ParserMatchClass = ConstantUImm5Plus33AsmOperandClass; } def uimm5_inssize_plus1 : Operand { let PrintMethod = "printUImm<6>"; let ParserMatchClass = ConstantUImm5Plus1AsmOperandClass; let EncoderMethod = "getSizeInsEncoding"; let DecoderMethod = "DecodeInsSize"; } def uimm5_plus32_normalize : Operand { let PrintMethod = "printUImm<5>"; let ParserMatchClass = ConstantUImm5Plus32NormalizeAsmOperandClass; } def uimm5_lsl2 : Operand { let EncoderMethod = "getUImm5Lsl2Encoding"; let DecoderMethod = "DecodeUImmWithOffsetAndScale<5, 0, 4>"; let ParserMatchClass = ConstantUImm5Lsl2AsmOperandClass; } def uimm5_plus32_normalize_64 : Operand { let PrintMethod = "printUImm<5>"; let ParserMatchClass = ConstantUImm5Plus32NormalizeAsmOperandClass; } def uimm6_lsl2 : Operand { let EncoderMethod = "getUImm6Lsl2Encoding"; let DecoderMethod = "DecodeUImmWithOffsetAndScale<6, 0, 4>"; let ParserMatchClass = ConstantUImm6Lsl2AsmOperandClass; } foreach I = {16} in def uimm # I : Operand { let PrintMethod = "printUImm<" # I # ">"; let ParserMatchClass = !cast("UImm" # I # "AsmOperandClass"); } // Like uimm16_64 but coerces simm16 to uimm16. def uimm16_relaxed : Operand { let PrintMethod = "printUImm<16>"; let ParserMatchClass = !cast("UImm16RelaxedAsmOperandClass"); } foreach I = {5} in def uimm # I # _64 : Operand { let PrintMethod = "printUImm<" # I # ">"; let ParserMatchClass = !cast("ConstantUImm" # I # "AsmOperandClass"); } foreach I = {16} in def uimm # I # _64 : Operand { let PrintMethod = "printUImm<" # I # ">"; let ParserMatchClass = !cast("UImm" # I # "AsmOperandClass"); } // Like uimm16_64 but coerces simm16 to uimm16. def uimm16_64_relaxed : Operand { let PrintMethod = "printUImm<16>"; let ParserMatchClass = !cast("UImm16RelaxedAsmOperandClass"); } def uimm16_altrelaxed : Operand { let PrintMethod = "printUImm<16>"; let ParserMatchClass = !cast("UImm16AltRelaxedAsmOperandClass"); } // Like uimm5 but reports a less confusing error for 32-63 when // an instruction alias permits that. def uimm5_report_uimm6 : Operand { let PrintMethod = "printUImm<6>"; let ParserMatchClass = ConstantUImm5ReportUImm6AsmOperandClass; } // Like uimm5_64 but reports a less confusing error for 32-63 when // an instruction alias permits that. def uimm5_64_report_uimm6 : Operand { let PrintMethod = "printUImm<5>"; let ParserMatchClass = ConstantUImm5ReportUImm6AsmOperandClass; } foreach I = {1, 2, 3, 4} in def uimm # I # _ptr : Operand { let PrintMethod = "printUImm<" # I # ">"; let ParserMatchClass = !cast("ConstantUImm" # I # "AsmOperandClass"); } foreach I = {1, 2, 3, 4, 5, 6, 8} in def vsplat_uimm # I : Operand { let PrintMethod = "printUImm<" # I # ">"; let ParserMatchClass = !cast("ConstantUImm" # I # "AsmOperandClass"); } // Signed operands foreach I = {4, 5, 6, 9, 10, 11} in def simm # I : Operand { let DecoderMethod = "DecodeSImmWithOffsetAndScale<" # I # ">"; let ParserMatchClass = !cast("ConstantSImm" # I # "AsmOperandClass"); } foreach I = {1, 2, 3} in def simm10_lsl # I : Operand { let DecoderMethod = "DecodeSImmWithOffsetAndScale<10, " # I # ">"; let ParserMatchClass = !cast("ConstantSImm10Lsl" # I # "AsmOperandClass"); } foreach I = {10} in def simm # I # _64 : Operand { let DecoderMethod = "DecodeSImmWithOffsetAndScale<" # I # ">"; let ParserMatchClass = !cast("ConstantSImm" # I # "AsmOperandClass"); } foreach I = {5, 10} in def vsplat_simm # I : Operand { let ParserMatchClass = !cast("ConstantSImm" # I # "AsmOperandClass"); } def simm7_lsl2 : Operand { let EncoderMethod = "getSImm7Lsl2Encoding"; let DecoderMethod = "DecodeSImmWithOffsetAndScale<" # I # ", 0, 4>"; let ParserMatchClass = ConstantSImm7Lsl2AsmOperandClass; } foreach I = {16, 32} in def simm # I : Operand { let DecoderMethod = "DecodeSImmWithOffsetAndScale<" # I # ">"; let ParserMatchClass = !cast("SImm" # I # "AsmOperandClass"); } // Like simm16 but coerces uimm16 to simm16. def simm16_relaxed : Operand { let DecoderMethod = "DecodeSImmWithOffsetAndScale<16>"; let ParserMatchClass = !cast("SImm16RelaxedAsmOperandClass"); } def simm16_64 : Operand { let DecoderMethod = "DecodeSImmWithOffsetAndScale<16>"; let ParserMatchClass = !cast("SImm16AsmOperandClass"); } // like simm32 but coerces simm32 to uimm32. def uimm32_coerced : Operand { let ParserMatchClass = !cast("UImm32CoercedAsmOperandClass"); } // Like simm32 but coerces uimm32 to simm32. def simm32_relaxed : Operand { let DecoderMethod = "DecodeSImmWithOffsetAndScale<32>"; let ParserMatchClass = !cast("SImm32RelaxedAsmOperandClass"); } // This is almost the same as a uimm7 but 0x7f is interpreted as -1. def li16_imm : Operand { let DecoderMethod = "DecodeLi16Imm"; let ParserMatchClass = ConstantUImm7Sub1AsmOperandClass; } def MipsMemAsmOperand : AsmOperandClass { let Name = "Mem"; let ParserMethod = "parseMemOperand"; } def MipsMemSimm9AsmOperand : AsmOperandClass { let Name = "MemOffsetSimm9"; let SuperClasses = [MipsMemAsmOperand]; let RenderMethod = "addMemOperands"; let ParserMethod = "parseMemOperand"; let PredicateMethod = "isMemWithSimmOffset<9>"; let DiagnosticType = "MemSImm9"; } def MipsMemSimm10AsmOperand : AsmOperandClass { let Name = "MemOffsetSimm10"; let SuperClasses = [MipsMemAsmOperand]; let RenderMethod = "addMemOperands"; let ParserMethod = "parseMemOperand"; let PredicateMethod = "isMemWithSimmOffset<10>"; let DiagnosticType = "MemSImm10"; } def MipsMemSimm12AsmOperand : AsmOperandClass { let Name = "MemOffsetSimm12"; let SuperClasses = [MipsMemAsmOperand]; let RenderMethod = "addMemOperands"; let ParserMethod = "parseMemOperand"; let PredicateMethod = "isMemWithSimmOffset<12>"; let DiagnosticType = "MemSImm12"; } foreach I = {1, 2, 3} in def MipsMemSimm10Lsl # I # AsmOperand : AsmOperandClass { let Name = "MemOffsetSimm10_" # I; let SuperClasses = [MipsMemAsmOperand]; let RenderMethod = "addMemOperands"; let ParserMethod = "parseMemOperand"; let PredicateMethod = "isMemWithSimmOffset<10, " # I # ">"; let DiagnosticType = "MemSImm10Lsl" # I; } def MipsMemSimm11AsmOperand : AsmOperandClass { let Name = "MemOffsetSimm11"; let SuperClasses = [MipsMemAsmOperand]; let RenderMethod = "addMemOperands"; let ParserMethod = "parseMemOperand"; let PredicateMethod = "isMemWithSimmOffset<11>"; let DiagnosticType = "MemSImm11"; } def MipsMemSimm16AsmOperand : AsmOperandClass { let Name = "MemOffsetSimm16"; let SuperClasses = [MipsMemAsmOperand]; let RenderMethod = "addMemOperands"; let ParserMethod = "parseMemOperand"; let PredicateMethod = "isMemWithSimmOffset<16>"; let DiagnosticType = "MemSImm16"; } def MipsMemSimmPtrAsmOperand : AsmOperandClass { let Name = "MemOffsetSimmPtr"; let SuperClasses = [MipsMemAsmOperand]; let RenderMethod = "addMemOperands"; let ParserMethod = "parseMemOperand"; let PredicateMethod = "isMemWithPtrSizeOffset"; let DiagnosticType = "MemSImmPtr"; } def MipsInvertedImmoperand : AsmOperandClass { let Name = "InvNum"; let RenderMethod = "addImmOperands"; let ParserMethod = "parseInvNum"; } def InvertedImOperand : Operand { let ParserMatchClass = MipsInvertedImmoperand; } def InvertedImOperand64 : Operand { let ParserMatchClass = MipsInvertedImmoperand; } class mem_generic : Operand { let PrintMethod = "printMemOperand"; let MIOperandInfo = (ops ptr_rc, simm16); let EncoderMethod = "getMemEncoding"; let ParserMatchClass = MipsMemAsmOperand; let OperandType = "OPERAND_MEMORY"; } // Address operand def mem : mem_generic; // MSA specific address operand def mem_msa : mem_generic { let MIOperandInfo = (ops ptr_rc, simm10); let EncoderMethod = "getMSAMemEncoding"; } def simm12 : Operand { let DecoderMethod = "DecodeSimm12"; } def mem_simm9 : mem_generic { let MIOperandInfo = (ops ptr_rc, simm9); let EncoderMethod = "getMemEncoding"; let ParserMatchClass = MipsMemSimm9AsmOperand; } def mem_simm10 : mem_generic { let MIOperandInfo = (ops ptr_rc, simm10); let EncoderMethod = "getMemEncoding"; let ParserMatchClass = MipsMemSimm10AsmOperand; } foreach I = {1, 2, 3} in def mem_simm10_lsl # I : mem_generic { let MIOperandInfo = (ops ptr_rc, !cast("simm10_lsl" # I)); let EncoderMethod = "getMemEncoding<" # I # ">"; let ParserMatchClass = !cast("MipsMemSimm10Lsl" # I # "AsmOperand"); } def mem_simm11 : mem_generic { let MIOperandInfo = (ops ptr_rc, simm11); let EncoderMethod = "getMemEncoding"; let ParserMatchClass = MipsMemSimm11AsmOperand; } def mem_simm12 : mem_generic { let MIOperandInfo = (ops ptr_rc, simm12); let EncoderMethod = "getMemEncoding"; let ParserMatchClass = MipsMemSimm12AsmOperand; } def mem_simm16 : mem_generic { let MIOperandInfo = (ops ptr_rc, simm16); let EncoderMethod = "getMemEncoding"; let ParserMatchClass = MipsMemSimm16AsmOperand; } def mem_simmptr : mem_generic { let ParserMatchClass = MipsMemSimmPtrAsmOperand; } def mem_ea : Operand { let PrintMethod = "printMemOperandEA"; let MIOperandInfo = (ops ptr_rc, simm16); let EncoderMethod = "getMemEncoding"; let OperandType = "OPERAND_MEMORY"; } def PtrRC : Operand { let MIOperandInfo = (ops ptr_rc); let DecoderMethod = "DecodePtrRegisterClass"; let ParserMatchClass = GPR32AsmOperand; } // size operand of ins instruction def size_ins : Operand { let EncoderMethod = "getSizeInsEncoding"; let DecoderMethod = "DecodeInsSize"; } // Transformation Function - get the lower 16 bits. def LO16 : SDNodeXFormgetZExtValue() & 0xFFFF); }]>; // Transformation Function - get the higher 16 bits. def HI16 : SDNodeXFormgetZExtValue() >> 16) & 0xFFFF); }]>; // Plus 1. def Plus1 : SDNodeXFormgetSExtValue() + 1); }]>; // Node immediate is zero (e.g. insve.d) def immz : PatLeaf<(imm), [{ return N->getSExtValue() == 0; }]>; // Node immediate fits as 16-bit sign extended on target immediate. // e.g. addi, andi def immSExt8 : PatLeaf<(imm), [{ return isInt<8>(N->getSExtValue()); }]>; // Node immediate fits as 16-bit sign extended on target immediate. // e.g. addi, andi def immSExt16 : PatLeaf<(imm), [{ return isInt<16>(N->getSExtValue()); }]>; // Node immediate fits as 7-bit zero extended on target immediate. def immZExt7 : PatLeaf<(imm), [{ return isUInt<7>(N->getZExtValue()); }]>; // Node immediate fits as 16-bit zero extended on target immediate. // The LO16 param means that only the lower 16 bits of the node // immediate are caught. // e.g. addiu, sltiu def immZExt16 : PatLeaf<(imm), [{ if (N->getValueType(0) == MVT::i32) return (uint32_t)N->getZExtValue() == (unsigned short)N->getZExtValue(); else return (uint64_t)N->getZExtValue() == (unsigned short)N->getZExtValue(); }], LO16>; // Immediate can be loaded with LUi (32-bit int with lower 16-bit cleared). def immSExt32Low16Zero : PatLeaf<(imm), [{ int64_t Val = N->getSExtValue(); return isInt<32>(Val) && !(Val & 0xffff); }]>; // Zero-extended 32-bit unsigned int with lower 16-bit cleared. def immZExt32Low16Zero : PatLeaf<(imm), [{ uint64_t Val = N->getZExtValue(); return isUInt<32>(Val) && !(Val & 0xffff); }]>; // Note immediate fits as a 32 bit signed extended on target immediate. def immSExt32 : PatLeaf<(imm), [{ return isInt<32>(N->getSExtValue()); }]>; // Note immediate fits as a 32 bit zero extended on target immediate. def immZExt32 : PatLeaf<(imm), [{ return isUInt<32>(N->getZExtValue()); }]>; // shamt field must fit in 5 bits. def immZExt5 : ImmLeaf; def immZExt5Plus1 : PatLeaf<(imm), [{ return isUInt<5>(N->getZExtValue() - 1); }]>; def immZExt5Plus32 : PatLeaf<(imm), [{ return isUInt<5>(N->getZExtValue() - 32); }]>; def immZExt5Plus33 : PatLeaf<(imm), [{ return isUInt<5>(N->getZExtValue() - 33); }]>; def immZExt5To31 : SDNodeXFormgetZExtValue()); }]>; // True if (N + 1) fits in 16-bit field. def immSExt16Plus1 : PatLeaf<(imm), [{ return isInt<17>(N->getSExtValue()) && isInt<16>(N->getSExtValue() + 1); }]>; def immZExtRange2To64 : PatLeaf<(imm), [{ return isUInt<7>(N->getZExtValue()) && (N->getZExtValue() >= 2) && (N->getZExtValue() <= 64); }]>; def ORiPred : PatLeaf<(imm), [{ return isUInt<16>(N->getZExtValue()) && !isInt<16>(N->getSExtValue()); }], LO16>; def LUiPred : PatLeaf<(imm), [{ int64_t Val = N->getSExtValue(); return !isInt<16>(Val) && isInt<32>(Val) && !(Val & 0xffff); }]>; def LUiORiPred : PatLeaf<(imm), [{ int64_t SVal = N->getSExtValue(); return isInt<32>(SVal) && (SVal & 0xffff); }]>; // Mips Address Mode! SDNode frameindex could possibily be a match // since load and store instructions from stack used it. def addr : ComplexPattern; def addrRegImm : ComplexPattern; def addrDefault : ComplexPattern; def addrimm10 : ComplexPattern; def addrimm10lsl1 : ComplexPattern; def addrimm10lsl2 : ComplexPattern; def addrimm10lsl3 : ComplexPattern; //===----------------------------------------------------------------------===// // Instructions specific format //===----------------------------------------------------------------------===// // Arithmetic and logical instructions with 3 register operands. class ArithLogicR: InstSE<(outs RO:$rd), (ins RO:$rs, RO:$rt), !strconcat(opstr, "\t$rd, $rs, $rt"), [(set RO:$rd, (OpNode RO:$rs, RO:$rt))], Itin, FrmR, opstr> { let isCommutable = isComm; let isReMaterializable = 1; let TwoOperandAliasConstraint = "$rd = $rs"; } // Arithmetic and logical instructions with 2 register operands. class ArithLogicI : InstSE<(outs RO:$rt), (ins RO:$rs, Od:$imm16), !strconcat(opstr, "\t$rt, $rs, $imm16"), [(set RO:$rt, (OpNode RO:$rs, imm_type:$imm16))], Itin, FrmI, opstr> { let isReMaterializable = 1; let TwoOperandAliasConstraint = "$rs = $rt"; } // Arithmetic Multiply ADD/SUB class MArithR : InstSE<(outs), (ins GPR32Opnd:$rs, GPR32Opnd:$rt), !strconcat(opstr, "\t$rs, $rt"), [], itin, FrmR, opstr> { let Defs = [HI0, LO0]; let Uses = [HI0, LO0]; let isCommutable = isComm; } // Logical class LogicNOR: InstSE<(outs RO:$rd), (ins RO:$rs, RO:$rt), !strconcat(opstr, "\t$rd, $rs, $rt"), [(set RO:$rd, (not (or RO:$rs, RO:$rt)))], II_NOR, FrmR, opstr> { let isCommutable = 1; } // Shifts class shift_rotate_imm : InstSE<(outs RO:$rd), (ins RO:$rt, ImmOpnd:$shamt), !strconcat(opstr, "\t$rd, $rt, $shamt"), [(set RO:$rd, (OpNode RO:$rt, PF:$shamt))], itin, FrmR, opstr> { let TwoOperandAliasConstraint = "$rt = $rd"; } class shift_rotate_reg: InstSE<(outs RO:$rd), (ins RO:$rt, GPR32Opnd:$rs), !strconcat(opstr, "\t$rd, $rt, $rs"), [(set RO:$rd, (OpNode RO:$rt, GPR32Opnd:$rs))], itin, FrmR, opstr>; // Load Upper Immediate class LoadUpper: InstSE<(outs RO:$rt), (ins Imm:$imm16), !strconcat(opstr, "\t$rt, $imm16"), [], II_LUI, FrmI, opstr>, IsAsCheapAsAMove { let hasSideEffects = 0; let isReMaterializable = 1; } // Memory Load/Store class LoadMemory : InstSE<(outs RO:$rt), (ins MO:$addr), !strconcat(opstr, "\t$rt, $addr"), [(set RO:$rt, (OpNode Addr:$addr))], Itin, FrmI, opstr> { let DecoderMethod = "DecodeMem"; let canFoldAsLoad = 1; string BaseOpcode = opstr; let mayLoad = 1; } class Load : LoadMemory; class StoreMemory : InstSE<(outs), (ins RO:$rt, MO:$addr), !strconcat(opstr, "\t$rt, $addr"), [(OpNode RO:$rt, Addr:$addr)], Itin, FrmI, opstr> { let DecoderMethod = "DecodeMem"; string BaseOpcode = opstr; let mayStore = 1; } class Store : StoreMemory; // Load/Store Left/Right let canFoldAsLoad = 1 in class LoadLeftRight : InstSE<(outs RO:$rt), (ins mem:$addr, RO:$src), !strconcat(opstr, "\t$rt, $addr"), [(set RO:$rt, (OpNode addr:$addr, RO:$src))], Itin, FrmI> { let DecoderMethod = "DecodeMem"; string Constraints = "$src = $rt"; let BaseOpcode = opstr; } class StoreLeftRight : InstSE<(outs), (ins RO:$rt, mem:$addr), !strconcat(opstr, "\t$rt, $addr"), [(OpNode RO:$rt, addr:$addr)], Itin, FrmI> { let DecoderMethod = "DecodeMem"; let BaseOpcode = opstr; } // COP2 Load/Store class LW_FT2 : InstSE<(outs RC:$rt), (ins mem_simm16:$addr), !strconcat(opstr, "\t$rt, $addr"), [(set RC:$rt, (OpNode addrDefault:$addr))], Itin, FrmFI, opstr> { let DecoderMethod = "DecodeFMem2"; let mayLoad = 1; } class SW_FT2 : InstSE<(outs), (ins RC:$rt, mem_simm16:$addr), !strconcat(opstr, "\t$rt, $addr"), [(OpNode RC:$rt, addrDefault:$addr)], Itin, FrmFI, opstr> { let DecoderMethod = "DecodeFMem2"; let mayStore = 1; } // COP3 Load/Store class LW_FT3 : InstSE<(outs RC:$rt), (ins mem:$addr), !strconcat(opstr, "\t$rt, $addr"), [(set RC:$rt, (OpNode addrDefault:$addr))], Itin, FrmFI, opstr> { let DecoderMethod = "DecodeFMem3"; let mayLoad = 1; } class SW_FT3 : InstSE<(outs), (ins RC:$rt, mem:$addr), !strconcat(opstr, "\t$rt, $addr"), [(OpNode RC:$rt, addrDefault:$addr)], Itin, FrmFI, opstr> { let DecoderMethod = "DecodeFMem3"; let mayStore = 1; } // Conditional Branch class CBranch : InstSE<(outs), (ins RO:$rs, RO:$rt, opnd:$offset), !strconcat(opstr, "\t$rs, $rt, $offset"), [(brcond (i32 (cond_op RO:$rs, RO:$rt)), bb:$offset)], II_BCC, FrmI, opstr> { let isBranch = 1; let isTerminator = 1; let hasDelaySlot = 1; let Defs = [AT]; bit isCTI = 1; } class CBranchLikely : InstSE<(outs), (ins RO:$rs, RO:$rt, opnd:$offset), !strconcat(opstr, "\t$rs, $rt, $offset"), [], II_BCC, FrmI, opstr> { let isBranch = 1; let isTerminator = 1; let hasDelaySlot = 1; let Defs = [AT]; bit isCTI = 1; } class CBranchZero : InstSE<(outs), (ins RO:$rs, opnd:$offset), !strconcat(opstr, "\t$rs, $offset"), [(brcond (i32 (cond_op RO:$rs, 0)), bb:$offset)], II_BCCZ, FrmI, opstr> { let isBranch = 1; let isTerminator = 1; let hasDelaySlot = 1; let Defs = [AT]; bit isCTI = 1; } class CBranchZeroLikely : InstSE<(outs), (ins RO:$rs, opnd:$offset), !strconcat(opstr, "\t$rs, $offset"), [], II_BCCZ, FrmI, opstr> { let isBranch = 1; let isTerminator = 1; let hasDelaySlot = 1; let Defs = [AT]; bit isCTI = 1; } // SetCC class SetCC_R : InstSE<(outs GPR32Opnd:$rd), (ins RO:$rs, RO:$rt), !strconcat(opstr, "\t$rd, $rs, $rt"), [(set GPR32Opnd:$rd, (cond_op RO:$rs, RO:$rt))], II_SLT_SLTU, FrmR, opstr>; class SetCC_I: InstSE<(outs GPR32Opnd:$rt), (ins RO:$rs, Od:$imm16), !strconcat(opstr, "\t$rt, $rs, $imm16"), [(set GPR32Opnd:$rt, (cond_op RO:$rs, imm_type:$imm16))], II_SLTI_SLTIU, FrmI, opstr>; // Jump class JumpFJ : InstSE<(outs), (ins opnd:$target), !strconcat(opstr, "\t$target"), [(operator targetoperator:$target)], II_J, FrmJ, bopstr> { let isTerminator=1; let isBarrier=1; let hasDelaySlot = 1; let DecoderMethod = "DecodeJumpTarget"; let Defs = [AT]; bit isCTI = 1; } // Unconditional branch class UncondBranch : PseudoSE<(outs), (ins brtarget:$offset), [(br bb:$offset)], II_B>, PseudoInstExpansion<(BEQInst ZERO, ZERO, opnd:$offset)> { let isBranch = 1; let isTerminator = 1; let isBarrier = 1; let hasDelaySlot = 1; let AdditionalPredicates = [RelocPIC]; let Defs = [AT]; bit isCTI = 1; } // Base class for indirect branch and return instruction classes. let isTerminator=1, isBarrier=1, hasDelaySlot = 1, isCTI = 1 in class JumpFR: InstSE<(outs), (ins RO:$rs), "jr\t$rs", [(operator RO:$rs)], II_JR, FrmR, opstr>; // Indirect branch class IndirectBranch : JumpFR { let isBranch = 1; let isIndirectBranch = 1; } // Jump and Link (Call) let isCall=1, hasDelaySlot=1, isCTI=1, Defs = [RA] in { class JumpLink : InstSE<(outs), (ins opnd:$target), !strconcat(opstr, "\t$target"), [(MipsJmpLink tglobaladdr:$target)], II_JAL, FrmJ, opstr> { let DecoderMethod = "DecodeJumpTarget"; } class JumpLinkRegPseudo: PseudoSE<(outs), (ins RO:$rs), [(MipsJmpLink RO:$rs)], II_JALR>, - PseudoInstExpansion<(JALRInst RetReg, ResRO:$rs)>; + PseudoInstExpansion<(JALRInst RetReg, ResRO:$rs)> { + let hasPostISelHook = 1; + } class JumpLinkReg: InstSE<(outs RO:$rd), (ins RO:$rs), !strconcat(opstr, "\t$rd, $rs"), - [], II_JALR, FrmR, opstr>; + [], II_JALR, FrmR, opstr> { + let hasPostISelHook = 1; + } class BGEZAL_FT : InstSE<(outs), (ins RO:$rs, opnd:$offset), !strconcat(opstr, "\t$rs, $offset"), [], II_BCCZAL, FrmI, opstr> { let hasDelaySlot = 1; } } let isCall = 1, isTerminator = 1, isReturn = 1, isBarrier = 1, hasDelaySlot = 1, hasExtraSrcRegAllocReq = 1, isCTI = 1, Defs = [AT] in { class TailCall : PseudoSE<(outs), (ins calltarget:$target), [], II_J>, PseudoInstExpansion<(JumpInst Opnd:$target)>; class TailCallReg : PseudoSE<(outs), (ins RO:$rs), [(MipsTailCall RO:$rs)], II_JR>, - PseudoInstExpansion<(JumpInst RO:$rs)>; + PseudoInstExpansion<(JumpInst RO:$rs)> { + let hasPostISelHook = 1; + } } class BAL_BR_Pseudo : PseudoSE<(outs), (ins opnd:$offset), [], II_BCCZAL>, PseudoInstExpansion<(RealInst ZERO, opnd:$offset)> { let isBranch = 1; let isTerminator = 1; let isBarrier = 1; let hasDelaySlot = 1; let Defs = [RA]; bit isCTI = 1; } let isCTI = 1 in { // Syscall class SYS_FT : InstSE<(outs), (ins ImmOp:$code_), !strconcat(opstr, "\t$code_"), [], itin, FrmI, opstr>; // Break class BRK_FT : InstSE<(outs), (ins uimm10:$code_1, uimm10:$code_2), !strconcat(opstr, "\t$code_1, $code_2"), [], II_BREAK, FrmOther, opstr>; // (D)Eret class ER_FT : InstSE<(outs), (ins), opstr, [], itin, FrmOther, opstr>; // Wait class WAIT_FT : InstSE<(outs), (ins), opstr, [], II_WAIT, FrmOther, opstr>; } // Interrupts class DEI_FT : InstSE<(outs RO:$rt), (ins), !strconcat(opstr, "\t$rt"), [], itin, FrmOther, opstr>; // Sync let hasSideEffects = 1 in class SYNC_FT : InstSE<(outs), (ins uimm5:$stype), "sync $stype", [(MipsSync immZExt5:$stype)], II_SYNC, FrmOther, opstr>; class SYNCI_FT : InstSE<(outs), (ins MO:$addr), !strconcat(opstr, "\t$addr"), [], II_SYNCI, FrmOther, opstr> { let hasSideEffects = 1; let DecoderMethod = "DecodeSyncI"; } let hasSideEffects = 1, isCTI = 1 in { class TEQ_FT : InstSE<(outs), (ins RO:$rs, RO:$rt, ImmOp:$code_), !strconcat(opstr, "\t$rs, $rt, $code_"), [], itin, FrmI, opstr>; class TEQI_FT : InstSE<(outs), (ins RO:$rs, simm16:$imm16), !strconcat(opstr, "\t$rs, $imm16"), [], itin, FrmOther, opstr>; } // Mul, Div class Mult DefRegs> : InstSE<(outs), (ins RO:$rs, RO:$rt), !strconcat(opstr, "\t$rs, $rt"), [], itin, FrmR, opstr> { let isCommutable = 1; let Defs = DefRegs; let hasSideEffects = 0; } // Pseudo multiply/divide instruction with explicit accumulator register // operands. class MultDivPseudo : PseudoSE<(outs R0:$ac), (ins R1:$rs, R1:$rt), [(set R0:$ac, (OpNode R1:$rs, R1:$rt))], Itin>, PseudoInstExpansion<(RealInst R1:$rs, R1:$rt)> { let isCommutable = IsComm; let hasSideEffects = HasSideEffects; let usesCustomInserter = UsesCustomInserter; } // Pseudo multiply add/sub instruction with explicit accumulator register // operands. class MAddSubPseudo : PseudoSE<(outs ACC64:$ac), (ins GPR32Opnd:$rs, GPR32Opnd:$rt, ACC64:$acin), [(set ACC64:$ac, (OpNode GPR32Opnd:$rs, GPR32Opnd:$rt, ACC64:$acin))], itin>, PseudoInstExpansion<(RealInst GPR32Opnd:$rs, GPR32Opnd:$rt)> { string Constraints = "$acin = $ac"; } class Div DefRegs> : InstSE<(outs), (ins RO:$rs, RO:$rt), !strconcat(opstr, "\t$$zero, $rs, $rt"), [], itin, FrmR, opstr> { let Defs = DefRegs; } // Move from Hi/Lo class PseudoMFLOHI : PseudoSE<(outs DstRC:$rd), (ins SrcRC:$hilo), [(set DstRC:$rd, (OpNode SrcRC:$hilo))], II_MFHI_MFLO>; class MoveFromLOHI: InstSE<(outs RO:$rd), (ins), !strconcat(opstr, "\t$rd"), [], II_MFHI_MFLO, FrmR, opstr> { let Uses = [UseReg]; let hasSideEffects = 0; let isMoveReg = 1; } class PseudoMTLOHI : PseudoSE<(outs DstRC:$lohi), (ins SrcRC:$lo, SrcRC:$hi), [(set DstRC:$lohi, (MipsMTLOHI SrcRC:$lo, SrcRC:$hi))], II_MTHI_MTLO>; class MoveToLOHI DefRegs>: InstSE<(outs), (ins RO:$rs), !strconcat(opstr, "\t$rs"), [], II_MTHI_MTLO, FrmR, opstr> { let Defs = DefRegs; let hasSideEffects = 0; let isMoveReg = 1; } class EffectiveAddress : InstSE<(outs RO:$rt), (ins mem_ea:$addr), !strconcat(opstr, "\t$rt, $addr"), [(set RO:$rt, addr:$addr)], II_ADDIU, FrmI, !strconcat(opstr, "_lea")> { let isCodeGenOnly = 1; let hasNoSchedulingInfo = 1; let DecoderMethod = "DecodeMem"; } // Count Leading Ones/Zeros in Word class CountLeading0: InstSE<(outs RO:$rd), (ins RO:$rs), !strconcat(opstr, "\t$rd, $rs"), [(set RO:$rd, (ctlz RO:$rs))], itin, FrmR, opstr>; class CountLeading1: InstSE<(outs RO:$rd), (ins RO:$rs), !strconcat(opstr, "\t$rd, $rs"), [(set RO:$rd, (ctlz (not RO:$rs)))], itin, FrmR, opstr>; // Sign Extend in Register. class SignExtInReg : InstSE<(outs RO:$rd), (ins RO:$rt), !strconcat(opstr, "\t$rd, $rt"), [(set RO:$rd, (sext_inreg RO:$rt, vt))], itin, FrmR, opstr>; // Subword Swap class SubwordSwap: InstSE<(outs RO:$rd), (ins RO:$rt), !strconcat(opstr, "\t$rd, $rt"), [], itin, FrmR, opstr> { let hasSideEffects = 0; } // Read Hardware class ReadHardware : InstSE<(outs CPURegOperand:$rt), (ins RO:$rd, uimm8:$sel), "rdhwr\t$rt, $rd, $sel", [], II_RDHWR, FrmR, "rdhwr">; // Ext and Ins class ExtBase : InstSE<(outs RO:$rt), (ins RO:$rs, PosOpnd:$pos, SizeOpnd:$size), !strconcat(opstr, "\t$rt, $rs, $pos, $size"), [(set RO:$rt, (Op RO:$rs, PosImm:$pos, SizeImm:$size))], II_EXT, FrmR, opstr>; // 'ins' and its' 64 bit variants are matched by C++ code. class InsBase: InstSE<(outs RO:$rt), (ins RO:$rs, PosOpnd:$pos, SizeOpnd:$size, RO:$src), !strconcat(opstr, "\t$rt, $rs, $pos, $size"), [(set RO:$rt, (null_frag RO:$rs, PosImm:$pos, SizeImm:$size, RO:$src))], II_INS, FrmR, opstr> { let Constraints = "$src = $rt"; } // Atomic instructions with 2 source operands (ATOMIC_SWAP & ATOMIC_LOAD_*). class Atomic2Ops : PseudoSE<(outs DRC:$dst), (ins PtrRC:$ptr, DRC:$incr), [(set DRC:$dst, (Op iPTR:$ptr, DRC:$incr))]>; class Atomic2OpsPostRA : PseudoSE<(outs RC:$dst), (ins PtrRC:$ptr, RC:$incr), []> { let mayLoad = 1; let mayStore = 1; } class Atomic2OpsSubwordPostRA : PseudoSE<(outs RC:$dst), (ins PtrRC:$ptr, RC:$incr, RC:$mask, RC:$mask2, RC:$shiftamnt), []>; // Atomic Compare & Swap. // Atomic compare and swap is lowered into two stages. The first stage happens // during ISelLowering, which produces the PostRA version of this instruction. class AtomicCmpSwap : PseudoSE<(outs DRC:$dst), (ins PtrRC:$ptr, DRC:$cmp, DRC:$swap), [(set DRC:$dst, (Op iPTR:$ptr, DRC:$cmp, DRC:$swap))]>; class AtomicCmpSwapPostRA : PseudoSE<(outs RC:$dst), (ins PtrRC:$ptr, RC:$cmp, RC:$swap), []> { let mayLoad = 1; let mayStore = 1; } class AtomicCmpSwapSubwordPostRA : PseudoSE<(outs RC:$dst), (ins PtrRC:$ptr, RC:$mask, RC:$ShiftCmpVal, RC:$mask2, RC:$ShiftNewVal, RC:$ShiftAmt), []> { let mayLoad = 1; let mayStore = 1; } class LLBase : InstSE<(outs RO:$rt), (ins MO:$addr), !strconcat(opstr, "\t$rt, $addr"), [], II_LL, FrmI, opstr> { let DecoderMethod = "DecodeMem"; let mayLoad = 1; } class SCBase : InstSE<(outs RO:$dst), (ins RO:$rt, mem:$addr), !strconcat(opstr, "\t$rt, $addr"), [], II_SC, FrmI> { let DecoderMethod = "DecodeMem"; let mayStore = 1; let Constraints = "$rt = $dst"; } class MFC3OP : InstSE<(outs RO:$rt), (ins RD:$rd, uimm3:$sel), !strconcat(asmstr, "\t$rt, $rd, $sel"), [], itin, FrmFR> { let BaseOpcode = asmstr; } class MTC3OP : InstSE<(outs RO:$rd), (ins RD:$rt, uimm3:$sel), !strconcat(asmstr, "\t$rt, $rd, $sel"), [], itin, FrmFR> { let BaseOpcode = asmstr; } class TrapBase : PseudoSE<(outs), (ins), [(trap)], II_TRAP>, PseudoInstExpansion<(RealInst 0, 0)> { let isBarrier = 1; let isTerminator = 1; let isCodeGenOnly = 1; let isCTI = 1; } //===----------------------------------------------------------------------===// // Pseudo instructions //===----------------------------------------------------------------------===// // Return RA. let isReturn=1, isTerminator=1, isBarrier=1, hasCtrlDep=1, isCTI=1 in { let hasDelaySlot=1 in def RetRA : PseudoSE<(outs), (ins), [(MipsRet)]>; let hasSideEffects=1 in def ERet : PseudoSE<(outs), (ins), [(MipsERet)]>; } let Defs = [SP], Uses = [SP], hasSideEffects = 1 in { def ADJCALLSTACKDOWN : MipsPseudo<(outs), (ins i32imm:$amt1, i32imm:$amt2), [(callseq_start timm:$amt1, timm:$amt2)]>; def ADJCALLSTACKUP : MipsPseudo<(outs), (ins i32imm:$amt1, i32imm:$amt2), [(callseq_end timm:$amt1, timm:$amt2)]>; } let usesCustomInserter = 1 in { def ATOMIC_LOAD_ADD_I8 : Atomic2Ops; def ATOMIC_LOAD_ADD_I16 : Atomic2Ops; def ATOMIC_LOAD_ADD_I32 : Atomic2Ops; def ATOMIC_LOAD_SUB_I8 : Atomic2Ops; def ATOMIC_LOAD_SUB_I16 : Atomic2Ops; def ATOMIC_LOAD_SUB_I32 : Atomic2Ops; def ATOMIC_LOAD_AND_I8 : Atomic2Ops; def ATOMIC_LOAD_AND_I16 : Atomic2Ops; def ATOMIC_LOAD_AND_I32 : Atomic2Ops; def ATOMIC_LOAD_OR_I8 : Atomic2Ops; def ATOMIC_LOAD_OR_I16 : Atomic2Ops; def ATOMIC_LOAD_OR_I32 : Atomic2Ops; def ATOMIC_LOAD_XOR_I8 : Atomic2Ops; def ATOMIC_LOAD_XOR_I16 : Atomic2Ops; def ATOMIC_LOAD_XOR_I32 : Atomic2Ops; def ATOMIC_LOAD_NAND_I8 : Atomic2Ops; def ATOMIC_LOAD_NAND_I16 : Atomic2Ops; def ATOMIC_LOAD_NAND_I32 : Atomic2Ops; def ATOMIC_SWAP_I8 : Atomic2Ops; def ATOMIC_SWAP_I16 : Atomic2Ops; def ATOMIC_SWAP_I32 : Atomic2Ops; def ATOMIC_CMP_SWAP_I8 : AtomicCmpSwap; def ATOMIC_CMP_SWAP_I16 : AtomicCmpSwap; def ATOMIC_CMP_SWAP_I32 : AtomicCmpSwap; } def ATOMIC_LOAD_ADD_I8_POSTRA : Atomic2OpsSubwordPostRA; def ATOMIC_LOAD_ADD_I16_POSTRA : Atomic2OpsSubwordPostRA; def ATOMIC_LOAD_ADD_I32_POSTRA : Atomic2OpsPostRA; def ATOMIC_LOAD_SUB_I8_POSTRA : Atomic2OpsSubwordPostRA; def ATOMIC_LOAD_SUB_I16_POSTRA : Atomic2OpsSubwordPostRA; def ATOMIC_LOAD_SUB_I32_POSTRA : Atomic2OpsPostRA; def ATOMIC_LOAD_AND_I8_POSTRA : Atomic2OpsSubwordPostRA; def ATOMIC_LOAD_AND_I16_POSTRA : Atomic2OpsSubwordPostRA; def ATOMIC_LOAD_AND_I32_POSTRA : Atomic2OpsPostRA; def ATOMIC_LOAD_OR_I8_POSTRA : Atomic2OpsSubwordPostRA; def ATOMIC_LOAD_OR_I16_POSTRA : Atomic2OpsSubwordPostRA; def ATOMIC_LOAD_OR_I32_POSTRA : Atomic2OpsPostRA; def ATOMIC_LOAD_XOR_I8_POSTRA : Atomic2OpsSubwordPostRA; def ATOMIC_LOAD_XOR_I16_POSTRA : Atomic2OpsSubwordPostRA; def ATOMIC_LOAD_XOR_I32_POSTRA : Atomic2OpsPostRA; def ATOMIC_LOAD_NAND_I8_POSTRA : Atomic2OpsSubwordPostRA; def ATOMIC_LOAD_NAND_I16_POSTRA : Atomic2OpsSubwordPostRA; def ATOMIC_LOAD_NAND_I32_POSTRA : Atomic2OpsPostRA; def ATOMIC_SWAP_I8_POSTRA : Atomic2OpsSubwordPostRA; def ATOMIC_SWAP_I16_POSTRA : Atomic2OpsSubwordPostRA; def ATOMIC_SWAP_I32_POSTRA : Atomic2OpsPostRA; def ATOMIC_CMP_SWAP_I8_POSTRA : AtomicCmpSwapSubwordPostRA; def ATOMIC_CMP_SWAP_I16_POSTRA : AtomicCmpSwapSubwordPostRA; def ATOMIC_CMP_SWAP_I32_POSTRA : AtomicCmpSwapPostRA; /// Pseudo instructions for loading and storing accumulator registers. let isPseudo = 1, isCodeGenOnly = 1, hasNoSchedulingInfo = 1 in { def LOAD_ACC64 : Load<"", ACC64>; def STORE_ACC64 : Store<"", ACC64>; } // We need these two pseudo instructions to avoid offset calculation for long // branches. See the comment in file MipsLongBranch.cpp for detailed // explanation. // Expands to: lui $dst, %highest/%higher/%hi/%lo($tgt - $baltgt) def LONG_BRANCH_LUi : PseudoSE<(outs GPR32Opnd:$dst), (ins brtarget:$tgt, brtarget:$baltgt), []>; // Expands to: lui $dst, highest/%higher/%hi/%lo($tgt) def LONG_BRANCH_LUi2Op : PseudoSE<(outs GPR32Opnd:$dst), (ins brtarget:$tgt), []>; // Expands to: addiu $dst, $src, %highest/%higher/%hi/%lo($tgt - $baltgt) def LONG_BRANCH_ADDiu : PseudoSE<(outs GPR32Opnd:$dst), (ins GPR32Opnd:$src, brtarget:$tgt, brtarget:$baltgt), []>; // Expands to: addiu $dst, $src, %highest/%higher/%hi/%lo($tgt) def LONG_BRANCH_ADDiu2Op : PseudoSE<(outs GPR32Opnd:$dst), (ins GPR32Opnd:$src, brtarget:$tgt), []>; //===----------------------------------------------------------------------===// // Instruction definition //===----------------------------------------------------------------------===// //===----------------------------------------------------------------------===// // MipsI Instructions //===----------------------------------------------------------------------===// /// Arithmetic Instructions (ALU Immediate) let AdditionalPredicates = [NotInMicroMips] in { def ADDiu : MMRel, StdMMR6Rel, ArithLogicI<"addiu", simm16_relaxed, GPR32Opnd, II_ADDIU, immSExt16, add>, ADDI_FM<0x9>, IsAsCheapAsAMove, ISA_MIPS1; def ANDi : MMRel, StdMMR6Rel, ArithLogicI<"andi", uimm16, GPR32Opnd, II_ANDI, immZExt16, and>, ADDI_FM<0xc>, ISA_MIPS1; def ORi : MMRel, StdMMR6Rel, ArithLogicI<"ori", uimm16, GPR32Opnd, II_ORI, immZExt16, or>, ADDI_FM<0xd>, ISA_MIPS1; def XORi : MMRel, StdMMR6Rel, ArithLogicI<"xori", uimm16, GPR32Opnd, II_XORI, immZExt16, xor>, ADDI_FM<0xe>, ISA_MIPS1; def ADDi : MMRel, ArithLogicI<"addi", simm16_relaxed, GPR32Opnd, II_ADDI>, ADDI_FM<0x8>, ISA_MIPS1_NOT_32R6_64R6; def SLTi : MMRel, SetCC_I<"slti", setlt, simm16, immSExt16, GPR32Opnd>, SLTI_FM<0xa>, ISA_MIPS1; def SLTiu : MMRel, SetCC_I<"sltiu", setult, simm16, immSExt16, GPR32Opnd>, SLTI_FM<0xb>, ISA_MIPS1; def LUi : MMRel, LoadUpper<"lui", GPR32Opnd, uimm16_relaxed>, LUI_FM, ISA_MIPS1; /// Arithmetic Instructions (3-Operand, R-Type) def ADDu : MMRel, StdMMR6Rel, ArithLogicR<"addu", GPR32Opnd, 1, II_ADDU, add>, ADD_FM<0, 0x21>, ISA_MIPS1; def SUBu : MMRel, StdMMR6Rel, ArithLogicR<"subu", GPR32Opnd, 0, II_SUBU, sub>, ADD_FM<0, 0x23>, ISA_MIPS1; let Defs = [HI0, LO0] in def MUL : MMRel, ArithLogicR<"mul", GPR32Opnd, 1, II_MUL, mul>, ADD_FM<0x1c, 2>, ISA_MIPS32_NOT_32R6_64R6; def ADD : MMRel, StdMMR6Rel, ArithLogicR<"add", GPR32Opnd, 1, II_ADD>, ADD_FM<0, 0x20>, ISA_MIPS1; def SUB : MMRel, StdMMR6Rel, ArithLogicR<"sub", GPR32Opnd, 0, II_SUB>, ADD_FM<0, 0x22>, ISA_MIPS1; def SLT : MMRel, SetCC_R<"slt", setlt, GPR32Opnd>, ADD_FM<0, 0x2a>, ISA_MIPS1; def SLTu : MMRel, SetCC_R<"sltu", setult, GPR32Opnd>, ADD_FM<0, 0x2b>, ISA_MIPS1; def AND : MMRel, StdMMR6Rel, ArithLogicR<"and", GPR32Opnd, 1, II_AND, and>, ADD_FM<0, 0x24>, ISA_MIPS1; def OR : MMRel, StdMMR6Rel, ArithLogicR<"or", GPR32Opnd, 1, II_OR, or>, ADD_FM<0, 0x25>, ISA_MIPS1; def XOR : MMRel, StdMMR6Rel, ArithLogicR<"xor", GPR32Opnd, 1, II_XOR, xor>, ADD_FM<0, 0x26>, ISA_MIPS1; def NOR : MMRel, StdMMR6Rel, LogicNOR<"nor", GPR32Opnd>, ADD_FM<0, 0x27>, ISA_MIPS1; } let AdditionalPredicates = [NotInMicroMips] in { /// Shift Instructions def SLL : MMRel, shift_rotate_imm<"sll", uimm5, GPR32Opnd, II_SLL, shl, immZExt5>, SRA_FM<0, 0>, ISA_MIPS1; def SRL : MMRel, shift_rotate_imm<"srl", uimm5, GPR32Opnd, II_SRL, srl, immZExt5>, SRA_FM<2, 0>, ISA_MIPS1; def SRA : MMRel, shift_rotate_imm<"sra", uimm5, GPR32Opnd, II_SRA, sra, immZExt5>, SRA_FM<3, 0>, ISA_MIPS1; def SLLV : MMRel, shift_rotate_reg<"sllv", GPR32Opnd, II_SLLV, shl>, SRLV_FM<4, 0>, ISA_MIPS1; def SRLV : MMRel, shift_rotate_reg<"srlv", GPR32Opnd, II_SRLV, srl>, SRLV_FM<6, 0>, ISA_MIPS1; def SRAV : MMRel, shift_rotate_reg<"srav", GPR32Opnd, II_SRAV, sra>, SRLV_FM<7, 0>, ISA_MIPS1; // Rotate Instructions def ROTR : MMRel, shift_rotate_imm<"rotr", uimm5, GPR32Opnd, II_ROTR, rotr, immZExt5>, SRA_FM<2, 1>, ISA_MIPS32R2; def ROTRV : MMRel, shift_rotate_reg<"rotrv", GPR32Opnd, II_ROTRV, rotr>, SRLV_FM<6, 1>, ISA_MIPS32R2; } /// Load and Store Instructions /// aligned let AdditionalPredicates = [NotInMicroMips] in { def LB : LoadMemory<"lb", GPR32Opnd, mem_simmptr, sextloadi8, II_LB>, MMRel, LW_FM<0x20>, ISA_MIPS1; def LBu : LoadMemory<"lbu", GPR32Opnd, mem_simmptr, zextloadi8, II_LBU, addrDefault>, MMRel, LW_FM<0x24>, ISA_MIPS1; def LH : LoadMemory<"lh", GPR32Opnd, mem_simmptr, sextloadi16, II_LH, addrDefault>, MMRel, LW_FM<0x21>, ISA_MIPS1; def LHu : LoadMemory<"lhu", GPR32Opnd, mem_simmptr, zextloadi16, II_LHU>, MMRel, LW_FM<0x25>, ISA_MIPS1; def LW : StdMMR6Rel, Load<"lw", GPR32Opnd, load, II_LW, addrDefault>, MMRel, LW_FM<0x23>, ISA_MIPS1; def SB : StdMMR6Rel, Store<"sb", GPR32Opnd, truncstorei8, II_SB>, MMRel, LW_FM<0x28>, ISA_MIPS1; def SH : Store<"sh", GPR32Opnd, truncstorei16, II_SH>, MMRel, LW_FM<0x29>, ISA_MIPS1; def SW : Store<"sw", GPR32Opnd, store, II_SW>, MMRel, LW_FM<0x2b>, ISA_MIPS1; } /// load/store left/right let AdditionalPredicates = [NotInMicroMips] in { def LWL : MMRel, LoadLeftRight<"lwl", MipsLWL, GPR32Opnd, II_LWL>, LW_FM<0x22>, ISA_MIPS1_NOT_32R6_64R6; def LWR : MMRel, LoadLeftRight<"lwr", MipsLWR, GPR32Opnd, II_LWR>, LW_FM<0x26>, ISA_MIPS1_NOT_32R6_64R6; def SWL : MMRel, StoreLeftRight<"swl", MipsSWL, GPR32Opnd, II_SWL>, LW_FM<0x2a>, ISA_MIPS1_NOT_32R6_64R6; def SWR : MMRel, StoreLeftRight<"swr", MipsSWR, GPR32Opnd, II_SWR>, LW_FM<0x2e>, ISA_MIPS1_NOT_32R6_64R6; // COP2 Memory Instructions def LWC2 : StdMMR6Rel, LW_FT2<"lwc2", COP2Opnd, II_LWC2, load>, LW_FM<0x32>, ISA_MIPS1_NOT_32R6_64R6; def SWC2 : StdMMR6Rel, SW_FT2<"swc2", COP2Opnd, II_SWC2, store>, LW_FM<0x3a>, ISA_MIPS1_NOT_32R6_64R6; def LDC2 : StdMMR6Rel, LW_FT2<"ldc2", COP2Opnd, II_LDC2, load>, LW_FM<0x36>, ISA_MIPS2_NOT_32R6_64R6; def SDC2 : StdMMR6Rel, SW_FT2<"sdc2", COP2Opnd, II_SDC2, store>, LW_FM<0x3e>, ISA_MIPS2_NOT_32R6_64R6; // COP3 Memory Instructions let DecoderNamespace = "COP3_" in { def LWC3 : LW_FT3<"lwc3", COP3Opnd, II_LWC3, load>, LW_FM<0x33>, ISA_MIPS1_NOT_32R6_64R6, NOT_ASE_CNMIPS; def SWC3 : SW_FT3<"swc3", COP3Opnd, II_SWC3, store>, LW_FM<0x3b>, ISA_MIPS1_NOT_32R6_64R6, NOT_ASE_CNMIPS; def LDC3 : LW_FT3<"ldc3", COP3Opnd, II_LDC3, load>, LW_FM<0x37>, ISA_MIPS2, NOT_ASE_CNMIPS; def SDC3 : SW_FT3<"sdc3", COP3Opnd, II_SDC3, store>, LW_FM<0x3f>, ISA_MIPS2, NOT_ASE_CNMIPS; } def SYNC : MMRel, StdMMR6Rel, SYNC_FT<"sync">, SYNC_FM, ISA_MIPS2; def SYNCI : MMRel, StdMMR6Rel, SYNCI_FT<"synci", mem_simm16>, SYNCI_FM, ISA_MIPS32R2; } let AdditionalPredicates = [NotInMicroMips] in { def TEQ : MMRel, TEQ_FT<"teq", GPR32Opnd, uimm10, II_TEQ>, TEQ_FM<0x34>, ISA_MIPS2; def TGE : MMRel, TEQ_FT<"tge", GPR32Opnd, uimm10, II_TGE>, TEQ_FM<0x30>, ISA_MIPS2; def TGEU : MMRel, TEQ_FT<"tgeu", GPR32Opnd, uimm10, II_TGEU>, TEQ_FM<0x31>, ISA_MIPS2; def TLT : MMRel, TEQ_FT<"tlt", GPR32Opnd, uimm10, II_TLT>, TEQ_FM<0x32>, ISA_MIPS2; def TLTU : MMRel, TEQ_FT<"tltu", GPR32Opnd, uimm10, II_TLTU>, TEQ_FM<0x33>, ISA_MIPS2; def TNE : MMRel, TEQ_FT<"tne", GPR32Opnd, uimm10, II_TNE>, TEQ_FM<0x36>, ISA_MIPS2; def TEQI : MMRel, TEQI_FT<"teqi", GPR32Opnd, II_TEQI>, TEQI_FM<0xc>, ISA_MIPS2_NOT_32R6_64R6; def TGEI : MMRel, TEQI_FT<"tgei", GPR32Opnd, II_TGEI>, TEQI_FM<0x8>, ISA_MIPS2_NOT_32R6_64R6; def TGEIU : MMRel, TEQI_FT<"tgeiu", GPR32Opnd, II_TGEIU>, TEQI_FM<0x9>, ISA_MIPS2_NOT_32R6_64R6; def TLTI : MMRel, TEQI_FT<"tlti", GPR32Opnd, II_TLTI>, TEQI_FM<0xa>, ISA_MIPS2_NOT_32R6_64R6; def TTLTIU : MMRel, TEQI_FT<"tltiu", GPR32Opnd, II_TTLTIU>, TEQI_FM<0xb>, ISA_MIPS2_NOT_32R6_64R6; def TNEI : MMRel, TEQI_FT<"tnei", GPR32Opnd, II_TNEI>, TEQI_FM<0xe>, ISA_MIPS2_NOT_32R6_64R6; } let AdditionalPredicates = [NotInMicroMips] in { def BREAK : MMRel, StdMMR6Rel, BRK_FT<"break">, BRK_FM<0xd>, ISA_MIPS1; def SYSCALL : MMRel, SYS_FT<"syscall", uimm20, II_SYSCALL>, SYS_FM<0xc>, ISA_MIPS1; def TRAP : TrapBase, ISA_MIPS1; def SDBBP : MMRel, SYS_FT<"sdbbp", uimm20, II_SDBBP>, SDBBP_FM, ISA_MIPS32_NOT_32R6_64R6; def ERET : MMRel, ER_FT<"eret", II_ERET>, ER_FM<0x18, 0x0>, INSN_MIPS3_32; def ERETNC : MMRel, ER_FT<"eretnc", II_ERETNC>, ER_FM<0x18, 0x1>, ISA_MIPS32R5; def DERET : MMRel, ER_FT<"deret", II_DERET>, ER_FM<0x1f, 0x0>, ISA_MIPS32; def EI : MMRel, StdMMR6Rel, DEI_FT<"ei", GPR32Opnd, II_EI>, EI_FM<1>, ISA_MIPS32R2; def DI : MMRel, StdMMR6Rel, DEI_FT<"di", GPR32Opnd, II_DI>, EI_FM<0>, ISA_MIPS32R2; def WAIT : MMRel, StdMMR6Rel, WAIT_FT<"wait">, WAIT_FM, INSN_MIPS3_32; } let AdditionalPredicates = [NotInMicroMips] in { /// Load-linked, Store-conditional def LL : LLBase<"ll", GPR32Opnd>, LW_FM<0x30>, PTR_32, ISA_MIPS2_NOT_32R6_64R6; def SC : SCBase<"sc", GPR32Opnd>, LW_FM<0x38>, PTR_32, ISA_MIPS2_NOT_32R6_64R6; } /// Jump and Branch Instructions let AdditionalPredicates = [NotInMicroMips, RelocNotPIC] in def J : MMRel, JumpFJ, FJ<2>, IsBranch, ISA_MIPS1; let AdditionalPredicates = [NotInMicroMips] in { def JR : MMRel, IndirectBranch<"jr", GPR32Opnd>, MTLO_FM<8>, ISA_MIPS1_NOT_32R6_64R6; def BEQ : MMRel, CBranch<"beq", brtarget, seteq, GPR32Opnd>, BEQ_FM<4>, ISA_MIPS1; def BEQL : MMRel, CBranchLikely<"beql", brtarget, GPR32Opnd>, BEQ_FM<20>, ISA_MIPS2_NOT_32R6_64R6; def BNE : MMRel, CBranch<"bne", brtarget, setne, GPR32Opnd>, BEQ_FM<5>, ISA_MIPS1; def BNEL : MMRel, CBranchLikely<"bnel", brtarget, GPR32Opnd>, BEQ_FM<21>, ISA_MIPS2_NOT_32R6_64R6; def BGEZ : MMRel, CBranchZero<"bgez", brtarget, setge, GPR32Opnd>, BGEZ_FM<1, 1>, ISA_MIPS1; def BGEZL : MMRel, CBranchZeroLikely<"bgezl", brtarget, GPR32Opnd>, BGEZ_FM<1, 3>, ISA_MIPS2_NOT_32R6_64R6; def BGTZ : MMRel, CBranchZero<"bgtz", brtarget, setgt, GPR32Opnd>, BGEZ_FM<7, 0>, ISA_MIPS1; def BGTZL : MMRel, CBranchZeroLikely<"bgtzl", brtarget, GPR32Opnd>, BGEZ_FM<23, 0>, ISA_MIPS2_NOT_32R6_64R6; def BLEZ : MMRel, CBranchZero<"blez", brtarget, setle, GPR32Opnd>, BGEZ_FM<6, 0>, ISA_MIPS1; def BLEZL : MMRel, CBranchZeroLikely<"blezl", brtarget, GPR32Opnd>, BGEZ_FM<22, 0>, ISA_MIPS2_NOT_32R6_64R6; def BLTZ : MMRel, CBranchZero<"bltz", brtarget, setlt, GPR32Opnd>, BGEZ_FM<1, 0>, ISA_MIPS1; def BLTZL : MMRel, CBranchZeroLikely<"bltzl", brtarget, GPR32Opnd>, BGEZ_FM<1, 2>, ISA_MIPS2_NOT_32R6_64R6; def B : UncondBranch, ISA_MIPS1; def JAL : MMRel, JumpLink<"jal", calltarget>, FJ<3>, ISA_MIPS1; } let AdditionalPredicates = [NotInMicroMips, NoIndirectJumpGuards] in { def JALR : JumpLinkReg<"jalr", GPR32Opnd>, JALR_FM, ISA_MIPS1; def JALRPseudo : JumpLinkRegPseudo, ISA_MIPS1; } let AdditionalPredicates = [NotInMicroMips] in { def JALX : MMRel, JumpLink<"jalx", calltarget>, FJ<0x1D>, ISA_MIPS32_NOT_32R6_64R6; def BGEZAL : MMRel, BGEZAL_FT<"bgezal", brtarget, GPR32Opnd>, BGEZAL_FM<0x11>, ISA_MIPS1_NOT_32R6_64R6; def BGEZALL : MMRel, BGEZAL_FT<"bgezall", brtarget, GPR32Opnd>, BGEZAL_FM<0x13>, ISA_MIPS2_NOT_32R6_64R6; def BLTZAL : MMRel, BGEZAL_FT<"bltzal", brtarget, GPR32Opnd>, BGEZAL_FM<0x10>, ISA_MIPS1_NOT_32R6_64R6; def BLTZALL : MMRel, BGEZAL_FT<"bltzall", brtarget, GPR32Opnd>, BGEZAL_FM<0x12>, ISA_MIPS2_NOT_32R6_64R6; def BAL_BR : BAL_BR_Pseudo, ISA_MIPS1; } let AdditionalPredicates = [NotInMips16Mode, NotInMicroMips] in { def TAILCALL : TailCall, ISA_MIPS1; } let AdditionalPredicates = [NotInMips16Mode, NotInMicroMips, NoIndirectJumpGuards] in def TAILCALLREG : TailCallReg, ISA_MIPS1_NOT_32R6_64R6; // Indirect branches are matched as PseudoIndirectBranch/PseudoIndirectBranch64 // then are expanded to JR, JR64, JALR, or JALR64 depending on the ISA. class PseudoIndirectBranchBase : MipsPseudo<(outs), (ins RO:$rs), [(brind RO:$rs)], II_IndirectBranchPseudo>, PseudoInstExpansion<(JumpInst RO:$rs)> { let isTerminator=1; let isBarrier=1; let hasDelaySlot = 1; let isBranch = 1; let isIndirectBranch = 1; bit isCTI = 1; } let AdditionalPredicates = [NotInMips16Mode, NotInMicroMips, NoIndirectJumpGuards] in def PseudoIndirectBranch : PseudoIndirectBranchBase, ISA_MIPS1_NOT_32R6_64R6; // Return instructions are matched as a RetRA instruction, then are expanded // into PseudoReturn/PseudoReturn64 after register allocation. Finally, // MipsAsmPrinter expands this into JR, JR64, JALR, or JALR64 depending on the // ISA. class PseudoReturnBase : MipsPseudo<(outs), (ins RO:$rs), [], II_ReturnPseudo> { let isTerminator = 1; let isBarrier = 1; let hasDelaySlot = 1; let isReturn = 1; let isCodeGenOnly = 1; let hasCtrlDep = 1; let hasExtraSrcRegAllocReq = 1; bit isCTI = 1; } def PseudoReturn : PseudoReturnBase; // Exception handling related node and instructions. // The conversion sequence is: // ISD::EH_RETURN -> MipsISD::EH_RETURN -> // MIPSeh_return -> (stack change + indirect branch) // // MIPSeh_return takes the place of regular return instruction // but takes two arguments (V1, V0) which are used for storing // the offset and return address respectively. def SDT_MipsEHRET : SDTypeProfile<0, 2, [SDTCisInt<0>, SDTCisPtrTy<1>]>; def MIPSehret : SDNode<"MipsISD::EH_RETURN", SDT_MipsEHRET, [SDNPHasChain, SDNPOptInGlue, SDNPVariadic]>; let Uses = [V0, V1], isTerminator = 1, isReturn = 1, isBarrier = 1, isCTI = 1 in { def MIPSeh_return32 : MipsPseudo<(outs), (ins GPR32:$spoff, GPR32:$dst), [(MIPSehret GPR32:$spoff, GPR32:$dst)]>; def MIPSeh_return64 : MipsPseudo<(outs), (ins GPR64:$spoff, GPR64:$dst), [(MIPSehret GPR64:$spoff, GPR64:$dst)]>; } /// Multiply and Divide Instructions. let AdditionalPredicates = [NotInMicroMips] in { def MULT : MMRel, Mult<"mult", II_MULT, GPR32Opnd, [HI0, LO0]>, MULT_FM<0, 0x18>, ISA_MIPS1_NOT_32R6_64R6; def MULTu : MMRel, Mult<"multu", II_MULTU, GPR32Opnd, [HI0, LO0]>, MULT_FM<0, 0x19>, ISA_MIPS1_NOT_32R6_64R6; def SDIV : MMRel, Div<"div", II_DIV, GPR32Opnd, [HI0, LO0]>, MULT_FM<0, 0x1a>, ISA_MIPS1_NOT_32R6_64R6; def UDIV : MMRel, Div<"divu", II_DIVU, GPR32Opnd, [HI0, LO0]>, MULT_FM<0, 0x1b>, ISA_MIPS1_NOT_32R6_64R6; def MTHI : MMRel, MoveToLOHI<"mthi", GPR32Opnd, [HI0]>, MTLO_FM<0x11>, ISA_MIPS1_NOT_32R6_64R6; def MTLO : MMRel, MoveToLOHI<"mtlo", GPR32Opnd, [LO0]>, MTLO_FM<0x13>, ISA_MIPS1_NOT_32R6_64R6; def MFHI : MMRel, MoveFromLOHI<"mfhi", GPR32Opnd, AC0>, MFLO_FM<0x10>, ISA_MIPS1_NOT_32R6_64R6; def MFLO : MMRel, MoveFromLOHI<"mflo", GPR32Opnd, AC0>, MFLO_FM<0x12>, ISA_MIPS1_NOT_32R6_64R6; /// Sign Ext In Register Instructions. def SEB : MMRel, StdMMR6Rel, SignExtInReg<"seb", i8, GPR32Opnd, II_SEB>, SEB_FM<0x10, 0x20>, ISA_MIPS32R2; def SEH : MMRel, StdMMR6Rel, SignExtInReg<"seh", i16, GPR32Opnd, II_SEH>, SEB_FM<0x18, 0x20>, ISA_MIPS32R2; /// Count Leading def CLZ : MMRel, CountLeading0<"clz", GPR32Opnd, II_CLZ>, CLO_FM<0x20>, ISA_MIPS32_NOT_32R6_64R6; def CLO : MMRel, CountLeading1<"clo", GPR32Opnd, II_CLO>, CLO_FM<0x21>, ISA_MIPS32_NOT_32R6_64R6; /// Word Swap Bytes Within Halfwords def WSBH : MMRel, SubwordSwap<"wsbh", GPR32Opnd, II_WSBH>, SEB_FM<2, 0x20>, ISA_MIPS32R2; /// No operation. def NOP : PseudoSE<(outs), (ins), []>, PseudoInstExpansion<(SLL ZERO, ZERO, 0)>, ISA_MIPS1; // FrameIndexes are legalized when they are operands from load/store // instructions. The same not happens for stack address copies, so an // add op with mem ComplexPattern is used and the stack address copy // can be matched. It's similar to Sparc LEA_ADDRi let AdditionalPredicates = [NotInMicroMips] in def LEA_ADDiu : MMRel, EffectiveAddress<"addiu", GPR32Opnd>, LW_FM<9>, ISA_MIPS1; // MADD*/MSUB* def MADD : MMRel, MArithR<"madd", II_MADD, 1>, MULT_FM<0x1c, 0>, ISA_MIPS32_NOT_32R6_64R6; def MADDU : MMRel, MArithR<"maddu", II_MADDU, 1>, MULT_FM<0x1c, 1>, ISA_MIPS32_NOT_32R6_64R6; def MSUB : MMRel, MArithR<"msub", II_MSUB>, MULT_FM<0x1c, 4>, ISA_MIPS32_NOT_32R6_64R6; def MSUBU : MMRel, MArithR<"msubu", II_MSUBU>, MULT_FM<0x1c, 5>, ISA_MIPS32_NOT_32R6_64R6; } let AdditionalPredicates = [NotDSP] in { def PseudoMULT : MultDivPseudo, ISA_MIPS1_NOT_32R6_64R6; def PseudoMULTu : MultDivPseudo, ISA_MIPS1_NOT_32R6_64R6; def PseudoMFHI : PseudoMFLOHI, ISA_MIPS1_NOT_32R6_64R6; def PseudoMFLO : PseudoMFLOHI, ISA_MIPS1_NOT_32R6_64R6; def PseudoMTLOHI : PseudoMTLOHI, ISA_MIPS1_NOT_32R6_64R6; def PseudoMADD : MAddSubPseudo, ISA_MIPS32_NOT_32R6_64R6; def PseudoMADDU : MAddSubPseudo, ISA_MIPS32_NOT_32R6_64R6; def PseudoMSUB : MAddSubPseudo, ISA_MIPS32_NOT_32R6_64R6; def PseudoMSUBU : MAddSubPseudo, ISA_MIPS32_NOT_32R6_64R6; } let AdditionalPredicates = [NotInMicroMips] in { def PseudoSDIV : MultDivPseudo, ISA_MIPS1_NOT_32R6_64R6; def PseudoUDIV : MultDivPseudo, ISA_MIPS1_NOT_32R6_64R6; def RDHWR : MMRel, ReadHardware, RDHWR_FM, ISA_MIPS1; // TODO: Add '0 < pos+size <= 32' constraint check to ext instruction def EXT : MMRel, StdMMR6Rel, ExtBase<"ext", GPR32Opnd, uimm5, uimm5_plus1, immZExt5, immZExt5Plus1, MipsExt>, EXT_FM<0>, ISA_MIPS32R2; def INS : MMRel, StdMMR6Rel, InsBase<"ins", GPR32Opnd, uimm5, uimm5_inssize_plus1, immZExt5, immZExt5Plus1>, EXT_FM<4>, ISA_MIPS32R2; } /// Move Control Registers From/To CPU Registers let AdditionalPredicates = [NotInMicroMips] in { def MTC0 : MTC3OP<"mtc0", COP0Opnd, GPR32Opnd, II_MTC0>, MFC3OP_FM<0x10, 4, 0>, ISA_MIPS1; def MFC0 : MFC3OP<"mfc0", GPR32Opnd, COP0Opnd, II_MFC0>, MFC3OP_FM<0x10, 0, 0>, ISA_MIPS1; def MFC2 : MFC3OP<"mfc2", GPR32Opnd, COP2Opnd, II_MFC2>, MFC3OP_FM<0x12, 0, 0>, ISA_MIPS1; def MTC2 : MTC3OP<"mtc2", COP2Opnd, GPR32Opnd, II_MTC2>, MFC3OP_FM<0x12, 4, 0>, ISA_MIPS1; } class Barrier : InstSE<(outs), (ins), asmstr, [], itin, FrmOther, asmstr>; let AdditionalPredicates = [NotInMicroMips] in { def SSNOP : MMRel, StdMMR6Rel, Barrier<"ssnop", II_SSNOP>, BARRIER_FM<1>, ISA_MIPS1; def EHB : MMRel, Barrier<"ehb", II_EHB>, BARRIER_FM<3>, ISA_MIPS1; let isCTI = 1 in def PAUSE : MMRel, StdMMR6Rel, Barrier<"pause", II_PAUSE>, BARRIER_FM<5>, ISA_MIPS32R2; } // JR_HB and JALR_HB are defined here using the new style naming // scheme because some of this code is shared with Mips32r6InstrInfo.td // and because of that it doesn't follow the naming convention of the // rest of the file. To avoid a mixture of old vs new style, the new // style was chosen. class JR_HB_DESC_BASE { dag OutOperandList = (outs); dag InOperandList = (ins GPROpnd:$rs); string AsmString = !strconcat(instr_asm, "\t$rs"); list Pattern = []; } class JALR_HB_DESC_BASE { dag OutOperandList = (outs GPROpnd:$rd); dag InOperandList = (ins GPROpnd:$rs); string AsmString = !strconcat(instr_asm, "\t$rd, $rs"); list Pattern = []; } class JR_HB_DESC : InstSE<(outs), (ins), "", [], II_JR_HB, FrmJ>, JR_HB_DESC_BASE<"jr.hb", RO> { let isBranch=1; let isIndirectBranch=1; let hasDelaySlot=1; let isTerminator=1; let isBarrier=1; bit isCTI = 1; } class JALR_HB_DESC : InstSE<(outs), (ins), "", [], II_JALR_HB, FrmJ>, JALR_HB_DESC_BASE<"jalr.hb", RO> { let isIndirectBranch=1; let hasDelaySlot=1; bit isCTI = 1; } class JR_HB_ENC : JR_HB_FM<8>; class JALR_HB_ENC : JALR_HB_FM<9>; def JR_HB : JR_HB_DESC, JR_HB_ENC, ISA_MIPS32R2_NOT_32R6_64R6; def JALR_HB : JALR_HB_DESC, JALR_HB_ENC, ISA_MIPS32; let AdditionalPredicates = [NotInMicroMips, UseIndirectJumpsHazard] in def JALRHBPseudo : JumpLinkRegPseudo; let AdditionalPredicates = [NotInMips16Mode, NotInMicroMips, UseIndirectJumpsHazard] in { def TAILCALLREGHB : TailCallReg, ISA_MIPS32_NOT_32R6_64R6; def PseudoIndirectHazardBranch : PseudoIndirectBranchBase, ISA_MIPS32R2_NOT_32R6_64R6; } class TLB : InstSE<(outs), (ins), asmstr, [], itin, FrmOther, asmstr>; let AdditionalPredicates = [NotInMicroMips] in { def TLBP : MMRel, TLB<"tlbp", II_TLBP>, COP0_TLB_FM<0x08>, ISA_MIPS1; def TLBR : MMRel, TLB<"tlbr", II_TLBR>, COP0_TLB_FM<0x01>, ISA_MIPS1; def TLBWI : MMRel, TLB<"tlbwi", II_TLBWI>, COP0_TLB_FM<0x02>, ISA_MIPS1; def TLBWR : MMRel, TLB<"tlbwr", II_TLBWR>, COP0_TLB_FM<0x06>, ISA_MIPS1; } class CacheOp : InstSE<(outs), (ins MemOpnd:$addr, uimm5:$hint), !strconcat(instr_asm, "\t$hint, $addr"), [], itin, FrmOther, instr_asm> { let DecoderMethod = "DecodeCacheOp"; } let AdditionalPredicates = [NotInMicroMips] in { def CACHE : MMRel, CacheOp<"cache", mem, II_CACHE>, CACHEOP_FM<0b101111>, INSN_MIPS3_32_NOT_32R6_64R6; def PREF : MMRel, CacheOp<"pref", mem, II_PREF>, CACHEOP_FM<0b110011>, INSN_MIPS3_32_NOT_32R6_64R6; } // FIXME: We are missing the prefx instruction. def ROL : MipsAsmPseudoInst<(outs), (ins GPR32Opnd:$rs, GPR32Opnd:$rt, GPR32Opnd:$rd), "rol\t$rs, $rt, $rd">; def ROLImm : MipsAsmPseudoInst<(outs), (ins GPR32Opnd:$rs, GPR32Opnd:$rt, simm16:$imm), "rol\t$rs, $rt, $imm">; def : MipsInstAlias<"rol $rd, $rs", (ROL GPR32Opnd:$rd, GPR32Opnd:$rd, GPR32Opnd:$rs), 0>; def : MipsInstAlias<"rol $rd, $imm", (ROLImm GPR32Opnd:$rd, GPR32Opnd:$rd, simm16:$imm), 0>; def ROR : MipsAsmPseudoInst<(outs), (ins GPR32Opnd:$rs, GPR32Opnd:$rt, GPR32Opnd:$rd), "ror\t$rs, $rt, $rd">; def RORImm : MipsAsmPseudoInst<(outs), (ins GPR32Opnd:$rs, GPR32Opnd:$rt, simm16:$imm), "ror\t$rs, $rt, $imm">; def : MipsInstAlias<"ror $rd, $rs", (ROR GPR32Opnd:$rd, GPR32Opnd:$rd, GPR32Opnd:$rs), 0>; def : MipsInstAlias<"ror $rd, $imm", (RORImm GPR32Opnd:$rd, GPR32Opnd:$rd, simm16:$imm), 0>; def DROL : MipsAsmPseudoInst<(outs), (ins GPR32Opnd:$rs, GPR32Opnd:$rt, GPR32Opnd:$rd), "drol\t$rs, $rt, $rd">, ISA_MIPS64; def DROLImm : MipsAsmPseudoInst<(outs), (ins GPR32Opnd:$rs, GPR32Opnd:$rt, simm16:$imm), "drol\t$rs, $rt, $imm">, ISA_MIPS64; def : MipsInstAlias<"drol $rd, $rs", (DROL GPR32Opnd:$rd, GPR32Opnd:$rd, GPR32Opnd:$rs), 0>, ISA_MIPS64; def : MipsInstAlias<"drol $rd, $imm", (DROLImm GPR32Opnd:$rd, GPR32Opnd:$rd, simm16:$imm), 0>, ISA_MIPS64; def DROR : MipsAsmPseudoInst<(outs), (ins GPR32Opnd:$rs, GPR32Opnd:$rt, GPR32Opnd:$rd), "dror\t$rs, $rt, $rd">, ISA_MIPS64; def DRORImm : MipsAsmPseudoInst<(outs), (ins GPR32Opnd:$rs, GPR32Opnd:$rt, simm16:$imm), "dror\t$rs, $rt, $imm">, ISA_MIPS64; def : MipsInstAlias<"dror $rd, $rs", (DROR GPR32Opnd:$rd, GPR32Opnd:$rd, GPR32Opnd:$rs), 0>, ISA_MIPS64; def : MipsInstAlias<"dror $rd, $imm", (DRORImm GPR32Opnd:$rd, GPR32Opnd:$rd, simm16:$imm), 0>, ISA_MIPS64; def ABSMacro : MipsAsmPseudoInst<(outs GPR32Opnd:$rd), (ins GPR32Opnd:$rs), "abs\t$rd, $rs">; def SEQMacro : MipsAsmPseudoInst<(outs GPR32Opnd:$rd), (ins GPR32Opnd:$rs, GPR32Opnd:$rt), "seq $rd, $rs, $rt">, NOT_ASE_CNMIPS; def : MipsInstAlias<"seq $rd, $rs", (SEQMacro GPR32Opnd:$rd, GPR32Opnd:$rd, GPR32Opnd:$rs), 0>, NOT_ASE_CNMIPS; def SEQIMacro : MipsAsmPseudoInst<(outs GPR32Opnd:$rd), (ins GPR32Opnd:$rs, simm32_relaxed:$imm), "seq $rd, $rs, $imm">, NOT_ASE_CNMIPS; def : MipsInstAlias<"seq $rd, $imm", (SEQIMacro GPR32Opnd:$rd, GPR32Opnd:$rd, simm32:$imm), 0>, NOT_ASE_CNMIPS; def MULImmMacro : MipsAsmPseudoInst<(outs), (ins GPR32Opnd:$rd, GPR32Opnd:$rs, simm32_relaxed:$imm), "mul\t$rd, $rs, $imm">, ISA_MIPS1_NOT_32R6_64R6; def MULOMacro : MipsAsmPseudoInst<(outs), (ins GPR32Opnd:$rd, GPR32Opnd:$rs, GPR32Opnd:$rt), "mulo\t$rd, $rs, $rt">, ISA_MIPS1_NOT_32R6_64R6; def MULOUMacro : MipsAsmPseudoInst<(outs), (ins GPR32Opnd:$rd, GPR32Opnd:$rs, GPR32Opnd:$rt), "mulou\t$rd, $rs, $rt">, ISA_MIPS1_NOT_32R6_64R6; // Virtualization ASE class HYPCALL_FT : InstSE<(outs), (ins uimm10:$code_), !strconcat(opstr, "\t$code_"), [], II_HYPCALL, FrmOther, opstr> { let BaseOpcode = opstr; } let AdditionalPredicates = [NotInMicroMips] in { def MFGC0 : MMRel, MFC3OP<"mfgc0", GPR32Opnd, COP0Opnd, II_MFGC0>, MFC3OP_FM<0x10, 3, 0>, ISA_MIPS32R5, ASE_VIRT; def MTGC0 : MMRel, MTC3OP<"mtgc0", COP0Opnd, GPR32Opnd, II_MTGC0>, MFC3OP_FM<0x10, 3, 2>, ISA_MIPS32R5, ASE_VIRT; def MFHGC0 : MMRel, MFC3OP<"mfhgc0", GPR32Opnd, COP0Opnd, II_MFHGC0>, MFC3OP_FM<0x10, 3, 4>, ISA_MIPS32R5, ASE_VIRT; def MTHGC0 : MMRel, MTC3OP<"mthgc0", COP0Opnd, GPR32Opnd, II_MTHGC0>, MFC3OP_FM<0x10, 3, 6>, ISA_MIPS32R5, ASE_VIRT; def TLBGINV : MMRel, TLB<"tlbginv", II_TLBGINV>, COP0_TLB_FM<0b001011>, ISA_MIPS32R5, ASE_VIRT; def TLBGINVF : MMRel, TLB<"tlbginvf", II_TLBGINVF>, COP0_TLB_FM<0b001100>, ISA_MIPS32R5, ASE_VIRT; def TLBGP : MMRel, TLB<"tlbgp", II_TLBGP>, COP0_TLB_FM<0b010000>, ISA_MIPS32R5, ASE_VIRT; def TLBGR : MMRel, TLB<"tlbgr", II_TLBGR>, COP0_TLB_FM<0b001001>, ISA_MIPS32R5, ASE_VIRT; def TLBGWI : MMRel, TLB<"tlbgwi", II_TLBGWI>, COP0_TLB_FM<0b001010>, ISA_MIPS32R5, ASE_VIRT; def TLBGWR : MMRel, TLB<"tlbgwr", II_TLBGWR>, COP0_TLB_FM<0b001110>, ISA_MIPS32R5, ASE_VIRT; def HYPCALL : MMRel, HYPCALL_FT<"hypcall">, HYPCALL_FM<0b101000>, ISA_MIPS32R5, ASE_VIRT; } //===----------------------------------------------------------------------===// // Instruction aliases //===----------------------------------------------------------------------===// multiclass OneOrTwoOperandMacroImmediateAlias { def : MipsInstAlias; def : MipsInstAlias; } let AdditionalPredicates = [NotInMicroMips] in { def : MipsInstAlias<"move $dst, $src", (OR GPR32Opnd:$dst, GPR32Opnd:$src, ZERO), 1>, GPR_32, ISA_MIPS1; def : MipsInstAlias<"move $dst, $src", (ADDu GPR32Opnd:$dst, GPR32Opnd:$src, ZERO), 1>, GPR_32, ISA_MIPS1; def : MipsInstAlias<"bal $offset", (BGEZAL ZERO, brtarget:$offset), 1>, ISA_MIPS1_NOT_32R6_64R6; def : MipsInstAlias<"j $rs", (JR GPR32Opnd:$rs), 0>, ISA_MIPS1; def : MipsInstAlias<"jalr $rs", (JALR RA, GPR32Opnd:$rs), 0>; def : MipsInstAlias<"jalr.hb $rs", (JALR_HB RA, GPR32Opnd:$rs), 1>, ISA_MIPS32; def : MipsInstAlias<"neg $rt, $rs", (SUB GPR32Opnd:$rt, ZERO, GPR32Opnd:$rs), 1>, ISA_MIPS1; def : MipsInstAlias<"neg $rt", (SUB GPR32Opnd:$rt, ZERO, GPR32Opnd:$rt), 1>, ISA_MIPS1; def : MipsInstAlias<"negu $rt, $rs", (SUBu GPR32Opnd:$rt, ZERO, GPR32Opnd:$rs), 1>, ISA_MIPS1; def : MipsInstAlias<"negu $rt", (SUBu GPR32Opnd:$rt, ZERO, GPR32Opnd:$rt), 1>, ISA_MIPS1; def : MipsInstAlias< "sgt $rd, $rs, $rt", (SLT GPR32Opnd:$rd, GPR32Opnd:$rt, GPR32Opnd:$rs), 0>, ISA_MIPS1; def : MipsInstAlias< "sgt $rs, $rt", (SLT GPR32Opnd:$rs, GPR32Opnd:$rt, GPR32Opnd:$rs), 0>, ISA_MIPS1; def : MipsInstAlias< "sgtu $rd, $rs, $rt", (SLTu GPR32Opnd:$rd, GPR32Opnd:$rt, GPR32Opnd:$rs), 0>, ISA_MIPS1; def : MipsInstAlias< "sgtu $$rs, $rt", (SLTu GPR32Opnd:$rs, GPR32Opnd:$rt, GPR32Opnd:$rs), 0>, ISA_MIPS1; def : MipsInstAlias< "not $rt, $rs", (NOR GPR32Opnd:$rt, GPR32Opnd:$rs, ZERO), 0>, ISA_MIPS1; def : MipsInstAlias< "not $rt", (NOR GPR32Opnd:$rt, GPR32Opnd:$rt, ZERO), 0>, ISA_MIPS1; def : MipsInstAlias<"nop", (SLL ZERO, ZERO, 0), 1>, ISA_MIPS1; defm : OneOrTwoOperandMacroImmediateAlias<"add", ADDi>, ISA_MIPS1_NOT_32R6_64R6; defm : OneOrTwoOperandMacroImmediateAlias<"addu", ADDiu>, ISA_MIPS1; defm : OneOrTwoOperandMacroImmediateAlias<"and", ANDi>, ISA_MIPS1, GPR_32; defm : OneOrTwoOperandMacroImmediateAlias<"or", ORi>, ISA_MIPS1, GPR_32; defm : OneOrTwoOperandMacroImmediateAlias<"xor", XORi>, ISA_MIPS1, GPR_32; defm : OneOrTwoOperandMacroImmediateAlias<"slt", SLTi>, ISA_MIPS1, GPR_32; defm : OneOrTwoOperandMacroImmediateAlias<"sltu", SLTiu>, ISA_MIPS1, GPR_32; def : MipsInstAlias<"mfgc0 $rt, $rd", (MFGC0 GPR32Opnd:$rt, COP0Opnd:$rd, 0), 0>, ISA_MIPS32R5, ASE_VIRT; def : MipsInstAlias<"mtgc0 $rt, $rd", (MTGC0 COP0Opnd:$rd, GPR32Opnd:$rt, 0), 0>, ISA_MIPS32R5, ASE_VIRT; def : MipsInstAlias<"mfhgc0 $rt, $rd", (MFHGC0 GPR32Opnd:$rt, COP0Opnd:$rd, 0), 0>, ISA_MIPS32R5, ASE_VIRT; def : MipsInstAlias<"mthgc0 $rt, $rd", (MTHGC0 COP0Opnd:$rd, GPR32Opnd:$rt, 0), 0>, ISA_MIPS32R5, ASE_VIRT; def : MipsInstAlias<"mfc0 $rt, $rd", (MFC0 GPR32Opnd:$rt, COP0Opnd:$rd, 0), 0>, ISA_MIPS1; def : MipsInstAlias<"mtc0 $rt, $rd", (MTC0 COP0Opnd:$rd, GPR32Opnd:$rt, 0), 0>, ISA_MIPS1; def : MipsInstAlias<"mfc2 $rt, $rd", (MFC2 GPR32Opnd:$rt, COP2Opnd:$rd, 0), 0>, ISA_MIPS1; def : MipsInstAlias<"mtc2 $rt, $rd", (MTC2 COP2Opnd:$rd, GPR32Opnd:$rt, 0), 0>, ISA_MIPS1; def : MipsInstAlias<"b $offset", (BEQ ZERO, ZERO, brtarget:$offset), 0>, ISA_MIPS1; def : MipsInstAlias<"bnez $rs,$offset", (BNE GPR32Opnd:$rs, ZERO, brtarget:$offset), 0>, ISA_MIPS1; def : MipsInstAlias<"bnezl $rs,$offset", (BNEL GPR32Opnd:$rs, ZERO, brtarget:$offset), 0>, ISA_MIPS2; def : MipsInstAlias<"beqz $rs,$offset", (BEQ GPR32Opnd:$rs, ZERO, brtarget:$offset), 0>, ISA_MIPS1; def : MipsInstAlias<"beqzl $rs,$offset", (BEQL GPR32Opnd:$rs, ZERO, brtarget:$offset), 0>, ISA_MIPS2; def : MipsInstAlias<"syscall", (SYSCALL 0), 1>, ISA_MIPS1; def : MipsInstAlias<"break", (BREAK 0, 0), 1>, ISA_MIPS1; def : MipsInstAlias<"break $imm", (BREAK uimm10:$imm, 0), 1>, ISA_MIPS1; def : MipsInstAlias<"ei", (EI ZERO), 1>, ISA_MIPS32R2; def : MipsInstAlias<"di", (DI ZERO), 1>, ISA_MIPS32R2; def : MipsInstAlias<"teq $rs, $rt", (TEQ GPR32Opnd:$rs, GPR32Opnd:$rt, 0), 1>, ISA_MIPS2; def : MipsInstAlias<"tge $rs, $rt", (TGE GPR32Opnd:$rs, GPR32Opnd:$rt, 0), 1>, ISA_MIPS2; def : MipsInstAlias<"tgeu $rs, $rt", (TGEU GPR32Opnd:$rs, GPR32Opnd:$rt, 0), 1>, ISA_MIPS2; def : MipsInstAlias<"tlt $rs, $rt", (TLT GPR32Opnd:$rs, GPR32Opnd:$rt, 0), 1>, ISA_MIPS2; def : MipsInstAlias<"tltu $rs, $rt", (TLTU GPR32Opnd:$rs, GPR32Opnd:$rt, 0), 1>, ISA_MIPS2; def : MipsInstAlias<"tne $rs, $rt", (TNE GPR32Opnd:$rs, GPR32Opnd:$rt, 0), 1>, ISA_MIPS2; def : MipsInstAlias<"rdhwr $rt, $rs", (RDHWR GPR32Opnd:$rt, HWRegsOpnd:$rs, 0), 1>, ISA_MIPS1; } def : MipsInstAlias<"sub, $rd, $rs, $imm", (ADDi GPR32Opnd:$rd, GPR32Opnd:$rs, InvertedImOperand:$imm), 0>, ISA_MIPS1_NOT_32R6_64R6; def : MipsInstAlias<"sub $rs, $imm", (ADDi GPR32Opnd:$rs, GPR32Opnd:$rs, InvertedImOperand:$imm), 0>, ISA_MIPS1_NOT_32R6_64R6; def : MipsInstAlias<"subu, $rd, $rs, $imm", (ADDiu GPR32Opnd:$rd, GPR32Opnd:$rs, InvertedImOperand:$imm), 0>; def : MipsInstAlias<"subu $rs, $imm", (ADDiu GPR32Opnd:$rs, GPR32Opnd:$rs, InvertedImOperand:$imm), 0>; let AdditionalPredicates = [NotInMicroMips] in { def : MipsInstAlias<"sll $rd, $rt, $rs", (SLLV GPR32Opnd:$rd, GPR32Opnd:$rt, GPR32Opnd:$rs), 0>; def : MipsInstAlias<"sra $rd, $rt, $rs", (SRAV GPR32Opnd:$rd, GPR32Opnd:$rt, GPR32Opnd:$rs), 0>; def : MipsInstAlias<"srl $rd, $rt, $rs", (SRLV GPR32Opnd:$rd, GPR32Opnd:$rt, GPR32Opnd:$rs), 0>; def : MipsInstAlias<"sll $rd, $rt", (SLLV GPR32Opnd:$rd, GPR32Opnd:$rd, GPR32Opnd:$rt), 0>; def : MipsInstAlias<"sra $rd, $rt", (SRAV GPR32Opnd:$rd, GPR32Opnd:$rd, GPR32Opnd:$rt), 0>; def : MipsInstAlias<"srl $rd, $rt", (SRLV GPR32Opnd:$rd, GPR32Opnd:$rd, GPR32Opnd:$rt), 0>; def : MipsInstAlias<"seh $rd", (SEH GPR32Opnd:$rd, GPR32Opnd:$rd), 0>, ISA_MIPS32R2; def : MipsInstAlias<"seb $rd", (SEB GPR32Opnd:$rd, GPR32Opnd:$rd), 0>, ISA_MIPS32R2; } def : MipsInstAlias<"sdbbp", (SDBBP 0)>, ISA_MIPS32_NOT_32R6_64R6; let AdditionalPredicates = [NotInMicroMips] in def : MipsInstAlias<"sync", (SYNC 0), 1>, ISA_MIPS2; def : MipsInstAlias<"mulo $rs, $rt", (MULOMacro GPR32Opnd:$rs, GPR32Opnd:$rs, GPR32Opnd:$rt), 0>, ISA_MIPS1_NOT_32R6_64R6; def : MipsInstAlias<"mulou $rs, $rt", (MULOUMacro GPR32Opnd:$rs, GPR32Opnd:$rs, GPR32Opnd:$rt), 0>, ISA_MIPS1_NOT_32R6_64R6; let AdditionalPredicates = [NotInMicroMips] in def : MipsInstAlias<"hypcall", (HYPCALL 0), 1>, ISA_MIPS32R5, ASE_VIRT; //===----------------------------------------------------------------------===// // Assembler Pseudo Instructions //===----------------------------------------------------------------------===// // We use uimm32_coerced to accept a 33 bit signed number that is rendered into // a 32 bit number. class LoadImmediate32 : MipsAsmPseudoInst<(outs RO:$rt), (ins Od:$imm32), !strconcat(instr_asm, "\t$rt, $imm32")> ; def LoadImm32 : LoadImmediate32<"li", uimm32_coerced, GPR32Opnd>; class LoadAddressFromReg32 : MipsAsmPseudoInst<(outs RO:$rt), (ins MemOpnd:$addr), !strconcat(instr_asm, "\t$rt, $addr")> ; def LoadAddrReg32 : LoadAddressFromReg32<"la", mem, GPR32Opnd>; class LoadAddressFromImm32 : MipsAsmPseudoInst<(outs RO:$rt), (ins Od:$imm32), !strconcat(instr_asm, "\t$rt, $imm32")> ; def LoadAddrImm32 : LoadAddressFromImm32<"la", i32imm, GPR32Opnd>; def JalTwoReg : MipsAsmPseudoInst<(outs GPR32Opnd:$rd), (ins GPR32Opnd:$rs), "jal\t$rd, $rs"> ; def JalOneReg : MipsAsmPseudoInst<(outs), (ins GPR32Opnd:$rs), "jal\t$rs"> ; class NORIMM_DESC_BASE : MipsAsmPseudoInst<(outs RO:$rs), (ins RO:$rt, Imm:$imm), "nor\t$rs, $rt, $imm">; def NORImm : NORIMM_DESC_BASE, GPR_32; def : MipsInstAlias<"nor\t$rs, $imm", (NORImm GPR32Opnd:$rs, GPR32Opnd:$rs, simm32_relaxed:$imm)>, GPR_32; let hasDelaySlot = 1, isCTI = 1 in { def BneImm : MipsAsmPseudoInst<(outs GPR32Opnd:$rt), (ins imm64:$imm64, brtarget:$offset), "bne\t$rt, $imm64, $offset">; def BeqImm : MipsAsmPseudoInst<(outs GPR32Opnd:$rt), (ins imm64:$imm64, brtarget:$offset), "beq\t$rt, $imm64, $offset">; class CondBranchPseudo : MipsAsmPseudoInst<(outs), (ins GPR32Opnd:$rs, GPR32Opnd:$rt, brtarget:$offset), !strconcat(instr_asm, "\t$rs, $rt, $offset")>; } def BLT : CondBranchPseudo<"blt">; def BLE : CondBranchPseudo<"ble">; def BGE : CondBranchPseudo<"bge">; def BGT : CondBranchPseudo<"bgt">; def BLTU : CondBranchPseudo<"bltu">; def BLEU : CondBranchPseudo<"bleu">; def BGEU : CondBranchPseudo<"bgeu">; def BGTU : CondBranchPseudo<"bgtu">; def BLTL : CondBranchPseudo<"bltl">, ISA_MIPS2_NOT_32R6_64R6; def BLEL : CondBranchPseudo<"blel">, ISA_MIPS2_NOT_32R6_64R6; def BGEL : CondBranchPseudo<"bgel">, ISA_MIPS2_NOT_32R6_64R6; def BGTL : CondBranchPseudo<"bgtl">, ISA_MIPS2_NOT_32R6_64R6; def BLTUL: CondBranchPseudo<"bltul">, ISA_MIPS2_NOT_32R6_64R6; def BLEUL: CondBranchPseudo<"bleul">, ISA_MIPS2_NOT_32R6_64R6; def BGEUL: CondBranchPseudo<"bgeul">, ISA_MIPS2_NOT_32R6_64R6; def BGTUL: CondBranchPseudo<"bgtul">, ISA_MIPS2_NOT_32R6_64R6; let isCTI = 1 in class CondBranchImmPseudo : MipsAsmPseudoInst<(outs), (ins GPR32Opnd:$rs, imm64:$imm, brtarget:$offset), !strconcat(instr_asm, "\t$rs, $imm, $offset")>; def BEQLImmMacro : CondBranchImmPseudo<"beql">, ISA_MIPS2_NOT_32R6_64R6; def BNELImmMacro : CondBranchImmPseudo<"bnel">, ISA_MIPS2_NOT_32R6_64R6; def BLTImmMacro : CondBranchImmPseudo<"blt">; def BLEImmMacro : CondBranchImmPseudo<"ble">; def BGEImmMacro : CondBranchImmPseudo<"bge">; def BGTImmMacro : CondBranchImmPseudo<"bgt">; def BLTUImmMacro : CondBranchImmPseudo<"bltu">; def BLEUImmMacro : CondBranchImmPseudo<"bleu">; def BGEUImmMacro : CondBranchImmPseudo<"bgeu">; def BGTUImmMacro : CondBranchImmPseudo<"bgtu">; def BLTLImmMacro : CondBranchImmPseudo<"bltl">, ISA_MIPS2_NOT_32R6_64R6; def BLELImmMacro : CondBranchImmPseudo<"blel">, ISA_MIPS2_NOT_32R6_64R6; def BGELImmMacro : CondBranchImmPseudo<"bgel">, ISA_MIPS2_NOT_32R6_64R6; def BGTLImmMacro : CondBranchImmPseudo<"bgtl">, ISA_MIPS2_NOT_32R6_64R6; def BLTULImmMacro : CondBranchImmPseudo<"bltul">, ISA_MIPS2_NOT_32R6_64R6; def BLEULImmMacro : CondBranchImmPseudo<"bleul">, ISA_MIPS2_NOT_32R6_64R6; def BGEULImmMacro : CondBranchImmPseudo<"bgeul">, ISA_MIPS2_NOT_32R6_64R6; def BGTULImmMacro : CondBranchImmPseudo<"bgtul">, ISA_MIPS2_NOT_32R6_64R6; // FIXME: Predicates are removed because instructions are matched regardless of // predicates, because PredicateControl was not in the hierarchy. This was // done to emit more precise error message from expansion function. // Once the tablegen-erated errors are made better, this needs to be fixed and // predicates needs to be restored. def SDivMacro : MipsAsmPseudoInst<(outs GPR32NonZeroOpnd:$rd), (ins GPR32Opnd:$rs, GPR32Opnd:$rt), "div\t$rd, $rs, $rt">, ISA_MIPS1_NOT_32R6_64R6; def SDivIMacro : MipsAsmPseudoInst<(outs GPR32Opnd:$rd), (ins GPR32Opnd:$rs, simm32:$imm), "div\t$rd, $rs, $imm">, ISA_MIPS1_NOT_32R6_64R6; def UDivMacro : MipsAsmPseudoInst<(outs GPR32Opnd:$rd), (ins GPR32Opnd:$rs, GPR32Opnd:$rt), "divu\t$rd, $rs, $rt">, ISA_MIPS1_NOT_32R6_64R6; def UDivIMacro : MipsAsmPseudoInst<(outs GPR32Opnd:$rd), (ins GPR32Opnd:$rs, simm32:$imm), "divu\t$rd, $rs, $imm">, ISA_MIPS1_NOT_32R6_64R6; def : MipsInstAlias<"div $rs, $rt", (SDIV GPR32ZeroOpnd:$rs, GPR32Opnd:$rt), 0>, ISA_MIPS1_NOT_32R6_64R6; def : MipsInstAlias<"div $rs, $rt", (SDivMacro GPR32NonZeroOpnd:$rs, GPR32NonZeroOpnd:$rs, GPR32Opnd:$rt), 0>, ISA_MIPS1_NOT_32R6_64R6; def : MipsInstAlias<"div $rd, $imm", (SDivIMacro GPR32Opnd:$rd, GPR32Opnd:$rd, simm32:$imm), 0>, ISA_MIPS1_NOT_32R6_64R6; def : MipsInstAlias<"divu $rt, $rs", (UDIV GPR32ZeroOpnd:$rt, GPR32Opnd:$rs), 0>, ISA_MIPS1_NOT_32R6_64R6; def : MipsInstAlias<"divu $rt, $rs", (UDivMacro GPR32NonZeroOpnd:$rt, GPR32NonZeroOpnd:$rt, GPR32Opnd:$rs), 0>, ISA_MIPS1_NOT_32R6_64R6; def : MipsInstAlias<"divu $rd, $imm", (UDivIMacro GPR32Opnd:$rd, GPR32Opnd:$rd, simm32:$imm), 0>, ISA_MIPS1_NOT_32R6_64R6; def SRemMacro : MipsAsmPseudoInst<(outs GPR32Opnd:$rd), (ins GPR32Opnd:$rs, GPR32Opnd:$rt), "rem\t$rd, $rs, $rt">, ISA_MIPS1_NOT_32R6_64R6; def SRemIMacro : MipsAsmPseudoInst<(outs GPR32Opnd:$rd), (ins GPR32Opnd:$rs, simm32_relaxed:$imm), "rem\t$rd, $rs, $imm">, ISA_MIPS1_NOT_32R6_64R6; def URemMacro : MipsAsmPseudoInst<(outs GPR32Opnd:$rd), (ins GPR32Opnd:$rs, GPR32Opnd:$rt), "remu\t$rd, $rs, $rt">, ISA_MIPS1_NOT_32R6_64R6; def URemIMacro : MipsAsmPseudoInst<(outs GPR32Opnd:$rd), (ins GPR32Opnd:$rs, simm32_relaxed:$imm), "remu\t$rd, $rs, $imm">, ISA_MIPS1_NOT_32R6_64R6; def : MipsInstAlias<"rem $rt, $rs", (SRemMacro GPR32Opnd:$rt, GPR32Opnd:$rt, GPR32Opnd:$rs), 0>, ISA_MIPS1_NOT_32R6_64R6; def : MipsInstAlias<"rem $rd, $imm", (SRemIMacro GPR32Opnd:$rd, GPR32Opnd:$rd, simm32_relaxed:$imm), 0>, ISA_MIPS1_NOT_32R6_64R6; def : MipsInstAlias<"remu $rt, $rs", (URemMacro GPR32Opnd:$rt, GPR32Opnd:$rt, GPR32Opnd:$rs), 0>, ISA_MIPS1_NOT_32R6_64R6; def : MipsInstAlias<"remu $rd, $imm", (URemIMacro GPR32Opnd:$rd, GPR32Opnd:$rd, simm32_relaxed:$imm), 0>, ISA_MIPS1_NOT_32R6_64R6; def Ulh : MipsAsmPseudoInst<(outs GPR32Opnd:$rt), (ins mem:$addr), "ulh\t$rt, $addr">; //, ISA_MIPS1_NOT_32R6_64R6; def Ulhu : MipsAsmPseudoInst<(outs GPR32Opnd:$rt), (ins mem:$addr), "ulhu\t$rt, $addr">; //, ISA_MIPS1_NOT_32R6_64R6; def Ulw : MipsAsmPseudoInst<(outs GPR32Opnd:$rt), (ins mem:$addr), "ulw\t$rt, $addr">; //, ISA_MIPS1_NOT_32R6_64R6; def Ush : MipsAsmPseudoInst<(outs GPR32Opnd:$rt), (ins mem:$addr), "ush\t$rt, $addr">; //, ISA_MIPS1_NOT_32R6_64R6; def Usw : MipsAsmPseudoInst<(outs GPR32Opnd:$rt), (ins mem:$addr), "usw\t$rt, $addr">; //, ISA_MIPS1_NOT_32R6_64R6; def LDMacro : MipsAsmPseudoInst<(outs GPR32Opnd:$rt), (ins mem_simm16:$addr), "ld $rt, $addr">, ISA_MIPS1_NOT_MIPS3; def SDMacro : MipsAsmPseudoInst<(outs GPR32Opnd:$rt), (ins mem_simm16:$addr), "sd $rt, $addr">, ISA_MIPS1_NOT_MIPS3; //===----------------------------------------------------------------------===// // Arbitrary patterns that map to one or more instructions //===----------------------------------------------------------------------===// // Load/store pattern templates. class LoadRegImmPat : MipsPat<(ValTy (Node addrRegImm:$a)), (LoadInst addrRegImm:$a)>; class StoreRegImmPat : MipsPat<(store ValTy:$v, addrRegImm:$a), (StoreInst ValTy:$v, addrRegImm:$a)>; // Materialize constants. multiclass MaterializeImms { // Constant synthesis previously relied on the ordering of the patterns below. // By making the predicates they use non-overlapping, the patterns were // reordered so that the effect of the newly introduced predicates can be // observed. // Arbitrary immediates def : MipsPat<(VT LUiORiPred:$imm), (ORiOp (LUiOp (HI16 imm:$imm)), (LO16 imm:$imm))>; // Bits 32-16 set, sign/zero extended. def : MipsPat<(VT LUiPred:$imm), (LUiOp (HI16 imm:$imm))>; // Small immediates def : MipsPat<(VT ORiPred:$imm), (ORiOp ZEROReg, imm:$imm)>; def : MipsPat<(VT immSExt16:$imm), (ADDiuOp ZEROReg, imm:$imm)>; } let AdditionalPredicates = [NotInMicroMips] in defm : MaterializeImms, ISA_MIPS1; // Carry MipsPatterns let AdditionalPredicates = [NotInMicroMips] in { def : MipsPat<(subc GPR32:$lhs, GPR32:$rhs), (SUBu GPR32:$lhs, GPR32:$rhs)>, ISA_MIPS1; } def : MipsPat<(addc GPR32:$lhs, GPR32:$rhs), (ADDu GPR32:$lhs, GPR32:$rhs)>, ISA_MIPS1, ASE_NOT_DSP; def : MipsPat<(addc GPR32:$src, immSExt16:$imm), (ADDiu GPR32:$src, imm:$imm)>, ISA_MIPS1, ASE_NOT_DSP; // Support multiplication for pre-Mips32 targets that don't have // the MUL instruction. def : MipsPat<(mul GPR32:$lhs, GPR32:$rhs), (PseudoMFLO (PseudoMULT GPR32:$lhs, GPR32:$rhs))>, ISA_MIPS1_NOT_32R6_64R6; // SYNC def : MipsPat<(MipsSync (i32 immz)), (SYNC 0)>, ISA_MIPS2; // Call def : MipsPat<(MipsJmpLink (i32 texternalsym:$dst)), (JAL texternalsym:$dst)>, ISA_MIPS1; //def : MipsPat<(MipsJmpLink GPR32:$dst), // (JALR GPR32:$dst)>; // Tail call let AdditionalPredicates = [NotInMicroMips] in { def : MipsPat<(MipsTailCall (iPTR tglobaladdr:$dst)), (TAILCALL tglobaladdr:$dst)>, ISA_MIPS1; def : MipsPat<(MipsTailCall (iPTR texternalsym:$dst)), (TAILCALL texternalsym:$dst)>, ISA_MIPS1; } // hi/lo relocs multiclass MipsHiLoRelocs { def : MipsPat<(MipsHi tglobaladdr:$in), (Lui tglobaladdr:$in)>; def : MipsPat<(MipsHi tblockaddress:$in), (Lui tblockaddress:$in)>; def : MipsPat<(MipsHi tjumptable:$in), (Lui tjumptable:$in)>; def : MipsPat<(MipsHi tconstpool:$in), (Lui tconstpool:$in)>; def : MipsPat<(MipsHi texternalsym:$in), (Lui texternalsym:$in)>; def : MipsPat<(MipsLo tglobaladdr:$in), (Addiu ZeroReg, tglobaladdr:$in)>; def : MipsPat<(MipsLo tblockaddress:$in), (Addiu ZeroReg, tblockaddress:$in)>; def : MipsPat<(MipsLo tjumptable:$in), (Addiu ZeroReg, tjumptable:$in)>; def : MipsPat<(MipsLo tconstpool:$in), (Addiu ZeroReg, tconstpool:$in)>; def : MipsPat<(MipsLo tglobaltlsaddr:$in), (Addiu ZeroReg, tglobaltlsaddr:$in)>; def : MipsPat<(MipsLo texternalsym:$in), (Addiu ZeroReg, texternalsym:$in)>; def : MipsPat<(add GPROpnd:$hi, (MipsLo tglobaladdr:$lo)), (Addiu GPROpnd:$hi, tglobaladdr:$lo)>; def : MipsPat<(add GPROpnd:$hi, (MipsLo tblockaddress:$lo)), (Addiu GPROpnd:$hi, tblockaddress:$lo)>; def : MipsPat<(add GPROpnd:$hi, (MipsLo tjumptable:$lo)), (Addiu GPROpnd:$hi, tjumptable:$lo)>; def : MipsPat<(add GPROpnd:$hi, (MipsLo tconstpool:$lo)), (Addiu GPROpnd:$hi, tconstpool:$lo)>; def : MipsPat<(add GPROpnd:$hi, (MipsLo tglobaltlsaddr:$lo)), (Addiu GPROpnd:$hi, tglobaltlsaddr:$lo)>; } // wrapper_pic class WrapperPat: MipsPat<(MipsWrapper RC:$gp, node:$in), (ADDiuOp RC:$gp, node:$in)>; let AdditionalPredicates = [NotInMicroMips] in { defm : MipsHiLoRelocs, ISA_MIPS1; def : MipsPat<(MipsGotHi tglobaladdr:$in), (LUi tglobaladdr:$in)>, ISA_MIPS1; def : MipsPat<(MipsGotHi texternalsym:$in), (LUi texternalsym:$in)>, ISA_MIPS1; def : MipsPat<(MipsTlsHi tglobaltlsaddr:$in), (LUi tglobaltlsaddr:$in)>, ISA_MIPS1; // gp_rel relocs def : MipsPat<(add GPR32:$gp, (MipsGPRel tglobaladdr:$in)), (ADDiu GPR32:$gp, tglobaladdr:$in)>, ISA_MIPS1, ABI_NOT_N64; def : MipsPat<(add GPR32:$gp, (MipsGPRel tconstpool:$in)), (ADDiu GPR32:$gp, tconstpool:$in)>, ISA_MIPS1, ABI_NOT_N64; def : WrapperPat, ISA_MIPS1; def : WrapperPat, ISA_MIPS1; def : WrapperPat, ISA_MIPS1; def : WrapperPat, ISA_MIPS1; def : WrapperPat, ISA_MIPS1; def : WrapperPat, ISA_MIPS1; // Mips does not have "not", so we expand our way def : MipsPat<(not GPR32:$in), (NOR GPR32Opnd:$in, ZERO)>, ISA_MIPS1; } // extended loads let AdditionalPredicates = [NotInMicroMips] in { def : MipsPat<(i32 (extloadi1 addr:$src)), (LBu addr:$src)>, ISA_MIPS1; def : MipsPat<(i32 (extloadi8 addr:$src)), (LBu addr:$src)>, ISA_MIPS1; def : MipsPat<(i32 (extloadi16 addr:$src)), (LHu addr:$src)>, ISA_MIPS1; // peepholes def : MipsPat<(store (i32 0), addr:$dst), (SW ZERO, addr:$dst)>, ISA_MIPS1; } // brcond patterns multiclass BrcondPats { def : MipsPat<(brcond (i32 (setne RC:$lhs, 0)), bb:$dst), (BNEOp RC:$lhs, ZEROReg, bb:$dst)>; def : MipsPat<(brcond (i32 (seteq RC:$lhs, 0)), bb:$dst), (BEQOp RC:$lhs, ZEROReg, bb:$dst)>; def : MipsPat<(brcond (i32 (setge RC:$lhs, RC:$rhs)), bb:$dst), (BEQOp1 (SLTOp RC:$lhs, RC:$rhs), ZERO, bb:$dst)>; def : MipsPat<(brcond (i32 (setuge RC:$lhs, RC:$rhs)), bb:$dst), (BEQOp1 (SLTuOp RC:$lhs, RC:$rhs), ZERO, bb:$dst)>; def : MipsPat<(brcond (i32 (setge RC:$lhs, immSExt16:$rhs)), bb:$dst), (BEQOp1 (SLTiOp RC:$lhs, immSExt16:$rhs), ZERO, bb:$dst)>; def : MipsPat<(brcond (i32 (setuge RC:$lhs, immSExt16:$rhs)), bb:$dst), (BEQOp1 (SLTiuOp RC:$lhs, immSExt16:$rhs), ZERO, bb:$dst)>; def : MipsPat<(brcond (i32 (setgt RC:$lhs, immSExt16Plus1:$rhs)), bb:$dst), (BEQOp1 (SLTiOp RC:$lhs, (Plus1 imm:$rhs)), ZERO, bb:$dst)>; def : MipsPat<(brcond (i32 (setugt RC:$lhs, immSExt16Plus1:$rhs)), bb:$dst), (BEQOp1 (SLTiuOp RC:$lhs, (Plus1 imm:$rhs)), ZERO, bb:$dst)>; def : MipsPat<(brcond (i32 (setle RC:$lhs, RC:$rhs)), bb:$dst), (BEQOp1 (SLTOp RC:$rhs, RC:$lhs), ZERO, bb:$dst)>; def : MipsPat<(brcond (i32 (setule RC:$lhs, RC:$rhs)), bb:$dst), (BEQOp1 (SLTuOp RC:$rhs, RC:$lhs), ZERO, bb:$dst)>; def : MipsPat<(brcond RC:$cond, bb:$dst), (BNEOp RC:$cond, ZEROReg, bb:$dst)>; } let AdditionalPredicates = [NotInMicroMips] in { defm : BrcondPats, ISA_MIPS1; def : MipsPat<(brcond (i32 (setlt i32:$lhs, 1)), bb:$dst), (BLEZ i32:$lhs, bb:$dst)>, ISA_MIPS1; def : MipsPat<(brcond (i32 (setgt i32:$lhs, -1)), bb:$dst), (BGEZ i32:$lhs, bb:$dst)>, ISA_MIPS1; } // setcc patterns multiclass SeteqPats { def : MipsPat<(seteq RC:$lhs, 0), (SLTiuOp RC:$lhs, 1)>; def : MipsPat<(setne RC:$lhs, 0), (SLTuOp ZEROReg, RC:$lhs)>; def : MipsPat<(seteq RC:$lhs, RC:$rhs), (SLTiuOp (XOROp RC:$lhs, RC:$rhs), 1)>; def : MipsPat<(setne RC:$lhs, RC:$rhs), (SLTuOp ZEROReg, (XOROp RC:$lhs, RC:$rhs))>; } multiclass SetlePats { def : MipsPat<(setle RC:$lhs, RC:$rhs), (XORiOp (SLTOp RC:$rhs, RC:$lhs), 1)>; def : MipsPat<(setule RC:$lhs, RC:$rhs), (XORiOp (SLTuOp RC:$rhs, RC:$lhs), 1)>; } multiclass SetgtPats { def : MipsPat<(setgt RC:$lhs, RC:$rhs), (SLTOp RC:$rhs, RC:$lhs)>; def : MipsPat<(setugt RC:$lhs, RC:$rhs), (SLTuOp RC:$rhs, RC:$lhs)>; } multiclass SetgePats { def : MipsPat<(setge RC:$lhs, RC:$rhs), (XORiOp (SLTOp RC:$lhs, RC:$rhs), 1)>; def : MipsPat<(setuge RC:$lhs, RC:$rhs), (XORiOp (SLTuOp RC:$lhs, RC:$rhs), 1)>; } multiclass SetgeImmPats { def : MipsPat<(setge RC:$lhs, immSExt16:$rhs), (XORiOp (SLTiOp RC:$lhs, immSExt16:$rhs), 1)>; def : MipsPat<(setuge RC:$lhs, immSExt16:$rhs), (XORiOp (SLTiuOp RC:$lhs, immSExt16:$rhs), 1)>; } let AdditionalPredicates = [NotInMicroMips] in { defm : SeteqPats, ISA_MIPS1; defm : SetlePats, ISA_MIPS1; defm : SetgtPats, ISA_MIPS1; defm : SetgePats, ISA_MIPS1; defm : SetgeImmPats, ISA_MIPS1; // bswap pattern def : MipsPat<(bswap GPR32:$rt), (ROTR (WSBH GPR32:$rt), 16)>, ISA_MIPS32R2; } // Load halfword/word patterns. let AdditionalPredicates = [NotInMicroMips] in { let AddedComplexity = 40 in { def : LoadRegImmPat, ISA_MIPS1; def : LoadRegImmPat, ISA_MIPS1; def : LoadRegImmPat, ISA_MIPS1; def : LoadRegImmPat, ISA_MIPS1; def : LoadRegImmPat, ISA_MIPS1; } // Atomic load patterns. def : MipsPat<(atomic_load_8 addr:$a), (LB addr:$a)>, ISA_MIPS1; def : MipsPat<(atomic_load_16 addr:$a), (LH addr:$a)>, ISA_MIPS1; def : MipsPat<(atomic_load_32 addr:$a), (LW addr:$a)>, ISA_MIPS1; // Atomic store patterns. def : MipsPat<(atomic_store_8 addr:$a, GPR32:$v), (SB GPR32:$v, addr:$a)>, ISA_MIPS1; def : MipsPat<(atomic_store_16 addr:$a, GPR32:$v), (SH GPR32:$v, addr:$a)>, ISA_MIPS1; def : MipsPat<(atomic_store_32 addr:$a, GPR32:$v), (SW GPR32:$v, addr:$a)>, ISA_MIPS1; } //===----------------------------------------------------------------------===// // Floating Point Support //===----------------------------------------------------------------------===// include "MipsInstrFPU.td" include "Mips64InstrInfo.td" include "MipsCondMov.td" include "Mips32r6InstrInfo.td" include "Mips64r6InstrInfo.td" // // Mips16 include "Mips16InstrFormats.td" include "Mips16InstrInfo.td" // DSP include "MipsDSPInstrFormats.td" include "MipsDSPInstrInfo.td" // MSA include "MipsMSAInstrFormats.td" include "MipsMSAInstrInfo.td" // EVA include "MipsEVAInstrFormats.td" include "MipsEVAInstrInfo.td" // MT include "MipsMTInstrFormats.td" include "MipsMTInstrInfo.td" // Micromips include "MicroMipsInstrFormats.td" include "MicroMipsInstrInfo.td" include "MicroMipsInstrFPU.td" // Micromips r6 include "MicroMips32r6InstrFormats.td" include "MicroMips32r6InstrInfo.td" // Micromips DSP include "MicroMipsDSPInstrFormats.td" include "MicroMipsDSPInstrInfo.td" Index: vendor/llvm/dist-release_80/lib/Target/Mips/MipsMCInstLower.cpp =================================================================== --- vendor/llvm/dist-release_80/lib/Target/Mips/MipsMCInstLower.cpp (revision 343793) +++ vendor/llvm/dist-release_80/lib/Target/Mips/MipsMCInstLower.cpp (revision 343794) @@ -1,329 +1,331 @@ //===- MipsMCInstLower.cpp - Convert Mips MachineInstr to MCInst ----------===// // // The LLVM Compiler Infrastructure // // This file is distributed under the University of Illinois Open Source // License. See LICENSE.TXT for details. // //===----------------------------------------------------------------------===// // // This file contains code to lower Mips MachineInstrs to their corresponding // MCInst records. // //===----------------------------------------------------------------------===// #include "MipsMCInstLower.h" #include "MCTargetDesc/MipsBaseInfo.h" #include "MCTargetDesc/MipsMCExpr.h" #include "MipsAsmPrinter.h" #include "llvm/CodeGen/MachineBasicBlock.h" #include "llvm/CodeGen/MachineInstr.h" #include "llvm/CodeGen/MachineOperand.h" #include "llvm/MC/MCExpr.h" #include "llvm/MC/MCInst.h" #include "llvm/Support/ErrorHandling.h" #include using namespace llvm; MipsMCInstLower::MipsMCInstLower(MipsAsmPrinter &asmprinter) : AsmPrinter(asmprinter) {} void MipsMCInstLower::Initialize(MCContext *C) { Ctx = C; } MCOperand MipsMCInstLower::LowerSymbolOperand(const MachineOperand &MO, MachineOperandType MOTy, unsigned Offset) const { MCSymbolRefExpr::VariantKind Kind = MCSymbolRefExpr::VK_None; MipsMCExpr::MipsExprKind TargetKind = MipsMCExpr::MEK_None; bool IsGpOff = false; const MCSymbol *Symbol; switch(MO.getTargetFlags()) { default: llvm_unreachable("Invalid target flag!"); case MipsII::MO_NO_FLAG: break; case MipsII::MO_GPREL: TargetKind = MipsMCExpr::MEK_GPREL; break; case MipsII::MO_GOT_CALL: TargetKind = MipsMCExpr::MEK_GOT_CALL; break; case MipsII::MO_GOT: TargetKind = MipsMCExpr::MEK_GOT; break; case MipsII::MO_ABS_HI: TargetKind = MipsMCExpr::MEK_HI; break; case MipsII::MO_ABS_LO: TargetKind = MipsMCExpr::MEK_LO; break; case MipsII::MO_TLSGD: TargetKind = MipsMCExpr::MEK_TLSGD; break; case MipsII::MO_TLSLDM: TargetKind = MipsMCExpr::MEK_TLSLDM; break; case MipsII::MO_DTPREL_HI: TargetKind = MipsMCExpr::MEK_DTPREL_HI; break; case MipsII::MO_DTPREL_LO: TargetKind = MipsMCExpr::MEK_DTPREL_LO; break; case MipsII::MO_GOTTPREL: TargetKind = MipsMCExpr::MEK_GOTTPREL; break; case MipsII::MO_TPREL_HI: TargetKind = MipsMCExpr::MEK_TPREL_HI; break; case MipsII::MO_TPREL_LO: TargetKind = MipsMCExpr::MEK_TPREL_LO; break; case MipsII::MO_GPOFF_HI: TargetKind = MipsMCExpr::MEK_HI; IsGpOff = true; break; case MipsII::MO_GPOFF_LO: TargetKind = MipsMCExpr::MEK_LO; IsGpOff = true; break; case MipsII::MO_GOT_DISP: TargetKind = MipsMCExpr::MEK_GOT_DISP; break; case MipsII::MO_GOT_HI16: TargetKind = MipsMCExpr::MEK_GOT_HI16; break; case MipsII::MO_GOT_LO16: TargetKind = MipsMCExpr::MEK_GOT_LO16; break; case MipsII::MO_GOT_PAGE: TargetKind = MipsMCExpr::MEK_GOT_PAGE; break; case MipsII::MO_GOT_OFST: TargetKind = MipsMCExpr::MEK_GOT_OFST; break; case MipsII::MO_HIGHER: TargetKind = MipsMCExpr::MEK_HIGHER; break; case MipsII::MO_HIGHEST: TargetKind = MipsMCExpr::MEK_HIGHEST; break; case MipsII::MO_CALL_HI16: TargetKind = MipsMCExpr::MEK_CALL_HI16; break; case MipsII::MO_CALL_LO16: TargetKind = MipsMCExpr::MEK_CALL_LO16; break; + case MipsII::MO_JALR: + return MCOperand(); } switch (MOTy) { case MachineOperand::MO_MachineBasicBlock: Symbol = MO.getMBB()->getSymbol(); break; case MachineOperand::MO_GlobalAddress: Symbol = AsmPrinter.getSymbol(MO.getGlobal()); Offset += MO.getOffset(); break; case MachineOperand::MO_BlockAddress: Symbol = AsmPrinter.GetBlockAddressSymbol(MO.getBlockAddress()); Offset += MO.getOffset(); break; case MachineOperand::MO_ExternalSymbol: Symbol = AsmPrinter.GetExternalSymbolSymbol(MO.getSymbolName()); Offset += MO.getOffset(); break; case MachineOperand::MO_MCSymbol: Symbol = MO.getMCSymbol(); Offset += MO.getOffset(); break; case MachineOperand::MO_JumpTableIndex: Symbol = AsmPrinter.GetJTISymbol(MO.getIndex()); break; case MachineOperand::MO_ConstantPoolIndex: Symbol = AsmPrinter.GetCPISymbol(MO.getIndex()); Offset += MO.getOffset(); break; default: llvm_unreachable(""); } const MCExpr *Expr = MCSymbolRefExpr::create(Symbol, Kind, *Ctx); if (Offset) { // Assume offset is never negative. assert(Offset > 0); Expr = MCBinaryExpr::createAdd(Expr, MCConstantExpr::create(Offset, *Ctx), *Ctx); } if (IsGpOff) Expr = MipsMCExpr::createGpOff(TargetKind, Expr, *Ctx); else if (TargetKind != MipsMCExpr::MEK_None) Expr = MipsMCExpr::create(TargetKind, Expr, *Ctx); return MCOperand::createExpr(Expr); } MCOperand MipsMCInstLower::LowerOperand(const MachineOperand &MO, unsigned offset) const { MachineOperandType MOTy = MO.getType(); switch (MOTy) { default: llvm_unreachable("unknown operand type"); case MachineOperand::MO_Register: // Ignore all implicit register operands. if (MO.isImplicit()) break; return MCOperand::createReg(MO.getReg()); case MachineOperand::MO_Immediate: return MCOperand::createImm(MO.getImm() + offset); case MachineOperand::MO_MachineBasicBlock: case MachineOperand::MO_GlobalAddress: case MachineOperand::MO_ExternalSymbol: case MachineOperand::MO_MCSymbol: case MachineOperand::MO_JumpTableIndex: case MachineOperand::MO_ConstantPoolIndex: case MachineOperand::MO_BlockAddress: return LowerSymbolOperand(MO, MOTy, offset); case MachineOperand::MO_RegisterMask: break; } return MCOperand(); } MCOperand MipsMCInstLower::createSub(MachineBasicBlock *BB1, MachineBasicBlock *BB2, MipsMCExpr::MipsExprKind Kind) const { const MCSymbolRefExpr *Sym1 = MCSymbolRefExpr::create(BB1->getSymbol(), *Ctx); const MCSymbolRefExpr *Sym2 = MCSymbolRefExpr::create(BB2->getSymbol(), *Ctx); const MCBinaryExpr *Sub = MCBinaryExpr::createSub(Sym1, Sym2, *Ctx); return MCOperand::createExpr(MipsMCExpr::create(Kind, Sub, *Ctx)); } void MipsMCInstLower:: lowerLongBranchLUi(const MachineInstr *MI, MCInst &OutMI) const { OutMI.setOpcode(Mips::LUi); // Lower register operand. OutMI.addOperand(LowerOperand(MI->getOperand(0))); MipsMCExpr::MipsExprKind Kind; unsigned TargetFlags = MI->getOperand(1).getTargetFlags(); switch (TargetFlags) { case MipsII::MO_HIGHEST: Kind = MipsMCExpr::MEK_HIGHEST; break; case MipsII::MO_HIGHER: Kind = MipsMCExpr::MEK_HIGHER; break; case MipsII::MO_ABS_HI: Kind = MipsMCExpr::MEK_HI; break; case MipsII::MO_ABS_LO: Kind = MipsMCExpr::MEK_LO; break; default: report_fatal_error("Unexpected flags for lowerLongBranchLUi"); } if (MI->getNumOperands() == 2) { const MCExpr *Expr = MCSymbolRefExpr::create(MI->getOperand(1).getMBB()->getSymbol(), *Ctx); const MipsMCExpr *MipsExpr = MipsMCExpr::create(Kind, Expr, *Ctx); OutMI.addOperand(MCOperand::createExpr(MipsExpr)); } else if (MI->getNumOperands() == 3) { // Create %hi($tgt-$baltgt). OutMI.addOperand(createSub(MI->getOperand(1).getMBB(), MI->getOperand(2).getMBB(), Kind)); } } void MipsMCInstLower::lowerLongBranchADDiu(const MachineInstr *MI, MCInst &OutMI, int Opcode) const { OutMI.setOpcode(Opcode); MipsMCExpr::MipsExprKind Kind; unsigned TargetFlags = MI->getOperand(2).getTargetFlags(); switch (TargetFlags) { case MipsII::MO_HIGHEST: Kind = MipsMCExpr::MEK_HIGHEST; break; case MipsII::MO_HIGHER: Kind = MipsMCExpr::MEK_HIGHER; break; case MipsII::MO_ABS_HI: Kind = MipsMCExpr::MEK_HI; break; case MipsII::MO_ABS_LO: Kind = MipsMCExpr::MEK_LO; break; default: report_fatal_error("Unexpected flags for lowerLongBranchADDiu"); } // Lower two register operands. for (unsigned I = 0, E = 2; I != E; ++I) { const MachineOperand &MO = MI->getOperand(I); OutMI.addOperand(LowerOperand(MO)); } if (MI->getNumOperands() == 3) { // Lower register operand. const MCExpr *Expr = MCSymbolRefExpr::create(MI->getOperand(2).getMBB()->getSymbol(), *Ctx); const MipsMCExpr *MipsExpr = MipsMCExpr::create(Kind, Expr, *Ctx); OutMI.addOperand(MCOperand::createExpr(MipsExpr)); } else if (MI->getNumOperands() == 4) { // Create %lo($tgt-$baltgt) or %hi($tgt-$baltgt). OutMI.addOperand(createSub(MI->getOperand(2).getMBB(), MI->getOperand(3).getMBB(), Kind)); } } bool MipsMCInstLower::lowerLongBranch(const MachineInstr *MI, MCInst &OutMI) const { switch (MI->getOpcode()) { default: return false; case Mips::LONG_BRANCH_LUi: case Mips::LONG_BRANCH_LUi2Op: case Mips::LONG_BRANCH_LUi2Op_64: lowerLongBranchLUi(MI, OutMI); return true; case Mips::LONG_BRANCH_ADDiu: case Mips::LONG_BRANCH_ADDiu2Op: lowerLongBranchADDiu(MI, OutMI, Mips::ADDiu); return true; case Mips::LONG_BRANCH_DADDiu: case Mips::LONG_BRANCH_DADDiu2Op: lowerLongBranchADDiu(MI, OutMI, Mips::DADDiu); return true; } } void MipsMCInstLower::Lower(const MachineInstr *MI, MCInst &OutMI) const { if (lowerLongBranch(MI, OutMI)) return; OutMI.setOpcode(MI->getOpcode()); for (unsigned i = 0, e = MI->getNumOperands(); i != e; ++i) { const MachineOperand &MO = MI->getOperand(i); MCOperand MCOp = LowerOperand(MO); if (MCOp.isValid()) OutMI.addOperand(MCOp); } } Index: vendor/llvm/dist-release_80/lib/Target/X86/X86DiscriminateMemOps.cpp =================================================================== --- vendor/llvm/dist-release_80/lib/Target/X86/X86DiscriminateMemOps.cpp (revision 343793) +++ vendor/llvm/dist-release_80/lib/Target/X86/X86DiscriminateMemOps.cpp (revision 343794) @@ -1,156 +1,167 @@ //===- X86DiscriminateMemOps.cpp - Unique IDs for Mem Ops -----------------===// // // The LLVM Compiler Infrastructure // // This file is distributed under the University of Illinois Open Source // License. See LICENSE.TXT for details. // //===----------------------------------------------------------------------===// /// /// This pass aids profile-driven cache prefetch insertion by ensuring all /// instructions that have a memory operand are distinguishible from each other. /// //===----------------------------------------------------------------------===// #include "X86.h" #include "X86InstrBuilder.h" #include "X86InstrInfo.h" #include "X86MachineFunctionInfo.h" #include "X86Subtarget.h" #include "llvm/CodeGen/MachineModuleInfo.h" #include "llvm/IR/DebugInfoMetadata.h" #include "llvm/ProfileData/SampleProf.h" #include "llvm/ProfileData/SampleProfReader.h" #include "llvm/Support/Debug.h" #include "llvm/Transforms/IPO/SampleProfile.h" using namespace llvm; #define DEBUG_TYPE "x86-discriminate-memops" +static cl::opt EnableDiscriminateMemops( + DEBUG_TYPE, cl::init(false), + cl::desc("Generate unique debug info for each instruction with a memory " + "operand. Should be enabled for profile-drived cache prefetching, " + "both in the build of the binary being profiled, as well as in " + "the build of the binary consuming the profile."), + cl::Hidden); + namespace { using Location = std::pair; Location diToLocation(const DILocation *Loc) { return std::make_pair(Loc->getFilename(), Loc->getLine()); } /// Ensure each instruction having a memory operand has a distinct pair. void updateDebugInfo(MachineInstr *MI, const DILocation *Loc) { DebugLoc DL(Loc); MI->setDebugLoc(DL); } class X86DiscriminateMemOps : public MachineFunctionPass { bool runOnMachineFunction(MachineFunction &MF) override; StringRef getPassName() const override { return "X86 Discriminate Memory Operands"; } public: static char ID; /// Default construct and initialize the pass. X86DiscriminateMemOps(); }; } // end anonymous namespace //===----------------------------------------------------------------------===// // Implementation //===----------------------------------------------------------------------===// char X86DiscriminateMemOps::ID = 0; /// Default construct and initialize the pass. X86DiscriminateMemOps::X86DiscriminateMemOps() : MachineFunctionPass(ID) {} bool X86DiscriminateMemOps::runOnMachineFunction(MachineFunction &MF) { + if (!EnableDiscriminateMemops) + return false; + DISubprogram *FDI = MF.getFunction().getSubprogram(); if (!FDI || !FDI->getUnit()->getDebugInfoForProfiling()) return false; // Have a default DILocation, if we find instructions with memops that don't // have any debug info. const DILocation *ReferenceDI = DILocation::get(FDI->getContext(), FDI->getLine(), 0, FDI); DenseMap MemOpDiscriminators; MemOpDiscriminators[diToLocation(ReferenceDI)] = 0; // Figure out the largest discriminator issued for each Location. When we // issue new discriminators, we can thus avoid issuing discriminators // belonging to instructions that don't have memops. This isn't a requirement // for the goals of this pass, however, it avoids unnecessary ambiguity. for (auto &MBB : MF) { for (auto &MI : MBB) { const auto &DI = MI.getDebugLoc(); if (!DI) continue; Location Loc = diToLocation(DI); MemOpDiscriminators[Loc] = std::max(MemOpDiscriminators[Loc], DI->getBaseDiscriminator()); } } // Keep track of the discriminators seen at each Location. If an instruction's // DebugInfo has a Location and discriminator we've already seen, replace its // discriminator with a new one, to guarantee uniqueness. DenseMap> Seen; bool Changed = false; for (auto &MBB : MF) { for (auto &MI : MBB) { if (X86II::getMemoryOperandNo(MI.getDesc().TSFlags) < 0) continue; const DILocation *DI = MI.getDebugLoc(); if (!DI) { DI = ReferenceDI; } Location L = diToLocation(DI); DenseSet &Set = Seen[L]; const std::pair::iterator, bool> TryInsert = Set.insert(DI->getBaseDiscriminator()); if (!TryInsert.second) { unsigned BF, DF, CI = 0; DILocation::decodeDiscriminator(DI->getDiscriminator(), BF, DF, CI); Optional EncodedDiscriminator = DILocation::encodeDiscriminator( MemOpDiscriminators[L] + 1, DF, CI); if (!EncodedDiscriminator) { // FIXME(mtrofin): The assumption is that this scenario is infrequent/OK // not to support. If evidence points otherwise, we can explore synthesizeing // unique DIs by adding fake line numbers, or by constructing 64 bit // discriminators. LLVM_DEBUG(dbgs() << "Unable to create a unique discriminator " "for instruction with memory operand in: " << DI->getFilename() << " Line: " << DI->getLine() << " Column: " << DI->getColumn() << ". This is likely due to a large macro expansion. \n"); continue; } // Since we were able to encode, bump the MemOpDiscriminators. ++MemOpDiscriminators[L]; DI = DI->cloneWithDiscriminator(EncodedDiscriminator.getValue()); updateDebugInfo(&MI, DI); Changed = true; std::pair::iterator, bool> MustInsert = Set.insert(DI->getBaseDiscriminator()); (void)MustInsert; // Silence warning in release build. assert(MustInsert.second && "New discriminator shouldn't be present in set"); } // Bump the reference DI to avoid cramming discriminators on line 0. // FIXME(mtrofin): pin ReferenceDI on blocks or first instruction with DI // in a block. It's more consistent than just relying on the last memop // instruction we happened to see. ReferenceDI = DI; } } return Changed; } FunctionPass *llvm::createX86DiscriminateMemOpsPass() { return new X86DiscriminateMemOps(); } Index: vendor/llvm/dist-release_80/lib/Target/X86/X86InsertPrefetch.cpp =================================================================== --- vendor/llvm/dist-release_80/lib/Target/X86/X86InsertPrefetch.cpp (revision 343793) +++ vendor/llvm/dist-release_80/lib/Target/X86/X86InsertPrefetch.cpp (revision 343794) @@ -1,253 +1,254 @@ //===------- X86InsertPrefetch.cpp - Insert cache prefetch hints ----------===// // // The LLVM Compiler Infrastructure // // This file is distributed under the University of Illinois Open Source // License. See LICENSE.TXT for details. // //===----------------------------------------------------------------------===// // // This pass applies cache prefetch instructions based on a profile. The pass // assumes DiscriminateMemOps ran immediately before, to ensure debug info // matches the one used at profile generation time. The profile is encoded in // afdo format (text or binary). It contains prefetch hints recommendations. // Each recommendation is made in terms of debug info locations, a type (i.e. // nta, t{0|1|2}) and a delta. The debug info identifies an instruction with a // memory operand (see X86DiscriminateMemOps). The prefetch will be made for // a location at that memory operand + the delta specified in the // recommendation. // //===----------------------------------------------------------------------===// #include "X86.h" #include "X86InstrBuilder.h" #include "X86InstrInfo.h" #include "X86MachineFunctionInfo.h" #include "X86Subtarget.h" #include "llvm/CodeGen/MachineModuleInfo.h" #include "llvm/IR/DebugInfoMetadata.h" #include "llvm/ProfileData/SampleProf.h" #include "llvm/ProfileData/SampleProfReader.h" #include "llvm/Transforms/IPO/SampleProfile.h" using namespace llvm; using namespace sampleprof; static cl::opt PrefetchHintsFile("prefetch-hints-file", - cl::desc("Path to the prefetch hints profile."), + cl::desc("Path to the prefetch hints profile. See also " + "-x86-discriminate-memops"), cl::Hidden); namespace { class X86InsertPrefetch : public MachineFunctionPass { void getAnalysisUsage(AnalysisUsage &AU) const override; bool doInitialization(Module &) override; bool runOnMachineFunction(MachineFunction &MF) override; struct PrefetchInfo { unsigned InstructionID; int64_t Delta; }; typedef SmallVectorImpl Prefetches; bool findPrefetchInfo(const FunctionSamples *Samples, const MachineInstr &MI, Prefetches &prefetches) const; public: static char ID; X86InsertPrefetch(const std::string &PrefetchHintsFilename); StringRef getPassName() const override { return "X86 Insert Cache Prefetches"; } private: std::string Filename; std::unique_ptr Reader; }; using PrefetchHints = SampleRecord::CallTargetMap; // Return any prefetching hints for the specified MachineInstruction. The hints // are returned as pairs (name, delta). ErrorOr getPrefetchHints(const FunctionSamples *TopSamples, const MachineInstr &MI) { if (const auto &Loc = MI.getDebugLoc()) if (const auto *Samples = TopSamples->findFunctionSamples(Loc)) return Samples->findCallTargetMapAt(FunctionSamples::getOffset(Loc), Loc->getBaseDiscriminator()); return std::error_code(); } // The prefetch instruction can't take memory operands involving vector // registers. bool IsMemOpCompatibleWithPrefetch(const MachineInstr &MI, int Op) { unsigned BaseReg = MI.getOperand(Op + X86::AddrBaseReg).getReg(); unsigned IndexReg = MI.getOperand(Op + X86::AddrIndexReg).getReg(); return (BaseReg == 0 || X86MCRegisterClasses[X86::GR64RegClassID].contains(BaseReg) || X86MCRegisterClasses[X86::GR32RegClassID].contains(BaseReg)) && (IndexReg == 0 || X86MCRegisterClasses[X86::GR64RegClassID].contains(IndexReg) || X86MCRegisterClasses[X86::GR32RegClassID].contains(IndexReg)); } } // end anonymous namespace //===----------------------------------------------------------------------===// // Implementation //===----------------------------------------------------------------------===// char X86InsertPrefetch::ID = 0; X86InsertPrefetch::X86InsertPrefetch(const std::string &PrefetchHintsFilename) : MachineFunctionPass(ID), Filename(PrefetchHintsFilename) {} /// Return true if the provided MachineInstruction has cache prefetch hints. In /// that case, the prefetch hints are stored, in order, in the Prefetches /// vector. bool X86InsertPrefetch::findPrefetchInfo(const FunctionSamples *TopSamples, const MachineInstr &MI, Prefetches &Prefetches) const { assert(Prefetches.empty() && "Expected caller passed empty PrefetchInfo vector."); static const std::pair HintTypes[] = { {"_nta_", X86::PREFETCHNTA}, {"_t0_", X86::PREFETCHT0}, {"_t1_", X86::PREFETCHT1}, {"_t2_", X86::PREFETCHT2}, }; static const char *SerializedPrefetchPrefix = "__prefetch"; const ErrorOr T = getPrefetchHints(TopSamples, MI); if (!T) return false; int16_t max_index = -1; // Convert serialized prefetch hints into PrefetchInfo objects, and populate // the Prefetches vector. for (const auto &S_V : *T) { StringRef Name = S_V.getKey(); if (Name.consume_front(SerializedPrefetchPrefix)) { int64_t D = static_cast(S_V.second); unsigned IID = 0; for (const auto &HintType : HintTypes) { if (Name.startswith(HintType.first)) { Name = Name.drop_front(HintType.first.size()); IID = HintType.second; break; } } if (IID == 0) return false; uint8_t index = 0; Name.consumeInteger(10, index); if (index >= Prefetches.size()) Prefetches.resize(index + 1); Prefetches[index] = {IID, D}; max_index = std::max(max_index, static_cast(index)); } } assert(max_index + 1 >= 0 && "Possible overflow: max_index + 1 should be positive."); assert(static_cast(max_index + 1) == Prefetches.size() && "The number of prefetch hints received should match the number of " "PrefetchInfo objects returned"); return !Prefetches.empty(); } bool X86InsertPrefetch::doInitialization(Module &M) { if (Filename.empty()) return false; LLVMContext &Ctx = M.getContext(); ErrorOr> ReaderOrErr = SampleProfileReader::create(Filename, Ctx); if (std::error_code EC = ReaderOrErr.getError()) { std::string Msg = "Could not open profile: " + EC.message(); Ctx.diagnose(DiagnosticInfoSampleProfile(Filename, Msg, DiagnosticSeverity::DS_Warning)); return false; } Reader = std::move(ReaderOrErr.get()); Reader->read(); return true; } void X86InsertPrefetch::getAnalysisUsage(AnalysisUsage &AU) const { AU.setPreservesAll(); AU.addRequired(); } bool X86InsertPrefetch::runOnMachineFunction(MachineFunction &MF) { if (!Reader) return false; const FunctionSamples *Samples = Reader->getSamplesFor(MF.getFunction()); if (!Samples) return false; bool Changed = false; const TargetInstrInfo *TII = MF.getSubtarget().getInstrInfo(); SmallVector Prefetches; for (auto &MBB : MF) { for (auto MI = MBB.instr_begin(); MI != MBB.instr_end();) { auto Current = MI; ++MI; int Offset = X86II::getMemoryOperandNo(Current->getDesc().TSFlags); if (Offset < 0) continue; unsigned Bias = X86II::getOperandBias(Current->getDesc()); int MemOpOffset = Offset + Bias; // FIXME(mtrofin): ORE message when the recommendation cannot be taken. if (!IsMemOpCompatibleWithPrefetch(*Current, MemOpOffset)) continue; Prefetches.clear(); if (!findPrefetchInfo(Samples, *Current, Prefetches)) continue; assert(!Prefetches.empty() && "The Prefetches vector should contain at least a value if " "findPrefetchInfo returned true."); for (auto &PrefInfo : Prefetches) { unsigned PFetchInstrID = PrefInfo.InstructionID; int64_t Delta = PrefInfo.Delta; const MCInstrDesc &Desc = TII->get(PFetchInstrID); MachineInstr *PFetch = MF.CreateMachineInstr(Desc, Current->getDebugLoc(), true); MachineInstrBuilder MIB(MF, PFetch); assert(X86::AddrBaseReg == 0 && X86::AddrScaleAmt == 1 && X86::AddrIndexReg == 2 && X86::AddrDisp == 3 && X86::AddrSegmentReg == 4 && "Unexpected change in X86 operand offset order."); // This assumes X86::AddBaseReg = 0, {...}ScaleAmt = 1, etc. // FIXME(mtrofin): consider adding a: // MachineInstrBuilder::set(unsigned offset, op). MIB.addReg(Current->getOperand(MemOpOffset + X86::AddrBaseReg).getReg()) .addImm( Current->getOperand(MemOpOffset + X86::AddrScaleAmt).getImm()) .addReg( Current->getOperand(MemOpOffset + X86::AddrIndexReg).getReg()) .addImm(Current->getOperand(MemOpOffset + X86::AddrDisp).getImm() + Delta) .addReg(Current->getOperand(MemOpOffset + X86::AddrSegmentReg) .getReg()); if (!Current->memoperands_empty()) { MachineMemOperand *CurrentOp = *(Current->memoperands_begin()); MIB.addMemOperand(MF.getMachineMemOperand( CurrentOp, CurrentOp->getOffset() + Delta, CurrentOp->getSize())); } // Insert before Current. This is because Current may clobber some of // the registers used to describe the input memory operand. MBB.insert(Current, PFetch); Changed = true; } } } return Changed; } FunctionPass *llvm::createX86InsertPrefetchPass() { return new X86InsertPrefetch(PrefetchHintsFile); } Index: vendor/llvm/dist-release_80/lib/Transforms/Utils/FunctionImportUtils.cpp =================================================================== --- vendor/llvm/dist-release_80/lib/Transforms/Utils/FunctionImportUtils.cpp (revision 343793) +++ vendor/llvm/dist-release_80/lib/Transforms/Utils/FunctionImportUtils.cpp (revision 343794) @@ -1,295 +1,313 @@ //===- lib/Transforms/Utils/FunctionImportUtils.cpp - Importing utilities -===// // // The LLVM Compiler Infrastructure // // This file is distributed under the University of Illinois Open Source // License. See LICENSE.TXT for details. // //===----------------------------------------------------------------------===// // // This file implements the FunctionImportGlobalProcessing class, used // to perform the necessary global value handling for function importing. // //===----------------------------------------------------------------------===// #include "llvm/Transforms/Utils/FunctionImportUtils.h" #include "llvm/IR/InstIterator.h" using namespace llvm; /// Checks if we should import SGV as a definition, otherwise import as a /// declaration. bool FunctionImportGlobalProcessing::doImportAsDefinition( const GlobalValue *SGV, SetVector *GlobalsToImport) { // Only import the globals requested for importing. if (!GlobalsToImport->count(const_cast(SGV))) return false; assert(!isa(SGV) && "Unexpected global alias in the import list."); // Otherwise yes. return true; } bool FunctionImportGlobalProcessing::doImportAsDefinition( const GlobalValue *SGV) { if (!isPerformingImport()) return false; return FunctionImportGlobalProcessing::doImportAsDefinition(SGV, GlobalsToImport); } bool FunctionImportGlobalProcessing::shouldPromoteLocalToGlobal( const GlobalValue *SGV) { assert(SGV->hasLocalLinkage()); // Both the imported references and the original local variable must // be promoted. if (!isPerformingImport() && !isModuleExporting()) return false; if (isPerformingImport()) { assert((!GlobalsToImport->count(const_cast(SGV)) || !isNonRenamableLocal(*SGV)) && "Attempting to promote non-renamable local"); // We don't know for sure yet if we are importing this value (as either // a reference or a def), since we are simply walking all values in the // module. But by necessity if we end up importing it and it is local, // it must be promoted, so unconditionally promote all values in the // importing module. return true; } // When exporting, consult the index. We can have more than one local // with the same GUID, in the case of same-named locals in different but // same-named source files that were compiled in their respective directories // (so the source file name and resulting GUID is the same). Find the one // in this module. auto Summary = ImportIndex.findSummaryInModule( SGV->getGUID(), SGV->getParent()->getModuleIdentifier()); assert(Summary && "Missing summary for global value when exporting"); auto Linkage = Summary->linkage(); if (!GlobalValue::isLocalLinkage(Linkage)) { assert(!isNonRenamableLocal(*SGV) && "Attempting to promote non-renamable local"); return true; } return false; } #ifndef NDEBUG bool FunctionImportGlobalProcessing::isNonRenamableLocal( const GlobalValue &GV) const { if (!GV.hasLocalLinkage()) return false; // This needs to stay in sync with the logic in buildModuleSummaryIndex. if (GV.hasSection()) return true; if (Used.count(const_cast(&GV))) return true; return false; } #endif std::string FunctionImportGlobalProcessing::getName(const GlobalValue *SGV, bool DoPromote) { // For locals that must be promoted to global scope, ensure that // the promoted name uniquely identifies the copy in the original module, // using the ID assigned during combined index creation. When importing, // we rename all locals (not just those that are promoted) in order to // avoid naming conflicts between locals imported from different modules. if (SGV->hasLocalLinkage() && (DoPromote || isPerformingImport())) return ModuleSummaryIndex::getGlobalNameForLocal( SGV->getName(), ImportIndex.getModuleHash(SGV->getParent()->getModuleIdentifier())); return SGV->getName(); } GlobalValue::LinkageTypes FunctionImportGlobalProcessing::getLinkage(const GlobalValue *SGV, bool DoPromote) { // Any local variable that is referenced by an exported function needs // to be promoted to global scope. Since we don't currently know which // functions reference which local variables/functions, we must treat // all as potentially exported if this module is exporting anything. if (isModuleExporting()) { if (SGV->hasLocalLinkage() && DoPromote) return GlobalValue::ExternalLinkage; return SGV->getLinkage(); } // Otherwise, if we aren't importing, no linkage change is needed. if (!isPerformingImport()) return SGV->getLinkage(); switch (SGV->getLinkage()) { case GlobalValue::LinkOnceODRLinkage: case GlobalValue::ExternalLinkage: // External and linkonce definitions are converted to available_externally // definitions upon import, so that they are available for inlining // and/or optimization, but are turned into declarations later // during the EliminateAvailableExternally pass. if (doImportAsDefinition(SGV) && !dyn_cast(SGV)) return GlobalValue::AvailableExternallyLinkage; // An imported external declaration stays external. return SGV->getLinkage(); case GlobalValue::AvailableExternallyLinkage: // An imported available_externally definition converts // to external if imported as a declaration. if (!doImportAsDefinition(SGV)) return GlobalValue::ExternalLinkage; // An imported available_externally declaration stays that way. return SGV->getLinkage(); case GlobalValue::LinkOnceAnyLinkage: case GlobalValue::WeakAnyLinkage: // Can't import linkonce_any/weak_any definitions correctly, or we might // change the program semantics, since the linker will pick the first // linkonce_any/weak_any definition and importing would change the order // they are seen by the linker. The module linking caller needs to enforce // this. assert(!doImportAsDefinition(SGV)); // If imported as a declaration, it becomes external_weak. return SGV->getLinkage(); case GlobalValue::WeakODRLinkage: // For weak_odr linkage, there is a guarantee that all copies will be // equivalent, so the issue described above for weak_any does not exist, // and the definition can be imported. It can be treated similarly // to an imported externally visible global value. if (doImportAsDefinition(SGV) && !dyn_cast(SGV)) return GlobalValue::AvailableExternallyLinkage; else return GlobalValue::ExternalLinkage; case GlobalValue::AppendingLinkage: // It would be incorrect to import an appending linkage variable, // since it would cause global constructors/destructors to be // executed multiple times. This should have already been handled // by linkIfNeeded, and we will assert in shouldLinkFromSource // if we try to import, so we simply return AppendingLinkage. return GlobalValue::AppendingLinkage; case GlobalValue::InternalLinkage: case GlobalValue::PrivateLinkage: // If we are promoting the local to global scope, it is handled // similarly to a normal externally visible global. if (DoPromote) { if (doImportAsDefinition(SGV) && !dyn_cast(SGV)) return GlobalValue::AvailableExternallyLinkage; else return GlobalValue::ExternalLinkage; } // A non-promoted imported local definition stays local. // The ThinLTO pass will eventually force-import their definitions. return SGV->getLinkage(); case GlobalValue::ExternalWeakLinkage: // External weak doesn't apply to definitions, must be a declaration. assert(!doImportAsDefinition(SGV)); // Linkage stays external_weak. return SGV->getLinkage(); case GlobalValue::CommonLinkage: // Linkage stays common on definitions. // The ThinLTO pass will eventually force-import their definitions. return SGV->getLinkage(); } llvm_unreachable("unknown linkage type"); } void FunctionImportGlobalProcessing::processGlobalForThinLTO(GlobalValue &GV) { ValueInfo VI; if (GV.hasName()) { VI = ImportIndex.getValueInfo(GV.getGUID()); // Set synthetic function entry counts. if (VI && ImportIndex.hasSyntheticEntryCounts()) { if (Function *F = dyn_cast(&GV)) { if (!F->isDeclaration()) { for (auto &S : VI.getSummaryList()) { FunctionSummary *FS = dyn_cast(S->getBaseObject()); if (FS->modulePath() == M.getModuleIdentifier()) { F->setEntryCount(Function::ProfileCount(FS->entryCount(), Function::PCT_Synthetic)); break; } } } } } // Check the summaries to see if the symbol gets resolved to a known local // definition. if (VI && VI.isDSOLocal()) { GV.setDSOLocal(true); if (GV.hasDLLImportStorageClass()) GV.setDLLStorageClass(GlobalValue::DefaultStorageClass); } } // Mark read-only variables which can be imported with specific attribute. // We can't internalize them now because IRMover will fail to link variable // definitions to their external declarations during ThinLTO import. We'll // internalize read-only variables later, after import is finished. // See internalizeImmutableGVs. // // If global value dead stripping is not enabled in summary then // propagateConstants hasn't been run. We can't internalize GV // in such case. if (!GV.isDeclaration() && VI && ImportIndex.withGlobalValueDeadStripping()) { const auto &SL = VI.getSummaryList(); auto *GVS = SL.empty() ? nullptr : dyn_cast(SL[0].get()); if (GVS && GVS->isReadOnly()) cast(&GV)->addAttribute("thinlto-internalize"); } bool DoPromote = false; if (GV.hasLocalLinkage() && ((DoPromote = shouldPromoteLocalToGlobal(&GV)) || isPerformingImport())) { + // Save the original name string before we rename GV below. + auto Name = GV.getName().str(); // Once we change the name or linkage it is difficult to determine // again whether we should promote since shouldPromoteLocalToGlobal needs // to locate the summary (based on GUID from name and linkage). Therefore, // use DoPromote result saved above. GV.setName(getName(&GV, DoPromote)); GV.setLinkage(getLinkage(&GV, DoPromote)); if (!GV.hasLocalLinkage()) GV.setVisibility(GlobalValue::HiddenVisibility); + + // If we are renaming a COMDAT leader, ensure that we record the COMDAT + // for later renaming as well. This is required for COFF. + if (const auto *C = GV.getComdat()) + if (C->getName() == Name) + RenamedComdats.try_emplace(C, M.getOrInsertComdat(GV.getName())); } else GV.setLinkage(getLinkage(&GV, /* DoPromote */ false)); // Remove functions imported as available externally defs from comdats, // as this is a declaration for the linker, and will be dropped eventually. // It is illegal for comdats to contain declarations. auto *GO = dyn_cast(&GV); if (GO && GO->isDeclarationForLinker() && GO->hasComdat()) { // The IRMover should not have placed any imported declarations in // a comdat, so the only declaration that should be in a comdat // at this point would be a definition imported as available_externally. assert(GO->hasAvailableExternallyLinkage() && "Expected comdat on definition (possibly available external)"); GO->setComdat(nullptr); } } void FunctionImportGlobalProcessing::processGlobalsForThinLTO() { for (GlobalVariable &GV : M.globals()) processGlobalForThinLTO(GV); for (Function &SF : M) processGlobalForThinLTO(SF); for (GlobalAlias &GA : M.aliases()) processGlobalForThinLTO(GA); + + // Replace any COMDATS that required renaming (because the COMDAT leader was + // promoted and renamed). + if (!RenamedComdats.empty()) + for (auto &GO : M.global_objects()) + if (auto *C = GO.getComdat()) { + auto Replacement = RenamedComdats.find(C); + if (Replacement != RenamedComdats.end()) + GO.setComdat(Replacement->second); + } } bool FunctionImportGlobalProcessing::run() { processGlobalsForThinLTO(); return false; } bool llvm::renameModuleForThinLTO(Module &M, const ModuleSummaryIndex &Index, SetVector *GlobalsToImport) { FunctionImportGlobalProcessing ThinLTOProcessing(M, Index, GlobalsToImport); return ThinLTOProcessing.run(); } Index: vendor/llvm/dist-release_80/lib/Transforms/Utils/LoopUtils.cpp =================================================================== --- vendor/llvm/dist-release_80/lib/Transforms/Utils/LoopUtils.cpp (revision 343793) +++ vendor/llvm/dist-release_80/lib/Transforms/Utils/LoopUtils.cpp (revision 343794) @@ -1,966 +1,969 @@ //===-- LoopUtils.cpp - Loop Utility functions -------------------------===// // // The LLVM Compiler Infrastructure // // This file is distributed under the University of Illinois Open Source // License. See LICENSE.TXT for details. // //===----------------------------------------------------------------------===// // // This file defines common loop utility functions. // //===----------------------------------------------------------------------===// #include "llvm/Transforms/Utils/LoopUtils.h" #include "llvm/ADT/ScopeExit.h" #include "llvm/Analysis/AliasAnalysis.h" #include "llvm/Analysis/BasicAliasAnalysis.h" #include "llvm/Analysis/GlobalsModRef.h" #include "llvm/Analysis/InstructionSimplify.h" #include "llvm/Analysis/LoopInfo.h" #include "llvm/Analysis/LoopPass.h" #include "llvm/Analysis/MustExecute.h" #include "llvm/Analysis/ScalarEvolution.h" #include "llvm/Analysis/ScalarEvolutionAliasAnalysis.h" #include "llvm/Analysis/ScalarEvolutionExpander.h" #include "llvm/Analysis/ScalarEvolutionExpressions.h" #include "llvm/Analysis/TargetTransformInfo.h" #include "llvm/Analysis/ValueTracking.h" #include "llvm/IR/DIBuilder.h" #include "llvm/IR/DomTreeUpdater.h" #include "llvm/IR/Dominators.h" #include "llvm/IR/Instructions.h" #include "llvm/IR/IntrinsicInst.h" #include "llvm/IR/Module.h" #include "llvm/IR/PatternMatch.h" #include "llvm/IR/ValueHandle.h" #include "llvm/Pass.h" #include "llvm/Support/Debug.h" #include "llvm/Support/KnownBits.h" #include "llvm/Transforms/Utils/BasicBlockUtils.h" using namespace llvm; using namespace llvm::PatternMatch; #define DEBUG_TYPE "loop-utils" static const char *LLVMLoopDisableNonforced = "llvm.loop.disable_nonforced"; bool llvm::formDedicatedExitBlocks(Loop *L, DominatorTree *DT, LoopInfo *LI, bool PreserveLCSSA) { bool Changed = false; // We re-use a vector for the in-loop predecesosrs. SmallVector InLoopPredecessors; auto RewriteExit = [&](BasicBlock *BB) { assert(InLoopPredecessors.empty() && "Must start with an empty predecessors list!"); auto Cleanup = make_scope_exit([&] { InLoopPredecessors.clear(); }); // See if there are any non-loop predecessors of this exit block and // keep track of the in-loop predecessors. bool IsDedicatedExit = true; for (auto *PredBB : predecessors(BB)) if (L->contains(PredBB)) { if (isa(PredBB->getTerminator())) // We cannot rewrite exiting edges from an indirectbr. return false; InLoopPredecessors.push_back(PredBB); } else { IsDedicatedExit = false; } assert(!InLoopPredecessors.empty() && "Must have *some* loop predecessor!"); // Nothing to do if this is already a dedicated exit. if (IsDedicatedExit) return false; auto *NewExitBB = SplitBlockPredecessors( BB, InLoopPredecessors, ".loopexit", DT, LI, nullptr, PreserveLCSSA); if (!NewExitBB) LLVM_DEBUG( dbgs() << "WARNING: Can't create a dedicated exit block for loop: " << *L << "\n"); else LLVM_DEBUG(dbgs() << "LoopSimplify: Creating dedicated exit block " << NewExitBB->getName() << "\n"); return true; }; // Walk the exit blocks directly rather than building up a data structure for // them, but only visit each one once. SmallPtrSet Visited; for (auto *BB : L->blocks()) for (auto *SuccBB : successors(BB)) { // We're looking for exit blocks so skip in-loop successors. if (L->contains(SuccBB)) continue; // Visit each exit block exactly once. if (!Visited.insert(SuccBB).second) continue; Changed |= RewriteExit(SuccBB); } return Changed; } /// Returns the instructions that use values defined in the loop. SmallVector llvm::findDefsUsedOutsideOfLoop(Loop *L) { SmallVector UsedOutside; for (auto *Block : L->getBlocks()) // FIXME: I believe that this could use copy_if if the Inst reference could // be adapted into a pointer. for (auto &Inst : *Block) { auto Users = Inst.users(); if (any_of(Users, [&](User *U) { auto *Use = cast(U); return !L->contains(Use->getParent()); })) UsedOutside.push_back(&Inst); } return UsedOutside; } void llvm::getLoopAnalysisUsage(AnalysisUsage &AU) { // By definition, all loop passes need the LoopInfo analysis and the // Dominator tree it depends on. Because they all participate in the loop // pass manager, they must also preserve these. AU.addRequired(); AU.addPreserved(); AU.addRequired(); AU.addPreserved(); // We must also preserve LoopSimplify and LCSSA. We locally access their IDs // here because users shouldn't directly get them from this header. extern char &LoopSimplifyID; extern char &LCSSAID; AU.addRequiredID(LoopSimplifyID); AU.addPreservedID(LoopSimplifyID); AU.addRequiredID(LCSSAID); AU.addPreservedID(LCSSAID); // This is used in the LPPassManager to perform LCSSA verification on passes // which preserve lcssa form AU.addRequired(); AU.addPreserved(); // Loop passes are designed to run inside of a loop pass manager which means // that any function analyses they require must be required by the first loop // pass in the manager (so that it is computed before the loop pass manager // runs) and preserved by all loop pasess in the manager. To make this // reasonably robust, the set needed for most loop passes is maintained here. // If your loop pass requires an analysis not listed here, you will need to // carefully audit the loop pass manager nesting structure that results. AU.addRequired(); AU.addPreserved(); AU.addPreserved(); AU.addPreserved(); AU.addPreserved(); AU.addRequired(); AU.addPreserved(); } /// Manually defined generic "LoopPass" dependency initialization. This is used /// to initialize the exact set of passes from above in \c /// getLoopAnalysisUsage. It can be used within a loop pass's initialization /// with: /// /// INITIALIZE_PASS_DEPENDENCY(LoopPass) /// /// As-if "LoopPass" were a pass. void llvm::initializeLoopPassPass(PassRegistry &Registry) { INITIALIZE_PASS_DEPENDENCY(DominatorTreeWrapperPass) INITIALIZE_PASS_DEPENDENCY(LoopInfoWrapperPass) INITIALIZE_PASS_DEPENDENCY(LoopSimplify) INITIALIZE_PASS_DEPENDENCY(LCSSAWrapperPass) INITIALIZE_PASS_DEPENDENCY(AAResultsWrapperPass) INITIALIZE_PASS_DEPENDENCY(BasicAAWrapperPass) INITIALIZE_PASS_DEPENDENCY(GlobalsAAWrapperPass) INITIALIZE_PASS_DEPENDENCY(SCEVAAWrapperPass) INITIALIZE_PASS_DEPENDENCY(ScalarEvolutionWrapperPass) } /// Find string metadata for loop /// /// If it has a value (e.g. {"llvm.distribute", 1} return the value as an /// operand or null otherwise. If the string metadata is not found return /// Optional's not-a-value. Optional llvm::findStringMetadataForLoop(const Loop *TheLoop, StringRef Name) { MDNode *MD = findOptionMDForLoop(TheLoop, Name); if (!MD) return None; switch (MD->getNumOperands()) { case 1: return nullptr; case 2: return &MD->getOperand(1); default: llvm_unreachable("loop metadata has 0 or 1 operand"); } } static Optional getOptionalBoolLoopAttribute(const Loop *TheLoop, StringRef Name) { MDNode *MD = findOptionMDForLoop(TheLoop, Name); if (!MD) return None; switch (MD->getNumOperands()) { case 1: // When the value is absent it is interpreted as 'attribute set'. return true; case 2: - return mdconst::extract_or_null(MD->getOperand(1).get()); + if (ConstantInt *IntMD = + mdconst::extract_or_null(MD->getOperand(1).get())) + return IntMD->getZExtValue(); + return true; } llvm_unreachable("unexpected number of options"); } static bool getBooleanLoopAttribute(const Loop *TheLoop, StringRef Name) { return getOptionalBoolLoopAttribute(TheLoop, Name).getValueOr(false); } llvm::Optional llvm::getOptionalIntLoopAttribute(Loop *TheLoop, StringRef Name) { const MDOperand *AttrMD = findStringMetadataForLoop(TheLoop, Name).getValueOr(nullptr); if (!AttrMD) return None; ConstantInt *IntMD = mdconst::extract_or_null(AttrMD->get()); if (!IntMD) return None; return IntMD->getSExtValue(); } Optional llvm::makeFollowupLoopID( MDNode *OrigLoopID, ArrayRef FollowupOptions, const char *InheritOptionsExceptPrefix, bool AlwaysNew) { if (!OrigLoopID) { if (AlwaysNew) return nullptr; return None; } assert(OrigLoopID->getOperand(0) == OrigLoopID); bool InheritAllAttrs = !InheritOptionsExceptPrefix; bool InheritSomeAttrs = InheritOptionsExceptPrefix && InheritOptionsExceptPrefix[0] != '\0'; SmallVector MDs; MDs.push_back(nullptr); bool Changed = false; if (InheritAllAttrs || InheritSomeAttrs) { for (const MDOperand &Existing : drop_begin(OrigLoopID->operands(), 1)) { MDNode *Op = cast(Existing.get()); auto InheritThisAttribute = [InheritSomeAttrs, InheritOptionsExceptPrefix](MDNode *Op) { if (!InheritSomeAttrs) return false; // Skip malformatted attribute metadata nodes. if (Op->getNumOperands() == 0) return true; Metadata *NameMD = Op->getOperand(0).get(); if (!isa(NameMD)) return true; StringRef AttrName = cast(NameMD)->getString(); // Do not inherit excluded attributes. return !AttrName.startswith(InheritOptionsExceptPrefix); }; if (InheritThisAttribute(Op)) MDs.push_back(Op); else Changed = true; } } else { // Modified if we dropped at least one attribute. Changed = OrigLoopID->getNumOperands() > 1; } bool HasAnyFollowup = false; for (StringRef OptionName : FollowupOptions) { MDNode *FollowupNode = findOptionMDForLoopID(OrigLoopID, OptionName); if (!FollowupNode) continue; HasAnyFollowup = true; for (const MDOperand &Option : drop_begin(FollowupNode->operands(), 1)) { MDs.push_back(Option.get()); Changed = true; } } // Attributes of the followup loop not specified explicity, so signal to the // transformation pass to add suitable attributes. if (!AlwaysNew && !HasAnyFollowup) return None; // If no attributes were added or remove, the previous loop Id can be reused. if (!AlwaysNew && !Changed) return OrigLoopID; // No attributes is equivalent to having no !llvm.loop metadata at all. if (MDs.size() == 1) return nullptr; // Build the new loop ID. MDTuple *FollowupLoopID = MDNode::get(OrigLoopID->getContext(), MDs); FollowupLoopID->replaceOperandWith(0, FollowupLoopID); return FollowupLoopID; } bool llvm::hasDisableAllTransformsHint(const Loop *L) { return getBooleanLoopAttribute(L, LLVMLoopDisableNonforced); } TransformationMode llvm::hasUnrollTransformation(Loop *L) { if (getBooleanLoopAttribute(L, "llvm.loop.unroll.disable")) return TM_SuppressedByUser; Optional Count = getOptionalIntLoopAttribute(L, "llvm.loop.unroll.count"); if (Count.hasValue()) return Count.getValue() == 1 ? TM_SuppressedByUser : TM_ForcedByUser; if (getBooleanLoopAttribute(L, "llvm.loop.unroll.enable")) return TM_ForcedByUser; if (getBooleanLoopAttribute(L, "llvm.loop.unroll.full")) return TM_ForcedByUser; if (hasDisableAllTransformsHint(L)) return TM_Disable; return TM_Unspecified; } TransformationMode llvm::hasUnrollAndJamTransformation(Loop *L) { if (getBooleanLoopAttribute(L, "llvm.loop.unroll_and_jam.disable")) return TM_SuppressedByUser; Optional Count = getOptionalIntLoopAttribute(L, "llvm.loop.unroll_and_jam.count"); if (Count.hasValue()) return Count.getValue() == 1 ? TM_SuppressedByUser : TM_ForcedByUser; if (getBooleanLoopAttribute(L, "llvm.loop.unroll_and_jam.enable")) return TM_ForcedByUser; if (hasDisableAllTransformsHint(L)) return TM_Disable; return TM_Unspecified; } TransformationMode llvm::hasVectorizeTransformation(Loop *L) { Optional Enable = getOptionalBoolLoopAttribute(L, "llvm.loop.vectorize.enable"); if (Enable == false) return TM_SuppressedByUser; Optional VectorizeWidth = getOptionalIntLoopAttribute(L, "llvm.loop.vectorize.width"); Optional InterleaveCount = getOptionalIntLoopAttribute(L, "llvm.loop.interleave.count"); - if (Enable == true) { - // 'Forcing' vector width and interleave count to one effectively disables - // this tranformation. - if (VectorizeWidth == 1 && InterleaveCount == 1) - return TM_SuppressedByUser; - return TM_ForcedByUser; - } + // 'Forcing' vector width and interleave count to one effectively disables + // this tranformation. + if (Enable == true && VectorizeWidth == 1 && InterleaveCount == 1) + return TM_SuppressedByUser; if (getBooleanLoopAttribute(L, "llvm.loop.isvectorized")) return TM_Disable; + + if (Enable == true) + return TM_ForcedByUser; if (VectorizeWidth == 1 && InterleaveCount == 1) return TM_Disable; if (VectorizeWidth > 1 || InterleaveCount > 1) return TM_Enable; if (hasDisableAllTransformsHint(L)) return TM_Disable; return TM_Unspecified; } TransformationMode llvm::hasDistributeTransformation(Loop *L) { if (getBooleanLoopAttribute(L, "llvm.loop.distribute.enable")) return TM_ForcedByUser; if (hasDisableAllTransformsHint(L)) return TM_Disable; return TM_Unspecified; } TransformationMode llvm::hasLICMVersioningTransformation(Loop *L) { if (getBooleanLoopAttribute(L, "llvm.loop.licm_versioning.disable")) return TM_SuppressedByUser; if (hasDisableAllTransformsHint(L)) return TM_Disable; return TM_Unspecified; } /// Does a BFS from a given node to all of its children inside a given loop. /// The returned vector of nodes includes the starting point. SmallVector llvm::collectChildrenInLoop(DomTreeNode *N, const Loop *CurLoop) { SmallVector Worklist; auto AddRegionToWorklist = [&](DomTreeNode *DTN) { // Only include subregions in the top level loop. BasicBlock *BB = DTN->getBlock(); if (CurLoop->contains(BB)) Worklist.push_back(DTN); }; AddRegionToWorklist(N); for (size_t I = 0; I < Worklist.size(); I++) for (DomTreeNode *Child : Worklist[I]->getChildren()) AddRegionToWorklist(Child); return Worklist; } void llvm::deleteDeadLoop(Loop *L, DominatorTree *DT = nullptr, ScalarEvolution *SE = nullptr, LoopInfo *LI = nullptr) { assert((!DT || L->isLCSSAForm(*DT)) && "Expected LCSSA!"); auto *Preheader = L->getLoopPreheader(); assert(Preheader && "Preheader should exist!"); // Now that we know the removal is safe, remove the loop by changing the // branch from the preheader to go to the single exit block. // // Because we're deleting a large chunk of code at once, the sequence in which // we remove things is very important to avoid invalidation issues. // Tell ScalarEvolution that the loop is deleted. Do this before // deleting the loop so that ScalarEvolution can look at the loop // to determine what it needs to clean up. if (SE) SE->forgetLoop(L); auto *ExitBlock = L->getUniqueExitBlock(); assert(ExitBlock && "Should have a unique exit block!"); assert(L->hasDedicatedExits() && "Loop should have dedicated exits!"); auto *OldBr = dyn_cast(Preheader->getTerminator()); assert(OldBr && "Preheader must end with a branch"); assert(OldBr->isUnconditional() && "Preheader must have a single successor"); // Connect the preheader to the exit block. Keep the old edge to the header // around to perform the dominator tree update in two separate steps // -- #1 insertion of the edge preheader -> exit and #2 deletion of the edge // preheader -> header. // // // 0. Preheader 1. Preheader 2. Preheader // | | | | // V | V | // Header <--\ | Header <--\ | Header <--\ // | | | | | | | | | | | // | V | | | V | | | V | // | Body --/ | | Body --/ | | Body --/ // V V V V V // Exit Exit Exit // // By doing this is two separate steps we can perform the dominator tree // update without using the batch update API. // // Even when the loop is never executed, we cannot remove the edge from the // source block to the exit block. Consider the case where the unexecuted loop // branches back to an outer loop. If we deleted the loop and removed the edge // coming to this inner loop, this will break the outer loop structure (by // deleting the backedge of the outer loop). If the outer loop is indeed a // non-loop, it will be deleted in a future iteration of loop deletion pass. IRBuilder<> Builder(OldBr); Builder.CreateCondBr(Builder.getFalse(), L->getHeader(), ExitBlock); // Remove the old branch. The conditional branch becomes a new terminator. OldBr->eraseFromParent(); // Rewrite phis in the exit block to get their inputs from the Preheader // instead of the exiting block. for (PHINode &P : ExitBlock->phis()) { // Set the zero'th element of Phi to be from the preheader and remove all // other incoming values. Given the loop has dedicated exits, all other // incoming values must be from the exiting blocks. int PredIndex = 0; P.setIncomingBlock(PredIndex, Preheader); // Removes all incoming values from all other exiting blocks (including // duplicate values from an exiting block). // Nuke all entries except the zero'th entry which is the preheader entry. // NOTE! We need to remove Incoming Values in the reverse order as done // below, to keep the indices valid for deletion (removeIncomingValues // updates getNumIncomingValues and shifts all values down into the operand // being deleted). for (unsigned i = 0, e = P.getNumIncomingValues() - 1; i != e; ++i) P.removeIncomingValue(e - i, false); assert((P.getNumIncomingValues() == 1 && P.getIncomingBlock(PredIndex) == Preheader) && "Should have exactly one value and that's from the preheader!"); } // Disconnect the loop body by branching directly to its exit. Builder.SetInsertPoint(Preheader->getTerminator()); Builder.CreateBr(ExitBlock); // Remove the old branch. Preheader->getTerminator()->eraseFromParent(); DomTreeUpdater DTU(DT, DomTreeUpdater::UpdateStrategy::Eager); if (DT) { // Update the dominator tree by informing it about the new edge from the // preheader to the exit. DTU.insertEdge(Preheader, ExitBlock); // Inform the dominator tree about the removed edge. DTU.deleteEdge(Preheader, L->getHeader()); } // Use a map to unique and a vector to guarantee deterministic ordering. llvm::SmallDenseSet, 4> DeadDebugSet; llvm::SmallVector DeadDebugInst; // Given LCSSA form is satisfied, we should not have users of instructions // within the dead loop outside of the loop. However, LCSSA doesn't take // unreachable uses into account. We handle them here. // We could do it after drop all references (in this case all users in the // loop will be already eliminated and we have less work to do but according // to API doc of User::dropAllReferences only valid operation after dropping // references, is deletion. So let's substitute all usages of // instruction from the loop with undef value of corresponding type first. for (auto *Block : L->blocks()) for (Instruction &I : *Block) { auto *Undef = UndefValue::get(I.getType()); for (Value::use_iterator UI = I.use_begin(), E = I.use_end(); UI != E;) { Use &U = *UI; ++UI; if (auto *Usr = dyn_cast(U.getUser())) if (L->contains(Usr->getParent())) continue; // If we have a DT then we can check that uses outside a loop only in // unreachable block. if (DT) assert(!DT->isReachableFromEntry(U) && "Unexpected user in reachable block"); U.set(Undef); } auto *DVI = dyn_cast(&I); if (!DVI) continue; auto Key = DeadDebugSet.find({DVI->getVariable(), DVI->getExpression()}); if (Key != DeadDebugSet.end()) continue; DeadDebugSet.insert({DVI->getVariable(), DVI->getExpression()}); DeadDebugInst.push_back(DVI); } // After the loop has been deleted all the values defined and modified // inside the loop are going to be unavailable. // Since debug values in the loop have been deleted, inserting an undef // dbg.value truncates the range of any dbg.value before the loop where the // loop used to be. This is particularly important for constant values. DIBuilder DIB(*ExitBlock->getModule()); for (auto *DVI : DeadDebugInst) DIB.insertDbgValueIntrinsic( UndefValue::get(Builder.getInt32Ty()), DVI->getVariable(), DVI->getExpression(), DVI->getDebugLoc(), ExitBlock->getFirstNonPHI()); // Remove the block from the reference counting scheme, so that we can // delete it freely later. for (auto *Block : L->blocks()) Block->dropAllReferences(); if (LI) { // Erase the instructions and the blocks without having to worry // about ordering because we already dropped the references. // NOTE: This iteration is safe because erasing the block does not remove // its entry from the loop's block list. We do that in the next section. for (Loop::block_iterator LpI = L->block_begin(), LpE = L->block_end(); LpI != LpE; ++LpI) (*LpI)->eraseFromParent(); // Finally, the blocks from loopinfo. This has to happen late because // otherwise our loop iterators won't work. SmallPtrSet blocks; blocks.insert(L->block_begin(), L->block_end()); for (BasicBlock *BB : blocks) LI->removeBlock(BB); // The last step is to update LoopInfo now that we've eliminated this loop. LI->erase(L); } } Optional llvm::getLoopEstimatedTripCount(Loop *L) { // Only support loops with a unique exiting block, and a latch. if (!L->getExitingBlock()) return None; // Get the branch weights for the loop's backedge. BranchInst *LatchBR = dyn_cast(L->getLoopLatch()->getTerminator()); if (!LatchBR || LatchBR->getNumSuccessors() != 2) return None; assert((LatchBR->getSuccessor(0) == L->getHeader() || LatchBR->getSuccessor(1) == L->getHeader()) && "At least one edge out of the latch must go to the header"); // To estimate the number of times the loop body was executed, we want to // know the number of times the backedge was taken, vs. the number of times // we exited the loop. uint64_t TrueVal, FalseVal; if (!LatchBR->extractProfMetadata(TrueVal, FalseVal)) return None; if (!TrueVal || !FalseVal) return 0; // Divide the count of the backedge by the count of the edge exiting the loop, // rounding to nearest. if (LatchBR->getSuccessor(0) == L->getHeader()) return (TrueVal + (FalseVal / 2)) / FalseVal; else return (FalseVal + (TrueVal / 2)) / TrueVal; } bool llvm::hasIterationCountInvariantInParent(Loop *InnerLoop, ScalarEvolution &SE) { Loop *OuterL = InnerLoop->getParentLoop(); if (!OuterL) return true; // Get the backedge taken count for the inner loop BasicBlock *InnerLoopLatch = InnerLoop->getLoopLatch(); const SCEV *InnerLoopBECountSC = SE.getExitCount(InnerLoop, InnerLoopLatch); if (isa(InnerLoopBECountSC) || !InnerLoopBECountSC->getType()->isIntegerTy()) return false; // Get whether count is invariant to the outer loop ScalarEvolution::LoopDisposition LD = SE.getLoopDisposition(InnerLoopBECountSC, OuterL); if (LD != ScalarEvolution::LoopInvariant) return false; return true; } /// Adds a 'fast' flag to floating point operations. static Value *addFastMathFlag(Value *V) { if (isa(V)) { FastMathFlags Flags; Flags.setFast(); cast(V)->setFastMathFlags(Flags); } return V; } Value *llvm::createMinMaxOp(IRBuilder<> &Builder, RecurrenceDescriptor::MinMaxRecurrenceKind RK, Value *Left, Value *Right) { CmpInst::Predicate P = CmpInst::ICMP_NE; switch (RK) { default: llvm_unreachable("Unknown min/max recurrence kind"); case RecurrenceDescriptor::MRK_UIntMin: P = CmpInst::ICMP_ULT; break; case RecurrenceDescriptor::MRK_UIntMax: P = CmpInst::ICMP_UGT; break; case RecurrenceDescriptor::MRK_SIntMin: P = CmpInst::ICMP_SLT; break; case RecurrenceDescriptor::MRK_SIntMax: P = CmpInst::ICMP_SGT; break; case RecurrenceDescriptor::MRK_FloatMin: P = CmpInst::FCMP_OLT; break; case RecurrenceDescriptor::MRK_FloatMax: P = CmpInst::FCMP_OGT; break; } // We only match FP sequences that are 'fast', so we can unconditionally // set it on any generated instructions. IRBuilder<>::FastMathFlagGuard FMFG(Builder); FastMathFlags FMF; FMF.setFast(); Builder.setFastMathFlags(FMF); Value *Cmp; if (RK == RecurrenceDescriptor::MRK_FloatMin || RK == RecurrenceDescriptor::MRK_FloatMax) Cmp = Builder.CreateFCmp(P, Left, Right, "rdx.minmax.cmp"); else Cmp = Builder.CreateICmp(P, Left, Right, "rdx.minmax.cmp"); Value *Select = Builder.CreateSelect(Cmp, Left, Right, "rdx.minmax.select"); return Select; } // Helper to generate an ordered reduction. Value * llvm::getOrderedReduction(IRBuilder<> &Builder, Value *Acc, Value *Src, unsigned Op, RecurrenceDescriptor::MinMaxRecurrenceKind MinMaxKind, ArrayRef RedOps) { unsigned VF = Src->getType()->getVectorNumElements(); // Extract and apply reduction ops in ascending order: // e.g. ((((Acc + Scl[0]) + Scl[1]) + Scl[2]) + ) ... + Scl[VF-1] Value *Result = Acc; for (unsigned ExtractIdx = 0; ExtractIdx != VF; ++ExtractIdx) { Value *Ext = Builder.CreateExtractElement(Src, Builder.getInt32(ExtractIdx)); if (Op != Instruction::ICmp && Op != Instruction::FCmp) { Result = Builder.CreateBinOp((Instruction::BinaryOps)Op, Result, Ext, "bin.rdx"); } else { assert(MinMaxKind != RecurrenceDescriptor::MRK_Invalid && "Invalid min/max"); Result = createMinMaxOp(Builder, MinMaxKind, Result, Ext); } if (!RedOps.empty()) propagateIRFlags(Result, RedOps); } return Result; } // Helper to generate a log2 shuffle reduction. Value * llvm::getShuffleReduction(IRBuilder<> &Builder, Value *Src, unsigned Op, RecurrenceDescriptor::MinMaxRecurrenceKind MinMaxKind, ArrayRef RedOps) { unsigned VF = Src->getType()->getVectorNumElements(); // VF is a power of 2 so we can emit the reduction using log2(VF) shuffles // and vector ops, reducing the set of values being computed by half each // round. assert(isPowerOf2_32(VF) && "Reduction emission only supported for pow2 vectors!"); Value *TmpVec = Src; SmallVector ShuffleMask(VF, nullptr); for (unsigned i = VF; i != 1; i >>= 1) { // Move the upper half of the vector to the lower half. for (unsigned j = 0; j != i / 2; ++j) ShuffleMask[j] = Builder.getInt32(i / 2 + j); // Fill the rest of the mask with undef. std::fill(&ShuffleMask[i / 2], ShuffleMask.end(), UndefValue::get(Builder.getInt32Ty())); Value *Shuf = Builder.CreateShuffleVector( TmpVec, UndefValue::get(TmpVec->getType()), ConstantVector::get(ShuffleMask), "rdx.shuf"); if (Op != Instruction::ICmp && Op != Instruction::FCmp) { // Floating point operations had to be 'fast' to enable the reduction. TmpVec = addFastMathFlag(Builder.CreateBinOp((Instruction::BinaryOps)Op, TmpVec, Shuf, "bin.rdx")); } else { assert(MinMaxKind != RecurrenceDescriptor::MRK_Invalid && "Invalid min/max"); TmpVec = createMinMaxOp(Builder, MinMaxKind, TmpVec, Shuf); } if (!RedOps.empty()) propagateIRFlags(TmpVec, RedOps); } // The result is in the first element of the vector. return Builder.CreateExtractElement(TmpVec, Builder.getInt32(0)); } /// Create a simple vector reduction specified by an opcode and some /// flags (if generating min/max reductions). Value *llvm::createSimpleTargetReduction( IRBuilder<> &Builder, const TargetTransformInfo *TTI, unsigned Opcode, Value *Src, TargetTransformInfo::ReductionFlags Flags, ArrayRef RedOps) { assert(isa(Src->getType()) && "Type must be a vector"); Value *ScalarUdf = UndefValue::get(Src->getType()->getVectorElementType()); std::function BuildFunc; using RD = RecurrenceDescriptor; RD::MinMaxRecurrenceKind MinMaxKind = RD::MRK_Invalid; // TODO: Support creating ordered reductions. FastMathFlags FMFFast; FMFFast.setFast(); switch (Opcode) { case Instruction::Add: BuildFunc = [&]() { return Builder.CreateAddReduce(Src); }; break; case Instruction::Mul: BuildFunc = [&]() { return Builder.CreateMulReduce(Src); }; break; case Instruction::And: BuildFunc = [&]() { return Builder.CreateAndReduce(Src); }; break; case Instruction::Or: BuildFunc = [&]() { return Builder.CreateOrReduce(Src); }; break; case Instruction::Xor: BuildFunc = [&]() { return Builder.CreateXorReduce(Src); }; break; case Instruction::FAdd: BuildFunc = [&]() { auto Rdx = Builder.CreateFAddReduce(ScalarUdf, Src); cast(Rdx)->setFastMathFlags(FMFFast); return Rdx; }; break; case Instruction::FMul: BuildFunc = [&]() { auto Rdx = Builder.CreateFMulReduce(ScalarUdf, Src); cast(Rdx)->setFastMathFlags(FMFFast); return Rdx; }; break; case Instruction::ICmp: if (Flags.IsMaxOp) { MinMaxKind = Flags.IsSigned ? RD::MRK_SIntMax : RD::MRK_UIntMax; BuildFunc = [&]() { return Builder.CreateIntMaxReduce(Src, Flags.IsSigned); }; } else { MinMaxKind = Flags.IsSigned ? RD::MRK_SIntMin : RD::MRK_UIntMin; BuildFunc = [&]() { return Builder.CreateIntMinReduce(Src, Flags.IsSigned); }; } break; case Instruction::FCmp: if (Flags.IsMaxOp) { MinMaxKind = RD::MRK_FloatMax; BuildFunc = [&]() { return Builder.CreateFPMaxReduce(Src, Flags.NoNaN); }; } else { MinMaxKind = RD::MRK_FloatMin; BuildFunc = [&]() { return Builder.CreateFPMinReduce(Src, Flags.NoNaN); }; } break; default: llvm_unreachable("Unhandled opcode"); break; } if (TTI->useReductionIntrinsic(Opcode, Src->getType(), Flags)) return BuildFunc(); return getShuffleReduction(Builder, Src, Opcode, MinMaxKind, RedOps); } /// Create a vector reduction using a given recurrence descriptor. Value *llvm::createTargetReduction(IRBuilder<> &B, const TargetTransformInfo *TTI, RecurrenceDescriptor &Desc, Value *Src, bool NoNaN) { // TODO: Support in-order reductions based on the recurrence descriptor. using RD = RecurrenceDescriptor; RD::RecurrenceKind RecKind = Desc.getRecurrenceKind(); TargetTransformInfo::ReductionFlags Flags; Flags.NoNaN = NoNaN; switch (RecKind) { case RD::RK_FloatAdd: return createSimpleTargetReduction(B, TTI, Instruction::FAdd, Src, Flags); case RD::RK_FloatMult: return createSimpleTargetReduction(B, TTI, Instruction::FMul, Src, Flags); case RD::RK_IntegerAdd: return createSimpleTargetReduction(B, TTI, Instruction::Add, Src, Flags); case RD::RK_IntegerMult: return createSimpleTargetReduction(B, TTI, Instruction::Mul, Src, Flags); case RD::RK_IntegerAnd: return createSimpleTargetReduction(B, TTI, Instruction::And, Src, Flags); case RD::RK_IntegerOr: return createSimpleTargetReduction(B, TTI, Instruction::Or, Src, Flags); case RD::RK_IntegerXor: return createSimpleTargetReduction(B, TTI, Instruction::Xor, Src, Flags); case RD::RK_IntegerMinMax: { RD::MinMaxRecurrenceKind MMKind = Desc.getMinMaxRecurrenceKind(); Flags.IsMaxOp = (MMKind == RD::MRK_SIntMax || MMKind == RD::MRK_UIntMax); Flags.IsSigned = (MMKind == RD::MRK_SIntMax || MMKind == RD::MRK_SIntMin); return createSimpleTargetReduction(B, TTI, Instruction::ICmp, Src, Flags); } case RD::RK_FloatMinMax: { Flags.IsMaxOp = Desc.getMinMaxRecurrenceKind() == RD::MRK_FloatMax; return createSimpleTargetReduction(B, TTI, Instruction::FCmp, Src, Flags); } default: llvm_unreachable("Unhandled RecKind"); } } void llvm::propagateIRFlags(Value *I, ArrayRef VL, Value *OpValue) { auto *VecOp = dyn_cast(I); if (!VecOp) return; auto *Intersection = (OpValue == nullptr) ? dyn_cast(VL[0]) : dyn_cast(OpValue); if (!Intersection) return; const unsigned Opcode = Intersection->getOpcode(); VecOp->copyIRFlags(Intersection); for (auto *V : VL) { auto *Instr = dyn_cast(V); if (!Instr) continue; if (OpValue == nullptr || Opcode == Instr->getOpcode()) VecOp->andIRFlags(V); } } bool llvm::isKnownNegativeInLoop(const SCEV *S, const Loop *L, ScalarEvolution &SE) { const SCEV *Zero = SE.getZero(S->getType()); return SE.isAvailableAtLoopEntry(S, L) && SE.isLoopEntryGuardedByCond(L, ICmpInst::ICMP_SLT, S, Zero); } bool llvm::isKnownNonNegativeInLoop(const SCEV *S, const Loop *L, ScalarEvolution &SE) { const SCEV *Zero = SE.getZero(S->getType()); return SE.isAvailableAtLoopEntry(S, L) && SE.isLoopEntryGuardedByCond(L, ICmpInst::ICMP_SGE, S, Zero); } bool llvm::cannotBeMinInLoop(const SCEV *S, const Loop *L, ScalarEvolution &SE, bool Signed) { unsigned BitWidth = cast(S->getType())->getBitWidth(); APInt Min = Signed ? APInt::getSignedMinValue(BitWidth) : APInt::getMinValue(BitWidth); auto Predicate = Signed ? ICmpInst::ICMP_SGT : ICmpInst::ICMP_UGT; return SE.isAvailableAtLoopEntry(S, L) && SE.isLoopEntryGuardedByCond(L, Predicate, S, SE.getConstant(Min)); } bool llvm::cannotBeMaxInLoop(const SCEV *S, const Loop *L, ScalarEvolution &SE, bool Signed) { unsigned BitWidth = cast(S->getType())->getBitWidth(); APInt Max = Signed ? APInt::getSignedMaxValue(BitWidth) : APInt::getMaxValue(BitWidth); auto Predicate = Signed ? ICmpInst::ICMP_SLT : ICmpInst::ICMP_ULT; return SE.isAvailableAtLoopEntry(S, L) && SE.isLoopEntryGuardedByCond(L, Predicate, S, SE.getConstant(Max)); } Index: vendor/llvm/dist-release_80/test/CodeGen/AArch64/build-vector-extract.ll =================================================================== --- vendor/llvm/dist-release_80/test/CodeGen/AArch64/build-vector-extract.ll (nonexistent) +++ vendor/llvm/dist-release_80/test/CodeGen/AArch64/build-vector-extract.ll (revision 343794) @@ -0,0 +1,441 @@ +; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py +; RUN: llc < %s -mtriple=aarch64-- | FileCheck %s + +define <2 x i64> @extract0_i32_zext_insert0_i64_undef(<4 x i32> %x) { +; CHECK-LABEL: extract0_i32_zext_insert0_i64_undef: +; CHECK: // %bb.0: +; CHECK-NEXT: movi v1.2d, #0000000000000000 +; CHECK-NEXT: zip1 v0.4s, v0.4s, v1.4s +; CHECK-NEXT: ret + %e = extractelement <4 x i32> %x, i32 0 + %z = zext i32 %e to i64 + %r = insertelement <2 x i64> undef, i64 %z, i32 0 + ret <2 x i64> %r +} + +define <2 x i64> @extract0_i32_zext_insert0_i64_zero(<4 x i32> %x) { +; CHECK-LABEL: extract0_i32_zext_insert0_i64_zero: +; CHECK: // %bb.0: +; CHECK-NEXT: fmov w8, s0 +; CHECK-NEXT: movi v0.2d, #0000000000000000 +; CHECK-NEXT: mov v0.d[0], x8 +; CHECK-NEXT: ret + %e = extractelement <4 x i32> %x, i32 0 + %z = zext i32 %e to i64 + %r = insertelement <2 x i64> zeroinitializer, i64 %z, i32 0 + ret <2 x i64> %r +} + +define <2 x i64> @extract1_i32_zext_insert0_i64_undef(<4 x i32> %x) { +; CHECK-LABEL: extract1_i32_zext_insert0_i64_undef: +; CHECK: // %bb.0: +; CHECK-NEXT: zip1 v0.4s, v0.4s, v0.4s +; CHECK-NEXT: movi v1.2d, #0000000000000000 +; CHECK-NEXT: ext v0.16b, v0.16b, v1.16b, #12 +; CHECK-NEXT: ret + %e = extractelement <4 x i32> %x, i32 1 + %z = zext i32 %e to i64 + %r = insertelement <2 x i64> undef, i64 %z, i32 0 + ret <2 x i64> %r +} + +define <2 x i64> @extract1_i32_zext_insert0_i64_zero(<4 x i32> %x) { +; CHECK-LABEL: extract1_i32_zext_insert0_i64_zero: +; CHECK: // %bb.0: +; CHECK-NEXT: mov w8, v0.s[1] +; CHECK-NEXT: movi v0.2d, #0000000000000000 +; CHECK-NEXT: mov v0.d[0], x8 +; CHECK-NEXT: ret + %e = extractelement <4 x i32> %x, i32 1 + %z = zext i32 %e to i64 + %r = insertelement <2 x i64> zeroinitializer, i64 %z, i32 0 + ret <2 x i64> %r +} + +define <2 x i64> @extract2_i32_zext_insert0_i64_undef(<4 x i32> %x) { +; CHECK-LABEL: extract2_i32_zext_insert0_i64_undef: +; CHECK: // %bb.0: +; CHECK-NEXT: uzp1 v0.4s, v0.4s, v0.4s +; CHECK-NEXT: movi v1.2d, #0000000000000000 +; CHECK-NEXT: ext v0.16b, v0.16b, v1.16b, #12 +; CHECK-NEXT: ret + %e = extractelement <4 x i32> %x, i32 2 + %z = zext i32 %e to i64 + %r = insertelement <2 x i64> undef, i64 %z, i32 0 + ret <2 x i64> %r +} + +define <2 x i64> @extract2_i32_zext_insert0_i64_zero(<4 x i32> %x) { +; CHECK-LABEL: extract2_i32_zext_insert0_i64_zero: +; CHECK: // %bb.0: +; CHECK-NEXT: mov w8, v0.s[2] +; CHECK-NEXT: movi v0.2d, #0000000000000000 +; CHECK-NEXT: mov v0.d[0], x8 +; CHECK-NEXT: ret + %e = extractelement <4 x i32> %x, i32 2 + %z = zext i32 %e to i64 + %r = insertelement <2 x i64> zeroinitializer, i64 %z, i32 0 + ret <2 x i64> %r +} + +define <2 x i64> @extract3_i32_zext_insert0_i64_undef(<4 x i32> %x) { +; CHECK-LABEL: extract3_i32_zext_insert0_i64_undef: +; CHECK: // %bb.0: +; CHECK-NEXT: movi v1.2d, #0000000000000000 +; CHECK-NEXT: ext v0.16b, v0.16b, v1.16b, #12 +; CHECK-NEXT: ret + %e = extractelement <4 x i32> %x, i32 3 + %z = zext i32 %e to i64 + %r = insertelement <2 x i64> undef, i64 %z, i32 0 + ret <2 x i64> %r +} + +define <2 x i64> @extract3_i32_zext_insert0_i64_zero(<4 x i32> %x) { +; CHECK-LABEL: extract3_i32_zext_insert0_i64_zero: +; CHECK: // %bb.0: +; CHECK-NEXT: mov w8, v0.s[3] +; CHECK-NEXT: movi v0.2d, #0000000000000000 +; CHECK-NEXT: mov v0.d[0], x8 +; CHECK-NEXT: ret + %e = extractelement <4 x i32> %x, i32 3 + %z = zext i32 %e to i64 + %r = insertelement <2 x i64> zeroinitializer, i64 %z, i32 0 + ret <2 x i64> %r +} + +define <2 x i64> @extract0_i32_zext_insert1_i64_undef(<4 x i32> %x) { +; CHECK-LABEL: extract0_i32_zext_insert1_i64_undef: +; CHECK: // %bb.0: +; CHECK-NEXT: movi v1.2d, #0000000000000000 +; CHECK-NEXT: zip1 v1.4s, v0.4s, v1.4s +; CHECK-NEXT: ext v0.16b, v0.16b, v1.16b, #8 +; CHECK-NEXT: ret + %e = extractelement <4 x i32> %x, i32 0 + %z = zext i32 %e to i64 + %r = insertelement <2 x i64> undef, i64 %z, i32 1 + ret <2 x i64> %r +} + +define <2 x i64> @extract0_i32_zext_insert1_i64_zero(<4 x i32> %x) { +; CHECK-LABEL: extract0_i32_zext_insert1_i64_zero: +; CHECK: // %bb.0: +; CHECK-NEXT: fmov w8, s0 +; CHECK-NEXT: movi v0.2d, #0000000000000000 +; CHECK-NEXT: mov v0.d[1], x8 +; CHECK-NEXT: ret + %e = extractelement <4 x i32> %x, i32 0 + %z = zext i32 %e to i64 + %r = insertelement <2 x i64> zeroinitializer, i64 %z, i32 1 + ret <2 x i64> %r +} + +define <2 x i64> @extract1_i32_zext_insert1_i64_undef(<4 x i32> %x) { +; CHECK-LABEL: extract1_i32_zext_insert1_i64_undef: +; CHECK: // %bb.0: +; CHECK-NEXT: ext v0.16b, v0.16b, v0.16b, #8 +; CHECK-NEXT: movi v1.2d, #0000000000000000 +; CHECK-NEXT: ext v0.16b, v0.16b, v1.16b, #4 +; CHECK-NEXT: ret + %e = extractelement <4 x i32> %x, i32 1 + %z = zext i32 %e to i64 + %r = insertelement <2 x i64> undef, i64 %z, i32 1 + ret <2 x i64> %r +} + +define <2 x i64> @extract1_i32_zext_insert1_i64_zero(<4 x i32> %x) { +; CHECK-LABEL: extract1_i32_zext_insert1_i64_zero: +; CHECK: // %bb.0: +; CHECK-NEXT: mov w8, v0.s[1] +; CHECK-NEXT: movi v0.2d, #0000000000000000 +; CHECK-NEXT: mov v0.d[1], x8 +; CHECK-NEXT: ret + %e = extractelement <4 x i32> %x, i32 1 + %z = zext i32 %e to i64 + %r = insertelement <2 x i64> zeroinitializer, i64 %z, i32 1 + ret <2 x i64> %r +} + +define <2 x i64> @extract2_i32_zext_insert1_i64_undef(<4 x i32> %x) { +; CHECK-LABEL: extract2_i32_zext_insert1_i64_undef: +; CHECK: // %bb.0: +; CHECK-NEXT: mov v0.s[3], wzr +; CHECK-NEXT: ret + %e = extractelement <4 x i32> %x, i32 2 + %z = zext i32 %e to i64 + %r = insertelement <2 x i64> undef, i64 %z, i32 1 + ret <2 x i64> %r +} + +define <2 x i64> @extract2_i32_zext_insert1_i64_zero(<4 x i32> %x) { +; CHECK-LABEL: extract2_i32_zext_insert1_i64_zero: +; CHECK: // %bb.0: +; CHECK-NEXT: mov w8, v0.s[2] +; CHECK-NEXT: movi v0.2d, #0000000000000000 +; CHECK-NEXT: mov v0.d[1], x8 +; CHECK-NEXT: ret + %e = extractelement <4 x i32> %x, i32 2 + %z = zext i32 %e to i64 + %r = insertelement <2 x i64> zeroinitializer, i64 %z, i32 1 + ret <2 x i64> %r +} + +define <2 x i64> @extract3_i32_zext_insert1_i64_undef(<4 x i32> %x) { +; CHECK-LABEL: extract3_i32_zext_insert1_i64_undef: +; CHECK: // %bb.0: +; CHECK-NEXT: movi v1.2d, #0000000000000000 +; CHECK-NEXT: ext v0.16b, v0.16b, v1.16b, #4 +; CHECK-NEXT: ret + %e = extractelement <4 x i32> %x, i32 3 + %z = zext i32 %e to i64 + %r = insertelement <2 x i64> undef, i64 %z, i32 1 + ret <2 x i64> %r +} + +define <2 x i64> @extract3_i32_zext_insert1_i64_zero(<4 x i32> %x) { +; CHECK-LABEL: extract3_i32_zext_insert1_i64_zero: +; CHECK: // %bb.0: +; CHECK-NEXT: mov w8, v0.s[3] +; CHECK-NEXT: movi v0.2d, #0000000000000000 +; CHECK-NEXT: mov v0.d[1], x8 +; CHECK-NEXT: ret + %e = extractelement <4 x i32> %x, i32 3 + %z = zext i32 %e to i64 + %r = insertelement <2 x i64> zeroinitializer, i64 %z, i32 1 + ret <2 x i64> %r +} + +define <2 x i64> @extract0_i16_zext_insert0_i64_undef(<8 x i16> %x) { +; CHECK-LABEL: extract0_i16_zext_insert0_i64_undef: +; CHECK: // %bb.0: +; CHECK-NEXT: umov w8, v0.h[0] +; CHECK-NEXT: and x8, x8, #0xffff +; CHECK-NEXT: fmov d0, x8 +; CHECK-NEXT: ret + %e = extractelement <8 x i16> %x, i32 0 + %z = zext i16 %e to i64 + %r = insertelement <2 x i64> undef, i64 %z, i32 0 + ret <2 x i64> %r +} + +define <2 x i64> @extract0_i16_zext_insert0_i64_zero(<8 x i16> %x) { +; CHECK-LABEL: extract0_i16_zext_insert0_i64_zero: +; CHECK: // %bb.0: +; CHECK-NEXT: umov w8, v0.h[0] +; CHECK-NEXT: and x8, x8, #0xffff +; CHECK-NEXT: movi v0.2d, #0000000000000000 +; CHECK-NEXT: mov v0.d[0], x8 +; CHECK-NEXT: ret + %e = extractelement <8 x i16> %x, i32 0 + %z = zext i16 %e to i64 + %r = insertelement <2 x i64> zeroinitializer, i64 %z, i32 0 + ret <2 x i64> %r +} + +define <2 x i64> @extract1_i16_zext_insert0_i64_undef(<8 x i16> %x) { +; CHECK-LABEL: extract1_i16_zext_insert0_i64_undef: +; CHECK: // %bb.0: +; CHECK-NEXT: umov w8, v0.h[1] +; CHECK-NEXT: and x8, x8, #0xffff +; CHECK-NEXT: fmov d0, x8 +; CHECK-NEXT: ret + %e = extractelement <8 x i16> %x, i32 1 + %z = zext i16 %e to i64 + %r = insertelement <2 x i64> undef, i64 %z, i32 0 + ret <2 x i64> %r +} + +define <2 x i64> @extract1_i16_zext_insert0_i64_zero(<8 x i16> %x) { +; CHECK-LABEL: extract1_i16_zext_insert0_i64_zero: +; CHECK: // %bb.0: +; CHECK-NEXT: umov w8, v0.h[1] +; CHECK-NEXT: and x8, x8, #0xffff +; CHECK-NEXT: movi v0.2d, #0000000000000000 +; CHECK-NEXT: mov v0.d[0], x8 +; CHECK-NEXT: ret + %e = extractelement <8 x i16> %x, i32 1 + %z = zext i16 %e to i64 + %r = insertelement <2 x i64> zeroinitializer, i64 %z, i32 0 + ret <2 x i64> %r +} + +define <2 x i64> @extract2_i16_zext_insert0_i64_undef(<8 x i16> %x) { +; CHECK-LABEL: extract2_i16_zext_insert0_i64_undef: +; CHECK: // %bb.0: +; CHECK-NEXT: umov w8, v0.h[2] +; CHECK-NEXT: and x8, x8, #0xffff +; CHECK-NEXT: fmov d0, x8 +; CHECK-NEXT: ret + %e = extractelement <8 x i16> %x, i32 2 + %z = zext i16 %e to i64 + %r = insertelement <2 x i64> undef, i64 %z, i32 0 + ret <2 x i64> %r +} + +define <2 x i64> @extract2_i16_zext_insert0_i64_zero(<8 x i16> %x) { +; CHECK-LABEL: extract2_i16_zext_insert0_i64_zero: +; CHECK: // %bb.0: +; CHECK-NEXT: umov w8, v0.h[2] +; CHECK-NEXT: and x8, x8, #0xffff +; CHECK-NEXT: movi v0.2d, #0000000000000000 +; CHECK-NEXT: mov v0.d[0], x8 +; CHECK-NEXT: ret + %e = extractelement <8 x i16> %x, i32 2 + %z = zext i16 %e to i64 + %r = insertelement <2 x i64> zeroinitializer, i64 %z, i32 0 + ret <2 x i64> %r +} + +define <2 x i64> @extract3_i16_zext_insert0_i64_undef(<8 x i16> %x) { +; CHECK-LABEL: extract3_i16_zext_insert0_i64_undef: +; CHECK: // %bb.0: +; CHECK-NEXT: umov w8, v0.h[3] +; CHECK-NEXT: and x8, x8, #0xffff +; CHECK-NEXT: fmov d0, x8 +; CHECK-NEXT: ret + %e = extractelement <8 x i16> %x, i32 3 + %z = zext i16 %e to i64 + %r = insertelement <2 x i64> undef, i64 %z, i32 0 + ret <2 x i64> %r +} + +define <2 x i64> @extract3_i16_zext_insert0_i64_zero(<8 x i16> %x) { +; CHECK-LABEL: extract3_i16_zext_insert0_i64_zero: +; CHECK: // %bb.0: +; CHECK-NEXT: umov w8, v0.h[3] +; CHECK-NEXT: and x8, x8, #0xffff +; CHECK-NEXT: movi v0.2d, #0000000000000000 +; CHECK-NEXT: mov v0.d[0], x8 +; CHECK-NEXT: ret + %e = extractelement <8 x i16> %x, i32 3 + %z = zext i16 %e to i64 + %r = insertelement <2 x i64> zeroinitializer, i64 %z, i32 0 + ret <2 x i64> %r +} + +define <2 x i64> @extract0_i16_zext_insert1_i64_undef(<8 x i16> %x) { +; CHECK-LABEL: extract0_i16_zext_insert1_i64_undef: +; CHECK: // %bb.0: +; CHECK-NEXT: umov w8, v0.h[0] +; CHECK-NEXT: and x8, x8, #0xffff +; CHECK-NEXT: dup v0.2d, x8 +; CHECK-NEXT: ret + %e = extractelement <8 x i16> %x, i32 0 + %z = zext i16 %e to i64 + %r = insertelement <2 x i64> undef, i64 %z, i32 1 + ret <2 x i64> %r +} + +define <2 x i64> @extract0_i16_zext_insert1_i64_zero(<8 x i16> %x) { +; CHECK-LABEL: extract0_i16_zext_insert1_i64_zero: +; CHECK: // %bb.0: +; CHECK-NEXT: umov w8, v0.h[0] +; CHECK-NEXT: and x8, x8, #0xffff +; CHECK-NEXT: movi v0.2d, #0000000000000000 +; CHECK-NEXT: mov v0.d[1], x8 +; CHECK-NEXT: ret + %e = extractelement <8 x i16> %x, i32 0 + %z = zext i16 %e to i64 + %r = insertelement <2 x i64> zeroinitializer, i64 %z, i32 1 + ret <2 x i64> %r +} + +define <2 x i64> @extract1_i16_zext_insert1_i64_undef(<8 x i16> %x) { +; CHECK-LABEL: extract1_i16_zext_insert1_i64_undef: +; CHECK: // %bb.0: +; CHECK-NEXT: umov w8, v0.h[1] +; CHECK-NEXT: and x8, x8, #0xffff +; CHECK-NEXT: dup v0.2d, x8 +; CHECK-NEXT: ret + %e = extractelement <8 x i16> %x, i32 1 + %z = zext i16 %e to i64 + %r = insertelement <2 x i64> undef, i64 %z, i32 1 + ret <2 x i64> %r +} + +define <2 x i64> @extract1_i16_zext_insert1_i64_zero(<8 x i16> %x) { +; CHECK-LABEL: extract1_i16_zext_insert1_i64_zero: +; CHECK: // %bb.0: +; CHECK-NEXT: umov w8, v0.h[1] +; CHECK-NEXT: and x8, x8, #0xffff +; CHECK-NEXT: movi v0.2d, #0000000000000000 +; CHECK-NEXT: mov v0.d[1], x8 +; CHECK-NEXT: ret + %e = extractelement <8 x i16> %x, i32 1 + %z = zext i16 %e to i64 + %r = insertelement <2 x i64> zeroinitializer, i64 %z, i32 1 + ret <2 x i64> %r +} + +define <2 x i64> @extract2_i16_zext_insert1_i64_undef(<8 x i16> %x) { +; CHECK-LABEL: extract2_i16_zext_insert1_i64_undef: +; CHECK: // %bb.0: +; CHECK-NEXT: umov w8, v0.h[2] +; CHECK-NEXT: and x8, x8, #0xffff +; CHECK-NEXT: dup v0.2d, x8 +; CHECK-NEXT: ret + %e = extractelement <8 x i16> %x, i32 2 + %z = zext i16 %e to i64 + %r = insertelement <2 x i64> undef, i64 %z, i32 1 + ret <2 x i64> %r +} + +define <2 x i64> @extract2_i16_zext_insert1_i64_zero(<8 x i16> %x) { +; CHECK-LABEL: extract2_i16_zext_insert1_i64_zero: +; CHECK: // %bb.0: +; CHECK-NEXT: umov w8, v0.h[2] +; CHECK-NEXT: and x8, x8, #0xffff +; CHECK-NEXT: movi v0.2d, #0000000000000000 +; CHECK-NEXT: mov v0.d[1], x8 +; CHECK-NEXT: ret + %e = extractelement <8 x i16> %x, i32 2 + %z = zext i16 %e to i64 + %r = insertelement <2 x i64> zeroinitializer, i64 %z, i32 1 + ret <2 x i64> %r +} + +define <2 x i64> @extract3_i16_zext_insert1_i64_undef(<8 x i16> %x) { +; CHECK-LABEL: extract3_i16_zext_insert1_i64_undef: +; CHECK: // %bb.0: +; CHECK-NEXT: umov w8, v0.h[3] +; CHECK-NEXT: and x8, x8, #0xffff +; CHECK-NEXT: dup v0.2d, x8 +; CHECK-NEXT: ret + %e = extractelement <8 x i16> %x, i32 3 + %z = zext i16 %e to i64 + %r = insertelement <2 x i64> undef, i64 %z, i32 1 + ret <2 x i64> %r +} + +define <2 x i64> @extract3_i16_zext_insert1_i64_zero(<8 x i16> %x) { +; CHECK-LABEL: extract3_i16_zext_insert1_i64_zero: +; CHECK: // %bb.0: +; CHECK-NEXT: umov w8, v0.h[3] +; CHECK-NEXT: and x8, x8, #0xffff +; CHECK-NEXT: movi v0.2d, #0000000000000000 +; CHECK-NEXT: mov v0.d[1], x8 +; CHECK-NEXT: ret + %e = extractelement <8 x i16> %x, i32 3 + %z = zext i16 %e to i64 + %r = insertelement <2 x i64> zeroinitializer, i64 %z, i32 1 + ret <2 x i64> %r +} + +; This would crash because we did not expect to create +; a shuffle for a vector where the source operand is +; not the same size as the result. +; TODO: Should we handle this pattern? Ie, is moving to/from +; registers the optimal code? + +define <4 x i32> @larger_bv_than_source(<4 x i16> %t0) { +; CHECK-LABEL: larger_bv_than_source: +; CHECK: // %bb.0: +; CHECK-NEXT: // kill: def $d0 killed $d0 def $q0 +; CHECK-NEXT: umov w8, v0.h[2] +; CHECK-NEXT: fmov s0, w8 +; CHECK-NEXT: ret + %t1 = extractelement <4 x i16> %t0, i32 2 + %vgetq_lane = zext i16 %t1 to i32 + %t2 = insertelement <4 x i32> undef, i32 %vgetq_lane, i64 0 + ret <4 x i32> %t2 +} + Index: vendor/llvm/dist-release_80/test/CodeGen/AArch64/eh_recoverfp.ll =================================================================== --- vendor/llvm/dist-release_80/test/CodeGen/AArch64/eh_recoverfp.ll (nonexistent) +++ vendor/llvm/dist-release_80/test/CodeGen/AArch64/eh_recoverfp.ll (revision 343794) @@ -0,0 +1,11 @@ +; RUN: llc -mtriple arm64-windows %s -o - 2>&1 | FileCheck %s + +define i8* @foo(i8* %a) { +; CHECK-LABEL: foo +; CHECK-NOT: llvm.x86.seh.recoverfp + %1 = call i8* @llvm.x86.seh.recoverfp(i8* bitcast (i32 ()* @f to i8*), i8* %a) + ret i8* %1 +} + +declare i8* @llvm.x86.seh.recoverfp(i8*, i8*) +declare i32 @f() Index: vendor/llvm/dist-release_80/test/CodeGen/AArch64/speculation-hardening-loads.ll =================================================================== --- vendor/llvm/dist-release_80/test/CodeGen/AArch64/speculation-hardening-loads.ll (revision 343793) +++ vendor/llvm/dist-release_80/test/CodeGen/AArch64/speculation-hardening-loads.ll (revision 343794) @@ -1,157 +1,157 @@ ; RUN: llc < %s -verify-machineinstrs -mtriple=aarch64-none-linux-gnu | FileCheck %s --dump-input-on-failure define i128 @ldp_single_csdb(i128* %p) speculative_load_hardening { entry: %0 = load i128, i128* %p, align 16 ret i128 %0 ; CHECK-LABEL: ldp_single_csdb ; CHECK: ldp x8, x1, [x0] ; CHECK-NEXT: cmp sp, #0 ; CHECK-NEXT: csetm x16, ne ; CHECK-NEXT: and x8, x8, x16 ; CHECK-NEXT: and x1, x1, x16 ; CHECK-NEXT: csdb -; CHECK-NEXT: mov x17, sp -; CHECK-NEXT: and x17, x17, x16 +; CHECK-NEXT: mov [[TMPREG:x[0-9]+]], sp +; CHECK-NEXT: and [[TMPREG]], [[TMPREG]], x16 ; CHECK-NEXT: mov x0, x8 -; CHECK-NEXT: mov sp, x17 +; CHECK-NEXT: mov sp, [[TMPREG]] ; CHECK-NEXT: ret } define double @ld_double(double* %p) speculative_load_hardening { entry: %0 = load double, double* %p, align 8 ret double %0 ; Checking that the address laoded from is masked for a floating point load. ; CHECK-LABEL: ld_double ; CHECK: cmp sp, #0 ; CHECK-NEXT: csetm x16, ne ; CHECK-NEXT: and x0, x0, x16 ; CHECK-NEXT: csdb ; CHECK-NEXT: ldr d0, [x0] -; CHECK-NEXT: mov x17, sp -; CHECK-NEXT: and x17, x17, x16 -; CHECK-NEXT: mov sp, x17 +; CHECK-NEXT: mov [[TMPREG:x[0-9]+]], sp +; CHECK-NEXT: and [[TMPREG]], [[TMPREG]], x16 +; CHECK-NEXT: mov sp, [[TMPREG]] ; CHECK-NEXT: ret } define i32 @csdb_emitted_for_subreg_use(i64* %p, i32 %b) speculative_load_hardening { entry: %X = load i64, i64* %p, align 8 %X_trunc = trunc i64 %X to i32 %add = add i32 %b, %X_trunc %iszero = icmp eq i64 %X, 0 %ret = select i1 %iszero, i32 %b, i32 %add ret i32 %ret ; Checking that the address laoded from is masked for a floating point load. ; CHECK-LABEL: csdb_emitted_for_subreg_use ; CHECK: ldr x8, [x0] ; CHECK-NEXT: cmp sp, #0 ; CHECK-NEXT: csetm x16, ne ; CHECK-NEXT: and x8, x8, x16 ; csdb instruction must occur before the add instruction with w8 as operand. ; CHECK-NEXT: csdb -; CHECK-NEXT: mov x17, sp ; CHECK-NEXT: add w9, w1, w8 ; CHECK-NEXT: cmp x8, #0 -; CHECK-NEXT: and x17, x17, x16 ; CHECK-NEXT: csel w0, w1, w9, eq -; CHECK-NEXT: mov sp, x17 +; CHECK-NEXT: mov [[TMPREG:x[0-9]+]], sp +; CHECK-NEXT: and [[TMPREG]], [[TMPREG]], x16 +; CHECK-NEXT: mov sp, [[TMPREG]] ; CHECK-NEXT: ret } define i64 @csdb_emitted_for_superreg_use(i32* %p, i64 %b) speculative_load_hardening { entry: %X = load i32, i32* %p, align 4 %X_ext = zext i32 %X to i64 %add = add i64 %b, %X_ext %iszero = icmp eq i32 %X, 0 %ret = select i1 %iszero, i64 %b, i64 %add ret i64 %ret ; Checking that the address laoded from is masked for a floating point load. ; CHECK-LABEL: csdb_emitted_for_superreg_use ; CHECK: ldr w8, [x0] ; CHECK-NEXT: cmp sp, #0 ; CHECK-NEXT: csetm x16, ne ; CHECK-NEXT: and w8, w8, w16 ; csdb instruction must occur before the add instruction with x8 as operand. ; CHECK-NEXT: csdb -; CHECK-NEXT: mov x17, sp ; CHECK-NEXT: add x9, x1, x8 ; CHECK-NEXT: cmp w8, #0 -; CHECK-NEXT: and x17, x17, x16 ; CHECK-NEXT: csel x0, x1, x9, eq -; CHECK-NEXT: mov sp, x17 +; CHECK-NEXT: mov [[TMPREG:x[0-9]+]], sp +; CHECK-NEXT: and [[TMPREG]], [[TMPREG]], x16 +; CHECK-NEXT: mov sp, [[TMPREG]] ; CHECK-NEXT: ret } define i64 @no_masking_with_full_control_flow_barriers(i64 %a, i64 %b, i64* %p) speculative_load_hardening { ; CHECK-LABEL: no_masking_with_full_control_flow_barriers ; CHECK: dsb sy ; CHECK: isb entry: %0 = tail call i64 asm "autia1716", "={x17},{x16},0"(i64 %b, i64 %a) %X = load i64, i64* %p, align 8 %ret = add i64 %X, %0 ; CHECK-NOT: csdb ; CHECK-NOT: and ; CHECK: ret ret i64 %ret } define void @f_implicitdef_vector_load(<4 x i32>* %dst, <2 x i32>* %src) speculative_load_hardening { entry: %0 = load <2 x i32>, <2 x i32>* %src, align 8 %shuffle = shufflevector <2 x i32> %0, <2 x i32> undef, <4 x i32> store <4 x i32> %shuffle, <4 x i32>* %dst, align 4 ret void ; CHECK-LABEL: f_implicitdef_vector_load ; CHECK: cmp sp, #0 ; CHECK-NEXT: csetm x16, ne ; CHECK-NEXT: and x1, x1, x16 ; CHECK-NEXT: csdb ; CHECK-NEXT: ldr d0, [x1] -; CHECK-NEXT: mov x17, sp -; CHECK-NEXT: and x17, x17, x16 ; CHECK-NEXT: mov v0.d[1], v0.d[0] ; CHECK-NEXT: str q0, [x0] -; CHECK-NEXT: mov sp, x17 +; CHECK-NEXT: mov [[TMPREG:x[0-9]+]], sp +; CHECK-NEXT: and [[TMPREG]], [[TMPREG]], x16 +; CHECK-NEXT: mov sp, [[TMPREG]] ; CHECK-NEXT: ret } define <2 x double> @f_usedefvectorload(double* %a, double* %b) speculative_load_hardening { entry: ; CHECK-LABEL: f_usedefvectorload ; CHECK: cmp sp, #0 ; CHECK-NEXT: csetm x16, ne ; CHECK-NEXT: movi v0.2d, #0000000000000000 ; CHECK-NEXT: and x1, x1, x16 ; CHECK-NEXT: csdb ; CHECK-NEXT: ld1 { v0.d }[0], [x1] -; CHECK-NEXT: mov x17, sp -; CHECK-NEXT: and x17, x17, x16 -; CHECK-NEXT: mov sp, x17 +; CHECK-NEXT: mov [[TMPREG:x[0-9]+]], sp +; CHECK-NEXT: and [[TMPREG]], [[TMPREG]], x16 +; CHECK-NEXT: mov sp, [[TMPREG]] ; CHECK-NEXT: ret %0 = load double, double* %b, align 16 %vld1_lane = insertelement <2 x double> , double %0, i32 0 ret <2 x double> %vld1_lane } define i32 @deadload() speculative_load_hardening { entry: ; CHECK-LABEL: deadload ; CHECK: cmp sp, #0 ; CHECK-NEXT: csetm x16, ne ; CHECK-NEXT: sub sp, sp, #16 ; CHECK-NEXT: .cfi_def_cfa_offset 16 ; CHECK-NEXT: ldr w8, [sp, #12] ; CHECK-NEXT: add sp, sp, #16 -; CHECK-NEXT: mov x17, sp -; CHECK-NEXT: and x17, x17, x16 -; CHECK-NEXT: mov sp, x17 +; CHECK-NEXT: mov [[TMPREG:x[0-9]+]], sp +; CHECK-NEXT: and [[TMPREG]], [[TMPREG]], x16 +; CHECK-NEXT: mov sp, [[TMPREG]] ; CHECK-NEXT: ret %a = alloca i32, align 4 %val = load volatile i32, i32* %a, align 4 ret i32 undef } Index: vendor/llvm/dist-release_80/test/CodeGen/AArch64/speculation-hardening.ll =================================================================== --- vendor/llvm/dist-release_80/test/CodeGen/AArch64/speculation-hardening.ll (revision 343793) +++ vendor/llvm/dist-release_80/test/CodeGen/AArch64/speculation-hardening.ll (revision 343794) @@ -1,156 +1,164 @@ -; RUN: sed -e 's/SLHATTR/speculative_load_hardening/' %s | llc -verify-machineinstrs -mtriple=aarch64-none-linux-gnu | FileCheck %s --check-prefixes=CHECK,SLH --dump-input-on-failure -; RUN: sed -e 's/SLHATTR//' %s | llc -verify-machineinstrs -mtriple=aarch64-none-linux-gnu | FileCheck %s --check-prefixes=CHECK,NOSLH --dump-input-on-failure -; RUN: sed -e 's/SLHATTR/speculative_load_hardening/' %s | llc -verify-machineinstrs -mtriple=aarch64-none-linux-gnu -global-isel | FileCheck %s --check-prefixes=CHECK,SLH --dump-input-on-failure -; RUN sed -e 's/SLHATTR//' %s | llc -verify-machineinstrs -mtriple=aarch64-none-linux-gnu -global-isel | FileCheck %s --check-prefixes=CHECK,NOSLH --dump-input-on-failure -; RUN: sed -e 's/SLHATTR/speculative_load_hardening/' %s | llc -verify-machineinstrs -mtriple=aarch64-none-linux-gnu -fast-isel | FileCheck %s --check-prefixes=CHECK,SLH --dump-input-on-failure -; RUN: sed -e 's/SLHATTR//' %s | llc -verify-machineinstrs -mtriple=aarch64-none-linux-gnu -fast-isel | FileCheck %s --check-prefixes=CHECK,NOSLH --dump-input-on-failure +; RUN: sed -e 's/SLHATTR/speculative_load_hardening/' %s | llc -verify-machineinstrs -mtriple=aarch64-none-linux-gnu | FileCheck %s --check-prefixes=CHECK,SLH,NOGISELSLH --dump-input-on-failure +; RUN: sed -e 's/SLHATTR//' %s | llc -verify-machineinstrs -mtriple=aarch64-none-linux-gnu | FileCheck %s --check-prefixes=CHECK,NOSLH,NOGISELNOSLH --dump-input-on-failure +; RUN: sed -e 's/SLHATTR/speculative_load_hardening/' %s | llc -verify-machineinstrs -mtriple=aarch64-none-linux-gnu -global-isel | FileCheck %s --check-prefixes=CHECK,SLH,GISELSLH --dump-input-on-failure +; RUN sed -e 's/SLHATTR//' %s | llc -verify-machineinstrs -mtriple=aarch64-none-linux-gnu -global-isel | FileCheck %s --check-prefixes=CHECK,NOSLH,GISELNOSLH --dump-input-on-failure +; RUN: sed -e 's/SLHATTR/speculative_load_hardening/' %s | llc -verify-machineinstrs -mtriple=aarch64-none-linux-gnu -fast-isel | FileCheck %s --check-prefixes=CHECK,SLH,NOGISELSLH --dump-input-on-failure +; RUN: sed -e 's/SLHATTR//' %s | llc -verify-machineinstrs -mtriple=aarch64-none-linux-gnu -fast-isel | FileCheck %s --check-prefixes=CHECK,NOSLH,NOGISELNOSLH --dump-input-on-failure define i32 @f(i8* nocapture readonly %p, i32 %i, i32 %N) local_unnamed_addr SLHATTR { ; CHECK-LABEL: f entry: ; SLH: cmp sp, #0 ; SLH: csetm x16, ne ; NOSLH-NOT: cmp sp, #0 ; NOSLH-NOT: csetm x16, ne -; SLH: mov x17, sp -; SLH: and x17, x17, x16 -; SLH: mov sp, x17 -; NOSLH-NOT: mov x17, sp -; NOSLH-NOT: and x17, x17, x16 -; NOSLH-NOT: mov sp, x17 +; SLH: mov [[TMPREG:x[0-9]+]], sp +; SLH: and [[TMPREG]], [[TMPREG]], x16 +; SLH: mov sp, [[TMPREG]] +; NOSLH-NOT: mov [[TMPREG:x[0-9]+]], sp +; NOSLH-NOT: and [[TMPREG]], [[TMPREG]], x16 +; NOSLH-NOT: mov sp, [[TMPREG]] %call = tail call i32 @tail_callee(i32 %i) ; SLH: cmp sp, #0 ; SLH: csetm x16, ne ; NOSLH-NOT: cmp sp, #0 ; NOSLH-NOT: csetm x16, ne %cmp = icmp slt i32 %call, %N br i1 %cmp, label %if.then, label %return ; GlobalISel lowers the branch to a b.ne sometimes instead of b.ge as expected.. ; CHECK: b.[[COND:(ge)|(lt)|(ne)]] if.then: ; preds = %entry ; NOSLH-NOT: csel x16, x16, xzr, {{(lt)|(ge)|(eq)}} ; SLH-DAG: csel x16, x16, xzr, {{(lt)|(ge)|(eq)}} %idxprom = sext i32 %i to i64 %arrayidx = getelementptr inbounds i8, i8* %p, i64 %idxprom %0 = load i8, i8* %arrayidx, align 1 ; CHECK-DAG: ldrb [[LOADED:w[0-9]+]], %conv = zext i8 %0 to i32 br label %return ; SLH-DAG: csel x16, x16, xzr, [[COND]] ; NOSLH-NOT: csel x16, x16, xzr, [[COND]] return: ; preds = %entry, %if.then %retval.0 = phi i32 [ %conv, %if.then ], [ 0, %entry ] -; SLH: mov x17, sp -; SLH: and x17, x17, x16 -; SLH: mov sp, x17 -; NOSLH-NOT: mov x17, sp -; NOSLH-NOT: and x17, x17, x16 -; NOSLH-NOT: mov sp, x17 +; SLH: mov [[TMPREG:x[0-9]+]], sp +; SLH: and [[TMPREG]], [[TMPREG]], x16 +; SLH: mov sp, [[TMPREG]] +; NOSLH-NOT: mov [[TMPREG:x[0-9]+]], sp +; NOSLH-NOT: and [[TMPREG]], [[TMPREG]], x16 +; NOSLH-NOT: mov sp, [[TMPREG]] ret i32 %retval.0 } ; Make sure that for a tail call, taint doesn't get put into SP twice. define i32 @tail_caller(i32 %a) local_unnamed_addr SLHATTR { ; CHECK-LABEL: tail_caller: -; SLH: mov x17, sp -; SLH: and x17, x17, x16 -; SLH: mov sp, x17 -; NOSLH-NOT: mov x17, sp -; NOSLH-NOT: and x17, x17, x16 -; NOSLH-NOT: mov sp, x17 +; NOGISELSLH: mov [[TMPREG:x[0-9]+]], sp +; NOGISELSLH: and [[TMPREG]], [[TMPREG]], x16 +; NOGISELSLH: mov sp, [[TMPREG]] +; NOGISELNOSLH-NOT: mov [[TMPREG:x[0-9]+]], sp +; NOGISELNOSLH-NOT: and [[TMPREG]], [[TMPREG]], x16 +; NOGISELNOSLH-NOT: mov sp, [[TMPREG]] +; GISELSLH: mov [[TMPREG:x[0-9]+]], sp +; GISELSLH: and [[TMPREG]], [[TMPREG]], x16 +; GISELSLH: mov sp, [[TMPREG]] +; GISELNOSLH-NOT: mov [[TMPREG:x[0-9]+]], sp +; GISELNOSLH-NOT: and [[TMPREG]], [[TMPREG]], x16 +; GISELNOSLH-NOT: mov sp, [[TMPREG]] ; GlobalISel doesn't optimize tail calls (yet?), so only check that ; cross-call taint register setup code is missing if a tail call was ; actually produced. -; SLH: {{(bl tail_callee[[:space:]] cmp sp, #0)|(b tail_callee)}} -; SLH-NOT: cmp sp, #0 +; NOGISELSLH: b tail_callee +; GISELSLH: bl tail_callee +; GISELSLH: cmp sp, #0 +; SLH-NOT: cmp sp, #0 %call = tail call i32 @tail_callee(i32 %a) ret i32 %call } declare i32 @tail_callee(i32) local_unnamed_addr ; Verify that no cb(n)z/tb(n)z instructions are produced when implementing ; SLH define i32 @compare_branch_zero(i32, i32) SLHATTR { ; CHECK-LABEL: compare_branch_zero %3 = icmp eq i32 %0, 0 br i1 %3, label %then, label %else ;SLH-NOT: cb{{n?}}z ;NOSLH: cb{{n?}}z then: %4 = sdiv i32 5, %1 ret i32 %4 else: %5 = sdiv i32 %1, %0 ret i32 %5 } define i32 @test_branch_zero(i32, i32) SLHATTR { ; CHECK-LABEL: test_branch_zero %3 = and i32 %0, 16 %4 = icmp eq i32 %3, 0 br i1 %4, label %then, label %else ;SLH-NOT: tb{{n?}}z ;NOSLH: tb{{n?}}z then: %5 = sdiv i32 5, %1 ret i32 %5 else: %6 = sdiv i32 %1, %0 ret i32 %6 } define i32 @landingpad(i32 %l0, i32 %l1) SLHATTR personality i8* bitcast (i32 (...)* @__gxx_personality_v0 to i8*) { ; CHECK-LABEL: landingpad entry: ; SLH: cmp sp, #0 ; SLH: csetm x16, ne ; NOSLH-NOT: cmp sp, #0 ; NOSLH-NOT: csetm x16, ne ; CHECK: bl _Z10throwing_fv invoke void @_Z10throwing_fv() to label %exit unwind label %lpad ; SLH: cmp sp, #0 ; SLH: csetm x16, ne lpad: %l4 = landingpad { i8*, i32 } catch i8* null ; SLH: cmp sp, #0 ; SLH: csetm x16, ne ; NOSLH-NOT: cmp sp, #0 ; NOSLH-NOT: csetm x16, ne %l5 = extractvalue { i8*, i32 } %l4, 0 %l6 = tail call i8* @__cxa_begin_catch(i8* %l5) %l7 = icmp sgt i32 %l0, %l1 br i1 %l7, label %then, label %else ; GlobalISel lowers the branch to a b.ne sometimes instead of b.ge as expected.. ; CHECK: b.[[COND:(le)|(gt)|(ne)]] then: ; SLH-DAG: csel x16, x16, xzr, [[COND]] %l9 = sdiv i32 %l0, %l1 br label %postif else: ; SLH-DAG: csel x16, x16, xzr, {{(gt)|(le)|(eq)}} %l11 = sdiv i32 %l1, %l0 br label %postif postif: %l13 = phi i32 [ %l9, %then ], [ %l11, %else ] tail call void @__cxa_end_catch() br label %exit exit: %l15 = phi i32 [ %l13, %postif ], [ 0, %entry ] ret i32 %l15 } declare i32 @__gxx_personality_v0(...) declare void @_Z10throwing_fv() local_unnamed_addr declare i8* @__cxa_begin_catch(i8*) local_unnamed_addr declare void @__cxa_end_catch() local_unnamed_addr Index: vendor/llvm/dist-release_80/test/CodeGen/AArch64/speculation-hardening.mir =================================================================== --- vendor/llvm/dist-release_80/test/CodeGen/AArch64/speculation-hardening.mir (revision 343793) +++ vendor/llvm/dist-release_80/test/CodeGen/AArch64/speculation-hardening.mir (revision 343794) @@ -1,117 +1,202 @@ # RUN: llc -verify-machineinstrs -mtriple=aarch64-none-linux-gnu \ # RUN: -start-before aarch64-speculation-hardening -o - %s \ # RUN: | FileCheck %s --dump-input-on-failure # Check that the speculation hardening pass generates code as expected for # basic blocks ending with a variety of branch patterns: # - (1) no branches (fallthrough) # - (2) one unconditional branch # - (3) one conditional branch + fall-through # - (4) one conditional branch + one unconditional branch # - other direct branches don't seem to be generated by the AArch64 codegen --- | define void @nobranch_fallthrough(i32 %a, i32 %b) speculative_load_hardening { ret void } define void @uncondbranch(i32 %a, i32 %b) speculative_load_hardening { ret void } define void @condbranch_fallthrough(i32 %a, i32 %b) speculative_load_hardening { ret void } define void @condbranch_uncondbranch(i32 %a, i32 %b) speculative_load_hardening { ret void } define void @indirectbranch(i32 %a, i32 %b) speculative_load_hardening { ret void } + ; Also check that a non-default temporary register gets picked correctly to + ; transfer the SP to to and it with the taint register when the default + ; temporary isn't available. + define void @indirect_call_x17(i32 %a, i32 %b) speculative_load_hardening { + ret void + } + @g = common dso_local local_unnamed_addr global i64 (...)* null, align 8 + define void @indirect_tailcall_x17(i32 %a, i32 %b) speculative_load_hardening { + ret void + } + define void @indirect_call_lr(i32 %a, i32 %b) speculative_load_hardening { + ret void + } + define void @RS_cannot_find_available_regs() speculative_load_hardening { + ret void + } ... --- name: nobranch_fallthrough tracksRegLiveness: true body: | ; CHECK-LABEL: nobranch_fallthrough bb.0: successors: %bb.1 liveins: $w0, $w1 ; CHECK-NOT: csel bb.1: liveins: $w0 RET undef $lr, implicit $w0 ... --- name: uncondbranch tracksRegLiveness: true body: | ; CHECK-LABEL: uncondbranch bb.0: successors: %bb.1 liveins: $w0, $w1 B %bb.1 ; CHECK-NOT: csel bb.1: liveins: $w0 RET undef $lr, implicit $w0 ... --- name: condbranch_fallthrough tracksRegLiveness: true body: | ; CHECK-LABEL: condbranch_fallthrough bb.0: successors: %bb.1, %bb.2 liveins: $w0, $w1 $wzr = SUBSWrs renamable $w0, renamable $w1, 0, implicit-def $nzcv, implicit-def $nzcv Bcc 11, %bb.2, implicit $nzcv ; CHECK: b.lt [[BB_LT_T:\.LBB[0-9_]+]] bb.1: liveins: $nzcv, $w0 ; CHECK: csel x16, x16, xzr, ge RET undef $lr, implicit $w0 bb.2: liveins: $nzcv, $w0 ; CHECK: csel x16, x16, xzr, lt RET undef $lr, implicit $w0 ... --- name: condbranch_uncondbranch tracksRegLiveness: true body: | ; CHECK-LABEL: condbranch_uncondbranch bb.0: successors: %bb.1, %bb.2 liveins: $w0, $w1 $wzr = SUBSWrs renamable $w0, renamable $w1, 0, implicit-def $nzcv, implicit-def $nzcv Bcc 11, %bb.2, implicit $nzcv B %bb.1, implicit $nzcv ; CHECK: b.lt [[BB_LT_T:\.LBB[0-9_]+]] bb.1: liveins: $nzcv, $w0 ; CHECK: csel x16, x16, xzr, ge RET undef $lr, implicit $w0 bb.2: liveins: $nzcv, $w0 ; CHECK: csel x16, x16, xzr, lt RET undef $lr, implicit $w0 ... --- name: indirectbranch tracksRegLiveness: true body: | ; Check that no instrumentation is done on indirect branches (for now). ; CHECK-LABEL: indirectbranch bb.0: successors: %bb.1, %bb.2 liveins: $x0 BR $x0 bb.1: liveins: $x0 ; CHECK-NOT: csel RET undef $lr, implicit $x0 bb.2: liveins: $x0 ; CHECK-NOT: csel RET undef $lr, implicit $x0 +... +--- +name: indirect_call_x17 +tracksRegLiveness: true +body: | + bb.0: + liveins: $x17 + ; CHECK-LABEL: indirect_call_x17 + ; CHECK: mov x0, sp + ; CHECK: and x0, x0, x16 + ; CHECK: mov sp, x0 + ; CHECK: blr x17 + BLR killed renamable $x17, implicit-def dead $lr, implicit $sp + RET undef $lr, implicit undef $w0 +... +--- +name: indirect_tailcall_x17 +tracksRegLiveness: true +body: | + bb.0: + liveins: $x0 + ; CHECK-LABEL: indirect_tailcall_x17 + ; CHECK: mov x1, sp + ; CHECK: and x1, x1, x16 + ; CHECK: mov sp, x1 + ; CHECK: br x17 + $x8 = ADRP target-flags(aarch64-page) @g + $x17 = LDRXui killed $x8, target-flags(aarch64-pageoff, aarch64-nc) @g + TCRETURNri killed $x17, 0, implicit $sp, implicit $x0 +... +--- +name: indirect_call_lr +tracksRegLiveness: true +body: | + bb.0: + ; CHECK-LABEL: indirect_call_lr + ; CHECK: mov x1, sp + ; CHECK-NEXT: and x1, x1, x16 + ; CHECK-NEXT: mov sp, x1 + ; CHECK-NEXT: blr x30 + liveins: $x0, $lr + BLR killed renamable $lr, implicit-def dead $lr, implicit $sp, implicit-def $sp, implicit-def $w0 + $w0 = nsw ADDWri killed $w0, 1, 0 + RET undef $lr, implicit $w0 +... +--- +name: RS_cannot_find_available_regs +tracksRegLiveness: true +body: | + bb.0: + ; In the rare case when no free temporary register is available for the + ; propagate taint-to-sp operation, just put in a full speculation barrier + ; (isb+dsb sy) at the start of the basic block. And don't put masks on + ; instructions for the rest of the basic block, since speculation in that + ; basic block was already done, so no need to do masking. + ; CHECK-LABEL: RS_cannot_find_available_regs + ; CHECK: dsb sy + ; CHECK-NEXT: isb + ; CHECK-NEXT: ldr x0, [x0] + ; The following 2 instructions come from propagating the taint encoded in + ; sp at function entry to x16. It turns out the taint info in x16 is not + ; used in this function, so those instructions could be optimized away. An + ; optimization for later if it turns out this situation occurs often enough. + ; CHECK-NEXT: cmp sp, #0 + ; CHECK-NEXT: csetm x16, ne + ; CHECK-NEXT: ret + liveins: $x0, $x1, $x2, $x3, $x4, $x5, $x6, $x7, $x8, $x9, $x10, $x11, $x12, $x13, $x14, $x15, $x17, $x18, $x19, $x20, $x21, $x22, $x23, $x24, $x25, $x26, $x27, $x28, $fp, $lr + $x0 = LDRXui killed $x0, 0 + RET undef $lr, implicit $x0 ... Index: vendor/llvm/dist-release_80/test/CodeGen/Mips/cconv/vector.ll =================================================================== --- vendor/llvm/dist-release_80/test/CodeGen/Mips/cconv/vector.ll (revision 343793) +++ vendor/llvm/dist-release_80/test/CodeGen/Mips/cconv/vector.ll (revision 343794) @@ -1,6897 +1,6897 @@ ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py ; RUN: llc < %s -mtriple=mips-unknown-linux-gnu -mcpu=mips32 -disable-mips-delay-filler | FileCheck %s --check-prefixes=ALL,MIPS32,MIPS32EB -; RUN: llc < %s -mtriple=mips64-unknown-linux-gnu -relocation-model=pic -mcpu=mips64 -disable-mips-delay-filler | FileCheck %s --check-prefixes=ALL,MIPS64,MIPS64EB +; RUN: llc < %s -mtriple=mips64-unknown-linux-gnu -relocation-model=pic -mcpu=mips64 -disable-mips-delay-filler -mips-jalr-reloc=false | FileCheck %s --check-prefixes=ALL,MIPS64,MIPS64EB ; RUN: llc < %s -mtriple=mips-unknown-linux-gnu -mcpu=mips32r5 -mattr=+fp64,+msa -disable-mips-delay-filler | FileCheck %s --check-prefixes=ALL,MIPS32R5,MIPS32R5EB -; RUN: llc < %s -mtriple=mips64-unknown-linux-gnu -relocation-model=pic -mcpu=mips64r5 -mattr=+fp64,+msa -disable-mips-delay-filler | FileCheck %s --check-prefixes=ALL,MIPS64R5,MIPS64R5EB +; RUN: llc < %s -mtriple=mips64-unknown-linux-gnu -relocation-model=pic -mcpu=mips64r5 -mattr=+fp64,+msa -disable-mips-delay-filler -mips-jalr-reloc=false | FileCheck %s --check-prefixes=ALL,MIPS64R5,MIPS64R5EB ; RUN: llc < %s -mtriple=mipsel-unknown-linux-gnu -mcpu=mips32 -disable-mips-delay-filler | FileCheck %s --check-prefixes=ALL,MIPS32,MIPS32EL -; RUN: llc < %s -mtriple=mips64el-unknown-linux-gnu -relocation-model=pic -mcpu=mips64 -disable-mips-delay-filler | FileCheck %s --check-prefixes=ALL,MIPS64,MIPS64EL +; RUN: llc < %s -mtriple=mips64el-unknown-linux-gnu -relocation-model=pic -mcpu=mips64 -disable-mips-delay-filler -mips-jalr-reloc=false | FileCheck %s --check-prefixes=ALL,MIPS64,MIPS64EL ; RUN: llc < %s -mtriple=mipsel-unknown-linux-gnu -mcpu=mips32r5 -mattr=+fp64,+msa -disable-mips-delay-filler | FileCheck %s --check-prefixes=ALL,MIPS32R5,MIPS32R5EL -; RUN: llc < %s -mtriple=mips64el-unknown-linux-gnu -relocation-model=pic -mcpu=mips64r5 -mattr=+fp64,+msa -disable-mips-delay-filler | FileCheck %s --check-prefixes=ALL,MIPS64R5,MIPS64R5EL +; RUN: llc < %s -mtriple=mips64el-unknown-linux-gnu -relocation-model=pic -mcpu=mips64r5 -mattr=+fp64,+msa -disable-mips-delay-filler -mips-jalr-reloc=false | FileCheck %s --check-prefixes=ALL,MIPS64R5,MIPS64R5EL ; Test that vector types are passed through the integer register set whether or ; not MSA is enabled. This is a ABI requirement for MIPS. For GCC compatibility ; we need to handle any power of 2 number of elements. We will test this ; exhaustively for combinations up to MSA register (128 bits) size. ; First set of tests are for argument passing. define <2 x i8> @i8_2(<2 x i8> %a, <2 x i8> %b) { ; MIPS32EB-LABEL: i8_2: ; MIPS32EB: # %bb.0: ; MIPS32EB-NEXT: srl $1, $5, 24 ; MIPS32EB-NEXT: srl $2, $4, 24 ; MIPS32EB-NEXT: addu $1, $2, $1 ; MIPS32EB-NEXT: sll $1, $1, 8 ; MIPS32EB-NEXT: srl $2, $5, 16 ; MIPS32EB-NEXT: srl $3, $4, 16 ; MIPS32EB-NEXT: addu $2, $3, $2 ; MIPS32EB-NEXT: andi $2, $2, 255 ; MIPS32EB-NEXT: or $2, $2, $1 ; MIPS32EB-NEXT: jr $ra ; MIPS32EB-NEXT: nop ; ; MIPS64EB-LABEL: i8_2: ; MIPS64EB: # %bb.0: ; MIPS64EB-NEXT: dsrl $1, $5, 56 ; MIPS64EB-NEXT: sll $1, $1, 0 ; MIPS64EB-NEXT: dsrl $2, $4, 56 ; MIPS64EB-NEXT: sll $2, $2, 0 ; MIPS64EB-NEXT: addu $1, $2, $1 ; MIPS64EB-NEXT: dsrl $2, $5, 48 ; MIPS64EB-NEXT: sll $1, $1, 8 ; MIPS64EB-NEXT: sll $2, $2, 0 ; MIPS64EB-NEXT: dsrl $3, $4, 48 ; MIPS64EB-NEXT: sll $3, $3, 0 ; MIPS64EB-NEXT: addu $2, $3, $2 ; MIPS64EB-NEXT: andi $2, $2, 255 ; MIPS64EB-NEXT: or $2, $2, $1 ; MIPS64EB-NEXT: jr $ra ; MIPS64EB-NEXT: nop ; ; MIPS32R5EB-LABEL: i8_2: ; MIPS32R5EB: # %bb.0: ; MIPS32R5EB-NEXT: addiu $sp, $sp, -48 ; MIPS32R5EB-NEXT: .cfi_def_cfa_offset 48 ; MIPS32R5EB-NEXT: sw $fp, 44($sp) # 4-byte Folded Spill ; MIPS32R5EB-NEXT: .cfi_offset 30, -4 ; MIPS32R5EB-NEXT: move $fp, $sp ; MIPS32R5EB-NEXT: .cfi_def_cfa_register 30 ; MIPS32R5EB-NEXT: addiu $1, $zero, -16 ; MIPS32R5EB-NEXT: and $sp, $sp, $1 ; MIPS32R5EB-NEXT: sw $5, 36($sp) ; MIPS32R5EB-NEXT: sw $4, 40($sp) ; MIPS32R5EB-NEXT: lbu $1, 37($sp) ; MIPS32R5EB-NEXT: sw $1, 28($sp) ; MIPS32R5EB-NEXT: lbu $1, 36($sp) ; MIPS32R5EB-NEXT: sw $1, 20($sp) ; MIPS32R5EB-NEXT: lbu $1, 41($sp) ; MIPS32R5EB-NEXT: sw $1, 12($sp) ; MIPS32R5EB-NEXT: lbu $1, 40($sp) ; MIPS32R5EB-NEXT: sw $1, 4($sp) ; MIPS32R5EB-NEXT: ld.d $w0, 16($sp) ; MIPS32R5EB-NEXT: ld.d $w1, 0($sp) ; MIPS32R5EB-NEXT: addv.d $w0, $w1, $w0 ; MIPS32R5EB-NEXT: shf.w $w0, $w0, 177 ; MIPS32R5EB-NEXT: copy_s.w $1, $w0[1] ; MIPS32R5EB-NEXT: copy_s.w $2, $w0[3] ; MIPS32R5EB-NEXT: sb $2, 33($sp) ; MIPS32R5EB-NEXT: sb $1, 32($sp) ; MIPS32R5EB-NEXT: lhu $2, 32($sp) ; MIPS32R5EB-NEXT: move $sp, $fp ; MIPS32R5EB-NEXT: lw $fp, 44($sp) # 4-byte Folded Reload ; MIPS32R5EB-NEXT: addiu $sp, $sp, 48 ; MIPS32R5EB-NEXT: jr $ra ; MIPS32R5EB-NEXT: nop ; ; MIPS64R5EB-LABEL: i8_2: ; MIPS64R5EB: # %bb.0: ; MIPS64R5EB-NEXT: daddiu $sp, $sp, -96 ; MIPS64R5EB-NEXT: .cfi_def_cfa_offset 96 ; MIPS64R5EB-NEXT: sd $4, 88($sp) ; MIPS64R5EB-NEXT: lbu $1, 89($sp) ; MIPS64R5EB-NEXT: sh $1, 2($sp) ; MIPS64R5EB-NEXT: lbu $1, 88($sp) ; MIPS64R5EB-NEXT: sh $1, 0($sp) ; MIPS64R5EB-NEXT: ld.h $w0, 0($sp) ; MIPS64R5EB-NEXT: copy_s.h $1, $w0[0] ; MIPS64R5EB-NEXT: copy_s.h $2, $w0[1] ; MIPS64R5EB-NEXT: sd $5, 80($sp) ; MIPS64R5EB-NEXT: lbu $3, 81($sp) ; MIPS64R5EB-NEXT: sh $3, 18($sp) ; MIPS64R5EB-NEXT: lbu $3, 80($sp) ; MIPS64R5EB-NEXT: sh $3, 16($sp) ; MIPS64R5EB-NEXT: ld.h $w0, 16($sp) ; MIPS64R5EB-NEXT: copy_s.h $3, $w0[0] ; MIPS64R5EB-NEXT: copy_s.h $4, $w0[1] ; MIPS64R5EB-NEXT: sw $4, 60($sp) ; MIPS64R5EB-NEXT: sw $3, 52($sp) ; MIPS64R5EB-NEXT: sw $2, 44($sp) ; MIPS64R5EB-NEXT: sw $1, 36($sp) ; MIPS64R5EB-NEXT: ld.d $w0, 48($sp) ; MIPS64R5EB-NEXT: ld.d $w1, 32($sp) ; MIPS64R5EB-NEXT: addv.d $w0, $w1, $w0 ; MIPS64R5EB-NEXT: copy_s.d $1, $w0[0] ; MIPS64R5EB-NEXT: copy_s.d $2, $w0[1] ; MIPS64R5EB-NEXT: sb $2, 77($sp) ; MIPS64R5EB-NEXT: sb $1, 76($sp) ; MIPS64R5EB-NEXT: lh $2, 76($sp) ; MIPS64R5EB-NEXT: daddiu $sp, $sp, 96 ; MIPS64R5EB-NEXT: jr $ra ; MIPS64R5EB-NEXT: nop ; ; MIPS32EL-LABEL: i8_2: ; MIPS32EL: # %bb.0: ; MIPS32EL-NEXT: addu $1, $4, $5 ; MIPS32EL-NEXT: andi $1, $1, 255 ; MIPS32EL-NEXT: andi $2, $5, 65280 ; MIPS32EL-NEXT: srl $2, $2, 8 ; MIPS32EL-NEXT: andi $3, $4, 65280 ; MIPS32EL-NEXT: srl $3, $3, 8 ; MIPS32EL-NEXT: addu $2, $3, $2 ; MIPS32EL-NEXT: sll $2, $2, 8 ; MIPS32EL-NEXT: or $2, $1, $2 ; MIPS32EL-NEXT: jr $ra ; MIPS32EL-NEXT: nop ; ; MIPS64EL-LABEL: i8_2: ; MIPS64EL: # %bb.0: ; MIPS64EL-NEXT: sll $1, $5, 0 ; MIPS64EL-NEXT: sll $2, $4, 0 ; MIPS64EL-NEXT: addu $3, $2, $1 ; MIPS64EL-NEXT: andi $3, $3, 255 ; MIPS64EL-NEXT: andi $1, $1, 65280 ; MIPS64EL-NEXT: srl $1, $1, 8 ; MIPS64EL-NEXT: andi $2, $2, 65280 ; MIPS64EL-NEXT: srl $2, $2, 8 ; MIPS64EL-NEXT: addu $1, $2, $1 ; MIPS64EL-NEXT: sll $1, $1, 8 ; MIPS64EL-NEXT: or $2, $3, $1 ; MIPS64EL-NEXT: jr $ra ; MIPS64EL-NEXT: nop ; ; MIPS32R5EL-LABEL: i8_2: ; MIPS32R5EL: # %bb.0: ; MIPS32R5EL-NEXT: addiu $sp, $sp, -48 ; MIPS32R5EL-NEXT: .cfi_def_cfa_offset 48 ; MIPS32R5EL-NEXT: sw $fp, 44($sp) # 4-byte Folded Spill ; MIPS32R5EL-NEXT: .cfi_offset 30, -4 ; MIPS32R5EL-NEXT: move $fp, $sp ; MIPS32R5EL-NEXT: .cfi_def_cfa_register 30 ; MIPS32R5EL-NEXT: addiu $1, $zero, -16 ; MIPS32R5EL-NEXT: and $sp, $sp, $1 ; MIPS32R5EL-NEXT: sw $5, 36($sp) ; MIPS32R5EL-NEXT: sw $4, 40($sp) ; MIPS32R5EL-NEXT: lbu $1, 37($sp) ; MIPS32R5EL-NEXT: sw $1, 24($sp) ; MIPS32R5EL-NEXT: lbu $1, 36($sp) ; MIPS32R5EL-NEXT: sw $1, 16($sp) ; MIPS32R5EL-NEXT: lbu $1, 41($sp) ; MIPS32R5EL-NEXT: sw $1, 8($sp) ; MIPS32R5EL-NEXT: lbu $1, 40($sp) ; MIPS32R5EL-NEXT: sw $1, 0($sp) ; MIPS32R5EL-NEXT: ld.d $w0, 16($sp) ; MIPS32R5EL-NEXT: ld.d $w1, 0($sp) ; MIPS32R5EL-NEXT: addv.d $w0, $w1, $w0 ; MIPS32R5EL-NEXT: copy_s.w $1, $w0[0] ; MIPS32R5EL-NEXT: copy_s.w $2, $w0[2] ; MIPS32R5EL-NEXT: sb $2, 33($sp) ; MIPS32R5EL-NEXT: sb $1, 32($sp) ; MIPS32R5EL-NEXT: lhu $2, 32($sp) ; MIPS32R5EL-NEXT: move $sp, $fp ; MIPS32R5EL-NEXT: lw $fp, 44($sp) # 4-byte Folded Reload ; MIPS32R5EL-NEXT: addiu $sp, $sp, 48 ; MIPS32R5EL-NEXT: jr $ra ; MIPS32R5EL-NEXT: nop ; ; MIPS64R5EL-LABEL: i8_2: ; MIPS64R5EL: # %bb.0: ; MIPS64R5EL-NEXT: daddiu $sp, $sp, -96 ; MIPS64R5EL-NEXT: .cfi_def_cfa_offset 96 ; MIPS64R5EL-NEXT: sd $4, 88($sp) ; MIPS64R5EL-NEXT: lbu $1, 89($sp) ; MIPS64R5EL-NEXT: sh $1, 2($sp) ; MIPS64R5EL-NEXT: lbu $1, 88($sp) ; MIPS64R5EL-NEXT: sh $1, 0($sp) ; MIPS64R5EL-NEXT: ld.h $w0, 0($sp) ; MIPS64R5EL-NEXT: copy_s.h $1, $w0[0] ; MIPS64R5EL-NEXT: copy_s.h $2, $w0[1] ; MIPS64R5EL-NEXT: sd $5, 80($sp) ; MIPS64R5EL-NEXT: lbu $3, 81($sp) ; MIPS64R5EL-NEXT: sh $3, 18($sp) ; MIPS64R5EL-NEXT: lbu $3, 80($sp) ; MIPS64R5EL-NEXT: sh $3, 16($sp) ; MIPS64R5EL-NEXT: ld.h $w0, 16($sp) ; MIPS64R5EL-NEXT: copy_s.h $3, $w0[0] ; MIPS64R5EL-NEXT: copy_s.h $4, $w0[1] ; MIPS64R5EL-NEXT: sw $4, 56($sp) ; MIPS64R5EL-NEXT: sw $3, 48($sp) ; MIPS64R5EL-NEXT: sw $2, 40($sp) ; MIPS64R5EL-NEXT: sw $1, 32($sp) ; MIPS64R5EL-NEXT: ld.d $w0, 48($sp) ; MIPS64R5EL-NEXT: ld.d $w1, 32($sp) ; MIPS64R5EL-NEXT: addv.d $w0, $w1, $w0 ; MIPS64R5EL-NEXT: copy_s.d $1, $w0[0] ; MIPS64R5EL-NEXT: copy_s.d $2, $w0[1] ; MIPS64R5EL-NEXT: sb $2, 77($sp) ; MIPS64R5EL-NEXT: sb $1, 76($sp) ; MIPS64R5EL-NEXT: lh $2, 76($sp) ; MIPS64R5EL-NEXT: daddiu $sp, $sp, 96 ; MIPS64R5EL-NEXT: jr $ra ; MIPS64R5EL-NEXT: nop %1 = add <2 x i8> %a, %b ret <2 x i8> %1 } ; Test that vector spilled to the outgoing argument area have the expected ; offset from $sp. define <2 x i8> @i8x2_7(<2 x i8> %a, <2 x i8> %b, <2 x i8> %c, <2 x i8> %d, <2 x i8> %e, <2 x i8> %f, <2 x i8> %g) { ; MIPS32EB-LABEL: i8x2_7: ; MIPS32EB: # %bb.0: # %entry ; MIPS32EB-NEXT: srl $1, $5, 24 ; MIPS32EB-NEXT: srl $2, $4, 24 ; MIPS32EB-NEXT: addu $1, $2, $1 ; MIPS32EB-NEXT: srl $2, $6, 24 ; MIPS32EB-NEXT: addu $1, $1, $2 ; MIPS32EB-NEXT: srl $2, $7, 24 ; MIPS32EB-NEXT: addu $1, $1, $2 ; MIPS32EB-NEXT: srl $2, $5, 16 ; MIPS32EB-NEXT: srl $3, $4, 16 ; MIPS32EB-NEXT: addu $2, $3, $2 ; MIPS32EB-NEXT: srl $3, $6, 16 ; MIPS32EB-NEXT: lbu $4, 16($sp) ; MIPS32EB-NEXT: addu $2, $2, $3 ; MIPS32EB-NEXT: addu $1, $1, $4 ; MIPS32EB-NEXT: lbu $3, 20($sp) ; MIPS32EB-NEXT: addu $1, $1, $3 ; MIPS32EB-NEXT: lbu $3, 24($sp) ; MIPS32EB-NEXT: addu $1, $1, $3 ; MIPS32EB-NEXT: srl $3, $7, 16 ; MIPS32EB-NEXT: sll $1, $1, 8 ; MIPS32EB-NEXT: addu $2, $2, $3 ; MIPS32EB-NEXT: lbu $3, 17($sp) ; MIPS32EB-NEXT: addu $2, $2, $3 ; MIPS32EB-NEXT: lbu $3, 21($sp) ; MIPS32EB-NEXT: addu $2, $2, $3 ; MIPS32EB-NEXT: lbu $3, 25($sp) ; MIPS32EB-NEXT: addu $2, $2, $3 ; MIPS32EB-NEXT: andi $2, $2, 255 ; MIPS32EB-NEXT: or $2, $2, $1 ; MIPS32EB-NEXT: jr $ra ; MIPS32EB-NEXT: nop ; ; MIPS64EB-LABEL: i8x2_7: ; MIPS64EB: # %bb.0: # %entry ; MIPS64EB-NEXT: dsrl $1, $5, 56 ; MIPS64EB-NEXT: dsrl $2, $6, 56 ; MIPS64EB-NEXT: sll $1, $1, 0 ; MIPS64EB-NEXT: dsrl $3, $4, 56 ; MIPS64EB-NEXT: sll $3, $3, 0 ; MIPS64EB-NEXT: addu $1, $3, $1 ; MIPS64EB-NEXT: dsrl $3, $6, 48 ; MIPS64EB-NEXT: sll $2, $2, 0 ; MIPS64EB-NEXT: dsrl $5, $5, 48 ; MIPS64EB-NEXT: sll $5, $5, 0 ; MIPS64EB-NEXT: dsrl $4, $4, 48 ; MIPS64EB-NEXT: sll $4, $4, 0 ; MIPS64EB-NEXT: addu $4, $4, $5 ; MIPS64EB-NEXT: addu $1, $1, $2 ; MIPS64EB-NEXT: dsrl $2, $8, 48 ; MIPS64EB-NEXT: dsrl $5, $8, 56 ; MIPS64EB-NEXT: sll $3, $3, 0 ; MIPS64EB-NEXT: dsrl $6, $7, 56 ; MIPS64EB-NEXT: sll $6, $6, 0 ; MIPS64EB-NEXT: addu $1, $1, $6 ; MIPS64EB-NEXT: addu $3, $4, $3 ; MIPS64EB-NEXT: sll $4, $5, 0 ; MIPS64EB-NEXT: dsrl $5, $7, 48 ; MIPS64EB-NEXT: sll $5, $5, 0 ; MIPS64EB-NEXT: addu $3, $3, $5 ; MIPS64EB-NEXT: dsrl $5, $10, 48 ; MIPS64EB-NEXT: sll $2, $2, 0 ; MIPS64EB-NEXT: dsrl $6, $10, 56 ; MIPS64EB-NEXT: addu $1, $1, $4 ; MIPS64EB-NEXT: dsrl $4, $9, 56 ; MIPS64EB-NEXT: sll $4, $4, 0 ; MIPS64EB-NEXT: addu $1, $1, $4 ; MIPS64EB-NEXT: sll $4, $6, 0 ; MIPS64EB-NEXT: addu $1, $1, $4 ; MIPS64EB-NEXT: sll $1, $1, 8 ; MIPS64EB-NEXT: addu $2, $3, $2 ; MIPS64EB-NEXT: dsrl $3, $9, 48 ; MIPS64EB-NEXT: sll $3, $3, 0 ; MIPS64EB-NEXT: addu $2, $2, $3 ; MIPS64EB-NEXT: sll $3, $5, 0 ; MIPS64EB-NEXT: addu $2, $2, $3 ; MIPS64EB-NEXT: andi $2, $2, 255 ; MIPS64EB-NEXT: or $2, $2, $1 ; MIPS64EB-NEXT: jr $ra ; MIPS64EB-NEXT: nop ; ; MIPS32R5EB-LABEL: i8x2_7: ; MIPS32R5EB: # %bb.0: # %entry ; MIPS32R5EB-NEXT: addiu $sp, $sp, -144 ; MIPS32R5EB-NEXT: .cfi_def_cfa_offset 144 ; MIPS32R5EB-NEXT: sw $fp, 140($sp) # 4-byte Folded Spill ; MIPS32R5EB-NEXT: .cfi_offset 30, -4 ; MIPS32R5EB-NEXT: move $fp, $sp ; MIPS32R5EB-NEXT: .cfi_def_cfa_register 30 ; MIPS32R5EB-NEXT: addiu $1, $zero, -16 ; MIPS32R5EB-NEXT: and $sp, $sp, $1 ; MIPS32R5EB-NEXT: sw $5, 132($sp) ; MIPS32R5EB-NEXT: sw $4, 136($sp) ; MIPS32R5EB-NEXT: lbu $1, 133($sp) ; MIPS32R5EB-NEXT: sw $1, 76($sp) ; MIPS32R5EB-NEXT: lbu $1, 132($sp) ; MIPS32R5EB-NEXT: sw $1, 68($sp) ; MIPS32R5EB-NEXT: lbu $1, 137($sp) ; MIPS32R5EB-NEXT: sw $1, 60($sp) ; MIPS32R5EB-NEXT: lbu $1, 136($sp) ; MIPS32R5EB-NEXT: sw $1, 52($sp) ; MIPS32R5EB-NEXT: ld.d $w0, 64($sp) ; MIPS32R5EB-NEXT: ld.d $w1, 48($sp) ; MIPS32R5EB-NEXT: addv.d $w0, $w1, $w0 ; MIPS32R5EB-NEXT: sw $6, 128($sp) ; MIPS32R5EB-NEXT: lbu $1, 129($sp) ; MIPS32R5EB-NEXT: sw $1, 92($sp) ; MIPS32R5EB-NEXT: lbu $1, 128($sp) ; MIPS32R5EB-NEXT: sw $1, 84($sp) ; MIPS32R5EB-NEXT: ld.d $w1, 80($sp) ; MIPS32R5EB-NEXT: addv.d $w0, $w0, $w1 ; MIPS32R5EB-NEXT: sw $7, 124($sp) ; MIPS32R5EB-NEXT: lbu $1, 125($sp) ; MIPS32R5EB-NEXT: sw $1, 108($sp) ; MIPS32R5EB-NEXT: lbu $1, 124($sp) ; MIPS32R5EB-NEXT: sw $1, 100($sp) ; MIPS32R5EB-NEXT: ld.d $w1, 96($sp) ; MIPS32R5EB-NEXT: addv.d $w0, $w0, $w1 ; MIPS32R5EB-NEXT: lbu $1, 161($fp) ; MIPS32R5EB-NEXT: sw $1, 12($sp) ; MIPS32R5EB-NEXT: lbu $1, 160($fp) ; MIPS32R5EB-NEXT: sw $1, 4($sp) ; MIPS32R5EB-NEXT: ld.d $w1, 0($sp) ; MIPS32R5EB-NEXT: addv.d $w0, $w0, $w1 ; MIPS32R5EB-NEXT: lbu $1, 165($fp) ; MIPS32R5EB-NEXT: sw $1, 28($sp) ; MIPS32R5EB-NEXT: lbu $1, 164($fp) ; MIPS32R5EB-NEXT: sw $1, 20($sp) ; MIPS32R5EB-NEXT: ld.d $w1, 16($sp) ; MIPS32R5EB-NEXT: addv.d $w0, $w0, $w1 ; MIPS32R5EB-NEXT: lbu $1, 169($fp) ; MIPS32R5EB-NEXT: sw $1, 44($sp) ; MIPS32R5EB-NEXT: lbu $1, 168($fp) ; MIPS32R5EB-NEXT: sw $1, 36($sp) ; MIPS32R5EB-NEXT: ld.d $w1, 32($sp) ; MIPS32R5EB-NEXT: addv.d $w0, $w0, $w1 ; MIPS32R5EB-NEXT: shf.w $w0, $w0, 177 ; MIPS32R5EB-NEXT: copy_s.w $1, $w0[1] ; MIPS32R5EB-NEXT: copy_s.w $2, $w0[3] ; MIPS32R5EB-NEXT: sb $2, 121($sp) ; MIPS32R5EB-NEXT: sb $1, 120($sp) ; MIPS32R5EB-NEXT: lhu $2, 120($sp) ; MIPS32R5EB-NEXT: move $sp, $fp ; MIPS32R5EB-NEXT: lw $fp, 140($sp) # 4-byte Folded Reload ; MIPS32R5EB-NEXT: addiu $sp, $sp, 144 ; MIPS32R5EB-NEXT: jr $ra ; MIPS32R5EB-NEXT: nop ; ; MIPS64R5EB-LABEL: i8x2_7: ; MIPS64R5EB: # %bb.0: # %entry ; MIPS64R5EB-NEXT: daddiu $sp, $sp, -288 ; MIPS64R5EB-NEXT: .cfi_def_cfa_offset 288 ; MIPS64R5EB-NEXT: sd $4, 280($sp) ; MIPS64R5EB-NEXT: lbu $1, 281($sp) ; MIPS64R5EB-NEXT: sh $1, 2($sp) ; MIPS64R5EB-NEXT: lbu $1, 280($sp) ; MIPS64R5EB-NEXT: sh $1, 0($sp) ; MIPS64R5EB-NEXT: ld.h $w0, 0($sp) ; MIPS64R5EB-NEXT: copy_s.h $1, $w0[0] ; MIPS64R5EB-NEXT: copy_s.h $2, $w0[1] ; MIPS64R5EB-NEXT: sd $5, 272($sp) ; MIPS64R5EB-NEXT: lbu $3, 273($sp) ; MIPS64R5EB-NEXT: sh $3, 18($sp) ; MIPS64R5EB-NEXT: lbu $3, 272($sp) ; MIPS64R5EB-NEXT: sh $3, 16($sp) ; MIPS64R5EB-NEXT: ld.h $w0, 16($sp) ; MIPS64R5EB-NEXT: copy_s.h $3, $w0[0] ; MIPS64R5EB-NEXT: copy_s.h $4, $w0[1] ; MIPS64R5EB-NEXT: sw $4, 140($sp) ; MIPS64R5EB-NEXT: sw $3, 132($sp) ; MIPS64R5EB-NEXT: sw $2, 124($sp) ; MIPS64R5EB-NEXT: sw $1, 116($sp) ; MIPS64R5EB-NEXT: ld.d $w0, 128($sp) ; MIPS64R5EB-NEXT: ld.d $w1, 112($sp) ; MIPS64R5EB-NEXT: addv.d $w0, $w1, $w0 ; MIPS64R5EB-NEXT: sd $6, 264($sp) ; MIPS64R5EB-NEXT: lbu $1, 265($sp) ; MIPS64R5EB-NEXT: sh $1, 34($sp) ; MIPS64R5EB-NEXT: lbu $1, 264($sp) ; MIPS64R5EB-NEXT: sh $1, 32($sp) ; MIPS64R5EB-NEXT: ld.h $w1, 32($sp) ; MIPS64R5EB-NEXT: copy_s.h $1, $w1[0] ; MIPS64R5EB-NEXT: copy_s.h $2, $w1[1] ; MIPS64R5EB-NEXT: sw $2, 156($sp) ; MIPS64R5EB-NEXT: sw $1, 148($sp) ; MIPS64R5EB-NEXT: ld.d $w1, 144($sp) ; MIPS64R5EB-NEXT: addv.d $w0, $w0, $w1 ; MIPS64R5EB-NEXT: sd $7, 256($sp) ; MIPS64R5EB-NEXT: lbu $1, 257($sp) ; MIPS64R5EB-NEXT: sh $1, 50($sp) ; MIPS64R5EB-NEXT: lbu $1, 256($sp) ; MIPS64R5EB-NEXT: sh $1, 48($sp) ; MIPS64R5EB-NEXT: ld.h $w1, 48($sp) ; MIPS64R5EB-NEXT: copy_s.h $1, $w1[0] ; MIPS64R5EB-NEXT: copy_s.h $2, $w1[1] ; MIPS64R5EB-NEXT: sw $2, 172($sp) ; MIPS64R5EB-NEXT: sw $1, 164($sp) ; MIPS64R5EB-NEXT: ld.d $w1, 160($sp) ; MIPS64R5EB-NEXT: addv.d $w0, $w0, $w1 ; MIPS64R5EB-NEXT: sd $8, 248($sp) ; MIPS64R5EB-NEXT: lbu $1, 249($sp) ; MIPS64R5EB-NEXT: sh $1, 66($sp) ; MIPS64R5EB-NEXT: lbu $1, 248($sp) ; MIPS64R5EB-NEXT: sh $1, 64($sp) ; MIPS64R5EB-NEXT: ld.h $w1, 64($sp) ; MIPS64R5EB-NEXT: copy_s.h $1, $w1[0] ; MIPS64R5EB-NEXT: copy_s.h $2, $w1[1] ; MIPS64R5EB-NEXT: sw $2, 188($sp) ; MIPS64R5EB-NEXT: sw $1, 180($sp) ; MIPS64R5EB-NEXT: ld.d $w1, 176($sp) ; MIPS64R5EB-NEXT: addv.d $w0, $w0, $w1 ; MIPS64R5EB-NEXT: sd $10, 232($sp) ; MIPS64R5EB-NEXT: lbu $1, 233($sp) ; MIPS64R5EB-NEXT: sh $1, 98($sp) ; MIPS64R5EB-NEXT: lbu $1, 232($sp) ; MIPS64R5EB-NEXT: sh $1, 96($sp) ; MIPS64R5EB-NEXT: ld.h $w1, 96($sp) ; MIPS64R5EB-NEXT: copy_s.h $1, $w1[0] ; MIPS64R5EB-NEXT: copy_s.h $2, $w1[1] ; MIPS64R5EB-NEXT: sd $9, 240($sp) ; MIPS64R5EB-NEXT: lbu $3, 241($sp) ; MIPS64R5EB-NEXT: sh $3, 82($sp) ; MIPS64R5EB-NEXT: lbu $3, 240($sp) ; MIPS64R5EB-NEXT: sh $3, 80($sp) ; MIPS64R5EB-NEXT: ld.h $w1, 80($sp) ; MIPS64R5EB-NEXT: copy_s.h $3, $w1[0] ; MIPS64R5EB-NEXT: copy_s.h $4, $w1[1] ; MIPS64R5EB-NEXT: sw $4, 204($sp) ; MIPS64R5EB-NEXT: sw $3, 196($sp) ; MIPS64R5EB-NEXT: ld.d $w1, 192($sp) ; MIPS64R5EB-NEXT: addv.d $w0, $w0, $w1 ; MIPS64R5EB-NEXT: sw $2, 220($sp) ; MIPS64R5EB-NEXT: sw $1, 212($sp) ; MIPS64R5EB-NEXT: ld.d $w1, 208($sp) ; MIPS64R5EB-NEXT: addv.d $w0, $w0, $w1 ; MIPS64R5EB-NEXT: copy_s.d $1, $w0[0] ; MIPS64R5EB-NEXT: copy_s.d $2, $w0[1] ; MIPS64R5EB-NEXT: sb $2, 229($sp) ; MIPS64R5EB-NEXT: sb $1, 228($sp) ; MIPS64R5EB-NEXT: lh $2, 228($sp) ; MIPS64R5EB-NEXT: daddiu $sp, $sp, 288 ; MIPS64R5EB-NEXT: jr $ra ; MIPS64R5EB-NEXT: nop ; ; MIPS32EL-LABEL: i8x2_7: ; MIPS32EL: # %bb.0: # %entry ; MIPS32EL-NEXT: addu $1, $4, $5 ; MIPS32EL-NEXT: addu $1, $1, $6 ; MIPS32EL-NEXT: addu $1, $1, $7 ; MIPS32EL-NEXT: andi $2, $5, 65280 ; MIPS32EL-NEXT: lbu $3, 16($sp) ; MIPS32EL-NEXT: addu $1, $1, $3 ; MIPS32EL-NEXT: srl $2, $2, 8 ; MIPS32EL-NEXT: andi $3, $4, 65280 ; MIPS32EL-NEXT: srl $3, $3, 8 ; MIPS32EL-NEXT: addu $2, $3, $2 ; MIPS32EL-NEXT: andi $3, $6, 65280 ; MIPS32EL-NEXT: srl $3, $3, 8 ; MIPS32EL-NEXT: lbu $4, 20($sp) ; MIPS32EL-NEXT: addu $2, $2, $3 ; MIPS32EL-NEXT: addu $1, $1, $4 ; MIPS32EL-NEXT: lbu $3, 24($sp) ; MIPS32EL-NEXT: addu $1, $1, $3 ; MIPS32EL-NEXT: andi $3, $7, 65280 ; MIPS32EL-NEXT: srl $3, $3, 8 ; MIPS32EL-NEXT: lbu $4, 25($sp) ; MIPS32EL-NEXT: andi $1, $1, 255 ; MIPS32EL-NEXT: addu $2, $2, $3 ; MIPS32EL-NEXT: lbu $3, 17($sp) ; MIPS32EL-NEXT: addu $2, $2, $3 ; MIPS32EL-NEXT: lbu $3, 21($sp) ; MIPS32EL-NEXT: addu $2, $2, $3 ; MIPS32EL-NEXT: addu $2, $2, $4 ; MIPS32EL-NEXT: sll $2, $2, 8 ; MIPS32EL-NEXT: or $2, $1, $2 ; MIPS32EL-NEXT: jr $ra ; MIPS32EL-NEXT: nop ; ; MIPS64EL-LABEL: i8x2_7: ; MIPS64EL: # %bb.0: # %entry ; MIPS64EL-NEXT: sll $1, $5, 0 ; MIPS64EL-NEXT: sll $2, $4, 0 ; MIPS64EL-NEXT: addu $3, $2, $1 ; MIPS64EL-NEXT: sll $4, $6, 0 ; MIPS64EL-NEXT: andi $1, $1, 65280 ; MIPS64EL-NEXT: srl $1, $1, 8 ; MIPS64EL-NEXT: andi $2, $2, 65280 ; MIPS64EL-NEXT: srl $2, $2, 8 ; MIPS64EL-NEXT: addu $1, $2, $1 ; MIPS64EL-NEXT: addu $2, $3, $4 ; MIPS64EL-NEXT: sll $3, $7, 0 ; MIPS64EL-NEXT: andi $5, $3, 65280 ; MIPS64EL-NEXT: andi $4, $4, 65280 ; MIPS64EL-NEXT: srl $4, $4, 8 ; MIPS64EL-NEXT: addu $2, $2, $3 ; MIPS64EL-NEXT: addu $1, $1, $4 ; MIPS64EL-NEXT: srl $3, $5, 8 ; MIPS64EL-NEXT: sll $4, $8, 0 ; MIPS64EL-NEXT: andi $5, $4, 65280 ; MIPS64EL-NEXT: srl $5, $5, 8 ; MIPS64EL-NEXT: addu $1, $1, $3 ; MIPS64EL-NEXT: addu $2, $2, $4 ; MIPS64EL-NEXT: sll $3, $9, 0 ; MIPS64EL-NEXT: addu $2, $2, $3 ; MIPS64EL-NEXT: sll $4, $10, 0 ; MIPS64EL-NEXT: addu $2, $2, $4 ; MIPS64EL-NEXT: andi $2, $2, 255 ; MIPS64EL-NEXT: addu $1, $1, $5 ; MIPS64EL-NEXT: andi $3, $3, 65280 ; MIPS64EL-NEXT: srl $3, $3, 8 ; MIPS64EL-NEXT: addu $1, $1, $3 ; MIPS64EL-NEXT: andi $3, $4, 65280 ; MIPS64EL-NEXT: srl $3, $3, 8 ; MIPS64EL-NEXT: addu $1, $1, $3 ; MIPS64EL-NEXT: sll $1, $1, 8 ; MIPS64EL-NEXT: or $2, $2, $1 ; MIPS64EL-NEXT: jr $ra ; MIPS64EL-NEXT: nop ; ; MIPS32R5EL-LABEL: i8x2_7: ; MIPS32R5EL: # %bb.0: # %entry ; MIPS32R5EL-NEXT: addiu $sp, $sp, -144 ; MIPS32R5EL-NEXT: .cfi_def_cfa_offset 144 ; MIPS32R5EL-NEXT: sw $fp, 140($sp) # 4-byte Folded Spill ; MIPS32R5EL-NEXT: .cfi_offset 30, -4 ; MIPS32R5EL-NEXT: move $fp, $sp ; MIPS32R5EL-NEXT: .cfi_def_cfa_register 30 ; MIPS32R5EL-NEXT: addiu $1, $zero, -16 ; MIPS32R5EL-NEXT: and $sp, $sp, $1 ; MIPS32R5EL-NEXT: sw $5, 132($sp) ; MIPS32R5EL-NEXT: sw $4, 136($sp) ; MIPS32R5EL-NEXT: lbu $1, 133($sp) ; MIPS32R5EL-NEXT: sw $1, 72($sp) ; MIPS32R5EL-NEXT: lbu $1, 132($sp) ; MIPS32R5EL-NEXT: sw $1, 64($sp) ; MIPS32R5EL-NEXT: lbu $1, 137($sp) ; MIPS32R5EL-NEXT: sw $1, 56($sp) ; MIPS32R5EL-NEXT: lbu $1, 136($sp) ; MIPS32R5EL-NEXT: sw $1, 48($sp) ; MIPS32R5EL-NEXT: ld.d $w0, 64($sp) ; MIPS32R5EL-NEXT: ld.d $w1, 48($sp) ; MIPS32R5EL-NEXT: addv.d $w0, $w1, $w0 ; MIPS32R5EL-NEXT: sw $6, 128($sp) ; MIPS32R5EL-NEXT: lbu $1, 129($sp) ; MIPS32R5EL-NEXT: sw $1, 88($sp) ; MIPS32R5EL-NEXT: lbu $1, 128($sp) ; MIPS32R5EL-NEXT: sw $1, 80($sp) ; MIPS32R5EL-NEXT: ld.d $w1, 80($sp) ; MIPS32R5EL-NEXT: addv.d $w0, $w0, $w1 ; MIPS32R5EL-NEXT: sw $7, 124($sp) ; MIPS32R5EL-NEXT: lbu $1, 125($sp) ; MIPS32R5EL-NEXT: sw $1, 104($sp) ; MIPS32R5EL-NEXT: lbu $1, 124($sp) ; MIPS32R5EL-NEXT: sw $1, 96($sp) ; MIPS32R5EL-NEXT: ld.d $w1, 96($sp) ; MIPS32R5EL-NEXT: addv.d $w0, $w0, $w1 ; MIPS32R5EL-NEXT: lbu $1, 161($fp) ; MIPS32R5EL-NEXT: sw $1, 8($sp) ; MIPS32R5EL-NEXT: lbu $1, 160($fp) ; MIPS32R5EL-NEXT: sw $1, 0($sp) ; MIPS32R5EL-NEXT: ld.d $w1, 0($sp) ; MIPS32R5EL-NEXT: addv.d $w0, $w0, $w1 ; MIPS32R5EL-NEXT: lbu $1, 165($fp) ; MIPS32R5EL-NEXT: sw $1, 24($sp) ; MIPS32R5EL-NEXT: lbu $1, 164($fp) ; MIPS32R5EL-NEXT: sw $1, 16($sp) ; MIPS32R5EL-NEXT: ld.d $w1, 16($sp) ; MIPS32R5EL-NEXT: addv.d $w0, $w0, $w1 ; MIPS32R5EL-NEXT: lbu $1, 169($fp) ; MIPS32R5EL-NEXT: sw $1, 40($sp) ; MIPS32R5EL-NEXT: lbu $1, 168($fp) ; MIPS32R5EL-NEXT: sw $1, 32($sp) ; MIPS32R5EL-NEXT: ld.d $w1, 32($sp) ; MIPS32R5EL-NEXT: addv.d $w0, $w0, $w1 ; MIPS32R5EL-NEXT: copy_s.w $1, $w0[0] ; MIPS32R5EL-NEXT: copy_s.w $2, $w0[2] ; MIPS32R5EL-NEXT: sb $2, 121($sp) ; MIPS32R5EL-NEXT: sb $1, 120($sp) ; MIPS32R5EL-NEXT: lhu $2, 120($sp) ; MIPS32R5EL-NEXT: move $sp, $fp ; MIPS32R5EL-NEXT: lw $fp, 140($sp) # 4-byte Folded Reload ; MIPS32R5EL-NEXT: addiu $sp, $sp, 144 ; MIPS32R5EL-NEXT: jr $ra ; MIPS32R5EL-NEXT: nop ; ; MIPS64R5EL-LABEL: i8x2_7: ; MIPS64R5EL: # %bb.0: # %entry ; MIPS64R5EL-NEXT: daddiu $sp, $sp, -288 ; MIPS64R5EL-NEXT: .cfi_def_cfa_offset 288 ; MIPS64R5EL-NEXT: sd $4, 280($sp) ; MIPS64R5EL-NEXT: lbu $1, 281($sp) ; MIPS64R5EL-NEXT: sh $1, 2($sp) ; MIPS64R5EL-NEXT: lbu $1, 280($sp) ; MIPS64R5EL-NEXT: sh $1, 0($sp) ; MIPS64R5EL-NEXT: ld.h $w0, 0($sp) ; MIPS64R5EL-NEXT: copy_s.h $1, $w0[0] ; MIPS64R5EL-NEXT: copy_s.h $2, $w0[1] ; MIPS64R5EL-NEXT: sd $5, 272($sp) ; MIPS64R5EL-NEXT: lbu $3, 273($sp) ; MIPS64R5EL-NEXT: sh $3, 18($sp) ; MIPS64R5EL-NEXT: lbu $3, 272($sp) ; MIPS64R5EL-NEXT: sh $3, 16($sp) ; MIPS64R5EL-NEXT: ld.h $w0, 16($sp) ; MIPS64R5EL-NEXT: copy_s.h $3, $w0[0] ; MIPS64R5EL-NEXT: copy_s.h $4, $w0[1] ; MIPS64R5EL-NEXT: sw $4, 136($sp) ; MIPS64R5EL-NEXT: sw $3, 128($sp) ; MIPS64R5EL-NEXT: sw $2, 120($sp) ; MIPS64R5EL-NEXT: sw $1, 112($sp) ; MIPS64R5EL-NEXT: ld.d $w0, 128($sp) ; MIPS64R5EL-NEXT: ld.d $w1, 112($sp) ; MIPS64R5EL-NEXT: addv.d $w0, $w1, $w0 ; MIPS64R5EL-NEXT: sd $6, 264($sp) ; MIPS64R5EL-NEXT: lbu $1, 265($sp) ; MIPS64R5EL-NEXT: sh $1, 34($sp) ; MIPS64R5EL-NEXT: lbu $1, 264($sp) ; MIPS64R5EL-NEXT: sh $1, 32($sp) ; MIPS64R5EL-NEXT: ld.h $w1, 32($sp) ; MIPS64R5EL-NEXT: copy_s.h $1, $w1[0] ; MIPS64R5EL-NEXT: copy_s.h $2, $w1[1] ; MIPS64R5EL-NEXT: sw $2, 152($sp) ; MIPS64R5EL-NEXT: sw $1, 144($sp) ; MIPS64R5EL-NEXT: ld.d $w1, 144($sp) ; MIPS64R5EL-NEXT: addv.d $w0, $w0, $w1 ; MIPS64R5EL-NEXT: sd $7, 256($sp) ; MIPS64R5EL-NEXT: lbu $1, 257($sp) ; MIPS64R5EL-NEXT: sh $1, 50($sp) ; MIPS64R5EL-NEXT: lbu $1, 256($sp) ; MIPS64R5EL-NEXT: sh $1, 48($sp) ; MIPS64R5EL-NEXT: ld.h $w1, 48($sp) ; MIPS64R5EL-NEXT: copy_s.h $1, $w1[0] ; MIPS64R5EL-NEXT: copy_s.h $2, $w1[1] ; MIPS64R5EL-NEXT: sw $2, 168($sp) ; MIPS64R5EL-NEXT: sw $1, 160($sp) ; MIPS64R5EL-NEXT: ld.d $w1, 160($sp) ; MIPS64R5EL-NEXT: addv.d $w0, $w0, $w1 ; MIPS64R5EL-NEXT: sd $8, 248($sp) ; MIPS64R5EL-NEXT: lbu $1, 249($sp) ; MIPS64R5EL-NEXT: sh $1, 66($sp) ; MIPS64R5EL-NEXT: lbu $1, 248($sp) ; MIPS64R5EL-NEXT: sh $1, 64($sp) ; MIPS64R5EL-NEXT: ld.h $w1, 64($sp) ; MIPS64R5EL-NEXT: copy_s.h $1, $w1[0] ; MIPS64R5EL-NEXT: copy_s.h $2, $w1[1] ; MIPS64R5EL-NEXT: sw $2, 184($sp) ; MIPS64R5EL-NEXT: sw $1, 176($sp) ; MIPS64R5EL-NEXT: ld.d $w1, 176($sp) ; MIPS64R5EL-NEXT: addv.d $w0, $w0, $w1 ; MIPS64R5EL-NEXT: sd $10, 232($sp) ; MIPS64R5EL-NEXT: lbu $1, 233($sp) ; MIPS64R5EL-NEXT: sh $1, 98($sp) ; MIPS64R5EL-NEXT: lbu $1, 232($sp) ; MIPS64R5EL-NEXT: sh $1, 96($sp) ; MIPS64R5EL-NEXT: ld.h $w1, 96($sp) ; MIPS64R5EL-NEXT: copy_s.h $1, $w1[0] ; MIPS64R5EL-NEXT: copy_s.h $2, $w1[1] ; MIPS64R5EL-NEXT: sd $9, 240($sp) ; MIPS64R5EL-NEXT: lbu $3, 241($sp) ; MIPS64R5EL-NEXT: sh $3, 82($sp) ; MIPS64R5EL-NEXT: lbu $3, 240($sp) ; MIPS64R5EL-NEXT: sh $3, 80($sp) ; MIPS64R5EL-NEXT: ld.h $w1, 80($sp) ; MIPS64R5EL-NEXT: copy_s.h $3, $w1[0] ; MIPS64R5EL-NEXT: copy_s.h $4, $w1[1] ; MIPS64R5EL-NEXT: sw $4, 200($sp) ; MIPS64R5EL-NEXT: sw $3, 192($sp) ; MIPS64R5EL-NEXT: ld.d $w1, 192($sp) ; MIPS64R5EL-NEXT: addv.d $w0, $w0, $w1 ; MIPS64R5EL-NEXT: sw $2, 216($sp) ; MIPS64R5EL-NEXT: sw $1, 208($sp) ; MIPS64R5EL-NEXT: ld.d $w1, 208($sp) ; MIPS64R5EL-NEXT: addv.d $w0, $w0, $w1 ; MIPS64R5EL-NEXT: copy_s.d $1, $w0[0] ; MIPS64R5EL-NEXT: copy_s.d $2, $w0[1] ; MIPS64R5EL-NEXT: sb $2, 229($sp) ; MIPS64R5EL-NEXT: sb $1, 228($sp) ; MIPS64R5EL-NEXT: lh $2, 228($sp) ; MIPS64R5EL-NEXT: daddiu $sp, $sp, 288 ; MIPS64R5EL-NEXT: jr $ra ; MIPS64R5EL-NEXT: nop entry: %0 = add <2 x i8> %a, %b %1 = add <2 x i8> %0, %c %2 = add <2 x i8> %1, %d %3 = add <2 x i8> %2, %e %4 = add <2 x i8> %3, %f %5 = add <2 x i8> %4, %g ret <2 x i8> %5 } define <4 x i8> @i8_4(<4 x i8> %a, <4 x i8> %b) { ; MIPS32-LABEL: i8_4: ; MIPS32: # %bb.0: ; MIPS32-NEXT: srl $1, $5, 24 ; MIPS32-NEXT: srl $2, $4, 24 ; MIPS32-NEXT: addu $1, $2, $1 ; MIPS32-NEXT: sll $1, $1, 8 ; MIPS32-NEXT: srl $2, $5, 16 ; MIPS32-NEXT: srl $3, $4, 16 ; MIPS32-NEXT: addu $2, $3, $2 ; MIPS32-NEXT: andi $2, $2, 255 ; MIPS32-NEXT: or $1, $2, $1 ; MIPS32-NEXT: addu $2, $4, $5 ; MIPS32-NEXT: sll $1, $1, 16 ; MIPS32-NEXT: andi $2, $2, 255 ; MIPS32-NEXT: srl $3, $5, 8 ; MIPS32-NEXT: srl $4, $4, 8 ; MIPS32-NEXT: addu $3, $4, $3 ; MIPS32-NEXT: sll $3, $3, 8 ; MIPS32-NEXT: or $2, $2, $3 ; MIPS32-NEXT: andi $2, $2, 65535 ; MIPS32-NEXT: or $2, $2, $1 ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: nop ; ; MIPS64-LABEL: i8_4: ; MIPS64: # %bb.0: ; MIPS64-NEXT: sll $1, $5, 0 ; MIPS64-NEXT: srl $2, $1, 24 ; MIPS64-NEXT: sll $3, $4, 0 ; MIPS64-NEXT: srl $4, $3, 24 ; MIPS64-NEXT: addu $2, $4, $2 ; MIPS64-NEXT: sll $2, $2, 8 ; MIPS64-NEXT: srl $4, $1, 16 ; MIPS64-NEXT: srl $5, $3, 16 ; MIPS64-NEXT: addu $4, $5, $4 ; MIPS64-NEXT: andi $4, $4, 255 ; MIPS64-NEXT: or $2, $4, $2 ; MIPS64-NEXT: addu $4, $3, $1 ; MIPS64-NEXT: sll $2, $2, 16 ; MIPS64-NEXT: andi $4, $4, 255 ; MIPS64-NEXT: srl $1, $1, 8 ; MIPS64-NEXT: srl $3, $3, 8 ; MIPS64-NEXT: addu $1, $3, $1 ; MIPS64-NEXT: sll $1, $1, 8 ; MIPS64-NEXT: or $1, $4, $1 ; MIPS64-NEXT: andi $1, $1, 65535 ; MIPS64-NEXT: or $2, $1, $2 ; MIPS64-NEXT: jr $ra ; MIPS64-NEXT: nop ; ; MIPS32R5-LABEL: i8_4: ; MIPS32R5: # %bb.0: ; MIPS32R5-NEXT: addiu $sp, $sp, -16 ; MIPS32R5-NEXT: .cfi_def_cfa_offset 16 ; MIPS32R5-NEXT: sw $5, 8($sp) ; MIPS32R5-NEXT: sw $4, 12($sp) ; MIPS32R5-NEXT: ldi.b $w0, 0 ; MIPS32R5-NEXT: lbu $1, 9($sp) ; MIPS32R5-NEXT: lbu $2, 8($sp) ; MIPS32R5-NEXT: move.v $w1, $w0 ; MIPS32R5-NEXT: insert.w $w1[0], $2 ; MIPS32R5-NEXT: insert.w $w1[1], $1 ; MIPS32R5-NEXT: lbu $1, 10($sp) ; MIPS32R5-NEXT: insert.w $w1[2], $1 ; MIPS32R5-NEXT: lbu $1, 12($sp) ; MIPS32R5-NEXT: lbu $2, 11($sp) ; MIPS32R5-NEXT: insert.w $w1[3], $2 ; MIPS32R5-NEXT: insert.w $w0[0], $1 ; MIPS32R5-NEXT: lbu $1, 13($sp) ; MIPS32R5-NEXT: insert.w $w0[1], $1 ; MIPS32R5-NEXT: lbu $1, 14($sp) ; MIPS32R5-NEXT: insert.w $w0[2], $1 ; MIPS32R5-NEXT: lbu $1, 15($sp) ; MIPS32R5-NEXT: insert.w $w0[3], $1 ; MIPS32R5-NEXT: addv.w $w0, $w0, $w1 ; MIPS32R5-NEXT: copy_s.w $1, $w0[0] ; MIPS32R5-NEXT: copy_s.w $2, $w0[1] ; MIPS32R5-NEXT: copy_s.w $3, $w0[2] ; MIPS32R5-NEXT: copy_s.w $4, $w0[3] ; MIPS32R5-NEXT: sb $4, 7($sp) ; MIPS32R5-NEXT: sb $3, 6($sp) ; MIPS32R5-NEXT: sb $2, 5($sp) ; MIPS32R5-NEXT: sb $1, 4($sp) ; MIPS32R5-NEXT: lw $2, 4($sp) ; MIPS32R5-NEXT: addiu $sp, $sp, 16 ; MIPS32R5-NEXT: jr $ra ; MIPS32R5-NEXT: nop ; ; MIPS64R5-LABEL: i8_4: ; MIPS64R5: # %bb.0: ; MIPS64R5-NEXT: daddiu $sp, $sp, -16 ; MIPS64R5-NEXT: .cfi_def_cfa_offset 16 ; MIPS64R5-NEXT: sll $1, $5, 0 ; MIPS64R5-NEXT: sw $1, 8($sp) ; MIPS64R5-NEXT: sll $1, $4, 0 ; MIPS64R5-NEXT: sw $1, 12($sp) ; MIPS64R5-NEXT: ldi.b $w0, 0 ; MIPS64R5-NEXT: lbu $1, 9($sp) ; MIPS64R5-NEXT: lbu $2, 8($sp) ; MIPS64R5-NEXT: move.v $w1, $w0 ; MIPS64R5-NEXT: insert.w $w1[0], $2 ; MIPS64R5-NEXT: insert.w $w1[1], $1 ; MIPS64R5-NEXT: lbu $1, 10($sp) ; MIPS64R5-NEXT: insert.w $w1[2], $1 ; MIPS64R5-NEXT: lbu $1, 12($sp) ; MIPS64R5-NEXT: lbu $2, 11($sp) ; MIPS64R5-NEXT: insert.w $w1[3], $2 ; MIPS64R5-NEXT: insert.w $w0[0], $1 ; MIPS64R5-NEXT: lbu $1, 13($sp) ; MIPS64R5-NEXT: insert.w $w0[1], $1 ; MIPS64R5-NEXT: lbu $1, 14($sp) ; MIPS64R5-NEXT: insert.w $w0[2], $1 ; MIPS64R5-NEXT: lbu $1, 15($sp) ; MIPS64R5-NEXT: insert.w $w0[3], $1 ; MIPS64R5-NEXT: addv.w $w0, $w0, $w1 ; MIPS64R5-NEXT: copy_s.w $1, $w0[0] ; MIPS64R5-NEXT: copy_s.w $2, $w0[1] ; MIPS64R5-NEXT: copy_s.w $3, $w0[2] ; MIPS64R5-NEXT: copy_s.w $4, $w0[3] ; MIPS64R5-NEXT: sb $4, 7($sp) ; MIPS64R5-NEXT: sb $3, 6($sp) ; MIPS64R5-NEXT: sb $2, 5($sp) ; MIPS64R5-NEXT: sb $1, 4($sp) ; MIPS64R5-NEXT: lw $2, 4($sp) ; MIPS64R5-NEXT: daddiu $sp, $sp, 16 ; MIPS64R5-NEXT: jr $ra ; MIPS64R5-NEXT: nop %1 = add <4 x i8> %a, %b ret <4 x i8> %1 } define <8 x i8> @i8_8(<8 x i8> %a, <8 x i8> %b) { ; MIPS32-LABEL: i8_8: ; MIPS32: # %bb.0: ; MIPS32-NEXT: srl $1, $6, 24 ; MIPS32-NEXT: srl $2, $4, 24 ; MIPS32-NEXT: addu $1, $2, $1 ; MIPS32-NEXT: sll $1, $1, 8 ; MIPS32-NEXT: srl $2, $6, 16 ; MIPS32-NEXT: srl $3, $4, 16 ; MIPS32-NEXT: addu $2, $3, $2 ; MIPS32-NEXT: andi $2, $2, 255 ; MIPS32-NEXT: srl $3, $7, 24 ; MIPS32-NEXT: srl $8, $5, 24 ; MIPS32-NEXT: or $1, $2, $1 ; MIPS32-NEXT: addu $2, $8, $3 ; MIPS32-NEXT: addu $3, $4, $6 ; MIPS32-NEXT: sll $2, $2, 8 ; MIPS32-NEXT: srl $8, $7, 16 ; MIPS32-NEXT: srl $9, $5, 16 ; MIPS32-NEXT: addu $8, $9, $8 ; MIPS32-NEXT: andi $8, $8, 255 ; MIPS32-NEXT: or $8, $8, $2 ; MIPS32-NEXT: sll $1, $1, 16 ; MIPS32-NEXT: andi $2, $3, 255 ; MIPS32-NEXT: srl $3, $6, 8 ; MIPS32-NEXT: srl $4, $4, 8 ; MIPS32-NEXT: addu $3, $4, $3 ; MIPS32-NEXT: sll $3, $3, 8 ; MIPS32-NEXT: or $2, $2, $3 ; MIPS32-NEXT: andi $2, $2, 65535 ; MIPS32-NEXT: addu $3, $5, $7 ; MIPS32-NEXT: or $2, $2, $1 ; MIPS32-NEXT: sll $1, $8, 16 ; MIPS32-NEXT: andi $3, $3, 255 ; MIPS32-NEXT: srl $4, $7, 8 ; MIPS32-NEXT: srl $5, $5, 8 ; MIPS32-NEXT: addu $4, $5, $4 ; MIPS32-NEXT: sll $4, $4, 8 ; MIPS32-NEXT: or $3, $3, $4 ; MIPS32-NEXT: andi $3, $3, 65535 ; MIPS32-NEXT: or $3, $3, $1 ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: nop ; ; MIPS64-LABEL: i8_8: ; MIPS64: # %bb.0: ; MIPS64-NEXT: dsrl $1, $5, 56 ; MIPS64-NEXT: sll $1, $1, 0 ; MIPS64-NEXT: dsrl $2, $4, 56 ; MIPS64-NEXT: sll $2, $2, 0 ; MIPS64-NEXT: addu $1, $2, $1 ; MIPS64-NEXT: dsrl $2, $5, 48 ; MIPS64-NEXT: sll $1, $1, 8 ; MIPS64-NEXT: sll $2, $2, 0 ; MIPS64-NEXT: dsrl $3, $4, 48 ; MIPS64-NEXT: sll $3, $3, 0 ; MIPS64-NEXT: addu $2, $3, $2 ; MIPS64-NEXT: andi $2, $2, 255 ; MIPS64-NEXT: dsrl $3, $5, 40 ; MIPS64-NEXT: or $1, $2, $1 ; MIPS64-NEXT: sll $2, $5, 0 ; MIPS64-NEXT: sll $3, $3, 0 ; MIPS64-NEXT: dsrl $6, $4, 40 ; MIPS64-NEXT: sll $6, $6, 0 ; MIPS64-NEXT: addu $3, $6, $3 ; MIPS64-NEXT: dsrl $5, $5, 32 ; MIPS64-NEXT: srl $6, $2, 24 ; MIPS64-NEXT: sll $7, $4, 0 ; MIPS64-NEXT: srl $8, $7, 24 ; MIPS64-NEXT: addu $6, $8, $6 ; MIPS64-NEXT: sll $1, $1, 16 ; MIPS64-NEXT: sll $3, $3, 8 ; MIPS64-NEXT: sll $5, $5, 0 ; MIPS64-NEXT: dsrl $4, $4, 32 ; MIPS64-NEXT: sll $4, $4, 0 ; MIPS64-NEXT: addu $4, $4, $5 ; MIPS64-NEXT: andi $4, $4, 255 ; MIPS64-NEXT: or $3, $4, $3 ; MIPS64-NEXT: andi $3, $3, 65535 ; MIPS64-NEXT: or $1, $3, $1 ; MIPS64-NEXT: sll $3, $6, 8 ; MIPS64-NEXT: srl $4, $2, 16 ; MIPS64-NEXT: srl $5, $7, 16 ; MIPS64-NEXT: addu $4, $5, $4 ; MIPS64-NEXT: andi $4, $4, 255 ; MIPS64-NEXT: or $3, $4, $3 ; MIPS64-NEXT: addu $4, $7, $2 ; MIPS64-NEXT: dsll $1, $1, 32 ; MIPS64-NEXT: sll $3, $3, 16 ; MIPS64-NEXT: andi $4, $4, 255 ; MIPS64-NEXT: srl $2, $2, 8 ; MIPS64-NEXT: srl $5, $7, 8 ; MIPS64-NEXT: addu $2, $5, $2 ; MIPS64-NEXT: sll $2, $2, 8 ; MIPS64-NEXT: or $2, $4, $2 ; MIPS64-NEXT: andi $2, $2, 65535 ; MIPS64-NEXT: or $2, $2, $3 ; MIPS64-NEXT: dsll $2, $2, 32 ; MIPS64-NEXT: dsrl $2, $2, 32 ; MIPS64-NEXT: or $2, $2, $1 ; MIPS64-NEXT: jr $ra ; MIPS64-NEXT: nop ; ; MIPS32R5EB-LABEL: i8_8: ; MIPS32R5EB: # %bb.0: ; MIPS32R5EB-NEXT: addiu $sp, $sp, -48 ; MIPS32R5EB-NEXT: .cfi_def_cfa_offset 48 ; MIPS32R5EB-NEXT: sw $fp, 44($sp) # 4-byte Folded Spill ; MIPS32R5EB-NEXT: .cfi_offset 30, -4 ; MIPS32R5EB-NEXT: move $fp, $sp ; MIPS32R5EB-NEXT: .cfi_def_cfa_register 30 ; MIPS32R5EB-NEXT: addiu $1, $zero, -16 ; MIPS32R5EB-NEXT: and $sp, $sp, $1 ; MIPS32R5EB-NEXT: sw $6, 24($sp) ; MIPS32R5EB-NEXT: ldi.b $w0, 0 ; MIPS32R5EB-NEXT: lbu $1, 25($sp) ; MIPS32R5EB-NEXT: lbu $2, 24($sp) ; MIPS32R5EB-NEXT: sw $7, 28($sp) ; MIPS32R5EB-NEXT: move.v $w1, $w0 ; MIPS32R5EB-NEXT: insert.h $w1[0], $2 ; MIPS32R5EB-NEXT: insert.h $w1[1], $1 ; MIPS32R5EB-NEXT: lbu $1, 26($sp) ; MIPS32R5EB-NEXT: sw $4, 32($sp) ; MIPS32R5EB-NEXT: insert.h $w1[2], $1 ; MIPS32R5EB-NEXT: lbu $1, 27($sp) ; MIPS32R5EB-NEXT: insert.h $w1[3], $1 ; MIPS32R5EB-NEXT: lbu $1, 28($sp) ; MIPS32R5EB-NEXT: sw $5, 36($sp) ; MIPS32R5EB-NEXT: insert.h $w1[4], $1 ; MIPS32R5EB-NEXT: lbu $1, 32($sp) ; MIPS32R5EB-NEXT: insert.h $w0[0], $1 ; MIPS32R5EB-NEXT: lbu $1, 33($sp) ; MIPS32R5EB-NEXT: insert.h $w0[1], $1 ; MIPS32R5EB-NEXT: lbu $1, 29($sp) ; MIPS32R5EB-NEXT: lbu $2, 34($sp) ; MIPS32R5EB-NEXT: insert.h $w0[2], $2 ; MIPS32R5EB-NEXT: insert.h $w1[5], $1 ; MIPS32R5EB-NEXT: lbu $1, 35($sp) ; MIPS32R5EB-NEXT: lbu $2, 31($sp) ; MIPS32R5EB-NEXT: lbu $3, 30($sp) ; MIPS32R5EB-NEXT: lbu $4, 39($sp) ; MIPS32R5EB-NEXT: insert.h $w1[6], $3 ; MIPS32R5EB-NEXT: insert.h $w1[7], $2 ; MIPS32R5EB-NEXT: insert.h $w0[3], $1 ; MIPS32R5EB-NEXT: lbu $1, 36($sp) ; MIPS32R5EB-NEXT: insert.h $w0[4], $1 ; MIPS32R5EB-NEXT: lbu $1, 37($sp) ; MIPS32R5EB-NEXT: insert.h $w0[5], $1 ; MIPS32R5EB-NEXT: lbu $1, 38($sp) ; MIPS32R5EB-NEXT: insert.h $w0[6], $1 ; MIPS32R5EB-NEXT: insert.h $w0[7], $4 ; MIPS32R5EB-NEXT: addv.h $w0, $w0, $w1 ; MIPS32R5EB-NEXT: copy_s.h $1, $w0[0] ; MIPS32R5EB-NEXT: copy_s.h $2, $w0[1] ; MIPS32R5EB-NEXT: copy_s.h $3, $w0[2] ; MIPS32R5EB-NEXT: copy_s.h $4, $w0[3] ; MIPS32R5EB-NEXT: copy_s.h $5, $w0[4] ; MIPS32R5EB-NEXT: copy_s.h $6, $w0[5] ; MIPS32R5EB-NEXT: copy_s.h $7, $w0[6] ; MIPS32R5EB-NEXT: copy_s.h $8, $w0[7] ; MIPS32R5EB-NEXT: sb $8, 23($sp) ; MIPS32R5EB-NEXT: sb $7, 22($sp) ; MIPS32R5EB-NEXT: sb $6, 21($sp) ; MIPS32R5EB-NEXT: sb $5, 20($sp) ; MIPS32R5EB-NEXT: sb $4, 19($sp) ; MIPS32R5EB-NEXT: sb $3, 18($sp) ; MIPS32R5EB-NEXT: sb $2, 17($sp) ; MIPS32R5EB-NEXT: sb $1, 16($sp) ; MIPS32R5EB-NEXT: lw $1, 20($sp) ; MIPS32R5EB-NEXT: sw $1, 12($sp) ; MIPS32R5EB-NEXT: lw $1, 16($sp) ; MIPS32R5EB-NEXT: sw $1, 4($sp) ; MIPS32R5EB-NEXT: ld.w $w0, 0($sp) ; MIPS32R5EB-NEXT: copy_s.w $2, $w0[1] ; MIPS32R5EB-NEXT: copy_s.w $3, $w0[3] ; MIPS32R5EB-NEXT: move $sp, $fp ; MIPS32R5EB-NEXT: lw $fp, 44($sp) # 4-byte Folded Reload ; MIPS32R5EB-NEXT: addiu $sp, $sp, 48 ; MIPS32R5EB-NEXT: jr $ra ; MIPS32R5EB-NEXT: nop ; ; MIPS64R5-LABEL: i8_8: ; MIPS64R5: # %bb.0: ; MIPS64R5-NEXT: daddiu $sp, $sp, -32 ; MIPS64R5-NEXT: .cfi_def_cfa_offset 32 ; MIPS64R5-NEXT: sd $5, 16($sp) ; MIPS64R5-NEXT: ldi.b $w0, 0 ; MIPS64R5-NEXT: lbu $1, 17($sp) ; MIPS64R5-NEXT: lbu $2, 16($sp) ; MIPS64R5-NEXT: sd $4, 24($sp) ; MIPS64R5-NEXT: move.v $w1, $w0 ; MIPS64R5-NEXT: insert.h $w1[0], $2 ; MIPS64R5-NEXT: insert.h $w1[1], $1 ; MIPS64R5-NEXT: lbu $1, 18($sp) ; MIPS64R5-NEXT: insert.h $w1[2], $1 ; MIPS64R5-NEXT: lbu $1, 19($sp) ; MIPS64R5-NEXT: insert.h $w1[3], $1 ; MIPS64R5-NEXT: lbu $1, 20($sp) ; MIPS64R5-NEXT: insert.h $w1[4], $1 ; MIPS64R5-NEXT: lbu $1, 24($sp) ; MIPS64R5-NEXT: insert.h $w0[0], $1 ; MIPS64R5-NEXT: lbu $1, 25($sp) ; MIPS64R5-NEXT: insert.h $w0[1], $1 ; MIPS64R5-NEXT: lbu $1, 21($sp) ; MIPS64R5-NEXT: lbu $2, 26($sp) ; MIPS64R5-NEXT: insert.h $w0[2], $2 ; MIPS64R5-NEXT: insert.h $w1[5], $1 ; MIPS64R5-NEXT: lbu $1, 27($sp) ; MIPS64R5-NEXT: lbu $2, 23($sp) ; MIPS64R5-NEXT: lbu $3, 22($sp) ; MIPS64R5-NEXT: lbu $4, 31($sp) ; MIPS64R5-NEXT: insert.h $w1[6], $3 ; MIPS64R5-NEXT: insert.h $w1[7], $2 ; MIPS64R5-NEXT: insert.h $w0[3], $1 ; MIPS64R5-NEXT: lbu $1, 28($sp) ; MIPS64R5-NEXT: insert.h $w0[4], $1 ; MIPS64R5-NEXT: lbu $1, 29($sp) ; MIPS64R5-NEXT: insert.h $w0[5], $1 ; MIPS64R5-NEXT: lbu $1, 30($sp) ; MIPS64R5-NEXT: insert.h $w0[6], $1 ; MIPS64R5-NEXT: insert.h $w0[7], $4 ; MIPS64R5-NEXT: addv.h $w0, $w0, $w1 ; MIPS64R5-NEXT: copy_s.h $1, $w0[0] ; MIPS64R5-NEXT: copy_s.h $2, $w0[1] ; MIPS64R5-NEXT: copy_s.h $3, $w0[2] ; MIPS64R5-NEXT: copy_s.h $4, $w0[3] ; MIPS64R5-NEXT: copy_s.h $5, $w0[4] ; MIPS64R5-NEXT: copy_s.h $6, $w0[5] ; MIPS64R5-NEXT: copy_s.h $7, $w0[6] ; MIPS64R5-NEXT: copy_s.h $8, $w0[7] ; MIPS64R5-NEXT: sb $8, 15($sp) ; MIPS64R5-NEXT: sb $7, 14($sp) ; MIPS64R5-NEXT: sb $6, 13($sp) ; MIPS64R5-NEXT: sb $5, 12($sp) ; MIPS64R5-NEXT: sb $4, 11($sp) ; MIPS64R5-NEXT: sb $3, 10($sp) ; MIPS64R5-NEXT: sb $2, 9($sp) ; MIPS64R5-NEXT: sb $1, 8($sp) ; MIPS64R5-NEXT: ld $2, 8($sp) ; MIPS64R5-NEXT: daddiu $sp, $sp, 32 ; MIPS64R5-NEXT: jr $ra ; MIPS64R5-NEXT: nop ; ; MIPS32R5EL-LABEL: i8_8: ; MIPS32R5EL: # %bb.0: ; MIPS32R5EL-NEXT: addiu $sp, $sp, -48 ; MIPS32R5EL-NEXT: .cfi_def_cfa_offset 48 ; MIPS32R5EL-NEXT: sw $fp, 44($sp) # 4-byte Folded Spill ; MIPS32R5EL-NEXT: .cfi_offset 30, -4 ; MIPS32R5EL-NEXT: move $fp, $sp ; MIPS32R5EL-NEXT: .cfi_def_cfa_register 30 ; MIPS32R5EL-NEXT: addiu $1, $zero, -16 ; MIPS32R5EL-NEXT: and $sp, $sp, $1 ; MIPS32R5EL-NEXT: sw $6, 24($sp) ; MIPS32R5EL-NEXT: ldi.b $w0, 0 ; MIPS32R5EL-NEXT: lbu $1, 25($sp) ; MIPS32R5EL-NEXT: lbu $2, 24($sp) ; MIPS32R5EL-NEXT: sw $7, 28($sp) ; MIPS32R5EL-NEXT: move.v $w1, $w0 ; MIPS32R5EL-NEXT: insert.h $w1[0], $2 ; MIPS32R5EL-NEXT: insert.h $w1[1], $1 ; MIPS32R5EL-NEXT: lbu $1, 26($sp) ; MIPS32R5EL-NEXT: sw $4, 32($sp) ; MIPS32R5EL-NEXT: insert.h $w1[2], $1 ; MIPS32R5EL-NEXT: lbu $1, 27($sp) ; MIPS32R5EL-NEXT: insert.h $w1[3], $1 ; MIPS32R5EL-NEXT: lbu $1, 28($sp) ; MIPS32R5EL-NEXT: sw $5, 36($sp) ; MIPS32R5EL-NEXT: insert.h $w1[4], $1 ; MIPS32R5EL-NEXT: lbu $1, 32($sp) ; MIPS32R5EL-NEXT: insert.h $w0[0], $1 ; MIPS32R5EL-NEXT: lbu $1, 33($sp) ; MIPS32R5EL-NEXT: insert.h $w0[1], $1 ; MIPS32R5EL-NEXT: lbu $1, 29($sp) ; MIPS32R5EL-NEXT: lbu $2, 34($sp) ; MIPS32R5EL-NEXT: insert.h $w0[2], $2 ; MIPS32R5EL-NEXT: insert.h $w1[5], $1 ; MIPS32R5EL-NEXT: lbu $1, 35($sp) ; MIPS32R5EL-NEXT: lbu $2, 31($sp) ; MIPS32R5EL-NEXT: lbu $3, 30($sp) ; MIPS32R5EL-NEXT: lbu $4, 39($sp) ; MIPS32R5EL-NEXT: insert.h $w1[6], $3 ; MIPS32R5EL-NEXT: insert.h $w1[7], $2 ; MIPS32R5EL-NEXT: insert.h $w0[3], $1 ; MIPS32R5EL-NEXT: lbu $1, 36($sp) ; MIPS32R5EL-NEXT: insert.h $w0[4], $1 ; MIPS32R5EL-NEXT: lbu $1, 37($sp) ; MIPS32R5EL-NEXT: insert.h $w0[5], $1 ; MIPS32R5EL-NEXT: lbu $1, 38($sp) ; MIPS32R5EL-NEXT: insert.h $w0[6], $1 ; MIPS32R5EL-NEXT: insert.h $w0[7], $4 ; MIPS32R5EL-NEXT: addv.h $w0, $w0, $w1 ; MIPS32R5EL-NEXT: copy_s.h $1, $w0[0] ; MIPS32R5EL-NEXT: copy_s.h $2, $w0[1] ; MIPS32R5EL-NEXT: copy_s.h $3, $w0[2] ; MIPS32R5EL-NEXT: copy_s.h $4, $w0[3] ; MIPS32R5EL-NEXT: copy_s.h $5, $w0[4] ; MIPS32R5EL-NEXT: copy_s.h $6, $w0[5] ; MIPS32R5EL-NEXT: copy_s.h $7, $w0[6] ; MIPS32R5EL-NEXT: copy_s.h $8, $w0[7] ; MIPS32R5EL-NEXT: sb $8, 23($sp) ; MIPS32R5EL-NEXT: sb $7, 22($sp) ; MIPS32R5EL-NEXT: sb $6, 21($sp) ; MIPS32R5EL-NEXT: sb $5, 20($sp) ; MIPS32R5EL-NEXT: sb $4, 19($sp) ; MIPS32R5EL-NEXT: sb $3, 18($sp) ; MIPS32R5EL-NEXT: sb $2, 17($sp) ; MIPS32R5EL-NEXT: sb $1, 16($sp) ; MIPS32R5EL-NEXT: lw $1, 20($sp) ; MIPS32R5EL-NEXT: sw $1, 8($sp) ; MIPS32R5EL-NEXT: lw $1, 16($sp) ; MIPS32R5EL-NEXT: sw $1, 0($sp) ; MIPS32R5EL-NEXT: ld.w $w0, 0($sp) ; MIPS32R5EL-NEXT: copy_s.w $2, $w0[0] ; MIPS32R5EL-NEXT: copy_s.w $3, $w0[2] ; MIPS32R5EL-NEXT: move $sp, $fp ; MIPS32R5EL-NEXT: lw $fp, 44($sp) # 4-byte Folded Reload ; MIPS32R5EL-NEXT: addiu $sp, $sp, 48 ; MIPS32R5EL-NEXT: jr $ra ; MIPS32R5EL-NEXT: nop %1 = add <8 x i8> %a, %b ret <8 x i8> %1 } define <16 x i8> @i8_16(<16 x i8> %a, <16 x i8> %b) { ; MIPS32-LABEL: i8_16: ; MIPS32: # %bb.0: ; MIPS32-NEXT: lw $1, 24($sp) ; MIPS32-NEXT: srl $2, $1, 24 ; MIPS32-NEXT: srl $3, $6, 24 ; MIPS32-NEXT: srl $8, $1, 16 ; MIPS32-NEXT: srl $9, $6, 16 ; MIPS32-NEXT: srl $10, $1, 8 ; MIPS32-NEXT: srl $11, $6, 8 ; MIPS32-NEXT: lw $12, 20($sp) ; MIPS32-NEXT: srl $13, $12, 8 ; MIPS32-NEXT: srl $14, $5, 8 ; MIPS32-NEXT: addu $13, $14, $13 ; MIPS32-NEXT: addu $14, $5, $12 ; MIPS32-NEXT: addu $10, $11, $10 ; MIPS32-NEXT: addu $1, $6, $1 ; MIPS32-NEXT: addu $6, $9, $8 ; MIPS32-NEXT: addu $2, $3, $2 ; MIPS32-NEXT: srl $3, $12, 24 ; MIPS32-NEXT: srl $8, $5, 24 ; MIPS32-NEXT: srl $9, $12, 16 ; MIPS32-NEXT: srl $5, $5, 16 ; MIPS32-NEXT: addu $5, $5, $9 ; MIPS32-NEXT: addu $3, $8, $3 ; MIPS32-NEXT: sll $2, $2, 8 ; MIPS32-NEXT: andi $6, $6, 255 ; MIPS32-NEXT: andi $1, $1, 255 ; MIPS32-NEXT: sll $8, $10, 8 ; MIPS32-NEXT: andi $9, $14, 255 ; MIPS32-NEXT: sll $10, $13, 8 ; MIPS32-NEXT: lw $11, 28($sp) ; MIPS32-NEXT: lw $12, 16($sp) ; MIPS32-NEXT: srl $13, $12, 24 ; MIPS32-NEXT: srl $14, $4, 24 ; MIPS32-NEXT: srl $15, $11, 24 ; MIPS32-NEXT: srl $24, $7, 24 ; MIPS32-NEXT: or $9, $9, $10 ; MIPS32-NEXT: or $1, $1, $8 ; MIPS32-NEXT: or $2, $6, $2 ; MIPS32-NEXT: addu $6, $24, $15 ; MIPS32-NEXT: sll $3, $3, 8 ; MIPS32-NEXT: andi $5, $5, 255 ; MIPS32-NEXT: addu $8, $14, $13 ; MIPS32-NEXT: sll $8, $8, 8 ; MIPS32-NEXT: srl $10, $12, 16 ; MIPS32-NEXT: srl $13, $4, 16 ; MIPS32-NEXT: addu $10, $13, $10 ; MIPS32-NEXT: andi $10, $10, 255 ; MIPS32-NEXT: or $8, $10, $8 ; MIPS32-NEXT: or $3, $5, $3 ; MIPS32-NEXT: addu $5, $4, $12 ; MIPS32-NEXT: sll $6, $6, 8 ; MIPS32-NEXT: srl $10, $11, 16 ; MIPS32-NEXT: srl $13, $7, 16 ; MIPS32-NEXT: addu $10, $13, $10 ; MIPS32-NEXT: andi $10, $10, 255 ; MIPS32-NEXT: or $6, $10, $6 ; MIPS32-NEXT: sll $10, $2, 16 ; MIPS32-NEXT: andi $1, $1, 65535 ; MIPS32-NEXT: sll $3, $3, 16 ; MIPS32-NEXT: andi $9, $9, 65535 ; MIPS32-NEXT: sll $2, $8, 16 ; MIPS32-NEXT: andi $5, $5, 255 ; MIPS32-NEXT: srl $8, $12, 8 ; MIPS32-NEXT: srl $4, $4, 8 ; MIPS32-NEXT: addu $4, $4, $8 ; MIPS32-NEXT: sll $4, $4, 8 ; MIPS32-NEXT: or $4, $5, $4 ; MIPS32-NEXT: andi $4, $4, 65535 ; MIPS32-NEXT: addu $5, $7, $11 ; MIPS32-NEXT: or $2, $4, $2 ; MIPS32-NEXT: or $3, $9, $3 ; MIPS32-NEXT: or $4, $1, $10 ; MIPS32-NEXT: sll $1, $6, 16 ; MIPS32-NEXT: andi $5, $5, 255 ; MIPS32-NEXT: srl $6, $11, 8 ; MIPS32-NEXT: srl $7, $7, 8 ; MIPS32-NEXT: addu $6, $7, $6 ; MIPS32-NEXT: sll $6, $6, 8 ; MIPS32-NEXT: or $5, $5, $6 ; MIPS32-NEXT: andi $5, $5, 65535 ; MIPS32-NEXT: or $5, $5, $1 ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: nop ; ; MIPS64-LABEL: i8_16: ; MIPS64: # %bb.0: ; MIPS64-NEXT: dsrl $1, $7, 56 ; MIPS64-NEXT: dsrl $2, $5, 56 ; MIPS64-NEXT: dsrl $3, $7, 48 ; MIPS64-NEXT: dsrl $8, $5, 48 ; MIPS64-NEXT: dsrl $9, $6, 56 ; MIPS64-NEXT: dsrl $10, $4, 56 ; MIPS64-NEXT: dsrl $11, $7, 32 ; MIPS64-NEXT: sll $1, $1, 0 ; MIPS64-NEXT: sll $2, $2, 0 ; MIPS64-NEXT: sll $3, $3, 0 ; MIPS64-NEXT: sll $8, $8, 0 ; MIPS64-NEXT: dsrl $12, $7, 40 ; MIPS64-NEXT: sll $12, $12, 0 ; MIPS64-NEXT: dsrl $13, $5, 40 ; MIPS64-NEXT: sll $13, $13, 0 ; MIPS64-NEXT: addu $12, $13, $12 ; MIPS64-NEXT: addu $3, $8, $3 ; MIPS64-NEXT: addu $1, $2, $1 ; MIPS64-NEXT: sll $2, $9, 0 ; MIPS64-NEXT: sll $8, $10, 0 ; MIPS64-NEXT: dsrl $9, $6, 48 ; MIPS64-NEXT: sll $9, $9, 0 ; MIPS64-NEXT: dsrl $10, $4, 48 ; MIPS64-NEXT: sll $10, $10, 0 ; MIPS64-NEXT: addu $9, $10, $9 ; MIPS64-NEXT: addu $2, $8, $2 ; MIPS64-NEXT: sll $1, $1, 8 ; MIPS64-NEXT: andi $3, $3, 255 ; MIPS64-NEXT: sll $8, $12, 8 ; MIPS64-NEXT: sll $10, $11, 0 ; MIPS64-NEXT: dsrl $11, $5, 32 ; MIPS64-NEXT: sll $11, $11, 0 ; MIPS64-NEXT: addu $10, $11, $10 ; MIPS64-NEXT: andi $10, $10, 255 ; MIPS64-NEXT: or $8, $10, $8 ; MIPS64-NEXT: sll $10, $6, 0 ; MIPS64-NEXT: or $1, $3, $1 ; MIPS64-NEXT: sll $2, $2, 8 ; MIPS64-NEXT: andi $3, $9, 255 ; MIPS64-NEXT: dsrl $9, $6, 40 ; MIPS64-NEXT: srl $11, $10, 24 ; MIPS64-NEXT: sll $12, $4, 0 ; MIPS64-NEXT: srl $13, $12, 24 ; MIPS64-NEXT: srl $14, $10, 16 ; MIPS64-NEXT: srl $15, $12, 16 ; MIPS64-NEXT: andi $8, $8, 65535 ; MIPS64-NEXT: addu $14, $15, $14 ; MIPS64-NEXT: addu $11, $13, $11 ; MIPS64-NEXT: sll $7, $7, 0 ; MIPS64-NEXT: or $2, $3, $2 ; MIPS64-NEXT: sll $1, $1, 16 ; MIPS64-NEXT: sll $3, $9, 0 ; MIPS64-NEXT: dsrl $9, $4, 40 ; MIPS64-NEXT: sll $9, $9, 0 ; MIPS64-NEXT: addu $3, $9, $3 ; MIPS64-NEXT: dsrl $6, $6, 32 ; MIPS64-NEXT: srl $9, $7, 24 ; MIPS64-NEXT: sll $5, $5, 0 ; MIPS64-NEXT: srl $13, $5, 24 ; MIPS64-NEXT: or $1, $8, $1 ; MIPS64-NEXT: addu $8, $13, $9 ; MIPS64-NEXT: sll $9, $11, 8 ; MIPS64-NEXT: andi $11, $14, 255 ; MIPS64-NEXT: sll $2, $2, 16 ; MIPS64-NEXT: sll $3, $3, 8 ; MIPS64-NEXT: sll $6, $6, 0 ; MIPS64-NEXT: dsrl $4, $4, 32 ; MIPS64-NEXT: sll $4, $4, 0 ; MIPS64-NEXT: addu $4, $4, $6 ; MIPS64-NEXT: andi $4, $4, 255 ; MIPS64-NEXT: or $3, $4, $3 ; MIPS64-NEXT: andi $3, $3, 65535 ; MIPS64-NEXT: or $2, $3, $2 ; MIPS64-NEXT: or $3, $11, $9 ; MIPS64-NEXT: addu $4, $12, $10 ; MIPS64-NEXT: sll $6, $8, 8 ; MIPS64-NEXT: srl $8, $7, 16 ; MIPS64-NEXT: srl $9, $5, 16 ; MIPS64-NEXT: addu $8, $9, $8 ; MIPS64-NEXT: andi $8, $8, 255 ; MIPS64-NEXT: or $6, $8, $6 ; MIPS64-NEXT: addu $8, $5, $7 ; MIPS64-NEXT: dsll $2, $2, 32 ; MIPS64-NEXT: sll $3, $3, 16 ; MIPS64-NEXT: andi $4, $4, 255 ; MIPS64-NEXT: srl $9, $10, 8 ; MIPS64-NEXT: srl $10, $12, 8 ; MIPS64-NEXT: addu $9, $10, $9 ; MIPS64-NEXT: sll $9, $9, 8 ; MIPS64-NEXT: or $4, $4, $9 ; MIPS64-NEXT: andi $4, $4, 65535 ; MIPS64-NEXT: or $3, $4, $3 ; MIPS64-NEXT: dsll $3, $3, 32 ; MIPS64-NEXT: dsrl $3, $3, 32 ; MIPS64-NEXT: or $2, $3, $2 ; MIPS64-NEXT: dsll $1, $1, 32 ; MIPS64-NEXT: sll $3, $6, 16 ; MIPS64-NEXT: andi $4, $8, 255 ; MIPS64-NEXT: srl $6, $7, 8 ; MIPS64-NEXT: srl $5, $5, 8 ; MIPS64-NEXT: addu $5, $5, $6 ; MIPS64-NEXT: sll $5, $5, 8 ; MIPS64-NEXT: or $4, $4, $5 ; MIPS64-NEXT: andi $4, $4, 65535 ; MIPS64-NEXT: or $3, $4, $3 ; MIPS64-NEXT: dsll $3, $3, 32 ; MIPS64-NEXT: dsrl $3, $3, 32 ; MIPS64-NEXT: or $3, $3, $1 ; MIPS64-NEXT: jr $ra ; MIPS64-NEXT: nop ; ; MIPS32R5EB-LABEL: i8_16: ; MIPS32R5EB: # %bb.0: ; MIPS32R5EB-NEXT: ldi.b $w0, 0 ; MIPS32R5EB-NEXT: lw $1, 20($sp) ; MIPS32R5EB-NEXT: lw $2, 16($sp) ; MIPS32R5EB-NEXT: move.v $w1, $w0 ; MIPS32R5EB-NEXT: insert.w $w1[0], $2 ; MIPS32R5EB-NEXT: insert.w $w1[1], $1 ; MIPS32R5EB-NEXT: lw $1, 24($sp) ; MIPS32R5EB-NEXT: insert.w $w0[0], $4 ; MIPS32R5EB-NEXT: insert.w $w1[2], $1 ; MIPS32R5EB-NEXT: lw $1, 28($sp) ; MIPS32R5EB-NEXT: insert.w $w1[3], $1 ; MIPS32R5EB-NEXT: shf.b $w1, $w1, 27 ; MIPS32R5EB-NEXT: insert.w $w0[1], $5 ; MIPS32R5EB-NEXT: insert.w $w0[2], $6 ; MIPS32R5EB-NEXT: insert.w $w0[3], $7 ; MIPS32R5EB-NEXT: shf.b $w0, $w0, 27 ; MIPS32R5EB-NEXT: addv.b $w0, $w0, $w1 ; MIPS32R5EB-NEXT: shf.b $w0, $w0, 27 ; MIPS32R5EB-NEXT: copy_s.w $2, $w0[0] ; MIPS32R5EB-NEXT: copy_s.w $3, $w0[1] ; MIPS32R5EB-NEXT: copy_s.w $4, $w0[2] ; MIPS32R5EB-NEXT: copy_s.w $5, $w0[3] ; MIPS32R5EB-NEXT: jr $ra ; MIPS32R5EB-NEXT: nop ; ; MIPS64R5EB-LABEL: i8_16: ; MIPS64R5EB: # %bb.0: ; MIPS64R5EB-NEXT: ldi.b $w0, 0 ; MIPS64R5EB-NEXT: move.v $w1, $w0 ; MIPS64R5EB-NEXT: insert.d $w1[0], $6 ; MIPS64R5EB-NEXT: insert.d $w1[1], $7 ; MIPS64R5EB-NEXT: shf.b $w1, $w1, 27 ; MIPS64R5EB-NEXT: shf.w $w1, $w1, 177 ; MIPS64R5EB-NEXT: insert.d $w0[0], $4 ; MIPS64R5EB-NEXT: insert.d $w0[1], $5 ; MIPS64R5EB-NEXT: shf.b $w0, $w0, 27 ; MIPS64R5EB-NEXT: shf.w $w0, $w0, 177 ; MIPS64R5EB-NEXT: addv.b $w0, $w0, $w1 ; MIPS64R5EB-NEXT: shf.b $w0, $w0, 27 ; MIPS64R5EB-NEXT: shf.w $w0, $w0, 177 ; MIPS64R5EB-NEXT: copy_s.d $2, $w0[0] ; MIPS64R5EB-NEXT: copy_s.d $3, $w0[1] ; MIPS64R5EB-NEXT: jr $ra ; MIPS64R5EB-NEXT: nop ; ; MIPS32R5EL-LABEL: i8_16: ; MIPS32R5EL: # %bb.0: ; MIPS32R5EL-NEXT: ldi.b $w0, 0 ; MIPS32R5EL-NEXT: lw $1, 20($sp) ; MIPS32R5EL-NEXT: lw $2, 16($sp) ; MIPS32R5EL-NEXT: move.v $w1, $w0 ; MIPS32R5EL-NEXT: insert.w $w1[0], $2 ; MIPS32R5EL-NEXT: insert.w $w1[1], $1 ; MIPS32R5EL-NEXT: lw $1, 24($sp) ; MIPS32R5EL-NEXT: insert.w $w1[2], $1 ; MIPS32R5EL-NEXT: lw $1, 28($sp) ; MIPS32R5EL-NEXT: insert.w $w1[3], $1 ; MIPS32R5EL-NEXT: insert.w $w0[0], $4 ; MIPS32R5EL-NEXT: insert.w $w0[1], $5 ; MIPS32R5EL-NEXT: insert.w $w0[2], $6 ; MIPS32R5EL-NEXT: insert.w $w0[3], $7 ; MIPS32R5EL-NEXT: addv.b $w0, $w0, $w1 ; MIPS32R5EL-NEXT: copy_s.w $2, $w0[0] ; MIPS32R5EL-NEXT: copy_s.w $3, $w0[1] ; MIPS32R5EL-NEXT: copy_s.w $4, $w0[2] ; MIPS32R5EL-NEXT: copy_s.w $5, $w0[3] ; MIPS32R5EL-NEXT: jr $ra ; MIPS32R5EL-NEXT: nop ; ; MIPS64R5EL-LABEL: i8_16: ; MIPS64R5EL: # %bb.0: ; MIPS64R5EL-NEXT: ldi.b $w0, 0 ; MIPS64R5EL-NEXT: move.v $w1, $w0 ; MIPS64R5EL-NEXT: insert.d $w1[0], $6 ; MIPS64R5EL-NEXT: insert.d $w1[1], $7 ; MIPS64R5EL-NEXT: insert.d $w0[0], $4 ; MIPS64R5EL-NEXT: insert.d $w0[1], $5 ; MIPS64R5EL-NEXT: addv.b $w0, $w0, $w1 ; MIPS64R5EL-NEXT: copy_s.d $2, $w0[0] ; MIPS64R5EL-NEXT: copy_s.d $3, $w0[1] ; MIPS64R5EL-NEXT: jr $ra ; MIPS64R5EL-NEXT: nop %1 = add <16 x i8> %a, %b ret <16 x i8> %1 } define <2 x i16> @i16_2(<2 x i16> %a, <2 x i16> %b) { ; MIPS32-LABEL: i16_2: ; MIPS32: # %bb.0: ; MIPS32-NEXT: addu $1, $4, $5 ; MIPS32-NEXT: andi $1, $1, 65535 ; MIPS32-NEXT: srl $2, $5, 16 ; MIPS32-NEXT: srl $3, $4, 16 ; MIPS32-NEXT: addu $2, $3, $2 ; MIPS32-NEXT: sll $2, $2, 16 ; MIPS32-NEXT: or $2, $1, $2 ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: nop ; ; MIPS64-LABEL: i16_2: ; MIPS64: # %bb.0: ; MIPS64-NEXT: sll $1, $5, 0 ; MIPS64-NEXT: sll $2, $4, 0 ; MIPS64-NEXT: addu $3, $2, $1 ; MIPS64-NEXT: andi $3, $3, 65535 ; MIPS64-NEXT: srl $1, $1, 16 ; MIPS64-NEXT: srl $2, $2, 16 ; MIPS64-NEXT: addu $1, $2, $1 ; MIPS64-NEXT: sll $1, $1, 16 ; MIPS64-NEXT: or $2, $3, $1 ; MIPS64-NEXT: jr $ra ; MIPS64-NEXT: nop ; ; MIPS32R5EB-LABEL: i16_2: ; MIPS32R5EB: # %bb.0: ; MIPS32R5EB-NEXT: addiu $sp, $sp, -48 ; MIPS32R5EB-NEXT: .cfi_def_cfa_offset 48 ; MIPS32R5EB-NEXT: sw $fp, 44($sp) # 4-byte Folded Spill ; MIPS32R5EB-NEXT: .cfi_offset 30, -4 ; MIPS32R5EB-NEXT: move $fp, $sp ; MIPS32R5EB-NEXT: .cfi_def_cfa_register 30 ; MIPS32R5EB-NEXT: addiu $1, $zero, -16 ; MIPS32R5EB-NEXT: and $sp, $sp, $1 ; MIPS32R5EB-NEXT: sw $5, 36($sp) ; MIPS32R5EB-NEXT: sw $4, 40($sp) ; MIPS32R5EB-NEXT: lhu $1, 38($sp) ; MIPS32R5EB-NEXT: sw $1, 28($sp) ; MIPS32R5EB-NEXT: lhu $1, 36($sp) ; MIPS32R5EB-NEXT: sw $1, 20($sp) ; MIPS32R5EB-NEXT: lhu $1, 42($sp) ; MIPS32R5EB-NEXT: sw $1, 12($sp) ; MIPS32R5EB-NEXT: lhu $1, 40($sp) ; MIPS32R5EB-NEXT: sw $1, 4($sp) ; MIPS32R5EB-NEXT: ld.d $w0, 16($sp) ; MIPS32R5EB-NEXT: ld.d $w1, 0($sp) ; MIPS32R5EB-NEXT: addv.d $w0, $w1, $w0 ; MIPS32R5EB-NEXT: shf.w $w0, $w0, 177 ; MIPS32R5EB-NEXT: copy_s.w $1, $w0[1] ; MIPS32R5EB-NEXT: copy_s.w $2, $w0[3] ; MIPS32R5EB-NEXT: sh $2, 34($sp) ; MIPS32R5EB-NEXT: sh $1, 32($sp) ; MIPS32R5EB-NEXT: lw $2, 32($sp) ; MIPS32R5EB-NEXT: move $sp, $fp ; MIPS32R5EB-NEXT: lw $fp, 44($sp) # 4-byte Folded Reload ; MIPS32R5EB-NEXT: addiu $sp, $sp, 48 ; MIPS32R5EB-NEXT: jr $ra ; MIPS32R5EB-NEXT: nop ; ; MIPS64R5-LABEL: i16_2: ; MIPS64R5: # %bb.0: ; MIPS64R5-NEXT: daddiu $sp, $sp, -16 ; MIPS64R5-NEXT: .cfi_def_cfa_offset 16 ; MIPS64R5-NEXT: sll $1, $5, 0 ; MIPS64R5-NEXT: sw $1, 8($sp) ; MIPS64R5-NEXT: sll $1, $4, 0 ; MIPS64R5-NEXT: sw $1, 12($sp) ; MIPS64R5-NEXT: ldi.b $w0, 0 ; MIPS64R5-NEXT: lh $1, 10($sp) ; MIPS64R5-NEXT: lh $2, 8($sp) ; MIPS64R5-NEXT: move.v $w1, $w0 ; MIPS64R5-NEXT: insert.d $w1[0], $2 ; MIPS64R5-NEXT: insert.d $w1[1], $1 ; MIPS64R5-NEXT: lh $1, 12($sp) ; MIPS64R5-NEXT: insert.d $w0[0], $1 ; MIPS64R5-NEXT: lh $1, 14($sp) ; MIPS64R5-NEXT: insert.d $w0[1], $1 ; MIPS64R5-NEXT: addv.d $w0, $w0, $w1 ; MIPS64R5-NEXT: copy_s.d $1, $w0[0] ; MIPS64R5-NEXT: copy_s.d $2, $w0[1] ; MIPS64R5-NEXT: sh $2, 6($sp) ; MIPS64R5-NEXT: sh $1, 4($sp) ; MIPS64R5-NEXT: lw $2, 4($sp) ; MIPS64R5-NEXT: daddiu $sp, $sp, 16 ; MIPS64R5-NEXT: jr $ra ; MIPS64R5-NEXT: nop ; ; MIPS32R5EL-LABEL: i16_2: ; MIPS32R5EL: # %bb.0: ; MIPS32R5EL-NEXT: addiu $sp, $sp, -48 ; MIPS32R5EL-NEXT: .cfi_def_cfa_offset 48 ; MIPS32R5EL-NEXT: sw $fp, 44($sp) # 4-byte Folded Spill ; MIPS32R5EL-NEXT: .cfi_offset 30, -4 ; MIPS32R5EL-NEXT: move $fp, $sp ; MIPS32R5EL-NEXT: .cfi_def_cfa_register 30 ; MIPS32R5EL-NEXT: addiu $1, $zero, -16 ; MIPS32R5EL-NEXT: and $sp, $sp, $1 ; MIPS32R5EL-NEXT: sw $5, 36($sp) ; MIPS32R5EL-NEXT: sw $4, 40($sp) ; MIPS32R5EL-NEXT: lhu $1, 38($sp) ; MIPS32R5EL-NEXT: sw $1, 24($sp) ; MIPS32R5EL-NEXT: lhu $1, 36($sp) ; MIPS32R5EL-NEXT: sw $1, 16($sp) ; MIPS32R5EL-NEXT: lhu $1, 42($sp) ; MIPS32R5EL-NEXT: sw $1, 8($sp) ; MIPS32R5EL-NEXT: lhu $1, 40($sp) ; MIPS32R5EL-NEXT: sw $1, 0($sp) ; MIPS32R5EL-NEXT: ld.d $w0, 16($sp) ; MIPS32R5EL-NEXT: ld.d $w1, 0($sp) ; MIPS32R5EL-NEXT: addv.d $w0, $w1, $w0 ; MIPS32R5EL-NEXT: copy_s.w $1, $w0[0] ; MIPS32R5EL-NEXT: copy_s.w $2, $w0[2] ; MIPS32R5EL-NEXT: sh $2, 34($sp) ; MIPS32R5EL-NEXT: sh $1, 32($sp) ; MIPS32R5EL-NEXT: lw $2, 32($sp) ; MIPS32R5EL-NEXT: move $sp, $fp ; MIPS32R5EL-NEXT: lw $fp, 44($sp) # 4-byte Folded Reload ; MIPS32R5EL-NEXT: addiu $sp, $sp, 48 ; MIPS32R5EL-NEXT: jr $ra ; MIPS32R5EL-NEXT: nop %1 = add <2 x i16> %a, %b ret <2 x i16> %1 } define <4 x i16> @i16_4(<4 x i16> %a, <4 x i16> %b) { ; MIPS32-LABEL: i16_4: ; MIPS32: # %bb.0: ; MIPS32-NEXT: addu $1, $4, $6 ; MIPS32-NEXT: andi $1, $1, 65535 ; MIPS32-NEXT: srl $2, $6, 16 ; MIPS32-NEXT: srl $3, $4, 16 ; MIPS32-NEXT: addu $2, $3, $2 ; MIPS32-NEXT: sll $2, $2, 16 ; MIPS32-NEXT: or $2, $1, $2 ; MIPS32-NEXT: addu $1, $5, $7 ; MIPS32-NEXT: andi $1, $1, 65535 ; MIPS32-NEXT: srl $3, $7, 16 ; MIPS32-NEXT: srl $4, $5, 16 ; MIPS32-NEXT: addu $3, $4, $3 ; MIPS32-NEXT: sll $3, $3, 16 ; MIPS32-NEXT: or $3, $1, $3 ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: nop ; ; MIPS64-LABEL: i16_4: ; MIPS64: # %bb.0: ; MIPS64-NEXT: dsrl $1, $5, 48 ; MIPS64-NEXT: sll $1, $1, 0 ; MIPS64-NEXT: dsrl $2, $4, 48 ; MIPS64-NEXT: sll $2, $2, 0 ; MIPS64-NEXT: addu $1, $2, $1 ; MIPS64-NEXT: dsrl $2, $5, 32 ; MIPS64-NEXT: sll $1, $1, 16 ; MIPS64-NEXT: sll $2, $2, 0 ; MIPS64-NEXT: dsrl $3, $4, 32 ; MIPS64-NEXT: sll $3, $3, 0 ; MIPS64-NEXT: addu $2, $3, $2 ; MIPS64-NEXT: andi $2, $2, 65535 ; MIPS64-NEXT: or $1, $2, $1 ; MIPS64-NEXT: sll $2, $5, 0 ; MIPS64-NEXT: sll $3, $4, 0 ; MIPS64-NEXT: addu $4, $3, $2 ; MIPS64-NEXT: dsll $1, $1, 32 ; MIPS64-NEXT: andi $4, $4, 65535 ; MIPS64-NEXT: srl $2, $2, 16 ; MIPS64-NEXT: srl $3, $3, 16 ; MIPS64-NEXT: addu $2, $3, $2 ; MIPS64-NEXT: sll $2, $2, 16 ; MIPS64-NEXT: or $2, $4, $2 ; MIPS64-NEXT: dsll $2, $2, 32 ; MIPS64-NEXT: dsrl $2, $2, 32 ; MIPS64-NEXT: or $2, $2, $1 ; MIPS64-NEXT: jr $ra ; MIPS64-NEXT: nop ; ; MIPS32R5EB-LABEL: i16_4: ; MIPS32R5EB: # %bb.0: ; MIPS32R5EB-NEXT: addiu $sp, $sp, -48 ; MIPS32R5EB-NEXT: .cfi_def_cfa_offset 48 ; MIPS32R5EB-NEXT: sw $fp, 44($sp) # 4-byte Folded Spill ; MIPS32R5EB-NEXT: .cfi_offset 30, -4 ; MIPS32R5EB-NEXT: move $fp, $sp ; MIPS32R5EB-NEXT: .cfi_def_cfa_register 30 ; MIPS32R5EB-NEXT: addiu $1, $zero, -16 ; MIPS32R5EB-NEXT: and $sp, $sp, $1 ; MIPS32R5EB-NEXT: sw $6, 24($sp) ; MIPS32R5EB-NEXT: sw $7, 28($sp) ; MIPS32R5EB-NEXT: sw $4, 32($sp) ; MIPS32R5EB-NEXT: ldi.b $w0, 0 ; MIPS32R5EB-NEXT: lhu $1, 26($sp) ; MIPS32R5EB-NEXT: lhu $2, 24($sp) ; MIPS32R5EB-NEXT: move.v $w1, $w0 ; MIPS32R5EB-NEXT: insert.w $w1[0], $2 ; MIPS32R5EB-NEXT: insert.w $w1[1], $1 ; MIPS32R5EB-NEXT: lhu $1, 28($sp) ; MIPS32R5EB-NEXT: sw $5, 36($sp) ; MIPS32R5EB-NEXT: insert.w $w1[2], $1 ; MIPS32R5EB-NEXT: lhu $1, 32($sp) ; MIPS32R5EB-NEXT: lhu $2, 30($sp) ; MIPS32R5EB-NEXT: insert.w $w1[3], $2 ; MIPS32R5EB-NEXT: insert.w $w0[0], $1 ; MIPS32R5EB-NEXT: lhu $1, 34($sp) ; MIPS32R5EB-NEXT: insert.w $w0[1], $1 ; MIPS32R5EB-NEXT: lhu $1, 36($sp) ; MIPS32R5EB-NEXT: insert.w $w0[2], $1 ; MIPS32R5EB-NEXT: lhu $1, 38($sp) ; MIPS32R5EB-NEXT: insert.w $w0[3], $1 ; MIPS32R5EB-NEXT: addv.w $w0, $w0, $w1 ; MIPS32R5EB-NEXT: copy_s.w $1, $w0[0] ; MIPS32R5EB-NEXT: copy_s.w $2, $w0[1] ; MIPS32R5EB-NEXT: copy_s.w $3, $w0[2] ; MIPS32R5EB-NEXT: copy_s.w $4, $w0[3] ; MIPS32R5EB-NEXT: sh $4, 22($sp) ; MIPS32R5EB-NEXT: sh $3, 20($sp) ; MIPS32R5EB-NEXT: sh $2, 18($sp) ; MIPS32R5EB-NEXT: sh $1, 16($sp) ; MIPS32R5EB-NEXT: lw $1, 20($sp) ; MIPS32R5EB-NEXT: sw $1, 12($sp) ; MIPS32R5EB-NEXT: lw $1, 16($sp) ; MIPS32R5EB-NEXT: sw $1, 4($sp) ; MIPS32R5EB-NEXT: ld.w $w0, 0($sp) ; MIPS32R5EB-NEXT: copy_s.w $2, $w0[1] ; MIPS32R5EB-NEXT: copy_s.w $3, $w0[3] ; MIPS32R5EB-NEXT: move $sp, $fp ; MIPS32R5EB-NEXT: lw $fp, 44($sp) # 4-byte Folded Reload ; MIPS32R5EB-NEXT: addiu $sp, $sp, 48 ; MIPS32R5EB-NEXT: jr $ra ; MIPS32R5EB-NEXT: nop ; ; MIPS64R5-LABEL: i16_4: ; MIPS64R5: # %bb.0: ; MIPS64R5-NEXT: daddiu $sp, $sp, -32 ; MIPS64R5-NEXT: .cfi_def_cfa_offset 32 ; MIPS64R5-NEXT: sd $5, 16($sp) ; MIPS64R5-NEXT: sd $4, 24($sp) ; MIPS64R5-NEXT: ldi.b $w0, 0 ; MIPS64R5-NEXT: lhu $1, 18($sp) ; MIPS64R5-NEXT: lhu $2, 16($sp) ; MIPS64R5-NEXT: move.v $w1, $w0 ; MIPS64R5-NEXT: insert.w $w1[0], $2 ; MIPS64R5-NEXT: insert.w $w1[1], $1 ; MIPS64R5-NEXT: lhu $1, 20($sp) ; MIPS64R5-NEXT: insert.w $w1[2], $1 ; MIPS64R5-NEXT: lhu $1, 24($sp) ; MIPS64R5-NEXT: lhu $2, 22($sp) ; MIPS64R5-NEXT: insert.w $w1[3], $2 ; MIPS64R5-NEXT: insert.w $w0[0], $1 ; MIPS64R5-NEXT: lhu $1, 26($sp) ; MIPS64R5-NEXT: insert.w $w0[1], $1 ; MIPS64R5-NEXT: lhu $1, 28($sp) ; MIPS64R5-NEXT: insert.w $w0[2], $1 ; MIPS64R5-NEXT: lhu $1, 30($sp) ; MIPS64R5-NEXT: insert.w $w0[3], $1 ; MIPS64R5-NEXT: addv.w $w0, $w0, $w1 ; MIPS64R5-NEXT: copy_s.w $1, $w0[0] ; MIPS64R5-NEXT: copy_s.w $2, $w0[1] ; MIPS64R5-NEXT: copy_s.w $3, $w0[2] ; MIPS64R5-NEXT: copy_s.w $4, $w0[3] ; MIPS64R5-NEXT: sh $4, 14($sp) ; MIPS64R5-NEXT: sh $3, 12($sp) ; MIPS64R5-NEXT: sh $2, 10($sp) ; MIPS64R5-NEXT: sh $1, 8($sp) ; MIPS64R5-NEXT: ld $2, 8($sp) ; MIPS64R5-NEXT: daddiu $sp, $sp, 32 ; MIPS64R5-NEXT: jr $ra ; MIPS64R5-NEXT: nop ; ; MIPS32R5EL-LABEL: i16_4: ; MIPS32R5EL: # %bb.0: ; MIPS32R5EL-NEXT: addiu $sp, $sp, -48 ; MIPS32R5EL-NEXT: .cfi_def_cfa_offset 48 ; MIPS32R5EL-NEXT: sw $fp, 44($sp) # 4-byte Folded Spill ; MIPS32R5EL-NEXT: .cfi_offset 30, -4 ; MIPS32R5EL-NEXT: move $fp, $sp ; MIPS32R5EL-NEXT: .cfi_def_cfa_register 30 ; MIPS32R5EL-NEXT: addiu $1, $zero, -16 ; MIPS32R5EL-NEXT: and $sp, $sp, $1 ; MIPS32R5EL-NEXT: sw $6, 24($sp) ; MIPS32R5EL-NEXT: sw $7, 28($sp) ; MIPS32R5EL-NEXT: sw $4, 32($sp) ; MIPS32R5EL-NEXT: ldi.b $w0, 0 ; MIPS32R5EL-NEXT: lhu $1, 26($sp) ; MIPS32R5EL-NEXT: lhu $2, 24($sp) ; MIPS32R5EL-NEXT: move.v $w1, $w0 ; MIPS32R5EL-NEXT: insert.w $w1[0], $2 ; MIPS32R5EL-NEXT: insert.w $w1[1], $1 ; MIPS32R5EL-NEXT: lhu $1, 28($sp) ; MIPS32R5EL-NEXT: sw $5, 36($sp) ; MIPS32R5EL-NEXT: insert.w $w1[2], $1 ; MIPS32R5EL-NEXT: lhu $1, 32($sp) ; MIPS32R5EL-NEXT: lhu $2, 30($sp) ; MIPS32R5EL-NEXT: insert.w $w1[3], $2 ; MIPS32R5EL-NEXT: insert.w $w0[0], $1 ; MIPS32R5EL-NEXT: lhu $1, 34($sp) ; MIPS32R5EL-NEXT: insert.w $w0[1], $1 ; MIPS32R5EL-NEXT: lhu $1, 36($sp) ; MIPS32R5EL-NEXT: insert.w $w0[2], $1 ; MIPS32R5EL-NEXT: lhu $1, 38($sp) ; MIPS32R5EL-NEXT: insert.w $w0[3], $1 ; MIPS32R5EL-NEXT: addv.w $w0, $w0, $w1 ; MIPS32R5EL-NEXT: copy_s.w $1, $w0[0] ; MIPS32R5EL-NEXT: copy_s.w $2, $w0[1] ; MIPS32R5EL-NEXT: copy_s.w $3, $w0[2] ; MIPS32R5EL-NEXT: copy_s.w $4, $w0[3] ; MIPS32R5EL-NEXT: sh $4, 22($sp) ; MIPS32R5EL-NEXT: sh $3, 20($sp) ; MIPS32R5EL-NEXT: sh $2, 18($sp) ; MIPS32R5EL-NEXT: sh $1, 16($sp) ; MIPS32R5EL-NEXT: lw $1, 20($sp) ; MIPS32R5EL-NEXT: sw $1, 8($sp) ; MIPS32R5EL-NEXT: lw $1, 16($sp) ; MIPS32R5EL-NEXT: sw $1, 0($sp) ; MIPS32R5EL-NEXT: ld.w $w0, 0($sp) ; MIPS32R5EL-NEXT: copy_s.w $2, $w0[0] ; MIPS32R5EL-NEXT: copy_s.w $3, $w0[2] ; MIPS32R5EL-NEXT: move $sp, $fp ; MIPS32R5EL-NEXT: lw $fp, 44($sp) # 4-byte Folded Reload ; MIPS32R5EL-NEXT: addiu $sp, $sp, 48 ; MIPS32R5EL-NEXT: jr $ra ; MIPS32R5EL-NEXT: nop %1 = add <4 x i16> %a, %b ret <4 x i16> %1 } define <8 x i16> @i16_8(<8 x i16> %a, <8 x i16> %b) { ; MIPS32-LABEL: i16_8: ; MIPS32: # %bb.0: ; MIPS32-NEXT: lw $1, 24($sp) ; MIPS32-NEXT: srl $2, $1, 16 ; MIPS32-NEXT: srl $3, $6, 16 ; MIPS32-NEXT: lw $8, 20($sp) ; MIPS32-NEXT: srl $9, $8, 16 ; MIPS32-NEXT: srl $10, $5, 16 ; MIPS32-NEXT: addu $9, $10, $9 ; MIPS32-NEXT: addu $5, $5, $8 ; MIPS32-NEXT: addu $2, $3, $2 ; MIPS32-NEXT: addu $1, $6, $1 ; MIPS32-NEXT: lw $3, 16($sp) ; MIPS32-NEXT: lw $6, 28($sp) ; MIPS32-NEXT: addu $8, $7, $6 ; MIPS32-NEXT: andi $1, $1, 65535 ; MIPS32-NEXT: sll $10, $2, 16 ; MIPS32-NEXT: andi $5, $5, 65535 ; MIPS32-NEXT: sll $9, $9, 16 ; MIPS32-NEXT: addu $2, $4, $3 ; MIPS32-NEXT: andi $2, $2, 65535 ; MIPS32-NEXT: srl $3, $3, 16 ; MIPS32-NEXT: srl $4, $4, 16 ; MIPS32-NEXT: addu $3, $4, $3 ; MIPS32-NEXT: sll $3, $3, 16 ; MIPS32-NEXT: or $2, $2, $3 ; MIPS32-NEXT: or $3, $5, $9 ; MIPS32-NEXT: or $4, $1, $10 ; MIPS32-NEXT: andi $1, $8, 65535 ; MIPS32-NEXT: srl $5, $6, 16 ; MIPS32-NEXT: srl $6, $7, 16 ; MIPS32-NEXT: addu $5, $6, $5 ; MIPS32-NEXT: sll $5, $5, 16 ; MIPS32-NEXT: or $5, $1, $5 ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: nop ; ; MIPS64-LABEL: i16_8: ; MIPS64: # %bb.0: ; MIPS64-NEXT: dsrl $1, $6, 48 ; MIPS64-NEXT: dsrl $2, $7, 48 ; MIPS64-NEXT: sll $1, $1, 0 ; MIPS64-NEXT: dsrl $3, $4, 48 ; MIPS64-NEXT: sll $3, $3, 0 ; MIPS64-NEXT: addu $1, $3, $1 ; MIPS64-NEXT: dsrl $3, $6, 32 ; MIPS64-NEXT: sll $2, $2, 0 ; MIPS64-NEXT: dsrl $8, $5, 48 ; MIPS64-NEXT: sll $8, $8, 0 ; MIPS64-NEXT: addu $2, $8, $2 ; MIPS64-NEXT: sll $1, $1, 16 ; MIPS64-NEXT: sll $3, $3, 0 ; MIPS64-NEXT: dsrl $8, $4, 32 ; MIPS64-NEXT: sll $8, $8, 0 ; MIPS64-NEXT: addu $3, $8, $3 ; MIPS64-NEXT: andi $3, $3, 65535 ; MIPS64-NEXT: dsrl $8, $7, 32 ; MIPS64-NEXT: or $1, $3, $1 ; MIPS64-NEXT: sll $2, $2, 16 ; MIPS64-NEXT: sll $3, $8, 0 ; MIPS64-NEXT: dsrl $8, $5, 32 ; MIPS64-NEXT: sll $8, $8, 0 ; MIPS64-NEXT: addu $3, $8, $3 ; MIPS64-NEXT: andi $3, $3, 65535 ; MIPS64-NEXT: or $3, $3, $2 ; MIPS64-NEXT: sll $2, $6, 0 ; MIPS64-NEXT: sll $4, $4, 0 ; MIPS64-NEXT: addu $6, $4, $2 ; MIPS64-NEXT: andi $6, $6, 65535 ; MIPS64-NEXT: srl $2, $2, 16 ; MIPS64-NEXT: srl $4, $4, 16 ; MIPS64-NEXT: addu $2, $4, $2 ; MIPS64-NEXT: sll $2, $2, 16 ; MIPS64-NEXT: dsll $1, $1, 32 ; MIPS64-NEXT: or $2, $6, $2 ; MIPS64-NEXT: dsll $2, $2, 32 ; MIPS64-NEXT: dsrl $2, $2, 32 ; MIPS64-NEXT: sll $4, $7, 0 ; MIPS64-NEXT: sll $5, $5, 0 ; MIPS64-NEXT: addu $6, $5, $4 ; MIPS64-NEXT: or $2, $2, $1 ; MIPS64-NEXT: dsll $1, $3, 32 ; MIPS64-NEXT: andi $3, $6, 65535 ; MIPS64-NEXT: srl $4, $4, 16 ; MIPS64-NEXT: srl $5, $5, 16 ; MIPS64-NEXT: addu $4, $5, $4 ; MIPS64-NEXT: sll $4, $4, 16 ; MIPS64-NEXT: or $3, $3, $4 ; MIPS64-NEXT: dsll $3, $3, 32 ; MIPS64-NEXT: dsrl $3, $3, 32 ; MIPS64-NEXT: or $3, $3, $1 ; MIPS64-NEXT: jr $ra ; MIPS64-NEXT: nop ; ; MIPS32R5EB-LABEL: i16_8: ; MIPS32R5EB: # %bb.0: ; MIPS32R5EB-NEXT: ldi.b $w0, 0 ; MIPS32R5EB-NEXT: lw $1, 20($sp) ; MIPS32R5EB-NEXT: lw $2, 16($sp) ; MIPS32R5EB-NEXT: move.v $w1, $w0 ; MIPS32R5EB-NEXT: insert.w $w1[0], $2 ; MIPS32R5EB-NEXT: insert.w $w1[1], $1 ; MIPS32R5EB-NEXT: lw $1, 24($sp) ; MIPS32R5EB-NEXT: insert.w $w0[0], $4 ; MIPS32R5EB-NEXT: insert.w $w1[2], $1 ; MIPS32R5EB-NEXT: lw $1, 28($sp) ; MIPS32R5EB-NEXT: insert.w $w1[3], $1 ; MIPS32R5EB-NEXT: shf.h $w1, $w1, 177 ; MIPS32R5EB-NEXT: insert.w $w0[1], $5 ; MIPS32R5EB-NEXT: insert.w $w0[2], $6 ; MIPS32R5EB-NEXT: insert.w $w0[3], $7 ; MIPS32R5EB-NEXT: shf.h $w0, $w0, 177 ; MIPS32R5EB-NEXT: addv.h $w0, $w0, $w1 ; MIPS32R5EB-NEXT: shf.h $w0, $w0, 177 ; MIPS32R5EB-NEXT: copy_s.w $2, $w0[0] ; MIPS32R5EB-NEXT: copy_s.w $3, $w0[1] ; MIPS32R5EB-NEXT: copy_s.w $4, $w0[2] ; MIPS32R5EB-NEXT: copy_s.w $5, $w0[3] ; MIPS32R5EB-NEXT: jr $ra ; MIPS32R5EB-NEXT: nop ; ; MIPS64R5EB-LABEL: i16_8: ; MIPS64R5EB: # %bb.0: ; MIPS64R5EB-NEXT: ldi.b $w0, 0 ; MIPS64R5EB-NEXT: move.v $w1, $w0 ; MIPS64R5EB-NEXT: insert.d $w1[0], $6 ; MIPS64R5EB-NEXT: insert.d $w1[1], $7 ; MIPS64R5EB-NEXT: shf.h $w1, $w1, 27 ; MIPS64R5EB-NEXT: insert.d $w0[0], $4 ; MIPS64R5EB-NEXT: insert.d $w0[1], $5 ; MIPS64R5EB-NEXT: shf.h $w0, $w0, 27 ; MIPS64R5EB-NEXT: addv.h $w0, $w0, $w1 ; MIPS64R5EB-NEXT: shf.h $w0, $w0, 27 ; MIPS64R5EB-NEXT: copy_s.d $2, $w0[0] ; MIPS64R5EB-NEXT: copy_s.d $3, $w0[1] ; MIPS64R5EB-NEXT: jr $ra ; MIPS64R5EB-NEXT: nop ; ; MIPS32R5EL-LABEL: i16_8: ; MIPS32R5EL: # %bb.0: ; MIPS32R5EL-NEXT: ldi.b $w0, 0 ; MIPS32R5EL-NEXT: lw $1, 20($sp) ; MIPS32R5EL-NEXT: lw $2, 16($sp) ; MIPS32R5EL-NEXT: move.v $w1, $w0 ; MIPS32R5EL-NEXT: insert.w $w1[0], $2 ; MIPS32R5EL-NEXT: insert.w $w1[1], $1 ; MIPS32R5EL-NEXT: lw $1, 24($sp) ; MIPS32R5EL-NEXT: insert.w $w1[2], $1 ; MIPS32R5EL-NEXT: lw $1, 28($sp) ; MIPS32R5EL-NEXT: insert.w $w1[3], $1 ; MIPS32R5EL-NEXT: insert.w $w0[0], $4 ; MIPS32R5EL-NEXT: insert.w $w0[1], $5 ; MIPS32R5EL-NEXT: insert.w $w0[2], $6 ; MIPS32R5EL-NEXT: insert.w $w0[3], $7 ; MIPS32R5EL-NEXT: addv.h $w0, $w0, $w1 ; MIPS32R5EL-NEXT: copy_s.w $2, $w0[0] ; MIPS32R5EL-NEXT: copy_s.w $3, $w0[1] ; MIPS32R5EL-NEXT: copy_s.w $4, $w0[2] ; MIPS32R5EL-NEXT: copy_s.w $5, $w0[3] ; MIPS32R5EL-NEXT: jr $ra ; MIPS32R5EL-NEXT: nop ; ; MIPS64R5EL-LABEL: i16_8: ; MIPS64R5EL: # %bb.0: ; MIPS64R5EL-NEXT: ldi.b $w0, 0 ; MIPS64R5EL-NEXT: move.v $w1, $w0 ; MIPS64R5EL-NEXT: insert.d $w1[0], $6 ; MIPS64R5EL-NEXT: insert.d $w1[1], $7 ; MIPS64R5EL-NEXT: insert.d $w0[0], $4 ; MIPS64R5EL-NEXT: insert.d $w0[1], $5 ; MIPS64R5EL-NEXT: addv.h $w0, $w0, $w1 ; MIPS64R5EL-NEXT: copy_s.d $2, $w0[0] ; MIPS64R5EL-NEXT: copy_s.d $3, $w0[1] ; MIPS64R5EL-NEXT: jr $ra ; MIPS64R5EL-NEXT: nop %1 = add <8 x i16> %a, %b ret <8 x i16> %1 } define <2 x i32> @i32_2(<2 x i32> %a, <2 x i32> %b) { ; MIPS32-LABEL: i32_2: ; MIPS32: # %bb.0: ; MIPS32-NEXT: addu $2, $4, $6 ; MIPS32-NEXT: addu $3, $5, $7 ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: nop ; ; MIPS64-LABEL: i32_2: ; MIPS64: # %bb.0: ; MIPS64-NEXT: sll $1, $5, 0 ; MIPS64-NEXT: sll $2, $4, 0 ; MIPS64-NEXT: addu $1, $2, $1 ; MIPS64-NEXT: dsll $1, $1, 32 ; MIPS64-NEXT: dsrl $2, $5, 32 ; MIPS64-NEXT: dsrl $1, $1, 32 ; MIPS64-NEXT: sll $2, $2, 0 ; MIPS64-NEXT: dsrl $3, $4, 32 ; MIPS64-NEXT: sll $3, $3, 0 ; MIPS64-NEXT: addu $2, $3, $2 ; MIPS64-NEXT: dsll $2, $2, 32 ; MIPS64-NEXT: or $2, $1, $2 ; MIPS64-NEXT: jr $ra ; MIPS64-NEXT: nop ; ; MIPS32R5EB-LABEL: i32_2: ; MIPS32R5EB: # %bb.0: ; MIPS32R5EB-NEXT: addiu $sp, $sp, -48 ; MIPS32R5EB-NEXT: .cfi_def_cfa_offset 48 ; MIPS32R5EB-NEXT: sw $fp, 44($sp) # 4-byte Folded Spill ; MIPS32R5EB-NEXT: .cfi_offset 30, -4 ; MIPS32R5EB-NEXT: move $fp, $sp ; MIPS32R5EB-NEXT: .cfi_def_cfa_register 30 ; MIPS32R5EB-NEXT: addiu $1, $zero, -16 ; MIPS32R5EB-NEXT: and $sp, $sp, $1 ; MIPS32R5EB-NEXT: sw $7, 28($sp) ; MIPS32R5EB-NEXT: sw $6, 20($sp) ; MIPS32R5EB-NEXT: sw $5, 12($sp) ; MIPS32R5EB-NEXT: sw $4, 4($sp) ; MIPS32R5EB-NEXT: ld.d $w0, 16($sp) ; MIPS32R5EB-NEXT: ld.d $w1, 0($sp) ; MIPS32R5EB-NEXT: addv.d $w0, $w1, $w0 ; MIPS32R5EB-NEXT: shf.w $w0, $w0, 177 ; MIPS32R5EB-NEXT: copy_s.w $2, $w0[1] ; MIPS32R5EB-NEXT: copy_s.w $3, $w0[3] ; MIPS32R5EB-NEXT: move $sp, $fp ; MIPS32R5EB-NEXT: lw $fp, 44($sp) # 4-byte Folded Reload ; MIPS32R5EB-NEXT: addiu $sp, $sp, 48 ; MIPS32R5EB-NEXT: jr $ra ; MIPS32R5EB-NEXT: nop ; ; MIPS64R5EB-LABEL: i32_2: ; MIPS64R5EB: # %bb.0: ; MIPS64R5EB-NEXT: daddiu $sp, $sp, -32 ; MIPS64R5EB-NEXT: .cfi_def_cfa_offset 32 ; MIPS64R5EB-NEXT: sd $5, 16($sp) ; MIPS64R5EB-NEXT: sd $4, 24($sp) ; MIPS64R5EB-NEXT: ldi.b $w0, 0 ; MIPS64R5EB-NEXT: lw $1, 16($sp) ; MIPS64R5EB-NEXT: move.v $w1, $w0 ; MIPS64R5EB-NEXT: insert.d $w1[0], $1 ; MIPS64R5EB-NEXT: insert.d $w1[1], $5 ; MIPS64R5EB-NEXT: lw $1, 24($sp) ; MIPS64R5EB-NEXT: insert.d $w0[0], $1 ; MIPS64R5EB-NEXT: insert.d $w0[1], $4 ; MIPS64R5EB-NEXT: addv.d $w0, $w0, $w1 ; MIPS64R5EB-NEXT: copy_s.d $1, $w0[0] ; MIPS64R5EB-NEXT: copy_s.d $2, $w0[1] ; MIPS64R5EB-NEXT: sw $2, 12($sp) ; MIPS64R5EB-NEXT: sw $1, 8($sp) ; MIPS64R5EB-NEXT: ld $2, 8($sp) ; MIPS64R5EB-NEXT: daddiu $sp, $sp, 32 ; MIPS64R5EB-NEXT: jr $ra ; MIPS64R5EB-NEXT: nop ; ; MIPS32R5EL-LABEL: i32_2: ; MIPS32R5EL: # %bb.0: ; MIPS32R5EL-NEXT: addiu $sp, $sp, -48 ; MIPS32R5EL-NEXT: .cfi_def_cfa_offset 48 ; MIPS32R5EL-NEXT: sw $fp, 44($sp) # 4-byte Folded Spill ; MIPS32R5EL-NEXT: .cfi_offset 30, -4 ; MIPS32R5EL-NEXT: move $fp, $sp ; MIPS32R5EL-NEXT: .cfi_def_cfa_register 30 ; MIPS32R5EL-NEXT: addiu $1, $zero, -16 ; MIPS32R5EL-NEXT: and $sp, $sp, $1 ; MIPS32R5EL-NEXT: sw $7, 24($sp) ; MIPS32R5EL-NEXT: sw $6, 16($sp) ; MIPS32R5EL-NEXT: sw $5, 8($sp) ; MIPS32R5EL-NEXT: sw $4, 0($sp) ; MIPS32R5EL-NEXT: ld.d $w0, 16($sp) ; MIPS32R5EL-NEXT: ld.d $w1, 0($sp) ; MIPS32R5EL-NEXT: addv.d $w0, $w1, $w0 ; MIPS32R5EL-NEXT: copy_s.w $2, $w0[0] ; MIPS32R5EL-NEXT: copy_s.w $3, $w0[2] ; MIPS32R5EL-NEXT: move $sp, $fp ; MIPS32R5EL-NEXT: lw $fp, 44($sp) # 4-byte Folded Reload ; MIPS32R5EL-NEXT: addiu $sp, $sp, 48 ; MIPS32R5EL-NEXT: jr $ra ; MIPS32R5EL-NEXT: nop ; ; MIPS64R5EL-LABEL: i32_2: ; MIPS64R5EL: # %bb.0: ; MIPS64R5EL-NEXT: daddiu $sp, $sp, -32 ; MIPS64R5EL-NEXT: .cfi_def_cfa_offset 32 ; MIPS64R5EL-NEXT: sd $5, 16($sp) ; MIPS64R5EL-NEXT: sd $4, 24($sp) ; MIPS64R5EL-NEXT: ldi.b $w0, 0 ; MIPS64R5EL-NEXT: lw $1, 20($sp) ; MIPS64R5EL-NEXT: move.v $w1, $w0 ; MIPS64R5EL-NEXT: insert.d $w1[0], $5 ; MIPS64R5EL-NEXT: insert.d $w1[1], $1 ; MIPS64R5EL-NEXT: insert.d $w0[0], $4 ; MIPS64R5EL-NEXT: lw $1, 28($sp) ; MIPS64R5EL-NEXT: insert.d $w0[1], $1 ; MIPS64R5EL-NEXT: addv.d $w0, $w0, $w1 ; MIPS64R5EL-NEXT: copy_s.d $1, $w0[0] ; MIPS64R5EL-NEXT: copy_s.d $2, $w0[1] ; MIPS64R5EL-NEXT: sw $2, 12($sp) ; MIPS64R5EL-NEXT: sw $1, 8($sp) ; MIPS64R5EL-NEXT: ld $2, 8($sp) ; MIPS64R5EL-NEXT: daddiu $sp, $sp, 32 ; MIPS64R5EL-NEXT: jr $ra ; MIPS64R5EL-NEXT: nop %1 = add <2 x i32> %a, %b ret <2 x i32> %1 } define <4 x i32> @i32_4(<4 x i32> %a, <4 x i32> %b) { ; MIPS32-LABEL: i32_4: ; MIPS32: # %bb.0: ; MIPS32-NEXT: lw $1, 20($sp) ; MIPS32-NEXT: lw $2, 16($sp) ; MIPS32-NEXT: addu $2, $4, $2 ; MIPS32-NEXT: addu $3, $5, $1 ; MIPS32-NEXT: lw $1, 24($sp) ; MIPS32-NEXT: addu $4, $6, $1 ; MIPS32-NEXT: lw $1, 28($sp) ; MIPS32-NEXT: addu $5, $7, $1 ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: nop ; ; MIPS64-LABEL: i32_4: ; MIPS64: # %bb.0: ; MIPS64-NEXT: sll $1, $6, 0 ; MIPS64-NEXT: sll $2, $4, 0 ; MIPS64-NEXT: addu $1, $2, $1 ; MIPS64-NEXT: dsll $1, $1, 32 ; MIPS64-NEXT: sll $2, $7, 0 ; MIPS64-NEXT: sll $3, $5, 0 ; MIPS64-NEXT: addu $2, $3, $2 ; MIPS64-NEXT: dsrl $3, $6, 32 ; MIPS64-NEXT: dsll $6, $2, 32 ; MIPS64-NEXT: dsrl $1, $1, 32 ; MIPS64-NEXT: sll $2, $3, 0 ; MIPS64-NEXT: dsrl $3, $4, 32 ; MIPS64-NEXT: sll $3, $3, 0 ; MIPS64-NEXT: addu $2, $3, $2 ; MIPS64-NEXT: dsll $2, $2, 32 ; MIPS64-NEXT: dsrl $3, $7, 32 ; MIPS64-NEXT: or $2, $1, $2 ; MIPS64-NEXT: dsrl $1, $6, 32 ; MIPS64-NEXT: sll $3, $3, 0 ; MIPS64-NEXT: dsrl $4, $5, 32 ; MIPS64-NEXT: sll $4, $4, 0 ; MIPS64-NEXT: addu $3, $4, $3 ; MIPS64-NEXT: dsll $3, $3, 32 ; MIPS64-NEXT: or $3, $1, $3 ; MIPS64-NEXT: jr $ra ; MIPS64-NEXT: nop ; ; MIPS32R5-LABEL: i32_4: ; MIPS32R5: # %bb.0: ; MIPS32R5-NEXT: ldi.b $w0, 0 ; MIPS32R5-NEXT: lw $1, 20($sp) ; MIPS32R5-NEXT: lw $2, 16($sp) ; MIPS32R5-NEXT: move.v $w1, $w0 ; MIPS32R5-NEXT: insert.w $w1[0], $2 ; MIPS32R5-NEXT: insert.w $w1[1], $1 ; MIPS32R5-NEXT: lw $1, 24($sp) ; MIPS32R5-NEXT: insert.w $w1[2], $1 ; MIPS32R5-NEXT: lw $1, 28($sp) ; MIPS32R5-NEXT: insert.w $w1[3], $1 ; MIPS32R5-NEXT: insert.w $w0[0], $4 ; MIPS32R5-NEXT: insert.w $w0[1], $5 ; MIPS32R5-NEXT: insert.w $w0[2], $6 ; MIPS32R5-NEXT: insert.w $w0[3], $7 ; MIPS32R5-NEXT: addv.w $w0, $w0, $w1 ; MIPS32R5-NEXT: copy_s.w $2, $w0[0] ; MIPS32R5-NEXT: copy_s.w $3, $w0[1] ; MIPS32R5-NEXT: copy_s.w $4, $w0[2] ; MIPS32R5-NEXT: copy_s.w $5, $w0[3] ; MIPS32R5-NEXT: jr $ra ; MIPS32R5-NEXT: nop ; ; MIPS64R5EB-LABEL: i32_4: ; MIPS64R5EB: # %bb.0: ; MIPS64R5EB-NEXT: ldi.b $w0, 0 ; MIPS64R5EB-NEXT: move.v $w1, $w0 ; MIPS64R5EB-NEXT: insert.d $w1[0], $6 ; MIPS64R5EB-NEXT: insert.d $w1[1], $7 ; MIPS64R5EB-NEXT: shf.w $w1, $w1, 177 ; MIPS64R5EB-NEXT: insert.d $w0[0], $4 ; MIPS64R5EB-NEXT: insert.d $w0[1], $5 ; MIPS64R5EB-NEXT: shf.w $w0, $w0, 177 ; MIPS64R5EB-NEXT: addv.w $w0, $w0, $w1 ; MIPS64R5EB-NEXT: shf.w $w0, $w0, 177 ; MIPS64R5EB-NEXT: copy_s.d $2, $w0[0] ; MIPS64R5EB-NEXT: copy_s.d $3, $w0[1] ; MIPS64R5EB-NEXT: jr $ra ; MIPS64R5EB-NEXT: nop ; ; MIPS64R5EL-LABEL: i32_4: ; MIPS64R5EL: # %bb.0: ; MIPS64R5EL-NEXT: ldi.b $w0, 0 ; MIPS64R5EL-NEXT: move.v $w1, $w0 ; MIPS64R5EL-NEXT: insert.d $w1[0], $6 ; MIPS64R5EL-NEXT: insert.d $w1[1], $7 ; MIPS64R5EL-NEXT: insert.d $w0[0], $4 ; MIPS64R5EL-NEXT: insert.d $w0[1], $5 ; MIPS64R5EL-NEXT: addv.w $w0, $w0, $w1 ; MIPS64R5EL-NEXT: copy_s.d $2, $w0[0] ; MIPS64R5EL-NEXT: copy_s.d $3, $w0[1] ; MIPS64R5EL-NEXT: jr $ra ; MIPS64R5EL-NEXT: nop %1 = add <4 x i32> %a, %b ret <4 x i32> %1 } define <2 x i64> @i64_2(<2 x i64> %a, <2 x i64> %b) { ; MIPS32EB-LABEL: i64_2: ; MIPS32EB: # %bb.0: ; MIPS32EB-NEXT: lw $1, 16($sp) ; MIPS32EB-NEXT: addu $1, $4, $1 ; MIPS32EB-NEXT: lw $2, 20($sp) ; MIPS32EB-NEXT: addu $3, $5, $2 ; MIPS32EB-NEXT: sltu $2, $3, $5 ; MIPS32EB-NEXT: lw $4, 24($sp) ; MIPS32EB-NEXT: addu $2, $1, $2 ; MIPS32EB-NEXT: addu $1, $6, $4 ; MIPS32EB-NEXT: lw $4, 28($sp) ; MIPS32EB-NEXT: addu $5, $7, $4 ; MIPS32EB-NEXT: sltu $4, $5, $7 ; MIPS32EB-NEXT: addu $4, $1, $4 ; MIPS32EB-NEXT: jr $ra ; MIPS32EB-NEXT: nop ; ; MIPS64-LABEL: i64_2: ; MIPS64: # %bb.0: ; MIPS64-NEXT: daddu $2, $4, $6 ; MIPS64-NEXT: daddu $3, $5, $7 ; MIPS64-NEXT: jr $ra ; MIPS64-NEXT: nop ; ; MIPS32R5EB-LABEL: i64_2: ; MIPS32R5EB: # %bb.0: ; MIPS32R5EB-NEXT: ldi.b $w0, 0 ; MIPS32R5EB-NEXT: lw $1, 20($sp) ; MIPS32R5EB-NEXT: lw $2, 16($sp) ; MIPS32R5EB-NEXT: move.v $w1, $w0 ; MIPS32R5EB-NEXT: insert.w $w1[0], $2 ; MIPS32R5EB-NEXT: insert.w $w1[1], $1 ; MIPS32R5EB-NEXT: lw $1, 24($sp) ; MIPS32R5EB-NEXT: insert.w $w0[0], $4 ; MIPS32R5EB-NEXT: insert.w $w1[2], $1 ; MIPS32R5EB-NEXT: lw $1, 28($sp) ; MIPS32R5EB-NEXT: insert.w $w1[3], $1 ; MIPS32R5EB-NEXT: shf.w $w1, $w1, 177 ; MIPS32R5EB-NEXT: insert.w $w0[1], $5 ; MIPS32R5EB-NEXT: insert.w $w0[2], $6 ; MIPS32R5EB-NEXT: insert.w $w0[3], $7 ; MIPS32R5EB-NEXT: shf.w $w0, $w0, 177 ; MIPS32R5EB-NEXT: addv.d $w0, $w0, $w1 ; MIPS32R5EB-NEXT: shf.w $w0, $w0, 177 ; MIPS32R5EB-NEXT: copy_s.w $2, $w0[0] ; MIPS32R5EB-NEXT: copy_s.w $3, $w0[1] ; MIPS32R5EB-NEXT: copy_s.w $4, $w0[2] ; MIPS32R5EB-NEXT: copy_s.w $5, $w0[3] ; MIPS32R5EB-NEXT: jr $ra ; MIPS32R5EB-NEXT: nop ; ; MIPS64R5-LABEL: i64_2: ; MIPS64R5: # %bb.0: ; MIPS64R5-NEXT: ldi.b $w0, 0 ; MIPS64R5-NEXT: move.v $w1, $w0 ; MIPS64R5-NEXT: insert.d $w1[0], $6 ; MIPS64R5-NEXT: insert.d $w1[1], $7 ; MIPS64R5-NEXT: insert.d $w0[0], $4 ; MIPS64R5-NEXT: insert.d $w0[1], $5 ; MIPS64R5-NEXT: addv.d $w0, $w0, $w1 ; MIPS64R5-NEXT: copy_s.d $2, $w0[0] ; MIPS64R5-NEXT: copy_s.d $3, $w0[1] ; MIPS64R5-NEXT: jr $ra ; MIPS64R5-NEXT: nop ; ; MIPS32EL-LABEL: i64_2: ; MIPS32EL: # %bb.0: ; MIPS32EL-NEXT: lw $1, 20($sp) ; MIPS32EL-NEXT: addu $1, $5, $1 ; MIPS32EL-NEXT: lw $2, 16($sp) ; MIPS32EL-NEXT: addu $2, $4, $2 ; MIPS32EL-NEXT: sltu $3, $2, $4 ; MIPS32EL-NEXT: lw $4, 28($sp) ; MIPS32EL-NEXT: addu $3, $1, $3 ; MIPS32EL-NEXT: addu $1, $7, $4 ; MIPS32EL-NEXT: lw $4, 24($sp) ; MIPS32EL-NEXT: addu $4, $6, $4 ; MIPS32EL-NEXT: sltu $5, $4, $6 ; MIPS32EL-NEXT: addu $5, $1, $5 ; MIPS32EL-NEXT: jr $ra ; MIPS32EL-NEXT: nop ; ; MIPS32R5EL-LABEL: i64_2: ; MIPS32R5EL: # %bb.0: ; MIPS32R5EL-NEXT: ldi.b $w0, 0 ; MIPS32R5EL-NEXT: lw $1, 20($sp) ; MIPS32R5EL-NEXT: lw $2, 16($sp) ; MIPS32R5EL-NEXT: move.v $w1, $w0 ; MIPS32R5EL-NEXT: insert.w $w1[0], $2 ; MIPS32R5EL-NEXT: insert.w $w1[1], $1 ; MIPS32R5EL-NEXT: lw $1, 24($sp) ; MIPS32R5EL-NEXT: insert.w $w1[2], $1 ; MIPS32R5EL-NEXT: lw $1, 28($sp) ; MIPS32R5EL-NEXT: insert.w $w1[3], $1 ; MIPS32R5EL-NEXT: insert.w $w0[0], $4 ; MIPS32R5EL-NEXT: insert.w $w0[1], $5 ; MIPS32R5EL-NEXT: insert.w $w0[2], $6 ; MIPS32R5EL-NEXT: insert.w $w0[3], $7 ; MIPS32R5EL-NEXT: addv.d $w0, $w0, $w1 ; MIPS32R5EL-NEXT: copy_s.w $2, $w0[0] ; MIPS32R5EL-NEXT: copy_s.w $3, $w0[1] ; MIPS32R5EL-NEXT: copy_s.w $4, $w0[2] ; MIPS32R5EL-NEXT: copy_s.w $5, $w0[3] ; MIPS32R5EL-NEXT: jr $ra ; MIPS32R5EL-NEXT: nop %1 = add <2 x i64> %a, %b ret <2 x i64> %1 } ; The MIPS vector ABI treats vectors of floats differently to vectors of ; integers. ; For arguments floating pointer vectors are bitcasted to integer vectors whose ; elements are of GPR width and where the element count is deduced from ; the length of the floating point vector divided by the size of the GPRs. ; For returns, integer vectors are passed via the GPR register set, but ; floating point vectors are returned via a hidden sret pointer. ; For testing purposes we skip returning values here and test them below ; instead. @float_res_v2f32 = external global <2 x float> define void @float_2(<2 x float> %a, <2 x float> %b) { ; MIPS32-LABEL: float_2: ; MIPS32: # %bb.0: ; MIPS32-NEXT: mtc1 $7, $f0 ; MIPS32-NEXT: mtc1 $5, $f1 ; MIPS32-NEXT: add.s $f0, $f1, $f0 ; MIPS32-NEXT: lui $1, %hi(float_res_v2f32) ; MIPS32-NEXT: addiu $2, $1, %lo(float_res_v2f32) ; MIPS32-NEXT: swc1 $f0, 4($2) ; MIPS32-NEXT: mtc1 $6, $f0 ; MIPS32-NEXT: mtc1 $4, $f1 ; MIPS32-NEXT: add.s $f0, $f1, $f0 ; MIPS32-NEXT: swc1 $f0, %lo(float_res_v2f32)($1) ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: nop ; ; MIPS64EB-LABEL: float_2: ; MIPS64EB: # %bb.0: ; MIPS64EB-NEXT: lui $1, %hi(%neg(%gp_rel(float_2))) ; MIPS64EB-NEXT: daddu $1, $1, $25 ; MIPS64EB-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(float_2))) ; MIPS64EB-NEXT: sll $2, $5, 0 ; MIPS64EB-NEXT: mtc1 $2, $f0 ; MIPS64EB-NEXT: sll $2, $4, 0 ; MIPS64EB-NEXT: mtc1 $2, $f1 ; MIPS64EB-NEXT: add.s $f0, $f1, $f0 ; MIPS64EB-NEXT: dsrl $2, $5, 32 ; MIPS64EB-NEXT: sll $2, $2, 0 ; MIPS64EB-NEXT: ld $1, %got_disp(float_res_v2f32)($1) ; MIPS64EB-NEXT: swc1 $f0, 4($1) ; MIPS64EB-NEXT: mtc1 $2, $f0 ; MIPS64EB-NEXT: dsrl $2, $4, 32 ; MIPS64EB-NEXT: sll $2, $2, 0 ; MIPS64EB-NEXT: mtc1 $2, $f1 ; MIPS64EB-NEXT: add.s $f0, $f1, $f0 ; MIPS64EB-NEXT: swc1 $f0, 0($1) ; MIPS64EB-NEXT: jr $ra ; MIPS64EB-NEXT: nop ; ; MIPS32R5-LABEL: float_2: ; MIPS32R5: # %bb.0: ; MIPS32R5-NEXT: addiu $sp, $sp, -48 ; MIPS32R5-NEXT: .cfi_def_cfa_offset 48 ; MIPS32R5-NEXT: sw $fp, 44($sp) # 4-byte Folded Spill ; MIPS32R5-NEXT: .cfi_offset 30, -4 ; MIPS32R5-NEXT: move $fp, $sp ; MIPS32R5-NEXT: .cfi_def_cfa_register 30 ; MIPS32R5-NEXT: addiu $1, $zero, -16 ; MIPS32R5-NEXT: and $sp, $sp, $1 ; MIPS32R5-NEXT: sw $7, 20($sp) ; MIPS32R5-NEXT: sw $6, 16($sp) ; MIPS32R5-NEXT: sw $5, 4($sp) ; MIPS32R5-NEXT: sw $4, 0($sp) ; MIPS32R5-NEXT: ld.w $w0, 16($sp) ; MIPS32R5-NEXT: ld.w $w1, 0($sp) ; MIPS32R5-NEXT: fadd.w $w0, $w1, $w0 ; MIPS32R5-NEXT: lui $1, %hi(float_res_v2f32) ; MIPS32R5-NEXT: addiu $2, $1, %lo(float_res_v2f32) ; MIPS32R5-NEXT: splati.w $w1, $w0[1] ; MIPS32R5-NEXT: swc1 $f1, 4($2) ; MIPS32R5-NEXT: swc1 $f0, %lo(float_res_v2f32)($1) ; MIPS32R5-NEXT: move $sp, $fp ; MIPS32R5-NEXT: lw $fp, 44($sp) # 4-byte Folded Reload ; MIPS32R5-NEXT: addiu $sp, $sp, 48 ; MIPS32R5-NEXT: jr $ra ; MIPS32R5-NEXT: nop ; ; MIPS64R5EB-LABEL: float_2: ; MIPS64R5EB: # %bb.0: ; MIPS64R5EB-NEXT: daddiu $sp, $sp, -32 ; MIPS64R5EB-NEXT: .cfi_def_cfa_offset 32 ; MIPS64R5EB-NEXT: lui $1, %hi(%neg(%gp_rel(float_2))) ; MIPS64R5EB-NEXT: daddu $1, $1, $25 ; MIPS64R5EB-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(float_2))) ; MIPS64R5EB-NEXT: sd $5, 16($sp) ; MIPS64R5EB-NEXT: sd $4, 0($sp) ; MIPS64R5EB-NEXT: ld.w $w0, 16($sp) ; MIPS64R5EB-NEXT: ld.w $w1, 0($sp) ; MIPS64R5EB-NEXT: fadd.w $w0, $w1, $w0 ; MIPS64R5EB-NEXT: shf.w $w0, $w0, 177 ; MIPS64R5EB-NEXT: copy_s.d $2, $w0[0] ; MIPS64R5EB-NEXT: ld $1, %got_disp(float_res_v2f32)($1) ; MIPS64R5EB-NEXT: sd $2, 0($1) ; MIPS64R5EB-NEXT: daddiu $sp, $sp, 32 ; MIPS64R5EB-NEXT: jr $ra ; MIPS64R5EB-NEXT: nop ; ; MIPS64EL-LABEL: float_2: ; MIPS64EL: # %bb.0: ; MIPS64EL-NEXT: lui $1, %hi(%neg(%gp_rel(float_2))) ; MIPS64EL-NEXT: daddu $1, $1, $25 ; MIPS64EL-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(float_2))) ; MIPS64EL-NEXT: sll $2, $5, 0 ; MIPS64EL-NEXT: mtc1 $2, $f0 ; MIPS64EL-NEXT: sll $2, $4, 0 ; MIPS64EL-NEXT: mtc1 $2, $f1 ; MIPS64EL-NEXT: add.s $f0, $f1, $f0 ; MIPS64EL-NEXT: dsrl $2, $5, 32 ; MIPS64EL-NEXT: sll $2, $2, 0 ; MIPS64EL-NEXT: ld $1, %got_disp(float_res_v2f32)($1) ; MIPS64EL-NEXT: swc1 $f0, 0($1) ; MIPS64EL-NEXT: mtc1 $2, $f0 ; MIPS64EL-NEXT: dsrl $2, $4, 32 ; MIPS64EL-NEXT: sll $2, $2, 0 ; MIPS64EL-NEXT: mtc1 $2, $f1 ; MIPS64EL-NEXT: add.s $f0, $f1, $f0 ; MIPS64EL-NEXT: swc1 $f0, 4($1) ; MIPS64EL-NEXT: jr $ra ; MIPS64EL-NEXT: nop ; ; MIPS64R5EL-LABEL: float_2: ; MIPS64R5EL: # %bb.0: ; MIPS64R5EL-NEXT: daddiu $sp, $sp, -32 ; MIPS64R5EL-NEXT: .cfi_def_cfa_offset 32 ; MIPS64R5EL-NEXT: lui $1, %hi(%neg(%gp_rel(float_2))) ; MIPS64R5EL-NEXT: daddu $1, $1, $25 ; MIPS64R5EL-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(float_2))) ; MIPS64R5EL-NEXT: sd $5, 16($sp) ; MIPS64R5EL-NEXT: sd $4, 0($sp) ; MIPS64R5EL-NEXT: ld.w $w0, 16($sp) ; MIPS64R5EL-NEXT: ld.w $w1, 0($sp) ; MIPS64R5EL-NEXT: fadd.w $w0, $w1, $w0 ; MIPS64R5EL-NEXT: copy_s.d $2, $w0[0] ; MIPS64R5EL-NEXT: ld $1, %got_disp(float_res_v2f32)($1) ; MIPS64R5EL-NEXT: sd $2, 0($1) ; MIPS64R5EL-NEXT: daddiu $sp, $sp, 32 ; MIPS64R5EL-NEXT: jr $ra ; MIPS64R5EL-NEXT: nop %1 = fadd <2 x float> %a, %b store <2 x float> %1, <2 x float> * @float_res_v2f32 ret void } @float_res_v4f32 = external global <4 x float> ; For MSA this case is suboptimal, the 4 loads can be combined into a single ; ld.w. define void @float_4(<4 x float> %a, <4 x float> %b) { ; MIPS32-LABEL: float_4: ; MIPS32: # %bb.0: ; MIPS32-NEXT: mtc1 $7, $f0 ; MIPS32-NEXT: mtc1 $6, $f1 ; MIPS32-NEXT: lwc1 $f2, 28($sp) ; MIPS32-NEXT: lwc1 $f3, 24($sp) ; MIPS32-NEXT: add.s $f1, $f1, $f3 ; MIPS32-NEXT: add.s $f0, $f0, $f2 ; MIPS32-NEXT: mtc1 $5, $f2 ; MIPS32-NEXT: lui $1, %hi(float_res_v4f32) ; MIPS32-NEXT: addiu $2, $1, %lo(float_res_v4f32) ; MIPS32-NEXT: lwc1 $f3, 20($sp) ; MIPS32-NEXT: swc1 $f0, 12($2) ; MIPS32-NEXT: swc1 $f1, 8($2) ; MIPS32-NEXT: add.s $f0, $f2, $f3 ; MIPS32-NEXT: swc1 $f0, 4($2) ; MIPS32-NEXT: mtc1 $4, $f0 ; MIPS32-NEXT: lwc1 $f1, 16($sp) ; MIPS32-NEXT: add.s $f0, $f0, $f1 ; MIPS32-NEXT: swc1 $f0, %lo(float_res_v4f32)($1) ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: nop ; ; MIPS64EB-LABEL: float_4: ; MIPS64EB: # %bb.0: ; MIPS64EB-NEXT: lui $1, %hi(%neg(%gp_rel(float_4))) ; MIPS64EB-NEXT: daddu $1, $1, $25 ; MIPS64EB-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(float_4))) ; MIPS64EB-NEXT: dsrl $2, $7, 32 ; MIPS64EB-NEXT: sll $2, $2, 0 ; MIPS64EB-NEXT: sll $3, $4, 0 ; MIPS64EB-NEXT: sll $8, $6, 0 ; MIPS64EB-NEXT: sll $7, $7, 0 ; MIPS64EB-NEXT: mtc1 $8, $f0 ; MIPS64EB-NEXT: mtc1 $3, $f1 ; MIPS64EB-NEXT: mtc1 $2, $f2 ; MIPS64EB-NEXT: dsrl $2, $5, 32 ; MIPS64EB-NEXT: sll $2, $2, 0 ; MIPS64EB-NEXT: mtc1 $2, $f3 ; MIPS64EB-NEXT: add.s $f2, $f3, $f2 ; MIPS64EB-NEXT: add.s $f0, $f1, $f0 ; MIPS64EB-NEXT: mtc1 $7, $f1 ; MIPS64EB-NEXT: sll $2, $5, 0 ; MIPS64EB-NEXT: mtc1 $2, $f3 ; MIPS64EB-NEXT: add.s $f1, $f3, $f1 ; MIPS64EB-NEXT: dsrl $2, $6, 32 ; MIPS64EB-NEXT: sll $2, $2, 0 ; MIPS64EB-NEXT: ld $1, %got_disp(float_res_v4f32)($1) ; MIPS64EB-NEXT: swc1 $f1, 12($1) ; MIPS64EB-NEXT: swc1 $f0, 4($1) ; MIPS64EB-NEXT: swc1 $f2, 8($1) ; MIPS64EB-NEXT: mtc1 $2, $f0 ; MIPS64EB-NEXT: dsrl $2, $4, 32 ; MIPS64EB-NEXT: sll $2, $2, 0 ; MIPS64EB-NEXT: mtc1 $2, $f1 ; MIPS64EB-NEXT: add.s $f0, $f1, $f0 ; MIPS64EB-NEXT: swc1 $f0, 0($1) ; MIPS64EB-NEXT: jr $ra ; MIPS64EB-NEXT: nop ; ; MIPS32R5-LABEL: float_4: ; MIPS32R5: # %bb.0: ; MIPS32R5-NEXT: ldi.b $w0, 0 ; MIPS32R5-NEXT: lw $1, 20($sp) ; MIPS32R5-NEXT: lw $2, 16($sp) ; MIPS32R5-NEXT: move.v $w1, $w0 ; MIPS32R5-NEXT: insert.w $w1[0], $2 ; MIPS32R5-NEXT: insert.w $w1[1], $1 ; MIPS32R5-NEXT: lw $1, 24($sp) ; MIPS32R5-NEXT: insert.w $w1[2], $1 ; MIPS32R5-NEXT: lw $1, 28($sp) ; MIPS32R5-NEXT: insert.w $w1[3], $1 ; MIPS32R5-NEXT: insert.w $w0[0], $4 ; MIPS32R5-NEXT: insert.w $w0[1], $5 ; MIPS32R5-NEXT: insert.w $w0[2], $6 ; MIPS32R5-NEXT: insert.w $w0[3], $7 ; MIPS32R5-NEXT: fadd.w $w0, $w0, $w1 ; MIPS32R5-NEXT: lui $1, %hi(float_res_v4f32) ; MIPS32R5-NEXT: addiu $1, $1, %lo(float_res_v4f32) ; MIPS32R5-NEXT: st.w $w0, 0($1) ; MIPS32R5-NEXT: jr $ra ; MIPS32R5-NEXT: nop ; ; MIPS64R5EB-LABEL: float_4: ; MIPS64R5EB: # %bb.0: ; MIPS64R5EB-NEXT: lui $1, %hi(%neg(%gp_rel(float_4))) ; MIPS64R5EB-NEXT: daddu $1, $1, $25 ; MIPS64R5EB-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(float_4))) ; MIPS64R5EB-NEXT: ldi.b $w0, 0 ; MIPS64R5EB-NEXT: move.v $w1, $w0 ; MIPS64R5EB-NEXT: insert.d $w1[0], $6 ; MIPS64R5EB-NEXT: insert.d $w1[1], $7 ; MIPS64R5EB-NEXT: shf.w $w1, $w1, 177 ; MIPS64R5EB-NEXT: insert.d $w0[0], $4 ; MIPS64R5EB-NEXT: insert.d $w0[1], $5 ; MIPS64R5EB-NEXT: shf.w $w0, $w0, 177 ; MIPS64R5EB-NEXT: fadd.w $w0, $w0, $w1 ; MIPS64R5EB-NEXT: ld $1, %got_disp(float_res_v4f32)($1) ; MIPS64R5EB-NEXT: st.w $w0, 0($1) ; MIPS64R5EB-NEXT: jr $ra ; MIPS64R5EB-NEXT: nop ; ; MIPS64EL-LABEL: float_4: ; MIPS64EL: # %bb.0: ; MIPS64EL-NEXT: lui $1, %hi(%neg(%gp_rel(float_4))) ; MIPS64EL-NEXT: daddu $1, $1, $25 ; MIPS64EL-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(float_4))) ; MIPS64EL-NEXT: dsrl $2, $7, 32 ; MIPS64EL-NEXT: sll $2, $2, 0 ; MIPS64EL-NEXT: sll $3, $4, 0 ; MIPS64EL-NEXT: sll $8, $6, 0 ; MIPS64EL-NEXT: sll $7, $7, 0 ; MIPS64EL-NEXT: mtc1 $8, $f0 ; MIPS64EL-NEXT: mtc1 $3, $f1 ; MIPS64EL-NEXT: mtc1 $2, $f2 ; MIPS64EL-NEXT: dsrl $2, $5, 32 ; MIPS64EL-NEXT: sll $2, $2, 0 ; MIPS64EL-NEXT: mtc1 $2, $f3 ; MIPS64EL-NEXT: add.s $f2, $f3, $f2 ; MIPS64EL-NEXT: add.s $f0, $f1, $f0 ; MIPS64EL-NEXT: mtc1 $7, $f1 ; MIPS64EL-NEXT: sll $2, $5, 0 ; MIPS64EL-NEXT: mtc1 $2, $f3 ; MIPS64EL-NEXT: add.s $f1, $f3, $f1 ; MIPS64EL-NEXT: dsrl $2, $6, 32 ; MIPS64EL-NEXT: sll $2, $2, 0 ; MIPS64EL-NEXT: ld $1, %got_disp(float_res_v4f32)($1) ; MIPS64EL-NEXT: swc1 $f1, 8($1) ; MIPS64EL-NEXT: swc1 $f0, 0($1) ; MIPS64EL-NEXT: swc1 $f2, 12($1) ; MIPS64EL-NEXT: mtc1 $2, $f0 ; MIPS64EL-NEXT: dsrl $2, $4, 32 ; MIPS64EL-NEXT: sll $2, $2, 0 ; MIPS64EL-NEXT: mtc1 $2, $f1 ; MIPS64EL-NEXT: add.s $f0, $f1, $f0 ; MIPS64EL-NEXT: swc1 $f0, 4($1) ; MIPS64EL-NEXT: jr $ra ; MIPS64EL-NEXT: nop ; ; MIPS64R5EL-LABEL: float_4: ; MIPS64R5EL: # %bb.0: ; MIPS64R5EL-NEXT: lui $1, %hi(%neg(%gp_rel(float_4))) ; MIPS64R5EL-NEXT: daddu $1, $1, $25 ; MIPS64R5EL-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(float_4))) ; MIPS64R5EL-NEXT: ldi.b $w0, 0 ; MIPS64R5EL-NEXT: move.v $w1, $w0 ; MIPS64R5EL-NEXT: insert.d $w1[0], $6 ; MIPS64R5EL-NEXT: insert.d $w1[1], $7 ; MIPS64R5EL-NEXT: insert.d $w0[0], $4 ; MIPS64R5EL-NEXT: insert.d $w0[1], $5 ; MIPS64R5EL-NEXT: fadd.w $w0, $w0, $w1 ; MIPS64R5EL-NEXT: ld $1, %got_disp(float_res_v4f32)($1) ; MIPS64R5EL-NEXT: st.w $w0, 0($1) ; MIPS64R5EL-NEXT: jr $ra ; MIPS64R5EL-NEXT: nop %1 = fadd <4 x float> %a, %b store <4 x float> %1, <4 x float> * @float_res_v4f32 ret void } @double_v2f64 = external global <2 x double> define void @double_2(<2 x double> %a, <2 x double> %b) { ; MIPS32-LABEL: double_2: ; MIPS32: # %bb.0: ; MIPS32-NEXT: addiu $sp, $sp, -32 ; MIPS32-NEXT: .cfi_def_cfa_offset 32 ; MIPS32-NEXT: lw $1, 60($sp) ; MIPS32-NEXT: sw $1, 12($sp) ; MIPS32-NEXT: lw $1, 56($sp) ; MIPS32-NEXT: sw $1, 8($sp) ; MIPS32-NEXT: sw $7, 28($sp) ; MIPS32-NEXT: sw $6, 24($sp) ; MIPS32-NEXT: ldc1 $f0, 8($sp) ; MIPS32-NEXT: ldc1 $f2, 24($sp) ; MIPS32-NEXT: add.d $f0, $f2, $f0 ; MIPS32-NEXT: lui $1, %hi(double_v2f64) ; MIPS32-NEXT: addiu $2, $1, %lo(double_v2f64) ; MIPS32-NEXT: lw $3, 52($sp) ; MIPS32-NEXT: sdc1 $f0, 8($2) ; MIPS32-NEXT: sw $3, 4($sp) ; MIPS32-NEXT: lw $2, 48($sp) ; MIPS32-NEXT: sw $2, 0($sp) ; MIPS32-NEXT: sw $5, 20($sp) ; MIPS32-NEXT: sw $4, 16($sp) ; MIPS32-NEXT: ldc1 $f0, 0($sp) ; MIPS32-NEXT: ldc1 $f2, 16($sp) ; MIPS32-NEXT: add.d $f0, $f2, $f0 ; MIPS32-NEXT: sdc1 $f0, %lo(double_v2f64)($1) ; MIPS32-NEXT: addiu $sp, $sp, 32 ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: nop ; ; MIPS64-LABEL: double_2: ; MIPS64: # %bb.0: ; MIPS64-NEXT: lui $1, %hi(%neg(%gp_rel(double_2))) ; MIPS64-NEXT: daddu $1, $1, $25 ; MIPS64-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(double_2))) ; MIPS64-NEXT: dmtc1 $7, $f0 ; MIPS64-NEXT: dmtc1 $5, $f1 ; MIPS64-NEXT: add.d $f0, $f1, $f0 ; MIPS64-NEXT: ld $1, %got_disp(double_v2f64)($1) ; MIPS64-NEXT: sdc1 $f0, 8($1) ; MIPS64-NEXT: dmtc1 $6, $f0 ; MIPS64-NEXT: dmtc1 $4, $f1 ; MIPS64-NEXT: add.d $f0, $f1, $f0 ; MIPS64-NEXT: sdc1 $f0, 0($1) ; MIPS64-NEXT: jr $ra ; MIPS64-NEXT: nop ; ; MIPS32R5EB-LABEL: double_2: ; MIPS32R5EB: # %bb.0: ; MIPS32R5EB-NEXT: ldi.b $w0, 0 ; MIPS32R5EB-NEXT: lw $1, 20($sp) ; MIPS32R5EB-NEXT: lw $2, 16($sp) ; MIPS32R5EB-NEXT: move.v $w1, $w0 ; MIPS32R5EB-NEXT: insert.w $w1[0], $2 ; MIPS32R5EB-NEXT: insert.w $w1[1], $1 ; MIPS32R5EB-NEXT: lw $1, 24($sp) ; MIPS32R5EB-NEXT: insert.w $w0[0], $4 ; MIPS32R5EB-NEXT: insert.w $w1[2], $1 ; MIPS32R5EB-NEXT: lw $1, 28($sp) ; MIPS32R5EB-NEXT: insert.w $w1[3], $1 ; MIPS32R5EB-NEXT: shf.w $w1, $w1, 177 ; MIPS32R5EB-NEXT: insert.w $w0[1], $5 ; MIPS32R5EB-NEXT: insert.w $w0[2], $6 ; MIPS32R5EB-NEXT: insert.w $w0[3], $7 ; MIPS32R5EB-NEXT: shf.w $w0, $w0, 177 ; MIPS32R5EB-NEXT: fadd.d $w0, $w0, $w1 ; MIPS32R5EB-NEXT: lui $1, %hi(double_v2f64) ; MIPS32R5EB-NEXT: addiu $1, $1, %lo(double_v2f64) ; MIPS32R5EB-NEXT: st.d $w0, 0($1) ; MIPS32R5EB-NEXT: jr $ra ; MIPS32R5EB-NEXT: nop ; ; MIPS64R5-LABEL: double_2: ; MIPS64R5: # %bb.0: ; MIPS64R5-NEXT: lui $1, %hi(%neg(%gp_rel(double_2))) ; MIPS64R5-NEXT: daddu $1, $1, $25 ; MIPS64R5-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(double_2))) ; MIPS64R5-NEXT: ldi.b $w0, 0 ; MIPS64R5-NEXT: move.v $w1, $w0 ; MIPS64R5-NEXT: insert.d $w1[0], $6 ; MIPS64R5-NEXT: insert.d $w1[1], $7 ; MIPS64R5-NEXT: insert.d $w0[0], $4 ; MIPS64R5-NEXT: insert.d $w0[1], $5 ; MIPS64R5-NEXT: fadd.d $w0, $w0, $w1 ; MIPS64R5-NEXT: ld $1, %got_disp(double_v2f64)($1) ; MIPS64R5-NEXT: st.d $w0, 0($1) ; MIPS64R5-NEXT: jr $ra ; MIPS64R5-NEXT: nop ; ; MIPS32R5EL-LABEL: double_2: ; MIPS32R5EL: # %bb.0: ; MIPS32R5EL-NEXT: ldi.b $w0, 0 ; MIPS32R5EL-NEXT: lw $1, 20($sp) ; MIPS32R5EL-NEXT: lw $2, 16($sp) ; MIPS32R5EL-NEXT: move.v $w1, $w0 ; MIPS32R5EL-NEXT: insert.w $w1[0], $2 ; MIPS32R5EL-NEXT: insert.w $w1[1], $1 ; MIPS32R5EL-NEXT: lw $1, 24($sp) ; MIPS32R5EL-NEXT: insert.w $w1[2], $1 ; MIPS32R5EL-NEXT: lw $1, 28($sp) ; MIPS32R5EL-NEXT: insert.w $w1[3], $1 ; MIPS32R5EL-NEXT: insert.w $w0[0], $4 ; MIPS32R5EL-NEXT: insert.w $w0[1], $5 ; MIPS32R5EL-NEXT: insert.w $w0[2], $6 ; MIPS32R5EL-NEXT: insert.w $w0[3], $7 ; MIPS32R5EL-NEXT: fadd.d $w0, $w0, $w1 ; MIPS32R5EL-NEXT: lui $1, %hi(double_v2f64) ; MIPS32R5EL-NEXT: addiu $1, $1, %lo(double_v2f64) ; MIPS32R5EL-NEXT: st.d $w0, 0($1) ; MIPS32R5EL-NEXT: jr $ra ; MIPS32R5EL-NEXT: nop %1 = fadd <2 x double> %a, %b store <2 x double> %1, <2 x double> * @double_v2f64 ret void } ; Return value testing. ; Integer vectors are returned in $2, $3, $4, $5 for O32, $2, $3 for N32/N64 ; Floating point vectors are returned through a hidden sret pointer. @gv2i8 = global <2 x i8> @gv4i8 = global <4 x i8> @gv8i8 = global <8 x i8> @gv16i8 = global <16 x i8> @gv2i16 = global <2 x i16> @gv4i16 = global <4 x i16> @gv8i16 = global <8 x i16> @gv2i32 = global <2 x i32> @gv4i32 = global <4 x i32> @gv2i64 = global <2 x i64> ; FIXME: why is this lh instead of lhu on mips64? define <2 x i8> @ret_2_i8() { ; MIPS32-LABEL: ret_2_i8: ; MIPS32: # %bb.0: ; MIPS32-NEXT: lui $1, %hi(gv2i8) ; MIPS32-NEXT: lhu $2, %lo(gv2i8)($1) ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: nop ; ; MIPS64-LABEL: ret_2_i8: ; MIPS64: # %bb.0: ; MIPS64-NEXT: lui $1, %hi(%neg(%gp_rel(ret_2_i8))) ; MIPS64-NEXT: daddu $1, $1, $25 ; MIPS64-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(ret_2_i8))) ; MIPS64-NEXT: ld $1, %got_disp(gv2i8)($1) ; MIPS64-NEXT: lh $2, 0($1) ; MIPS64-NEXT: jr $ra ; MIPS64-NEXT: nop ; ; MIPS32R5-LABEL: ret_2_i8: ; MIPS32R5: # %bb.0: ; MIPS32R5-NEXT: lui $1, %hi(gv2i8) ; MIPS32R5-NEXT: lhu $2, %lo(gv2i8)($1) ; MIPS32R5-NEXT: jr $ra ; MIPS32R5-NEXT: nop ; ; MIPS64R5-LABEL: ret_2_i8: ; MIPS64R5: # %bb.0: ; MIPS64R5-NEXT: lui $1, %hi(%neg(%gp_rel(ret_2_i8))) ; MIPS64R5-NEXT: daddu $1, $1, $25 ; MIPS64R5-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(ret_2_i8))) ; MIPS64R5-NEXT: ld $1, %got_disp(gv2i8)($1) ; MIPS64R5-NEXT: lh $2, 0($1) ; MIPS64R5-NEXT: jr $ra ; MIPS64R5-NEXT: nop %1 = load <2 x i8>, <2 x i8> * @gv2i8 ret <2 x i8> %1 } define <4 x i8> @ret_4_i8() { ; MIPS32-LABEL: ret_4_i8: ; MIPS32: # %bb.0: ; MIPS32-NEXT: lui $1, %hi(gv4i8) ; MIPS32-NEXT: lw $2, %lo(gv4i8)($1) ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: nop ; ; MIPS64-LABEL: ret_4_i8: ; MIPS64: # %bb.0: ; MIPS64-NEXT: lui $1, %hi(%neg(%gp_rel(ret_4_i8))) ; MIPS64-NEXT: daddu $1, $1, $25 ; MIPS64-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(ret_4_i8))) ; MIPS64-NEXT: ld $1, %got_disp(gv4i8)($1) ; MIPS64-NEXT: lw $2, 0($1) ; MIPS64-NEXT: jr $ra ; MIPS64-NEXT: nop ; ; MIPS32R5-LABEL: ret_4_i8: ; MIPS32R5: # %bb.0: ; MIPS32R5-NEXT: lui $1, %hi(gv4i8) ; MIPS32R5-NEXT: lw $2, %lo(gv4i8)($1) ; MIPS32R5-NEXT: jr $ra ; MIPS32R5-NEXT: nop ; ; MIPS64R5-LABEL: ret_4_i8: ; MIPS64R5: # %bb.0: ; MIPS64R5-NEXT: lui $1, %hi(%neg(%gp_rel(ret_4_i8))) ; MIPS64R5-NEXT: daddu $1, $1, $25 ; MIPS64R5-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(ret_4_i8))) ; MIPS64R5-NEXT: ld $1, %got_disp(gv4i8)($1) ; MIPS64R5-NEXT: lw $2, 0($1) ; MIPS64R5-NEXT: jr $ra ; MIPS64R5-NEXT: nop %1 = load <4 x i8>, <4 x i8> * @gv4i8 ret <4 x i8> %1 } define <8 x i8> @ret_8_i8() { ; MIPS32-LABEL: ret_8_i8: ; MIPS32: # %bb.0: ; MIPS32-NEXT: lui $1, %hi(gv8i8) ; MIPS32-NEXT: lw $2, %lo(gv8i8)($1) ; MIPS32-NEXT: addiu $1, $1, %lo(gv8i8) ; MIPS32-NEXT: lw $3, 4($1) ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: nop ; ; MIPS64-LABEL: ret_8_i8: ; MIPS64: # %bb.0: ; MIPS64-NEXT: lui $1, %hi(%neg(%gp_rel(ret_8_i8))) ; MIPS64-NEXT: daddu $1, $1, $25 ; MIPS64-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(ret_8_i8))) ; MIPS64-NEXT: ld $1, %got_disp(gv8i8)($1) ; MIPS64-NEXT: ld $2, 0($1) ; MIPS64-NEXT: jr $ra ; MIPS64-NEXT: nop ; ; MIPS32R5EB-LABEL: ret_8_i8: ; MIPS32R5EB: # %bb.0: ; MIPS32R5EB-NEXT: addiu $sp, $sp, -32 ; MIPS32R5EB-NEXT: .cfi_def_cfa_offset 32 ; MIPS32R5EB-NEXT: sw $fp, 28($sp) # 4-byte Folded Spill ; MIPS32R5EB-NEXT: .cfi_offset 30, -4 ; MIPS32R5EB-NEXT: move $fp, $sp ; MIPS32R5EB-NEXT: .cfi_def_cfa_register 30 ; MIPS32R5EB-NEXT: addiu $1, $zero, -16 ; MIPS32R5EB-NEXT: and $sp, $sp, $1 ; MIPS32R5EB-NEXT: lui $1, %hi(gv8i8) ; MIPS32R5EB-NEXT: lw $2, %lo(gv8i8)($1) ; MIPS32R5EB-NEXT: sw $2, 4($sp) ; MIPS32R5EB-NEXT: addiu $1, $1, %lo(gv8i8) ; MIPS32R5EB-NEXT: lw $1, 4($1) ; MIPS32R5EB-NEXT: sw $1, 12($sp) ; MIPS32R5EB-NEXT: ld.w $w0, 0($sp) ; MIPS32R5EB-NEXT: copy_s.w $2, $w0[1] ; MIPS32R5EB-NEXT: copy_s.w $3, $w0[3] ; MIPS32R5EB-NEXT: move $sp, $fp ; MIPS32R5EB-NEXT: lw $fp, 28($sp) # 4-byte Folded Reload ; MIPS32R5EB-NEXT: addiu $sp, $sp, 32 ; MIPS32R5EB-NEXT: jr $ra ; MIPS32R5EB-NEXT: nop ; ; MIPS64R5-LABEL: ret_8_i8: ; MIPS64R5: # %bb.0: ; MIPS64R5-NEXT: lui $1, %hi(%neg(%gp_rel(ret_8_i8))) ; MIPS64R5-NEXT: daddu $1, $1, $25 ; MIPS64R5-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(ret_8_i8))) ; MIPS64R5-NEXT: ld $1, %got_disp(gv8i8)($1) ; MIPS64R5-NEXT: ld $2, 0($1) ; MIPS64R5-NEXT: jr $ra ; MIPS64R5-NEXT: nop ; ; MIPS32R5EL-LABEL: ret_8_i8: ; MIPS32R5EL: # %bb.0: ; MIPS32R5EL-NEXT: addiu $sp, $sp, -32 ; MIPS32R5EL-NEXT: .cfi_def_cfa_offset 32 ; MIPS32R5EL-NEXT: sw $fp, 28($sp) # 4-byte Folded Spill ; MIPS32R5EL-NEXT: .cfi_offset 30, -4 ; MIPS32R5EL-NEXT: move $fp, $sp ; MIPS32R5EL-NEXT: .cfi_def_cfa_register 30 ; MIPS32R5EL-NEXT: addiu $1, $zero, -16 ; MIPS32R5EL-NEXT: and $sp, $sp, $1 ; MIPS32R5EL-NEXT: lui $1, %hi(gv8i8) ; MIPS32R5EL-NEXT: lw $2, %lo(gv8i8)($1) ; MIPS32R5EL-NEXT: sw $2, 0($sp) ; MIPS32R5EL-NEXT: addiu $1, $1, %lo(gv8i8) ; MIPS32R5EL-NEXT: lw $1, 4($1) ; MIPS32R5EL-NEXT: sw $1, 8($sp) ; MIPS32R5EL-NEXT: ld.w $w0, 0($sp) ; MIPS32R5EL-NEXT: copy_s.w $2, $w0[0] ; MIPS32R5EL-NEXT: copy_s.w $3, $w0[2] ; MIPS32R5EL-NEXT: move $sp, $fp ; MIPS32R5EL-NEXT: lw $fp, 28($sp) # 4-byte Folded Reload ; MIPS32R5EL-NEXT: addiu $sp, $sp, 32 ; MIPS32R5EL-NEXT: jr $ra ; MIPS32R5EL-NEXT: nop %1 = load <8 x i8>, <8 x i8> * @gv8i8 ret <8 x i8> %1 } define <16 x i8> @ret_16_i8() { ; MIPS32-LABEL: ret_16_i8: ; MIPS32: # %bb.0: ; MIPS32-NEXT: lui $1, %hi(gv16i8) ; MIPS32-NEXT: lw $2, %lo(gv16i8)($1) ; MIPS32-NEXT: addiu $1, $1, %lo(gv16i8) ; MIPS32-NEXT: lw $3, 4($1) ; MIPS32-NEXT: lw $4, 8($1) ; MIPS32-NEXT: lw $5, 12($1) ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: nop ; ; MIPS64-LABEL: ret_16_i8: ; MIPS64: # %bb.0: ; MIPS64-NEXT: lui $1, %hi(%neg(%gp_rel(ret_16_i8))) ; MIPS64-NEXT: daddu $1, $1, $25 ; MIPS64-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(ret_16_i8))) ; MIPS64-NEXT: ld $1, %got_disp(gv16i8)($1) ; MIPS64-NEXT: ld $2, 0($1) ; MIPS64-NEXT: ld $3, 8($1) ; MIPS64-NEXT: jr $ra ; MIPS64-NEXT: nop ; ; MIPS32R5-LABEL: ret_16_i8: ; MIPS32R5: # %bb.0: ; MIPS32R5-NEXT: lui $1, %hi(gv16i8) ; MIPS32R5-NEXT: addiu $1, $1, %lo(gv16i8) ; MIPS32R5-NEXT: ld.w $w0, 0($1) ; MIPS32R5-NEXT: copy_s.w $2, $w0[0] ; MIPS32R5-NEXT: copy_s.w $3, $w0[1] ; MIPS32R5-NEXT: copy_s.w $4, $w0[2] ; MIPS32R5-NEXT: copy_s.w $5, $w0[3] ; MIPS32R5-NEXT: jr $ra ; MIPS32R5-NEXT: nop ; ; MIPS64R5-LABEL: ret_16_i8: ; MIPS64R5: # %bb.0: ; MIPS64R5-NEXT: lui $1, %hi(%neg(%gp_rel(ret_16_i8))) ; MIPS64R5-NEXT: daddu $1, $1, $25 ; MIPS64R5-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(ret_16_i8))) ; MIPS64R5-NEXT: ld $1, %got_disp(gv16i8)($1) ; MIPS64R5-NEXT: ld.d $w0, 0($1) ; MIPS64R5-NEXT: copy_s.d $2, $w0[0] ; MIPS64R5-NEXT: copy_s.d $3, $w0[1] ; MIPS64R5-NEXT: jr $ra ; MIPS64R5-NEXT: nop %1 = load <16 x i8>, <16 x i8> * @gv16i8 ret <16 x i8> %1 } define <2 x i16> @ret_2_i16() { ; MIPS32-LABEL: ret_2_i16: ; MIPS32: # %bb.0: ; MIPS32-NEXT: lui $1, %hi(gv2i16) ; MIPS32-NEXT: lw $2, %lo(gv2i16)($1) ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: nop ; ; MIPS64-LABEL: ret_2_i16: ; MIPS64: # %bb.0: ; MIPS64-NEXT: lui $1, %hi(%neg(%gp_rel(ret_2_i16))) ; MIPS64-NEXT: daddu $1, $1, $25 ; MIPS64-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(ret_2_i16))) ; MIPS64-NEXT: ld $1, %got_disp(gv2i16)($1) ; MIPS64-NEXT: lw $2, 0($1) ; MIPS64-NEXT: jr $ra ; MIPS64-NEXT: nop ; ; MIPS32R5-LABEL: ret_2_i16: ; MIPS32R5: # %bb.0: ; MIPS32R5-NEXT: lui $1, %hi(gv2i16) ; MIPS32R5-NEXT: lw $2, %lo(gv2i16)($1) ; MIPS32R5-NEXT: jr $ra ; MIPS32R5-NEXT: nop ; ; MIPS64R5-LABEL: ret_2_i16: ; MIPS64R5: # %bb.0: ; MIPS64R5-NEXT: lui $1, %hi(%neg(%gp_rel(ret_2_i16))) ; MIPS64R5-NEXT: daddu $1, $1, $25 ; MIPS64R5-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(ret_2_i16))) ; MIPS64R5-NEXT: ld $1, %got_disp(gv2i16)($1) ; MIPS64R5-NEXT: lw $2, 0($1) ; MIPS64R5-NEXT: jr $ra ; MIPS64R5-NEXT: nop %1 = load <2 x i16>, <2 x i16> * @gv2i16 ret <2 x i16> %1 } define <4 x i16> @ret_4_i16() { ; MIPS32-LABEL: ret_4_i16: ; MIPS32: # %bb.0: ; MIPS32-NEXT: lui $1, %hi(gv4i16) ; MIPS32-NEXT: lw $2, %lo(gv4i16)($1) ; MIPS32-NEXT: addiu $1, $1, %lo(gv4i16) ; MIPS32-NEXT: lw $3, 4($1) ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: nop ; ; MIPS64-LABEL: ret_4_i16: ; MIPS64: # %bb.0: ; MIPS64-NEXT: lui $1, %hi(%neg(%gp_rel(ret_4_i16))) ; MIPS64-NEXT: daddu $1, $1, $25 ; MIPS64-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(ret_4_i16))) ; MIPS64-NEXT: ld $1, %got_disp(gv4i16)($1) ; MIPS64-NEXT: ld $2, 0($1) ; MIPS64-NEXT: jr $ra ; MIPS64-NEXT: nop ; ; MIPS32R5EB-LABEL: ret_4_i16: ; MIPS32R5EB: # %bb.0: ; MIPS32R5EB-NEXT: addiu $sp, $sp, -32 ; MIPS32R5EB-NEXT: .cfi_def_cfa_offset 32 ; MIPS32R5EB-NEXT: sw $fp, 28($sp) # 4-byte Folded Spill ; MIPS32R5EB-NEXT: .cfi_offset 30, -4 ; MIPS32R5EB-NEXT: move $fp, $sp ; MIPS32R5EB-NEXT: .cfi_def_cfa_register 30 ; MIPS32R5EB-NEXT: addiu $1, $zero, -16 ; MIPS32R5EB-NEXT: and $sp, $sp, $1 ; MIPS32R5EB-NEXT: lui $1, %hi(gv4i16) ; MIPS32R5EB-NEXT: lw $2, %lo(gv4i16)($1) ; MIPS32R5EB-NEXT: sw $2, 4($sp) ; MIPS32R5EB-NEXT: addiu $1, $1, %lo(gv4i16) ; MIPS32R5EB-NEXT: lw $1, 4($1) ; MIPS32R5EB-NEXT: sw $1, 12($sp) ; MIPS32R5EB-NEXT: ld.w $w0, 0($sp) ; MIPS32R5EB-NEXT: copy_s.w $2, $w0[1] ; MIPS32R5EB-NEXT: copy_s.w $3, $w0[3] ; MIPS32R5EB-NEXT: move $sp, $fp ; MIPS32R5EB-NEXT: lw $fp, 28($sp) # 4-byte Folded Reload ; MIPS32R5EB-NEXT: addiu $sp, $sp, 32 ; MIPS32R5EB-NEXT: jr $ra ; MIPS32R5EB-NEXT: nop ; ; MIPS64R5-LABEL: ret_4_i16: ; MIPS64R5: # %bb.0: ; MIPS64R5-NEXT: lui $1, %hi(%neg(%gp_rel(ret_4_i16))) ; MIPS64R5-NEXT: daddu $1, $1, $25 ; MIPS64R5-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(ret_4_i16))) ; MIPS64R5-NEXT: ld $1, %got_disp(gv4i16)($1) ; MIPS64R5-NEXT: ld $2, 0($1) ; MIPS64R5-NEXT: jr $ra ; MIPS64R5-NEXT: nop ; ; MIPS32R5EL-LABEL: ret_4_i16: ; MIPS32R5EL: # %bb.0: ; MIPS32R5EL-NEXT: addiu $sp, $sp, -32 ; MIPS32R5EL-NEXT: .cfi_def_cfa_offset 32 ; MIPS32R5EL-NEXT: sw $fp, 28($sp) # 4-byte Folded Spill ; MIPS32R5EL-NEXT: .cfi_offset 30, -4 ; MIPS32R5EL-NEXT: move $fp, $sp ; MIPS32R5EL-NEXT: .cfi_def_cfa_register 30 ; MIPS32R5EL-NEXT: addiu $1, $zero, -16 ; MIPS32R5EL-NEXT: and $sp, $sp, $1 ; MIPS32R5EL-NEXT: lui $1, %hi(gv4i16) ; MIPS32R5EL-NEXT: lw $2, %lo(gv4i16)($1) ; MIPS32R5EL-NEXT: sw $2, 0($sp) ; MIPS32R5EL-NEXT: addiu $1, $1, %lo(gv4i16) ; MIPS32R5EL-NEXT: lw $1, 4($1) ; MIPS32R5EL-NEXT: sw $1, 8($sp) ; MIPS32R5EL-NEXT: ld.w $w0, 0($sp) ; MIPS32R5EL-NEXT: copy_s.w $2, $w0[0] ; MIPS32R5EL-NEXT: copy_s.w $3, $w0[2] ; MIPS32R5EL-NEXT: move $sp, $fp ; MIPS32R5EL-NEXT: lw $fp, 28($sp) # 4-byte Folded Reload ; MIPS32R5EL-NEXT: addiu $sp, $sp, 32 ; MIPS32R5EL-NEXT: jr $ra ; MIPS32R5EL-NEXT: nop %1 = load <4 x i16>, <4 x i16> * @gv4i16 ret <4 x i16> %1 } define <8 x i16> @ret_8_i16() { ; MIPS32-LABEL: ret_8_i16: ; MIPS32: # %bb.0: ; MIPS32-NEXT: lui $1, %hi(gv8i16) ; MIPS32-NEXT: lw $2, %lo(gv8i16)($1) ; MIPS32-NEXT: addiu $1, $1, %lo(gv8i16) ; MIPS32-NEXT: lw $3, 4($1) ; MIPS32-NEXT: lw $4, 8($1) ; MIPS32-NEXT: lw $5, 12($1) ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: nop ; ; MIPS64-LABEL: ret_8_i16: ; MIPS64: # %bb.0: ; MIPS64-NEXT: lui $1, %hi(%neg(%gp_rel(ret_8_i16))) ; MIPS64-NEXT: daddu $1, $1, $25 ; MIPS64-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(ret_8_i16))) ; MIPS64-NEXT: ld $1, %got_disp(gv8i16)($1) ; MIPS64-NEXT: ld $2, 0($1) ; MIPS64-NEXT: ld $3, 8($1) ; MIPS64-NEXT: jr $ra ; MIPS64-NEXT: nop ; ; MIPS32R5-LABEL: ret_8_i16: ; MIPS32R5: # %bb.0: ; MIPS32R5-NEXT: lui $1, %hi(gv8i16) ; MIPS32R5-NEXT: addiu $1, $1, %lo(gv8i16) ; MIPS32R5-NEXT: ld.w $w0, 0($1) ; MIPS32R5-NEXT: copy_s.w $2, $w0[0] ; MIPS32R5-NEXT: copy_s.w $3, $w0[1] ; MIPS32R5-NEXT: copy_s.w $4, $w0[2] ; MIPS32R5-NEXT: copy_s.w $5, $w0[3] ; MIPS32R5-NEXT: jr $ra ; MIPS32R5-NEXT: nop ; ; MIPS64R5-LABEL: ret_8_i16: ; MIPS64R5: # %bb.0: ; MIPS64R5-NEXT: lui $1, %hi(%neg(%gp_rel(ret_8_i16))) ; MIPS64R5-NEXT: daddu $1, $1, $25 ; MIPS64R5-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(ret_8_i16))) ; MIPS64R5-NEXT: ld $1, %got_disp(gv8i16)($1) ; MIPS64R5-NEXT: ld.d $w0, 0($1) ; MIPS64R5-NEXT: copy_s.d $2, $w0[0] ; MIPS64R5-NEXT: copy_s.d $3, $w0[1] ; MIPS64R5-NEXT: jr $ra ; MIPS64R5-NEXT: nop %1 = load <8 x i16>, <8 x i16> * @gv8i16 ret <8 x i16> %1 } define <2 x i32> @ret_2_i32() { ; MIPS32-LABEL: ret_2_i32: ; MIPS32: # %bb.0: ; MIPS32-NEXT: lui $1, %hi(gv2i32) ; MIPS32-NEXT: lw $2, %lo(gv2i32)($1) ; MIPS32-NEXT: addiu $1, $1, %lo(gv2i32) ; MIPS32-NEXT: lw $3, 4($1) ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: nop ; ; MIPS64-LABEL: ret_2_i32: ; MIPS64: # %bb.0: ; MIPS64-NEXT: lui $1, %hi(%neg(%gp_rel(ret_2_i32))) ; MIPS64-NEXT: daddu $1, $1, $25 ; MIPS64-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(ret_2_i32))) ; MIPS64-NEXT: ld $1, %got_disp(gv2i32)($1) ; MIPS64-NEXT: ld $2, 0($1) ; MIPS64-NEXT: jr $ra ; MIPS64-NEXT: nop ; ; MIPS32R5EB-LABEL: ret_2_i32: ; MIPS32R5EB: # %bb.0: ; MIPS32R5EB-NEXT: addiu $sp, $sp, -32 ; MIPS32R5EB-NEXT: .cfi_def_cfa_offset 32 ; MIPS32R5EB-NEXT: sw $fp, 28($sp) # 4-byte Folded Spill ; MIPS32R5EB-NEXT: .cfi_offset 30, -4 ; MIPS32R5EB-NEXT: move $fp, $sp ; MIPS32R5EB-NEXT: .cfi_def_cfa_register 30 ; MIPS32R5EB-NEXT: addiu $1, $zero, -16 ; MIPS32R5EB-NEXT: and $sp, $sp, $1 ; MIPS32R5EB-NEXT: lui $1, %hi(gv2i32) ; MIPS32R5EB-NEXT: lw $2, %lo(gv2i32)($1) ; MIPS32R5EB-NEXT: sw $2, 4($sp) ; MIPS32R5EB-NEXT: addiu $1, $1, %lo(gv2i32) ; MIPS32R5EB-NEXT: lw $1, 4($1) ; MIPS32R5EB-NEXT: sw $1, 12($sp) ; MIPS32R5EB-NEXT: ld.w $w0, 0($sp) ; MIPS32R5EB-NEXT: copy_s.w $2, $w0[1] ; MIPS32R5EB-NEXT: copy_s.w $3, $w0[3] ; MIPS32R5EB-NEXT: move $sp, $fp ; MIPS32R5EB-NEXT: lw $fp, 28($sp) # 4-byte Folded Reload ; MIPS32R5EB-NEXT: addiu $sp, $sp, 32 ; MIPS32R5EB-NEXT: jr $ra ; MIPS32R5EB-NEXT: nop ; ; MIPS64R5-LABEL: ret_2_i32: ; MIPS64R5: # %bb.0: ; MIPS64R5-NEXT: lui $1, %hi(%neg(%gp_rel(ret_2_i32))) ; MIPS64R5-NEXT: daddu $1, $1, $25 ; MIPS64R5-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(ret_2_i32))) ; MIPS64R5-NEXT: ld $1, %got_disp(gv2i32)($1) ; MIPS64R5-NEXT: ld $2, 0($1) ; MIPS64R5-NEXT: jr $ra ; MIPS64R5-NEXT: nop ; ; MIPS32R5EL-LABEL: ret_2_i32: ; MIPS32R5EL: # %bb.0: ; MIPS32R5EL-NEXT: addiu $sp, $sp, -32 ; MIPS32R5EL-NEXT: .cfi_def_cfa_offset 32 ; MIPS32R5EL-NEXT: sw $fp, 28($sp) # 4-byte Folded Spill ; MIPS32R5EL-NEXT: .cfi_offset 30, -4 ; MIPS32R5EL-NEXT: move $fp, $sp ; MIPS32R5EL-NEXT: .cfi_def_cfa_register 30 ; MIPS32R5EL-NEXT: addiu $1, $zero, -16 ; MIPS32R5EL-NEXT: and $sp, $sp, $1 ; MIPS32R5EL-NEXT: lui $1, %hi(gv2i32) ; MIPS32R5EL-NEXT: lw $2, %lo(gv2i32)($1) ; MIPS32R5EL-NEXT: sw $2, 0($sp) ; MIPS32R5EL-NEXT: addiu $1, $1, %lo(gv2i32) ; MIPS32R5EL-NEXT: lw $1, 4($1) ; MIPS32R5EL-NEXT: sw $1, 8($sp) ; MIPS32R5EL-NEXT: ld.w $w0, 0($sp) ; MIPS32R5EL-NEXT: copy_s.w $2, $w0[0] ; MIPS32R5EL-NEXT: copy_s.w $3, $w0[2] ; MIPS32R5EL-NEXT: move $sp, $fp ; MIPS32R5EL-NEXT: lw $fp, 28($sp) # 4-byte Folded Reload ; MIPS32R5EL-NEXT: addiu $sp, $sp, 32 ; MIPS32R5EL-NEXT: jr $ra ; MIPS32R5EL-NEXT: nop %1 = load <2 x i32>, <2 x i32> * @gv2i32 ret <2 x i32> %1 } define <4 x i32> @ret_4_i32() { ; MIPS32-LABEL: ret_4_i32: ; MIPS32: # %bb.0: ; MIPS32-NEXT: lui $1, %hi(gv4i32) ; MIPS32-NEXT: lw $2, %lo(gv4i32)($1) ; MIPS32-NEXT: addiu $1, $1, %lo(gv4i32) ; MIPS32-NEXT: lw $3, 4($1) ; MIPS32-NEXT: lw $4, 8($1) ; MIPS32-NEXT: lw $5, 12($1) ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: nop ; ; MIPS64-LABEL: ret_4_i32: ; MIPS64: # %bb.0: ; MIPS64-NEXT: lui $1, %hi(%neg(%gp_rel(ret_4_i32))) ; MIPS64-NEXT: daddu $1, $1, $25 ; MIPS64-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(ret_4_i32))) ; MIPS64-NEXT: ld $1, %got_disp(gv4i32)($1) ; MIPS64-NEXT: ld $2, 0($1) ; MIPS64-NEXT: ld $3, 8($1) ; MIPS64-NEXT: jr $ra ; MIPS64-NEXT: nop ; ; MIPS32R5-LABEL: ret_4_i32: ; MIPS32R5: # %bb.0: ; MIPS32R5-NEXT: lui $1, %hi(gv4i32) ; MIPS32R5-NEXT: addiu $1, $1, %lo(gv4i32) ; MIPS32R5-NEXT: ld.w $w0, 0($1) ; MIPS32R5-NEXT: copy_s.w $2, $w0[0] ; MIPS32R5-NEXT: copy_s.w $3, $w0[1] ; MIPS32R5-NEXT: copy_s.w $4, $w0[2] ; MIPS32R5-NEXT: copy_s.w $5, $w0[3] ; MIPS32R5-NEXT: jr $ra ; MIPS32R5-NEXT: nop ; ; MIPS64R5-LABEL: ret_4_i32: ; MIPS64R5: # %bb.0: ; MIPS64R5-NEXT: lui $1, %hi(%neg(%gp_rel(ret_4_i32))) ; MIPS64R5-NEXT: daddu $1, $1, $25 ; MIPS64R5-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(ret_4_i32))) ; MIPS64R5-NEXT: ld $1, %got_disp(gv4i32)($1) ; MIPS64R5-NEXT: ld.d $w0, 0($1) ; MIPS64R5-NEXT: copy_s.d $2, $w0[0] ; MIPS64R5-NEXT: copy_s.d $3, $w0[1] ; MIPS64R5-NEXT: jr $ra ; MIPS64R5-NEXT: nop %1 = load <4 x i32>, <4 x i32> * @gv4i32 ret <4 x i32> %1 } define <2 x i64> @ret_2_i64() { ; MIPS32-LABEL: ret_2_i64: ; MIPS32: # %bb.0: ; MIPS32-NEXT: lui $1, %hi(gv2i64) ; MIPS32-NEXT: lw $2, %lo(gv2i64)($1) ; MIPS32-NEXT: addiu $1, $1, %lo(gv2i64) ; MIPS32-NEXT: lw $3, 4($1) ; MIPS32-NEXT: lw $4, 8($1) ; MIPS32-NEXT: lw $5, 12($1) ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: nop ; ; MIPS64-LABEL: ret_2_i64: ; MIPS64: # %bb.0: ; MIPS64-NEXT: lui $1, %hi(%neg(%gp_rel(ret_2_i64))) ; MIPS64-NEXT: daddu $1, $1, $25 ; MIPS64-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(ret_2_i64))) ; MIPS64-NEXT: ld $1, %got_disp(gv2i64)($1) ; MIPS64-NEXT: ld $2, 0($1) ; MIPS64-NEXT: ld $3, 8($1) ; MIPS64-NEXT: jr $ra ; MIPS64-NEXT: nop ; ; MIPS32R5-LABEL: ret_2_i64: ; MIPS32R5: # %bb.0: ; MIPS32R5-NEXT: lui $1, %hi(gv2i64) ; MIPS32R5-NEXT: addiu $1, $1, %lo(gv2i64) ; MIPS32R5-NEXT: ld.w $w0, 0($1) ; MIPS32R5-NEXT: copy_s.w $2, $w0[0] ; MIPS32R5-NEXT: copy_s.w $3, $w0[1] ; MIPS32R5-NEXT: copy_s.w $4, $w0[2] ; MIPS32R5-NEXT: copy_s.w $5, $w0[3] ; MIPS32R5-NEXT: jr $ra ; MIPS32R5-NEXT: nop ; ; MIPS64R5-LABEL: ret_2_i64: ; MIPS64R5: # %bb.0: ; MIPS64R5-NEXT: lui $1, %hi(%neg(%gp_rel(ret_2_i64))) ; MIPS64R5-NEXT: daddu $1, $1, $25 ; MIPS64R5-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(ret_2_i64))) ; MIPS64R5-NEXT: ld $1, %got_disp(gv2i64)($1) ; MIPS64R5-NEXT: ld.d $w0, 0($1) ; MIPS64R5-NEXT: copy_s.d $2, $w0[0] ; MIPS64R5-NEXT: copy_s.d $3, $w0[1] ; MIPS64R5-NEXT: jr $ra ; MIPS64R5-NEXT: nop %1 = load <2 x i64>, <2 x i64> * @gv2i64 ret <2 x i64> %1 } @gv2f32 = global <2 x float> @gv4f32 = global <4 x float> define <2 x float> @ret_float_2() { ; MIPS32-LABEL: ret_float_2: ; MIPS32: # %bb.0: # %entry ; MIPS32-NEXT: lui $1, %hi(gv2f32) ; MIPS32-NEXT: addiu $2, $1, %lo(gv2f32) ; MIPS32-NEXT: lwc1 $f0, 4($2) ; MIPS32-NEXT: swc1 $f0, 4($4) ; MIPS32-NEXT: lwc1 $f0, %lo(gv2f32)($1) ; MIPS32-NEXT: swc1 $f0, 0($4) ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: nop ; ; MIPS64-LABEL: ret_float_2: ; MIPS64: # %bb.0: # %entry ; MIPS64-NEXT: lui $1, %hi(%neg(%gp_rel(ret_float_2))) ; MIPS64-NEXT: daddu $1, $1, $25 ; MIPS64-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(ret_float_2))) ; MIPS64-NEXT: ld $1, %got_disp(gv2f32)($1) ; MIPS64-NEXT: ld $2, 0($1) ; MIPS64-NEXT: jr $ra ; MIPS64-NEXT: nop ; ; MIPS32R5-LABEL: ret_float_2: ; MIPS32R5: # %bb.0: # %entry ; MIPS32R5-NEXT: lui $1, %hi(gv2f32) ; MIPS32R5-NEXT: addiu $2, $1, %lo(gv2f32) ; MIPS32R5-NEXT: lwc1 $f0, 4($2) ; MIPS32R5-NEXT: swc1 $f0, 4($4) ; MIPS32R5-NEXT: lwc1 $f0, %lo(gv2f32)($1) ; MIPS32R5-NEXT: swc1 $f0, 0($4) ; MIPS32R5-NEXT: jr $ra ; MIPS32R5-NEXT: nop ; ; MIPS64R5-LABEL: ret_float_2: ; MIPS64R5: # %bb.0: # %entry ; MIPS64R5-NEXT: lui $1, %hi(%neg(%gp_rel(ret_float_2))) ; MIPS64R5-NEXT: daddu $1, $1, $25 ; MIPS64R5-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(ret_float_2))) ; MIPS64R5-NEXT: ld $1, %got_disp(gv2f32)($1) ; MIPS64R5-NEXT: ld $2, 0($1) ; MIPS64R5-NEXT: jr $ra ; MIPS64R5-NEXT: nop entry: %0 = load <2 x float>, <2 x float> * @gv2f32 ret <2 x float> %0 } define <4 x float> @ret_float_4() { ; MIPS32-LABEL: ret_float_4: ; MIPS32: # %bb.0: # %entry ; MIPS32-NEXT: lui $1, %hi(gv4f32) ; MIPS32-NEXT: addiu $2, $1, %lo(gv4f32) ; MIPS32-NEXT: lwc1 $f0, 12($2) ; MIPS32-NEXT: swc1 $f0, 12($4) ; MIPS32-NEXT: lwc1 $f0, 8($2) ; MIPS32-NEXT: swc1 $f0, 8($4) ; MIPS32-NEXT: lwc1 $f0, 4($2) ; MIPS32-NEXT: swc1 $f0, 4($4) ; MIPS32-NEXT: lwc1 $f0, %lo(gv4f32)($1) ; MIPS32-NEXT: swc1 $f0, 0($4) ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: nop ; ; MIPS64-LABEL: ret_float_4: ; MIPS64: # %bb.0: # %entry ; MIPS64-NEXT: lui $1, %hi(%neg(%gp_rel(ret_float_4))) ; MIPS64-NEXT: daddu $1, $1, $25 ; MIPS64-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(ret_float_4))) ; MIPS64-NEXT: ld $1, %got_disp(gv4f32)($1) ; MIPS64-NEXT: ld $2, 0($1) ; MIPS64-NEXT: ld $3, 8($1) ; MIPS64-NEXT: jr $ra ; MIPS64-NEXT: nop ; ; MIPS32R5-LABEL: ret_float_4: ; MIPS32R5: # %bb.0: # %entry ; MIPS32R5-NEXT: lui $1, %hi(gv4f32) ; MIPS32R5-NEXT: addiu $1, $1, %lo(gv4f32) ; MIPS32R5-NEXT: ld.w $w0, 0($1) ; MIPS32R5-NEXT: st.w $w0, 0($4) ; MIPS32R5-NEXT: jr $ra ; MIPS32R5-NEXT: nop ; ; MIPS64R5-LABEL: ret_float_4: ; MIPS64R5: # %bb.0: # %entry ; MIPS64R5-NEXT: lui $1, %hi(%neg(%gp_rel(ret_float_4))) ; MIPS64R5-NEXT: daddu $1, $1, $25 ; MIPS64R5-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(ret_float_4))) ; MIPS64R5-NEXT: ld $1, %got_disp(gv4f32)($1) ; MIPS64R5-NEXT: ld.d $w0, 0($1) ; MIPS64R5-NEXT: copy_s.d $2, $w0[0] ; MIPS64R5-NEXT: copy_s.d $3, $w0[1] ; MIPS64R5-NEXT: jr $ra ; MIPS64R5-NEXT: nop entry: %0 = load <4 x float>, <4 x float> * @gv4f32 ret <4 x float> %0 } @gv2f64 = global <2 x double> define <2 x double> @ret_double_2() { ; MIPS32-LABEL: ret_double_2: ; MIPS32: # %bb.0: # %entry ; MIPS32-NEXT: lui $1, %hi(gv2f64) ; MIPS32-NEXT: addiu $2, $1, %lo(gv2f64) ; MIPS32-NEXT: ldc1 $f0, 8($2) ; MIPS32-NEXT: sdc1 $f0, 8($4) ; MIPS32-NEXT: ldc1 $f0, %lo(gv2f64)($1) ; MIPS32-NEXT: sdc1 $f0, 0($4) ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: nop ; ; MIPS64-LABEL: ret_double_2: ; MIPS64: # %bb.0: # %entry ; MIPS64-NEXT: lui $1, %hi(%neg(%gp_rel(ret_double_2))) ; MIPS64-NEXT: daddu $1, $1, $25 ; MIPS64-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(ret_double_2))) ; MIPS64-NEXT: ld $1, %got_disp(gv2f64)($1) ; MIPS64-NEXT: ld $2, 0($1) ; MIPS64-NEXT: ld $3, 8($1) ; MIPS64-NEXT: jr $ra ; MIPS64-NEXT: nop ; ; MIPS32R5-LABEL: ret_double_2: ; MIPS32R5: # %bb.0: # %entry ; MIPS32R5-NEXT: lui $1, %hi(gv2f64) ; MIPS32R5-NEXT: addiu $1, $1, %lo(gv2f64) ; MIPS32R5-NEXT: ld.d $w0, 0($1) ; MIPS32R5-NEXT: st.d $w0, 0($4) ; MIPS32R5-NEXT: jr $ra ; MIPS32R5-NEXT: nop ; ; MIPS64R5-LABEL: ret_double_2: ; MIPS64R5: # %bb.0: # %entry ; MIPS64R5-NEXT: lui $1, %hi(%neg(%gp_rel(ret_double_2))) ; MIPS64R5-NEXT: daddu $1, $1, $25 ; MIPS64R5-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(ret_double_2))) ; MIPS64R5-NEXT: ld $1, %got_disp(gv2f64)($1) ; MIPS64R5-NEXT: ld.d $w0, 0($1) ; MIPS64R5-NEXT: copy_s.d $2, $w0[0] ; MIPS64R5-NEXT: copy_s.d $3, $w0[1] ; MIPS64R5-NEXT: jr $ra ; MIPS64R5-NEXT: nop entry: %0 = load <2 x double>, <2 x double> * @gv2f64 ret <2 x double> %0 } ; Test argument lowering and call result lowering. define void @call_i8_2() { ; MIPS32EB-LABEL: call_i8_2: ; MIPS32EB: # %bb.0: # %entry ; MIPS32EB-NEXT: addiu $sp, $sp, -24 ; MIPS32EB-NEXT: .cfi_def_cfa_offset 24 ; MIPS32EB-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; MIPS32EB-NEXT: .cfi_offset 31, -4 ; MIPS32EB-NEXT: addiu $4, $zero, 1543 ; MIPS32EB-NEXT: addiu $5, $zero, 3080 ; MIPS32EB-NEXT: jal i8_2 ; MIPS32EB-NEXT: nop ; MIPS32EB-NEXT: srl $1, $2, 16 ; MIPS32EB-NEXT: lui $3, %hi(gv2i8) ; MIPS32EB-NEXT: addiu $4, $3, %lo(gv2i8) ; MIPS32EB-NEXT: sb $1, 1($4) ; MIPS32EB-NEXT: srl $1, $2, 24 ; MIPS32EB-NEXT: sb $1, %lo(gv2i8)($3) ; MIPS32EB-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; MIPS32EB-NEXT: addiu $sp, $sp, 24 ; MIPS32EB-NEXT: jr $ra ; MIPS32EB-NEXT: nop ; ; MIPS64EB-LABEL: call_i8_2: ; MIPS64EB: # %bb.0: # %entry ; MIPS64EB-NEXT: daddiu $sp, $sp, -16 ; MIPS64EB-NEXT: .cfi_def_cfa_offset 16 ; MIPS64EB-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; MIPS64EB-NEXT: sd $gp, 0($sp) # 8-byte Folded Spill ; MIPS64EB-NEXT: .cfi_offset 31, -8 ; MIPS64EB-NEXT: .cfi_offset 28, -16 ; MIPS64EB-NEXT: lui $1, %hi(%neg(%gp_rel(call_i8_2))) ; MIPS64EB-NEXT: daddu $1, $1, $25 ; MIPS64EB-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(call_i8_2))) ; MIPS64EB-NEXT: ld $25, %call16(i8_2)($gp) ; MIPS64EB-NEXT: daddiu $4, $zero, 1543 ; MIPS64EB-NEXT: daddiu $5, $zero, 3080 ; MIPS64EB-NEXT: jalr $25 ; MIPS64EB-NEXT: nop ; MIPS64EB-NEXT: dsrl $1, $2, 48 ; MIPS64EB-NEXT: ld $3, %got_disp(gv2i8)($gp) ; MIPS64EB-NEXT: sb $1, 1($3) ; MIPS64EB-NEXT: dsrl $1, $2, 56 ; MIPS64EB-NEXT: sb $1, 0($3) ; MIPS64EB-NEXT: ld $gp, 0($sp) # 8-byte Folded Reload ; MIPS64EB-NEXT: ld $ra, 8($sp) # 8-byte Folded Reload ; MIPS64EB-NEXT: daddiu $sp, $sp, 16 ; MIPS64EB-NEXT: jr $ra ; MIPS64EB-NEXT: nop ; ; MIPS32R5EB-LABEL: call_i8_2: ; MIPS32R5EB: # %bb.0: # %entry ; MIPS32R5EB-NEXT: addiu $sp, $sp, -32 ; MIPS32R5EB-NEXT: .cfi_def_cfa_offset 32 ; MIPS32R5EB-NEXT: sw $ra, 28($sp) # 4-byte Folded Spill ; MIPS32R5EB-NEXT: .cfi_offset 31, -4 ; MIPS32R5EB-NEXT: addiu $4, $zero, 1543 ; MIPS32R5EB-NEXT: addiu $5, $zero, 3080 ; MIPS32R5EB-NEXT: jal i8_2 ; MIPS32R5EB-NEXT: nop ; MIPS32R5EB-NEXT: sw $2, 16($sp) ; MIPS32R5EB-NEXT: lui $1, %hi(gv2i8) ; MIPS32R5EB-NEXT: lhu $2, 16($sp) ; MIPS32R5EB-NEXT: sh $2, %lo(gv2i8)($1) ; MIPS32R5EB-NEXT: lw $ra, 28($sp) # 4-byte Folded Reload ; MIPS32R5EB-NEXT: addiu $sp, $sp, 32 ; MIPS32R5EB-NEXT: jr $ra ; MIPS32R5EB-NEXT: nop ; ; MIPS64R5EB-LABEL: call_i8_2: ; MIPS64R5EB: # %bb.0: # %entry ; MIPS64R5EB-NEXT: daddiu $sp, $sp, -64 ; MIPS64R5EB-NEXT: .cfi_def_cfa_offset 64 ; MIPS64R5EB-NEXT: sd $ra, 56($sp) # 8-byte Folded Spill ; MIPS64R5EB-NEXT: sd $gp, 48($sp) # 8-byte Folded Spill ; MIPS64R5EB-NEXT: .cfi_offset 31, -8 ; MIPS64R5EB-NEXT: .cfi_offset 28, -16 ; MIPS64R5EB-NEXT: lui $1, %hi(%neg(%gp_rel(call_i8_2))) ; MIPS64R5EB-NEXT: daddu $1, $1, $25 ; MIPS64R5EB-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(call_i8_2))) ; MIPS64R5EB-NEXT: addiu $1, $zero, 1543 ; MIPS64R5EB-NEXT: sh $1, 40($sp) ; MIPS64R5EB-NEXT: addiu $1, $zero, 3080 ; MIPS64R5EB-NEXT: sh $1, 44($sp) ; MIPS64R5EB-NEXT: ld $25, %call16(i8_2)($gp) ; MIPS64R5EB-NEXT: lh $4, 40($sp) ; MIPS64R5EB-NEXT: lh $5, 44($sp) ; MIPS64R5EB-NEXT: jalr $25 ; MIPS64R5EB-NEXT: nop ; MIPS64R5EB-NEXT: sd $2, 32($sp) ; MIPS64R5EB-NEXT: lbu $1, 33($sp) ; MIPS64R5EB-NEXT: sh $1, 2($sp) ; MIPS64R5EB-NEXT: lbu $1, 32($sp) ; MIPS64R5EB-NEXT: sh $1, 0($sp) ; MIPS64R5EB-NEXT: ld.h $w0, 0($sp) ; MIPS64R5EB-NEXT: copy_s.h $1, $w0[0] ; MIPS64R5EB-NEXT: copy_s.h $2, $w0[1] ; MIPS64R5EB-NEXT: sw $2, 28($sp) ; MIPS64R5EB-NEXT: sw $1, 20($sp) ; MIPS64R5EB-NEXT: ld.d $w0, 16($sp) ; MIPS64R5EB-NEXT: copy_s.d $1, $w0[0] ; MIPS64R5EB-NEXT: copy_s.d $2, $w0[1] ; MIPS64R5EB-NEXT: ld $3, %got_disp(gv2i8)($gp) ; MIPS64R5EB-NEXT: sb $2, 1($3) ; MIPS64R5EB-NEXT: sb $1, 0($3) ; MIPS64R5EB-NEXT: ld $gp, 48($sp) # 8-byte Folded Reload ; MIPS64R5EB-NEXT: ld $ra, 56($sp) # 8-byte Folded Reload ; MIPS64R5EB-NEXT: daddiu $sp, $sp, 64 ; MIPS64R5EB-NEXT: jr $ra ; MIPS64R5EB-NEXT: nop ; ; MIPS32EL-LABEL: call_i8_2: ; MIPS32EL: # %bb.0: # %entry ; MIPS32EL-NEXT: addiu $sp, $sp, -24 ; MIPS32EL-NEXT: .cfi_def_cfa_offset 24 ; MIPS32EL-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; MIPS32EL-NEXT: .cfi_offset 31, -4 ; MIPS32EL-NEXT: addiu $4, $zero, 1798 ; MIPS32EL-NEXT: addiu $5, $zero, 2060 ; MIPS32EL-NEXT: jal i8_2 ; MIPS32EL-NEXT: nop ; MIPS32EL-NEXT: lui $1, %hi(gv2i8) ; MIPS32EL-NEXT: sb $2, %lo(gv2i8)($1) ; MIPS32EL-NEXT: srl $2, $2, 8 ; MIPS32EL-NEXT: addiu $1, $1, %lo(gv2i8) ; MIPS32EL-NEXT: sb $2, 1($1) ; MIPS32EL-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; MIPS32EL-NEXT: addiu $sp, $sp, 24 ; MIPS32EL-NEXT: jr $ra ; MIPS32EL-NEXT: nop ; ; MIPS64EL-LABEL: call_i8_2: ; MIPS64EL: # %bb.0: # %entry ; MIPS64EL-NEXT: daddiu $sp, $sp, -16 ; MIPS64EL-NEXT: .cfi_def_cfa_offset 16 ; MIPS64EL-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; MIPS64EL-NEXT: sd $gp, 0($sp) # 8-byte Folded Spill ; MIPS64EL-NEXT: .cfi_offset 31, -8 ; MIPS64EL-NEXT: .cfi_offset 28, -16 ; MIPS64EL-NEXT: lui $1, %hi(%neg(%gp_rel(call_i8_2))) ; MIPS64EL-NEXT: daddu $1, $1, $25 ; MIPS64EL-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(call_i8_2))) ; MIPS64EL-NEXT: ld $25, %call16(i8_2)($gp) ; MIPS64EL-NEXT: daddiu $4, $zero, 1798 ; MIPS64EL-NEXT: daddiu $5, $zero, 2060 ; MIPS64EL-NEXT: jalr $25 ; MIPS64EL-NEXT: nop ; MIPS64EL-NEXT: sll $1, $2, 0 ; MIPS64EL-NEXT: ld $2, %got_disp(gv2i8)($gp) ; MIPS64EL-NEXT: sb $1, 0($2) ; MIPS64EL-NEXT: srl $1, $1, 8 ; MIPS64EL-NEXT: sb $1, 1($2) ; MIPS64EL-NEXT: ld $gp, 0($sp) # 8-byte Folded Reload ; MIPS64EL-NEXT: ld $ra, 8($sp) # 8-byte Folded Reload ; MIPS64EL-NEXT: daddiu $sp, $sp, 16 ; MIPS64EL-NEXT: jr $ra ; MIPS64EL-NEXT: nop ; ; MIPS32R5EL-LABEL: call_i8_2: ; MIPS32R5EL: # %bb.0: # %entry ; MIPS32R5EL-NEXT: addiu $sp, $sp, -32 ; MIPS32R5EL-NEXT: .cfi_def_cfa_offset 32 ; MIPS32R5EL-NEXT: sw $ra, 28($sp) # 4-byte Folded Spill ; MIPS32R5EL-NEXT: .cfi_offset 31, -4 ; MIPS32R5EL-NEXT: addiu $4, $zero, 1798 ; MIPS32R5EL-NEXT: addiu $5, $zero, 2060 ; MIPS32R5EL-NEXT: jal i8_2 ; MIPS32R5EL-NEXT: nop ; MIPS32R5EL-NEXT: sw $2, 16($sp) ; MIPS32R5EL-NEXT: lui $1, %hi(gv2i8) ; MIPS32R5EL-NEXT: lhu $2, 16($sp) ; MIPS32R5EL-NEXT: sh $2, %lo(gv2i8)($1) ; MIPS32R5EL-NEXT: lw $ra, 28($sp) # 4-byte Folded Reload ; MIPS32R5EL-NEXT: addiu $sp, $sp, 32 ; MIPS32R5EL-NEXT: jr $ra ; MIPS32R5EL-NEXT: nop ; ; MIPS64R5EL-LABEL: call_i8_2: ; MIPS64R5EL: # %bb.0: # %entry ; MIPS64R5EL-NEXT: daddiu $sp, $sp, -64 ; MIPS64R5EL-NEXT: .cfi_def_cfa_offset 64 ; MIPS64R5EL-NEXT: sd $ra, 56($sp) # 8-byte Folded Spill ; MIPS64R5EL-NEXT: sd $gp, 48($sp) # 8-byte Folded Spill ; MIPS64R5EL-NEXT: .cfi_offset 31, -8 ; MIPS64R5EL-NEXT: .cfi_offset 28, -16 ; MIPS64R5EL-NEXT: lui $1, %hi(%neg(%gp_rel(call_i8_2))) ; MIPS64R5EL-NEXT: daddu $1, $1, $25 ; MIPS64R5EL-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(call_i8_2))) ; MIPS64R5EL-NEXT: addiu $1, $zero, 1798 ; MIPS64R5EL-NEXT: sh $1, 40($sp) ; MIPS64R5EL-NEXT: addiu $1, $zero, 2060 ; MIPS64R5EL-NEXT: sh $1, 44($sp) ; MIPS64R5EL-NEXT: ld $25, %call16(i8_2)($gp) ; MIPS64R5EL-NEXT: lh $4, 40($sp) ; MIPS64R5EL-NEXT: lh $5, 44($sp) ; MIPS64R5EL-NEXT: jalr $25 ; MIPS64R5EL-NEXT: nop ; MIPS64R5EL-NEXT: sd $2, 32($sp) ; MIPS64R5EL-NEXT: lbu $1, 33($sp) ; MIPS64R5EL-NEXT: sh $1, 2($sp) ; MIPS64R5EL-NEXT: lbu $1, 32($sp) ; MIPS64R5EL-NEXT: sh $1, 0($sp) ; MIPS64R5EL-NEXT: ld.h $w0, 0($sp) ; MIPS64R5EL-NEXT: copy_s.h $1, $w0[0] ; MIPS64R5EL-NEXT: copy_s.h $2, $w0[1] ; MIPS64R5EL-NEXT: sw $2, 24($sp) ; MIPS64R5EL-NEXT: sw $1, 16($sp) ; MIPS64R5EL-NEXT: ld.d $w0, 16($sp) ; MIPS64R5EL-NEXT: copy_s.d $1, $w0[0] ; MIPS64R5EL-NEXT: copy_s.d $2, $w0[1] ; MIPS64R5EL-NEXT: ld $3, %got_disp(gv2i8)($gp) ; MIPS64R5EL-NEXT: sb $2, 1($3) ; MIPS64R5EL-NEXT: sb $1, 0($3) ; MIPS64R5EL-NEXT: ld $gp, 48($sp) # 8-byte Folded Reload ; MIPS64R5EL-NEXT: ld $ra, 56($sp) # 8-byte Folded Reload ; MIPS64R5EL-NEXT: daddiu $sp, $sp, 64 ; MIPS64R5EL-NEXT: jr $ra ; MIPS64R5EL-NEXT: nop entry: %0 = call <2 x i8> @i8_2(<2 x i8> , <2 x i8> ) store <2 x i8> %0, <2 x i8> * @gv2i8 ret void } define void @call_i8_4() { ; MIPS32EB-LABEL: call_i8_4: ; MIPS32EB: # %bb.0: # %entry ; MIPS32EB-NEXT: addiu $sp, $sp, -24 ; MIPS32EB-NEXT: .cfi_def_cfa_offset 24 ; MIPS32EB-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; MIPS32EB-NEXT: .cfi_offset 31, -4 ; MIPS32EB-NEXT: lui $1, 1543 ; MIPS32EB-NEXT: ori $4, $1, 2314 ; MIPS32EB-NEXT: lui $1, 3080 ; MIPS32EB-NEXT: ori $5, $1, 2314 ; MIPS32EB-NEXT: jal i8_4 ; MIPS32EB-NEXT: nop ; MIPS32EB-NEXT: lui $1, %hi(gv4i8) ; MIPS32EB-NEXT: sw $2, %lo(gv4i8)($1) ; MIPS32EB-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; MIPS32EB-NEXT: addiu $sp, $sp, 24 ; MIPS32EB-NEXT: jr $ra ; MIPS32EB-NEXT: nop ; ; MIPS64EB-LABEL: call_i8_4: ; MIPS64EB: # %bb.0: # %entry ; MIPS64EB-NEXT: daddiu $sp, $sp, -16 ; MIPS64EB-NEXT: .cfi_def_cfa_offset 16 ; MIPS64EB-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; MIPS64EB-NEXT: sd $gp, 0($sp) # 8-byte Folded Spill ; MIPS64EB-NEXT: .cfi_offset 31, -8 ; MIPS64EB-NEXT: .cfi_offset 28, -16 ; MIPS64EB-NEXT: lui $1, %hi(%neg(%gp_rel(call_i8_4))) ; MIPS64EB-NEXT: daddu $1, $1, $25 ; MIPS64EB-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(call_i8_4))) ; MIPS64EB-NEXT: lui $1, 1543 ; MIPS64EB-NEXT: ori $4, $1, 2314 ; MIPS64EB-NEXT: lui $1, 3080 ; MIPS64EB-NEXT: ori $5, $1, 2314 ; MIPS64EB-NEXT: ld $25, %call16(i8_4)($gp) ; MIPS64EB-NEXT: jalr $25 ; MIPS64EB-NEXT: nop ; MIPS64EB-NEXT: ld $1, %got_disp(gv4i8)($gp) ; MIPS64EB-NEXT: sw $2, 0($1) ; MIPS64EB-NEXT: ld $gp, 0($sp) # 8-byte Folded Reload ; MIPS64EB-NEXT: ld $ra, 8($sp) # 8-byte Folded Reload ; MIPS64EB-NEXT: daddiu $sp, $sp, 16 ; MIPS64EB-NEXT: jr $ra ; MIPS64EB-NEXT: nop ; ; MIPS32R5EB-LABEL: call_i8_4: ; MIPS32R5EB: # %bb.0: # %entry ; MIPS32R5EB-NEXT: addiu $sp, $sp, -32 ; MIPS32R5EB-NEXT: .cfi_def_cfa_offset 32 ; MIPS32R5EB-NEXT: sw $ra, 28($sp) # 4-byte Folded Spill ; MIPS32R5EB-NEXT: .cfi_offset 31, -4 ; MIPS32R5EB-NEXT: lui $1, 1543 ; MIPS32R5EB-NEXT: ori $4, $1, 2314 ; MIPS32R5EB-NEXT: lui $1, 3080 ; MIPS32R5EB-NEXT: ori $5, $1, 2314 ; MIPS32R5EB-NEXT: jal i8_4 ; MIPS32R5EB-NEXT: nop ; MIPS32R5EB-NEXT: lui $1, %hi(gv4i8) ; MIPS32R5EB-NEXT: sw $2, %lo(gv4i8)($1) ; MIPS32R5EB-NEXT: lw $ra, 28($sp) # 4-byte Folded Reload ; MIPS32R5EB-NEXT: addiu $sp, $sp, 32 ; MIPS32R5EB-NEXT: jr $ra ; MIPS32R5EB-NEXT: nop ; ; MIPS64R5EB-LABEL: call_i8_4: ; MIPS64R5EB: # %bb.0: # %entry ; MIPS64R5EB-NEXT: daddiu $sp, $sp, -32 ; MIPS64R5EB-NEXT: .cfi_def_cfa_offset 32 ; MIPS64R5EB-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64R5EB-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64R5EB-NEXT: .cfi_offset 31, -8 ; MIPS64R5EB-NEXT: .cfi_offset 28, -16 ; MIPS64R5EB-NEXT: lui $1, %hi(%neg(%gp_rel(call_i8_4))) ; MIPS64R5EB-NEXT: daddu $1, $1, $25 ; MIPS64R5EB-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(call_i8_4))) ; MIPS64R5EB-NEXT: lui $1, 1543 ; MIPS64R5EB-NEXT: ori $4, $1, 2314 ; MIPS64R5EB-NEXT: lui $1, 3080 ; MIPS64R5EB-NEXT: ori $5, $1, 2314 ; MIPS64R5EB-NEXT: ld $25, %call16(i8_4)($gp) ; MIPS64R5EB-NEXT: jalr $25 ; MIPS64R5EB-NEXT: nop ; MIPS64R5EB-NEXT: ld $1, %got_disp(gv4i8)($gp) ; MIPS64R5EB-NEXT: sw $2, 0($1) ; MIPS64R5EB-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64R5EB-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64R5EB-NEXT: daddiu $sp, $sp, 32 ; MIPS64R5EB-NEXT: jr $ra ; MIPS64R5EB-NEXT: nop ; ; MIPS32EL-LABEL: call_i8_4: ; MIPS32EL: # %bb.0: # %entry ; MIPS32EL-NEXT: addiu $sp, $sp, -24 ; MIPS32EL-NEXT: .cfi_def_cfa_offset 24 ; MIPS32EL-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; MIPS32EL-NEXT: .cfi_offset 31, -4 ; MIPS32EL-NEXT: lui $1, 2569 ; MIPS32EL-NEXT: ori $4, $1, 1798 ; MIPS32EL-NEXT: ori $5, $1, 2060 ; MIPS32EL-NEXT: jal i8_4 ; MIPS32EL-NEXT: nop ; MIPS32EL-NEXT: lui $1, %hi(gv4i8) ; MIPS32EL-NEXT: sw $2, %lo(gv4i8)($1) ; MIPS32EL-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; MIPS32EL-NEXT: addiu $sp, $sp, 24 ; MIPS32EL-NEXT: jr $ra ; MIPS32EL-NEXT: nop ; ; MIPS64EL-LABEL: call_i8_4: ; MIPS64EL: # %bb.0: # %entry ; MIPS64EL-NEXT: daddiu $sp, $sp, -16 ; MIPS64EL-NEXT: .cfi_def_cfa_offset 16 ; MIPS64EL-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; MIPS64EL-NEXT: sd $gp, 0($sp) # 8-byte Folded Spill ; MIPS64EL-NEXT: .cfi_offset 31, -8 ; MIPS64EL-NEXT: .cfi_offset 28, -16 ; MIPS64EL-NEXT: lui $1, %hi(%neg(%gp_rel(call_i8_4))) ; MIPS64EL-NEXT: daddu $1, $1, $25 ; MIPS64EL-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(call_i8_4))) ; MIPS64EL-NEXT: lui $1, 2569 ; MIPS64EL-NEXT: ori $4, $1, 1798 ; MIPS64EL-NEXT: ori $5, $1, 2060 ; MIPS64EL-NEXT: ld $25, %call16(i8_4)($gp) ; MIPS64EL-NEXT: jalr $25 ; MIPS64EL-NEXT: nop ; MIPS64EL-NEXT: ld $1, %got_disp(gv4i8)($gp) ; MIPS64EL-NEXT: sw $2, 0($1) ; MIPS64EL-NEXT: ld $gp, 0($sp) # 8-byte Folded Reload ; MIPS64EL-NEXT: ld $ra, 8($sp) # 8-byte Folded Reload ; MIPS64EL-NEXT: daddiu $sp, $sp, 16 ; MIPS64EL-NEXT: jr $ra ; MIPS64EL-NEXT: nop ; ; MIPS32R5EL-LABEL: call_i8_4: ; MIPS32R5EL: # %bb.0: # %entry ; MIPS32R5EL-NEXT: addiu $sp, $sp, -32 ; MIPS32R5EL-NEXT: .cfi_def_cfa_offset 32 ; MIPS32R5EL-NEXT: sw $ra, 28($sp) # 4-byte Folded Spill ; MIPS32R5EL-NEXT: .cfi_offset 31, -4 ; MIPS32R5EL-NEXT: lui $1, 2569 ; MIPS32R5EL-NEXT: ori $4, $1, 1798 ; MIPS32R5EL-NEXT: ori $5, $1, 2060 ; MIPS32R5EL-NEXT: jal i8_4 ; MIPS32R5EL-NEXT: nop ; MIPS32R5EL-NEXT: lui $1, %hi(gv4i8) ; MIPS32R5EL-NEXT: sw $2, %lo(gv4i8)($1) ; MIPS32R5EL-NEXT: lw $ra, 28($sp) # 4-byte Folded Reload ; MIPS32R5EL-NEXT: addiu $sp, $sp, 32 ; MIPS32R5EL-NEXT: jr $ra ; MIPS32R5EL-NEXT: nop ; ; MIPS64R5EL-LABEL: call_i8_4: ; MIPS64R5EL: # %bb.0: # %entry ; MIPS64R5EL-NEXT: daddiu $sp, $sp, -32 ; MIPS64R5EL-NEXT: .cfi_def_cfa_offset 32 ; MIPS64R5EL-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64R5EL-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64R5EL-NEXT: .cfi_offset 31, -8 ; MIPS64R5EL-NEXT: .cfi_offset 28, -16 ; MIPS64R5EL-NEXT: lui $1, %hi(%neg(%gp_rel(call_i8_4))) ; MIPS64R5EL-NEXT: daddu $1, $1, $25 ; MIPS64R5EL-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(call_i8_4))) ; MIPS64R5EL-NEXT: lui $1, 2569 ; MIPS64R5EL-NEXT: ori $4, $1, 1798 ; MIPS64R5EL-NEXT: ori $5, $1, 2060 ; MIPS64R5EL-NEXT: ld $25, %call16(i8_4)($gp) ; MIPS64R5EL-NEXT: jalr $25 ; MIPS64R5EL-NEXT: nop ; MIPS64R5EL-NEXT: ld $1, %got_disp(gv4i8)($gp) ; MIPS64R5EL-NEXT: sw $2, 0($1) ; MIPS64R5EL-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64R5EL-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64R5EL-NEXT: daddiu $sp, $sp, 32 ; MIPS64R5EL-NEXT: jr $ra ; MIPS64R5EL-NEXT: nop entry: %0 = call <4 x i8> @i8_4(<4 x i8> , <4 x i8> ) store <4 x i8> %0, <4 x i8> * @gv4i8 ret void } define void @call_i8_8() { ; MIPS32EB-LABEL: call_i8_8: ; MIPS32EB: # %bb.0: # %entry ; MIPS32EB-NEXT: addiu $sp, $sp, -24 ; MIPS32EB-NEXT: .cfi_def_cfa_offset 24 ; MIPS32EB-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; MIPS32EB-NEXT: .cfi_offset 31, -4 ; MIPS32EB-NEXT: lui $1, 3080 ; MIPS32EB-NEXT: ori $6, $1, 2314 ; MIPS32EB-NEXT: lui $1, 1543 ; MIPS32EB-NEXT: ori $4, $1, 2314 ; MIPS32EB-NEXT: move $5, $4 ; MIPS32EB-NEXT: move $7, $4 ; MIPS32EB-NEXT: jal i8_8 ; MIPS32EB-NEXT: nop ; MIPS32EB-NEXT: lui $1, %hi(gv8i8) ; MIPS32EB-NEXT: addiu $4, $1, %lo(gv8i8) ; MIPS32EB-NEXT: sw $3, 4($4) ; MIPS32EB-NEXT: sw $2, %lo(gv8i8)($1) ; MIPS32EB-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; MIPS32EB-NEXT: addiu $sp, $sp, 24 ; MIPS32EB-NEXT: jr $ra ; MIPS32EB-NEXT: nop ; ; MIPS64EB-LABEL: call_i8_8: ; MIPS64EB: # %bb.0: # %entry ; MIPS64EB-NEXT: daddiu $sp, $sp, -16 ; MIPS64EB-NEXT: .cfi_def_cfa_offset 16 ; MIPS64EB-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; MIPS64EB-NEXT: sd $gp, 0($sp) # 8-byte Folded Spill ; MIPS64EB-NEXT: .cfi_offset 31, -8 ; MIPS64EB-NEXT: .cfi_offset 28, -16 ; MIPS64EB-NEXT: lui $1, %hi(%neg(%gp_rel(call_i8_8))) ; MIPS64EB-NEXT: daddu $1, $1, $25 ; MIPS64EB-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(call_i8_8))) ; MIPS64EB-NEXT: lui $1, 772 ; MIPS64EB-NEXT: daddiu $1, $1, -31611 ; MIPS64EB-NEXT: dsll $1, $1, 17 ; MIPS64EB-NEXT: daddiu $1, $1, 1543 ; MIPS64EB-NEXT: dsll $1, $1, 16 ; MIPS64EB-NEXT: daddiu $4, $1, 2314 ; MIPS64EB-NEXT: lui $1, 1540 ; MIPS64EB-NEXT: daddiu $1, $1, 1157 ; MIPS64EB-NEXT: dsll $1, $1, 17 ; MIPS64EB-NEXT: daddiu $1, $1, 1543 ; MIPS64EB-NEXT: dsll $1, $1, 16 ; MIPS64EB-NEXT: daddiu $5, $1, 2314 ; MIPS64EB-NEXT: ld $25, %call16(i8_8)($gp) ; MIPS64EB-NEXT: jalr $25 ; MIPS64EB-NEXT: nop ; MIPS64EB-NEXT: ld $1, %got_disp(gv8i8)($gp) ; MIPS64EB-NEXT: sd $2, 0($1) ; MIPS64EB-NEXT: ld $gp, 0($sp) # 8-byte Folded Reload ; MIPS64EB-NEXT: ld $ra, 8($sp) # 8-byte Folded Reload ; MIPS64EB-NEXT: daddiu $sp, $sp, 16 ; MIPS64EB-NEXT: jr $ra ; MIPS64EB-NEXT: nop ; ; MIPS32R5EB-LABEL: call_i8_8: ; MIPS32R5EB: # %bb.0: # %entry ; MIPS32R5EB-NEXT: addiu $sp, $sp, -24 ; MIPS32R5EB-NEXT: .cfi_def_cfa_offset 24 ; MIPS32R5EB-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; MIPS32R5EB-NEXT: .cfi_offset 31, -4 ; MIPS32R5EB-NEXT: lui $1, 3080 ; MIPS32R5EB-NEXT: ori $6, $1, 2314 ; MIPS32R5EB-NEXT: lui $1, 1543 ; MIPS32R5EB-NEXT: ori $4, $1, 2314 ; MIPS32R5EB-NEXT: move $5, $4 ; MIPS32R5EB-NEXT: move $7, $4 ; MIPS32R5EB-NEXT: jal i8_8 ; MIPS32R5EB-NEXT: nop ; MIPS32R5EB-NEXT: lui $1, %hi(gv8i8) ; MIPS32R5EB-NEXT: addiu $4, $1, %lo(gv8i8) ; MIPS32R5EB-NEXT: sw $3, 4($4) ; MIPS32R5EB-NEXT: sw $2, %lo(gv8i8)($1) ; MIPS32R5EB-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; MIPS32R5EB-NEXT: addiu $sp, $sp, 24 ; MIPS32R5EB-NEXT: jr $ra ; MIPS32R5EB-NEXT: nop ; ; MIPS64R5EB-LABEL: call_i8_8: ; MIPS64R5EB: # %bb.0: # %entry ; MIPS64R5EB-NEXT: daddiu $sp, $sp, -32 ; MIPS64R5EB-NEXT: .cfi_def_cfa_offset 32 ; MIPS64R5EB-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64R5EB-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64R5EB-NEXT: .cfi_offset 31, -8 ; MIPS64R5EB-NEXT: .cfi_offset 28, -16 ; MIPS64R5EB-NEXT: lui $1, %hi(%neg(%gp_rel(call_i8_8))) ; MIPS64R5EB-NEXT: daddu $1, $1, $25 ; MIPS64R5EB-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(call_i8_8))) ; MIPS64R5EB-NEXT: lui $1, 772 ; MIPS64R5EB-NEXT: daddiu $1, $1, -31611 ; MIPS64R5EB-NEXT: dsll $1, $1, 17 ; MIPS64R5EB-NEXT: daddiu $1, $1, 1543 ; MIPS64R5EB-NEXT: dsll $1, $1, 16 ; MIPS64R5EB-NEXT: daddiu $4, $1, 2314 ; MIPS64R5EB-NEXT: lui $1, 1540 ; MIPS64R5EB-NEXT: daddiu $1, $1, 1157 ; MIPS64R5EB-NEXT: dsll $1, $1, 17 ; MIPS64R5EB-NEXT: daddiu $1, $1, 1543 ; MIPS64R5EB-NEXT: dsll $1, $1, 16 ; MIPS64R5EB-NEXT: daddiu $5, $1, 2314 ; MIPS64R5EB-NEXT: ld $25, %call16(i8_8)($gp) ; MIPS64R5EB-NEXT: jalr $25 ; MIPS64R5EB-NEXT: nop ; MIPS64R5EB-NEXT: ld $1, %got_disp(gv8i8)($gp) ; MIPS64R5EB-NEXT: sd $2, 0($1) ; MIPS64R5EB-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64R5EB-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64R5EB-NEXT: daddiu $sp, $sp, 32 ; MIPS64R5EB-NEXT: jr $ra ; MIPS64R5EB-NEXT: nop ; ; MIPS32EL-LABEL: call_i8_8: ; MIPS32EL: # %bb.0: # %entry ; MIPS32EL-NEXT: addiu $sp, $sp, -24 ; MIPS32EL-NEXT: .cfi_def_cfa_offset 24 ; MIPS32EL-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; MIPS32EL-NEXT: .cfi_offset 31, -4 ; MIPS32EL-NEXT: lui $1, 2569 ; MIPS32EL-NEXT: ori $6, $1, 2060 ; MIPS32EL-NEXT: ori $4, $1, 1798 ; MIPS32EL-NEXT: move $5, $4 ; MIPS32EL-NEXT: move $7, $4 ; MIPS32EL-NEXT: jal i8_8 ; MIPS32EL-NEXT: nop ; MIPS32EL-NEXT: lui $1, %hi(gv8i8) ; MIPS32EL-NEXT: addiu $4, $1, %lo(gv8i8) ; MIPS32EL-NEXT: sw $3, 4($4) ; MIPS32EL-NEXT: sw $2, %lo(gv8i8)($1) ; MIPS32EL-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; MIPS32EL-NEXT: addiu $sp, $sp, 24 ; MIPS32EL-NEXT: jr $ra ; MIPS32EL-NEXT: nop ; ; MIPS64EL-LABEL: call_i8_8: ; MIPS64EL: # %bb.0: # %entry ; MIPS64EL-NEXT: daddiu $sp, $sp, -16 ; MIPS64EL-NEXT: .cfi_def_cfa_offset 16 ; MIPS64EL-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; MIPS64EL-NEXT: sd $gp, 0($sp) # 8-byte Folded Spill ; MIPS64EL-NEXT: .cfi_offset 31, -8 ; MIPS64EL-NEXT: .cfi_offset 28, -16 ; MIPS64EL-NEXT: lui $1, %hi(%neg(%gp_rel(call_i8_8))) ; MIPS64EL-NEXT: daddu $1, $1, $25 ; MIPS64EL-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(call_i8_8))) ; MIPS64EL-NEXT: lui $1, 1285 ; MIPS64EL-NEXT: daddiu $1, $1, -31869 ; MIPS64EL-NEXT: dsll $1, $1, 17 ; MIPS64EL-NEXT: daddiu $1, $1, 2569 ; MIPS64EL-NEXT: dsll $1, $1, 16 ; MIPS64EL-NEXT: daddiu $4, $1, 1798 ; MIPS64EL-NEXT: daddiu $5, $1, 2060 ; MIPS64EL-NEXT: ld $25, %call16(i8_8)($gp) ; MIPS64EL-NEXT: jalr $25 ; MIPS64EL-NEXT: nop ; MIPS64EL-NEXT: ld $1, %got_disp(gv8i8)($gp) ; MIPS64EL-NEXT: sd $2, 0($1) ; MIPS64EL-NEXT: ld $gp, 0($sp) # 8-byte Folded Reload ; MIPS64EL-NEXT: ld $ra, 8($sp) # 8-byte Folded Reload ; MIPS64EL-NEXT: daddiu $sp, $sp, 16 ; MIPS64EL-NEXT: jr $ra ; MIPS64EL-NEXT: nop ; ; MIPS32R5EL-LABEL: call_i8_8: ; MIPS32R5EL: # %bb.0: # %entry ; MIPS32R5EL-NEXT: addiu $sp, $sp, -24 ; MIPS32R5EL-NEXT: .cfi_def_cfa_offset 24 ; MIPS32R5EL-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; MIPS32R5EL-NEXT: .cfi_offset 31, -4 ; MIPS32R5EL-NEXT: lui $1, 2569 ; MIPS32R5EL-NEXT: ori $6, $1, 2060 ; MIPS32R5EL-NEXT: ori $4, $1, 1798 ; MIPS32R5EL-NEXT: move $5, $4 ; MIPS32R5EL-NEXT: move $7, $4 ; MIPS32R5EL-NEXT: jal i8_8 ; MIPS32R5EL-NEXT: nop ; MIPS32R5EL-NEXT: lui $1, %hi(gv8i8) ; MIPS32R5EL-NEXT: addiu $4, $1, %lo(gv8i8) ; MIPS32R5EL-NEXT: sw $3, 4($4) ; MIPS32R5EL-NEXT: sw $2, %lo(gv8i8)($1) ; MIPS32R5EL-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; MIPS32R5EL-NEXT: addiu $sp, $sp, 24 ; MIPS32R5EL-NEXT: jr $ra ; MIPS32R5EL-NEXT: nop ; ; MIPS64R5EL-LABEL: call_i8_8: ; MIPS64R5EL: # %bb.0: # %entry ; MIPS64R5EL-NEXT: daddiu $sp, $sp, -32 ; MIPS64R5EL-NEXT: .cfi_def_cfa_offset 32 ; MIPS64R5EL-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64R5EL-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64R5EL-NEXT: .cfi_offset 31, -8 ; MIPS64R5EL-NEXT: .cfi_offset 28, -16 ; MIPS64R5EL-NEXT: lui $1, %hi(%neg(%gp_rel(call_i8_8))) ; MIPS64R5EL-NEXT: daddu $1, $1, $25 ; MIPS64R5EL-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(call_i8_8))) ; MIPS64R5EL-NEXT: lui $1, 1285 ; MIPS64R5EL-NEXT: daddiu $1, $1, -31869 ; MIPS64R5EL-NEXT: dsll $1, $1, 17 ; MIPS64R5EL-NEXT: daddiu $1, $1, 2569 ; MIPS64R5EL-NEXT: dsll $1, $1, 16 ; MIPS64R5EL-NEXT: daddiu $4, $1, 1798 ; MIPS64R5EL-NEXT: daddiu $5, $1, 2060 ; MIPS64R5EL-NEXT: ld $25, %call16(i8_8)($gp) ; MIPS64R5EL-NEXT: jalr $25 ; MIPS64R5EL-NEXT: nop ; MIPS64R5EL-NEXT: ld $1, %got_disp(gv8i8)($gp) ; MIPS64R5EL-NEXT: sd $2, 0($1) ; MIPS64R5EL-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64R5EL-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64R5EL-NEXT: daddiu $sp, $sp, 32 ; MIPS64R5EL-NEXT: jr $ra ; MIPS64R5EL-NEXT: nop entry: %0 = call <8 x i8> @i8_8(<8 x i8> , <8 x i8> ) store <8 x i8> %0, <8 x i8> * @gv8i8 ret void } define void @calli8_16() { ; MIPS32EB-LABEL: calli8_16: ; MIPS32EB: # %bb.0: # %entry ; MIPS32EB-NEXT: addiu $sp, $sp, -40 ; MIPS32EB-NEXT: .cfi_def_cfa_offset 40 ; MIPS32EB-NEXT: sw $ra, 36($sp) # 4-byte Folded Spill ; MIPS32EB-NEXT: .cfi_offset 31, -4 ; MIPS32EB-NEXT: lui $1, 3080 ; MIPS32EB-NEXT: ori $1, $1, 2314 ; MIPS32EB-NEXT: lui $2, 1801 ; MIPS32EB-NEXT: sw $1, 28($sp) ; MIPS32EB-NEXT: ori $1, $2, 1801 ; MIPS32EB-NEXT: sw $1, 24($sp) ; MIPS32EB-NEXT: sw $1, 20($sp) ; MIPS32EB-NEXT: sw $1, 16($sp) ; MIPS32EB-NEXT: lui $1, 1543 ; MIPS32EB-NEXT: ori $4, $1, 1543 ; MIPS32EB-NEXT: ori $7, $1, 2314 ; MIPS32EB-NEXT: move $5, $4 ; MIPS32EB-NEXT: move $6, $4 ; MIPS32EB-NEXT: jal i8_16 ; MIPS32EB-NEXT: nop ; MIPS32EB-NEXT: lui $1, %hi(gv16i8) ; MIPS32EB-NEXT: addiu $6, $1, %lo(gv16i8) ; MIPS32EB-NEXT: sw $5, 12($6) ; MIPS32EB-NEXT: sw $4, 8($6) ; MIPS32EB-NEXT: sw $3, 4($6) ; MIPS32EB-NEXT: sw $2, %lo(gv16i8)($1) ; MIPS32EB-NEXT: lw $ra, 36($sp) # 4-byte Folded Reload ; MIPS32EB-NEXT: addiu $sp, $sp, 40 ; MIPS32EB-NEXT: jr $ra ; MIPS32EB-NEXT: nop ; ; MIPS64EB-LABEL: calli8_16: ; MIPS64EB: # %bb.0: # %entry ; MIPS64EB-NEXT: daddiu $sp, $sp, -16 ; MIPS64EB-NEXT: .cfi_def_cfa_offset 16 ; MIPS64EB-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; MIPS64EB-NEXT: sd $gp, 0($sp) # 8-byte Folded Spill ; MIPS64EB-NEXT: .cfi_offset 31, -8 ; MIPS64EB-NEXT: .cfi_offset 28, -16 ; MIPS64EB-NEXT: lui $1, %hi(%neg(%gp_rel(calli8_16))) ; MIPS64EB-NEXT: daddu $1, $1, $25 ; MIPS64EB-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(calli8_16))) ; MIPS64EB-NEXT: lui $1, 1801 ; MIPS64EB-NEXT: daddiu $1, $1, 1801 ; MIPS64EB-NEXT: dsll $1, $1, 16 ; MIPS64EB-NEXT: daddiu $1, $1, 1801 ; MIPS64EB-NEXT: lui $2, 1543 ; MIPS64EB-NEXT: dsll $1, $1, 16 ; MIPS64EB-NEXT: daddiu $2, $2, 1543 ; MIPS64EB-NEXT: dsll $2, $2, 16 ; MIPS64EB-NEXT: daddiu $2, $2, 1543 ; MIPS64EB-NEXT: dsll $2, $2, 16 ; MIPS64EB-NEXT: daddiu $4, $2, 1543 ; MIPS64EB-NEXT: daddiu $5, $2, 2314 ; MIPS64EB-NEXT: daddiu $6, $1, 1801 ; MIPS64EB-NEXT: lui $1, 225 ; MIPS64EB-NEXT: daddiu $1, $1, 8417 ; MIPS64EB-NEXT: dsll $1, $1, 16 ; MIPS64EB-NEXT: daddiu $1, $1, 8577 ; MIPS64EB-NEXT: dsll $1, $1, 19 ; MIPS64EB-NEXT: daddiu $7, $1, 2314 ; MIPS64EB-NEXT: ld $25, %call16(i8_16)($gp) ; MIPS64EB-NEXT: jalr $25 ; MIPS64EB-NEXT: nop ; MIPS64EB-NEXT: ld $1, %got_disp(gv16i8)($gp) ; MIPS64EB-NEXT: sd $3, 8($1) ; MIPS64EB-NEXT: sd $2, 0($1) ; MIPS64EB-NEXT: ld $gp, 0($sp) # 8-byte Folded Reload ; MIPS64EB-NEXT: ld $ra, 8($sp) # 8-byte Folded Reload ; MIPS64EB-NEXT: daddiu $sp, $sp, 16 ; MIPS64EB-NEXT: jr $ra ; MIPS64EB-NEXT: nop ; ; MIPS32R5-LABEL: calli8_16: ; MIPS32R5: # %bb.0: # %entry ; MIPS32R5-NEXT: addiu $sp, $sp, -40 ; MIPS32R5-NEXT: .cfi_def_cfa_offset 40 ; MIPS32R5-NEXT: sw $ra, 36($sp) # 4-byte Folded Spill ; MIPS32R5-NEXT: .cfi_offset 31, -4 ; MIPS32R5-NEXT: lui $1, %hi($CPI30_0) ; MIPS32R5-NEXT: addiu $1, $1, %lo($CPI30_0) ; MIPS32R5-NEXT: ld.w $w0, 0($1) ; MIPS32R5-NEXT: copy_s.w $4, $w0[0] ; MIPS32R5-NEXT: copy_s.w $5, $w0[1] ; MIPS32R5-NEXT: copy_s.w $6, $w0[2] ; MIPS32R5-NEXT: copy_s.w $7, $w0[3] ; MIPS32R5-NEXT: lui $1, %hi($CPI30_1) ; MIPS32R5-NEXT: addiu $1, $1, %lo($CPI30_1) ; MIPS32R5-NEXT: ld.w $w0, 0($1) ; MIPS32R5-NEXT: copy_s.w $1, $w0[0] ; MIPS32R5-NEXT: copy_s.w $2, $w0[1] ; MIPS32R5-NEXT: copy_s.w $3, $w0[2] ; MIPS32R5-NEXT: copy_s.w $8, $w0[3] ; MIPS32R5-NEXT: sw $8, 28($sp) ; MIPS32R5-NEXT: sw $3, 24($sp) ; MIPS32R5-NEXT: sw $2, 20($sp) ; MIPS32R5-NEXT: sw $1, 16($sp) ; MIPS32R5-NEXT: jal i8_16 ; MIPS32R5-NEXT: nop ; MIPS32R5-NEXT: ldi.b $w0, 0 ; MIPS32R5-NEXT: insert.w $w0[0], $2 ; MIPS32R5-NEXT: lui $1, %hi(gv16i8) ; MIPS32R5-NEXT: insert.w $w0[1], $3 ; MIPS32R5-NEXT: addiu $1, $1, %lo(gv16i8) ; MIPS32R5-NEXT: insert.w $w0[2], $4 ; MIPS32R5-NEXT: insert.w $w0[3], $5 ; MIPS32R5-NEXT: st.w $w0, 0($1) ; MIPS32R5-NEXT: lw $ra, 36($sp) # 4-byte Folded Reload ; MIPS32R5-NEXT: addiu $sp, $sp, 40 ; MIPS32R5-NEXT: jr $ra ; MIPS32R5-NEXT: nop ; ; MIPS64R5-LABEL: calli8_16: ; MIPS64R5: # %bb.0: # %entry ; MIPS64R5-NEXT: daddiu $sp, $sp, -16 ; MIPS64R5-NEXT: .cfi_def_cfa_offset 16 ; MIPS64R5-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; MIPS64R5-NEXT: sd $gp, 0($sp) # 8-byte Folded Spill ; MIPS64R5-NEXT: .cfi_offset 31, -8 ; MIPS64R5-NEXT: .cfi_offset 28, -16 ; MIPS64R5-NEXT: lui $1, %hi(%neg(%gp_rel(calli8_16))) ; MIPS64R5-NEXT: daddu $1, $1, $25 ; MIPS64R5-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(calli8_16))) ; MIPS64R5-NEXT: ld $1, %got_page(.LCPI30_0)($gp) ; MIPS64R5-NEXT: daddiu $1, $1, %got_ofst(.LCPI30_0) ; MIPS64R5-NEXT: ld.d $w0, 0($1) ; MIPS64R5-NEXT: copy_s.d $4, $w0[0] ; MIPS64R5-NEXT: copy_s.d $5, $w0[1] ; MIPS64R5-NEXT: ld $1, %got_page(.LCPI30_1)($gp) ; MIPS64R5-NEXT: daddiu $1, $1, %got_ofst(.LCPI30_1) ; MIPS64R5-NEXT: ld.d $w0, 0($1) ; MIPS64R5-NEXT: copy_s.d $6, $w0[0] ; MIPS64R5-NEXT: copy_s.d $7, $w0[1] ; MIPS64R5-NEXT: ld $25, %call16(i8_16)($gp) ; MIPS64R5-NEXT: jalr $25 ; MIPS64R5-NEXT: nop ; MIPS64R5-NEXT: ldi.b $w0, 0 ; MIPS64R5-NEXT: insert.d $w0[0], $2 ; MIPS64R5-NEXT: insert.d $w0[1], $3 ; MIPS64R5-NEXT: ld $1, %got_disp(gv16i8)($gp) ; MIPS64R5-NEXT: st.d $w0, 0($1) ; MIPS64R5-NEXT: ld $gp, 0($sp) # 8-byte Folded Reload ; MIPS64R5-NEXT: ld $ra, 8($sp) # 8-byte Folded Reload ; MIPS64R5-NEXT: daddiu $sp, $sp, 16 ; MIPS64R5-NEXT: jr $ra ; MIPS64R5-NEXT: nop ; ; MIPS32EL-LABEL: calli8_16: ; MIPS32EL: # %bb.0: # %entry ; MIPS32EL-NEXT: addiu $sp, $sp, -40 ; MIPS32EL-NEXT: .cfi_def_cfa_offset 40 ; MIPS32EL-NEXT: sw $ra, 36($sp) # 4-byte Folded Spill ; MIPS32EL-NEXT: .cfi_offset 31, -4 ; MIPS32EL-NEXT: lui $1, 2569 ; MIPS32EL-NEXT: ori $2, $1, 2060 ; MIPS32EL-NEXT: lui $3, 2311 ; MIPS32EL-NEXT: sw $2, 28($sp) ; MIPS32EL-NEXT: ori $2, $3, 2311 ; MIPS32EL-NEXT: sw $2, 24($sp) ; MIPS32EL-NEXT: sw $2, 20($sp) ; MIPS32EL-NEXT: sw $2, 16($sp) ; MIPS32EL-NEXT: lui $2, 1798 ; MIPS32EL-NEXT: ori $4, $2, 1798 ; MIPS32EL-NEXT: ori $7, $1, 1798 ; MIPS32EL-NEXT: move $5, $4 ; MIPS32EL-NEXT: move $6, $4 ; MIPS32EL-NEXT: jal i8_16 ; MIPS32EL-NEXT: nop ; MIPS32EL-NEXT: lui $1, %hi(gv16i8) ; MIPS32EL-NEXT: addiu $6, $1, %lo(gv16i8) ; MIPS32EL-NEXT: sw $5, 12($6) ; MIPS32EL-NEXT: sw $4, 8($6) ; MIPS32EL-NEXT: sw $3, 4($6) ; MIPS32EL-NEXT: sw $2, %lo(gv16i8)($1) ; MIPS32EL-NEXT: lw $ra, 36($sp) # 4-byte Folded Reload ; MIPS32EL-NEXT: addiu $sp, $sp, 40 ; MIPS32EL-NEXT: jr $ra ; MIPS32EL-NEXT: nop ; ; MIPS64EL-LABEL: calli8_16: ; MIPS64EL: # %bb.0: # %entry ; MIPS64EL-NEXT: daddiu $sp, $sp, -16 ; MIPS64EL-NEXT: .cfi_def_cfa_offset 16 ; MIPS64EL-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; MIPS64EL-NEXT: sd $gp, 0($sp) # 8-byte Folded Spill ; MIPS64EL-NEXT: .cfi_offset 31, -8 ; MIPS64EL-NEXT: .cfi_offset 28, -16 ; MIPS64EL-NEXT: lui $1, %hi(%neg(%gp_rel(calli8_16))) ; MIPS64EL-NEXT: daddu $1, $1, $25 ; MIPS64EL-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(calli8_16))) ; MIPS64EL-NEXT: lui $1, 1285 ; MIPS64EL-NEXT: daddiu $1, $1, -31869 ; MIPS64EL-NEXT: dsll $1, $1, 16 ; MIPS64EL-NEXT: daddiu $1, $1, 899 ; MIPS64EL-NEXT: lui $2, 2311 ; MIPS64EL-NEXT: daddiu $2, $2, 2311 ; MIPS64EL-NEXT: dsll $2, $2, 16 ; MIPS64EL-NEXT: daddiu $2, $2, 2311 ; MIPS64EL-NEXT: dsll $2, $2, 16 ; MIPS64EL-NEXT: dsll $1, $1, 17 ; MIPS64EL-NEXT: lui $3, 899 ; MIPS64EL-NEXT: daddiu $3, $3, 899 ; MIPS64EL-NEXT: dsll $3, $3, 16 ; MIPS64EL-NEXT: daddiu $3, $3, 899 ; MIPS64EL-NEXT: dsll $3, $3, 17 ; MIPS64EL-NEXT: daddiu $4, $3, 1798 ; MIPS64EL-NEXT: daddiu $5, $1, 1798 ; MIPS64EL-NEXT: daddiu $6, $2, 2311 ; MIPS64EL-NEXT: lui $1, 642 ; MIPS64EL-NEXT: daddiu $1, $1, 16899 ; MIPS64EL-NEXT: dsll $1, $1, 18 ; MIPS64EL-NEXT: daddiu $1, $1, 2311 ; MIPS64EL-NEXT: dsll $1, $1, 16 ; MIPS64EL-NEXT: daddiu $7, $1, 2311 ; MIPS64EL-NEXT: ld $25, %call16(i8_16)($gp) ; MIPS64EL-NEXT: jalr $25 ; MIPS64EL-NEXT: nop ; MIPS64EL-NEXT: ld $1, %got_disp(gv16i8)($gp) ; MIPS64EL-NEXT: sd $3, 8($1) ; MIPS64EL-NEXT: sd $2, 0($1) ; MIPS64EL-NEXT: ld $gp, 0($sp) # 8-byte Folded Reload ; MIPS64EL-NEXT: ld $ra, 8($sp) # 8-byte Folded Reload ; MIPS64EL-NEXT: daddiu $sp, $sp, 16 ; MIPS64EL-NEXT: jr $ra ; MIPS64EL-NEXT: nop entry: %0 = call <16 x i8> @i8_16(<16 x i8> , <16 x i8> ) store <16 x i8> %0, <16 x i8> * @gv16i8 ret void } define void @calli16_2() { ; MIPS32EB-LABEL: calli16_2: ; MIPS32EB: # %bb.0: # %entry ; MIPS32EB-NEXT: addiu $sp, $sp, -24 ; MIPS32EB-NEXT: .cfi_def_cfa_offset 24 ; MIPS32EB-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; MIPS32EB-NEXT: .cfi_offset 31, -4 ; MIPS32EB-NEXT: lui $1, 6 ; MIPS32EB-NEXT: ori $4, $1, 7 ; MIPS32EB-NEXT: lui $1, 12 ; MIPS32EB-NEXT: ori $5, $1, 8 ; MIPS32EB-NEXT: jal i16_2 ; MIPS32EB-NEXT: nop ; MIPS32EB-NEXT: lui $1, %hi(gv2i16) ; MIPS32EB-NEXT: sw $2, %lo(gv2i16)($1) ; MIPS32EB-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; MIPS32EB-NEXT: addiu $sp, $sp, 24 ; MIPS32EB-NEXT: jr $ra ; MIPS32EB-NEXT: nop ; ; MIPS64EB-LABEL: calli16_2: ; MIPS64EB: # %bb.0: # %entry ; MIPS64EB-NEXT: daddiu $sp, $sp, -16 ; MIPS64EB-NEXT: .cfi_def_cfa_offset 16 ; MIPS64EB-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; MIPS64EB-NEXT: sd $gp, 0($sp) # 8-byte Folded Spill ; MIPS64EB-NEXT: .cfi_offset 31, -8 ; MIPS64EB-NEXT: .cfi_offset 28, -16 ; MIPS64EB-NEXT: lui $1, %hi(%neg(%gp_rel(calli16_2))) ; MIPS64EB-NEXT: daddu $1, $1, $25 ; MIPS64EB-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(calli16_2))) ; MIPS64EB-NEXT: lui $1, 6 ; MIPS64EB-NEXT: ori $4, $1, 7 ; MIPS64EB-NEXT: lui $1, 12 ; MIPS64EB-NEXT: ori $5, $1, 8 ; MIPS64EB-NEXT: ld $25, %call16(i16_2)($gp) ; MIPS64EB-NEXT: jalr $25 ; MIPS64EB-NEXT: nop ; MIPS64EB-NEXT: ld $1, %got_disp(gv2i16)($gp) ; MIPS64EB-NEXT: sw $2, 0($1) ; MIPS64EB-NEXT: ld $gp, 0($sp) # 8-byte Folded Reload ; MIPS64EB-NEXT: ld $ra, 8($sp) # 8-byte Folded Reload ; MIPS64EB-NEXT: daddiu $sp, $sp, 16 ; MIPS64EB-NEXT: jr $ra ; MIPS64EB-NEXT: nop ; ; MIPS32R5EB-LABEL: calli16_2: ; MIPS32R5EB: # %bb.0: # %entry ; MIPS32R5EB-NEXT: addiu $sp, $sp, -32 ; MIPS32R5EB-NEXT: .cfi_def_cfa_offset 32 ; MIPS32R5EB-NEXT: sw $ra, 28($sp) # 4-byte Folded Spill ; MIPS32R5EB-NEXT: .cfi_offset 31, -4 ; MIPS32R5EB-NEXT: lui $1, 6 ; MIPS32R5EB-NEXT: ori $4, $1, 7 ; MIPS32R5EB-NEXT: lui $1, 12 ; MIPS32R5EB-NEXT: ori $5, $1, 8 ; MIPS32R5EB-NEXT: jal i16_2 ; MIPS32R5EB-NEXT: nop ; MIPS32R5EB-NEXT: lui $1, %hi(gv2i16) ; MIPS32R5EB-NEXT: sw $2, %lo(gv2i16)($1) ; MIPS32R5EB-NEXT: lw $ra, 28($sp) # 4-byte Folded Reload ; MIPS32R5EB-NEXT: addiu $sp, $sp, 32 ; MIPS32R5EB-NEXT: jr $ra ; MIPS32R5EB-NEXT: nop ; ; MIPS64R5EB-LABEL: calli16_2: ; MIPS64R5EB: # %bb.0: # %entry ; MIPS64R5EB-NEXT: daddiu $sp, $sp, -32 ; MIPS64R5EB-NEXT: .cfi_def_cfa_offset 32 ; MIPS64R5EB-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64R5EB-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64R5EB-NEXT: .cfi_offset 31, -8 ; MIPS64R5EB-NEXT: .cfi_offset 28, -16 ; MIPS64R5EB-NEXT: lui $1, %hi(%neg(%gp_rel(calli16_2))) ; MIPS64R5EB-NEXT: daddu $1, $1, $25 ; MIPS64R5EB-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(calli16_2))) ; MIPS64R5EB-NEXT: lui $1, 6 ; MIPS64R5EB-NEXT: ori $4, $1, 7 ; MIPS64R5EB-NEXT: lui $1, 12 ; MIPS64R5EB-NEXT: ori $5, $1, 8 ; MIPS64R5EB-NEXT: ld $25, %call16(i16_2)($gp) ; MIPS64R5EB-NEXT: jalr $25 ; MIPS64R5EB-NEXT: nop ; MIPS64R5EB-NEXT: ld $1, %got_disp(gv2i16)($gp) ; MIPS64R5EB-NEXT: sw $2, 0($1) ; MIPS64R5EB-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64R5EB-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64R5EB-NEXT: daddiu $sp, $sp, 32 ; MIPS64R5EB-NEXT: jr $ra ; MIPS64R5EB-NEXT: nop ; ; MIPS32EL-LABEL: calli16_2: ; MIPS32EL: # %bb.0: # %entry ; MIPS32EL-NEXT: addiu $sp, $sp, -24 ; MIPS32EL-NEXT: .cfi_def_cfa_offset 24 ; MIPS32EL-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; MIPS32EL-NEXT: .cfi_offset 31, -4 ; MIPS32EL-NEXT: lui $1, 7 ; MIPS32EL-NEXT: ori $4, $1, 6 ; MIPS32EL-NEXT: lui $1, 8 ; MIPS32EL-NEXT: ori $5, $1, 12 ; MIPS32EL-NEXT: jal i16_2 ; MIPS32EL-NEXT: nop ; MIPS32EL-NEXT: lui $1, %hi(gv2i16) ; MIPS32EL-NEXT: sw $2, %lo(gv2i16)($1) ; MIPS32EL-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; MIPS32EL-NEXT: addiu $sp, $sp, 24 ; MIPS32EL-NEXT: jr $ra ; MIPS32EL-NEXT: nop ; ; MIPS64EL-LABEL: calli16_2: ; MIPS64EL: # %bb.0: # %entry ; MIPS64EL-NEXT: daddiu $sp, $sp, -16 ; MIPS64EL-NEXT: .cfi_def_cfa_offset 16 ; MIPS64EL-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; MIPS64EL-NEXT: sd $gp, 0($sp) # 8-byte Folded Spill ; MIPS64EL-NEXT: .cfi_offset 31, -8 ; MIPS64EL-NEXT: .cfi_offset 28, -16 ; MIPS64EL-NEXT: lui $1, %hi(%neg(%gp_rel(calli16_2))) ; MIPS64EL-NEXT: daddu $1, $1, $25 ; MIPS64EL-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(calli16_2))) ; MIPS64EL-NEXT: lui $1, 7 ; MIPS64EL-NEXT: ori $4, $1, 6 ; MIPS64EL-NEXT: lui $1, 8 ; MIPS64EL-NEXT: ori $5, $1, 12 ; MIPS64EL-NEXT: ld $25, %call16(i16_2)($gp) ; MIPS64EL-NEXT: jalr $25 ; MIPS64EL-NEXT: nop ; MIPS64EL-NEXT: ld $1, %got_disp(gv2i16)($gp) ; MIPS64EL-NEXT: sw $2, 0($1) ; MIPS64EL-NEXT: ld $gp, 0($sp) # 8-byte Folded Reload ; MIPS64EL-NEXT: ld $ra, 8($sp) # 8-byte Folded Reload ; MIPS64EL-NEXT: daddiu $sp, $sp, 16 ; MIPS64EL-NEXT: jr $ra ; MIPS64EL-NEXT: nop ; ; MIPS32R5EL-LABEL: calli16_2: ; MIPS32R5EL: # %bb.0: # %entry ; MIPS32R5EL-NEXT: addiu $sp, $sp, -32 ; MIPS32R5EL-NEXT: .cfi_def_cfa_offset 32 ; MIPS32R5EL-NEXT: sw $ra, 28($sp) # 4-byte Folded Spill ; MIPS32R5EL-NEXT: .cfi_offset 31, -4 ; MIPS32R5EL-NEXT: lui $1, 7 ; MIPS32R5EL-NEXT: ori $4, $1, 6 ; MIPS32R5EL-NEXT: lui $1, 8 ; MIPS32R5EL-NEXT: ori $5, $1, 12 ; MIPS32R5EL-NEXT: jal i16_2 ; MIPS32R5EL-NEXT: nop ; MIPS32R5EL-NEXT: lui $1, %hi(gv2i16) ; MIPS32R5EL-NEXT: sw $2, %lo(gv2i16)($1) ; MIPS32R5EL-NEXT: lw $ra, 28($sp) # 4-byte Folded Reload ; MIPS32R5EL-NEXT: addiu $sp, $sp, 32 ; MIPS32R5EL-NEXT: jr $ra ; MIPS32R5EL-NEXT: nop ; ; MIPS64R5EL-LABEL: calli16_2: ; MIPS64R5EL: # %bb.0: # %entry ; MIPS64R5EL-NEXT: daddiu $sp, $sp, -32 ; MIPS64R5EL-NEXT: .cfi_def_cfa_offset 32 ; MIPS64R5EL-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64R5EL-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64R5EL-NEXT: .cfi_offset 31, -8 ; MIPS64R5EL-NEXT: .cfi_offset 28, -16 ; MIPS64R5EL-NEXT: lui $1, %hi(%neg(%gp_rel(calli16_2))) ; MIPS64R5EL-NEXT: daddu $1, $1, $25 ; MIPS64R5EL-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(calli16_2))) ; MIPS64R5EL-NEXT: lui $1, 7 ; MIPS64R5EL-NEXT: ori $4, $1, 6 ; MIPS64R5EL-NEXT: lui $1, 8 ; MIPS64R5EL-NEXT: ori $5, $1, 12 ; MIPS64R5EL-NEXT: ld $25, %call16(i16_2)($gp) ; MIPS64R5EL-NEXT: jalr $25 ; MIPS64R5EL-NEXT: nop ; MIPS64R5EL-NEXT: ld $1, %got_disp(gv2i16)($gp) ; MIPS64R5EL-NEXT: sw $2, 0($1) ; MIPS64R5EL-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64R5EL-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64R5EL-NEXT: daddiu $sp, $sp, 32 ; MIPS64R5EL-NEXT: jr $ra ; MIPS64R5EL-NEXT: nop entry: %0 = call <2 x i16> @i16_2(<2 x i16> , <2 x i16> ) store <2 x i16> %0, <2 x i16> * @gv2i16 ret void } define void @calli16_4() { ; MIPS32EB-LABEL: calli16_4: ; MIPS32EB: # %bb.0: # %entry ; MIPS32EB-NEXT: addiu $sp, $sp, -24 ; MIPS32EB-NEXT: .cfi_def_cfa_offset 24 ; MIPS32EB-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; MIPS32EB-NEXT: .cfi_offset 31, -4 ; MIPS32EB-NEXT: lui $1, 6 ; MIPS32EB-NEXT: ori $4, $1, 7 ; MIPS32EB-NEXT: lui $1, 12 ; MIPS32EB-NEXT: ori $6, $1, 8 ; MIPS32EB-NEXT: lui $1, 9 ; MIPS32EB-NEXT: ori $5, $1, 10 ; MIPS32EB-NEXT: move $7, $5 ; MIPS32EB-NEXT: jal i16_4 ; MIPS32EB-NEXT: nop ; MIPS32EB-NEXT: lui $1, %hi(gv4i16) ; MIPS32EB-NEXT: addiu $4, $1, %lo(gv4i16) ; MIPS32EB-NEXT: sw $3, 4($4) ; MIPS32EB-NEXT: sw $2, %lo(gv4i16)($1) ; MIPS32EB-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; MIPS32EB-NEXT: addiu $sp, $sp, 24 ; MIPS32EB-NEXT: jr $ra ; MIPS32EB-NEXT: nop ; ; MIPS64EB-LABEL: calli16_4: ; MIPS64EB: # %bb.0: # %entry ; MIPS64EB-NEXT: daddiu $sp, $sp, -16 ; MIPS64EB-NEXT: .cfi_def_cfa_offset 16 ; MIPS64EB-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; MIPS64EB-NEXT: sd $gp, 0($sp) # 8-byte Folded Spill ; MIPS64EB-NEXT: .cfi_offset 31, -8 ; MIPS64EB-NEXT: .cfi_offset 28, -16 ; MIPS64EB-NEXT: lui $1, %hi(%neg(%gp_rel(calli16_4))) ; MIPS64EB-NEXT: daddu $1, $1, $25 ; MIPS64EB-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(calli16_4))) ; MIPS64EB-NEXT: lui $1, 6 ; MIPS64EB-NEXT: daddiu $1, $1, 7 ; MIPS64EB-NEXT: dsll $1, $1, 16 ; MIPS64EB-NEXT: daddiu $1, $1, 9 ; MIPS64EB-NEXT: dsll $1, $1, 16 ; MIPS64EB-NEXT: daddiu $4, $1, 10 ; MIPS64EB-NEXT: lui $1, 2 ; MIPS64EB-NEXT: daddiu $1, $1, -32767 ; MIPS64EB-NEXT: dsll $1, $1, 19 ; MIPS64EB-NEXT: daddiu $1, $1, 9 ; MIPS64EB-NEXT: dsll $1, $1, 16 ; MIPS64EB-NEXT: daddiu $5, $1, 10 ; MIPS64EB-NEXT: ld $25, %call16(i16_4)($gp) ; MIPS64EB-NEXT: jalr $25 ; MIPS64EB-NEXT: nop ; MIPS64EB-NEXT: ld $1, %got_disp(gv4i16)($gp) ; MIPS64EB-NEXT: sd $2, 0($1) ; MIPS64EB-NEXT: ld $gp, 0($sp) # 8-byte Folded Reload ; MIPS64EB-NEXT: ld $ra, 8($sp) # 8-byte Folded Reload ; MIPS64EB-NEXT: daddiu $sp, $sp, 16 ; MIPS64EB-NEXT: jr $ra ; MIPS64EB-NEXT: nop ; ; MIPS32R5EB-LABEL: calli16_4: ; MIPS32R5EB: # %bb.0: # %entry ; MIPS32R5EB-NEXT: addiu $sp, $sp, -24 ; MIPS32R5EB-NEXT: .cfi_def_cfa_offset 24 ; MIPS32R5EB-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; MIPS32R5EB-NEXT: .cfi_offset 31, -4 ; MIPS32R5EB-NEXT: lui $1, 6 ; MIPS32R5EB-NEXT: ori $4, $1, 7 ; MIPS32R5EB-NEXT: lui $1, 12 ; MIPS32R5EB-NEXT: ori $6, $1, 8 ; MIPS32R5EB-NEXT: lui $1, 9 ; MIPS32R5EB-NEXT: ori $5, $1, 10 ; MIPS32R5EB-NEXT: move $7, $5 ; MIPS32R5EB-NEXT: jal i16_4 ; MIPS32R5EB-NEXT: nop ; MIPS32R5EB-NEXT: lui $1, %hi(gv4i16) ; MIPS32R5EB-NEXT: addiu $4, $1, %lo(gv4i16) ; MIPS32R5EB-NEXT: sw $3, 4($4) ; MIPS32R5EB-NEXT: sw $2, %lo(gv4i16)($1) ; MIPS32R5EB-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; MIPS32R5EB-NEXT: addiu $sp, $sp, 24 ; MIPS32R5EB-NEXT: jr $ra ; MIPS32R5EB-NEXT: nop ; ; MIPS64R5EB-LABEL: calli16_4: ; MIPS64R5EB: # %bb.0: # %entry ; MIPS64R5EB-NEXT: daddiu $sp, $sp, -32 ; MIPS64R5EB-NEXT: .cfi_def_cfa_offset 32 ; MIPS64R5EB-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64R5EB-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64R5EB-NEXT: .cfi_offset 31, -8 ; MIPS64R5EB-NEXT: .cfi_offset 28, -16 ; MIPS64R5EB-NEXT: lui $1, %hi(%neg(%gp_rel(calli16_4))) ; MIPS64R5EB-NEXT: daddu $1, $1, $25 ; MIPS64R5EB-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(calli16_4))) ; MIPS64R5EB-NEXT: lui $1, 6 ; MIPS64R5EB-NEXT: daddiu $1, $1, 7 ; MIPS64R5EB-NEXT: dsll $1, $1, 16 ; MIPS64R5EB-NEXT: daddiu $1, $1, 9 ; MIPS64R5EB-NEXT: dsll $1, $1, 16 ; MIPS64R5EB-NEXT: daddiu $4, $1, 10 ; MIPS64R5EB-NEXT: lui $1, 2 ; MIPS64R5EB-NEXT: daddiu $1, $1, -32767 ; MIPS64R5EB-NEXT: dsll $1, $1, 19 ; MIPS64R5EB-NEXT: daddiu $1, $1, 9 ; MIPS64R5EB-NEXT: dsll $1, $1, 16 ; MIPS64R5EB-NEXT: daddiu $5, $1, 10 ; MIPS64R5EB-NEXT: ld $25, %call16(i16_4)($gp) ; MIPS64R5EB-NEXT: jalr $25 ; MIPS64R5EB-NEXT: nop ; MIPS64R5EB-NEXT: ld $1, %got_disp(gv4i16)($gp) ; MIPS64R5EB-NEXT: sd $2, 0($1) ; MIPS64R5EB-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64R5EB-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64R5EB-NEXT: daddiu $sp, $sp, 32 ; MIPS64R5EB-NEXT: jr $ra ; MIPS64R5EB-NEXT: nop ; ; MIPS32EL-LABEL: calli16_4: ; MIPS32EL: # %bb.0: # %entry ; MIPS32EL-NEXT: addiu $sp, $sp, -24 ; MIPS32EL-NEXT: .cfi_def_cfa_offset 24 ; MIPS32EL-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; MIPS32EL-NEXT: .cfi_offset 31, -4 ; MIPS32EL-NEXT: lui $1, 7 ; MIPS32EL-NEXT: ori $4, $1, 6 ; MIPS32EL-NEXT: lui $1, 8 ; MIPS32EL-NEXT: ori $6, $1, 12 ; MIPS32EL-NEXT: lui $1, 10 ; MIPS32EL-NEXT: ori $5, $1, 9 ; MIPS32EL-NEXT: move $7, $5 ; MIPS32EL-NEXT: jal i16_4 ; MIPS32EL-NEXT: nop ; MIPS32EL-NEXT: lui $1, %hi(gv4i16) ; MIPS32EL-NEXT: addiu $4, $1, %lo(gv4i16) ; MIPS32EL-NEXT: sw $3, 4($4) ; MIPS32EL-NEXT: sw $2, %lo(gv4i16)($1) ; MIPS32EL-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; MIPS32EL-NEXT: addiu $sp, $sp, 24 ; MIPS32EL-NEXT: jr $ra ; MIPS32EL-NEXT: nop ; ; MIPS64EL-LABEL: calli16_4: ; MIPS64EL: # %bb.0: # %entry ; MIPS64EL-NEXT: daddiu $sp, $sp, -16 ; MIPS64EL-NEXT: .cfi_def_cfa_offset 16 ; MIPS64EL-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; MIPS64EL-NEXT: sd $gp, 0($sp) # 8-byte Folded Spill ; MIPS64EL-NEXT: .cfi_offset 31, -8 ; MIPS64EL-NEXT: .cfi_offset 28, -16 ; MIPS64EL-NEXT: lui $1, %hi(%neg(%gp_rel(calli16_4))) ; MIPS64EL-NEXT: daddu $1, $1, $25 ; MIPS64EL-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(calli16_4))) ; MIPS64EL-NEXT: lui $1, 10 ; MIPS64EL-NEXT: daddiu $1, $1, 9 ; MIPS64EL-NEXT: dsll $1, $1, 16 ; MIPS64EL-NEXT: daddiu $1, $1, 7 ; MIPS64EL-NEXT: dsll $1, $1, 16 ; MIPS64EL-NEXT: daddiu $4, $1, 6 ; MIPS64EL-NEXT: lui $1, 1 ; MIPS64EL-NEXT: daddiu $1, $1, 16385 ; MIPS64EL-NEXT: dsll $1, $1, 16 ; MIPS64EL-NEXT: daddiu $1, $1, 8193 ; MIPS64EL-NEXT: dsll $1, $1, 19 ; MIPS64EL-NEXT: daddiu $5, $1, 12 ; MIPS64EL-NEXT: ld $25, %call16(i16_4)($gp) ; MIPS64EL-NEXT: jalr $25 ; MIPS64EL-NEXT: nop ; MIPS64EL-NEXT: ld $1, %got_disp(gv4i16)($gp) ; MIPS64EL-NEXT: sd $2, 0($1) ; MIPS64EL-NEXT: ld $gp, 0($sp) # 8-byte Folded Reload ; MIPS64EL-NEXT: ld $ra, 8($sp) # 8-byte Folded Reload ; MIPS64EL-NEXT: daddiu $sp, $sp, 16 ; MIPS64EL-NEXT: jr $ra ; MIPS64EL-NEXT: nop ; ; MIPS32R5EL-LABEL: calli16_4: ; MIPS32R5EL: # %bb.0: # %entry ; MIPS32R5EL-NEXT: addiu $sp, $sp, -24 ; MIPS32R5EL-NEXT: .cfi_def_cfa_offset 24 ; MIPS32R5EL-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; MIPS32R5EL-NEXT: .cfi_offset 31, -4 ; MIPS32R5EL-NEXT: lui $1, 7 ; MIPS32R5EL-NEXT: ori $4, $1, 6 ; MIPS32R5EL-NEXT: lui $1, 8 ; MIPS32R5EL-NEXT: ori $6, $1, 12 ; MIPS32R5EL-NEXT: lui $1, 10 ; MIPS32R5EL-NEXT: ori $5, $1, 9 ; MIPS32R5EL-NEXT: move $7, $5 ; MIPS32R5EL-NEXT: jal i16_4 ; MIPS32R5EL-NEXT: nop ; MIPS32R5EL-NEXT: lui $1, %hi(gv4i16) ; MIPS32R5EL-NEXT: addiu $4, $1, %lo(gv4i16) ; MIPS32R5EL-NEXT: sw $3, 4($4) ; MIPS32R5EL-NEXT: sw $2, %lo(gv4i16)($1) ; MIPS32R5EL-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; MIPS32R5EL-NEXT: addiu $sp, $sp, 24 ; MIPS32R5EL-NEXT: jr $ra ; MIPS32R5EL-NEXT: nop ; ; MIPS64R5EL-LABEL: calli16_4: ; MIPS64R5EL: # %bb.0: # %entry ; MIPS64R5EL-NEXT: daddiu $sp, $sp, -32 ; MIPS64R5EL-NEXT: .cfi_def_cfa_offset 32 ; MIPS64R5EL-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64R5EL-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64R5EL-NEXT: .cfi_offset 31, -8 ; MIPS64R5EL-NEXT: .cfi_offset 28, -16 ; MIPS64R5EL-NEXT: lui $1, %hi(%neg(%gp_rel(calli16_4))) ; MIPS64R5EL-NEXT: daddu $1, $1, $25 ; MIPS64R5EL-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(calli16_4))) ; MIPS64R5EL-NEXT: lui $1, 10 ; MIPS64R5EL-NEXT: daddiu $1, $1, 9 ; MIPS64R5EL-NEXT: dsll $1, $1, 16 ; MIPS64R5EL-NEXT: daddiu $1, $1, 7 ; MIPS64R5EL-NEXT: dsll $1, $1, 16 ; MIPS64R5EL-NEXT: daddiu $4, $1, 6 ; MIPS64R5EL-NEXT: lui $1, 1 ; MIPS64R5EL-NEXT: daddiu $1, $1, 16385 ; MIPS64R5EL-NEXT: dsll $1, $1, 16 ; MIPS64R5EL-NEXT: daddiu $1, $1, 8193 ; MIPS64R5EL-NEXT: dsll $1, $1, 19 ; MIPS64R5EL-NEXT: daddiu $5, $1, 12 ; MIPS64R5EL-NEXT: ld $25, %call16(i16_4)($gp) ; MIPS64R5EL-NEXT: jalr $25 ; MIPS64R5EL-NEXT: nop ; MIPS64R5EL-NEXT: ld $1, %got_disp(gv4i16)($gp) ; MIPS64R5EL-NEXT: sd $2, 0($1) ; MIPS64R5EL-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64R5EL-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64R5EL-NEXT: daddiu $sp, $sp, 32 ; MIPS64R5EL-NEXT: jr $ra ; MIPS64R5EL-NEXT: nop entry: %0 = call <4 x i16> @i16_4(<4 x i16> , <4 x i16> ) store <4 x i16> %0, <4 x i16> * @gv4i16 ret void } define void @calli16_8() { ; MIPS32EB-LABEL: calli16_8: ; MIPS32EB: # %bb.0: # %entry ; MIPS32EB-NEXT: addiu $sp, $sp, -40 ; MIPS32EB-NEXT: .cfi_def_cfa_offset 40 ; MIPS32EB-NEXT: sw $ra, 36($sp) # 4-byte Folded Spill ; MIPS32EB-NEXT: .cfi_offset 31, -4 ; MIPS32EB-NEXT: lui $1, 9 ; MIPS32EB-NEXT: ori $5, $1, 10 ; MIPS32EB-NEXT: sw $5, 28($sp) ; MIPS32EB-NEXT: lui $1, 12 ; MIPS32EB-NEXT: ori $1, $1, 8 ; MIPS32EB-NEXT: sw $1, 24($sp) ; MIPS32EB-NEXT: sw $5, 20($sp) ; MIPS32EB-NEXT: lui $1, 6 ; MIPS32EB-NEXT: ori $4, $1, 7 ; MIPS32EB-NEXT: sw $4, 16($sp) ; MIPS32EB-NEXT: move $6, $4 ; MIPS32EB-NEXT: move $7, $5 ; MIPS32EB-NEXT: jal i16_8 ; MIPS32EB-NEXT: nop ; MIPS32EB-NEXT: lui $1, %hi(gv8i16) ; MIPS32EB-NEXT: addiu $6, $1, %lo(gv8i16) ; MIPS32EB-NEXT: sw $5, 12($6) ; MIPS32EB-NEXT: sw $4, 8($6) ; MIPS32EB-NEXT: sw $3, 4($6) ; MIPS32EB-NEXT: sw $2, %lo(gv8i16)($1) ; MIPS32EB-NEXT: lw $ra, 36($sp) # 4-byte Folded Reload ; MIPS32EB-NEXT: addiu $sp, $sp, 40 ; MIPS32EB-NEXT: jr $ra ; MIPS32EB-NEXT: nop ; ; MIPS64EB-LABEL: calli16_8: ; MIPS64EB: # %bb.0: # %entry ; MIPS64EB-NEXT: daddiu $sp, $sp, -16 ; MIPS64EB-NEXT: .cfi_def_cfa_offset 16 ; MIPS64EB-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; MIPS64EB-NEXT: sd $gp, 0($sp) # 8-byte Folded Spill ; MIPS64EB-NEXT: .cfi_offset 31, -8 ; MIPS64EB-NEXT: .cfi_offset 28, -16 ; MIPS64EB-NEXT: lui $1, %hi(%neg(%gp_rel(calli16_8))) ; MIPS64EB-NEXT: daddu $1, $1, $25 ; MIPS64EB-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(calli16_8))) ; MIPS64EB-NEXT: lui $1, 6 ; MIPS64EB-NEXT: daddiu $1, $1, 7 ; MIPS64EB-NEXT: dsll $1, $1, 16 ; MIPS64EB-NEXT: daddiu $1, $1, 9 ; MIPS64EB-NEXT: dsll $1, $1, 16 ; MIPS64EB-NEXT: daddiu $4, $1, 10 ; MIPS64EB-NEXT: lui $1, 2 ; MIPS64EB-NEXT: daddiu $1, $1, -32767 ; MIPS64EB-NEXT: dsll $1, $1, 19 ; MIPS64EB-NEXT: daddiu $1, $1, 9 ; MIPS64EB-NEXT: dsll $1, $1, 16 ; MIPS64EB-NEXT: daddiu $7, $1, 10 ; MIPS64EB-NEXT: ld $25, %call16(i16_8)($gp) ; MIPS64EB-NEXT: move $5, $4 ; MIPS64EB-NEXT: move $6, $4 ; MIPS64EB-NEXT: jalr $25 ; MIPS64EB-NEXT: nop ; MIPS64EB-NEXT: ld $1, %got_disp(gv8i16)($gp) ; MIPS64EB-NEXT: sd $3, 8($1) ; MIPS64EB-NEXT: sd $2, 0($1) ; MIPS64EB-NEXT: ld $gp, 0($sp) # 8-byte Folded Reload ; MIPS64EB-NEXT: ld $ra, 8($sp) # 8-byte Folded Reload ; MIPS64EB-NEXT: daddiu $sp, $sp, 16 ; MIPS64EB-NEXT: jr $ra ; MIPS64EB-NEXT: nop ; ; MIPS32R5EB-LABEL: calli16_8: ; MIPS32R5EB: # %bb.0: # %entry ; MIPS32R5EB-NEXT: addiu $sp, $sp, -40 ; MIPS32R5EB-NEXT: .cfi_def_cfa_offset 40 ; MIPS32R5EB-NEXT: sw $ra, 36($sp) # 4-byte Folded Spill ; MIPS32R5EB-NEXT: .cfi_offset 31, -4 ; MIPS32R5EB-NEXT: lui $1, 6 ; MIPS32R5EB-NEXT: ori $1, $1, 7 ; MIPS32R5EB-NEXT: lui $2, 9 ; MIPS32R5EB-NEXT: ori $2, $2, 10 ; MIPS32R5EB-NEXT: fill.w $w0, $2 ; MIPS32R5EB-NEXT: insert.w $w0[1], $1 ; MIPS32R5EB-NEXT: splati.d $w0, $w0[0] ; MIPS32R5EB-NEXT: copy_s.w $4, $w0[0] ; MIPS32R5EB-NEXT: copy_s.w $5, $w0[1] ; MIPS32R5EB-NEXT: copy_s.w $6, $w0[2] ; MIPS32R5EB-NEXT: copy_s.w $7, $w0[3] ; MIPS32R5EB-NEXT: lui $1, %hi($CPI33_0) ; MIPS32R5EB-NEXT: addiu $1, $1, %lo($CPI33_0) ; MIPS32R5EB-NEXT: ld.w $w0, 0($1) ; MIPS32R5EB-NEXT: copy_s.w $1, $w0[0] ; MIPS32R5EB-NEXT: copy_s.w $2, $w0[1] ; MIPS32R5EB-NEXT: copy_s.w $3, $w0[2] ; MIPS32R5EB-NEXT: copy_s.w $8, $w0[3] ; MIPS32R5EB-NEXT: sw $8, 28($sp) ; MIPS32R5EB-NEXT: sw $3, 24($sp) ; MIPS32R5EB-NEXT: sw $2, 20($sp) ; MIPS32R5EB-NEXT: sw $1, 16($sp) ; MIPS32R5EB-NEXT: jal i16_8 ; MIPS32R5EB-NEXT: nop ; MIPS32R5EB-NEXT: lui $1, %hi(gv8i16) ; MIPS32R5EB-NEXT: addiu $1, $1, %lo(gv8i16) ; MIPS32R5EB-NEXT: ldi.b $w0, 0 ; MIPS32R5EB-NEXT: insert.w $w0[0], $2 ; MIPS32R5EB-NEXT: insert.w $w0[1], $3 ; MIPS32R5EB-NEXT: insert.w $w0[2], $4 ; MIPS32R5EB-NEXT: insert.w $w0[3], $5 ; MIPS32R5EB-NEXT: st.w $w0, 0($1) ; MIPS32R5EB-NEXT: lw $ra, 36($sp) # 4-byte Folded Reload ; MIPS32R5EB-NEXT: addiu $sp, $sp, 40 ; MIPS32R5EB-NEXT: jr $ra ; MIPS32R5EB-NEXT: nop ; ; MIPS64R5EB-LABEL: calli16_8: ; MIPS64R5EB: # %bb.0: # %entry ; MIPS64R5EB-NEXT: daddiu $sp, $sp, -16 ; MIPS64R5EB-NEXT: .cfi_def_cfa_offset 16 ; MIPS64R5EB-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; MIPS64R5EB-NEXT: sd $gp, 0($sp) # 8-byte Folded Spill ; MIPS64R5EB-NEXT: .cfi_offset 31, -8 ; MIPS64R5EB-NEXT: .cfi_offset 28, -16 ; MIPS64R5EB-NEXT: lui $1, %hi(%neg(%gp_rel(calli16_8))) ; MIPS64R5EB-NEXT: daddu $1, $1, $25 ; MIPS64R5EB-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(calli16_8))) ; MIPS64R5EB-NEXT: lui $1, 9 ; MIPS64R5EB-NEXT: ori $1, $1, 10 ; MIPS64R5EB-NEXT: lui $2, 6 ; MIPS64R5EB-NEXT: ori $2, $2, 7 ; MIPS64R5EB-NEXT: dinsu $1, $2, 32, 32 ; MIPS64R5EB-NEXT: fill.d $w0, $1 ; MIPS64R5EB-NEXT: copy_s.d $4, $w0[0] ; MIPS64R5EB-NEXT: copy_s.d $5, $w0[1] ; MIPS64R5EB-NEXT: ld $1, %got_page(.LCPI33_0)($gp) ; MIPS64R5EB-NEXT: daddiu $1, $1, %got_ofst(.LCPI33_0) ; MIPS64R5EB-NEXT: ld.d $w0, 0($1) ; MIPS64R5EB-NEXT: copy_s.d $6, $w0[0] ; MIPS64R5EB-NEXT: copy_s.d $7, $w0[1] ; MIPS64R5EB-NEXT: ld $25, %call16(i16_8)($gp) ; MIPS64R5EB-NEXT: jalr $25 ; MIPS64R5EB-NEXT: nop ; MIPS64R5EB-NEXT: ldi.b $w0, 0 ; MIPS64R5EB-NEXT: ld $1, %got_disp(gv8i16)($gp) ; MIPS64R5EB-NEXT: insert.d $w0[0], $2 ; MIPS64R5EB-NEXT: insert.d $w0[1], $3 ; MIPS64R5EB-NEXT: st.d $w0, 0($1) ; MIPS64R5EB-NEXT: ld $gp, 0($sp) # 8-byte Folded Reload ; MIPS64R5EB-NEXT: ld $ra, 8($sp) # 8-byte Folded Reload ; MIPS64R5EB-NEXT: daddiu $sp, $sp, 16 ; MIPS64R5EB-NEXT: jr $ra ; MIPS64R5EB-NEXT: nop ; ; MIPS32EL-LABEL: calli16_8: ; MIPS32EL: # %bb.0: # %entry ; MIPS32EL-NEXT: addiu $sp, $sp, -40 ; MIPS32EL-NEXT: .cfi_def_cfa_offset 40 ; MIPS32EL-NEXT: sw $ra, 36($sp) # 4-byte Folded Spill ; MIPS32EL-NEXT: .cfi_offset 31, -4 ; MIPS32EL-NEXT: lui $1, 10 ; MIPS32EL-NEXT: ori $5, $1, 9 ; MIPS32EL-NEXT: sw $5, 28($sp) ; MIPS32EL-NEXT: lui $1, 8 ; MIPS32EL-NEXT: ori $1, $1, 12 ; MIPS32EL-NEXT: sw $1, 24($sp) ; MIPS32EL-NEXT: sw $5, 20($sp) ; MIPS32EL-NEXT: lui $1, 7 ; MIPS32EL-NEXT: ori $4, $1, 6 ; MIPS32EL-NEXT: sw $4, 16($sp) ; MIPS32EL-NEXT: move $6, $4 ; MIPS32EL-NEXT: move $7, $5 ; MIPS32EL-NEXT: jal i16_8 ; MIPS32EL-NEXT: nop ; MIPS32EL-NEXT: lui $1, %hi(gv8i16) ; MIPS32EL-NEXT: addiu $6, $1, %lo(gv8i16) ; MIPS32EL-NEXT: sw $5, 12($6) ; MIPS32EL-NEXT: sw $4, 8($6) ; MIPS32EL-NEXT: sw $3, 4($6) ; MIPS32EL-NEXT: sw $2, %lo(gv8i16)($1) ; MIPS32EL-NEXT: lw $ra, 36($sp) # 4-byte Folded Reload ; MIPS32EL-NEXT: addiu $sp, $sp, 40 ; MIPS32EL-NEXT: jr $ra ; MIPS32EL-NEXT: nop ; ; MIPS64EL-LABEL: calli16_8: ; MIPS64EL: # %bb.0: # %entry ; MIPS64EL-NEXT: daddiu $sp, $sp, -16 ; MIPS64EL-NEXT: .cfi_def_cfa_offset 16 ; MIPS64EL-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; MIPS64EL-NEXT: sd $gp, 0($sp) # 8-byte Folded Spill ; MIPS64EL-NEXT: .cfi_offset 31, -8 ; MIPS64EL-NEXT: .cfi_offset 28, -16 ; MIPS64EL-NEXT: lui $1, %hi(%neg(%gp_rel(calli16_8))) ; MIPS64EL-NEXT: daddu $1, $1, $25 ; MIPS64EL-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(calli16_8))) ; MIPS64EL-NEXT: lui $1, 10 ; MIPS64EL-NEXT: daddiu $1, $1, 9 ; MIPS64EL-NEXT: dsll $1, $1, 16 ; MIPS64EL-NEXT: daddiu $1, $1, 7 ; MIPS64EL-NEXT: dsll $1, $1, 16 ; MIPS64EL-NEXT: daddiu $4, $1, 6 ; MIPS64EL-NEXT: lui $1, 1 ; MIPS64EL-NEXT: daddiu $1, $1, 16385 ; MIPS64EL-NEXT: dsll $1, $1, 16 ; MIPS64EL-NEXT: daddiu $1, $1, 8193 ; MIPS64EL-NEXT: dsll $1, $1, 19 ; MIPS64EL-NEXT: daddiu $7, $1, 12 ; MIPS64EL-NEXT: ld $25, %call16(i16_8)($gp) ; MIPS64EL-NEXT: move $5, $4 ; MIPS64EL-NEXT: move $6, $4 ; MIPS64EL-NEXT: jalr $25 ; MIPS64EL-NEXT: nop ; MIPS64EL-NEXT: ld $1, %got_disp(gv8i16)($gp) ; MIPS64EL-NEXT: sd $3, 8($1) ; MIPS64EL-NEXT: sd $2, 0($1) ; MIPS64EL-NEXT: ld $gp, 0($sp) # 8-byte Folded Reload ; MIPS64EL-NEXT: ld $ra, 8($sp) # 8-byte Folded Reload ; MIPS64EL-NEXT: daddiu $sp, $sp, 16 ; MIPS64EL-NEXT: jr $ra ; MIPS64EL-NEXT: nop ; ; MIPS32R5EL-LABEL: calli16_8: ; MIPS32R5EL: # %bb.0: # %entry ; MIPS32R5EL-NEXT: addiu $sp, $sp, -40 ; MIPS32R5EL-NEXT: .cfi_def_cfa_offset 40 ; MIPS32R5EL-NEXT: sw $ra, 36($sp) # 4-byte Folded Spill ; MIPS32R5EL-NEXT: .cfi_offset 31, -4 ; MIPS32R5EL-NEXT: lui $1, 10 ; MIPS32R5EL-NEXT: ori $1, $1, 9 ; MIPS32R5EL-NEXT: lui $2, 7 ; MIPS32R5EL-NEXT: ori $2, $2, 6 ; MIPS32R5EL-NEXT: fill.w $w0, $2 ; MIPS32R5EL-NEXT: insert.w $w0[1], $1 ; MIPS32R5EL-NEXT: splati.d $w0, $w0[0] ; MIPS32R5EL-NEXT: copy_s.w $4, $w0[0] ; MIPS32R5EL-NEXT: copy_s.w $5, $w0[1] ; MIPS32R5EL-NEXT: copy_s.w $6, $w0[2] ; MIPS32R5EL-NEXT: copy_s.w $7, $w0[3] ; MIPS32R5EL-NEXT: lui $1, %hi($CPI33_0) ; MIPS32R5EL-NEXT: addiu $1, $1, %lo($CPI33_0) ; MIPS32R5EL-NEXT: ld.w $w0, 0($1) ; MIPS32R5EL-NEXT: copy_s.w $1, $w0[0] ; MIPS32R5EL-NEXT: copy_s.w $2, $w0[1] ; MIPS32R5EL-NEXT: copy_s.w $3, $w0[2] ; MIPS32R5EL-NEXT: copy_s.w $8, $w0[3] ; MIPS32R5EL-NEXT: sw $8, 28($sp) ; MIPS32R5EL-NEXT: sw $3, 24($sp) ; MIPS32R5EL-NEXT: sw $2, 20($sp) ; MIPS32R5EL-NEXT: sw $1, 16($sp) ; MIPS32R5EL-NEXT: jal i16_8 ; MIPS32R5EL-NEXT: nop ; MIPS32R5EL-NEXT: lui $1, %hi(gv8i16) ; MIPS32R5EL-NEXT: addiu $1, $1, %lo(gv8i16) ; MIPS32R5EL-NEXT: ldi.b $w0, 0 ; MIPS32R5EL-NEXT: insert.w $w0[0], $2 ; MIPS32R5EL-NEXT: insert.w $w0[1], $3 ; MIPS32R5EL-NEXT: insert.w $w0[2], $4 ; MIPS32R5EL-NEXT: insert.w $w0[3], $5 ; MIPS32R5EL-NEXT: st.w $w0, 0($1) ; MIPS32R5EL-NEXT: lw $ra, 36($sp) # 4-byte Folded Reload ; MIPS32R5EL-NEXT: addiu $sp, $sp, 40 ; MIPS32R5EL-NEXT: jr $ra ; MIPS32R5EL-NEXT: nop ; ; MIPS64R5EL-LABEL: calli16_8: ; MIPS64R5EL: # %bb.0: # %entry ; MIPS64R5EL-NEXT: daddiu $sp, $sp, -16 ; MIPS64R5EL-NEXT: .cfi_def_cfa_offset 16 ; MIPS64R5EL-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; MIPS64R5EL-NEXT: sd $gp, 0($sp) # 8-byte Folded Spill ; MIPS64R5EL-NEXT: .cfi_offset 31, -8 ; MIPS64R5EL-NEXT: .cfi_offset 28, -16 ; MIPS64R5EL-NEXT: lui $1, %hi(%neg(%gp_rel(calli16_8))) ; MIPS64R5EL-NEXT: daddu $1, $1, $25 ; MIPS64R5EL-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(calli16_8))) ; MIPS64R5EL-NEXT: lui $1, 7 ; MIPS64R5EL-NEXT: ori $1, $1, 6 ; MIPS64R5EL-NEXT: lui $2, 10 ; MIPS64R5EL-NEXT: ori $2, $2, 9 ; MIPS64R5EL-NEXT: dinsu $1, $2, 32, 32 ; MIPS64R5EL-NEXT: fill.d $w0, $1 ; MIPS64R5EL-NEXT: copy_s.d $4, $w0[0] ; MIPS64R5EL-NEXT: copy_s.d $5, $w0[1] ; MIPS64R5EL-NEXT: ld $1, %got_page(.LCPI33_0)($gp) ; MIPS64R5EL-NEXT: daddiu $1, $1, %got_ofst(.LCPI33_0) ; MIPS64R5EL-NEXT: ld.d $w0, 0($1) ; MIPS64R5EL-NEXT: copy_s.d $6, $w0[0] ; MIPS64R5EL-NEXT: copy_s.d $7, $w0[1] ; MIPS64R5EL-NEXT: ld $25, %call16(i16_8)($gp) ; MIPS64R5EL-NEXT: jalr $25 ; MIPS64R5EL-NEXT: nop ; MIPS64R5EL-NEXT: ldi.b $w0, 0 ; MIPS64R5EL-NEXT: ld $1, %got_disp(gv8i16)($gp) ; MIPS64R5EL-NEXT: insert.d $w0[0], $2 ; MIPS64R5EL-NEXT: insert.d $w0[1], $3 ; MIPS64R5EL-NEXT: st.d $w0, 0($1) ; MIPS64R5EL-NEXT: ld $gp, 0($sp) # 8-byte Folded Reload ; MIPS64R5EL-NEXT: ld $ra, 8($sp) # 8-byte Folded Reload ; MIPS64R5EL-NEXT: daddiu $sp, $sp, 16 ; MIPS64R5EL-NEXT: jr $ra ; MIPS64R5EL-NEXT: nop entry: %0 = call <8 x i16> @i16_8(<8 x i16> , <8 x i16> ) store <8 x i16> %0, <8 x i16> * @gv8i16 ret void } define void @calli32_2() { ; MIPS32-LABEL: calli32_2: ; MIPS32: # %bb.0: # %entry ; MIPS32-NEXT: addiu $sp, $sp, -24 ; MIPS32-NEXT: .cfi_def_cfa_offset 24 ; MIPS32-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; MIPS32-NEXT: .cfi_offset 31, -4 ; MIPS32-NEXT: addiu $4, $zero, 6 ; MIPS32-NEXT: addiu $5, $zero, 7 ; MIPS32-NEXT: addiu $6, $zero, 12 ; MIPS32-NEXT: addiu $7, $zero, 8 ; MIPS32-NEXT: jal i32_2 ; MIPS32-NEXT: nop ; MIPS32-NEXT: lui $1, %hi(gv2i32) ; MIPS32-NEXT: addiu $4, $1, %lo(gv2i32) ; MIPS32-NEXT: sw $3, 4($4) ; MIPS32-NEXT: sw $2, %lo(gv2i32)($1) ; MIPS32-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; MIPS32-NEXT: addiu $sp, $sp, 24 ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: nop ; ; MIPS64EB-LABEL: calli32_2: ; MIPS64EB: # %bb.0: # %entry ; MIPS64EB-NEXT: daddiu $sp, $sp, -16 ; MIPS64EB-NEXT: .cfi_def_cfa_offset 16 ; MIPS64EB-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; MIPS64EB-NEXT: sd $gp, 0($sp) # 8-byte Folded Spill ; MIPS64EB-NEXT: .cfi_offset 31, -8 ; MIPS64EB-NEXT: .cfi_offset 28, -16 ; MIPS64EB-NEXT: lui $1, %hi(%neg(%gp_rel(calli32_2))) ; MIPS64EB-NEXT: daddu $1, $1, $25 ; MIPS64EB-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(calli32_2))) ; MIPS64EB-NEXT: daddiu $1, $zero, 3 ; MIPS64EB-NEXT: dsll $2, $1, 33 ; MIPS64EB-NEXT: daddiu $4, $2, 7 ; MIPS64EB-NEXT: dsll $1, $1, 34 ; MIPS64EB-NEXT: daddiu $5, $1, 8 ; MIPS64EB-NEXT: ld $25, %call16(i32_2)($gp) ; MIPS64EB-NEXT: jalr $25 ; MIPS64EB-NEXT: nop ; MIPS64EB-NEXT: ld $1, %got_disp(gv2i32)($gp) ; MIPS64EB-NEXT: sd $2, 0($1) ; MIPS64EB-NEXT: ld $gp, 0($sp) # 8-byte Folded Reload ; MIPS64EB-NEXT: ld $ra, 8($sp) # 8-byte Folded Reload ; MIPS64EB-NEXT: daddiu $sp, $sp, 16 ; MIPS64EB-NEXT: jr $ra ; MIPS64EB-NEXT: nop ; ; MIPS32R5-LABEL: calli32_2: ; MIPS32R5: # %bb.0: # %entry ; MIPS32R5-NEXT: addiu $sp, $sp, -24 ; MIPS32R5-NEXT: .cfi_def_cfa_offset 24 ; MIPS32R5-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; MIPS32R5-NEXT: .cfi_offset 31, -4 ; MIPS32R5-NEXT: addiu $4, $zero, 6 ; MIPS32R5-NEXT: addiu $5, $zero, 7 ; MIPS32R5-NEXT: addiu $6, $zero, 12 ; MIPS32R5-NEXT: addiu $7, $zero, 8 ; MIPS32R5-NEXT: jal i32_2 ; MIPS32R5-NEXT: nop ; MIPS32R5-NEXT: lui $1, %hi(gv2i32) ; MIPS32R5-NEXT: addiu $4, $1, %lo(gv2i32) ; MIPS32R5-NEXT: sw $3, 4($4) ; MIPS32R5-NEXT: sw $2, %lo(gv2i32)($1) ; MIPS32R5-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; MIPS32R5-NEXT: addiu $sp, $sp, 24 ; MIPS32R5-NEXT: jr $ra ; MIPS32R5-NEXT: nop ; ; MIPS64R5EB-LABEL: calli32_2: ; MIPS64R5EB: # %bb.0: # %entry ; MIPS64R5EB-NEXT: daddiu $sp, $sp, -32 ; MIPS64R5EB-NEXT: .cfi_def_cfa_offset 32 ; MIPS64R5EB-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64R5EB-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64R5EB-NEXT: .cfi_offset 31, -8 ; MIPS64R5EB-NEXT: .cfi_offset 28, -16 ; MIPS64R5EB-NEXT: lui $1, %hi(%neg(%gp_rel(calli32_2))) ; MIPS64R5EB-NEXT: daddu $1, $1, $25 ; MIPS64R5EB-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(calli32_2))) ; MIPS64R5EB-NEXT: daddiu $1, $zero, 3 ; MIPS64R5EB-NEXT: dsll $2, $1, 33 ; MIPS64R5EB-NEXT: daddiu $4, $2, 7 ; MIPS64R5EB-NEXT: dsll $1, $1, 34 ; MIPS64R5EB-NEXT: daddiu $5, $1, 8 ; MIPS64R5EB-NEXT: ld $25, %call16(i32_2)($gp) ; MIPS64R5EB-NEXT: jalr $25 ; MIPS64R5EB-NEXT: nop ; MIPS64R5EB-NEXT: ld $1, %got_disp(gv2i32)($gp) ; MIPS64R5EB-NEXT: sd $2, 0($1) ; MIPS64R5EB-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64R5EB-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64R5EB-NEXT: daddiu $sp, $sp, 32 ; MIPS64R5EB-NEXT: jr $ra ; MIPS64R5EB-NEXT: nop ; ; MIPS64EL-LABEL: calli32_2: ; MIPS64EL: # %bb.0: # %entry ; MIPS64EL-NEXT: daddiu $sp, $sp, -16 ; MIPS64EL-NEXT: .cfi_def_cfa_offset 16 ; MIPS64EL-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; MIPS64EL-NEXT: sd $gp, 0($sp) # 8-byte Folded Spill ; MIPS64EL-NEXT: .cfi_offset 31, -8 ; MIPS64EL-NEXT: .cfi_offset 28, -16 ; MIPS64EL-NEXT: lui $1, %hi(%neg(%gp_rel(calli32_2))) ; MIPS64EL-NEXT: daddu $1, $1, $25 ; MIPS64EL-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(calli32_2))) ; MIPS64EL-NEXT: daddiu $1, $zero, 7 ; MIPS64EL-NEXT: dsll $1, $1, 32 ; MIPS64EL-NEXT: daddiu $4, $1, 6 ; MIPS64EL-NEXT: daddiu $1, $zero, 1 ; MIPS64EL-NEXT: dsll $1, $1, 35 ; MIPS64EL-NEXT: daddiu $5, $1, 12 ; MIPS64EL-NEXT: ld $25, %call16(i32_2)($gp) ; MIPS64EL-NEXT: jalr $25 ; MIPS64EL-NEXT: nop ; MIPS64EL-NEXT: ld $1, %got_disp(gv2i32)($gp) ; MIPS64EL-NEXT: sd $2, 0($1) ; MIPS64EL-NEXT: ld $gp, 0($sp) # 8-byte Folded Reload ; MIPS64EL-NEXT: ld $ra, 8($sp) # 8-byte Folded Reload ; MIPS64EL-NEXT: daddiu $sp, $sp, 16 ; MIPS64EL-NEXT: jr $ra ; MIPS64EL-NEXT: nop ; ; MIPS64R5EL-LABEL: calli32_2: ; MIPS64R5EL: # %bb.0: # %entry ; MIPS64R5EL-NEXT: daddiu $sp, $sp, -32 ; MIPS64R5EL-NEXT: .cfi_def_cfa_offset 32 ; MIPS64R5EL-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64R5EL-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64R5EL-NEXT: .cfi_offset 31, -8 ; MIPS64R5EL-NEXT: .cfi_offset 28, -16 ; MIPS64R5EL-NEXT: lui $1, %hi(%neg(%gp_rel(calli32_2))) ; MIPS64R5EL-NEXT: daddu $1, $1, $25 ; MIPS64R5EL-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(calli32_2))) ; MIPS64R5EL-NEXT: daddiu $1, $zero, 7 ; MIPS64R5EL-NEXT: dsll $1, $1, 32 ; MIPS64R5EL-NEXT: daddiu $4, $1, 6 ; MIPS64R5EL-NEXT: daddiu $1, $zero, 1 ; MIPS64R5EL-NEXT: dsll $1, $1, 35 ; MIPS64R5EL-NEXT: daddiu $5, $1, 12 ; MIPS64R5EL-NEXT: ld $25, %call16(i32_2)($gp) ; MIPS64R5EL-NEXT: jalr $25 ; MIPS64R5EL-NEXT: nop ; MIPS64R5EL-NEXT: ld $1, %got_disp(gv2i32)($gp) ; MIPS64R5EL-NEXT: sd $2, 0($1) ; MIPS64R5EL-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64R5EL-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64R5EL-NEXT: daddiu $sp, $sp, 32 ; MIPS64R5EL-NEXT: jr $ra ; MIPS64R5EL-NEXT: nop entry: %0 = call <2 x i32> @i32_2(<2 x i32> , <2 x i32> ) store <2 x i32> %0, <2 x i32> * @gv2i32 ret void } define void @calli32_4() { ; MIPS32-LABEL: calli32_4: ; MIPS32: # %bb.0: # %entry ; MIPS32-NEXT: addiu $sp, $sp, -40 ; MIPS32-NEXT: .cfi_def_cfa_offset 40 ; MIPS32-NEXT: sw $ra, 36($sp) # 4-byte Folded Spill ; MIPS32-NEXT: .cfi_offset 31, -4 ; MIPS32-NEXT: addiu $1, $zero, 9 ; MIPS32-NEXT: addiu $2, $zero, 10 ; MIPS32-NEXT: sw $2, 28($sp) ; MIPS32-NEXT: sw $1, 24($sp) ; MIPS32-NEXT: addiu $1, $zero, 8 ; MIPS32-NEXT: sw $1, 20($sp) ; MIPS32-NEXT: addiu $1, $zero, 12 ; MIPS32-NEXT: sw $1, 16($sp) ; MIPS32-NEXT: addiu $4, $zero, 6 ; MIPS32-NEXT: addiu $5, $zero, 7 ; MIPS32-NEXT: addiu $6, $zero, 9 ; MIPS32-NEXT: addiu $7, $zero, 10 ; MIPS32-NEXT: jal i32_4 ; MIPS32-NEXT: nop ; MIPS32-NEXT: lui $1, %hi(gv4i32) ; MIPS32-NEXT: addiu $6, $1, %lo(gv4i32) ; MIPS32-NEXT: sw $5, 12($6) ; MIPS32-NEXT: sw $4, 8($6) ; MIPS32-NEXT: sw $3, 4($6) ; MIPS32-NEXT: sw $2, %lo(gv4i32)($1) ; MIPS32-NEXT: lw $ra, 36($sp) # 4-byte Folded Reload ; MIPS32-NEXT: addiu $sp, $sp, 40 ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: nop ; ; MIPS64EB-LABEL: calli32_4: ; MIPS64EB: # %bb.0: # %entry ; MIPS64EB-NEXT: daddiu $sp, $sp, -16 ; MIPS64EB-NEXT: .cfi_def_cfa_offset 16 ; MIPS64EB-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; MIPS64EB-NEXT: sd $gp, 0($sp) # 8-byte Folded Spill ; MIPS64EB-NEXT: .cfi_offset 31, -8 ; MIPS64EB-NEXT: .cfi_offset 28, -16 ; MIPS64EB-NEXT: lui $1, %hi(%neg(%gp_rel(calli32_4))) ; MIPS64EB-NEXT: daddu $1, $1, $25 ; MIPS64EB-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(calli32_4))) ; MIPS64EB-NEXT: daddiu $1, $zero, 3 ; MIPS64EB-NEXT: dsll $2, $1, 33 ; MIPS64EB-NEXT: daddiu $4, $2, 7 ; MIPS64EB-NEXT: dsll $1, $1, 34 ; MIPS64EB-NEXT: daddiu $6, $1, 8 ; MIPS64EB-NEXT: daddiu $1, $zero, 9 ; MIPS64EB-NEXT: dsll $1, $1, 32 ; MIPS64EB-NEXT: daddiu $5, $1, 10 ; MIPS64EB-NEXT: ld $25, %call16(i32_4)($gp) ; MIPS64EB-NEXT: move $7, $5 ; MIPS64EB-NEXT: jalr $25 ; MIPS64EB-NEXT: nop ; MIPS64EB-NEXT: ld $1, %got_disp(gv4i32)($gp) ; MIPS64EB-NEXT: sd $3, 8($1) ; MIPS64EB-NEXT: sd $2, 0($1) ; MIPS64EB-NEXT: ld $gp, 0($sp) # 8-byte Folded Reload ; MIPS64EB-NEXT: ld $ra, 8($sp) # 8-byte Folded Reload ; MIPS64EB-NEXT: daddiu $sp, $sp, 16 ; MIPS64EB-NEXT: jr $ra ; MIPS64EB-NEXT: nop ; ; MIPS32R5-LABEL: calli32_4: ; MIPS32R5: # %bb.0: # %entry ; MIPS32R5-NEXT: addiu $sp, $sp, -40 ; MIPS32R5-NEXT: .cfi_def_cfa_offset 40 ; MIPS32R5-NEXT: sw $ra, 36($sp) # 4-byte Folded Spill ; MIPS32R5-NEXT: .cfi_offset 31, -4 ; MIPS32R5-NEXT: addiu $1, $zero, 9 ; MIPS32R5-NEXT: addiu $2, $zero, 10 ; MIPS32R5-NEXT: sw $2, 28($sp) ; MIPS32R5-NEXT: sw $1, 24($sp) ; MIPS32R5-NEXT: addiu $1, $zero, 8 ; MIPS32R5-NEXT: sw $1, 20($sp) ; MIPS32R5-NEXT: addiu $1, $zero, 12 ; MIPS32R5-NEXT: sw $1, 16($sp) ; MIPS32R5-NEXT: addiu $4, $zero, 6 ; MIPS32R5-NEXT: addiu $5, $zero, 7 ; MIPS32R5-NEXT: addiu $6, $zero, 9 ; MIPS32R5-NEXT: addiu $7, $zero, 10 ; MIPS32R5-NEXT: jal i32_4 ; MIPS32R5-NEXT: nop ; MIPS32R5-NEXT: ldi.b $w0, 0 ; MIPS32R5-NEXT: insert.w $w0[0], $2 ; MIPS32R5-NEXT: insert.w $w0[1], $3 ; MIPS32R5-NEXT: insert.w $w0[2], $4 ; MIPS32R5-NEXT: lui $1, %hi(gv4i32) ; MIPS32R5-NEXT: insert.w $w0[3], $5 ; MIPS32R5-NEXT: addiu $1, $1, %lo(gv4i32) ; MIPS32R5-NEXT: st.w $w0, 0($1) ; MIPS32R5-NEXT: lw $ra, 36($sp) # 4-byte Folded Reload ; MIPS32R5-NEXT: addiu $sp, $sp, 40 ; MIPS32R5-NEXT: jr $ra ; MIPS32R5-NEXT: nop ; ; MIPS64R5-LABEL: calli32_4: ; MIPS64R5: # %bb.0: # %entry ; MIPS64R5-NEXT: daddiu $sp, $sp, -16 ; MIPS64R5-NEXT: .cfi_def_cfa_offset 16 ; MIPS64R5-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; MIPS64R5-NEXT: sd $gp, 0($sp) # 8-byte Folded Spill ; MIPS64R5-NEXT: .cfi_offset 31, -8 ; MIPS64R5-NEXT: .cfi_offset 28, -16 ; MIPS64R5-NEXT: lui $1, %hi(%neg(%gp_rel(calli32_4))) ; MIPS64R5-NEXT: daddu $1, $1, $25 ; MIPS64R5-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(calli32_4))) ; MIPS64R5-NEXT: ld $1, %got_page(.LCPI35_0)($gp) ; MIPS64R5-NEXT: daddiu $1, $1, %got_ofst(.LCPI35_0) ; MIPS64R5-NEXT: ld.d $w0, 0($1) ; MIPS64R5-NEXT: copy_s.d $4, $w0[0] ; MIPS64R5-NEXT: copy_s.d $5, $w0[1] ; MIPS64R5-NEXT: ld $1, %got_page(.LCPI35_1)($gp) ; MIPS64R5-NEXT: daddiu $1, $1, %got_ofst(.LCPI35_1) ; MIPS64R5-NEXT: ld.d $w0, 0($1) ; MIPS64R5-NEXT: copy_s.d $6, $w0[0] ; MIPS64R5-NEXT: copy_s.d $7, $w0[1] ; MIPS64R5-NEXT: ld $25, %call16(i32_4)($gp) ; MIPS64R5-NEXT: jalr $25 ; MIPS64R5-NEXT: nop ; MIPS64R5-NEXT: ldi.b $w0, 0 ; MIPS64R5-NEXT: insert.d $w0[0], $2 ; MIPS64R5-NEXT: insert.d $w0[1], $3 ; MIPS64R5-NEXT: ld $1, %got_disp(gv4i32)($gp) ; MIPS64R5-NEXT: st.d $w0, 0($1) ; MIPS64R5-NEXT: ld $gp, 0($sp) # 8-byte Folded Reload ; MIPS64R5-NEXT: ld $ra, 8($sp) # 8-byte Folded Reload ; MIPS64R5-NEXT: daddiu $sp, $sp, 16 ; MIPS64R5-NEXT: jr $ra ; MIPS64R5-NEXT: nop ; ; MIPS64EL-LABEL: calli32_4: ; MIPS64EL: # %bb.0: # %entry ; MIPS64EL-NEXT: daddiu $sp, $sp, -16 ; MIPS64EL-NEXT: .cfi_def_cfa_offset 16 ; MIPS64EL-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; MIPS64EL-NEXT: sd $gp, 0($sp) # 8-byte Folded Spill ; MIPS64EL-NEXT: .cfi_offset 31, -8 ; MIPS64EL-NEXT: .cfi_offset 28, -16 ; MIPS64EL-NEXT: lui $1, %hi(%neg(%gp_rel(calli32_4))) ; MIPS64EL-NEXT: daddu $1, $1, $25 ; MIPS64EL-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(calli32_4))) ; MIPS64EL-NEXT: daddiu $1, $zero, 7 ; MIPS64EL-NEXT: dsll $1, $1, 32 ; MIPS64EL-NEXT: daddiu $4, $1, 6 ; MIPS64EL-NEXT: daddiu $1, $zero, 1 ; MIPS64EL-NEXT: dsll $1, $1, 35 ; MIPS64EL-NEXT: daddiu $6, $1, 12 ; MIPS64EL-NEXT: daddiu $1, $zero, 5 ; MIPS64EL-NEXT: dsll $1, $1, 33 ; MIPS64EL-NEXT: daddiu $5, $1, 9 ; MIPS64EL-NEXT: ld $25, %call16(i32_4)($gp) ; MIPS64EL-NEXT: move $7, $5 ; MIPS64EL-NEXT: jalr $25 ; MIPS64EL-NEXT: nop ; MIPS64EL-NEXT: ld $1, %got_disp(gv4i32)($gp) ; MIPS64EL-NEXT: sd $3, 8($1) ; MIPS64EL-NEXT: sd $2, 0($1) ; MIPS64EL-NEXT: ld $gp, 0($sp) # 8-byte Folded Reload ; MIPS64EL-NEXT: ld $ra, 8($sp) # 8-byte Folded Reload ; MIPS64EL-NEXT: daddiu $sp, $sp, 16 ; MIPS64EL-NEXT: jr $ra ; MIPS64EL-NEXT: nop entry: %0 = call <4 x i32> @i32_4(<4 x i32> , <4 x i32> ) store <4 x i32> %0, <4 x i32> * @gv4i32 ret void } define void @calli64_2() { ; MIPS32EB-LABEL: calli64_2: ; MIPS32EB: # %bb.0: # %entry ; MIPS32EB-NEXT: addiu $sp, $sp, -40 ; MIPS32EB-NEXT: .cfi_def_cfa_offset 40 ; MIPS32EB-NEXT: sw $ra, 36($sp) # 4-byte Folded Spill ; MIPS32EB-NEXT: .cfi_offset 31, -4 ; MIPS32EB-NEXT: addiu $1, $zero, 8 ; MIPS32EB-NEXT: sw $1, 28($sp) ; MIPS32EB-NEXT: addiu $1, $zero, 12 ; MIPS32EB-NEXT: sw $1, 20($sp) ; MIPS32EB-NEXT: sw $zero, 24($sp) ; MIPS32EB-NEXT: sw $zero, 16($sp) ; MIPS32EB-NEXT: addiu $4, $zero, 0 ; MIPS32EB-NEXT: addiu $5, $zero, 6 ; MIPS32EB-NEXT: addiu $6, $zero, 0 ; MIPS32EB-NEXT: addiu $7, $zero, 7 ; MIPS32EB-NEXT: jal i64_2 ; MIPS32EB-NEXT: nop ; MIPS32EB-NEXT: lui $1, %hi(gv2i64) ; MIPS32EB-NEXT: addiu $6, $1, %lo(gv2i64) ; MIPS32EB-NEXT: sw $5, 12($6) ; MIPS32EB-NEXT: sw $4, 8($6) ; MIPS32EB-NEXT: sw $3, 4($6) ; MIPS32EB-NEXT: sw $2, %lo(gv2i64)($1) ; MIPS32EB-NEXT: lw $ra, 36($sp) # 4-byte Folded Reload ; MIPS32EB-NEXT: addiu $sp, $sp, 40 ; MIPS32EB-NEXT: jr $ra ; MIPS32EB-NEXT: nop ; ; MIPS64-LABEL: calli64_2: ; MIPS64: # %bb.0: # %entry ; MIPS64-NEXT: daddiu $sp, $sp, -16 ; MIPS64-NEXT: .cfi_def_cfa_offset 16 ; MIPS64-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; MIPS64-NEXT: sd $gp, 0($sp) # 8-byte Folded Spill ; MIPS64-NEXT: .cfi_offset 31, -8 ; MIPS64-NEXT: .cfi_offset 28, -16 ; MIPS64-NEXT: lui $1, %hi(%neg(%gp_rel(calli64_2))) ; MIPS64-NEXT: daddu $1, $1, $25 ; MIPS64-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(calli64_2))) ; MIPS64-NEXT: ld $25, %call16(i64_2)($gp) ; MIPS64-NEXT: daddiu $4, $zero, 6 ; MIPS64-NEXT: daddiu $5, $zero, 7 ; MIPS64-NEXT: daddiu $6, $zero, 12 ; MIPS64-NEXT: daddiu $7, $zero, 8 ; MIPS64-NEXT: jalr $25 ; MIPS64-NEXT: nop ; MIPS64-NEXT: ld $1, %got_disp(gv2i64)($gp) ; MIPS64-NEXT: sd $3, 8($1) ; MIPS64-NEXT: sd $2, 0($1) ; MIPS64-NEXT: ld $gp, 0($sp) # 8-byte Folded Reload ; MIPS64-NEXT: ld $ra, 8($sp) # 8-byte Folded Reload ; MIPS64-NEXT: daddiu $sp, $sp, 16 ; MIPS64-NEXT: jr $ra ; MIPS64-NEXT: nop ; ; MIPS32R5-LABEL: calli64_2: ; MIPS32R5: # %bb.0: # %entry ; MIPS32R5-NEXT: addiu $sp, $sp, -40 ; MIPS32R5-NEXT: .cfi_def_cfa_offset 40 ; MIPS32R5-NEXT: sw $ra, 36($sp) # 4-byte Folded Spill ; MIPS32R5-NEXT: .cfi_offset 31, -4 ; MIPS32R5-NEXT: lui $1, %hi($CPI36_0) ; MIPS32R5-NEXT: addiu $1, $1, %lo($CPI36_0) ; MIPS32R5-NEXT: ld.w $w0, 0($1) ; MIPS32R5-NEXT: copy_s.w $4, $w0[0] ; MIPS32R5-NEXT: copy_s.w $5, $w0[1] ; MIPS32R5-NEXT: copy_s.w $6, $w0[2] ; MIPS32R5-NEXT: copy_s.w $7, $w0[3] ; MIPS32R5-NEXT: lui $1, %hi($CPI36_1) ; MIPS32R5-NEXT: addiu $1, $1, %lo($CPI36_1) ; MIPS32R5-NEXT: ld.w $w0, 0($1) ; MIPS32R5-NEXT: copy_s.w $1, $w0[0] ; MIPS32R5-NEXT: copy_s.w $2, $w0[1] ; MIPS32R5-NEXT: copy_s.w $3, $w0[2] ; MIPS32R5-NEXT: copy_s.w $8, $w0[3] ; MIPS32R5-NEXT: sw $8, 28($sp) ; MIPS32R5-NEXT: sw $3, 24($sp) ; MIPS32R5-NEXT: sw $2, 20($sp) ; MIPS32R5-NEXT: sw $1, 16($sp) ; MIPS32R5-NEXT: jal i64_2 ; MIPS32R5-NEXT: nop ; MIPS32R5-NEXT: ldi.b $w0, 0 ; MIPS32R5-NEXT: insert.w $w0[0], $2 ; MIPS32R5-NEXT: lui $1, %hi(gv2i64) ; MIPS32R5-NEXT: insert.w $w0[1], $3 ; MIPS32R5-NEXT: addiu $1, $1, %lo(gv2i64) ; MIPS32R5-NEXT: insert.w $w0[2], $4 ; MIPS32R5-NEXT: insert.w $w0[3], $5 ; MIPS32R5-NEXT: st.w $w0, 0($1) ; MIPS32R5-NEXT: lw $ra, 36($sp) # 4-byte Folded Reload ; MIPS32R5-NEXT: addiu $sp, $sp, 40 ; MIPS32R5-NEXT: jr $ra ; MIPS32R5-NEXT: nop ; ; MIPS64R5-LABEL: calli64_2: ; MIPS64R5: # %bb.0: # %entry ; MIPS64R5-NEXT: daddiu $sp, $sp, -16 ; MIPS64R5-NEXT: .cfi_def_cfa_offset 16 ; MIPS64R5-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; MIPS64R5-NEXT: sd $gp, 0($sp) # 8-byte Folded Spill ; MIPS64R5-NEXT: .cfi_offset 31, -8 ; MIPS64R5-NEXT: .cfi_offset 28, -16 ; MIPS64R5-NEXT: lui $1, %hi(%neg(%gp_rel(calli64_2))) ; MIPS64R5-NEXT: daddu $1, $1, $25 ; MIPS64R5-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(calli64_2))) ; MIPS64R5-NEXT: ld $25, %call16(i64_2)($gp) ; MIPS64R5-NEXT: daddiu $4, $zero, 6 ; MIPS64R5-NEXT: daddiu $5, $zero, 7 ; MIPS64R5-NEXT: daddiu $6, $zero, 12 ; MIPS64R5-NEXT: daddiu $7, $zero, 8 ; MIPS64R5-NEXT: jalr $25 ; MIPS64R5-NEXT: nop ; MIPS64R5-NEXT: ldi.b $w0, 0 ; MIPS64R5-NEXT: insert.d $w0[0], $2 ; MIPS64R5-NEXT: insert.d $w0[1], $3 ; MIPS64R5-NEXT: ld $1, %got_disp(gv2i64)($gp) ; MIPS64R5-NEXT: st.d $w0, 0($1) ; MIPS64R5-NEXT: ld $gp, 0($sp) # 8-byte Folded Reload ; MIPS64R5-NEXT: ld $ra, 8($sp) # 8-byte Folded Reload ; MIPS64R5-NEXT: daddiu $sp, $sp, 16 ; MIPS64R5-NEXT: jr $ra ; MIPS64R5-NEXT: nop ; ; MIPS32EL-LABEL: calli64_2: ; MIPS32EL: # %bb.0: # %entry ; MIPS32EL-NEXT: addiu $sp, $sp, -40 ; MIPS32EL-NEXT: .cfi_def_cfa_offset 40 ; MIPS32EL-NEXT: sw $ra, 36($sp) # 4-byte Folded Spill ; MIPS32EL-NEXT: .cfi_offset 31, -4 ; MIPS32EL-NEXT: addiu $1, $zero, 8 ; MIPS32EL-NEXT: sw $1, 24($sp) ; MIPS32EL-NEXT: addiu $1, $zero, 12 ; MIPS32EL-NEXT: sw $1, 16($sp) ; MIPS32EL-NEXT: sw $zero, 28($sp) ; MIPS32EL-NEXT: sw $zero, 20($sp) ; MIPS32EL-NEXT: addiu $4, $zero, 6 ; MIPS32EL-NEXT: addiu $5, $zero, 0 ; MIPS32EL-NEXT: addiu $6, $zero, 7 ; MIPS32EL-NEXT: addiu $7, $zero, 0 ; MIPS32EL-NEXT: jal i64_2 ; MIPS32EL-NEXT: nop ; MIPS32EL-NEXT: lui $1, %hi(gv2i64) ; MIPS32EL-NEXT: addiu $6, $1, %lo(gv2i64) ; MIPS32EL-NEXT: sw $5, 12($6) ; MIPS32EL-NEXT: sw $4, 8($6) ; MIPS32EL-NEXT: sw $3, 4($6) ; MIPS32EL-NEXT: sw $2, %lo(gv2i64)($1) ; MIPS32EL-NEXT: lw $ra, 36($sp) # 4-byte Folded Reload ; MIPS32EL-NEXT: addiu $sp, $sp, 40 ; MIPS32EL-NEXT: jr $ra ; MIPS32EL-NEXT: nop entry: %0 = call <2 x i64> @i64_2(<2 x i64> , <2 x i64> ) store <2 x i64> %0, <2 x i64> * @gv2i64 ret void } declare <2 x float> @float2_extern(<2 x float>, <2 x float>) declare <4 x float> @float4_extern(<4 x float>, <4 x float>) declare <2 x double> @double2_extern(<2 x double>, <2 x double>) define void @callfloat_2() { ; MIPS32-LABEL: callfloat_2: ; MIPS32: # %bb.0: # %entry ; MIPS32-NEXT: addiu $sp, $sp, -40 ; MIPS32-NEXT: .cfi_def_cfa_offset 40 ; MIPS32-NEXT: sw $ra, 36($sp) # 4-byte Folded Spill ; MIPS32-NEXT: .cfi_offset 31, -4 ; MIPS32-NEXT: lui $1, 16736 ; MIPS32-NEXT: sw $1, 20($sp) ; MIPS32-NEXT: lui $1, 16704 ; MIPS32-NEXT: sw $1, 16($sp) ; MIPS32-NEXT: addiu $4, $sp, 24 ; MIPS32-NEXT: addiu $6, $zero, 0 ; MIPS32-NEXT: lui $7, 49024 ; MIPS32-NEXT: jal float2_extern ; MIPS32-NEXT: nop ; MIPS32-NEXT: lui $1, %hi(gv2f32) ; MIPS32-NEXT: addiu $2, $1, %lo(gv2f32) ; MIPS32-NEXT: lwc1 $f0, 28($sp) ; MIPS32-NEXT: swc1 $f0, 4($2) ; MIPS32-NEXT: lwc1 $f0, 24($sp) ; MIPS32-NEXT: swc1 $f0, %lo(gv2f32)($1) ; MIPS32-NEXT: lw $ra, 36($sp) # 4-byte Folded Reload ; MIPS32-NEXT: addiu $sp, $sp, 40 ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: nop ; ; MIPS64EB-LABEL: callfloat_2: ; MIPS64EB: # %bb.0: # %entry ; MIPS64EB-NEXT: daddiu $sp, $sp, -16 ; MIPS64EB-NEXT: .cfi_def_cfa_offset 16 ; MIPS64EB-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; MIPS64EB-NEXT: sd $gp, 0($sp) # 8-byte Folded Spill ; MIPS64EB-NEXT: .cfi_offset 31, -8 ; MIPS64EB-NEXT: .cfi_offset 28, -16 ; MIPS64EB-NEXT: lui $1, %hi(%neg(%gp_rel(callfloat_2))) ; MIPS64EB-NEXT: daddu $1, $1, $25 ; MIPS64EB-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(callfloat_2))) ; MIPS64EB-NEXT: daddiu $1, $zero, 383 ; MIPS64EB-NEXT: dsll $4, $1, 23 ; MIPS64EB-NEXT: daddiu $1, $zero, 261 ; MIPS64EB-NEXT: dsll $1, $1, 33 ; MIPS64EB-NEXT: daddiu $1, $1, 523 ; MIPS64EB-NEXT: dsll $5, $1, 21 ; MIPS64EB-NEXT: ld $25, %call16(float2_extern)($gp) ; MIPS64EB-NEXT: jalr $25 ; MIPS64EB-NEXT: nop ; MIPS64EB-NEXT: ld $1, %got_disp(gv2f32)($gp) ; MIPS64EB-NEXT: sd $2, 0($1) ; MIPS64EB-NEXT: ld $gp, 0($sp) # 8-byte Folded Reload ; MIPS64EB-NEXT: ld $ra, 8($sp) # 8-byte Folded Reload ; MIPS64EB-NEXT: daddiu $sp, $sp, 16 ; MIPS64EB-NEXT: jr $ra ; MIPS64EB-NEXT: nop ; ; MIPS32R5-LABEL: callfloat_2: ; MIPS32R5: # %bb.0: # %entry ; MIPS32R5-NEXT: addiu $sp, $sp, -40 ; MIPS32R5-NEXT: .cfi_def_cfa_offset 40 ; MIPS32R5-NEXT: sw $ra, 36($sp) # 4-byte Folded Spill ; MIPS32R5-NEXT: .cfi_offset 31, -4 ; MIPS32R5-NEXT: lui $1, 16736 ; MIPS32R5-NEXT: sw $1, 20($sp) ; MIPS32R5-NEXT: lui $1, 16704 ; MIPS32R5-NEXT: sw $1, 16($sp) ; MIPS32R5-NEXT: addiu $4, $sp, 24 ; MIPS32R5-NEXT: addiu $6, $zero, 0 ; MIPS32R5-NEXT: lui $7, 49024 ; MIPS32R5-NEXT: jal float2_extern ; MIPS32R5-NEXT: nop ; MIPS32R5-NEXT: lui $1, %hi(gv2f32) ; MIPS32R5-NEXT: addiu $2, $1, %lo(gv2f32) ; MIPS32R5-NEXT: lwc1 $f0, 28($sp) ; MIPS32R5-NEXT: swc1 $f0, 4($2) ; MIPS32R5-NEXT: lwc1 $f0, 24($sp) ; MIPS32R5-NEXT: swc1 $f0, %lo(gv2f32)($1) ; MIPS32R5-NEXT: lw $ra, 36($sp) # 4-byte Folded Reload ; MIPS32R5-NEXT: addiu $sp, $sp, 40 ; MIPS32R5-NEXT: jr $ra ; MIPS32R5-NEXT: nop ; ; MIPS64R5-LABEL: callfloat_2: ; MIPS64R5: # %bb.0: # %entry ; MIPS64R5-NEXT: daddiu $sp, $sp, -16 ; MIPS64R5-NEXT: .cfi_def_cfa_offset 16 ; MIPS64R5-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; MIPS64R5-NEXT: sd $gp, 0($sp) # 8-byte Folded Spill ; MIPS64R5-NEXT: .cfi_offset 31, -8 ; MIPS64R5-NEXT: .cfi_offset 28, -16 ; MIPS64R5-NEXT: lui $1, %hi(%neg(%gp_rel(callfloat_2))) ; MIPS64R5-NEXT: daddu $1, $1, $25 ; MIPS64R5-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(callfloat_2))) ; MIPS64R5-NEXT: ld $1, %got_page(.LCPI37_0)($gp) ; MIPS64R5-NEXT: daddiu $1, $1, %got_ofst(.LCPI37_0) ; MIPS64R5-NEXT: ld.d $w0, 0($1) ; MIPS64R5-NEXT: copy_s.d $4, $w0[0] ; MIPS64R5-NEXT: ld $1, %got_page(.LCPI37_1)($gp) ; MIPS64R5-NEXT: daddiu $1, $1, %got_ofst(.LCPI37_1) ; MIPS64R5-NEXT: ld.d $w0, 0($1) ; MIPS64R5-NEXT: copy_s.d $5, $w0[0] ; MIPS64R5-NEXT: ld $25, %call16(float2_extern)($gp) ; MIPS64R5-NEXT: jalr $25 ; MIPS64R5-NEXT: nop ; MIPS64R5-NEXT: ld $1, %got_disp(gv2f32)($gp) ; MIPS64R5-NEXT: sd $2, 0($1) ; MIPS64R5-NEXT: ld $gp, 0($sp) # 8-byte Folded Reload ; MIPS64R5-NEXT: ld $ra, 8($sp) # 8-byte Folded Reload ; MIPS64R5-NEXT: daddiu $sp, $sp, 16 ; MIPS64R5-NEXT: jr $ra ; MIPS64R5-NEXT: nop ; ; MIPS64EL-LABEL: callfloat_2: ; MIPS64EL: # %bb.0: # %entry ; MIPS64EL-NEXT: daddiu $sp, $sp, -16 ; MIPS64EL-NEXT: .cfi_def_cfa_offset 16 ; MIPS64EL-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; MIPS64EL-NEXT: sd $gp, 0($sp) # 8-byte Folded Spill ; MIPS64EL-NEXT: .cfi_offset 31, -8 ; MIPS64EL-NEXT: .cfi_offset 28, -16 ; MIPS64EL-NEXT: lui $1, %hi(%neg(%gp_rel(callfloat_2))) ; MIPS64EL-NEXT: daddu $1, $1, $25 ; MIPS64EL-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(callfloat_2))) ; MIPS64EL-NEXT: daddiu $1, $zero, 383 ; MIPS64EL-NEXT: dsll $4, $1, 55 ; MIPS64EL-NEXT: daddiu $1, $zero, 523 ; MIPS64EL-NEXT: dsll $1, $1, 31 ; MIPS64EL-NEXT: daddiu $1, $1, 261 ; MIPS64EL-NEXT: dsll $5, $1, 22 ; MIPS64EL-NEXT: ld $25, %call16(float2_extern)($gp) ; MIPS64EL-NEXT: jalr $25 ; MIPS64EL-NEXT: nop ; MIPS64EL-NEXT: ld $1, %got_disp(gv2f32)($gp) ; MIPS64EL-NEXT: sd $2, 0($1) ; MIPS64EL-NEXT: ld $gp, 0($sp) # 8-byte Folded Reload ; MIPS64EL-NEXT: ld $ra, 8($sp) # 8-byte Folded Reload ; MIPS64EL-NEXT: daddiu $sp, $sp, 16 ; MIPS64EL-NEXT: jr $ra ; MIPS64EL-NEXT: nop entry: %0 = call <2 x float> @float2_extern(<2 x float> , <2 x float> ) store <2 x float> %0, <2 x float> * @gv2f32 ret void } define void @callfloat_4() { ; MIPS32-LABEL: callfloat_4: ; MIPS32: # %bb.0: # %entry ; MIPS32-NEXT: addiu $sp, $sp, -80 ; MIPS32-NEXT: .cfi_def_cfa_offset 80 ; MIPS32-NEXT: sw $ra, 76($sp) # 4-byte Folded Spill ; MIPS32-NEXT: sw $fp, 72($sp) # 4-byte Folded Spill ; MIPS32-NEXT: .cfi_offset 31, -4 ; MIPS32-NEXT: .cfi_offset 30, -8 ; MIPS32-NEXT: move $fp, $sp ; MIPS32-NEXT: .cfi_def_cfa_register 30 ; MIPS32-NEXT: addiu $1, $zero, -16 ; MIPS32-NEXT: and $sp, $sp, $1 ; MIPS32-NEXT: lui $1, 16704 ; MIPS32-NEXT: lui $2, 16736 ; MIPS32-NEXT: lui $3, 16752 ; MIPS32-NEXT: lui $4, 16768 ; MIPS32-NEXT: sw $4, 36($sp) ; MIPS32-NEXT: sw $3, 32($sp) ; MIPS32-NEXT: sw $2, 28($sp) ; MIPS32-NEXT: sw $1, 24($sp) ; MIPS32-NEXT: lui $1, 16512 ; MIPS32-NEXT: sw $1, 20($sp) ; MIPS32-NEXT: lui $1, 16384 ; MIPS32-NEXT: sw $1, 16($sp) ; MIPS32-NEXT: addiu $4, $sp, 48 ; MIPS32-NEXT: addiu $6, $zero, 0 ; MIPS32-NEXT: lui $7, 49024 ; MIPS32-NEXT: jal float4_extern ; MIPS32-NEXT: nop ; MIPS32-NEXT: lui $1, %hi(gv4f32) ; MIPS32-NEXT: addiu $2, $1, %lo(gv4f32) ; MIPS32-NEXT: lwc1 $f0, 60($sp) ; MIPS32-NEXT: swc1 $f0, 12($2) ; MIPS32-NEXT: lwc1 $f0, 56($sp) ; MIPS32-NEXT: swc1 $f0, 8($2) ; MIPS32-NEXT: lwc1 $f0, 52($sp) ; MIPS32-NEXT: swc1 $f0, 4($2) ; MIPS32-NEXT: lwc1 $f0, 48($sp) ; MIPS32-NEXT: swc1 $f0, %lo(gv4f32)($1) ; MIPS32-NEXT: move $sp, $fp ; MIPS32-NEXT: lw $fp, 72($sp) # 4-byte Folded Reload ; MIPS32-NEXT: lw $ra, 76($sp) # 4-byte Folded Reload ; MIPS32-NEXT: addiu $sp, $sp, 80 ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: nop ; ; MIPS64EB-LABEL: callfloat_4: ; MIPS64EB: # %bb.0: # %entry ; MIPS64EB-NEXT: daddiu $sp, $sp, -16 ; MIPS64EB-NEXT: .cfi_def_cfa_offset 16 ; MIPS64EB-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; MIPS64EB-NEXT: sd $gp, 0($sp) # 8-byte Folded Spill ; MIPS64EB-NEXT: .cfi_offset 31, -8 ; MIPS64EB-NEXT: .cfi_offset 28, -16 ; MIPS64EB-NEXT: lui $1, %hi(%neg(%gp_rel(callfloat_4))) ; MIPS64EB-NEXT: daddu $1, $1, $25 ; MIPS64EB-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(callfloat_4))) ; MIPS64EB-NEXT: daddiu $1, $zero, 1 ; MIPS64EB-NEXT: dsll $1, $1, 39 ; MIPS64EB-NEXT: daddiu $1, $1, 129 ; MIPS64EB-NEXT: daddiu $2, $zero, 261 ; MIPS64EB-NEXT: dsll $2, $2, 33 ; MIPS64EB-NEXT: daddiu $3, $zero, 383 ; MIPS64EB-NEXT: dsll $4, $3, 23 ; MIPS64EB-NEXT: dsll $5, $1, 23 ; MIPS64EB-NEXT: daddiu $1, $2, 523 ; MIPS64EB-NEXT: dsll $6, $1, 21 ; MIPS64EB-NEXT: daddiu $1, $zero, 1047 ; MIPS64EB-NEXT: dsll $1, $1, 29 ; MIPS64EB-NEXT: daddiu $1, $1, 131 ; MIPS64EB-NEXT: dsll $7, $1, 23 ; MIPS64EB-NEXT: ld $25, %call16(float4_extern)($gp) ; MIPS64EB-NEXT: jalr $25 ; MIPS64EB-NEXT: nop ; MIPS64EB-NEXT: ld $1, %got_disp(gv4f32)($gp) ; MIPS64EB-NEXT: sd $3, 8($1) ; MIPS64EB-NEXT: sd $2, 0($1) ; MIPS64EB-NEXT: ld $gp, 0($sp) # 8-byte Folded Reload ; MIPS64EB-NEXT: ld $ra, 8($sp) # 8-byte Folded Reload ; MIPS64EB-NEXT: daddiu $sp, $sp, 16 ; MIPS64EB-NEXT: jr $ra ; MIPS64EB-NEXT: nop ; ; MIPS32R5-LABEL: callfloat_4: ; MIPS32R5: # %bb.0: # %entry ; MIPS32R5-NEXT: addiu $sp, $sp, -80 ; MIPS32R5-NEXT: .cfi_def_cfa_offset 80 ; MIPS32R5-NEXT: sw $ra, 76($sp) # 4-byte Folded Spill ; MIPS32R5-NEXT: sw $fp, 72($sp) # 4-byte Folded Spill ; MIPS32R5-NEXT: .cfi_offset 31, -4 ; MIPS32R5-NEXT: .cfi_offset 30, -8 ; MIPS32R5-NEXT: move $fp, $sp ; MIPS32R5-NEXT: .cfi_def_cfa_register 30 ; MIPS32R5-NEXT: addiu $1, $zero, -16 ; MIPS32R5-NEXT: and $sp, $sp, $1 ; MIPS32R5-NEXT: lui $1, %hi($CPI38_0) ; MIPS32R5-NEXT: addiu $1, $1, %lo($CPI38_0) ; MIPS32R5-NEXT: ld.w $w0, 0($1) ; MIPS32R5-NEXT: copy_s.w $6, $w0[0] ; MIPS32R5-NEXT: copy_s.w $7, $w0[1] ; MIPS32R5-NEXT: copy_s.w $1, $w0[2] ; MIPS32R5-NEXT: copy_s.w $2, $w0[3] ; MIPS32R5-NEXT: lui $3, %hi($CPI38_1) ; MIPS32R5-NEXT: addiu $3, $3, %lo($CPI38_1) ; MIPS32R5-NEXT: ld.w $w0, 0($3) ; MIPS32R5-NEXT: copy_s.w $3, $w0[0] ; MIPS32R5-NEXT: copy_s.w $4, $w0[1] ; MIPS32R5-NEXT: copy_s.w $5, $w0[2] ; MIPS32R5-NEXT: copy_s.w $8, $w0[3] ; MIPS32R5-NEXT: sw $8, 36($sp) ; MIPS32R5-NEXT: sw $5, 32($sp) ; MIPS32R5-NEXT: sw $4, 28($sp) ; MIPS32R5-NEXT: sw $3, 24($sp) ; MIPS32R5-NEXT: sw $2, 20($sp) ; MIPS32R5-NEXT: sw $1, 16($sp) ; MIPS32R5-NEXT: addiu $4, $sp, 48 ; MIPS32R5-NEXT: jal float4_extern ; MIPS32R5-NEXT: nop ; MIPS32R5-NEXT: lui $1, %hi(gv4f32) ; MIPS32R5-NEXT: addiu $1, $1, %lo(gv4f32) ; MIPS32R5-NEXT: ld.w $w0, 48($sp) ; MIPS32R5-NEXT: st.w $w0, 0($1) ; MIPS32R5-NEXT: move $sp, $fp ; MIPS32R5-NEXT: lw $fp, 72($sp) # 4-byte Folded Reload ; MIPS32R5-NEXT: lw $ra, 76($sp) # 4-byte Folded Reload ; MIPS32R5-NEXT: addiu $sp, $sp, 80 ; MIPS32R5-NEXT: jr $ra ; MIPS32R5-NEXT: nop ; ; MIPS64R5-LABEL: callfloat_4: ; MIPS64R5: # %bb.0: # %entry ; MIPS64R5-NEXT: daddiu $sp, $sp, -16 ; MIPS64R5-NEXT: .cfi_def_cfa_offset 16 ; MIPS64R5-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; MIPS64R5-NEXT: sd $gp, 0($sp) # 8-byte Folded Spill ; MIPS64R5-NEXT: .cfi_offset 31, -8 ; MIPS64R5-NEXT: .cfi_offset 28, -16 ; MIPS64R5-NEXT: lui $1, %hi(%neg(%gp_rel(callfloat_4))) ; MIPS64R5-NEXT: daddu $1, $1, $25 ; MIPS64R5-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(callfloat_4))) ; MIPS64R5-NEXT: ld $1, %got_page(.LCPI38_0)($gp) ; MIPS64R5-NEXT: daddiu $1, $1, %got_ofst(.LCPI38_0) ; MIPS64R5-NEXT: ld.d $w0, 0($1) ; MIPS64R5-NEXT: copy_s.d $4, $w0[0] ; MIPS64R5-NEXT: copy_s.d $5, $w0[1] ; MIPS64R5-NEXT: ld $1, %got_page(.LCPI38_1)($gp) ; MIPS64R5-NEXT: daddiu $1, $1, %got_ofst(.LCPI38_1) ; MIPS64R5-NEXT: ld.d $w0, 0($1) ; MIPS64R5-NEXT: copy_s.d $6, $w0[0] ; MIPS64R5-NEXT: copy_s.d $7, $w0[1] ; MIPS64R5-NEXT: ld $25, %call16(float4_extern)($gp) ; MIPS64R5-NEXT: jalr $25 ; MIPS64R5-NEXT: nop ; MIPS64R5-NEXT: ldi.b $w0, 0 ; MIPS64R5-NEXT: insert.d $w0[0], $2 ; MIPS64R5-NEXT: insert.d $w0[1], $3 ; MIPS64R5-NEXT: ld $1, %got_disp(gv4f32)($gp) ; MIPS64R5-NEXT: st.d $w0, 0($1) ; MIPS64R5-NEXT: ld $gp, 0($sp) # 8-byte Folded Reload ; MIPS64R5-NEXT: ld $ra, 8($sp) # 8-byte Folded Reload ; MIPS64R5-NEXT: daddiu $sp, $sp, 16 ; MIPS64R5-NEXT: jr $ra ; MIPS64R5-NEXT: nop ; ; MIPS64EL-LABEL: callfloat_4: ; MIPS64EL: # %bb.0: # %entry ; MIPS64EL-NEXT: daddiu $sp, $sp, -16 ; MIPS64EL-NEXT: .cfi_def_cfa_offset 16 ; MIPS64EL-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; MIPS64EL-NEXT: sd $gp, 0($sp) # 8-byte Folded Spill ; MIPS64EL-NEXT: .cfi_offset 31, -8 ; MIPS64EL-NEXT: .cfi_offset 28, -16 ; MIPS64EL-NEXT: lui $1, %hi(%neg(%gp_rel(callfloat_4))) ; MIPS64EL-NEXT: daddu $1, $1, $25 ; MIPS64EL-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(callfloat_4))) ; MIPS64EL-NEXT: daddiu $1, $zero, 129 ; MIPS64EL-NEXT: dsll $1, $1, 25 ; MIPS64EL-NEXT: daddiu $1, $1, 1 ; MIPS64EL-NEXT: daddiu $2, $zero, 523 ; MIPS64EL-NEXT: dsll $2, $2, 31 ; MIPS64EL-NEXT: daddiu $3, $zero, 383 ; MIPS64EL-NEXT: dsll $4, $3, 55 ; MIPS64EL-NEXT: dsll $5, $1, 30 ; MIPS64EL-NEXT: daddiu $1, $2, 261 ; MIPS64EL-NEXT: dsll $6, $1, 22 ; MIPS64EL-NEXT: daddiu $1, $zero, 131 ; MIPS64EL-NEXT: dsll $1, $1, 35 ; MIPS64EL-NEXT: daddiu $1, $1, 1047 ; MIPS64EL-NEXT: dsll $7, $1, 20 ; MIPS64EL-NEXT: ld $25, %call16(float4_extern)($gp) ; MIPS64EL-NEXT: jalr $25 ; MIPS64EL-NEXT: nop ; MIPS64EL-NEXT: ld $1, %got_disp(gv4f32)($gp) ; MIPS64EL-NEXT: sd $3, 8($1) ; MIPS64EL-NEXT: sd $2, 0($1) ; MIPS64EL-NEXT: ld $gp, 0($sp) # 8-byte Folded Reload ; MIPS64EL-NEXT: ld $ra, 8($sp) # 8-byte Folded Reload ; MIPS64EL-NEXT: daddiu $sp, $sp, 16 ; MIPS64EL-NEXT: jr $ra ; MIPS64EL-NEXT: nop entry: %0 = call <4 x float> @float4_extern(<4 x float> , <4 x float> ) store <4 x float> %0, <4 x float> * @gv4f32 ret void } define void @calldouble_2() { ; MIPS32EB-LABEL: calldouble_2: ; MIPS32EB: # %bb.0: # %entry ; MIPS32EB-NEXT: addiu $sp, $sp, -80 ; MIPS32EB-NEXT: .cfi_def_cfa_offset 80 ; MIPS32EB-NEXT: sw $ra, 76($sp) # 4-byte Folded Spill ; MIPS32EB-NEXT: sw $fp, 72($sp) # 4-byte Folded Spill ; MIPS32EB-NEXT: .cfi_offset 31, -4 ; MIPS32EB-NEXT: .cfi_offset 30, -8 ; MIPS32EB-NEXT: move $fp, $sp ; MIPS32EB-NEXT: .cfi_def_cfa_register 30 ; MIPS32EB-NEXT: addiu $1, $zero, -16 ; MIPS32EB-NEXT: and $sp, $sp, $1 ; MIPS32EB-NEXT: lui $1, 16424 ; MIPS32EB-NEXT: lui $2, 16428 ; MIPS32EB-NEXT: sw $2, 32($sp) ; MIPS32EB-NEXT: sw $1, 24($sp) ; MIPS32EB-NEXT: lui $1, 49136 ; MIPS32EB-NEXT: sw $1, 16($sp) ; MIPS32EB-NEXT: sw $zero, 36($sp) ; MIPS32EB-NEXT: sw $zero, 28($sp) ; MIPS32EB-NEXT: sw $zero, 20($sp) ; MIPS32EB-NEXT: addiu $4, $sp, 48 ; MIPS32EB-NEXT: addiu $6, $zero, 0 ; MIPS32EB-NEXT: addiu $7, $zero, 0 ; MIPS32EB-NEXT: jal double2_extern ; MIPS32EB-NEXT: nop ; MIPS32EB-NEXT: lui $1, %hi(gv2f64) ; MIPS32EB-NEXT: addiu $2, $1, %lo(gv2f64) ; MIPS32EB-NEXT: ldc1 $f0, 56($sp) ; MIPS32EB-NEXT: sdc1 $f0, 8($2) ; MIPS32EB-NEXT: ldc1 $f0, 48($sp) ; MIPS32EB-NEXT: sdc1 $f0, %lo(gv2f64)($1) ; MIPS32EB-NEXT: move $sp, $fp ; MIPS32EB-NEXT: lw $fp, 72($sp) # 4-byte Folded Reload ; MIPS32EB-NEXT: lw $ra, 76($sp) # 4-byte Folded Reload ; MIPS32EB-NEXT: addiu $sp, $sp, 80 ; MIPS32EB-NEXT: jr $ra ; MIPS32EB-NEXT: nop ; ; MIPS64-LABEL: calldouble_2: ; MIPS64: # %bb.0: # %entry ; MIPS64-NEXT: daddiu $sp, $sp, -16 ; MIPS64-NEXT: .cfi_def_cfa_offset 16 ; MIPS64-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; MIPS64-NEXT: sd $gp, 0($sp) # 8-byte Folded Spill ; MIPS64-NEXT: .cfi_offset 31, -8 ; MIPS64-NEXT: .cfi_offset 28, -16 ; MIPS64-NEXT: lui $1, %hi(%neg(%gp_rel(calldouble_2))) ; MIPS64-NEXT: daddu $1, $1, $25 ; MIPS64-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(calldouble_2))) ; MIPS64-NEXT: daddiu $1, $zero, 3071 ; MIPS64-NEXT: dsll $5, $1, 52 ; MIPS64-NEXT: daddiu $1, $zero, 2053 ; MIPS64-NEXT: dsll $6, $1, 51 ; MIPS64-NEXT: daddiu $1, $zero, 4107 ; MIPS64-NEXT: dsll $7, $1, 50 ; MIPS64-NEXT: ld $25, %call16(double2_extern)($gp) ; MIPS64-NEXT: daddiu $4, $zero, 0 ; MIPS64-NEXT: jalr $25 ; MIPS64-NEXT: nop ; MIPS64-NEXT: ld $1, %got_disp(gv2f64)($gp) ; MIPS64-NEXT: sd $3, 8($1) ; MIPS64-NEXT: sd $2, 0($1) ; MIPS64-NEXT: ld $gp, 0($sp) # 8-byte Folded Reload ; MIPS64-NEXT: ld $ra, 8($sp) # 8-byte Folded Reload ; MIPS64-NEXT: daddiu $sp, $sp, 16 ; MIPS64-NEXT: jr $ra ; MIPS64-NEXT: nop ; ; MIPS32R5-LABEL: calldouble_2: ; MIPS32R5: # %bb.0: # %entry ; MIPS32R5-NEXT: addiu $sp, $sp, -80 ; MIPS32R5-NEXT: .cfi_def_cfa_offset 80 ; MIPS32R5-NEXT: sw $ra, 76($sp) # 4-byte Folded Spill ; MIPS32R5-NEXT: sw $fp, 72($sp) # 4-byte Folded Spill ; MIPS32R5-NEXT: .cfi_offset 31, -4 ; MIPS32R5-NEXT: .cfi_offset 30, -8 ; MIPS32R5-NEXT: move $fp, $sp ; MIPS32R5-NEXT: .cfi_def_cfa_register 30 ; MIPS32R5-NEXT: addiu $1, $zero, -16 ; MIPS32R5-NEXT: and $sp, $sp, $1 ; MIPS32R5-NEXT: lui $1, %hi($CPI39_0) ; MIPS32R5-NEXT: addiu $1, $1, %lo($CPI39_0) ; MIPS32R5-NEXT: ld.w $w0, 0($1) ; MIPS32R5-NEXT: copy_s.w $6, $w0[0] ; MIPS32R5-NEXT: copy_s.w $7, $w0[1] ; MIPS32R5-NEXT: copy_s.w $1, $w0[2] ; MIPS32R5-NEXT: copy_s.w $2, $w0[3] ; MIPS32R5-NEXT: lui $3, %hi($CPI39_1) ; MIPS32R5-NEXT: addiu $3, $3, %lo($CPI39_1) ; MIPS32R5-NEXT: ld.w $w0, 0($3) ; MIPS32R5-NEXT: copy_s.w $3, $w0[0] ; MIPS32R5-NEXT: copy_s.w $4, $w0[1] ; MIPS32R5-NEXT: copy_s.w $5, $w0[2] ; MIPS32R5-NEXT: copy_s.w $8, $w0[3] ; MIPS32R5-NEXT: sw $8, 36($sp) ; MIPS32R5-NEXT: sw $5, 32($sp) ; MIPS32R5-NEXT: sw $4, 28($sp) ; MIPS32R5-NEXT: sw $3, 24($sp) ; MIPS32R5-NEXT: sw $2, 20($sp) ; MIPS32R5-NEXT: sw $1, 16($sp) ; MIPS32R5-NEXT: addiu $4, $sp, 48 ; MIPS32R5-NEXT: jal double2_extern ; MIPS32R5-NEXT: nop ; MIPS32R5-NEXT: lui $1, %hi(gv2f64) ; MIPS32R5-NEXT: addiu $1, $1, %lo(gv2f64) ; MIPS32R5-NEXT: ld.d $w0, 48($sp) ; MIPS32R5-NEXT: st.d $w0, 0($1) ; MIPS32R5-NEXT: move $sp, $fp ; MIPS32R5-NEXT: lw $fp, 72($sp) # 4-byte Folded Reload ; MIPS32R5-NEXT: lw $ra, 76($sp) # 4-byte Folded Reload ; MIPS32R5-NEXT: addiu $sp, $sp, 80 ; MIPS32R5-NEXT: jr $ra ; MIPS32R5-NEXT: nop ; ; MIPS64R5-LABEL: calldouble_2: ; MIPS64R5: # %bb.0: # %entry ; MIPS64R5-NEXT: daddiu $sp, $sp, -16 ; MIPS64R5-NEXT: .cfi_def_cfa_offset 16 ; MIPS64R5-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; MIPS64R5-NEXT: sd $gp, 0($sp) # 8-byte Folded Spill ; MIPS64R5-NEXT: .cfi_offset 31, -8 ; MIPS64R5-NEXT: .cfi_offset 28, -16 ; MIPS64R5-NEXT: lui $1, %hi(%neg(%gp_rel(calldouble_2))) ; MIPS64R5-NEXT: daddu $1, $1, $25 ; MIPS64R5-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(calldouble_2))) ; MIPS64R5-NEXT: ld $1, %got_page(.LCPI39_0)($gp) ; MIPS64R5-NEXT: daddiu $1, $1, %got_ofst(.LCPI39_0) ; MIPS64R5-NEXT: ld.d $w0, 0($1) ; MIPS64R5-NEXT: copy_s.d $4, $w0[0] ; MIPS64R5-NEXT: copy_s.d $5, $w0[1] ; MIPS64R5-NEXT: ld $1, %got_page(.LCPI39_1)($gp) ; MIPS64R5-NEXT: daddiu $1, $1, %got_ofst(.LCPI39_1) ; MIPS64R5-NEXT: ld.d $w0, 0($1) ; MIPS64R5-NEXT: copy_s.d $6, $w0[0] ; MIPS64R5-NEXT: copy_s.d $7, $w0[1] ; MIPS64R5-NEXT: ld $25, %call16(double2_extern)($gp) ; MIPS64R5-NEXT: jalr $25 ; MIPS64R5-NEXT: nop ; MIPS64R5-NEXT: ldi.b $w0, 0 ; MIPS64R5-NEXT: insert.d $w0[0], $2 ; MIPS64R5-NEXT: insert.d $w0[1], $3 ; MIPS64R5-NEXT: ld $1, %got_disp(gv2f64)($gp) ; MIPS64R5-NEXT: st.d $w0, 0($1) ; MIPS64R5-NEXT: ld $gp, 0($sp) # 8-byte Folded Reload ; MIPS64R5-NEXT: ld $ra, 8($sp) # 8-byte Folded Reload ; MIPS64R5-NEXT: daddiu $sp, $sp, 16 ; MIPS64R5-NEXT: jr $ra ; MIPS64R5-NEXT: nop ; ; MIPS32EL-LABEL: calldouble_2: ; MIPS32EL: # %bb.0: # %entry ; MIPS32EL-NEXT: addiu $sp, $sp, -80 ; MIPS32EL-NEXT: .cfi_def_cfa_offset 80 ; MIPS32EL-NEXT: sw $ra, 76($sp) # 4-byte Folded Spill ; MIPS32EL-NEXT: sw $fp, 72($sp) # 4-byte Folded Spill ; MIPS32EL-NEXT: .cfi_offset 31, -4 ; MIPS32EL-NEXT: .cfi_offset 30, -8 ; MIPS32EL-NEXT: move $fp, $sp ; MIPS32EL-NEXT: .cfi_def_cfa_register 30 ; MIPS32EL-NEXT: addiu $1, $zero, -16 ; MIPS32EL-NEXT: and $sp, $sp, $1 ; MIPS32EL-NEXT: lui $1, 16424 ; MIPS32EL-NEXT: lui $2, 16428 ; MIPS32EL-NEXT: sw $2, 36($sp) ; MIPS32EL-NEXT: sw $1, 28($sp) ; MIPS32EL-NEXT: lui $1, 49136 ; MIPS32EL-NEXT: sw $1, 20($sp) ; MIPS32EL-NEXT: sw $zero, 32($sp) ; MIPS32EL-NEXT: sw $zero, 24($sp) ; MIPS32EL-NEXT: sw $zero, 16($sp) ; MIPS32EL-NEXT: addiu $4, $sp, 48 ; MIPS32EL-NEXT: addiu $6, $zero, 0 ; MIPS32EL-NEXT: addiu $7, $zero, 0 ; MIPS32EL-NEXT: jal double2_extern ; MIPS32EL-NEXT: nop ; MIPS32EL-NEXT: lui $1, %hi(gv2f64) ; MIPS32EL-NEXT: addiu $2, $1, %lo(gv2f64) ; MIPS32EL-NEXT: ldc1 $f0, 56($sp) ; MIPS32EL-NEXT: sdc1 $f0, 8($2) ; MIPS32EL-NEXT: ldc1 $f0, 48($sp) ; MIPS32EL-NEXT: sdc1 $f0, %lo(gv2f64)($1) ; MIPS32EL-NEXT: move $sp, $fp ; MIPS32EL-NEXT: lw $fp, 72($sp) # 4-byte Folded Reload ; MIPS32EL-NEXT: lw $ra, 76($sp) # 4-byte Folded Reload ; MIPS32EL-NEXT: addiu $sp, $sp, 80 ; MIPS32EL-NEXT: jr $ra ; MIPS32EL-NEXT: nop entry: %0 = call <2 x double> @double2_extern(<2 x double> , <2 x double> ) store <2 x double> %0, <2 x double> * @gv2f64 ret void } ; The mixed tests show that due to alignment requirements, $5 is not used ; in argument passing. define float @mixed_i8(<2 x float> %a, i8 %b, <2 x float> %c) { ; MIPS32-LABEL: mixed_i8: ; MIPS32: # %bb.0: # %entry ; MIPS32-NEXT: mtc1 $5, $f0 ; MIPS32-NEXT: andi $1, $6, 255 ; MIPS32-NEXT: mtc1 $1, $f1 ; MIPS32-NEXT: cvt.s.w $f1, $f1 ; MIPS32-NEXT: add.s $f0, $f1, $f0 ; MIPS32-NEXT: lwc1 $f2, 20($sp) ; MIPS32-NEXT: add.s $f0, $f0, $f2 ; MIPS32-NEXT: mtc1 $4, $f2 ; MIPS32-NEXT: add.s $f1, $f1, $f2 ; MIPS32-NEXT: lwc1 $f2, 16($sp) ; MIPS32-NEXT: add.s $f1, $f1, $f2 ; MIPS32-NEXT: add.s $f0, $f1, $f0 ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: nop ; ; MIPS64EB-LABEL: mixed_i8: ; MIPS64EB: # %bb.0: # %entry ; MIPS64EB-NEXT: sll $1, $4, 0 ; MIPS64EB-NEXT: mtc1 $1, $f0 ; MIPS64EB-NEXT: sll $1, $5, 0 ; MIPS64EB-NEXT: andi $1, $1, 255 ; MIPS64EB-NEXT: mtc1 $1, $f1 ; MIPS64EB-NEXT: cvt.s.w $f1, $f1 ; MIPS64EB-NEXT: add.s $f0, $f1, $f0 ; MIPS64EB-NEXT: dsrl $1, $4, 32 ; MIPS64EB-NEXT: sll $1, $1, 0 ; MIPS64EB-NEXT: sll $2, $6, 0 ; MIPS64EB-NEXT: mtc1 $2, $f2 ; MIPS64EB-NEXT: add.s $f0, $f0, $f2 ; MIPS64EB-NEXT: mtc1 $1, $f2 ; MIPS64EB-NEXT: add.s $f1, $f1, $f2 ; MIPS64EB-NEXT: dsrl $1, $6, 32 ; MIPS64EB-NEXT: sll $1, $1, 0 ; MIPS64EB-NEXT: mtc1 $1, $f2 ; MIPS64EB-NEXT: add.s $f1, $f1, $f2 ; MIPS64EB-NEXT: add.s $f0, $f1, $f0 ; MIPS64EB-NEXT: jr $ra ; MIPS64EB-NEXT: nop ; ; MIPS32R5-LABEL: mixed_i8: ; MIPS32R5: # %bb.0: # %entry ; MIPS32R5-NEXT: addiu $sp, $sp, -64 ; MIPS32R5-NEXT: .cfi_def_cfa_offset 64 ; MIPS32R5-NEXT: sw $fp, 60($sp) # 4-byte Folded Spill ; MIPS32R5-NEXT: .cfi_offset 30, -4 ; MIPS32R5-NEXT: move $fp, $sp ; MIPS32R5-NEXT: .cfi_def_cfa_register 30 ; MIPS32R5-NEXT: addiu $1, $zero, -16 ; MIPS32R5-NEXT: and $sp, $sp, $1 ; MIPS32R5-NEXT: andi $1, $6, 255 ; MIPS32R5-NEXT: mtc1 $1, $f0 ; MIPS32R5-NEXT: cvt.s.w $f0, $f0 ; MIPS32R5-NEXT: swc1 $f0, 36($sp) ; MIPS32R5-NEXT: swc1 $f0, 32($sp) ; MIPS32R5-NEXT: sw $5, 4($sp) ; MIPS32R5-NEXT: sw $4, 0($sp) ; MIPS32R5-NEXT: ld.w $w0, 0($sp) ; MIPS32R5-NEXT: ld.w $w1, 32($sp) ; MIPS32R5-NEXT: fadd.w $w0, $w1, $w0 ; MIPS32R5-NEXT: lw $1, 84($fp) ; MIPS32R5-NEXT: sw $1, 20($sp) ; MIPS32R5-NEXT: lw $1, 80($fp) ; MIPS32R5-NEXT: sw $1, 16($sp) ; MIPS32R5-NEXT: ld.w $w1, 16($sp) ; MIPS32R5-NEXT: fadd.w $w0, $w0, $w1 ; MIPS32R5-NEXT: splati.w $w1, $w0[1] ; MIPS32R5-NEXT: add.s $f0, $f0, $f1 ; MIPS32R5-NEXT: move $sp, $fp ; MIPS32R5-NEXT: lw $fp, 60($sp) # 4-byte Folded Reload ; MIPS32R5-NEXT: addiu $sp, $sp, 64 ; MIPS32R5-NEXT: jr $ra ; MIPS32R5-NEXT: nop ; ; MIPS64R5-LABEL: mixed_i8: ; MIPS64R5: # %bb.0: # %entry ; MIPS64R5-NEXT: daddiu $sp, $sp, -48 ; MIPS64R5-NEXT: .cfi_def_cfa_offset 48 ; MIPS64R5-NEXT: sll $1, $5, 0 ; MIPS64R5-NEXT: andi $1, $1, 255 ; MIPS64R5-NEXT: mtc1 $1, $f0 ; MIPS64R5-NEXT: cvt.s.w $f0, $f0 ; MIPS64R5-NEXT: swc1 $f0, 36($sp) ; MIPS64R5-NEXT: swc1 $f0, 32($sp) ; MIPS64R5-NEXT: sd $4, 0($sp) ; MIPS64R5-NEXT: ld.w $w0, 0($sp) ; MIPS64R5-NEXT: ld.w $w1, 32($sp) ; MIPS64R5-NEXT: fadd.w $w0, $w1, $w0 ; MIPS64R5-NEXT: sd $6, 16($sp) ; MIPS64R5-NEXT: ld.w $w1, 16($sp) ; MIPS64R5-NEXT: fadd.w $w0, $w0, $w1 ; MIPS64R5-NEXT: splati.w $w1, $w0[1] ; MIPS64R5-NEXT: add.s $f0, $f0, $f1 ; MIPS64R5-NEXT: daddiu $sp, $sp, 48 ; MIPS64R5-NEXT: jr $ra ; MIPS64R5-NEXT: nop ; ; MIPS64EL-LABEL: mixed_i8: ; MIPS64EL: # %bb.0: # %entry ; MIPS64EL-NEXT: dsrl $1, $4, 32 ; MIPS64EL-NEXT: sll $1, $1, 0 ; MIPS64EL-NEXT: mtc1 $1, $f0 ; MIPS64EL-NEXT: sll $1, $5, 0 ; MIPS64EL-NEXT: andi $1, $1, 255 ; MIPS64EL-NEXT: mtc1 $1, $f1 ; MIPS64EL-NEXT: cvt.s.w $f1, $f1 ; MIPS64EL-NEXT: add.s $f0, $f1, $f0 ; MIPS64EL-NEXT: dsrl $1, $6, 32 ; MIPS64EL-NEXT: sll $1, $1, 0 ; MIPS64EL-NEXT: mtc1 $1, $f2 ; MIPS64EL-NEXT: add.s $f0, $f0, $f2 ; MIPS64EL-NEXT: sll $1, $4, 0 ; MIPS64EL-NEXT: mtc1 $1, $f2 ; MIPS64EL-NEXT: add.s $f1, $f1, $f2 ; MIPS64EL-NEXT: sll $1, $6, 0 ; MIPS64EL-NEXT: mtc1 $1, $f2 ; MIPS64EL-NEXT: add.s $f1, $f1, $f2 ; MIPS64EL-NEXT: add.s $f0, $f1, $f0 ; MIPS64EL-NEXT: jr $ra ; MIPS64EL-NEXT: nop entry: %0 = zext i8 %b to i32 %1 = uitofp i32 %0 to float %2 = insertelement <2 x float> undef, float %1, i32 0 %3 = insertelement <2 x float> %2, float %1, i32 1 %4 = fadd <2 x float> %3, %a %5 = fadd <2 x float> %4, %c %6 = extractelement <2 x float> %5, i32 0 %7 = extractelement <2 x float> %5, i32 1 %8 = fadd float %6, %7 ret float %8 } define <4 x float> @mixed_32(<4 x float> %a, i32 %b) { ; MIPS32EB-LABEL: mixed_32: ; MIPS32EB: # %bb.0: # %entry ; MIPS32EB-NEXT: addiu $sp, $sp, -8 ; MIPS32EB-NEXT: .cfi_def_cfa_offset 8 ; MIPS32EB-NEXT: lui $1, 17200 ; MIPS32EB-NEXT: sw $1, 0($sp) ; MIPS32EB-NEXT: lw $1, 32($sp) ; MIPS32EB-NEXT: sw $1, 4($sp) ; MIPS32EB-NEXT: lui $1, %hi($CPI41_0) ; MIPS32EB-NEXT: ldc1 $f0, %lo($CPI41_0)($1) ; MIPS32EB-NEXT: ldc1 $f2, 0($sp) ; MIPS32EB-NEXT: sub.d $f0, $f2, $f0 ; MIPS32EB-NEXT: cvt.s.d $f0, $f0 ; MIPS32EB-NEXT: lwc1 $f1, 28($sp) ; MIPS32EB-NEXT: lwc1 $f2, 24($sp) ; MIPS32EB-NEXT: add.s $f2, $f0, $f2 ; MIPS32EB-NEXT: add.s $f1, $f0, $f1 ; MIPS32EB-NEXT: swc1 $f1, 12($4) ; MIPS32EB-NEXT: swc1 $f2, 8($4) ; MIPS32EB-NEXT: mtc1 $7, $f1 ; MIPS32EB-NEXT: add.s $f1, $f0, $f1 ; MIPS32EB-NEXT: swc1 $f1, 4($4) ; MIPS32EB-NEXT: mtc1 $6, $f1 ; MIPS32EB-NEXT: add.s $f0, $f0, $f1 ; MIPS32EB-NEXT: swc1 $f0, 0($4) ; MIPS32EB-NEXT: addiu $sp, $sp, 8 ; MIPS32EB-NEXT: jr $ra ; MIPS32EB-NEXT: nop ; ; MIPS64EB-LABEL: mixed_32: ; MIPS64EB: # %bb.0: # %entry ; MIPS64EB-NEXT: daddiu $sp, $sp, -16 ; MIPS64EB-NEXT: .cfi_def_cfa_offset 16 ; MIPS64EB-NEXT: lui $1, %hi(%neg(%gp_rel(mixed_32))) ; MIPS64EB-NEXT: daddu $1, $1, $25 ; MIPS64EB-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(mixed_32))) ; MIPS64EB-NEXT: lui $2, 17200 ; MIPS64EB-NEXT: sw $2, 8($sp) ; MIPS64EB-NEXT: sll $2, $6, 0 ; MIPS64EB-NEXT: sw $2, 12($sp) ; MIPS64EB-NEXT: ld $1, %got_page(.LCPI41_0)($1) ; MIPS64EB-NEXT: ldc1 $f0, %got_ofst(.LCPI41_0)($1) ; MIPS64EB-NEXT: ldc1 $f1, 8($sp) ; MIPS64EB-NEXT: sub.d $f0, $f1, $f0 ; MIPS64EB-NEXT: cvt.s.d $f0, $f0 ; MIPS64EB-NEXT: dsrl $1, $4, 32 ; MIPS64EB-NEXT: sll $1, $1, 0 ; MIPS64EB-NEXT: mtc1 $1, $f1 ; MIPS64EB-NEXT: add.s $f1, $f0, $f1 ; MIPS64EB-NEXT: dsrl $1, $5, 32 ; MIPS64EB-NEXT: mfc1 $2, $f1 ; MIPS64EB-NEXT: sll $3, $4, 0 ; MIPS64EB-NEXT: sll $1, $1, 0 ; MIPS64EB-NEXT: mtc1 $1, $f1 ; MIPS64EB-NEXT: add.s $f1, $f0, $f1 ; MIPS64EB-NEXT: mfc1 $1, $f1 ; MIPS64EB-NEXT: mtc1 $3, $f1 ; MIPS64EB-NEXT: sll $3, $5, 0 ; MIPS64EB-NEXT: mtc1 $3, $f2 ; MIPS64EB-NEXT: dsll $2, $2, 32 ; MIPS64EB-NEXT: add.s $f1, $f0, $f1 ; MIPS64EB-NEXT: mfc1 $3, $f1 ; MIPS64EB-NEXT: dsll $3, $3, 32 ; MIPS64EB-NEXT: dsrl $3, $3, 32 ; MIPS64EB-NEXT: or $2, $3, $2 ; MIPS64EB-NEXT: dsll $1, $1, 32 ; MIPS64EB-NEXT: add.s $f0, $f0, $f2 ; MIPS64EB-NEXT: mfc1 $3, $f0 ; MIPS64EB-NEXT: dsll $3, $3, 32 ; MIPS64EB-NEXT: dsrl $3, $3, 32 ; MIPS64EB-NEXT: or $3, $3, $1 ; MIPS64EB-NEXT: daddiu $sp, $sp, 16 ; MIPS64EB-NEXT: jr $ra ; MIPS64EB-NEXT: nop ; ; MIPS32R5EB-LABEL: mixed_32: ; MIPS32R5EB: # %bb.0: # %entry ; MIPS32R5EB-NEXT: addiu $sp, $sp, -8 ; MIPS32R5EB-NEXT: .cfi_def_cfa_offset 8 ; MIPS32R5EB-NEXT: lui $1, 17200 ; MIPS32R5EB-NEXT: sw $1, 0($sp) ; MIPS32R5EB-NEXT: lw $1, 32($sp) ; MIPS32R5EB-NEXT: sw $1, 4($sp) ; MIPS32R5EB-NEXT: lui $1, %hi($CPI41_0) ; MIPS32R5EB-NEXT: ldc1 $f0, %lo($CPI41_0)($1) ; MIPS32R5EB-NEXT: ldc1 $f1, 0($sp) ; MIPS32R5EB-NEXT: sub.d $f0, $f1, $f0 ; MIPS32R5EB-NEXT: cvt.s.d $f0, $f0 ; MIPS32R5EB-NEXT: ldi.b $w1, 0 ; MIPS32R5EB-NEXT: splati.w $w0, $w0[0] ; MIPS32R5EB-NEXT: insert.w $w1[0], $6 ; MIPS32R5EB-NEXT: insert.w $w1[1], $7 ; MIPS32R5EB-NEXT: lw $1, 24($sp) ; MIPS32R5EB-NEXT: insert.w $w1[2], $1 ; MIPS32R5EB-NEXT: lw $1, 28($sp) ; MIPS32R5EB-NEXT: insert.w $w1[3], $1 ; MIPS32R5EB-NEXT: fadd.w $w0, $w0, $w1 ; MIPS32R5EB-NEXT: st.w $w0, 0($4) ; MIPS32R5EB-NEXT: addiu $sp, $sp, 8 ; MIPS32R5EB-NEXT: jr $ra ; MIPS32R5EB-NEXT: nop ; ; MIPS64R5EB-LABEL: mixed_32: ; MIPS64R5EB: # %bb.0: # %entry ; MIPS64R5EB-NEXT: daddiu $sp, $sp, -16 ; MIPS64R5EB-NEXT: .cfi_def_cfa_offset 16 ; MIPS64R5EB-NEXT: lui $1, %hi(%neg(%gp_rel(mixed_32))) ; MIPS64R5EB-NEXT: daddu $1, $1, $25 ; MIPS64R5EB-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(mixed_32))) ; MIPS64R5EB-NEXT: lui $2, 17200 ; MIPS64R5EB-NEXT: sw $2, 8($sp) ; MIPS64R5EB-NEXT: sll $2, $6, 0 ; MIPS64R5EB-NEXT: sw $2, 12($sp) ; MIPS64R5EB-NEXT: ld $1, %got_page(.LCPI41_0)($1) ; MIPS64R5EB-NEXT: ldc1 $f0, %got_ofst(.LCPI41_0)($1) ; MIPS64R5EB-NEXT: ldc1 $f1, 8($sp) ; MIPS64R5EB-NEXT: sub.d $f0, $f1, $f0 ; MIPS64R5EB-NEXT: ldi.b $w1, 0 ; MIPS64R5EB-NEXT: insert.d $w1[0], $4 ; MIPS64R5EB-NEXT: insert.d $w1[1], $5 ; MIPS64R5EB-NEXT: shf.w $w1, $w1, 177 ; MIPS64R5EB-NEXT: cvt.s.d $f0, $f0 ; MIPS64R5EB-NEXT: splati.w $w0, $w0[0] ; MIPS64R5EB-NEXT: fadd.w $w0, $w0, $w1 ; MIPS64R5EB-NEXT: shf.w $w0, $w0, 177 ; MIPS64R5EB-NEXT: copy_s.d $2, $w0[0] ; MIPS64R5EB-NEXT: copy_s.d $3, $w0[1] ; MIPS64R5EB-NEXT: daddiu $sp, $sp, 16 ; MIPS64R5EB-NEXT: jr $ra ; MIPS64R5EB-NEXT: nop ; ; MIPS32EL-LABEL: mixed_32: ; MIPS32EL: # %bb.0: # %entry ; MIPS32EL-NEXT: addiu $sp, $sp, -8 ; MIPS32EL-NEXT: .cfi_def_cfa_offset 8 ; MIPS32EL-NEXT: lui $1, 17200 ; MIPS32EL-NEXT: sw $1, 4($sp) ; MIPS32EL-NEXT: lw $1, 32($sp) ; MIPS32EL-NEXT: sw $1, 0($sp) ; MIPS32EL-NEXT: lui $1, %hi($CPI41_0) ; MIPS32EL-NEXT: ldc1 $f0, %lo($CPI41_0)($1) ; MIPS32EL-NEXT: ldc1 $f2, 0($sp) ; MIPS32EL-NEXT: sub.d $f0, $f2, $f0 ; MIPS32EL-NEXT: cvt.s.d $f0, $f0 ; MIPS32EL-NEXT: lwc1 $f1, 28($sp) ; MIPS32EL-NEXT: lwc1 $f2, 24($sp) ; MIPS32EL-NEXT: add.s $f2, $f0, $f2 ; MIPS32EL-NEXT: add.s $f1, $f0, $f1 ; MIPS32EL-NEXT: swc1 $f1, 12($4) ; MIPS32EL-NEXT: swc1 $f2, 8($4) ; MIPS32EL-NEXT: mtc1 $7, $f1 ; MIPS32EL-NEXT: add.s $f1, $f0, $f1 ; MIPS32EL-NEXT: swc1 $f1, 4($4) ; MIPS32EL-NEXT: mtc1 $6, $f1 ; MIPS32EL-NEXT: add.s $f0, $f0, $f1 ; MIPS32EL-NEXT: swc1 $f0, 0($4) ; MIPS32EL-NEXT: addiu $sp, $sp, 8 ; MIPS32EL-NEXT: jr $ra ; MIPS32EL-NEXT: nop ; ; MIPS64EL-LABEL: mixed_32: ; MIPS64EL: # %bb.0: # %entry ; MIPS64EL-NEXT: daddiu $sp, $sp, -16 ; MIPS64EL-NEXT: .cfi_def_cfa_offset 16 ; MIPS64EL-NEXT: lui $1, %hi(%neg(%gp_rel(mixed_32))) ; MIPS64EL-NEXT: daddu $1, $1, $25 ; MIPS64EL-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(mixed_32))) ; MIPS64EL-NEXT: lui $2, 17200 ; MIPS64EL-NEXT: sw $2, 12($sp) ; MIPS64EL-NEXT: sll $2, $6, 0 ; MIPS64EL-NEXT: sw $2, 8($sp) ; MIPS64EL-NEXT: ld $1, %got_page(.LCPI41_0)($1) ; MIPS64EL-NEXT: ldc1 $f0, %got_ofst(.LCPI41_0)($1) ; MIPS64EL-NEXT: ldc1 $f1, 8($sp) ; MIPS64EL-NEXT: sub.d $f0, $f1, $f0 ; MIPS64EL-NEXT: cvt.s.d $f0, $f0 ; MIPS64EL-NEXT: dsrl $1, $4, 32 ; MIPS64EL-NEXT: sll $1, $1, 0 ; MIPS64EL-NEXT: mtc1 $1, $f1 ; MIPS64EL-NEXT: add.s $f1, $f0, $f1 ; MIPS64EL-NEXT: dsrl $1, $5, 32 ; MIPS64EL-NEXT: mfc1 $2, $f1 ; MIPS64EL-NEXT: sll $3, $4, 0 ; MIPS64EL-NEXT: sll $1, $1, 0 ; MIPS64EL-NEXT: mtc1 $1, $f1 ; MIPS64EL-NEXT: add.s $f1, $f0, $f1 ; MIPS64EL-NEXT: mfc1 $1, $f1 ; MIPS64EL-NEXT: mtc1 $3, $f1 ; MIPS64EL-NEXT: sll $3, $5, 0 ; MIPS64EL-NEXT: mtc1 $3, $f2 ; MIPS64EL-NEXT: dsll $2, $2, 32 ; MIPS64EL-NEXT: add.s $f1, $f0, $f1 ; MIPS64EL-NEXT: mfc1 $3, $f1 ; MIPS64EL-NEXT: dsll $3, $3, 32 ; MIPS64EL-NEXT: dsrl $3, $3, 32 ; MIPS64EL-NEXT: or $2, $3, $2 ; MIPS64EL-NEXT: dsll $1, $1, 32 ; MIPS64EL-NEXT: add.s $f0, $f0, $f2 ; MIPS64EL-NEXT: mfc1 $3, $f0 ; MIPS64EL-NEXT: dsll $3, $3, 32 ; MIPS64EL-NEXT: dsrl $3, $3, 32 ; MIPS64EL-NEXT: or $3, $3, $1 ; MIPS64EL-NEXT: daddiu $sp, $sp, 16 ; MIPS64EL-NEXT: jr $ra ; MIPS64EL-NEXT: nop ; ; MIPS32R5EL-LABEL: mixed_32: ; MIPS32R5EL: # %bb.0: # %entry ; MIPS32R5EL-NEXT: addiu $sp, $sp, -8 ; MIPS32R5EL-NEXT: .cfi_def_cfa_offset 8 ; MIPS32R5EL-NEXT: lui $1, 17200 ; MIPS32R5EL-NEXT: sw $1, 4($sp) ; MIPS32R5EL-NEXT: lw $1, 32($sp) ; MIPS32R5EL-NEXT: sw $1, 0($sp) ; MIPS32R5EL-NEXT: lui $1, %hi($CPI41_0) ; MIPS32R5EL-NEXT: ldc1 $f0, %lo($CPI41_0)($1) ; MIPS32R5EL-NEXT: ldc1 $f1, 0($sp) ; MIPS32R5EL-NEXT: sub.d $f0, $f1, $f0 ; MIPS32R5EL-NEXT: cvt.s.d $f0, $f0 ; MIPS32R5EL-NEXT: ldi.b $w1, 0 ; MIPS32R5EL-NEXT: splati.w $w0, $w0[0] ; MIPS32R5EL-NEXT: insert.w $w1[0], $6 ; MIPS32R5EL-NEXT: insert.w $w1[1], $7 ; MIPS32R5EL-NEXT: lw $1, 24($sp) ; MIPS32R5EL-NEXT: insert.w $w1[2], $1 ; MIPS32R5EL-NEXT: lw $1, 28($sp) ; MIPS32R5EL-NEXT: insert.w $w1[3], $1 ; MIPS32R5EL-NEXT: fadd.w $w0, $w0, $w1 ; MIPS32R5EL-NEXT: st.w $w0, 0($4) ; MIPS32R5EL-NEXT: addiu $sp, $sp, 8 ; MIPS32R5EL-NEXT: jr $ra ; MIPS32R5EL-NEXT: nop ; ; MIPS64R5EL-LABEL: mixed_32: ; MIPS64R5EL: # %bb.0: # %entry ; MIPS64R5EL-NEXT: daddiu $sp, $sp, -16 ; MIPS64R5EL-NEXT: .cfi_def_cfa_offset 16 ; MIPS64R5EL-NEXT: lui $1, %hi(%neg(%gp_rel(mixed_32))) ; MIPS64R5EL-NEXT: daddu $1, $1, $25 ; MIPS64R5EL-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(mixed_32))) ; MIPS64R5EL-NEXT: lui $2, 17200 ; MIPS64R5EL-NEXT: sw $2, 12($sp) ; MIPS64R5EL-NEXT: sll $2, $6, 0 ; MIPS64R5EL-NEXT: sw $2, 8($sp) ; MIPS64R5EL-NEXT: ld $1, %got_page(.LCPI41_0)($1) ; MIPS64R5EL-NEXT: ldc1 $f0, %got_ofst(.LCPI41_0)($1) ; MIPS64R5EL-NEXT: ldc1 $f1, 8($sp) ; MIPS64R5EL-NEXT: sub.d $f0, $f1, $f0 ; MIPS64R5EL-NEXT: ldi.b $w1, 0 ; MIPS64R5EL-NEXT: insert.d $w1[0], $4 ; MIPS64R5EL-NEXT: insert.d $w1[1], $5 ; MIPS64R5EL-NEXT: cvt.s.d $f0, $f0 ; MIPS64R5EL-NEXT: splati.w $w0, $w0[0] ; MIPS64R5EL-NEXT: fadd.w $w0, $w0, $w1 ; MIPS64R5EL-NEXT: copy_s.d $2, $w0[0] ; MIPS64R5EL-NEXT: copy_s.d $3, $w0[1] ; MIPS64R5EL-NEXT: daddiu $sp, $sp, 16 ; MIPS64R5EL-NEXT: jr $ra ; MIPS64R5EL-NEXT: nop entry: %0 = uitofp i32 %b to float %1 = insertelement <4 x float> undef, float %0, i32 0 %2 = insertelement <4 x float> %1, float %0, i32 1 %3 = insertelement <4 x float> %2, float %0, i32 2 %4 = insertelement <4 x float> %3, float %0, i32 3 %5 = fadd <4 x float> %4, %a ret <4 x float> %5 } ; This test is slightly more fragile than I'd like as the offset into the ; outgoing arguments area is dependant on the size of the stack frame for ; this function. define <4 x float> @cast(<4 x i32> %a) { ; MIPS32EB-LABEL: cast: ; MIPS32EB: # %bb.0: # %entry ; MIPS32EB-NEXT: addiu $sp, $sp, -32 ; MIPS32EB-NEXT: .cfi_def_cfa_offset 32 ; MIPS32EB-NEXT: lw $1, 52($sp) ; MIPS32EB-NEXT: lui $2, 17200 ; MIPS32EB-NEXT: sw $2, 24($sp) ; MIPS32EB-NEXT: sw $1, 28($sp) ; MIPS32EB-NEXT: lw $1, 48($sp) ; MIPS32EB-NEXT: sw $2, 16($sp) ; MIPS32EB-NEXT: sw $1, 20($sp) ; MIPS32EB-NEXT: lui $1, %hi($CPI42_0) ; MIPS32EB-NEXT: sw $2, 8($sp) ; MIPS32EB-NEXT: sw $7, 12($sp) ; MIPS32EB-NEXT: ldc1 $f0, %lo($CPI42_0)($1) ; MIPS32EB-NEXT: ldc1 $f2, 24($sp) ; MIPS32EB-NEXT: sub.d $f2, $f2, $f0 ; MIPS32EB-NEXT: ldc1 $f4, 16($sp) ; MIPS32EB-NEXT: sub.d $f4, $f4, $f0 ; MIPS32EB-NEXT: ldc1 $f6, 8($sp) ; MIPS32EB-NEXT: sub.d $f6, $f6, $f0 ; MIPS32EB-NEXT: cvt.s.d $f6, $f6 ; MIPS32EB-NEXT: cvt.s.d $f4, $f4 ; MIPS32EB-NEXT: cvt.s.d $f2, $f2 ; MIPS32EB-NEXT: swc1 $f2, 12($4) ; MIPS32EB-NEXT: swc1 $f4, 8($4) ; MIPS32EB-NEXT: swc1 $f6, 4($4) ; MIPS32EB-NEXT: sw $2, 0($sp) ; MIPS32EB-NEXT: sw $6, 4($sp) ; MIPS32EB-NEXT: ldc1 $f2, 0($sp) ; MIPS32EB-NEXT: sub.d $f0, $f2, $f0 ; MIPS32EB-NEXT: cvt.s.d $f0, $f0 ; MIPS32EB-NEXT: swc1 $f0, 0($4) ; MIPS32EB-NEXT: addiu $sp, $sp, 32 ; MIPS32EB-NEXT: jr $ra ; MIPS32EB-NEXT: nop ; ; MIPS64EB-LABEL: cast: ; MIPS64EB: # %bb.0: # %entry ; MIPS64EB-NEXT: daddiu $sp, $sp, -32 ; MIPS64EB-NEXT: .cfi_def_cfa_offset 32 ; MIPS64EB-NEXT: lui $1, %hi(%neg(%gp_rel(cast))) ; MIPS64EB-NEXT: daddu $1, $1, $25 ; MIPS64EB-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(cast))) ; MIPS64EB-NEXT: sll $2, $4, 0 ; MIPS64EB-NEXT: lui $3, 17200 ; MIPS64EB-NEXT: sw $3, 0($sp) ; MIPS64EB-NEXT: sw $2, 4($sp) ; MIPS64EB-NEXT: sll $2, $5, 0 ; MIPS64EB-NEXT: sw $3, 8($sp) ; MIPS64EB-NEXT: sw $2, 12($sp) ; MIPS64EB-NEXT: ld $1, %got_page(.LCPI42_0)($1) ; MIPS64EB-NEXT: ldc1 $f0, %got_ofst(.LCPI42_0)($1) ; MIPS64EB-NEXT: ldc1 $f1, 0($sp) ; MIPS64EB-NEXT: sub.d $f1, $f1, $f0 ; MIPS64EB-NEXT: cvt.s.d $f1, $f1 ; MIPS64EB-NEXT: ldc1 $f2, 8($sp) ; MIPS64EB-NEXT: sub.d $f2, $f2, $f0 ; MIPS64EB-NEXT: mfc1 $1, $f1 ; MIPS64EB-NEXT: dsrl $2, $4, 32 ; MIPS64EB-NEXT: sll $2, $2, 0 ; MIPS64EB-NEXT: sw $3, 16($sp) ; MIPS64EB-NEXT: sw $2, 20($sp) ; MIPS64EB-NEXT: sw $3, 24($sp) ; MIPS64EB-NEXT: dsll $1, $1, 32 ; MIPS64EB-NEXT: cvt.s.d $f1, $f2 ; MIPS64EB-NEXT: dsrl $2, $5, 32 ; MIPS64EB-NEXT: sll $2, $2, 0 ; MIPS64EB-NEXT: sw $2, 28($sp) ; MIPS64EB-NEXT: mfc1 $2, $f1 ; MIPS64EB-NEXT: dsll $3, $2, 32 ; MIPS64EB-NEXT: dsrl $1, $1, 32 ; MIPS64EB-NEXT: ldc1 $f1, 16($sp) ; MIPS64EB-NEXT: sub.d $f1, $f1, $f0 ; MIPS64EB-NEXT: cvt.s.d $f1, $f1 ; MIPS64EB-NEXT: mfc1 $2, $f1 ; MIPS64EB-NEXT: dsll $2, $2, 32 ; MIPS64EB-NEXT: or $2, $1, $2 ; MIPS64EB-NEXT: dsrl $1, $3, 32 ; MIPS64EB-NEXT: ldc1 $f1, 24($sp) ; MIPS64EB-NEXT: sub.d $f0, $f1, $f0 ; MIPS64EB-NEXT: cvt.s.d $f0, $f0 ; MIPS64EB-NEXT: mfc1 $3, $f0 ; MIPS64EB-NEXT: dsll $3, $3, 32 ; MIPS64EB-NEXT: or $3, $1, $3 ; MIPS64EB-NEXT: daddiu $sp, $sp, 32 ; MIPS64EB-NEXT: jr $ra ; MIPS64EB-NEXT: nop ; ; MIPS32R5-LABEL: cast: ; MIPS32R5: # %bb.0: # %entry ; MIPS32R5-NEXT: ldi.b $w0, 0 ; MIPS32R5-NEXT: insert.w $w0[0], $6 ; MIPS32R5-NEXT: insert.w $w0[1], $7 ; MIPS32R5-NEXT: lw $1, 16($sp) ; MIPS32R5-NEXT: insert.w $w0[2], $1 ; MIPS32R5-NEXT: lw $1, 20($sp) ; MIPS32R5-NEXT: insert.w $w0[3], $1 ; MIPS32R5-NEXT: ffint_u.w $w0, $w0 ; MIPS32R5-NEXT: st.w $w0, 0($4) ; MIPS32R5-NEXT: jr $ra ; MIPS32R5-NEXT: nop ; ; MIPS64R5EB-LABEL: cast: ; MIPS64R5EB: # %bb.0: # %entry ; MIPS64R5EB-NEXT: ldi.b $w0, 0 ; MIPS64R5EB-NEXT: insert.d $w0[0], $4 ; MIPS64R5EB-NEXT: insert.d $w0[1], $5 ; MIPS64R5EB-NEXT: shf.w $w0, $w0, 177 ; MIPS64R5EB-NEXT: ffint_u.w $w0, $w0 ; MIPS64R5EB-NEXT: shf.w $w0, $w0, 177 ; MIPS64R5EB-NEXT: copy_s.d $2, $w0[0] ; MIPS64R5EB-NEXT: copy_s.d $3, $w0[1] ; MIPS64R5EB-NEXT: jr $ra ; MIPS64R5EB-NEXT: nop ; ; MIPS32EL-LABEL: cast: ; MIPS32EL: # %bb.0: # %entry ; MIPS32EL-NEXT: addiu $sp, $sp, -32 ; MIPS32EL-NEXT: .cfi_def_cfa_offset 32 ; MIPS32EL-NEXT: lw $1, 52($sp) ; MIPS32EL-NEXT: lui $2, 17200 ; MIPS32EL-NEXT: sw $2, 28($sp) ; MIPS32EL-NEXT: sw $1, 24($sp) ; MIPS32EL-NEXT: lw $1, 48($sp) ; MIPS32EL-NEXT: sw $2, 20($sp) ; MIPS32EL-NEXT: sw $1, 16($sp) ; MIPS32EL-NEXT: lui $1, %hi($CPI42_0) ; MIPS32EL-NEXT: sw $2, 12($sp) ; MIPS32EL-NEXT: sw $7, 8($sp) ; MIPS32EL-NEXT: ldc1 $f0, %lo($CPI42_0)($1) ; MIPS32EL-NEXT: ldc1 $f2, 24($sp) ; MIPS32EL-NEXT: sub.d $f2, $f2, $f0 ; MIPS32EL-NEXT: ldc1 $f4, 16($sp) ; MIPS32EL-NEXT: sub.d $f4, $f4, $f0 ; MIPS32EL-NEXT: ldc1 $f6, 8($sp) ; MIPS32EL-NEXT: sub.d $f6, $f6, $f0 ; MIPS32EL-NEXT: cvt.s.d $f6, $f6 ; MIPS32EL-NEXT: cvt.s.d $f4, $f4 ; MIPS32EL-NEXT: cvt.s.d $f2, $f2 ; MIPS32EL-NEXT: swc1 $f2, 12($4) ; MIPS32EL-NEXT: swc1 $f4, 8($4) ; MIPS32EL-NEXT: swc1 $f6, 4($4) ; MIPS32EL-NEXT: sw $2, 4($sp) ; MIPS32EL-NEXT: sw $6, 0($sp) ; MIPS32EL-NEXT: ldc1 $f2, 0($sp) ; MIPS32EL-NEXT: sub.d $f0, $f2, $f0 ; MIPS32EL-NEXT: cvt.s.d $f0, $f0 ; MIPS32EL-NEXT: swc1 $f0, 0($4) ; MIPS32EL-NEXT: addiu $sp, $sp, 32 ; MIPS32EL-NEXT: jr $ra ; MIPS32EL-NEXT: nop ; ; MIPS64EL-LABEL: cast: ; MIPS64EL: # %bb.0: # %entry ; MIPS64EL-NEXT: daddiu $sp, $sp, -32 ; MIPS64EL-NEXT: .cfi_def_cfa_offset 32 ; MIPS64EL-NEXT: lui $1, %hi(%neg(%gp_rel(cast))) ; MIPS64EL-NEXT: daddu $1, $1, $25 ; MIPS64EL-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(cast))) ; MIPS64EL-NEXT: sll $2, $4, 0 ; MIPS64EL-NEXT: lui $3, 17200 ; MIPS64EL-NEXT: sw $3, 4($sp) ; MIPS64EL-NEXT: sw $2, 0($sp) ; MIPS64EL-NEXT: sll $2, $5, 0 ; MIPS64EL-NEXT: sw $3, 12($sp) ; MIPS64EL-NEXT: sw $2, 8($sp) ; MIPS64EL-NEXT: ld $1, %got_page(.LCPI42_0)($1) ; MIPS64EL-NEXT: ldc1 $f0, %got_ofst(.LCPI42_0)($1) ; MIPS64EL-NEXT: ldc1 $f1, 0($sp) ; MIPS64EL-NEXT: sub.d $f1, $f1, $f0 ; MIPS64EL-NEXT: cvt.s.d $f1, $f1 ; MIPS64EL-NEXT: ldc1 $f2, 8($sp) ; MIPS64EL-NEXT: sub.d $f2, $f2, $f0 ; MIPS64EL-NEXT: mfc1 $1, $f1 ; MIPS64EL-NEXT: dsrl $2, $4, 32 ; MIPS64EL-NEXT: sll $2, $2, 0 ; MIPS64EL-NEXT: sw $3, 20($sp) ; MIPS64EL-NEXT: sw $2, 16($sp) ; MIPS64EL-NEXT: sw $3, 28($sp) ; MIPS64EL-NEXT: dsll $1, $1, 32 ; MIPS64EL-NEXT: cvt.s.d $f1, $f2 ; MIPS64EL-NEXT: dsrl $2, $5, 32 ; MIPS64EL-NEXT: sll $2, $2, 0 ; MIPS64EL-NEXT: sw $2, 24($sp) ; MIPS64EL-NEXT: mfc1 $2, $f1 ; MIPS64EL-NEXT: dsll $3, $2, 32 ; MIPS64EL-NEXT: dsrl $1, $1, 32 ; MIPS64EL-NEXT: ldc1 $f1, 16($sp) ; MIPS64EL-NEXT: sub.d $f1, $f1, $f0 ; MIPS64EL-NEXT: cvt.s.d $f1, $f1 ; MIPS64EL-NEXT: mfc1 $2, $f1 ; MIPS64EL-NEXT: dsll $2, $2, 32 ; MIPS64EL-NEXT: or $2, $1, $2 ; MIPS64EL-NEXT: dsrl $1, $3, 32 ; MIPS64EL-NEXT: ldc1 $f1, 24($sp) ; MIPS64EL-NEXT: sub.d $f0, $f1, $f0 ; MIPS64EL-NEXT: cvt.s.d $f0, $f0 ; MIPS64EL-NEXT: mfc1 $3, $f0 ; MIPS64EL-NEXT: dsll $3, $3, 32 ; MIPS64EL-NEXT: or $3, $1, $3 ; MIPS64EL-NEXT: daddiu $sp, $sp, 32 ; MIPS64EL-NEXT: jr $ra ; MIPS64EL-NEXT: nop ; ; MIPS64R5EL-LABEL: cast: ; MIPS64R5EL: # %bb.0: # %entry ; MIPS64R5EL-NEXT: ldi.b $w0, 0 ; MIPS64R5EL-NEXT: insert.d $w0[0], $4 ; MIPS64R5EL-NEXT: insert.d $w0[1], $5 ; MIPS64R5EL-NEXT: ffint_u.w $w0, $w0 ; MIPS64R5EL-NEXT: copy_s.d $2, $w0[0] ; MIPS64R5EL-NEXT: copy_s.d $3, $w0[1] ; MIPS64R5EL-NEXT: jr $ra ; MIPS64R5EL-NEXT: nop entry: %0 = uitofp <4 x i32> %a to <4 x float> ret <4 x float> %0 } define <4 x float> @select(<4 x i32> %cond, <4 x float> %arg1, <4 x float> %arg2) { ; MIPS32-LABEL: select: ; MIPS32: # %bb.0: # %entry ; MIPS32-NEXT: andi $1, $7, 1 ; MIPS32-NEXT: lw $2, 16($sp) ; MIPS32-NEXT: andi $2, $2, 1 ; MIPS32-NEXT: addiu $3, $sp, 44 ; MIPS32-NEXT: addiu $5, $sp, 28 ; MIPS32-NEXT: addiu $7, $sp, 48 ; MIPS32-NEXT: addiu $8, $sp, 32 ; MIPS32-NEXT: movn $7, $8, $2 ; MIPS32-NEXT: movn $3, $5, $1 ; MIPS32-NEXT: andi $1, $6, 1 ; MIPS32-NEXT: addiu $2, $sp, 40 ; MIPS32-NEXT: addiu $5, $sp, 24 ; MIPS32-NEXT: movn $2, $5, $1 ; MIPS32-NEXT: lw $1, 20($sp) ; MIPS32-NEXT: lwc1 $f0, 0($2) ; MIPS32-NEXT: lwc1 $f1, 0($3) ; MIPS32-NEXT: lwc1 $f2, 0($7) ; MIPS32-NEXT: andi $1, $1, 1 ; MIPS32-NEXT: addiu $2, $sp, 52 ; MIPS32-NEXT: addiu $3, $sp, 36 ; MIPS32-NEXT: movn $2, $3, $1 ; MIPS32-NEXT: lwc1 $f3, 0($2) ; MIPS32-NEXT: swc1 $f3, 12($4) ; MIPS32-NEXT: swc1 $f2, 8($4) ; MIPS32-NEXT: swc1 $f1, 4($4) ; MIPS32-NEXT: swc1 $f0, 0($4) ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: nop ; ; MIPS64-LABEL: select: ; MIPS64: # %bb.0: # %entry ; MIPS64-NEXT: sll $1, $8, 0 ; MIPS64-NEXT: mtc1 $1, $f0 ; MIPS64-NEXT: sll $1, $4, 0 ; MIPS64-NEXT: andi $1, $1, 1 ; MIPS64-NEXT: sll $2, $6, 0 ; MIPS64-NEXT: mtc1 $2, $f1 ; MIPS64-NEXT: movn.s $f0, $f1, $1 ; MIPS64-NEXT: dsrl $1, $8, 32 ; MIPS64-NEXT: dsrl $2, $4, 32 ; MIPS64-NEXT: sll $1, $1, 0 ; MIPS64-NEXT: mfc1 $3, $f0 ; MIPS64-NEXT: sll $4, $9, 0 ; MIPS64-NEXT: mtc1 $1, $f0 ; MIPS64-NEXT: sll $1, $2, 0 ; MIPS64-NEXT: andi $1, $1, 1 ; MIPS64-NEXT: dsrl $2, $6, 32 ; MIPS64-NEXT: sll $2, $2, 0 ; MIPS64-NEXT: mtc1 $2, $f1 ; MIPS64-NEXT: movn.s $f0, $f1, $1 ; MIPS64-NEXT: dsll $1, $3, 32 ; MIPS64-NEXT: mtc1 $4, $f1 ; MIPS64-NEXT: sll $2, $5, 0 ; MIPS64-NEXT: andi $2, $2, 1 ; MIPS64-NEXT: sll $3, $7, 0 ; MIPS64-NEXT: mtc1 $3, $f2 ; MIPS64-NEXT: movn.s $f1, $f2, $2 ; MIPS64-NEXT: mfc1 $2, $f1 ; MIPS64-NEXT: dsll $3, $2, 32 ; MIPS64-NEXT: dsrl $1, $1, 32 ; MIPS64-NEXT: mfc1 $2, $f0 ; MIPS64-NEXT: dsrl $4, $5, 32 ; MIPS64-NEXT: dsrl $5, $9, 32 ; MIPS64-NEXT: dsll $2, $2, 32 ; MIPS64-NEXT: sll $5, $5, 0 ; MIPS64-NEXT: or $2, $1, $2 ; MIPS64-NEXT: dsrl $1, $3, 32 ; MIPS64-NEXT: mtc1 $5, $f0 ; MIPS64-NEXT: sll $3, $4, 0 ; MIPS64-NEXT: andi $3, $3, 1 ; MIPS64-NEXT: dsrl $4, $7, 32 ; MIPS64-NEXT: sll $4, $4, 0 ; MIPS64-NEXT: mtc1 $4, $f1 ; MIPS64-NEXT: movn.s $f0, $f1, $3 ; MIPS64-NEXT: mfc1 $3, $f0 ; MIPS64-NEXT: dsll $3, $3, 32 ; MIPS64-NEXT: or $3, $1, $3 ; MIPS64-NEXT: jr $ra ; MIPS64-NEXT: nop ; ; MIPS32R5-LABEL: select: ; MIPS32R5: # %bb.0: # %entry ; MIPS32R5-NEXT: ldi.b $w0, 0 ; MIPS32R5-NEXT: lw $1, 44($sp) ; MIPS32R5-NEXT: lw $2, 40($sp) ; MIPS32R5-NEXT: move.v $w1, $w0 ; MIPS32R5-NEXT: insert.w $w1[0], $2 ; MIPS32R5-NEXT: insert.w $w1[1], $1 ; MIPS32R5-NEXT: lw $1, 48($sp) ; MIPS32R5-NEXT: insert.w $w1[2], $1 ; MIPS32R5-NEXT: lw $1, 28($sp) ; MIPS32R5-NEXT: lw $2, 52($sp) ; MIPS32R5-NEXT: lw $3, 24($sp) ; MIPS32R5-NEXT: move.v $w2, $w0 ; MIPS32R5-NEXT: insert.w $w2[0], $3 ; MIPS32R5-NEXT: insert.w $w0[0], $6 ; MIPS32R5-NEXT: insert.w $w1[3], $2 ; MIPS32R5-NEXT: insert.w $w2[1], $1 ; MIPS32R5-NEXT: lw $1, 32($sp) ; MIPS32R5-NEXT: insert.w $w2[2], $1 ; MIPS32R5-NEXT: lw $1, 36($sp) ; MIPS32R5-NEXT: insert.w $w2[3], $1 ; MIPS32R5-NEXT: insert.w $w0[1], $7 ; MIPS32R5-NEXT: lw $1, 16($sp) ; MIPS32R5-NEXT: insert.w $w0[2], $1 ; MIPS32R5-NEXT: lw $1, 20($sp) ; MIPS32R5-NEXT: insert.w $w0[3], $1 ; MIPS32R5-NEXT: slli.w $w0, $w0, 31 ; MIPS32R5-NEXT: srai.w $w0, $w0, 31 ; MIPS32R5-NEXT: bsel.v $w0, $w1, $w2 ; MIPS32R5-NEXT: st.w $w0, 0($4) ; MIPS32R5-NEXT: jr $ra ; MIPS32R5-NEXT: nop ; ; MIPS64R5EB-LABEL: select: ; MIPS64R5EB: # %bb.0: # %entry ; MIPS64R5EB-NEXT: ldi.b $w0, 0 ; MIPS64R5EB-NEXT: move.v $w1, $w0 ; MIPS64R5EB-NEXT: insert.d $w1[0], $8 ; MIPS64R5EB-NEXT: insert.d $w1[1], $9 ; MIPS64R5EB-NEXT: shf.w $w1, $w1, 177 ; MIPS64R5EB-NEXT: move.v $w2, $w0 ; MIPS64R5EB-NEXT: insert.d $w2[0], $6 ; MIPS64R5EB-NEXT: insert.d $w2[1], $7 ; MIPS64R5EB-NEXT: shf.w $w2, $w2, 177 ; MIPS64R5EB-NEXT: insert.d $w0[0], $4 ; MIPS64R5EB-NEXT: insert.d $w0[1], $5 ; MIPS64R5EB-NEXT: shf.w $w0, $w0, 177 ; MIPS64R5EB-NEXT: slli.w $w0, $w0, 31 ; MIPS64R5EB-NEXT: srai.w $w0, $w0, 31 ; MIPS64R5EB-NEXT: bsel.v $w0, $w1, $w2 ; MIPS64R5EB-NEXT: shf.w $w0, $w0, 177 ; MIPS64R5EB-NEXT: copy_s.d $2, $w0[0] ; MIPS64R5EB-NEXT: copy_s.d $3, $w0[1] ; MIPS64R5EB-NEXT: jr $ra ; MIPS64R5EB-NEXT: nop ; ; MIPS64R5EL-LABEL: select: ; MIPS64R5EL: # %bb.0: # %entry ; MIPS64R5EL-NEXT: ldi.b $w0, 0 ; MIPS64R5EL-NEXT: move.v $w1, $w0 ; MIPS64R5EL-NEXT: insert.d $w1[0], $8 ; MIPS64R5EL-NEXT: insert.d $w1[1], $9 ; MIPS64R5EL-NEXT: move.v $w2, $w0 ; MIPS64R5EL-NEXT: insert.d $w2[0], $6 ; MIPS64R5EL-NEXT: insert.d $w2[1], $7 ; MIPS64R5EL-NEXT: insert.d $w0[0], $4 ; MIPS64R5EL-NEXT: insert.d $w0[1], $5 ; MIPS64R5EL-NEXT: slli.w $w0, $w0, 31 ; MIPS64R5EL-NEXT: srai.w $w0, $w0, 31 ; MIPS64R5EL-NEXT: bsel.v $w0, $w1, $w2 ; MIPS64R5EL-NEXT: copy_s.d $2, $w0[0] ; MIPS64R5EL-NEXT: copy_s.d $3, $w0[1] ; MIPS64R5EL-NEXT: jr $ra ; MIPS64R5EL-NEXT: nop entry: %cond.t = trunc <4 x i32> %cond to <4 x i1> %res = select <4 x i1> %cond.t, <4 x float> %arg1, <4 x float> %arg2 ret <4 x float> %res } Index: vendor/llvm/dist-release_80/test/CodeGen/Mips/gprestore.ll =================================================================== --- vendor/llvm/dist-release_80/test/CodeGen/Mips/gprestore.ll (revision 343793) +++ vendor/llvm/dist-release_80/test/CodeGen/Mips/gprestore.ll (revision 343794) @@ -1,227 +1,227 @@ ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py -; RUN: llc -mtriple=mips-mti-linux-gnu < %s -relocation-model=pic | FileCheck %s --check-prefix=O32 -; RUN: llc -mtriple=mips64-mti-linux-gnu < %s -relocation-model=pic | FileCheck %s --check-prefix=N64 -; RUN: llc -mtriple=mips64-mti-linux-gnu < %s -relocation-model=pic -target-abi n32 | FileCheck %s --check-prefix=N32 -; RUN: llc -mtriple=mips-mti-linux-gnu < %s -relocation-model=pic -O3 | FileCheck %s --check-prefix=O3O32 -; RUN: llc -mtriple=mips64-mti-linux-gnu < %s -relocation-model=pic -O3 | FileCheck %s --check-prefix=O3N64 -; RUN: llc -mtriple=mips64-mti-linux-gnu < %s -relocation-model=pic -target-abi n32 -O3 | FileCheck %s --check-prefix=O3N32 +; RUN: llc -mtriple=mips-mti-linux-gnu < %s -relocation-model=pic -mips-jalr-reloc=false | FileCheck %s --check-prefix=O32 +; RUN: llc -mtriple=mips64-mti-linux-gnu < %s -relocation-model=pic -mips-jalr-reloc=false | FileCheck %s --check-prefix=N64 +; RUN: llc -mtriple=mips64-mti-linux-gnu < %s -relocation-model=pic -target-abi n32 -mips-jalr-reloc=false | FileCheck %s --check-prefix=N32 +; RUN: llc -mtriple=mips-mti-linux-gnu < %s -relocation-model=pic -O3 -mips-jalr-reloc=false | FileCheck %s --check-prefix=O3O32 +; RUN: llc -mtriple=mips64-mti-linux-gnu < %s -relocation-model=pic -O3 -mips-jalr-reloc=false | FileCheck %s --check-prefix=O3N64 +; RUN: llc -mtriple=mips64-mti-linux-gnu < %s -relocation-model=pic -target-abi n32 -O3 -mips-jalr-reloc=false | FileCheck %s --check-prefix=O3N32 ; Test that PIC calls use the $25 register. This is an ABI requirement. @p = external global i32 @q = external global i32 @r = external global i32 define void @f0() nounwind { ; O32-LABEL: f0: ; O32: # %bb.0: # %entry ; O32-NEXT: lui $2, %hi(_gp_disp) ; O32-NEXT: addiu $2, $2, %lo(_gp_disp) ; O32-NEXT: addiu $sp, $sp, -32 ; O32-NEXT: sw $ra, 28($sp) # 4-byte Folded Spill ; O32-NEXT: sw $17, 24($sp) # 4-byte Folded Spill ; O32-NEXT: sw $16, 20($sp) # 4-byte Folded Spill ; O32-NEXT: addu $16, $2, $25 ; O32-NEXT: lw $25, %call16(f1)($16) ; O32-NEXT: jalr $25 ; O32-NEXT: move $gp, $16 ; O32-NEXT: lw $1, %got(p)($16) ; O32-NEXT: lw $4, 0($1) ; O32-NEXT: lw $25, %call16(f2)($16) ; O32-NEXT: jalr $25 ; O32-NEXT: move $gp, $16 ; O32-NEXT: lw $1, %got(q)($16) ; O32-NEXT: lw $17, 0($1) ; O32-NEXT: lw $25, %call16(f2)($16) ; O32-NEXT: jalr $25 ; O32-NEXT: move $4, $17 ; O32-NEXT: lw $1, %got(r)($16) ; O32-NEXT: lw $5, 0($1) ; O32-NEXT: lw $25, %call16(f3)($16) ; O32-NEXT: move $4, $17 ; O32-NEXT: jalr $25 ; O32-NEXT: move $gp, $16 ; O32-NEXT: lw $16, 20($sp) # 4-byte Folded Reload ; O32-NEXT: lw $17, 24($sp) # 4-byte Folded Reload ; O32-NEXT: lw $ra, 28($sp) # 4-byte Folded Reload ; O32-NEXT: jr $ra ; O32-NEXT: addiu $sp, $sp, 32 ; ; N64-LABEL: f0: ; N64: # %bb.0: # %entry ; N64-NEXT: daddiu $sp, $sp, -32 ; N64-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; N64-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; N64-NEXT: sd $16, 8($sp) # 8-byte Folded Spill ; N64-NEXT: lui $1, %hi(%neg(%gp_rel(f0))) ; N64-NEXT: daddu $1, $1, $25 ; N64-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(f0))) ; N64-NEXT: ld $25, %call16(f1)($gp) ; N64-NEXT: jalr $25 ; N64-NEXT: nop ; N64-NEXT: ld $1, %got_disp(p)($gp) ; N64-NEXT: ld $25, %call16(f2)($gp) ; N64-NEXT: jalr $25 ; N64-NEXT: lw $4, 0($1) ; N64-NEXT: ld $1, %got_disp(q)($gp) ; N64-NEXT: lw $16, 0($1) ; N64-NEXT: ld $25, %call16(f2)($gp) ; N64-NEXT: jalr $25 ; N64-NEXT: move $4, $16 ; N64-NEXT: ld $1, %got_disp(r)($gp) ; N64-NEXT: lw $5, 0($1) ; N64-NEXT: ld $25, %call16(f3)($gp) ; N64-NEXT: jalr $25 ; N64-NEXT: move $4, $16 ; N64-NEXT: ld $16, 8($sp) # 8-byte Folded Reload ; N64-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; N64-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; N64-NEXT: jr $ra ; N64-NEXT: daddiu $sp, $sp, 32 ; ; N32-LABEL: f0: ; N32: # %bb.0: # %entry ; N32-NEXT: addiu $sp, $sp, -32 ; N32-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; N32-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; N32-NEXT: sd $16, 8($sp) # 8-byte Folded Spill ; N32-NEXT: lui $1, %hi(%neg(%gp_rel(f0))) ; N32-NEXT: addu $1, $1, $25 ; N32-NEXT: addiu $gp, $1, %lo(%neg(%gp_rel(f0))) ; N32-NEXT: lw $25, %call16(f1)($gp) ; N32-NEXT: jalr $25 ; N32-NEXT: nop ; N32-NEXT: lw $1, %got_disp(p)($gp) ; N32-NEXT: lw $25, %call16(f2)($gp) ; N32-NEXT: jalr $25 ; N32-NEXT: lw $4, 0($1) ; N32-NEXT: lw $1, %got_disp(q)($gp) ; N32-NEXT: lw $16, 0($1) ; N32-NEXT: lw $25, %call16(f2)($gp) ; N32-NEXT: jalr $25 ; N32-NEXT: move $4, $16 ; N32-NEXT: lw $1, %got_disp(r)($gp) ; N32-NEXT: lw $5, 0($1) ; N32-NEXT: lw $25, %call16(f3)($gp) ; N32-NEXT: jalr $25 ; N32-NEXT: move $4, $16 ; N32-NEXT: ld $16, 8($sp) # 8-byte Folded Reload ; N32-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; N32-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; N32-NEXT: jr $ra ; N32-NEXT: addiu $sp, $sp, 32 ; ; O3O32-LABEL: f0: ; O3O32: # %bb.0: # %entry ; O3O32-NEXT: lui $2, %hi(_gp_disp) ; O3O32-NEXT: addiu $2, $2, %lo(_gp_disp) ; O3O32-NEXT: addiu $sp, $sp, -32 ; O3O32-NEXT: sw $ra, 28($sp) # 4-byte Folded Spill ; O3O32-NEXT: sw $17, 24($sp) # 4-byte Folded Spill ; O3O32-NEXT: sw $16, 20($sp) # 4-byte Folded Spill ; O3O32-NEXT: addu $16, $2, $25 ; O3O32-NEXT: lw $25, %call16(f1)($16) ; O3O32-NEXT: jalr $25 ; O3O32-NEXT: move $gp, $16 ; O3O32-NEXT: lw $1, %got(p)($16) ; O3O32-NEXT: lw $25, %call16(f2)($16) ; O3O32-NEXT: move $gp, $16 ; O3O32-NEXT: jalr $25 ; O3O32-NEXT: lw $4, 0($1) ; O3O32-NEXT: lw $1, %got(q)($16) ; O3O32-NEXT: lw $25, %call16(f2)($16) ; O3O32-NEXT: lw $17, 0($1) ; O3O32-NEXT: jalr $25 ; O3O32-NEXT: move $4, $17 ; O3O32-NEXT: lw $1, %got(r)($16) ; O3O32-NEXT: lw $25, %call16(f3)($16) ; O3O32-NEXT: move $4, $17 ; O3O32-NEXT: move $gp, $16 ; O3O32-NEXT: jalr $25 ; O3O32-NEXT: lw $5, 0($1) ; O3O32-NEXT: lw $16, 20($sp) # 4-byte Folded Reload ; O3O32-NEXT: lw $17, 24($sp) # 4-byte Folded Reload ; O3O32-NEXT: lw $ra, 28($sp) # 4-byte Folded Reload ; O3O32-NEXT: jr $ra ; O3O32-NEXT: addiu $sp, $sp, 32 ; ; O3N64-LABEL: f0: ; O3N64: # %bb.0: # %entry ; O3N64-NEXT: daddiu $sp, $sp, -32 ; O3N64-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; O3N64-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; O3N64-NEXT: sd $16, 8($sp) # 8-byte Folded Spill ; O3N64-NEXT: lui $1, %hi(%neg(%gp_rel(f0))) ; O3N64-NEXT: daddu $1, $1, $25 ; O3N64-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(f0))) ; O3N64-NEXT: ld $25, %call16(f1)($gp) ; O3N64-NEXT: jalr $25 ; O3N64-NEXT: nop ; O3N64-NEXT: ld $1, %got_disp(p)($gp) ; O3N64-NEXT: ld $25, %call16(f2)($gp) ; O3N64-NEXT: jalr $25 ; O3N64-NEXT: lw $4, 0($1) ; O3N64-NEXT: ld $1, %got_disp(q)($gp) ; O3N64-NEXT: ld $25, %call16(f2)($gp) ; O3N64-NEXT: lw $16, 0($1) ; O3N64-NEXT: jalr $25 ; O3N64-NEXT: move $4, $16 ; O3N64-NEXT: ld $1, %got_disp(r)($gp) ; O3N64-NEXT: ld $25, %call16(f3)($gp) ; O3N64-NEXT: move $4, $16 ; O3N64-NEXT: jalr $25 ; O3N64-NEXT: lw $5, 0($1) ; O3N64-NEXT: ld $16, 8($sp) # 8-byte Folded Reload ; O3N64-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; O3N64-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; O3N64-NEXT: jr $ra ; O3N64-NEXT: daddiu $sp, $sp, 32 ; ; O3N32-LABEL: f0: ; O3N32: # %bb.0: # %entry ; O3N32-NEXT: addiu $sp, $sp, -32 ; O3N32-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; O3N32-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; O3N32-NEXT: sd $16, 8($sp) # 8-byte Folded Spill ; O3N32-NEXT: lui $1, %hi(%neg(%gp_rel(f0))) ; O3N32-NEXT: addu $1, $1, $25 ; O3N32-NEXT: addiu $gp, $1, %lo(%neg(%gp_rel(f0))) ; O3N32-NEXT: lw $25, %call16(f1)($gp) ; O3N32-NEXT: jalr $25 ; O3N32-NEXT: nop ; O3N32-NEXT: lw $1, %got_disp(p)($gp) ; O3N32-NEXT: lw $25, %call16(f2)($gp) ; O3N32-NEXT: jalr $25 ; O3N32-NEXT: lw $4, 0($1) ; O3N32-NEXT: lw $1, %got_disp(q)($gp) ; O3N32-NEXT: lw $25, %call16(f2)($gp) ; O3N32-NEXT: lw $16, 0($1) ; O3N32-NEXT: jalr $25 ; O3N32-NEXT: move $4, $16 ; O3N32-NEXT: lw $1, %got_disp(r)($gp) ; O3N32-NEXT: lw $25, %call16(f3)($gp) ; O3N32-NEXT: move $4, $16 ; O3N32-NEXT: jalr $25 ; O3N32-NEXT: lw $5, 0($1) ; O3N32-NEXT: ld $16, 8($sp) # 8-byte Folded Reload ; O3N32-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; O3N32-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; O3N32-NEXT: jr $ra ; O3N32-NEXT: addiu $sp, $sp, 32 entry: tail call void @f1() nounwind %tmp = load i32, i32* @p, align 4 tail call void @f2(i32 %tmp) nounwind %tmp1 = load i32, i32* @q, align 4 tail call void @f2(i32 %tmp1) nounwind %tmp2 = load i32, i32* @r, align 4 tail call void @f3(i32 %tmp1, i32 %tmp2) nounwind ret void } declare void @f1() declare void @f2(i32) declare void @f3(i32, i32) Index: vendor/llvm/dist-release_80/test/CodeGen/Mips/llvm-ir/sdiv.ll =================================================================== --- vendor/llvm/dist-release_80/test/CodeGen/Mips/llvm-ir/sdiv.ll (revision 343793) +++ vendor/llvm/dist-release_80/test/CodeGen/Mips/llvm-ir/sdiv.ll (revision 343794) @@ -1,484 +1,486 @@ ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py ; RUN: llc < %s -mtriple=mips -mcpu=mips2 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefixes=GP32,GP32R0R2 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefixes=GP32,GP32R0R2 ; RUN: llc < %s -mtriple=mips -mcpu=mips32 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefixes=GP32,GP32R0R2 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefixes=GP32,GP32R0R2 ; RUN: llc < %s -mtriple=mips -mcpu=mips32r2 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefixes=GP32,GP32R2R5 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefixes=GP32,GP32R2R5 ; RUN: llc < %s -mtriple=mips -mcpu=mips32r3 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefixes=GP32,GP32R2R5 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefixes=GP32,GP32R2R5 ; RUN: llc < %s -mtriple=mips -mcpu=mips32r5 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefixes=GP32,GP32R2R5 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefixes=GP32,GP32R2R5 ; RUN: llc < %s -mtriple=mips -mcpu=mips32r6 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefix=GP32R6 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefix=GP32R6 ; RUN: llc < %s -mtriple=mips64 -mcpu=mips3 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefixes=GP64,GP64R0R1 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefixes=GP64,GP64R0R1 ; RUN: llc < %s -mtriple=mips64 -mcpu=mips4 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefixes=GP64,GP64R0R1 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefixes=GP64,GP64R0R1 ; RUN: llc < %s -mtriple=mips64 -mcpu=mips64 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefixes=GP64,GP64R0R1 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefixes=GP64,GP64R0R1 ; RUN: llc < %s -mtriple=mips64 -mcpu=mips64r2 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefixes=GP64,GP64R2R5 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefixes=GP64,GP64R2R5 ; RUN: llc < %s -mtriple=mips64 -mcpu=mips64r3 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefixes=GP64,GP64R2R5 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefixes=GP64,GP64R2R5 ; RUN: llc < %s -mtriple=mips64 -mcpu=mips64r5 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefixes=GP64,GP64R2R5 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefixes=GP64,GP64R2R5 ; RUN: llc < %s -mtriple=mips64 -mcpu=mips64r6 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefix=GP64R6 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefix=GP64R6 -; RUN: llc < %s -mtriple=mips -mcpu=mips32r3 -mattr=+micromips -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefix=MMR3 -; RUN: llc < %s -mtriple=mips -mcpu=mips32r6 -mattr=+micromips -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefix=MMR6 +; RUN: llc < %s -mtriple=mips -mcpu=mips32r3 -mattr=+micromips \ +; RUN: -relocation-model=pic -mips-jalr-reloc=false | \ +; RUN: FileCheck %s -check-prefix=MMR3 +; RUN: llc < %s -mtriple=mips -mcpu=mips32r6 -mattr=+micromips \ +; RUN: -relocation-model=pic -mips-jalr-reloc=false | \ +; RUN: FileCheck %s -check-prefix=MMR6 define signext i1 @sdiv_i1(i1 signext %a, i1 signext %b) { ; GP32-LABEL: sdiv_i1: ; GP32: # %bb.0: # %entry ; GP32-NEXT: jr $ra ; GP32-NEXT: move $2, $4 ; ; GP32R6-LABEL: sdiv_i1: ; GP32R6: # %bb.0: # %entry ; GP32R6-NEXT: jr $ra ; GP32R6-NEXT: move $2, $4 ; ; GP64-LABEL: sdiv_i1: ; GP64: # %bb.0: # %entry ; GP64-NEXT: jr $ra ; GP64-NEXT: move $2, $4 ; ; GP64R6-LABEL: sdiv_i1: ; GP64R6: # %bb.0: # %entry ; GP64R6-NEXT: jr $ra ; GP64R6-NEXT: move $2, $4 ; ; MMR3-LABEL: sdiv_i1: ; MMR3: # %bb.0: # %entry ; MMR3-NEXT: move $2, $4 ; MMR3-NEXT: jrc $ra ; ; MMR6-LABEL: sdiv_i1: ; MMR6: # %bb.0: # %entry ; MMR6-NEXT: move $2, $4 ; MMR6-NEXT: jrc $ra entry: %r = sdiv i1 %a, %b ret i1 %r } define signext i8 @sdiv_i8(i8 signext %a, i8 signext %b) { ; GP32R0R2-LABEL: sdiv_i8: ; GP32R0R2: # %bb.0: # %entry ; GP32R0R2-NEXT: div $zero, $4, $5 ; GP32R0R2-NEXT: teq $5, $zero, 7 ; GP32R0R2-NEXT: mflo $1 ; GP32R0R2-NEXT: sll $1, $1, 24 ; GP32R0R2-NEXT: jr $ra ; GP32R0R2-NEXT: sra $2, $1, 24 ; ; GP32R2R5-LABEL: sdiv_i8: ; GP32R2R5: # %bb.0: # %entry ; GP32R2R5-NEXT: div $zero, $4, $5 ; GP32R2R5-NEXT: teq $5, $zero, 7 ; GP32R2R5-NEXT: mflo $1 ; GP32R2R5-NEXT: jr $ra ; GP32R2R5-NEXT: seb $2, $1 ; ; GP32R6-LABEL: sdiv_i8: ; GP32R6: # %bb.0: # %entry ; GP32R6-NEXT: div $1, $4, $5 ; GP32R6-NEXT: teq $5, $zero, 7 ; GP32R6-NEXT: jr $ra ; GP32R6-NEXT: seb $2, $1 ; ; GP64R0R1-LABEL: sdiv_i8: ; GP64R0R1: # %bb.0: # %entry ; GP64R0R1-NEXT: div $zero, $4, $5 ; GP64R0R1-NEXT: teq $5, $zero, 7 ; GP64R0R1-NEXT: mflo $1 ; GP64R0R1-NEXT: sll $1, $1, 24 ; GP64R0R1-NEXT: jr $ra ; GP64R0R1-NEXT: sra $2, $1, 24 ; ; GP64R2R5-LABEL: sdiv_i8: ; GP64R2R5: # %bb.0: # %entry ; GP64R2R5-NEXT: div $zero, $4, $5 ; GP64R2R5-NEXT: teq $5, $zero, 7 ; GP64R2R5-NEXT: mflo $1 ; GP64R2R5-NEXT: jr $ra ; GP64R2R5-NEXT: seb $2, $1 ; ; GP64R6-LABEL: sdiv_i8: ; GP64R6: # %bb.0: # %entry ; GP64R6-NEXT: div $1, $4, $5 ; GP64R6-NEXT: teq $5, $zero, 7 ; GP64R6-NEXT: jr $ra ; GP64R6-NEXT: seb $2, $1 ; ; MMR3-LABEL: sdiv_i8: ; MMR3: # %bb.0: # %entry ; MMR3-NEXT: div $zero, $4, $5 ; MMR3-NEXT: teq $5, $zero, 7 ; MMR3-NEXT: mflo16 $1 ; MMR3-NEXT: jr $ra ; MMR3-NEXT: seb $2, $1 ; ; MMR6-LABEL: sdiv_i8: ; MMR6: # %bb.0: # %entry ; MMR6-NEXT: div $1, $4, $5 ; MMR6-NEXT: teq $5, $zero, 7 ; MMR6-NEXT: seb $2, $1 ; MMR6-NEXT: jrc $ra entry: %r = sdiv i8 %a, %b ret i8 %r } define signext i16 @sdiv_i16(i16 signext %a, i16 signext %b) { ; GP32R0R2-LABEL: sdiv_i16: ; GP32R0R2: # %bb.0: # %entry ; GP32R0R2-NEXT: div $zero, $4, $5 ; GP32R0R2-NEXT: teq $5, $zero, 7 ; GP32R0R2-NEXT: mflo $1 ; GP32R0R2-NEXT: sll $1, $1, 16 ; GP32R0R2-NEXT: jr $ra ; GP32R0R2-NEXT: sra $2, $1, 16 ; ; GP32R2R5-LABEL: sdiv_i16: ; GP32R2R5: # %bb.0: # %entry ; GP32R2R5-NEXT: div $zero, $4, $5 ; GP32R2R5-NEXT: teq $5, $zero, 7 ; GP32R2R5-NEXT: mflo $1 ; GP32R2R5-NEXT: jr $ra ; GP32R2R5-NEXT: seh $2, $1 ; ; GP32R6-LABEL: sdiv_i16: ; GP32R6: # %bb.0: # %entry ; GP32R6-NEXT: div $1, $4, $5 ; GP32R6-NEXT: teq $5, $zero, 7 ; GP32R6-NEXT: jr $ra ; GP32R6-NEXT: seh $2, $1 ; ; GP64R0R1-LABEL: sdiv_i16: ; GP64R0R1: # %bb.0: # %entry ; GP64R0R1-NEXT: div $zero, $4, $5 ; GP64R0R1-NEXT: teq $5, $zero, 7 ; GP64R0R1-NEXT: mflo $1 ; GP64R0R1-NEXT: sll $1, $1, 16 ; GP64R0R1-NEXT: jr $ra ; GP64R0R1-NEXT: sra $2, $1, 16 ; ; GP64R2R5-LABEL: sdiv_i16: ; GP64R2R5: # %bb.0: # %entry ; GP64R2R5-NEXT: div $zero, $4, $5 ; GP64R2R5-NEXT: teq $5, $zero, 7 ; GP64R2R5-NEXT: mflo $1 ; GP64R2R5-NEXT: jr $ra ; GP64R2R5-NEXT: seh $2, $1 ; ; GP64R6-LABEL: sdiv_i16: ; GP64R6: # %bb.0: # %entry ; GP64R6-NEXT: div $1, $4, $5 ; GP64R6-NEXT: teq $5, $zero, 7 ; GP64R6-NEXT: jr $ra ; GP64R6-NEXT: seh $2, $1 ; ; MMR3-LABEL: sdiv_i16: ; MMR3: # %bb.0: # %entry ; MMR3-NEXT: div $zero, $4, $5 ; MMR3-NEXT: teq $5, $zero, 7 ; MMR3-NEXT: mflo16 $1 ; MMR3-NEXT: jr $ra ; MMR3-NEXT: seh $2, $1 ; ; MMR6-LABEL: sdiv_i16: ; MMR6: # %bb.0: # %entry ; MMR6-NEXT: div $1, $4, $5 ; MMR6-NEXT: teq $5, $zero, 7 ; MMR6-NEXT: seh $2, $1 ; MMR6-NEXT: jrc $ra entry: %r = sdiv i16 %a, %b ret i16 %r } define signext i32 @sdiv_i32(i32 signext %a, i32 signext %b) { ; GP32-LABEL: sdiv_i32: ; GP32: # %bb.0: # %entry ; GP32-NEXT: div $zero, $4, $5 ; GP32-NEXT: teq $5, $zero, 7 ; GP32-NEXT: jr $ra ; GP32-NEXT: mflo $2 ; ; GP32R6-LABEL: sdiv_i32: ; GP32R6: # %bb.0: # %entry ; GP32R6-NEXT: div $2, $4, $5 ; GP32R6-NEXT: teq $5, $zero, 7 ; GP32R6-NEXT: jrc $ra ; ; GP64-LABEL: sdiv_i32: ; GP64: # %bb.0: # %entry ; GP64-NEXT: div $zero, $4, $5 ; GP64-NEXT: teq $5, $zero, 7 ; GP64-NEXT: jr $ra ; GP64-NEXT: mflo $2 ; ; GP64R6-LABEL: sdiv_i32: ; GP64R6: # %bb.0: # %entry ; GP64R6-NEXT: div $2, $4, $5 ; GP64R6-NEXT: teq $5, $zero, 7 ; GP64R6-NEXT: jrc $ra ; ; MMR3-LABEL: sdiv_i32: ; MMR3: # %bb.0: # %entry ; MMR3-NEXT: div $zero, $4, $5 ; MMR3-NEXT: teq $5, $zero, 7 ; MMR3-NEXT: mflo16 $2 ; MMR3-NEXT: jrc $ra ; ; MMR6-LABEL: sdiv_i32: ; MMR6: # %bb.0: # %entry ; MMR6-NEXT: div $2, $4, $5 ; MMR6-NEXT: teq $5, $zero, 7 ; MMR6-NEXT: jrc $ra entry: %r = sdiv i32 %a, %b ret i32 %r } define signext i64 @sdiv_i64(i64 signext %a, i64 signext %b) { ; GP32-LABEL: sdiv_i64: ; GP32: # %bb.0: # %entry ; GP32-NEXT: lui $2, %hi(_gp_disp) ; GP32-NEXT: addiu $2, $2, %lo(_gp_disp) ; GP32-NEXT: addiu $sp, $sp, -24 ; GP32-NEXT: .cfi_def_cfa_offset 24 ; GP32-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; GP32-NEXT: .cfi_offset 31, -4 ; GP32-NEXT: addu $gp, $2, $25 ; GP32-NEXT: lw $25, %call16(__divdi3)($gp) ; GP32-NEXT: jalr $25 ; GP32-NEXT: nop ; GP32-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; GP32-NEXT: jr $ra ; GP32-NEXT: addiu $sp, $sp, 24 ; ; GP32R6-LABEL: sdiv_i64: ; GP32R6: # %bb.0: # %entry ; GP32R6-NEXT: lui $2, %hi(_gp_disp) ; GP32R6-NEXT: addiu $2, $2, %lo(_gp_disp) ; GP32R6-NEXT: addiu $sp, $sp, -24 ; GP32R6-NEXT: .cfi_def_cfa_offset 24 ; GP32R6-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; GP32R6-NEXT: .cfi_offset 31, -4 ; GP32R6-NEXT: addu $gp, $2, $25 ; GP32R6-NEXT: lw $25, %call16(__divdi3)($gp) ; GP32R6-NEXT: jalrc $25 ; GP32R6-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; GP32R6-NEXT: jr $ra ; GP32R6-NEXT: addiu $sp, $sp, 24 ; ; GP64-LABEL: sdiv_i64: ; GP64: # %bb.0: # %entry ; GP64-NEXT: ddiv $zero, $4, $5 ; GP64-NEXT: teq $5, $zero, 7 ; GP64-NEXT: jr $ra ; GP64-NEXT: mflo $2 ; ; GP64R6-LABEL: sdiv_i64: ; GP64R6: # %bb.0: # %entry ; GP64R6-NEXT: ddiv $2, $4, $5 ; GP64R6-NEXT: teq $5, $zero, 7 ; GP64R6-NEXT: jrc $ra ; ; MMR3-LABEL: sdiv_i64: ; MMR3: # %bb.0: # %entry ; MMR3-NEXT: lui $2, %hi(_gp_disp) ; MMR3-NEXT: addiu $2, $2, %lo(_gp_disp) ; MMR3-NEXT: addiusp -24 ; MMR3-NEXT: .cfi_def_cfa_offset 24 ; MMR3-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; MMR3-NEXT: .cfi_offset 31, -4 ; MMR3-NEXT: addu $2, $2, $25 ; MMR3-NEXT: lw $25, %call16(__divdi3)($2) ; MMR3-NEXT: move $gp, $2 ; MMR3-NEXT: jalr $25 ; MMR3-NEXT: nop ; MMR3-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; MMR3-NEXT: addiusp 24 ; MMR3-NEXT: jrc $ra ; ; MMR6-LABEL: sdiv_i64: ; MMR6: # %bb.0: # %entry ; MMR6-NEXT: lui $2, %hi(_gp_disp) ; MMR6-NEXT: addiu $2, $2, %lo(_gp_disp) ; MMR6-NEXT: addiu $sp, $sp, -24 ; MMR6-NEXT: .cfi_def_cfa_offset 24 ; MMR6-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; MMR6-NEXT: .cfi_offset 31, -4 ; MMR6-NEXT: addu $2, $2, $25 ; MMR6-NEXT: lw $25, %call16(__divdi3)($2) ; MMR6-NEXT: move $gp, $2 ; MMR6-NEXT: jalr $25 ; MMR6-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; MMR6-NEXT: addiu $sp, $sp, 24 ; MMR6-NEXT: jrc $ra entry: %r = sdiv i64 %a, %b ret i64 %r } define signext i128 @sdiv_i128(i128 signext %a, i128 signext %b) { ; GP32-LABEL: sdiv_i128: ; GP32: # %bb.0: # %entry ; GP32-NEXT: lui $2, %hi(_gp_disp) ; GP32-NEXT: addiu $2, $2, %lo(_gp_disp) ; GP32-NEXT: addiu $sp, $sp, -40 ; GP32-NEXT: .cfi_def_cfa_offset 40 ; GP32-NEXT: sw $ra, 36($sp) # 4-byte Folded Spill ; GP32-NEXT: .cfi_offset 31, -4 ; GP32-NEXT: addu $gp, $2, $25 ; GP32-NEXT: lw $1, 60($sp) ; GP32-NEXT: lw $2, 64($sp) ; GP32-NEXT: lw $3, 68($sp) ; GP32-NEXT: sw $3, 28($sp) ; GP32-NEXT: sw $2, 24($sp) ; GP32-NEXT: sw $1, 20($sp) ; GP32-NEXT: lw $1, 56($sp) ; GP32-NEXT: sw $1, 16($sp) ; GP32-NEXT: lw $25, %call16(__divti3)($gp) ; GP32-NEXT: jalr $25 ; GP32-NEXT: nop ; GP32-NEXT: lw $ra, 36($sp) # 4-byte Folded Reload ; GP32-NEXT: jr $ra ; GP32-NEXT: addiu $sp, $sp, 40 ; ; GP32R6-LABEL: sdiv_i128: ; GP32R6: # %bb.0: # %entry ; GP32R6-NEXT: lui $2, %hi(_gp_disp) ; GP32R6-NEXT: addiu $2, $2, %lo(_gp_disp) ; GP32R6-NEXT: addiu $sp, $sp, -40 ; GP32R6-NEXT: .cfi_def_cfa_offset 40 ; GP32R6-NEXT: sw $ra, 36($sp) # 4-byte Folded Spill ; GP32R6-NEXT: .cfi_offset 31, -4 ; GP32R6-NEXT: addu $gp, $2, $25 ; GP32R6-NEXT: lw $1, 60($sp) ; GP32R6-NEXT: lw $2, 64($sp) ; GP32R6-NEXT: lw $3, 68($sp) ; GP32R6-NEXT: sw $3, 28($sp) ; GP32R6-NEXT: sw $2, 24($sp) ; GP32R6-NEXT: sw $1, 20($sp) ; GP32R6-NEXT: lw $1, 56($sp) ; GP32R6-NEXT: sw $1, 16($sp) ; GP32R6-NEXT: lw $25, %call16(__divti3)($gp) ; GP32R6-NEXT: jalrc $25 ; GP32R6-NEXT: lw $ra, 36($sp) # 4-byte Folded Reload ; GP32R6-NEXT: jr $ra ; GP32R6-NEXT: addiu $sp, $sp, 40 ; ; GP64-LABEL: sdiv_i128: ; GP64: # %bb.0: # %entry ; GP64-NEXT: daddiu $sp, $sp, -16 ; GP64-NEXT: .cfi_def_cfa_offset 16 ; GP64-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; GP64-NEXT: sd $gp, 0($sp) # 8-byte Folded Spill ; GP64-NEXT: .cfi_offset 31, -8 ; GP64-NEXT: .cfi_offset 28, -16 ; GP64-NEXT: lui $1, %hi(%neg(%gp_rel(sdiv_i128))) ; GP64-NEXT: daddu $1, $1, $25 ; GP64-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(sdiv_i128))) ; GP64-NEXT: ld $25, %call16(__divti3)($gp) ; GP64-NEXT: jalr $25 ; GP64-NEXT: nop ; GP64-NEXT: ld $gp, 0($sp) # 8-byte Folded Reload ; GP64-NEXT: ld $ra, 8($sp) # 8-byte Folded Reload ; GP64-NEXT: jr $ra ; GP64-NEXT: daddiu $sp, $sp, 16 ; ; GP64R6-LABEL: sdiv_i128: ; GP64R6: # %bb.0: # %entry ; GP64R6-NEXT: daddiu $sp, $sp, -16 ; GP64R6-NEXT: .cfi_def_cfa_offset 16 ; GP64R6-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; GP64R6-NEXT: sd $gp, 0($sp) # 8-byte Folded Spill ; GP64R6-NEXT: .cfi_offset 31, -8 ; GP64R6-NEXT: .cfi_offset 28, -16 ; GP64R6-NEXT: lui $1, %hi(%neg(%gp_rel(sdiv_i128))) ; GP64R6-NEXT: daddu $1, $1, $25 ; GP64R6-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(sdiv_i128))) ; GP64R6-NEXT: ld $25, %call16(__divti3)($gp) ; GP64R6-NEXT: jalrc $25 ; GP64R6-NEXT: ld $gp, 0($sp) # 8-byte Folded Reload ; GP64R6-NEXT: ld $ra, 8($sp) # 8-byte Folded Reload ; GP64R6-NEXT: jr $ra ; GP64R6-NEXT: daddiu $sp, $sp, 16 ; ; MMR3-LABEL: sdiv_i128: ; MMR3: # %bb.0: # %entry ; MMR3-NEXT: lui $2, %hi(_gp_disp) ; MMR3-NEXT: addiu $2, $2, %lo(_gp_disp) ; MMR3-NEXT: addiusp -48 ; MMR3-NEXT: .cfi_def_cfa_offset 48 ; MMR3-NEXT: sw $ra, 44($sp) # 4-byte Folded Spill ; MMR3-NEXT: swp $16, 36($sp) ; MMR3-NEXT: .cfi_offset 31, -4 ; MMR3-NEXT: .cfi_offset 17, -8 ; MMR3-NEXT: .cfi_offset 16, -12 ; MMR3-NEXT: addu $16, $2, $25 ; MMR3-NEXT: move $1, $7 ; MMR3-NEXT: lw $7, 68($sp) ; MMR3-NEXT: lw $17, 72($sp) ; MMR3-NEXT: lw $3, 76($sp) ; MMR3-NEXT: move $2, $sp ; MMR3-NEXT: sw16 $3, 28($2) ; MMR3-NEXT: sw16 $17, 24($2) ; MMR3-NEXT: sw16 $7, 20($2) ; MMR3-NEXT: lw $3, 64($sp) ; MMR3-NEXT: sw16 $3, 16($2) ; MMR3-NEXT: lw $25, %call16(__divti3)($16) ; MMR3-NEXT: move $7, $1 ; MMR3-NEXT: move $gp, $16 ; MMR3-NEXT: jalr $25 ; MMR3-NEXT: nop ; MMR3-NEXT: lwp $16, 36($sp) ; MMR3-NEXT: lw $ra, 44($sp) # 4-byte Folded Reload ; MMR3-NEXT: addiusp 48 ; MMR3-NEXT: jrc $ra ; ; MMR6-LABEL: sdiv_i128: ; MMR6: # %bb.0: # %entry ; MMR6-NEXT: lui $2, %hi(_gp_disp) ; MMR6-NEXT: addiu $2, $2, %lo(_gp_disp) ; MMR6-NEXT: addiu $sp, $sp, -48 ; MMR6-NEXT: .cfi_def_cfa_offset 48 ; MMR6-NEXT: sw $ra, 44($sp) # 4-byte Folded Spill ; MMR6-NEXT: sw $17, 40($sp) # 4-byte Folded Spill ; MMR6-NEXT: sw $16, 36($sp) # 4-byte Folded Spill ; MMR6-NEXT: .cfi_offset 31, -4 ; MMR6-NEXT: .cfi_offset 17, -8 ; MMR6-NEXT: .cfi_offset 16, -12 ; MMR6-NEXT: addu $16, $2, $25 ; MMR6-NEXT: move $1, $7 ; MMR6-NEXT: lw $7, 68($sp) ; MMR6-NEXT: lw $17, 72($sp) ; MMR6-NEXT: lw $3, 76($sp) ; MMR6-NEXT: move $2, $sp ; MMR6-NEXT: sw16 $3, 28($2) ; MMR6-NEXT: sw16 $17, 24($2) ; MMR6-NEXT: sw16 $7, 20($2) ; MMR6-NEXT: lw $3, 64($sp) ; MMR6-NEXT: sw16 $3, 16($2) ; MMR6-NEXT: lw $25, %call16(__divti3)($16) ; MMR6-NEXT: move $7, $1 ; MMR6-NEXT: move $gp, $16 ; MMR6-NEXT: jalr $25 ; MMR6-NEXT: lw $16, 36($sp) # 4-byte Folded Reload ; MMR6-NEXT: lw $17, 40($sp) # 4-byte Folded Reload ; MMR6-NEXT: lw $ra, 44($sp) # 4-byte Folded Reload ; MMR6-NEXT: addiu $sp, $sp, 48 ; MMR6-NEXT: jrc $ra entry: %r = sdiv i128 %a, %b ret i128 %r } Index: vendor/llvm/dist-release_80/test/CodeGen/Mips/llvm-ir/srem.ll =================================================================== --- vendor/llvm/dist-release_80/test/CodeGen/Mips/llvm-ir/srem.ll (revision 343793) +++ vendor/llvm/dist-release_80/test/CodeGen/Mips/llvm-ir/srem.ll (revision 343794) @@ -1,484 +1,486 @@ ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py ; RUN: llc < %s -mtriple=mips -mcpu=mips2 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefixes=GP32,GP32R0R2 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefixes=GP32,GP32R0R2 ; RUN: llc < %s -mtriple=mips -mcpu=mips32 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefixes=GP32,GP32R0R2 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefixes=GP32,GP32R0R2 ; RUN: llc < %s -mtriple=mips -mcpu=mips32r2 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefixes=GP32,GP32R2R5 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefixes=GP32,GP32R2R5 ; RUN: llc < %s -mtriple=mips -mcpu=mips32r3 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefixes=GP32,GP32R2R5 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefixes=GP32,GP32R2R5 ; RUN: llc < %s -mtriple=mips -mcpu=mips32r5 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefixes=GP32,GP32R2R5 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefixes=GP32,GP32R2R5 ; RUN: llc < %s -mtriple=mips -mcpu=mips32r6 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefix=GP32R6 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefix=GP32R6 ; RUN: llc < %s -mtriple=mips64 -mcpu=mips3 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefixes=GP64,GP64R0R1 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefixes=GP64,GP64R0R1 ; RUN: llc < %s -mtriple=mips64 -mcpu=mips4 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefixes=GP64,GP64R0R1 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefixes=GP64,GP64R0R1 ; RUN: llc < %s -mtriple=mips64 -mcpu=mips64 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefixes=GP64,GP64R0R1 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefixes=GP64,GP64R0R1 ; RUN: llc < %s -mtriple=mips64 -mcpu=mips64r2 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefixes=GP64,GP64R2R5 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefixes=GP64,GP64R2R5 ; RUN: llc < %s -mtriple=mips64 -mcpu=mips64r3 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefixes=GP64,GP64R2R5 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefixes=GP64,GP64R2R5 ; RUN: llc < %s -mtriple=mips64 -mcpu=mips64r5 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefixes=GP64,GP64R2R5 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefixes=GP64,GP64R2R5 ; RUN: llc < %s -mtriple=mips64 -mcpu=mips64r6 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefix=GP64R6 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefix=GP64R6 -; RUN: llc < %s -mtriple=mips -mcpu=mips32r3 -mattr=+micromips -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefix=MMR3 -; RUN: llc < %s -mtriple=mips -mcpu=mips32r6 -mattr=+micromips -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefix=MMR6 +; RUN: llc < %s -mtriple=mips -mcpu=mips32r3 -mattr=+micromips \ +; RUN: -relocation-model=pic -mips-jalr-reloc=false | \ +; RUN: FileCheck %s -check-prefix=MMR3 +; RUN: llc < %s -mtriple=mips -mcpu=mips32r6 -mattr=+micromips \ +; RUN: -relocation-model=pic -mips-jalr-reloc=false | \ +; RUN: FileCheck %s -check-prefix=MMR6 define signext i1 @srem_i1(i1 signext %a, i1 signext %b) { ; GP32-LABEL: srem_i1: ; GP32: # %bb.0: # %entry ; GP32-NEXT: jr $ra ; GP32-NEXT: addiu $2, $zero, 0 ; ; GP32R6-LABEL: srem_i1: ; GP32R6: # %bb.0: # %entry ; GP32R6-NEXT: jr $ra ; GP32R6-NEXT: addiu $2, $zero, 0 ; ; GP64-LABEL: srem_i1: ; GP64: # %bb.0: # %entry ; GP64-NEXT: jr $ra ; GP64-NEXT: addiu $2, $zero, 0 ; ; GP64R6-LABEL: srem_i1: ; GP64R6: # %bb.0: # %entry ; GP64R6-NEXT: jr $ra ; GP64R6-NEXT: addiu $2, $zero, 0 ; ; MMR3-LABEL: srem_i1: ; MMR3: # %bb.0: # %entry ; MMR3-NEXT: li16 $2, 0 ; MMR3-NEXT: jrc $ra ; ; MMR6-LABEL: srem_i1: ; MMR6: # %bb.0: # %entry ; MMR6-NEXT: li16 $2, 0 ; MMR6-NEXT: jrc $ra entry: %r = srem i1 %a, %b ret i1 %r } define signext i8 @srem_i8(i8 signext %a, i8 signext %b) { ; GP32R0R2-LABEL: srem_i8: ; GP32R0R2: # %bb.0: # %entry ; GP32R0R2-NEXT: div $zero, $4, $5 ; GP32R0R2-NEXT: teq $5, $zero, 7 ; GP32R0R2-NEXT: mfhi $1 ; GP32R0R2-NEXT: sll $1, $1, 24 ; GP32R0R2-NEXT: jr $ra ; GP32R0R2-NEXT: sra $2, $1, 24 ; ; GP32R2R5-LABEL: srem_i8: ; GP32R2R5: # %bb.0: # %entry ; GP32R2R5-NEXT: div $zero, $4, $5 ; GP32R2R5-NEXT: teq $5, $zero, 7 ; GP32R2R5-NEXT: mfhi $1 ; GP32R2R5-NEXT: jr $ra ; GP32R2R5-NEXT: seb $2, $1 ; ; GP32R6-LABEL: srem_i8: ; GP32R6: # %bb.0: # %entry ; GP32R6-NEXT: mod $1, $4, $5 ; GP32R6-NEXT: teq $5, $zero, 7 ; GP32R6-NEXT: jr $ra ; GP32R6-NEXT: seb $2, $1 ; ; GP64R0R1-LABEL: srem_i8: ; GP64R0R1: # %bb.0: # %entry ; GP64R0R1-NEXT: div $zero, $4, $5 ; GP64R0R1-NEXT: teq $5, $zero, 7 ; GP64R0R1-NEXT: mfhi $1 ; GP64R0R1-NEXT: sll $1, $1, 24 ; GP64R0R1-NEXT: jr $ra ; GP64R0R1-NEXT: sra $2, $1, 24 ; ; GP64R2R5-LABEL: srem_i8: ; GP64R2R5: # %bb.0: # %entry ; GP64R2R5-NEXT: div $zero, $4, $5 ; GP64R2R5-NEXT: teq $5, $zero, 7 ; GP64R2R5-NEXT: mfhi $1 ; GP64R2R5-NEXT: jr $ra ; GP64R2R5-NEXT: seb $2, $1 ; ; GP64R6-LABEL: srem_i8: ; GP64R6: # %bb.0: # %entry ; GP64R6-NEXT: mod $1, $4, $5 ; GP64R6-NEXT: teq $5, $zero, 7 ; GP64R6-NEXT: jr $ra ; GP64R6-NEXT: seb $2, $1 ; ; MMR3-LABEL: srem_i8: ; MMR3: # %bb.0: # %entry ; MMR3-NEXT: div $zero, $4, $5 ; MMR3-NEXT: teq $5, $zero, 7 ; MMR3-NEXT: mfhi16 $1 ; MMR3-NEXT: jr $ra ; MMR3-NEXT: seb $2, $1 ; ; MMR6-LABEL: srem_i8: ; MMR6: # %bb.0: # %entry ; MMR6-NEXT: mod $1, $4, $5 ; MMR6-NEXT: teq $5, $zero, 7 ; MMR6-NEXT: seb $2, $1 ; MMR6-NEXT: jrc $ra entry: %r = srem i8 %a, %b ret i8 %r } define signext i16 @srem_i16(i16 signext %a, i16 signext %b) { ; GP32R0R2-LABEL: srem_i16: ; GP32R0R2: # %bb.0: # %entry ; GP32R0R2-NEXT: div $zero, $4, $5 ; GP32R0R2-NEXT: teq $5, $zero, 7 ; GP32R0R2-NEXT: mfhi $1 ; GP32R0R2-NEXT: sll $1, $1, 16 ; GP32R0R2-NEXT: jr $ra ; GP32R0R2-NEXT: sra $2, $1, 16 ; ; GP32R2R5-LABEL: srem_i16: ; GP32R2R5: # %bb.0: # %entry ; GP32R2R5-NEXT: div $zero, $4, $5 ; GP32R2R5-NEXT: teq $5, $zero, 7 ; GP32R2R5-NEXT: mfhi $1 ; GP32R2R5-NEXT: jr $ra ; GP32R2R5-NEXT: seh $2, $1 ; ; GP32R6-LABEL: srem_i16: ; GP32R6: # %bb.0: # %entry ; GP32R6-NEXT: mod $1, $4, $5 ; GP32R6-NEXT: teq $5, $zero, 7 ; GP32R6-NEXT: jr $ra ; GP32R6-NEXT: seh $2, $1 ; ; GP64R0R1-LABEL: srem_i16: ; GP64R0R1: # %bb.0: # %entry ; GP64R0R1-NEXT: div $zero, $4, $5 ; GP64R0R1-NEXT: teq $5, $zero, 7 ; GP64R0R1-NEXT: mfhi $1 ; GP64R0R1-NEXT: sll $1, $1, 16 ; GP64R0R1-NEXT: jr $ra ; GP64R0R1-NEXT: sra $2, $1, 16 ; ; GP64R2R5-LABEL: srem_i16: ; GP64R2R5: # %bb.0: # %entry ; GP64R2R5-NEXT: div $zero, $4, $5 ; GP64R2R5-NEXT: teq $5, $zero, 7 ; GP64R2R5-NEXT: mfhi $1 ; GP64R2R5-NEXT: jr $ra ; GP64R2R5-NEXT: seh $2, $1 ; ; GP64R6-LABEL: srem_i16: ; GP64R6: # %bb.0: # %entry ; GP64R6-NEXT: mod $1, $4, $5 ; GP64R6-NEXT: teq $5, $zero, 7 ; GP64R6-NEXT: jr $ra ; GP64R6-NEXT: seh $2, $1 ; ; MMR3-LABEL: srem_i16: ; MMR3: # %bb.0: # %entry ; MMR3-NEXT: div $zero, $4, $5 ; MMR3-NEXT: teq $5, $zero, 7 ; MMR3-NEXT: mfhi16 $1 ; MMR3-NEXT: jr $ra ; MMR3-NEXT: seh $2, $1 ; ; MMR6-LABEL: srem_i16: ; MMR6: # %bb.0: # %entry ; MMR6-NEXT: mod $1, $4, $5 ; MMR6-NEXT: teq $5, $zero, 7 ; MMR6-NEXT: seh $2, $1 ; MMR6-NEXT: jrc $ra entry: %r = srem i16 %a, %b ret i16 %r } define signext i32 @srem_i32(i32 signext %a, i32 signext %b) { ; GP32-LABEL: srem_i32: ; GP32: # %bb.0: # %entry ; GP32-NEXT: div $zero, $4, $5 ; GP32-NEXT: teq $5, $zero, 7 ; GP32-NEXT: jr $ra ; GP32-NEXT: mfhi $2 ; ; GP32R6-LABEL: srem_i32: ; GP32R6: # %bb.0: # %entry ; GP32R6-NEXT: mod $2, $4, $5 ; GP32R6-NEXT: teq $5, $zero, 7 ; GP32R6-NEXT: jrc $ra ; ; GP64-LABEL: srem_i32: ; GP64: # %bb.0: # %entry ; GP64-NEXT: div $zero, $4, $5 ; GP64-NEXT: teq $5, $zero, 7 ; GP64-NEXT: jr $ra ; GP64-NEXT: mfhi $2 ; ; GP64R6-LABEL: srem_i32: ; GP64R6: # %bb.0: # %entry ; GP64R6-NEXT: mod $2, $4, $5 ; GP64R6-NEXT: teq $5, $zero, 7 ; GP64R6-NEXT: jrc $ra ; ; MMR3-LABEL: srem_i32: ; MMR3: # %bb.0: # %entry ; MMR3-NEXT: div $zero, $4, $5 ; MMR3-NEXT: teq $5, $zero, 7 ; MMR3-NEXT: mfhi16 $2 ; MMR3-NEXT: jrc $ra ; ; MMR6-LABEL: srem_i32: ; MMR6: # %bb.0: # %entry ; MMR6-NEXT: mod $2, $4, $5 ; MMR6-NEXT: teq $5, $zero, 7 ; MMR6-NEXT: jrc $ra entry: %r = srem i32 %a, %b ret i32 %r } define signext i64 @srem_i64(i64 signext %a, i64 signext %b) { ; GP32-LABEL: srem_i64: ; GP32: # %bb.0: # %entry ; GP32-NEXT: lui $2, %hi(_gp_disp) ; GP32-NEXT: addiu $2, $2, %lo(_gp_disp) ; GP32-NEXT: addiu $sp, $sp, -24 ; GP32-NEXT: .cfi_def_cfa_offset 24 ; GP32-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; GP32-NEXT: .cfi_offset 31, -4 ; GP32-NEXT: addu $gp, $2, $25 ; GP32-NEXT: lw $25, %call16(__moddi3)($gp) ; GP32-NEXT: jalr $25 ; GP32-NEXT: nop ; GP32-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; GP32-NEXT: jr $ra ; GP32-NEXT: addiu $sp, $sp, 24 ; ; GP32R6-LABEL: srem_i64: ; GP32R6: # %bb.0: # %entry ; GP32R6-NEXT: lui $2, %hi(_gp_disp) ; GP32R6-NEXT: addiu $2, $2, %lo(_gp_disp) ; GP32R6-NEXT: addiu $sp, $sp, -24 ; GP32R6-NEXT: .cfi_def_cfa_offset 24 ; GP32R6-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; GP32R6-NEXT: .cfi_offset 31, -4 ; GP32R6-NEXT: addu $gp, $2, $25 ; GP32R6-NEXT: lw $25, %call16(__moddi3)($gp) ; GP32R6-NEXT: jalrc $25 ; GP32R6-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; GP32R6-NEXT: jr $ra ; GP32R6-NEXT: addiu $sp, $sp, 24 ; ; GP64-LABEL: srem_i64: ; GP64: # %bb.0: # %entry ; GP64-NEXT: ddiv $zero, $4, $5 ; GP64-NEXT: teq $5, $zero, 7 ; GP64-NEXT: jr $ra ; GP64-NEXT: mfhi $2 ; ; GP64R6-LABEL: srem_i64: ; GP64R6: # %bb.0: # %entry ; GP64R6-NEXT: dmod $2, $4, $5 ; GP64R6-NEXT: teq $5, $zero, 7 ; GP64R6-NEXT: jrc $ra ; ; MMR3-LABEL: srem_i64: ; MMR3: # %bb.0: # %entry ; MMR3-NEXT: lui $2, %hi(_gp_disp) ; MMR3-NEXT: addiu $2, $2, %lo(_gp_disp) ; MMR3-NEXT: addiusp -24 ; MMR3-NEXT: .cfi_def_cfa_offset 24 ; MMR3-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; MMR3-NEXT: .cfi_offset 31, -4 ; MMR3-NEXT: addu $2, $2, $25 ; MMR3-NEXT: lw $25, %call16(__moddi3)($2) ; MMR3-NEXT: move $gp, $2 ; MMR3-NEXT: jalr $25 ; MMR3-NEXT: nop ; MMR3-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; MMR3-NEXT: addiusp 24 ; MMR3-NEXT: jrc $ra ; ; MMR6-LABEL: srem_i64: ; MMR6: # %bb.0: # %entry ; MMR6-NEXT: lui $2, %hi(_gp_disp) ; MMR6-NEXT: addiu $2, $2, %lo(_gp_disp) ; MMR6-NEXT: addiu $sp, $sp, -24 ; MMR6-NEXT: .cfi_def_cfa_offset 24 ; MMR6-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; MMR6-NEXT: .cfi_offset 31, -4 ; MMR6-NEXT: addu $2, $2, $25 ; MMR6-NEXT: lw $25, %call16(__moddi3)($2) ; MMR6-NEXT: move $gp, $2 ; MMR6-NEXT: jalr $25 ; MMR6-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; MMR6-NEXT: addiu $sp, $sp, 24 ; MMR6-NEXT: jrc $ra entry: %r = srem i64 %a, %b ret i64 %r } define signext i128 @srem_i128(i128 signext %a, i128 signext %b) { ; GP32-LABEL: srem_i128: ; GP32: # %bb.0: # %entry ; GP32-NEXT: lui $2, %hi(_gp_disp) ; GP32-NEXT: addiu $2, $2, %lo(_gp_disp) ; GP32-NEXT: addiu $sp, $sp, -40 ; GP32-NEXT: .cfi_def_cfa_offset 40 ; GP32-NEXT: sw $ra, 36($sp) # 4-byte Folded Spill ; GP32-NEXT: .cfi_offset 31, -4 ; GP32-NEXT: addu $gp, $2, $25 ; GP32-NEXT: lw $1, 60($sp) ; GP32-NEXT: lw $2, 64($sp) ; GP32-NEXT: lw $3, 68($sp) ; GP32-NEXT: sw $3, 28($sp) ; GP32-NEXT: sw $2, 24($sp) ; GP32-NEXT: sw $1, 20($sp) ; GP32-NEXT: lw $1, 56($sp) ; GP32-NEXT: sw $1, 16($sp) ; GP32-NEXT: lw $25, %call16(__modti3)($gp) ; GP32-NEXT: jalr $25 ; GP32-NEXT: nop ; GP32-NEXT: lw $ra, 36($sp) # 4-byte Folded Reload ; GP32-NEXT: jr $ra ; GP32-NEXT: addiu $sp, $sp, 40 ; ; GP32R6-LABEL: srem_i128: ; GP32R6: # %bb.0: # %entry ; GP32R6-NEXT: lui $2, %hi(_gp_disp) ; GP32R6-NEXT: addiu $2, $2, %lo(_gp_disp) ; GP32R6-NEXT: addiu $sp, $sp, -40 ; GP32R6-NEXT: .cfi_def_cfa_offset 40 ; GP32R6-NEXT: sw $ra, 36($sp) # 4-byte Folded Spill ; GP32R6-NEXT: .cfi_offset 31, -4 ; GP32R6-NEXT: addu $gp, $2, $25 ; GP32R6-NEXT: lw $1, 60($sp) ; GP32R6-NEXT: lw $2, 64($sp) ; GP32R6-NEXT: lw $3, 68($sp) ; GP32R6-NEXT: sw $3, 28($sp) ; GP32R6-NEXT: sw $2, 24($sp) ; GP32R6-NEXT: sw $1, 20($sp) ; GP32R6-NEXT: lw $1, 56($sp) ; GP32R6-NEXT: sw $1, 16($sp) ; GP32R6-NEXT: lw $25, %call16(__modti3)($gp) ; GP32R6-NEXT: jalrc $25 ; GP32R6-NEXT: lw $ra, 36($sp) # 4-byte Folded Reload ; GP32R6-NEXT: jr $ra ; GP32R6-NEXT: addiu $sp, $sp, 40 ; ; GP64-LABEL: srem_i128: ; GP64: # %bb.0: # %entry ; GP64-NEXT: daddiu $sp, $sp, -16 ; GP64-NEXT: .cfi_def_cfa_offset 16 ; GP64-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; GP64-NEXT: sd $gp, 0($sp) # 8-byte Folded Spill ; GP64-NEXT: .cfi_offset 31, -8 ; GP64-NEXT: .cfi_offset 28, -16 ; GP64-NEXT: lui $1, %hi(%neg(%gp_rel(srem_i128))) ; GP64-NEXT: daddu $1, $1, $25 ; GP64-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(srem_i128))) ; GP64-NEXT: ld $25, %call16(__modti3)($gp) ; GP64-NEXT: jalr $25 ; GP64-NEXT: nop ; GP64-NEXT: ld $gp, 0($sp) # 8-byte Folded Reload ; GP64-NEXT: ld $ra, 8($sp) # 8-byte Folded Reload ; GP64-NEXT: jr $ra ; GP64-NEXT: daddiu $sp, $sp, 16 ; ; GP64R6-LABEL: srem_i128: ; GP64R6: # %bb.0: # %entry ; GP64R6-NEXT: daddiu $sp, $sp, -16 ; GP64R6-NEXT: .cfi_def_cfa_offset 16 ; GP64R6-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; GP64R6-NEXT: sd $gp, 0($sp) # 8-byte Folded Spill ; GP64R6-NEXT: .cfi_offset 31, -8 ; GP64R6-NEXT: .cfi_offset 28, -16 ; GP64R6-NEXT: lui $1, %hi(%neg(%gp_rel(srem_i128))) ; GP64R6-NEXT: daddu $1, $1, $25 ; GP64R6-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(srem_i128))) ; GP64R6-NEXT: ld $25, %call16(__modti3)($gp) ; GP64R6-NEXT: jalrc $25 ; GP64R6-NEXT: ld $gp, 0($sp) # 8-byte Folded Reload ; GP64R6-NEXT: ld $ra, 8($sp) # 8-byte Folded Reload ; GP64R6-NEXT: jr $ra ; GP64R6-NEXT: daddiu $sp, $sp, 16 ; ; MMR3-LABEL: srem_i128: ; MMR3: # %bb.0: # %entry ; MMR3-NEXT: lui $2, %hi(_gp_disp) ; MMR3-NEXT: addiu $2, $2, %lo(_gp_disp) ; MMR3-NEXT: addiusp -48 ; MMR3-NEXT: .cfi_def_cfa_offset 48 ; MMR3-NEXT: sw $ra, 44($sp) # 4-byte Folded Spill ; MMR3-NEXT: swp $16, 36($sp) ; MMR3-NEXT: .cfi_offset 31, -4 ; MMR3-NEXT: .cfi_offset 17, -8 ; MMR3-NEXT: .cfi_offset 16, -12 ; MMR3-NEXT: addu $16, $2, $25 ; MMR3-NEXT: move $1, $7 ; MMR3-NEXT: lw $7, 68($sp) ; MMR3-NEXT: lw $17, 72($sp) ; MMR3-NEXT: lw $3, 76($sp) ; MMR3-NEXT: move $2, $sp ; MMR3-NEXT: sw16 $3, 28($2) ; MMR3-NEXT: sw16 $17, 24($2) ; MMR3-NEXT: sw16 $7, 20($2) ; MMR3-NEXT: lw $3, 64($sp) ; MMR3-NEXT: sw16 $3, 16($2) ; MMR3-NEXT: lw $25, %call16(__modti3)($16) ; MMR3-NEXT: move $7, $1 ; MMR3-NEXT: move $gp, $16 ; MMR3-NEXT: jalr $25 ; MMR3-NEXT: nop ; MMR3-NEXT: lwp $16, 36($sp) ; MMR3-NEXT: lw $ra, 44($sp) # 4-byte Folded Reload ; MMR3-NEXT: addiusp 48 ; MMR3-NEXT: jrc $ra ; ; MMR6-LABEL: srem_i128: ; MMR6: # %bb.0: # %entry ; MMR6-NEXT: lui $2, %hi(_gp_disp) ; MMR6-NEXT: addiu $2, $2, %lo(_gp_disp) ; MMR6-NEXT: addiu $sp, $sp, -48 ; MMR6-NEXT: .cfi_def_cfa_offset 48 ; MMR6-NEXT: sw $ra, 44($sp) # 4-byte Folded Spill ; MMR6-NEXT: sw $17, 40($sp) # 4-byte Folded Spill ; MMR6-NEXT: sw $16, 36($sp) # 4-byte Folded Spill ; MMR6-NEXT: .cfi_offset 31, -4 ; MMR6-NEXT: .cfi_offset 17, -8 ; MMR6-NEXT: .cfi_offset 16, -12 ; MMR6-NEXT: addu $16, $2, $25 ; MMR6-NEXT: move $1, $7 ; MMR6-NEXT: lw $7, 68($sp) ; MMR6-NEXT: lw $17, 72($sp) ; MMR6-NEXT: lw $3, 76($sp) ; MMR6-NEXT: move $2, $sp ; MMR6-NEXT: sw16 $3, 28($2) ; MMR6-NEXT: sw16 $17, 24($2) ; MMR6-NEXT: sw16 $7, 20($2) ; MMR6-NEXT: lw $3, 64($sp) ; MMR6-NEXT: sw16 $3, 16($2) ; MMR6-NEXT: lw $25, %call16(__modti3)($16) ; MMR6-NEXT: move $7, $1 ; MMR6-NEXT: move $gp, $16 ; MMR6-NEXT: jalr $25 ; MMR6-NEXT: lw $16, 36($sp) # 4-byte Folded Reload ; MMR6-NEXT: lw $17, 40($sp) # 4-byte Folded Reload ; MMR6-NEXT: lw $ra, 44($sp) # 4-byte Folded Reload ; MMR6-NEXT: addiu $sp, $sp, 48 ; MMR6-NEXT: jrc $ra entry: %r = srem i128 %a, %b ret i128 %r } Index: vendor/llvm/dist-release_80/test/CodeGen/Mips/llvm-ir/udiv.ll =================================================================== --- vendor/llvm/dist-release_80/test/CodeGen/Mips/llvm-ir/udiv.ll (revision 343793) +++ vendor/llvm/dist-release_80/test/CodeGen/Mips/llvm-ir/udiv.ll (revision 343794) @@ -1,436 +1,438 @@ ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py ; RUN: llc < %s -mtriple=mips -mcpu=mips2 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefixes=GP32,GP32R0R1 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefixes=GP32,GP32R0R1 ; RUN: llc < %s -mtriple=mips -mcpu=mips32 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefixes=GP32,GP32R0R1 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefixes=GP32,GP32R0R1 ; RUN: llc < %s -mtriple=mips -mcpu=mips32r2 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefixes=GP32,GP32R2R5 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefixes=GP32,GP32R2R5 ; RUN: llc < %s -mtriple=mips -mcpu=mips32r3 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefixes=GP32,GP32R2R5 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefixes=GP32,GP32R2R5 ; RUN: llc < %s -mtriple=mips -mcpu=mips32r5 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefixes=GP32,GP32R2R5 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefixes=GP32,GP32R2R5 ; RUN: llc < %s -mtriple=mips -mcpu=mips32r6 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefix=GP32R6 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefix=GP32R6 ; RUN: llc < %s -mtriple=mips64 -mcpu=mips3 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefixes=GP64,GP64R0R1 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefixes=GP64,GP64R0R1 ; RUN: llc < %s -mtriple=mips64 -mcpu=mips4 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefixes=GP64,GP64R0R1 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefixes=GP64,GP64R0R1 ; RUN: llc < %s -mtriple=mips64 -mcpu=mips64 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefixes=GP64,GP64R0R2 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefixes=GP64,GP64R0R2 ; RUN: llc < %s -mtriple=mips64 -mcpu=mips64r2 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefixes=GP64,GP64R2R5 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefixes=GP64,GP64R2R5 ; RUN: llc < %s -mtriple=mips64 -mcpu=mips64r3 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefixes=GP64,GP64R2R5 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefixes=GP64,GP64R2R5 ; RUN: llc < %s -mtriple=mips64 -mcpu=mips64r5 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefixes=GP64,GP64R2R5 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefixes=GP64,GP64R2R5 ; RUN: llc < %s -mtriple=mips64 -mcpu=mips64r6 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefix=GP64R6 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefix=GP64R6 -; RUN: llc < %s -mtriple=mips -mcpu=mips32r3 -mattr=+micromips -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefix=MMR3 -; RUN: llc < %s -mtriple=mips -mcpu=mips32r6 -mattr=+micromips -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefix=MMR6 +; RUN: llc < %s -mtriple=mips -mcpu=mips32r3 -mattr=+micromips \ +; RUN: -relocation-model=pic -mips-jalr-reloc=false | \ +; RUN: FileCheck %s -check-prefix=MMR3 +; RUN: llc < %s -mtriple=mips -mcpu=mips32r6 -mattr=+micromips \ +; RUN: -relocation-model=pic -mips-jalr-reloc=false | \ +; RUN: FileCheck %s -check-prefix=MMR6 define zeroext i1 @udiv_i1(i1 zeroext %a, i1 zeroext %b) { ; GP32-LABEL: udiv_i1: ; GP32: # %bb.0: # %entry ; GP32-NEXT: jr $ra ; GP32-NEXT: move $2, $4 ; ; GP32R6-LABEL: udiv_i1: ; GP32R6: # %bb.0: # %entry ; GP32R6-NEXT: jr $ra ; GP32R6-NEXT: move $2, $4 ; ; GP64-LABEL: udiv_i1: ; GP64: # %bb.0: # %entry ; GP64-NEXT: jr $ra ; GP64-NEXT: move $2, $4 ; ; GP64R6-LABEL: udiv_i1: ; GP64R6: # %bb.0: # %entry ; GP64R6-NEXT: jr $ra ; GP64R6-NEXT: move $2, $4 ; ; MMR3-LABEL: udiv_i1: ; MMR3: # %bb.0: # %entry ; MMR3-NEXT: move $2, $4 ; MMR3-NEXT: jrc $ra ; ; MMR6-LABEL: udiv_i1: ; MMR6: # %bb.0: # %entry ; MMR6-NEXT: move $2, $4 ; MMR6-NEXT: jrc $ra entry: %r = udiv i1 %a, %b ret i1 %r } define zeroext i8 @udiv_i8(i8 zeroext %a, i8 zeroext %b) { ; GP32-LABEL: udiv_i8: ; GP32: # %bb.0: # %entry ; GP32-NEXT: divu $zero, $4, $5 ; GP32-NEXT: teq $5, $zero, 7 ; GP32-NEXT: jr $ra ; GP32-NEXT: mflo $2 ; ; GP32R6-LABEL: udiv_i8: ; GP32R6: # %bb.0: # %entry ; GP32R6-NEXT: divu $2, $4, $5 ; GP32R6-NEXT: teq $5, $zero, 7 ; GP32R6-NEXT: jrc $ra ; ; GP64-LABEL: udiv_i8: ; GP64: # %bb.0: # %entry ; GP64-NEXT: divu $zero, $4, $5 ; GP64-NEXT: teq $5, $zero, 7 ; GP64-NEXT: jr $ra ; GP64-NEXT: mflo $2 ; ; GP64R6-LABEL: udiv_i8: ; GP64R6: # %bb.0: # %entry ; GP64R6-NEXT: divu $2, $4, $5 ; GP64R6-NEXT: teq $5, $zero, 7 ; GP64R6-NEXT: jrc $ra ; ; MMR3-LABEL: udiv_i8: ; MMR3: # %bb.0: # %entry ; MMR3-NEXT: divu $zero, $4, $5 ; MMR3-NEXT: teq $5, $zero, 7 ; MMR3-NEXT: mflo16 $2 ; MMR3-NEXT: jrc $ra ; ; MMR6-LABEL: udiv_i8: ; MMR6: # %bb.0: # %entry ; MMR6-NEXT: divu $2, $4, $5 ; MMR6-NEXT: teq $5, $zero, 7 ; MMR6-NEXT: jrc $ra entry: %r = udiv i8 %a, %b ret i8 %r } define zeroext i16 @udiv_i16(i16 zeroext %a, i16 zeroext %b) { ; GP32-LABEL: udiv_i16: ; GP32: # %bb.0: # %entry ; GP32-NEXT: divu $zero, $4, $5 ; GP32-NEXT: teq $5, $zero, 7 ; GP32-NEXT: jr $ra ; GP32-NEXT: mflo $2 ; ; GP32R6-LABEL: udiv_i16: ; GP32R6: # %bb.0: # %entry ; GP32R6-NEXT: divu $2, $4, $5 ; GP32R6-NEXT: teq $5, $zero, 7 ; GP32R6-NEXT: jrc $ra ; ; GP64-LABEL: udiv_i16: ; GP64: # %bb.0: # %entry ; GP64-NEXT: divu $zero, $4, $5 ; GP64-NEXT: teq $5, $zero, 7 ; GP64-NEXT: jr $ra ; GP64-NEXT: mflo $2 ; ; GP64R6-LABEL: udiv_i16: ; GP64R6: # %bb.0: # %entry ; GP64R6-NEXT: divu $2, $4, $5 ; GP64R6-NEXT: teq $5, $zero, 7 ; GP64R6-NEXT: jrc $ra ; ; MMR3-LABEL: udiv_i16: ; MMR3: # %bb.0: # %entry ; MMR3-NEXT: divu $zero, $4, $5 ; MMR3-NEXT: teq $5, $zero, 7 ; MMR3-NEXT: mflo16 $2 ; MMR3-NEXT: jrc $ra ; ; MMR6-LABEL: udiv_i16: ; MMR6: # %bb.0: # %entry ; MMR6-NEXT: divu $2, $4, $5 ; MMR6-NEXT: teq $5, $zero, 7 ; MMR6-NEXT: jrc $ra entry: %r = udiv i16 %a, %b ret i16 %r } define signext i32 @udiv_i32(i32 signext %a, i32 signext %b) { ; GP32-LABEL: udiv_i32: ; GP32: # %bb.0: # %entry ; GP32-NEXT: divu $zero, $4, $5 ; GP32-NEXT: teq $5, $zero, 7 ; GP32-NEXT: jr $ra ; GP32-NEXT: mflo $2 ; ; GP32R6-LABEL: udiv_i32: ; GP32R6: # %bb.0: # %entry ; GP32R6-NEXT: divu $2, $4, $5 ; GP32R6-NEXT: teq $5, $zero, 7 ; GP32R6-NEXT: jrc $ra ; ; GP64-LABEL: udiv_i32: ; GP64: # %bb.0: # %entry ; GP64-NEXT: divu $zero, $4, $5 ; GP64-NEXT: teq $5, $zero, 7 ; GP64-NEXT: jr $ra ; GP64-NEXT: mflo $2 ; ; GP64R6-LABEL: udiv_i32: ; GP64R6: # %bb.0: # %entry ; GP64R6-NEXT: divu $2, $4, $5 ; GP64R6-NEXT: teq $5, $zero, 7 ; GP64R6-NEXT: jrc $ra ; ; MMR3-LABEL: udiv_i32: ; MMR3: # %bb.0: # %entry ; MMR3-NEXT: divu $zero, $4, $5 ; MMR3-NEXT: teq $5, $zero, 7 ; MMR3-NEXT: mflo16 $2 ; MMR3-NEXT: jrc $ra ; ; MMR6-LABEL: udiv_i32: ; MMR6: # %bb.0: # %entry ; MMR6-NEXT: divu $2, $4, $5 ; MMR6-NEXT: teq $5, $zero, 7 ; MMR6-NEXT: jrc $ra entry: %r = udiv i32 %a, %b ret i32 %r } define signext i64 @udiv_i64(i64 signext %a, i64 signext %b) { ; GP32-LABEL: udiv_i64: ; GP32: # %bb.0: # %entry ; GP32-NEXT: lui $2, %hi(_gp_disp) ; GP32-NEXT: addiu $2, $2, %lo(_gp_disp) ; GP32-NEXT: addiu $sp, $sp, -24 ; GP32-NEXT: .cfi_def_cfa_offset 24 ; GP32-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; GP32-NEXT: .cfi_offset 31, -4 ; GP32-NEXT: addu $gp, $2, $25 ; GP32-NEXT: lw $25, %call16(__udivdi3)($gp) ; GP32-NEXT: jalr $25 ; GP32-NEXT: nop ; GP32-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; GP32-NEXT: jr $ra ; GP32-NEXT: addiu $sp, $sp, 24 ; ; GP32R6-LABEL: udiv_i64: ; GP32R6: # %bb.0: # %entry ; GP32R6-NEXT: lui $2, %hi(_gp_disp) ; GP32R6-NEXT: addiu $2, $2, %lo(_gp_disp) ; GP32R6-NEXT: addiu $sp, $sp, -24 ; GP32R6-NEXT: .cfi_def_cfa_offset 24 ; GP32R6-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; GP32R6-NEXT: .cfi_offset 31, -4 ; GP32R6-NEXT: addu $gp, $2, $25 ; GP32R6-NEXT: lw $25, %call16(__udivdi3)($gp) ; GP32R6-NEXT: jalrc $25 ; GP32R6-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; GP32R6-NEXT: jr $ra ; GP32R6-NEXT: addiu $sp, $sp, 24 ; ; GP64-LABEL: udiv_i64: ; GP64: # %bb.0: # %entry ; GP64-NEXT: ddivu $zero, $4, $5 ; GP64-NEXT: teq $5, $zero, 7 ; GP64-NEXT: jr $ra ; GP64-NEXT: mflo $2 ; ; GP64R6-LABEL: udiv_i64: ; GP64R6: # %bb.0: # %entry ; GP64R6-NEXT: ddivu $2, $4, $5 ; GP64R6-NEXT: teq $5, $zero, 7 ; GP64R6-NEXT: jrc $ra ; ; MMR3-LABEL: udiv_i64: ; MMR3: # %bb.0: # %entry ; MMR3-NEXT: lui $2, %hi(_gp_disp) ; MMR3-NEXT: addiu $2, $2, %lo(_gp_disp) ; MMR3-NEXT: addiusp -24 ; MMR3-NEXT: .cfi_def_cfa_offset 24 ; MMR3-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; MMR3-NEXT: .cfi_offset 31, -4 ; MMR3-NEXT: addu $2, $2, $25 ; MMR3-NEXT: lw $25, %call16(__udivdi3)($2) ; MMR3-NEXT: move $gp, $2 ; MMR3-NEXT: jalr $25 ; MMR3-NEXT: nop ; MMR3-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; MMR3-NEXT: addiusp 24 ; MMR3-NEXT: jrc $ra ; ; MMR6-LABEL: udiv_i64: ; MMR6: # %bb.0: # %entry ; MMR6-NEXT: lui $2, %hi(_gp_disp) ; MMR6-NEXT: addiu $2, $2, %lo(_gp_disp) ; MMR6-NEXT: addiu $sp, $sp, -24 ; MMR6-NEXT: .cfi_def_cfa_offset 24 ; MMR6-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; MMR6-NEXT: .cfi_offset 31, -4 ; MMR6-NEXT: addu $2, $2, $25 ; MMR6-NEXT: lw $25, %call16(__udivdi3)($2) ; MMR6-NEXT: move $gp, $2 ; MMR6-NEXT: jalr $25 ; MMR6-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; MMR6-NEXT: addiu $sp, $sp, 24 ; MMR6-NEXT: jrc $ra entry: %r = udiv i64 %a, %b ret i64 %r } define signext i128 @udiv_i128(i128 signext %a, i128 signext %b) { ; GP32-LABEL: udiv_i128: ; GP32: # %bb.0: # %entry ; GP32-NEXT: lui $2, %hi(_gp_disp) ; GP32-NEXT: addiu $2, $2, %lo(_gp_disp) ; GP32-NEXT: addiu $sp, $sp, -40 ; GP32-NEXT: .cfi_def_cfa_offset 40 ; GP32-NEXT: sw $ra, 36($sp) # 4-byte Folded Spill ; GP32-NEXT: .cfi_offset 31, -4 ; GP32-NEXT: addu $gp, $2, $25 ; GP32-NEXT: lw $1, 60($sp) ; GP32-NEXT: lw $2, 64($sp) ; GP32-NEXT: lw $3, 68($sp) ; GP32-NEXT: sw $3, 28($sp) ; GP32-NEXT: sw $2, 24($sp) ; GP32-NEXT: sw $1, 20($sp) ; GP32-NEXT: lw $1, 56($sp) ; GP32-NEXT: sw $1, 16($sp) ; GP32-NEXT: lw $25, %call16(__udivti3)($gp) ; GP32-NEXT: jalr $25 ; GP32-NEXT: nop ; GP32-NEXT: lw $ra, 36($sp) # 4-byte Folded Reload ; GP32-NEXT: jr $ra ; GP32-NEXT: addiu $sp, $sp, 40 ; ; GP32R6-LABEL: udiv_i128: ; GP32R6: # %bb.0: # %entry ; GP32R6-NEXT: lui $2, %hi(_gp_disp) ; GP32R6-NEXT: addiu $2, $2, %lo(_gp_disp) ; GP32R6-NEXT: addiu $sp, $sp, -40 ; GP32R6-NEXT: .cfi_def_cfa_offset 40 ; GP32R6-NEXT: sw $ra, 36($sp) # 4-byte Folded Spill ; GP32R6-NEXT: .cfi_offset 31, -4 ; GP32R6-NEXT: addu $gp, $2, $25 ; GP32R6-NEXT: lw $1, 60($sp) ; GP32R6-NEXT: lw $2, 64($sp) ; GP32R6-NEXT: lw $3, 68($sp) ; GP32R6-NEXT: sw $3, 28($sp) ; GP32R6-NEXT: sw $2, 24($sp) ; GP32R6-NEXT: sw $1, 20($sp) ; GP32R6-NEXT: lw $1, 56($sp) ; GP32R6-NEXT: sw $1, 16($sp) ; GP32R6-NEXT: lw $25, %call16(__udivti3)($gp) ; GP32R6-NEXT: jalrc $25 ; GP32R6-NEXT: lw $ra, 36($sp) # 4-byte Folded Reload ; GP32R6-NEXT: jr $ra ; GP32R6-NEXT: addiu $sp, $sp, 40 ; ; GP64-LABEL: udiv_i128: ; GP64: # %bb.0: # %entry ; GP64-NEXT: daddiu $sp, $sp, -16 ; GP64-NEXT: .cfi_def_cfa_offset 16 ; GP64-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; GP64-NEXT: sd $gp, 0($sp) # 8-byte Folded Spill ; GP64-NEXT: .cfi_offset 31, -8 ; GP64-NEXT: .cfi_offset 28, -16 ; GP64-NEXT: lui $1, %hi(%neg(%gp_rel(udiv_i128))) ; GP64-NEXT: daddu $1, $1, $25 ; GP64-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(udiv_i128))) ; GP64-NEXT: ld $25, %call16(__udivti3)($gp) ; GP64-NEXT: jalr $25 ; GP64-NEXT: nop ; GP64-NEXT: ld $gp, 0($sp) # 8-byte Folded Reload ; GP64-NEXT: ld $ra, 8($sp) # 8-byte Folded Reload ; GP64-NEXT: jr $ra ; GP64-NEXT: daddiu $sp, $sp, 16 ; ; GP64R6-LABEL: udiv_i128: ; GP64R6: # %bb.0: # %entry ; GP64R6-NEXT: daddiu $sp, $sp, -16 ; GP64R6-NEXT: .cfi_def_cfa_offset 16 ; GP64R6-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; GP64R6-NEXT: sd $gp, 0($sp) # 8-byte Folded Spill ; GP64R6-NEXT: .cfi_offset 31, -8 ; GP64R6-NEXT: .cfi_offset 28, -16 ; GP64R6-NEXT: lui $1, %hi(%neg(%gp_rel(udiv_i128))) ; GP64R6-NEXT: daddu $1, $1, $25 ; GP64R6-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(udiv_i128))) ; GP64R6-NEXT: ld $25, %call16(__udivti3)($gp) ; GP64R6-NEXT: jalrc $25 ; GP64R6-NEXT: ld $gp, 0($sp) # 8-byte Folded Reload ; GP64R6-NEXT: ld $ra, 8($sp) # 8-byte Folded Reload ; GP64R6-NEXT: jr $ra ; GP64R6-NEXT: daddiu $sp, $sp, 16 ; ; MMR3-LABEL: udiv_i128: ; MMR3: # %bb.0: # %entry ; MMR3-NEXT: lui $2, %hi(_gp_disp) ; MMR3-NEXT: addiu $2, $2, %lo(_gp_disp) ; MMR3-NEXT: addiusp -48 ; MMR3-NEXT: .cfi_def_cfa_offset 48 ; MMR3-NEXT: sw $ra, 44($sp) # 4-byte Folded Spill ; MMR3-NEXT: swp $16, 36($sp) ; MMR3-NEXT: .cfi_offset 31, -4 ; MMR3-NEXT: .cfi_offset 17, -8 ; MMR3-NEXT: .cfi_offset 16, -12 ; MMR3-NEXT: addu $16, $2, $25 ; MMR3-NEXT: move $1, $7 ; MMR3-NEXT: lw $7, 68($sp) ; MMR3-NEXT: lw $17, 72($sp) ; MMR3-NEXT: lw $3, 76($sp) ; MMR3-NEXT: move $2, $sp ; MMR3-NEXT: sw16 $3, 28($2) ; MMR3-NEXT: sw16 $17, 24($2) ; MMR3-NEXT: sw16 $7, 20($2) ; MMR3-NEXT: lw $3, 64($sp) ; MMR3-NEXT: sw16 $3, 16($2) ; MMR3-NEXT: lw $25, %call16(__udivti3)($16) ; MMR3-NEXT: move $7, $1 ; MMR3-NEXT: move $gp, $16 ; MMR3-NEXT: jalr $25 ; MMR3-NEXT: nop ; MMR3-NEXT: lwp $16, 36($sp) ; MMR3-NEXT: lw $ra, 44($sp) # 4-byte Folded Reload ; MMR3-NEXT: addiusp 48 ; MMR3-NEXT: jrc $ra ; ; MMR6-LABEL: udiv_i128: ; MMR6: # %bb.0: # %entry ; MMR6-NEXT: lui $2, %hi(_gp_disp) ; MMR6-NEXT: addiu $2, $2, %lo(_gp_disp) ; MMR6-NEXT: addiu $sp, $sp, -48 ; MMR6-NEXT: .cfi_def_cfa_offset 48 ; MMR6-NEXT: sw $ra, 44($sp) # 4-byte Folded Spill ; MMR6-NEXT: sw $17, 40($sp) # 4-byte Folded Spill ; MMR6-NEXT: sw $16, 36($sp) # 4-byte Folded Spill ; MMR6-NEXT: .cfi_offset 31, -4 ; MMR6-NEXT: .cfi_offset 17, -8 ; MMR6-NEXT: .cfi_offset 16, -12 ; MMR6-NEXT: addu $16, $2, $25 ; MMR6-NEXT: move $1, $7 ; MMR6-NEXT: lw $7, 68($sp) ; MMR6-NEXT: lw $17, 72($sp) ; MMR6-NEXT: lw $3, 76($sp) ; MMR6-NEXT: move $2, $sp ; MMR6-NEXT: sw16 $3, 28($2) ; MMR6-NEXT: sw16 $17, 24($2) ; MMR6-NEXT: sw16 $7, 20($2) ; MMR6-NEXT: lw $3, 64($sp) ; MMR6-NEXT: sw16 $3, 16($2) ; MMR6-NEXT: lw $25, %call16(__udivti3)($16) ; MMR6-NEXT: move $7, $1 ; MMR6-NEXT: move $gp, $16 ; MMR6-NEXT: jalr $25 ; MMR6-NEXT: lw $16, 36($sp) # 4-byte Folded Reload ; MMR6-NEXT: lw $17, 40($sp) # 4-byte Folded Reload ; MMR6-NEXT: lw $ra, 44($sp) # 4-byte Folded Reload ; MMR6-NEXT: addiu $sp, $sp, 48 ; MMR6-NEXT: jrc $ra entry: %r = udiv i128 %a, %b ret i128 %r } Index: vendor/llvm/dist-release_80/test/CodeGen/Mips/llvm-ir/urem.ll =================================================================== --- vendor/llvm/dist-release_80/test/CodeGen/Mips/llvm-ir/urem.ll (revision 343793) +++ vendor/llvm/dist-release_80/test/CodeGen/Mips/llvm-ir/urem.ll (revision 343794) @@ -1,516 +1,518 @@ ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py ; RUN: llc < %s -mtriple=mips -mcpu=mips2 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefixes=GP32,GP32R0R2 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefixes=GP32,GP32R0R2 ; RUN: llc < %s -mtriple=mips -mcpu=mips32 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefixes=GP32,GP32R0R2 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefixes=GP32,GP32R0R2 ; RUN: llc < %s -mtriple=mips -mcpu=mips32r2 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefixes=GP32,GP32R2R5 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefixes=GP32,GP32R2R5 ; RUN: llc < %s -mtriple=mips -mcpu=mips32r3 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefixes=GP32,GP32R2R5 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefixes=GP32,GP32R2R5 ; RUN: llc < %s -mtriple=mips -mcpu=mips32r5 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefixes=GP32,GP32R2R5 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefixes=GP32,GP32R2R5 ; RUN: llc < %s -mtriple=mips -mcpu=mips32r6 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefix=GP32R6 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefix=GP32R6 ; RUN: llc < %s -mtriple=mips64 -mcpu=mips3 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefixes=GP64,GP64R0R1 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefixes=GP64,GP64R0R1 ; RUN: llc < %s -mtriple=mips64 -mcpu=mips4 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefixes=GP64,GP64R0R1 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefixes=GP64,GP64R0R1 ; RUN: llc < %s -mtriple=mips64 -mcpu=mips64 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefixes=GP64,GP64R0R1 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefixes=GP64,GP64R0R1 ; RUN: llc < %s -mtriple=mips64 -mcpu=mips64r2 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefixes=GP64,GP64R2R5 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefixes=GP64,GP64R2R5 ; RUN: llc < %s -mtriple=mips64 -mcpu=mips64r3 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefixes=GP64,GP64R2R5 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefixes=GP64,GP64R2R5 ; RUN: llc < %s -mtriple=mips64 -mcpu=mips64r5 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefixes=GP64,GP64R2R5 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefixes=GP64,GP64R2R5 ; RUN: llc < %s -mtriple=mips64 -mcpu=mips64r6 -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefix=GP64R6 +; RUN: -mips-jalr-reloc=false | FileCheck %s -check-prefix=GP64R6 -; RUN: llc < %s -mtriple=mips -mcpu=mips32r3 -mattr=+micromips -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefix=MMR3 -; RUN: llc < %s -mtriple=mips -mcpu=mips32r6 -mattr=+micromips -relocation-model=pic \ -; RUN: | FileCheck %s -check-prefix=MMR6 +; RUN: llc < %s -mtriple=mips -mcpu=mips32r3 -mattr=+micromips \ +; RUN: -relocation-model=pic -mips-jalr-reloc=false | \ +; RUN: FileCheck %s -check-prefix=MMR3 +; RUN: llc < %s -mtriple=mips -mcpu=mips32r6 -mattr=+micromips \ +; RUN: -relocation-model=pic -mips-jalr-reloc=false | \ +; RUN: FileCheck %s -check-prefix=MMR6 define signext i1 @urem_i1(i1 signext %a, i1 signext %b) { ; GP32-LABEL: urem_i1: ; GP32: # %bb.0: # %entry ; GP32-NEXT: jr $ra ; GP32-NEXT: addiu $2, $zero, 0 ; ; GP32R6-LABEL: urem_i1: ; GP32R6: # %bb.0: # %entry ; GP32R6-NEXT: jr $ra ; GP32R6-NEXT: addiu $2, $zero, 0 ; ; GP64-LABEL: urem_i1: ; GP64: # %bb.0: # %entry ; GP64-NEXT: jr $ra ; GP64-NEXT: addiu $2, $zero, 0 ; ; GP64R6-LABEL: urem_i1: ; GP64R6: # %bb.0: # %entry ; GP64R6-NEXT: jr $ra ; GP64R6-NEXT: addiu $2, $zero, 0 ; ; MMR3-LABEL: urem_i1: ; MMR3: # %bb.0: # %entry ; MMR3-NEXT: li16 $2, 0 ; MMR3-NEXT: jrc $ra ; ; MMR6-LABEL: urem_i1: ; MMR6: # %bb.0: # %entry ; MMR6-NEXT: li16 $2, 0 ; MMR6-NEXT: jrc $ra entry: %r = urem i1 %a, %b ret i1 %r } define signext i8 @urem_i8(i8 signext %a, i8 signext %b) { ; GP32R0R2-LABEL: urem_i8: ; GP32R0R2: # %bb.0: # %entry ; GP32R0R2-NEXT: andi $1, $5, 255 ; GP32R0R2-NEXT: andi $2, $4, 255 ; GP32R0R2-NEXT: divu $zero, $2, $1 ; GP32R0R2-NEXT: teq $1, $zero, 7 ; GP32R0R2-NEXT: mfhi $1 ; GP32R0R2-NEXT: sll $1, $1, 24 ; GP32R0R2-NEXT: jr $ra ; GP32R0R2-NEXT: sra $2, $1, 24 ; ; GP32R2R5-LABEL: urem_i8: ; GP32R2R5: # %bb.0: # %entry ; GP32R2R5-NEXT: andi $1, $5, 255 ; GP32R2R5-NEXT: andi $2, $4, 255 ; GP32R2R5-NEXT: divu $zero, $2, $1 ; GP32R2R5-NEXT: teq $1, $zero, 7 ; GP32R2R5-NEXT: mfhi $1 ; GP32R2R5-NEXT: jr $ra ; GP32R2R5-NEXT: seb $2, $1 ; ; GP32R6-LABEL: urem_i8: ; GP32R6: # %bb.0: # %entry ; GP32R6-NEXT: andi $1, $5, 255 ; GP32R6-NEXT: andi $2, $4, 255 ; GP32R6-NEXT: modu $2, $2, $1 ; GP32R6-NEXT: teq $1, $zero, 7 ; GP32R6-NEXT: jr $ra ; GP32R6-NEXT: seb $2, $2 ; ; GP64R0R1-LABEL: urem_i8: ; GP64R0R1: # %bb.0: # %entry ; GP64R0R1-NEXT: andi $1, $5, 255 ; GP64R0R1-NEXT: andi $2, $4, 255 ; GP64R0R1-NEXT: divu $zero, $2, $1 ; GP64R0R1-NEXT: teq $1, $zero, 7 ; GP64R0R1-NEXT: mfhi $1 ; GP64R0R1-NEXT: sll $1, $1, 24 ; GP64R0R1-NEXT: jr $ra ; GP64R0R1-NEXT: sra $2, $1, 24 ; ; GP64R2R5-LABEL: urem_i8: ; GP64R2R5: # %bb.0: # %entry ; GP64R2R5-NEXT: andi $1, $5, 255 ; GP64R2R5-NEXT: andi $2, $4, 255 ; GP64R2R5-NEXT: divu $zero, $2, $1 ; GP64R2R5-NEXT: teq $1, $zero, 7 ; GP64R2R5-NEXT: mfhi $1 ; GP64R2R5-NEXT: jr $ra ; GP64R2R5-NEXT: seb $2, $1 ; ; GP64R6-LABEL: urem_i8: ; GP64R6: # %bb.0: # %entry ; GP64R6-NEXT: andi $1, $5, 255 ; GP64R6-NEXT: andi $2, $4, 255 ; GP64R6-NEXT: modu $2, $2, $1 ; GP64R6-NEXT: teq $1, $zero, 7 ; GP64R6-NEXT: jr $ra ; GP64R6-NEXT: seb $2, $2 ; ; MMR3-LABEL: urem_i8: ; MMR3: # %bb.0: # %entry ; MMR3-NEXT: andi16 $2, $5, 255 ; MMR3-NEXT: andi16 $3, $4, 255 ; MMR3-NEXT: divu $zero, $3, $2 ; MMR3-NEXT: teq $2, $zero, 7 ; MMR3-NEXT: mfhi16 $1 ; MMR3-NEXT: jr $ra ; MMR3-NEXT: seb $2, $1 ; ; MMR6-LABEL: urem_i8: ; MMR6: # %bb.0: # %entry ; MMR6-NEXT: andi16 $2, $5, 255 ; MMR6-NEXT: andi16 $3, $4, 255 ; MMR6-NEXT: modu $1, $3, $2 ; MMR6-NEXT: teq $2, $zero, 7 ; MMR6-NEXT: seb $2, $1 ; MMR6-NEXT: jrc $ra entry: %r = urem i8 %a, %b ret i8 %r } define signext i16 @urem_i16(i16 signext %a, i16 signext %b) { ; GP32R0R2-LABEL: urem_i16: ; GP32R0R2: # %bb.0: # %entry ; GP32R0R2-NEXT: andi $1, $5, 65535 ; GP32R0R2-NEXT: andi $2, $4, 65535 ; GP32R0R2-NEXT: divu $zero, $2, $1 ; GP32R0R2-NEXT: teq $1, $zero, 7 ; GP32R0R2-NEXT: mfhi $1 ; GP32R0R2-NEXT: sll $1, $1, 16 ; GP32R0R2-NEXT: jr $ra ; GP32R0R2-NEXT: sra $2, $1, 16 ; ; GP32R2R5-LABEL: urem_i16: ; GP32R2R5: # %bb.0: # %entry ; GP32R2R5-NEXT: andi $1, $5, 65535 ; GP32R2R5-NEXT: andi $2, $4, 65535 ; GP32R2R5-NEXT: divu $zero, $2, $1 ; GP32R2R5-NEXT: teq $1, $zero, 7 ; GP32R2R5-NEXT: mfhi $1 ; GP32R2R5-NEXT: jr $ra ; GP32R2R5-NEXT: seh $2, $1 ; ; GP32R6-LABEL: urem_i16: ; GP32R6: # %bb.0: # %entry ; GP32R6-NEXT: andi $1, $5, 65535 ; GP32R6-NEXT: andi $2, $4, 65535 ; GP32R6-NEXT: modu $2, $2, $1 ; GP32R6-NEXT: teq $1, $zero, 7 ; GP32R6-NEXT: jr $ra ; GP32R6-NEXT: seh $2, $2 ; ; GP64R0R1-LABEL: urem_i16: ; GP64R0R1: # %bb.0: # %entry ; GP64R0R1-NEXT: andi $1, $5, 65535 ; GP64R0R1-NEXT: andi $2, $4, 65535 ; GP64R0R1-NEXT: divu $zero, $2, $1 ; GP64R0R1-NEXT: teq $1, $zero, 7 ; GP64R0R1-NEXT: mfhi $1 ; GP64R0R1-NEXT: sll $1, $1, 16 ; GP64R0R1-NEXT: jr $ra ; GP64R0R1-NEXT: sra $2, $1, 16 ; ; GP64R2R5-LABEL: urem_i16: ; GP64R2R5: # %bb.0: # %entry ; GP64R2R5-NEXT: andi $1, $5, 65535 ; GP64R2R5-NEXT: andi $2, $4, 65535 ; GP64R2R5-NEXT: divu $zero, $2, $1 ; GP64R2R5-NEXT: teq $1, $zero, 7 ; GP64R2R5-NEXT: mfhi $1 ; GP64R2R5-NEXT: jr $ra ; GP64R2R5-NEXT: seh $2, $1 ; ; GP64R6-LABEL: urem_i16: ; GP64R6: # %bb.0: # %entry ; GP64R6-NEXT: andi $1, $5, 65535 ; GP64R6-NEXT: andi $2, $4, 65535 ; GP64R6-NEXT: modu $2, $2, $1 ; GP64R6-NEXT: teq $1, $zero, 7 ; GP64R6-NEXT: jr $ra ; GP64R6-NEXT: seh $2, $2 ; ; MMR3-LABEL: urem_i16: ; MMR3: # %bb.0: # %entry ; MMR3-NEXT: andi16 $2, $5, 65535 ; MMR3-NEXT: andi16 $3, $4, 65535 ; MMR3-NEXT: divu $zero, $3, $2 ; MMR3-NEXT: teq $2, $zero, 7 ; MMR3-NEXT: mfhi16 $1 ; MMR3-NEXT: jr $ra ; MMR3-NEXT: seh $2, $1 ; ; MMR6-LABEL: urem_i16: ; MMR6: # %bb.0: # %entry ; MMR6-NEXT: andi16 $2, $5, 65535 ; MMR6-NEXT: andi16 $3, $4, 65535 ; MMR6-NEXT: modu $1, $3, $2 ; MMR6-NEXT: teq $2, $zero, 7 ; MMR6-NEXT: seh $2, $1 ; MMR6-NEXT: jrc $ra entry: %r = urem i16 %a, %b ret i16 %r } define signext i32 @urem_i32(i32 signext %a, i32 signext %b) { ; GP32-LABEL: urem_i32: ; GP32: # %bb.0: # %entry ; GP32-NEXT: divu $zero, $4, $5 ; GP32-NEXT: teq $5, $zero, 7 ; GP32-NEXT: jr $ra ; GP32-NEXT: mfhi $2 ; ; GP32R6-LABEL: urem_i32: ; GP32R6: # %bb.0: # %entry ; GP32R6-NEXT: modu $2, $4, $5 ; GP32R6-NEXT: teq $5, $zero, 7 ; GP32R6-NEXT: jrc $ra ; ; GP64-LABEL: urem_i32: ; GP64: # %bb.0: # %entry ; GP64-NEXT: divu $zero, $4, $5 ; GP64-NEXT: teq $5, $zero, 7 ; GP64-NEXT: jr $ra ; GP64-NEXT: mfhi $2 ; ; GP64R6-LABEL: urem_i32: ; GP64R6: # %bb.0: # %entry ; GP64R6-NEXT: modu $2, $4, $5 ; GP64R6-NEXT: teq $5, $zero, 7 ; GP64R6-NEXT: jrc $ra ; ; MMR3-LABEL: urem_i32: ; MMR3: # %bb.0: # %entry ; MMR3-NEXT: divu $zero, $4, $5 ; MMR3-NEXT: teq $5, $zero, 7 ; MMR3-NEXT: mfhi16 $2 ; MMR3-NEXT: jrc $ra ; ; MMR6-LABEL: urem_i32: ; MMR6: # %bb.0: # %entry ; MMR6-NEXT: modu $2, $4, $5 ; MMR6-NEXT: teq $5, $zero, 7 ; MMR6-NEXT: jrc $ra entry: %r = urem i32 %a, %b ret i32 %r } define signext i64 @urem_i64(i64 signext %a, i64 signext %b) { ; GP32-LABEL: urem_i64: ; GP32: # %bb.0: # %entry ; GP32-NEXT: lui $2, %hi(_gp_disp) ; GP32-NEXT: addiu $2, $2, %lo(_gp_disp) ; GP32-NEXT: addiu $sp, $sp, -24 ; GP32-NEXT: .cfi_def_cfa_offset 24 ; GP32-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; GP32-NEXT: .cfi_offset 31, -4 ; GP32-NEXT: addu $gp, $2, $25 ; GP32-NEXT: lw $25, %call16(__umoddi3)($gp) ; GP32-NEXT: jalr $25 ; GP32-NEXT: nop ; GP32-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; GP32-NEXT: jr $ra ; GP32-NEXT: addiu $sp, $sp, 24 ; ; GP32R6-LABEL: urem_i64: ; GP32R6: # %bb.0: # %entry ; GP32R6-NEXT: lui $2, %hi(_gp_disp) ; GP32R6-NEXT: addiu $2, $2, %lo(_gp_disp) ; GP32R6-NEXT: addiu $sp, $sp, -24 ; GP32R6-NEXT: .cfi_def_cfa_offset 24 ; GP32R6-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; GP32R6-NEXT: .cfi_offset 31, -4 ; GP32R6-NEXT: addu $gp, $2, $25 ; GP32R6-NEXT: lw $25, %call16(__umoddi3)($gp) ; GP32R6-NEXT: jalrc $25 ; GP32R6-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; GP32R6-NEXT: jr $ra ; GP32R6-NEXT: addiu $sp, $sp, 24 ; ; GP64-LABEL: urem_i64: ; GP64: # %bb.0: # %entry ; GP64-NEXT: ddivu $zero, $4, $5 ; GP64-NEXT: teq $5, $zero, 7 ; GP64-NEXT: jr $ra ; GP64-NEXT: mfhi $2 ; ; GP64R6-LABEL: urem_i64: ; GP64R6: # %bb.0: # %entry ; GP64R6-NEXT: dmodu $2, $4, $5 ; GP64R6-NEXT: teq $5, $zero, 7 ; GP64R6-NEXT: jrc $ra ; ; MMR3-LABEL: urem_i64: ; MMR3: # %bb.0: # %entry ; MMR3-NEXT: lui $2, %hi(_gp_disp) ; MMR3-NEXT: addiu $2, $2, %lo(_gp_disp) ; MMR3-NEXT: addiusp -24 ; MMR3-NEXT: .cfi_def_cfa_offset 24 ; MMR3-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; MMR3-NEXT: .cfi_offset 31, -4 ; MMR3-NEXT: addu $2, $2, $25 ; MMR3-NEXT: lw $25, %call16(__umoddi3)($2) ; MMR3-NEXT: move $gp, $2 ; MMR3-NEXT: jalr $25 ; MMR3-NEXT: nop ; MMR3-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; MMR3-NEXT: addiusp 24 ; MMR3-NEXT: jrc $ra ; ; MMR6-LABEL: urem_i64: ; MMR6: # %bb.0: # %entry ; MMR6-NEXT: lui $2, %hi(_gp_disp) ; MMR6-NEXT: addiu $2, $2, %lo(_gp_disp) ; MMR6-NEXT: addiu $sp, $sp, -24 ; MMR6-NEXT: .cfi_def_cfa_offset 24 ; MMR6-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; MMR6-NEXT: .cfi_offset 31, -4 ; MMR6-NEXT: addu $2, $2, $25 ; MMR6-NEXT: lw $25, %call16(__umoddi3)($2) ; MMR6-NEXT: move $gp, $2 ; MMR6-NEXT: jalr $25 ; MMR6-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; MMR6-NEXT: addiu $sp, $sp, 24 ; MMR6-NEXT: jrc $ra entry: %r = urem i64 %a, %b ret i64 %r } define signext i128 @urem_i128(i128 signext %a, i128 signext %b) { ; GP32-LABEL: urem_i128: ; GP32: # %bb.0: # %entry ; GP32-NEXT: lui $2, %hi(_gp_disp) ; GP32-NEXT: addiu $2, $2, %lo(_gp_disp) ; GP32-NEXT: addiu $sp, $sp, -40 ; GP32-NEXT: .cfi_def_cfa_offset 40 ; GP32-NEXT: sw $ra, 36($sp) # 4-byte Folded Spill ; GP32-NEXT: .cfi_offset 31, -4 ; GP32-NEXT: addu $gp, $2, $25 ; GP32-NEXT: lw $1, 60($sp) ; GP32-NEXT: lw $2, 64($sp) ; GP32-NEXT: lw $3, 68($sp) ; GP32-NEXT: sw $3, 28($sp) ; GP32-NEXT: sw $2, 24($sp) ; GP32-NEXT: sw $1, 20($sp) ; GP32-NEXT: lw $1, 56($sp) ; GP32-NEXT: sw $1, 16($sp) ; GP32-NEXT: lw $25, %call16(__umodti3)($gp) ; GP32-NEXT: jalr $25 ; GP32-NEXT: nop ; GP32-NEXT: lw $ra, 36($sp) # 4-byte Folded Reload ; GP32-NEXT: jr $ra ; GP32-NEXT: addiu $sp, $sp, 40 ; ; GP32R6-LABEL: urem_i128: ; GP32R6: # %bb.0: # %entry ; GP32R6-NEXT: lui $2, %hi(_gp_disp) ; GP32R6-NEXT: addiu $2, $2, %lo(_gp_disp) ; GP32R6-NEXT: addiu $sp, $sp, -40 ; GP32R6-NEXT: .cfi_def_cfa_offset 40 ; GP32R6-NEXT: sw $ra, 36($sp) # 4-byte Folded Spill ; GP32R6-NEXT: .cfi_offset 31, -4 ; GP32R6-NEXT: addu $gp, $2, $25 ; GP32R6-NEXT: lw $1, 60($sp) ; GP32R6-NEXT: lw $2, 64($sp) ; GP32R6-NEXT: lw $3, 68($sp) ; GP32R6-NEXT: sw $3, 28($sp) ; GP32R6-NEXT: sw $2, 24($sp) ; GP32R6-NEXT: sw $1, 20($sp) ; GP32R6-NEXT: lw $1, 56($sp) ; GP32R6-NEXT: sw $1, 16($sp) ; GP32R6-NEXT: lw $25, %call16(__umodti3)($gp) ; GP32R6-NEXT: jalrc $25 ; GP32R6-NEXT: lw $ra, 36($sp) # 4-byte Folded Reload ; GP32R6-NEXT: jr $ra ; GP32R6-NEXT: addiu $sp, $sp, 40 ; ; GP64-LABEL: urem_i128: ; GP64: # %bb.0: # %entry ; GP64-NEXT: daddiu $sp, $sp, -16 ; GP64-NEXT: .cfi_def_cfa_offset 16 ; GP64-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; GP64-NEXT: sd $gp, 0($sp) # 8-byte Folded Spill ; GP64-NEXT: .cfi_offset 31, -8 ; GP64-NEXT: .cfi_offset 28, -16 ; GP64-NEXT: lui $1, %hi(%neg(%gp_rel(urem_i128))) ; GP64-NEXT: daddu $1, $1, $25 ; GP64-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(urem_i128))) ; GP64-NEXT: ld $25, %call16(__umodti3)($gp) ; GP64-NEXT: jalr $25 ; GP64-NEXT: nop ; GP64-NEXT: ld $gp, 0($sp) # 8-byte Folded Reload ; GP64-NEXT: ld $ra, 8($sp) # 8-byte Folded Reload ; GP64-NEXT: jr $ra ; GP64-NEXT: daddiu $sp, $sp, 16 ; ; GP64R6-LABEL: urem_i128: ; GP64R6: # %bb.0: # %entry ; GP64R6-NEXT: daddiu $sp, $sp, -16 ; GP64R6-NEXT: .cfi_def_cfa_offset 16 ; GP64R6-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; GP64R6-NEXT: sd $gp, 0($sp) # 8-byte Folded Spill ; GP64R6-NEXT: .cfi_offset 31, -8 ; GP64R6-NEXT: .cfi_offset 28, -16 ; GP64R6-NEXT: lui $1, %hi(%neg(%gp_rel(urem_i128))) ; GP64R6-NEXT: daddu $1, $1, $25 ; GP64R6-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(urem_i128))) ; GP64R6-NEXT: ld $25, %call16(__umodti3)($gp) ; GP64R6-NEXT: jalrc $25 ; GP64R6-NEXT: ld $gp, 0($sp) # 8-byte Folded Reload ; GP64R6-NEXT: ld $ra, 8($sp) # 8-byte Folded Reload ; GP64R6-NEXT: jr $ra ; GP64R6-NEXT: daddiu $sp, $sp, 16 ; ; MMR3-LABEL: urem_i128: ; MMR3: # %bb.0: # %entry ; MMR3-NEXT: lui $2, %hi(_gp_disp) ; MMR3-NEXT: addiu $2, $2, %lo(_gp_disp) ; MMR3-NEXT: addiusp -48 ; MMR3-NEXT: .cfi_def_cfa_offset 48 ; MMR3-NEXT: sw $ra, 44($sp) # 4-byte Folded Spill ; MMR3-NEXT: swp $16, 36($sp) ; MMR3-NEXT: .cfi_offset 31, -4 ; MMR3-NEXT: .cfi_offset 17, -8 ; MMR3-NEXT: .cfi_offset 16, -12 ; MMR3-NEXT: addu $16, $2, $25 ; MMR3-NEXT: move $1, $7 ; MMR3-NEXT: lw $7, 68($sp) ; MMR3-NEXT: lw $17, 72($sp) ; MMR3-NEXT: lw $3, 76($sp) ; MMR3-NEXT: move $2, $sp ; MMR3-NEXT: sw16 $3, 28($2) ; MMR3-NEXT: sw16 $17, 24($2) ; MMR3-NEXT: sw16 $7, 20($2) ; MMR3-NEXT: lw $3, 64($sp) ; MMR3-NEXT: sw16 $3, 16($2) ; MMR3-NEXT: lw $25, %call16(__umodti3)($16) ; MMR3-NEXT: move $7, $1 ; MMR3-NEXT: move $gp, $16 ; MMR3-NEXT: jalr $25 ; MMR3-NEXT: nop ; MMR3-NEXT: lwp $16, 36($sp) ; MMR3-NEXT: lw $ra, 44($sp) # 4-byte Folded Reload ; MMR3-NEXT: addiusp 48 ; MMR3-NEXT: jrc $ra ; ; MMR6-LABEL: urem_i128: ; MMR6: # %bb.0: # %entry ; MMR6-NEXT: lui $2, %hi(_gp_disp) ; MMR6-NEXT: addiu $2, $2, %lo(_gp_disp) ; MMR6-NEXT: addiu $sp, $sp, -48 ; MMR6-NEXT: .cfi_def_cfa_offset 48 ; MMR6-NEXT: sw $ra, 44($sp) # 4-byte Folded Spill ; MMR6-NEXT: sw $17, 40($sp) # 4-byte Folded Spill ; MMR6-NEXT: sw $16, 36($sp) # 4-byte Folded Spill ; MMR6-NEXT: .cfi_offset 31, -4 ; MMR6-NEXT: .cfi_offset 17, -8 ; MMR6-NEXT: .cfi_offset 16, -12 ; MMR6-NEXT: addu $16, $2, $25 ; MMR6-NEXT: move $1, $7 ; MMR6-NEXT: lw $7, 68($sp) ; MMR6-NEXT: lw $17, 72($sp) ; MMR6-NEXT: lw $3, 76($sp) ; MMR6-NEXT: move $2, $sp ; MMR6-NEXT: sw16 $3, 28($2) ; MMR6-NEXT: sw16 $17, 24($2) ; MMR6-NEXT: sw16 $7, 20($2) ; MMR6-NEXT: lw $3, 64($sp) ; MMR6-NEXT: sw16 $3, 16($2) ; MMR6-NEXT: lw $25, %call16(__umodti3)($16) ; MMR6-NEXT: move $7, $1 ; MMR6-NEXT: move $gp, $16 ; MMR6-NEXT: jalr $25 ; MMR6-NEXT: lw $16, 36($sp) # 4-byte Folded Reload ; MMR6-NEXT: lw $17, 40($sp) # 4-byte Folded Reload ; MMR6-NEXT: lw $ra, 44($sp) # 4-byte Folded Reload ; MMR6-NEXT: addiu $sp, $sp, 48 ; MMR6-NEXT: jrc $ra entry: %r = urem i128 %a, %b ret i128 %r } Index: vendor/llvm/dist-release_80/test/CodeGen/Mips/long-call-attr.ll =================================================================== --- vendor/llvm/dist-release_80/test/CodeGen/Mips/long-call-attr.ll (revision 343793) +++ vendor/llvm/dist-release_80/test/CodeGen/Mips/long-call-attr.ll (revision 343794) @@ -1,42 +1,42 @@ ; RUN: llc -march=mips -target-abi o32 --mattr=+long-calls,+noabicalls < %s \ -; RUN: | FileCheck -check-prefix=O32 %s +; RUN: -mips-jalr-reloc=false | FileCheck -check-prefix=O32 %s ; RUN: llc -march=mips -target-abi o32 --mattr=-long-calls,+noabicalls < %s \ -; RUN: | FileCheck -check-prefix=O32 %s +; RUN: -mips-jalr-reloc=false | FileCheck -check-prefix=O32 %s ; RUN: llc -march=mips64 -target-abi n64 --mattr=+long-calls,+noabicalls < %s \ -; RUN: | FileCheck -check-prefix=N64 %s +; RUN: -mips-jalr-reloc=false | FileCheck -check-prefix=N64 %s ; RUN: llc -march=mips64 -target-abi n64 --mattr=-long-calls,+noabicalls < %s \ -; RUN: | FileCheck -check-prefix=N64 %s +; RUN: -mips-jalr-reloc=false | FileCheck -check-prefix=N64 %s declare void @far() #0 define void @near() #1 { ret void } define void @foo() { call void @far() ; O32-LABEL: foo: ; O32: lui $1, %hi(far) ; O32-NEXT: addiu $25, $1, %lo(far) ; O32-NEXT: jalr $25 ; N64-LABEL: foo: ; N64: lui $1, %highest(far) ; N64-NEXT: daddiu $1, $1, %higher(far) ; N64-NEXT: dsll $1, $1, 16 ; N64-NEXT: daddiu $1, $1, %hi(far) ; N64-NEXT: dsll $1, $1, 16 ; N64-NEXT: daddiu $25, $1, %lo(far) ; N64-NEXT: jalr $25 call void @near() ; O32: jal near ; N64: jal near ret void } attributes #0 = { "long-call" } attributes #1 = { "short-call" } Index: vendor/llvm/dist-release_80/test/CodeGen/Mips/long-call-mcount.ll =================================================================== --- vendor/llvm/dist-release_80/test/CodeGen/Mips/long-call-mcount.ll (revision 343793) +++ vendor/llvm/dist-release_80/test/CodeGen/Mips/long-call-mcount.ll (revision 343794) @@ -1,19 +1,19 @@ ; Check call to mcount in case of long/short call options. ; RUN: llc -march=mips -target-abi o32 --mattr=+long-calls,+noabicalls < %s \ -; RUN: | FileCheck -check-prefixes=CHECK,LONG %s +; RUN: -mips-jalr-reloc=false | FileCheck -check-prefixes=CHECK,LONG %s ; RUN: llc -march=mips -target-abi o32 --mattr=-long-calls,+noabicalls < %s \ -; RUN: | FileCheck -check-prefixes=CHECK,SHORT %s +; RUN: -mips-jalr-reloc=false | FileCheck -check-prefixes=CHECK,SHORT %s ; Function Attrs: noinline nounwind optnone define void @foo() #0 { entry: ret void ; CHECK-LABEL: foo ; LONG: lui $1, %hi(_mcount) ; LONG-NEXT: addiu $25, $1, %lo(_mcount) ; LONG-NEXT: jalr $25 ; SHORT: jal _mcount } attributes #0 = { "instrument-function-entry-inlined"="_mcount" } Index: vendor/llvm/dist-release_80/test/CodeGen/Mips/msa/f16-llvm-ir.ll =================================================================== --- vendor/llvm/dist-release_80/test/CodeGen/Mips/msa/f16-llvm-ir.ll (revision 343793) +++ vendor/llvm/dist-release_80/test/CodeGen/Mips/msa/f16-llvm-ir.ll (revision 343794) @@ -1,3312 +1,3312 @@ ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py ; RUN: llc -relocation-model=pic -mtriple=mipsel-- -mcpu=mips32r5 \ -; RUN: -mattr=+fp64,+msa -verify-machineinstrs < %s | FileCheck %s \ +; RUN: -mattr=+fp64,+msa -verify-machineinstrs -mips-jalr-reloc=false < %s | FileCheck %s \ ; RUN: --check-prefixes=ALL,MIPS32,MIPSR5,MIPS32-O32,MIPS32R5-O32 ; RUN: llc -relocation-model=pic -mtriple=mips64el-- -mcpu=mips64r5 \ -; RUN: -mattr=+fp64,+msa -verify-machineinstrs -target-abi n32 < %s | FileCheck %s \ +; RUN: -mattr=+fp64,+msa -verify-machineinstrs -target-abi n32 -mips-jalr-reloc=false < %s | FileCheck %s \ ; RUN: --check-prefixes=ALL,MIPS64,MIPSR5,MIPS64-N32,MIPS64R5-N32 ; RUN: llc -relocation-model=pic -mtriple=mips64el-- -mcpu=mips64r5 \ -; RUN: -mattr=+fp64,+msa -verify-machineinstrs -target-abi n64 < %s | FileCheck %s \ +; RUN: -mattr=+fp64,+msa -verify-machineinstrs -target-abi n64 -mips-jalr-reloc=false < %s | FileCheck %s \ ; RUN: --check-prefixes=ALL,MIPS64,MIPSR5,MIPS64-N64,MIPS64R5-N64 ; RUN: llc -relocation-model=pic -mtriple=mipsel-- -mcpu=mips32r6 \ -; RUN: -mattr=+fp64,+msa -verify-machineinstrs < %s | FileCheck %s \ +; RUN: -mattr=+fp64,+msa -verify-machineinstrs -mips-jalr-reloc=false < %s | FileCheck %s \ ; RUN: --check-prefixes=ALL,MIPS32,MIPSR6,MIPSR6-O32 ; RUN: llc -relocation-model=pic -mtriple=mips64el-- -mcpu=mips64r6 \ -; RUN: -mattr=+fp64,+msa -verify-machineinstrs -target-abi n32 < %s | FileCheck %s \ +; RUN: -mattr=+fp64,+msa -verify-machineinstrs -target-abi n32 -mips-jalr-reloc=false < %s | FileCheck %s \ ; RUN: --check-prefixes=ALL,MIPS64,MIPSR6,MIPS64-N32,MIPSR6-N32 ; RUN: llc -relocation-model=pic -mtriple=mips64el-- -mcpu=mips64r6 \ -; RUN: -mattr=+fp64,+msa -verify-machineinstrs -target-abi n64 < %s | FileCheck %s \ +; RUN: -mattr=+fp64,+msa -verify-machineinstrs -target-abi n64 -mips-jalr-reloc=false < %s | FileCheck %s \ ; RUN: --check-prefixes=ALL,MIPS64,MIPSR6,MIPS64-N64,MIPSR6-N64 ; Check the use of frame indexes in the msa pseudo f16 instructions. @k = external global float declare float @k2(half *) define void @f3(i16 %b) { ; MIPS32-LABEL: f3: ; MIPS32: # %bb.0: # %entry ; MIPS32-NEXT: lui $2, %hi(_gp_disp) ; MIPS32-NEXT: addiu $2, $2, %lo(_gp_disp) ; MIPS32-NEXT: addiu $sp, $sp, -32 ; MIPS32-NEXT: .cfi_def_cfa_offset 32 ; MIPS32-NEXT: sw $ra, 28($sp) # 4-byte Folded Spill ; MIPS32-NEXT: sw $16, 24($sp) # 4-byte Folded Spill ; MIPS32-NEXT: .cfi_offset 31, -4 ; MIPS32-NEXT: .cfi_offset 16, -8 ; MIPS32-NEXT: addu $16, $2, $25 ; MIPS32-NEXT: sh $4, 22($sp) ; MIPS32-NEXT: addiu $4, $sp, 22 ; MIPS32-NEXT: lw $25, %call16(k2)($16) ; MIPS32-NEXT: jalr $25 ; MIPS32-NEXT: move $gp, $16 ; MIPS32-NEXT: lw $1, %got(k)($16) ; MIPS32-NEXT: swc1 $f0, 0($1) ; MIPS32-NEXT: lw $16, 24($sp) # 4-byte Folded Reload ; MIPS32-NEXT: lw $ra, 28($sp) # 4-byte Folded Reload ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: addiu $sp, $sp, 32 ; ; MIPS64-N32-LABEL: f3: ; MIPS64-N32: # %bb.0: # %entry ; MIPS64-N32-NEXT: addiu $sp, $sp, -32 ; MIPS64-N32-NEXT: .cfi_def_cfa_offset 32 ; MIPS64-N32-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: .cfi_offset 31, -8 ; MIPS64-N32-NEXT: .cfi_offset 28, -16 ; MIPS64-N32-NEXT: lui $1, %hi(%neg(%gp_rel(f3))) ; MIPS64-N32-NEXT: addu $1, $1, $25 ; MIPS64-N32-NEXT: addiu $gp, $1, %lo(%neg(%gp_rel(f3))) ; MIPS64-N32-NEXT: sh $4, 14($sp) ; MIPS64-N32-NEXT: lw $25, %call16(k2)($gp) ; MIPS64-N32-NEXT: jalr $25 ; MIPS64-N32-NEXT: addiu $4, $sp, 14 ; MIPS64-N32-NEXT: lw $1, %got_disp(k)($gp) ; MIPS64-N32-NEXT: swc1 $f0, 0($1) ; MIPS64-N32-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: jr $ra ; MIPS64-N32-NEXT: addiu $sp, $sp, 32 ; ; MIPS64-N64-LABEL: f3: ; MIPS64-N64: # %bb.0: # %entry ; MIPS64-N64-NEXT: daddiu $sp, $sp, -32 ; MIPS64-N64-NEXT: .cfi_def_cfa_offset 32 ; MIPS64-N64-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: .cfi_offset 31, -8 ; MIPS64-N64-NEXT: .cfi_offset 28, -16 ; MIPS64-N64-NEXT: lui $1, %hi(%neg(%gp_rel(f3))) ; MIPS64-N64-NEXT: daddu $1, $1, $25 ; MIPS64-N64-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(f3))) ; MIPS64-N64-NEXT: sh $4, 14($sp) ; MIPS64-N64-NEXT: ld $25, %call16(k2)($gp) ; MIPS64-N64-NEXT: jalr $25 ; MIPS64-N64-NEXT: daddiu $4, $sp, 14 ; MIPS64-N64-NEXT: ld $1, %got_disp(k)($gp) ; MIPS64-N64-NEXT: swc1 $f0, 0($1) ; MIPS64-N64-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: jr $ra ; MIPS64-N64-NEXT: daddiu $sp, $sp, 32 entry: %0 = alloca half %1 = bitcast i16 %b to half store half %1, half * %0 %2 = call float @k2(half * %0) store float %2, float * @k ret void } define void @f(i16 %b) { ; MIPS32-LABEL: f: ; MIPS32: # %bb.0: ; MIPS32-NEXT: lui $2, %hi(_gp_disp) ; MIPS32-NEXT: addiu $2, $2, %lo(_gp_disp) ; MIPS32-NEXT: addiu $sp, $sp, -8 ; MIPS32-NEXT: .cfi_def_cfa_offset 8 ; MIPS32-NEXT: addu $1, $2, $25 ; MIPS32-NEXT: sh $4, 4($sp) ; MIPS32-NEXT: lh $2, 4($sp) ; MIPS32-NEXT: fill.h $w0, $2 ; MIPS32-NEXT: fexupr.w $w0, $w0 ; MIPS32-NEXT: copy_s.w $2, $w0[0] ; MIPS32-NEXT: mtc1 $2, $f0 ; MIPS32-NEXT: lw $1, %got(k)($1) ; MIPS32-NEXT: swc1 $f0, 0($1) ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: addiu $sp, $sp, 8 ; ; MIPS64-N32-LABEL: f: ; MIPS64-N32: # %bb.0: ; MIPS64-N32-NEXT: addiu $sp, $sp, -16 ; MIPS64-N32-NEXT: .cfi_def_cfa_offset 16 ; MIPS64-N32-NEXT: lui $1, %hi(%neg(%gp_rel(f))) ; MIPS64-N32-NEXT: addu $1, $1, $25 ; MIPS64-N32-NEXT: addiu $1, $1, %lo(%neg(%gp_rel(f))) ; MIPS64-N32-NEXT: sh $4, 12($sp) ; MIPS64-N32-NEXT: lh $2, 12($sp) ; MIPS64-N32-NEXT: fill.h $w0, $2 ; MIPS64-N32-NEXT: fexupr.w $w0, $w0 ; MIPS64-N32-NEXT: copy_s.w $2, $w0[0] ; MIPS64-N32-NEXT: mtc1 $2, $f0 ; MIPS64-N32-NEXT: lw $1, %got_disp(k)($1) ; MIPS64-N32-NEXT: swc1 $f0, 0($1) ; MIPS64-N32-NEXT: jr $ra ; MIPS64-N32-NEXT: addiu $sp, $sp, 16 ; ; MIPS64-N64-LABEL: f: ; MIPS64-N64: # %bb.0: ; MIPS64-N64-NEXT: daddiu $sp, $sp, -16 ; MIPS64-N64-NEXT: .cfi_def_cfa_offset 16 ; MIPS64-N64-NEXT: lui $1, %hi(%neg(%gp_rel(f))) ; MIPS64-N64-NEXT: daddu $1, $1, $25 ; MIPS64-N64-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(f))) ; MIPS64-N64-NEXT: sh $4, 12($sp) ; MIPS64-N64-NEXT: lh $2, 12($sp) ; MIPS64-N64-NEXT: fill.h $w0, $2 ; MIPS64-N64-NEXT: fexupr.w $w0, $w0 ; MIPS64-N64-NEXT: copy_s.w $2, $w0[0] ; MIPS64-N64-NEXT: mtc1 $2, $f0 ; MIPS64-N64-NEXT: ld $1, %got_disp(k)($1) ; MIPS64-N64-NEXT: swc1 $f0, 0($1) ; MIPS64-N64-NEXT: jr $ra ; MIPS64-N64-NEXT: daddiu $sp, $sp, 16 %1 = bitcast i16 %b to half %2 = fpext half %1 to float store float %2, float * @k ret void } @g = external global i16, align 2 @h = external global half, align 2 ; Check that fext f16 to double has a fexupr.w, fexupr.d sequence. ; Check that ftrunc double to f16 has fexdo.w, fexdo.h sequence. ; Check that MIPS64R5+ uses 64-bit floating point <-> 64-bit GPR transfers. ; We don't need to check if pre-MIPSR5 expansions occur, the MSA ASE requires ; MIPSR5. Additionally, fp64 mode / FR=1 is required to use MSA. define void @fadd_f64() { ; MIPS32-LABEL: fadd_f64: ; MIPS32: # %bb.0: # %entry ; MIPS32-NEXT: lui $2, %hi(_gp_disp) ; MIPS32-NEXT: addiu $2, $2, %lo(_gp_disp) ; MIPS32-NEXT: addu $1, $2, $25 ; MIPS32-NEXT: lw $1, %got(h)($1) ; MIPS32-NEXT: lh $2, 0($1) ; MIPS32-NEXT: fill.h $w0, $2 ; MIPS32-NEXT: fexupr.w $w0, $w0 ; MIPS32-NEXT: fexupr.d $w0, $w0 ; MIPS32-NEXT: copy_s.w $2, $w0[0] ; MIPS32-NEXT: mtc1 $2, $f1 ; MIPS32-NEXT: copy_s.w $2, $w0[1] ; MIPS32-NEXT: mthc1 $2, $f1 ; MIPS32-NEXT: add.d $f0, $f1, $f1 ; MIPS32-NEXT: mfc1 $2, $f0 ; MIPS32-NEXT: fill.w $w1, $2 ; MIPS32-NEXT: mfhc1 $2, $f0 ; MIPS32-NEXT: insert.w $w1[1], $2 ; MIPS32-NEXT: insert.w $w1[3], $2 ; MIPS32-NEXT: fexdo.w $w0, $w1, $w1 ; MIPS32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS32-NEXT: copy_u.h $2, $w0[0] ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: sh $2, 0($1) ; ; MIPS64-N32-LABEL: fadd_f64: ; MIPS64-N32: # %bb.0: # %entry ; MIPS64-N32-NEXT: lui $1, %hi(%neg(%gp_rel(fadd_f64))) ; MIPS64-N32-NEXT: addu $1, $1, $25 ; MIPS64-N32-NEXT: addiu $1, $1, %lo(%neg(%gp_rel(fadd_f64))) ; MIPS64-N32-NEXT: lw $1, %got_disp(h)($1) ; MIPS64-N32-NEXT: lh $2, 0($1) ; MIPS64-N32-NEXT: fill.h $w0, $2 ; MIPS64-N32-NEXT: fexupr.w $w0, $w0 ; MIPS64-N32-NEXT: fexupr.d $w0, $w0 ; MIPS64-N32-NEXT: copy_s.d $2, $w0[0] ; MIPS64-N32-NEXT: dmtc1 $2, $f0 ; MIPS64-N32-NEXT: add.d $f0, $f0, $f0 ; MIPS64-N32-NEXT: dmfc1 $2, $f0 ; MIPS64-N32-NEXT: fill.d $w0, $2 ; MIPS64-N32-NEXT: fexdo.w $w0, $w0, $w0 ; MIPS64-N32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N32-NEXT: copy_u.h $2, $w0[0] ; MIPS64-N32-NEXT: jr $ra ; MIPS64-N32-NEXT: sh $2, 0($1) ; ; MIPS64-N64-LABEL: fadd_f64: ; MIPS64-N64: # %bb.0: # %entry ; MIPS64-N64-NEXT: lui $1, %hi(%neg(%gp_rel(fadd_f64))) ; MIPS64-N64-NEXT: daddu $1, $1, $25 ; MIPS64-N64-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(fadd_f64))) ; MIPS64-N64-NEXT: ld $1, %got_disp(h)($1) ; MIPS64-N64-NEXT: lh $2, 0($1) ; MIPS64-N64-NEXT: fill.h $w0, $2 ; MIPS64-N64-NEXT: fexupr.w $w0, $w0 ; MIPS64-N64-NEXT: fexupr.d $w0, $w0 ; MIPS64-N64-NEXT: copy_s.d $2, $w0[0] ; MIPS64-N64-NEXT: dmtc1 $2, $f0 ; MIPS64-N64-NEXT: add.d $f0, $f0, $f0 ; MIPS64-N64-NEXT: dmfc1 $2, $f0 ; MIPS64-N64-NEXT: fill.d $w0, $2 ; MIPS64-N64-NEXT: fexdo.w $w0, $w0, $w0 ; MIPS64-N64-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N64-NEXT: copy_u.h $2, $w0[0] ; MIPS64-N64-NEXT: jr $ra ; MIPS64-N64-NEXT: sh $2, 0($1) entry: %0 = load half, half * @h, align 2 %1 = fpext half %0 to double %2 = load half, half * @h, align 2 %3 = fpext half %2 to double %add = fadd double %1, %3 %4 = fptrunc double %add to half store half %4, half * @h, align 2 ret void } ; Entire fp16 (unsigned) range fits into (signed) i32. define i32 @ffptoui() { ; MIPS32-LABEL: ffptoui: ; MIPS32: # %bb.0: # %entry ; MIPS32-NEXT: lui $2, %hi(_gp_disp) ; MIPS32-NEXT: addiu $2, $2, %lo(_gp_disp) ; MIPS32-NEXT: addu $1, $2, $25 ; MIPS32-NEXT: lw $1, %got(h)($1) ; MIPS32-NEXT: lh $1, 0($1) ; MIPS32-NEXT: fill.h $w0, $1 ; MIPS32-NEXT: fexupr.w $w0, $w0 ; MIPS32-NEXT: fexupr.d $w0, $w0 ; MIPS32-NEXT: copy_s.w $1, $w0[0] ; MIPS32-NEXT: mtc1 $1, $f1 ; MIPS32-NEXT: copy_s.w $1, $w0[1] ; MIPS32-NEXT: mthc1 $1, $f1 ; MIPS32-NEXT: trunc.w.d $f0, $f1 ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: mfc1 $2, $f0 ; ; MIPS64-N32-LABEL: ffptoui: ; MIPS64-N32: # %bb.0: # %entry ; MIPS64-N32-NEXT: lui $1, %hi(%neg(%gp_rel(ffptoui))) ; MIPS64-N32-NEXT: addu $1, $1, $25 ; MIPS64-N32-NEXT: addiu $1, $1, %lo(%neg(%gp_rel(ffptoui))) ; MIPS64-N32-NEXT: lw $1, %got_disp(h)($1) ; MIPS64-N32-NEXT: lh $1, 0($1) ; MIPS64-N32-NEXT: fill.h $w0, $1 ; MIPS64-N32-NEXT: fexupr.w $w0, $w0 ; MIPS64-N32-NEXT: fexupr.d $w0, $w0 ; MIPS64-N32-NEXT: copy_s.d $1, $w0[0] ; MIPS64-N32-NEXT: dmtc1 $1, $f0 ; MIPS64-N32-NEXT: trunc.w.d $f0, $f0 ; MIPS64-N32-NEXT: jr $ra ; MIPS64-N32-NEXT: mfc1 $2, $f0 ; ; MIPS64-N64-LABEL: ffptoui: ; MIPS64-N64: # %bb.0: # %entry ; MIPS64-N64-NEXT: lui $1, %hi(%neg(%gp_rel(ffptoui))) ; MIPS64-N64-NEXT: daddu $1, $1, $25 ; MIPS64-N64-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(ffptoui))) ; MIPS64-N64-NEXT: ld $1, %got_disp(h)($1) ; MIPS64-N64-NEXT: lh $1, 0($1) ; MIPS64-N64-NEXT: fill.h $w0, $1 ; MIPS64-N64-NEXT: fexupr.w $w0, $w0 ; MIPS64-N64-NEXT: fexupr.d $w0, $w0 ; MIPS64-N64-NEXT: copy_s.d $1, $w0[0] ; MIPS64-N64-NEXT: dmtc1 $1, $f0 ; MIPS64-N64-NEXT: trunc.w.d $f0, $f0 ; MIPS64-N64-NEXT: jr $ra ; MIPS64-N64-NEXT: mfc1 $2, $f0 entry: %0 = load half, half * @h, align 2 %1 = fptoui half %0 to i32 ret i32 %1 } define i32 @ffptosi() { ; MIPS32-LABEL: ffptosi: ; MIPS32: # %bb.0: # %entry ; MIPS32-NEXT: lui $2, %hi(_gp_disp) ; MIPS32-NEXT: addiu $2, $2, %lo(_gp_disp) ; MIPS32-NEXT: addu $1, $2, $25 ; MIPS32-NEXT: lw $1, %got(h)($1) ; MIPS32-NEXT: lh $1, 0($1) ; MIPS32-NEXT: fill.h $w0, $1 ; MIPS32-NEXT: fexupr.w $w0, $w0 ; MIPS32-NEXT: fexupr.d $w0, $w0 ; MIPS32-NEXT: copy_s.w $1, $w0[0] ; MIPS32-NEXT: mtc1 $1, $f1 ; MIPS32-NEXT: copy_s.w $1, $w0[1] ; MIPS32-NEXT: mthc1 $1, $f1 ; MIPS32-NEXT: trunc.w.d $f0, $f1 ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: mfc1 $2, $f0 ; ; MIPS64-N32-LABEL: ffptosi: ; MIPS64-N32: # %bb.0: # %entry ; MIPS64-N32-NEXT: lui $1, %hi(%neg(%gp_rel(ffptosi))) ; MIPS64-N32-NEXT: addu $1, $1, $25 ; MIPS64-N32-NEXT: addiu $1, $1, %lo(%neg(%gp_rel(ffptosi))) ; MIPS64-N32-NEXT: lw $1, %got_disp(h)($1) ; MIPS64-N32-NEXT: lh $1, 0($1) ; MIPS64-N32-NEXT: fill.h $w0, $1 ; MIPS64-N32-NEXT: fexupr.w $w0, $w0 ; MIPS64-N32-NEXT: fexupr.d $w0, $w0 ; MIPS64-N32-NEXT: copy_s.d $1, $w0[0] ; MIPS64-N32-NEXT: dmtc1 $1, $f0 ; MIPS64-N32-NEXT: trunc.w.d $f0, $f0 ; MIPS64-N32-NEXT: jr $ra ; MIPS64-N32-NEXT: mfc1 $2, $f0 ; ; MIPS64-N64-LABEL: ffptosi: ; MIPS64-N64: # %bb.0: # %entry ; MIPS64-N64-NEXT: lui $1, %hi(%neg(%gp_rel(ffptosi))) ; MIPS64-N64-NEXT: daddu $1, $1, $25 ; MIPS64-N64-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(ffptosi))) ; MIPS64-N64-NEXT: ld $1, %got_disp(h)($1) ; MIPS64-N64-NEXT: lh $1, 0($1) ; MIPS64-N64-NEXT: fill.h $w0, $1 ; MIPS64-N64-NEXT: fexupr.w $w0, $w0 ; MIPS64-N64-NEXT: fexupr.d $w0, $w0 ; MIPS64-N64-NEXT: copy_s.d $1, $w0[0] ; MIPS64-N64-NEXT: dmtc1 $1, $f0 ; MIPS64-N64-NEXT: trunc.w.d $f0, $f0 ; MIPS64-N64-NEXT: jr $ra ; MIPS64-N64-NEXT: mfc1 $2, $f0 entry: %0 = load half, half * @h, align 2 %1 = fptosi half %0 to i32 ret i32 %1 } define void @uitofp(i32 %a) { ; MIPS32-LABEL: uitofp: ; MIPS32: # %bb.0: # %entry ; MIPS32-NEXT: lui $2, %hi(_gp_disp) ; MIPS32-NEXT: addiu $2, $2, %lo(_gp_disp) ; MIPS32-NEXT: addiu $sp, $sp, -8 ; MIPS32-NEXT: .cfi_def_cfa_offset 8 ; MIPS32-NEXT: addu $1, $2, $25 ; MIPS32-NEXT: lui $2, 17200 ; MIPS32-NEXT: sw $2, 4($sp) ; MIPS32-NEXT: sw $4, 0($sp) ; MIPS32-NEXT: lw $2, %got($CPI5_0)($1) ; MIPS32-NEXT: ldc1 $f0, %lo($CPI5_0)($2) ; MIPS32-NEXT: ldc1 $f1, 0($sp) ; MIPS32-NEXT: sub.d $f0, $f1, $f0 ; MIPS32-NEXT: mfc1 $2, $f0 ; MIPS32-NEXT: fill.w $w1, $2 ; MIPS32-NEXT: mfhc1 $2, $f0 ; MIPS32-NEXT: insert.w $w1[1], $2 ; MIPS32-NEXT: insert.w $w1[3], $2 ; MIPS32-NEXT: fexdo.w $w0, $w1, $w1 ; MIPS32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS32-NEXT: lw $1, %got(h)($1) ; MIPS32-NEXT: copy_u.h $2, $w0[0] ; MIPS32-NEXT: sh $2, 0($1) ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: addiu $sp, $sp, 8 ; ; MIPS64-N32-LABEL: uitofp: ; MIPS64-N32: # %bb.0: # %entry ; MIPS64-N32-NEXT: addiu $sp, $sp, -16 ; MIPS64-N32-NEXT: .cfi_def_cfa_offset 16 ; MIPS64-N32-NEXT: lui $1, %hi(%neg(%gp_rel(uitofp))) ; MIPS64-N32-NEXT: addu $1, $1, $25 ; MIPS64-N32-NEXT: addiu $1, $1, %lo(%neg(%gp_rel(uitofp))) ; MIPS64-N32-NEXT: lui $2, 17200 ; MIPS64-N32-NEXT: sw $2, 12($sp) ; MIPS64-N32-NEXT: sll $2, $4, 0 ; MIPS64-N32-NEXT: sw $2, 8($sp) ; MIPS64-N32-NEXT: lw $2, %got_page(.LCPI5_0)($1) ; MIPS64-N32-NEXT: ldc1 $f0, %got_ofst(.LCPI5_0)($2) ; MIPS64-N32-NEXT: ldc1 $f1, 8($sp) ; MIPS64-N32-NEXT: sub.d $f0, $f1, $f0 ; MIPS64-N32-NEXT: dmfc1 $2, $f0 ; MIPS64-N32-NEXT: fill.d $w0, $2 ; MIPS64-N32-NEXT: fexdo.w $w0, $w0, $w0 ; MIPS64-N32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N32-NEXT: lw $1, %got_disp(h)($1) ; MIPS64-N32-NEXT: copy_u.h $2, $w0[0] ; MIPS64-N32-NEXT: sh $2, 0($1) ; MIPS64-N32-NEXT: jr $ra ; MIPS64-N32-NEXT: addiu $sp, $sp, 16 ; ; MIPS64-N64-LABEL: uitofp: ; MIPS64-N64: # %bb.0: # %entry ; MIPS64-N64-NEXT: daddiu $sp, $sp, -16 ; MIPS64-N64-NEXT: .cfi_def_cfa_offset 16 ; MIPS64-N64-NEXT: lui $1, %hi(%neg(%gp_rel(uitofp))) ; MIPS64-N64-NEXT: daddu $1, $1, $25 ; MIPS64-N64-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(uitofp))) ; MIPS64-N64-NEXT: lui $2, 17200 ; MIPS64-N64-NEXT: sw $2, 12($sp) ; MIPS64-N64-NEXT: sll $2, $4, 0 ; MIPS64-N64-NEXT: sw $2, 8($sp) ; MIPS64-N64-NEXT: ld $2, %got_page(.LCPI5_0)($1) ; MIPS64-N64-NEXT: ldc1 $f0, %got_ofst(.LCPI5_0)($2) ; MIPS64-N64-NEXT: ldc1 $f1, 8($sp) ; MIPS64-N64-NEXT: sub.d $f0, $f1, $f0 ; MIPS64-N64-NEXT: dmfc1 $2, $f0 ; MIPS64-N64-NEXT: fill.d $w0, $2 ; MIPS64-N64-NEXT: fexdo.w $w0, $w0, $w0 ; MIPS64-N64-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N64-NEXT: ld $1, %got_disp(h)($1) ; MIPS64-N64-NEXT: copy_u.h $2, $w0[0] ; MIPS64-N64-NEXT: sh $2, 0($1) ; MIPS64-N64-NEXT: jr $ra ; MIPS64-N64-NEXT: daddiu $sp, $sp, 16 entry: %0 = uitofp i32 %a to half store half %0, half * @h, align 2 ret void } ; Check that f16 is expanded to f32 and relevant transfer ops occur. ; We don't check f16 -> f64 expansion occurs, as we expand f16 to f32. define void @fadd() { ; MIPS32-LABEL: fadd: ; MIPS32: # %bb.0: # %entry ; MIPS32-NEXT: lui $2, %hi(_gp_disp) ; MIPS32-NEXT: addiu $2, $2, %lo(_gp_disp) ; MIPS32-NEXT: addu $1, $2, $25 ; MIPS32-NEXT: lw $1, %got(g)($1) ; MIPS32-NEXT: lh $2, 0($1) ; MIPS32-NEXT: fill.h $w0, $2 ; MIPS32-NEXT: fexupr.w $w0, $w0 ; MIPS32-NEXT: copy_s.w $2, $w0[0] ; MIPS32-NEXT: mtc1 $2, $f0 ; MIPS32-NEXT: add.s $f0, $f0, $f0 ; MIPS32-NEXT: mfc1 $2, $f0 ; MIPS32-NEXT: fill.w $w0, $2 ; MIPS32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS32-NEXT: copy_u.h $2, $w0[0] ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: sh $2, 0($1) ; ; MIPS64-N32-LABEL: fadd: ; MIPS64-N32: # %bb.0: # %entry ; MIPS64-N32-NEXT: lui $1, %hi(%neg(%gp_rel(fadd))) ; MIPS64-N32-NEXT: addu $1, $1, $25 ; MIPS64-N32-NEXT: addiu $1, $1, %lo(%neg(%gp_rel(fadd))) ; MIPS64-N32-NEXT: lw $1, %got_disp(g)($1) ; MIPS64-N32-NEXT: lh $2, 0($1) ; MIPS64-N32-NEXT: fill.h $w0, $2 ; MIPS64-N32-NEXT: fexupr.w $w0, $w0 ; MIPS64-N32-NEXT: copy_s.w $2, $w0[0] ; MIPS64-N32-NEXT: mtc1 $2, $f0 ; MIPS64-N32-NEXT: add.s $f0, $f0, $f0 ; MIPS64-N32-NEXT: mfc1 $2, $f0 ; MIPS64-N32-NEXT: fill.w $w0, $2 ; MIPS64-N32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N32-NEXT: copy_u.h $2, $w0[0] ; MIPS64-N32-NEXT: jr $ra ; MIPS64-N32-NEXT: sh $2, 0($1) ; ; MIPS64-N64-LABEL: fadd: ; MIPS64-N64: # %bb.0: # %entry ; MIPS64-N64-NEXT: lui $1, %hi(%neg(%gp_rel(fadd))) ; MIPS64-N64-NEXT: daddu $1, $1, $25 ; MIPS64-N64-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(fadd))) ; MIPS64-N64-NEXT: ld $1, %got_disp(g)($1) ; MIPS64-N64-NEXT: lh $2, 0($1) ; MIPS64-N64-NEXT: fill.h $w0, $2 ; MIPS64-N64-NEXT: fexupr.w $w0, $w0 ; MIPS64-N64-NEXT: copy_s.w $2, $w0[0] ; MIPS64-N64-NEXT: mtc1 $2, $f0 ; MIPS64-N64-NEXT: add.s $f0, $f0, $f0 ; MIPS64-N64-NEXT: mfc1 $2, $f0 ; MIPS64-N64-NEXT: fill.w $w0, $2 ; MIPS64-N64-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N64-NEXT: copy_u.h $2, $w0[0] ; MIPS64-N64-NEXT: jr $ra ; MIPS64-N64-NEXT: sh $2, 0($1) entry: %0 = load i16, i16* @g, align 2 %1 = call float @llvm.convert.from.fp16.f32(i16 %0) %2 = load i16, i16* @g, align 2 %3 = call float @llvm.convert.from.fp16.f32(i16 %2) %add = fadd float %1, %3 %4 = call i16 @llvm.convert.to.fp16.f32(float %add) store i16 %4, i16* @g, align 2 ret void } ; Function Attrs: nounwind readnone declare float @llvm.convert.from.fp16.f32(i16) ; Function Attrs: nounwind readnone declare i16 @llvm.convert.to.fp16.f32(float) ; Function Attrs: nounwind define void @fsub() { ; MIPS32-LABEL: fsub: ; MIPS32: # %bb.0: # %entry ; MIPS32-NEXT: lui $2, %hi(_gp_disp) ; MIPS32-NEXT: addiu $2, $2, %lo(_gp_disp) ; MIPS32-NEXT: addu $1, $2, $25 ; MIPS32-NEXT: lw $1, %got(g)($1) ; MIPS32-NEXT: lh $2, 0($1) ; MIPS32-NEXT: fill.h $w0, $2 ; MIPS32-NEXT: fexupr.w $w0, $w0 ; MIPS32-NEXT: copy_s.w $2, $w0[0] ; MIPS32-NEXT: mtc1 $2, $f0 ; MIPS32-NEXT: sub.s $f0, $f0, $f0 ; MIPS32-NEXT: mfc1 $2, $f0 ; MIPS32-NEXT: fill.w $w0, $2 ; MIPS32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS32-NEXT: copy_u.h $2, $w0[0] ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: sh $2, 0($1) ; ; MIPS64-N32-LABEL: fsub: ; MIPS64-N32: # %bb.0: # %entry ; MIPS64-N32-NEXT: lui $1, %hi(%neg(%gp_rel(fsub))) ; MIPS64-N32-NEXT: addu $1, $1, $25 ; MIPS64-N32-NEXT: addiu $1, $1, %lo(%neg(%gp_rel(fsub))) ; MIPS64-N32-NEXT: lw $1, %got_disp(g)($1) ; MIPS64-N32-NEXT: lh $2, 0($1) ; MIPS64-N32-NEXT: fill.h $w0, $2 ; MIPS64-N32-NEXT: fexupr.w $w0, $w0 ; MIPS64-N32-NEXT: copy_s.w $2, $w0[0] ; MIPS64-N32-NEXT: mtc1 $2, $f0 ; MIPS64-N32-NEXT: sub.s $f0, $f0, $f0 ; MIPS64-N32-NEXT: mfc1 $2, $f0 ; MIPS64-N32-NEXT: fill.w $w0, $2 ; MIPS64-N32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N32-NEXT: copy_u.h $2, $w0[0] ; MIPS64-N32-NEXT: jr $ra ; MIPS64-N32-NEXT: sh $2, 0($1) ; ; MIPS64-N64-LABEL: fsub: ; MIPS64-N64: # %bb.0: # %entry ; MIPS64-N64-NEXT: lui $1, %hi(%neg(%gp_rel(fsub))) ; MIPS64-N64-NEXT: daddu $1, $1, $25 ; MIPS64-N64-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(fsub))) ; MIPS64-N64-NEXT: ld $1, %got_disp(g)($1) ; MIPS64-N64-NEXT: lh $2, 0($1) ; MIPS64-N64-NEXT: fill.h $w0, $2 ; MIPS64-N64-NEXT: fexupr.w $w0, $w0 ; MIPS64-N64-NEXT: copy_s.w $2, $w0[0] ; MIPS64-N64-NEXT: mtc1 $2, $f0 ; MIPS64-N64-NEXT: sub.s $f0, $f0, $f0 ; MIPS64-N64-NEXT: mfc1 $2, $f0 ; MIPS64-N64-NEXT: fill.w $w0, $2 ; MIPS64-N64-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N64-NEXT: copy_u.h $2, $w0[0] ; MIPS64-N64-NEXT: jr $ra ; MIPS64-N64-NEXT: sh $2, 0($1) entry: %0 = load i16, i16* @g, align 2 %1 = call float @llvm.convert.from.fp16.f32(i16 %0) %2 = load i16, i16* @g, align 2 %3 = call float @llvm.convert.from.fp16.f32(i16 %2) %sub = fsub float %1, %3 %4 = call i16 @llvm.convert.to.fp16.f32(float %sub) store i16 %4, i16* @g, align 2 ret void } define void @fmult() { ; MIPS32-LABEL: fmult: ; MIPS32: # %bb.0: # %entry ; MIPS32-NEXT: lui $2, %hi(_gp_disp) ; MIPS32-NEXT: addiu $2, $2, %lo(_gp_disp) ; MIPS32-NEXT: addu $1, $2, $25 ; MIPS32-NEXT: lw $1, %got(g)($1) ; MIPS32-NEXT: lh $2, 0($1) ; MIPS32-NEXT: fill.h $w0, $2 ; MIPS32-NEXT: fexupr.w $w0, $w0 ; MIPS32-NEXT: copy_s.w $2, $w0[0] ; MIPS32-NEXT: mtc1 $2, $f0 ; MIPS32-NEXT: mul.s $f0, $f0, $f0 ; MIPS32-NEXT: mfc1 $2, $f0 ; MIPS32-NEXT: fill.w $w0, $2 ; MIPS32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS32-NEXT: copy_u.h $2, $w0[0] ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: sh $2, 0($1) ; ; MIPS64-N32-LABEL: fmult: ; MIPS64-N32: # %bb.0: # %entry ; MIPS64-N32-NEXT: lui $1, %hi(%neg(%gp_rel(fmult))) ; MIPS64-N32-NEXT: addu $1, $1, $25 ; MIPS64-N32-NEXT: addiu $1, $1, %lo(%neg(%gp_rel(fmult))) ; MIPS64-N32-NEXT: lw $1, %got_disp(g)($1) ; MIPS64-N32-NEXT: lh $2, 0($1) ; MIPS64-N32-NEXT: fill.h $w0, $2 ; MIPS64-N32-NEXT: fexupr.w $w0, $w0 ; MIPS64-N32-NEXT: copy_s.w $2, $w0[0] ; MIPS64-N32-NEXT: mtc1 $2, $f0 ; MIPS64-N32-NEXT: mul.s $f0, $f0, $f0 ; MIPS64-N32-NEXT: mfc1 $2, $f0 ; MIPS64-N32-NEXT: fill.w $w0, $2 ; MIPS64-N32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N32-NEXT: copy_u.h $2, $w0[0] ; MIPS64-N32-NEXT: jr $ra ; MIPS64-N32-NEXT: sh $2, 0($1) ; ; MIPS64-N64-LABEL: fmult: ; MIPS64-N64: # %bb.0: # %entry ; MIPS64-N64-NEXT: lui $1, %hi(%neg(%gp_rel(fmult))) ; MIPS64-N64-NEXT: daddu $1, $1, $25 ; MIPS64-N64-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(fmult))) ; MIPS64-N64-NEXT: ld $1, %got_disp(g)($1) ; MIPS64-N64-NEXT: lh $2, 0($1) ; MIPS64-N64-NEXT: fill.h $w0, $2 ; MIPS64-N64-NEXT: fexupr.w $w0, $w0 ; MIPS64-N64-NEXT: copy_s.w $2, $w0[0] ; MIPS64-N64-NEXT: mtc1 $2, $f0 ; MIPS64-N64-NEXT: mul.s $f0, $f0, $f0 ; MIPS64-N64-NEXT: mfc1 $2, $f0 ; MIPS64-N64-NEXT: fill.w $w0, $2 ; MIPS64-N64-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N64-NEXT: copy_u.h $2, $w0[0] ; MIPS64-N64-NEXT: jr $ra ; MIPS64-N64-NEXT: sh $2, 0($1) entry: %0 = load i16, i16* @g, align 2 %1 = call float @llvm.convert.from.fp16.f32(i16 %0) %2 = load i16, i16* @g, align 2 %3 = call float @llvm.convert.from.fp16.f32(i16 %2) %mul = fmul float %1, %3 %4 = call i16 @llvm.convert.to.fp16.f32(float %mul) store i16 %4, i16* @g, align 2 ret void } define void @fdiv() { ; MIPS32-LABEL: fdiv: ; MIPS32: # %bb.0: # %entry ; MIPS32-NEXT: lui $2, %hi(_gp_disp) ; MIPS32-NEXT: addiu $2, $2, %lo(_gp_disp) ; MIPS32-NEXT: addu $1, $2, $25 ; MIPS32-NEXT: lw $1, %got(g)($1) ; MIPS32-NEXT: lh $2, 0($1) ; MIPS32-NEXT: fill.h $w0, $2 ; MIPS32-NEXT: fexupr.w $w0, $w0 ; MIPS32-NEXT: copy_s.w $2, $w0[0] ; MIPS32-NEXT: mtc1 $2, $f0 ; MIPS32-NEXT: div.s $f0, $f0, $f0 ; MIPS32-NEXT: mfc1 $2, $f0 ; MIPS32-NEXT: fill.w $w0, $2 ; MIPS32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS32-NEXT: copy_u.h $2, $w0[0] ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: sh $2, 0($1) ; ; MIPS64-N32-LABEL: fdiv: ; MIPS64-N32: # %bb.0: # %entry ; MIPS64-N32-NEXT: lui $1, %hi(%neg(%gp_rel(fdiv))) ; MIPS64-N32-NEXT: addu $1, $1, $25 ; MIPS64-N32-NEXT: addiu $1, $1, %lo(%neg(%gp_rel(fdiv))) ; MIPS64-N32-NEXT: lw $1, %got_disp(g)($1) ; MIPS64-N32-NEXT: lh $2, 0($1) ; MIPS64-N32-NEXT: fill.h $w0, $2 ; MIPS64-N32-NEXT: fexupr.w $w0, $w0 ; MIPS64-N32-NEXT: copy_s.w $2, $w0[0] ; MIPS64-N32-NEXT: mtc1 $2, $f0 ; MIPS64-N32-NEXT: div.s $f0, $f0, $f0 ; MIPS64-N32-NEXT: mfc1 $2, $f0 ; MIPS64-N32-NEXT: fill.w $w0, $2 ; MIPS64-N32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N32-NEXT: copy_u.h $2, $w0[0] ; MIPS64-N32-NEXT: jr $ra ; MIPS64-N32-NEXT: sh $2, 0($1) ; ; MIPS64-N64-LABEL: fdiv: ; MIPS64-N64: # %bb.0: # %entry ; MIPS64-N64-NEXT: lui $1, %hi(%neg(%gp_rel(fdiv))) ; MIPS64-N64-NEXT: daddu $1, $1, $25 ; MIPS64-N64-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(fdiv))) ; MIPS64-N64-NEXT: ld $1, %got_disp(g)($1) ; MIPS64-N64-NEXT: lh $2, 0($1) ; MIPS64-N64-NEXT: fill.h $w0, $2 ; MIPS64-N64-NEXT: fexupr.w $w0, $w0 ; MIPS64-N64-NEXT: copy_s.w $2, $w0[0] ; MIPS64-N64-NEXT: mtc1 $2, $f0 ; MIPS64-N64-NEXT: div.s $f0, $f0, $f0 ; MIPS64-N64-NEXT: mfc1 $2, $f0 ; MIPS64-N64-NEXT: fill.w $w0, $2 ; MIPS64-N64-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N64-NEXT: copy_u.h $2, $w0[0] ; MIPS64-N64-NEXT: jr $ra ; MIPS64-N64-NEXT: sh $2, 0($1) entry: %0 = load i16, i16* @g, align 2 %1 = call float @llvm.convert.from.fp16.f32(i16 %0) %2 = load i16, i16* @g, align 2 %3 = call float @llvm.convert.from.fp16.f32(i16 %2) %div = fdiv float %1, %3 %4 = call i16 @llvm.convert.to.fp16.f32(float %div) store i16 %4, i16* @g, align 2 ret void } define void @frem() { ; MIPS32-LABEL: frem: ; MIPS32: # %bb.0: # %entry ; MIPS32-NEXT: lui $2, %hi(_gp_disp) ; MIPS32-NEXT: addiu $2, $2, %lo(_gp_disp) ; MIPS32-NEXT: addiu $sp, $sp, -24 ; MIPS32-NEXT: .cfi_def_cfa_offset 24 ; MIPS32-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; MIPS32-NEXT: sw $16, 16($sp) # 4-byte Folded Spill ; MIPS32-NEXT: .cfi_offset 31, -4 ; MIPS32-NEXT: .cfi_offset 16, -8 ; MIPS32-NEXT: addu $gp, $2, $25 ; MIPS32-NEXT: lw $16, %got(g)($gp) ; MIPS32-NEXT: lh $1, 0($16) ; MIPS32-NEXT: fill.h $w0, $1 ; MIPS32-NEXT: fexupr.w $w0, $w0 ; MIPS32-NEXT: copy_s.w $1, $w0[0] ; MIPS32-NEXT: mtc1 $1, $f12 ; MIPS32-NEXT: lw $25, %call16(fmodf)($gp) ; MIPS32-NEXT: jalr $25 ; MIPS32-NEXT: mov.s $f14, $f12 ; MIPS32-NEXT: mfc1 $1, $f0 ; MIPS32-NEXT: fill.w $w0, $1 ; MIPS32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS32-NEXT: copy_u.h $1, $w0[0] ; MIPS32-NEXT: sh $1, 0($16) ; MIPS32-NEXT: lw $16, 16($sp) # 4-byte Folded Reload ; MIPS32-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: addiu $sp, $sp, 24 ; ; MIPS64-N32-LABEL: frem: ; MIPS64-N32: # %bb.0: # %entry ; MIPS64-N32-NEXT: addiu $sp, $sp, -32 ; MIPS64-N32-NEXT: .cfi_def_cfa_offset 32 ; MIPS64-N32-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: sd $16, 8($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: .cfi_offset 31, -8 ; MIPS64-N32-NEXT: .cfi_offset 28, -16 ; MIPS64-N32-NEXT: .cfi_offset 16, -24 ; MIPS64-N32-NEXT: lui $1, %hi(%neg(%gp_rel(frem))) ; MIPS64-N32-NEXT: addu $1, $1, $25 ; MIPS64-N32-NEXT: addiu $gp, $1, %lo(%neg(%gp_rel(frem))) ; MIPS64-N32-NEXT: lw $16, %got_disp(g)($gp) ; MIPS64-N32-NEXT: lh $1, 0($16) ; MIPS64-N32-NEXT: fill.h $w0, $1 ; MIPS64-N32-NEXT: fexupr.w $w0, $w0 ; MIPS64-N32-NEXT: copy_s.w $1, $w0[0] ; MIPS64-N32-NEXT: mtc1 $1, $f12 ; MIPS64-N32-NEXT: lw $25, %call16(fmodf)($gp) ; MIPS64-N32-NEXT: jalr $25 ; MIPS64-N32-NEXT: mov.s $f13, $f12 ; MIPS64-N32-NEXT: mfc1 $1, $f0 ; MIPS64-N32-NEXT: fill.w $w0, $1 ; MIPS64-N32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N32-NEXT: copy_u.h $1, $w0[0] ; MIPS64-N32-NEXT: sh $1, 0($16) ; MIPS64-N32-NEXT: ld $16, 8($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: jr $ra ; MIPS64-N32-NEXT: addiu $sp, $sp, 32 ; ; MIPS64-N64-LABEL: frem: ; MIPS64-N64: # %bb.0: # %entry ; MIPS64-N64-NEXT: daddiu $sp, $sp, -32 ; MIPS64-N64-NEXT: .cfi_def_cfa_offset 32 ; MIPS64-N64-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: sd $16, 8($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: .cfi_offset 31, -8 ; MIPS64-N64-NEXT: .cfi_offset 28, -16 ; MIPS64-N64-NEXT: .cfi_offset 16, -24 ; MIPS64-N64-NEXT: lui $1, %hi(%neg(%gp_rel(frem))) ; MIPS64-N64-NEXT: daddu $1, $1, $25 ; MIPS64-N64-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(frem))) ; MIPS64-N64-NEXT: ld $16, %got_disp(g)($gp) ; MIPS64-N64-NEXT: lh $1, 0($16) ; MIPS64-N64-NEXT: fill.h $w0, $1 ; MIPS64-N64-NEXT: fexupr.w $w0, $w0 ; MIPS64-N64-NEXT: copy_s.w $1, $w0[0] ; MIPS64-N64-NEXT: mtc1 $1, $f12 ; MIPS64-N64-NEXT: ld $25, %call16(fmodf)($gp) ; MIPS64-N64-NEXT: jalr $25 ; MIPS64-N64-NEXT: mov.s $f13, $f12 ; MIPS64-N64-NEXT: mfc1 $1, $f0 ; MIPS64-N64-NEXT: fill.w $w0, $1 ; MIPS64-N64-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N64-NEXT: copy_u.h $1, $w0[0] ; MIPS64-N64-NEXT: sh $1, 0($16) ; MIPS64-N64-NEXT: ld $16, 8($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: jr $ra ; MIPS64-N64-NEXT: daddiu $sp, $sp, 32 entry: %0 = load i16, i16* @g, align 2 %1 = call float @llvm.convert.from.fp16.f32(i16 %0) %2 = load i16, i16* @g, align 2 %3 = call float @llvm.convert.from.fp16.f32(i16 %2) %rem = frem float %1, %3 %4 = call i16 @llvm.convert.to.fp16.f32(float %rem) store i16 %4, i16* @g, align 2 ret void } @i1 = external global i16, align 1 define void @fcmp() { ; MIPS32-O32-LABEL: fcmp: ; MIPS32-O32: # %bb.0: # %entry ; MIPS32-O32-NEXT: lui $2, %hi(_gp_disp) ; MIPS32-O32-NEXT: addiu $2, $2, %lo(_gp_disp) ; MIPS32-O32-NEXT: addu $1, $2, $25 ; MIPS32-O32-NEXT: lw $2, %got(g)($1) ; MIPS32-O32-NEXT: lh $2, 0($2) ; MIPS32-O32-NEXT: fill.h $w0, $2 ; MIPS32-O32-NEXT: fexupr.w $w0, $w0 ; MIPS32-O32-NEXT: copy_s.w $2, $w0[0] ; MIPS32-O32-NEXT: mtc1 $2, $f0 ; MIPS32-O32-NEXT: addiu $2, $zero, 1 ; MIPS32-O32-NEXT: c.un.s $f0, $f0 ; MIPS32-O32-NEXT: movt $2, $zero, $fcc0 ; MIPS32-O32-NEXT: lw $1, %got(i1)($1) ; MIPS32-O32-NEXT: jr $ra ; MIPS32-O32-NEXT: sh $2, 0($1) ; ; MIPS64R5-N32-LABEL: fcmp: ; MIPS64R5-N32: # %bb.0: # %entry ; MIPS64R5-N32-NEXT: lui $1, %hi(%neg(%gp_rel(fcmp))) ; MIPS64R5-N32-NEXT: addu $1, $1, $25 ; MIPS64R5-N32-NEXT: addiu $1, $1, %lo(%neg(%gp_rel(fcmp))) ; MIPS64R5-N32-NEXT: lw $2, %got_disp(g)($1) ; MIPS64R5-N32-NEXT: lh $2, 0($2) ; MIPS64R5-N32-NEXT: fill.h $w0, $2 ; MIPS64R5-N32-NEXT: fexupr.w $w0, $w0 ; MIPS64R5-N32-NEXT: copy_s.w $2, $w0[0] ; MIPS64R5-N32-NEXT: mtc1 $2, $f0 ; MIPS64R5-N32-NEXT: addiu $2, $zero, 1 ; MIPS64R5-N32-NEXT: c.un.s $f0, $f0 ; MIPS64R5-N32-NEXT: movt $2, $zero, $fcc0 ; MIPS64R5-N32-NEXT: lw $1, %got_disp(i1)($1) ; MIPS64R5-N32-NEXT: jr $ra ; MIPS64R5-N32-NEXT: sh $2, 0($1) ; ; MIPS64R5-N64-LABEL: fcmp: ; MIPS64R5-N64: # %bb.0: # %entry ; MIPS64R5-N64-NEXT: lui $1, %hi(%neg(%gp_rel(fcmp))) ; MIPS64R5-N64-NEXT: daddu $1, $1, $25 ; MIPS64R5-N64-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(fcmp))) ; MIPS64R5-N64-NEXT: ld $2, %got_disp(g)($1) ; MIPS64R5-N64-NEXT: lh $2, 0($2) ; MIPS64R5-N64-NEXT: fill.h $w0, $2 ; MIPS64R5-N64-NEXT: fexupr.w $w0, $w0 ; MIPS64R5-N64-NEXT: copy_s.w $2, $w0[0] ; MIPS64R5-N64-NEXT: mtc1 $2, $f0 ; MIPS64R5-N64-NEXT: addiu $2, $zero, 1 ; MIPS64R5-N64-NEXT: c.un.s $f0, $f0 ; MIPS64R5-N64-NEXT: movt $2, $zero, $fcc0 ; MIPS64R5-N64-NEXT: ld $1, %got_disp(i1)($1) ; MIPS64R5-N64-NEXT: jr $ra ; MIPS64R5-N64-NEXT: sh $2, 0($1) ; ; MIPSR6-O32-LABEL: fcmp: ; MIPSR6-O32: # %bb.0: # %entry ; MIPSR6-O32-NEXT: lui $2, %hi(_gp_disp) ; MIPSR6-O32-NEXT: addiu $2, $2, %lo(_gp_disp) ; MIPSR6-O32-NEXT: addu $1, $2, $25 ; MIPSR6-O32-NEXT: lw $2, %got(g)($1) ; MIPSR6-O32-NEXT: lh $2, 0($2) ; MIPSR6-O32-NEXT: fill.h $w0, $2 ; MIPSR6-O32-NEXT: fexupr.w $w0, $w0 ; MIPSR6-O32-NEXT: copy_s.w $2, $w0[0] ; MIPSR6-O32-NEXT: mtc1 $2, $f0 ; MIPSR6-O32-NEXT: cmp.un.s $f0, $f0, $f0 ; MIPSR6-O32-NEXT: mfc1 $2, $f0 ; MIPSR6-O32-NEXT: not $2, $2 ; MIPSR6-O32-NEXT: andi $2, $2, 1 ; MIPSR6-O32-NEXT: lw $1, %got(i1)($1) ; MIPSR6-O32-NEXT: jr $ra ; MIPSR6-O32-NEXT: sh $2, 0($1) ; ; MIPSR6-N32-LABEL: fcmp: ; MIPSR6-N32: # %bb.0: # %entry ; MIPSR6-N32-NEXT: lui $1, %hi(%neg(%gp_rel(fcmp))) ; MIPSR6-N32-NEXT: addu $1, $1, $25 ; MIPSR6-N32-NEXT: addiu $1, $1, %lo(%neg(%gp_rel(fcmp))) ; MIPSR6-N32-NEXT: lw $2, %got_disp(g)($1) ; MIPSR6-N32-NEXT: lh $2, 0($2) ; MIPSR6-N32-NEXT: fill.h $w0, $2 ; MIPSR6-N32-NEXT: fexupr.w $w0, $w0 ; MIPSR6-N32-NEXT: copy_s.w $2, $w0[0] ; MIPSR6-N32-NEXT: mtc1 $2, $f0 ; MIPSR6-N32-NEXT: cmp.un.s $f0, $f0, $f0 ; MIPSR6-N32-NEXT: mfc1 $2, $f0 ; MIPSR6-N32-NEXT: not $2, $2 ; MIPSR6-N32-NEXT: andi $2, $2, 1 ; MIPSR6-N32-NEXT: lw $1, %got_disp(i1)($1) ; MIPSR6-N32-NEXT: jr $ra ; MIPSR6-N32-NEXT: sh $2, 0($1) ; ; MIPSR6-N64-LABEL: fcmp: ; MIPSR6-N64: # %bb.0: # %entry ; MIPSR6-N64-NEXT: lui $1, %hi(%neg(%gp_rel(fcmp))) ; MIPSR6-N64-NEXT: daddu $1, $1, $25 ; MIPSR6-N64-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(fcmp))) ; MIPSR6-N64-NEXT: ld $2, %got_disp(g)($1) ; MIPSR6-N64-NEXT: lh $2, 0($2) ; MIPSR6-N64-NEXT: fill.h $w0, $2 ; MIPSR6-N64-NEXT: fexupr.w $w0, $w0 ; MIPSR6-N64-NEXT: copy_s.w $2, $w0[0] ; MIPSR6-N64-NEXT: mtc1 $2, $f0 ; MIPSR6-N64-NEXT: cmp.un.s $f0, $f0, $f0 ; MIPSR6-N64-NEXT: mfc1 $2, $f0 ; MIPSR6-N64-NEXT: not $2, $2 ; MIPSR6-N64-NEXT: andi $2, $2, 1 ; MIPSR6-N64-NEXT: ld $1, %got_disp(i1)($1) ; MIPSR6-N64-NEXT: jr $ra ; MIPSR6-N64-NEXT: sh $2, 0($1) entry: %0 = load i16, i16* @g, align 2 %1 = call float @llvm.convert.from.fp16.f32(i16 %0) %2 = load i16, i16* @g, align 2 %3 = call float @llvm.convert.from.fp16.f32(i16 %2) %fcmp = fcmp oeq float %1, %3 %4 = zext i1 %fcmp to i16 store i16 %4, i16* @i1, align 2 ret void } declare float @llvm.powi.f32(float, i32) define void @fpowi() { ; MIPS32-LABEL: fpowi: ; MIPS32: # %bb.0: # %entry ; MIPS32-NEXT: lui $2, %hi(_gp_disp) ; MIPS32-NEXT: addiu $2, $2, %lo(_gp_disp) ; MIPS32-NEXT: addu $1, $2, $25 ; MIPS32-NEXT: lw $1, %got(g)($1) ; MIPS32-NEXT: lh $2, 0($1) ; MIPS32-NEXT: fill.h $w0, $2 ; MIPS32-NEXT: fexupr.w $w0, $w0 ; MIPS32-NEXT: copy_s.w $2, $w0[0] ; MIPS32-NEXT: mtc1 $2, $f0 ; MIPS32-NEXT: mul.s $f0, $f0, $f0 ; MIPS32-NEXT: mfc1 $2, $f0 ; MIPS32-NEXT: fill.w $w0, $2 ; MIPS32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS32-NEXT: copy_u.h $2, $w0[0] ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: sh $2, 0($1) ; ; MIPS64-N32-LABEL: fpowi: ; MIPS64-N32: # %bb.0: # %entry ; MIPS64-N32-NEXT: lui $1, %hi(%neg(%gp_rel(fpowi))) ; MIPS64-N32-NEXT: addu $1, $1, $25 ; MIPS64-N32-NEXT: addiu $1, $1, %lo(%neg(%gp_rel(fpowi))) ; MIPS64-N32-NEXT: lw $1, %got_disp(g)($1) ; MIPS64-N32-NEXT: lh $2, 0($1) ; MIPS64-N32-NEXT: fill.h $w0, $2 ; MIPS64-N32-NEXT: fexupr.w $w0, $w0 ; MIPS64-N32-NEXT: copy_s.w $2, $w0[0] ; MIPS64-N32-NEXT: mtc1 $2, $f0 ; MIPS64-N32-NEXT: mul.s $f0, $f0, $f0 ; MIPS64-N32-NEXT: mfc1 $2, $f0 ; MIPS64-N32-NEXT: fill.w $w0, $2 ; MIPS64-N32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N32-NEXT: copy_u.h $2, $w0[0] ; MIPS64-N32-NEXT: jr $ra ; MIPS64-N32-NEXT: sh $2, 0($1) ; ; MIPS64-N64-LABEL: fpowi: ; MIPS64-N64: # %bb.0: # %entry ; MIPS64-N64-NEXT: lui $1, %hi(%neg(%gp_rel(fpowi))) ; MIPS64-N64-NEXT: daddu $1, $1, $25 ; MIPS64-N64-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(fpowi))) ; MIPS64-N64-NEXT: ld $1, %got_disp(g)($1) ; MIPS64-N64-NEXT: lh $2, 0($1) ; MIPS64-N64-NEXT: fill.h $w0, $2 ; MIPS64-N64-NEXT: fexupr.w $w0, $w0 ; MIPS64-N64-NEXT: copy_s.w $2, $w0[0] ; MIPS64-N64-NEXT: mtc1 $2, $f0 ; MIPS64-N64-NEXT: mul.s $f0, $f0, $f0 ; MIPS64-N64-NEXT: mfc1 $2, $f0 ; MIPS64-N64-NEXT: fill.w $w0, $2 ; MIPS64-N64-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N64-NEXT: copy_u.h $2, $w0[0] ; MIPS64-N64-NEXT: jr $ra ; MIPS64-N64-NEXT: sh $2, 0($1) entry: %0 = load i16, i16* @g, align 2 %1 = call float @llvm.convert.from.fp16.f32(i16 %0) %powi = call float @llvm.powi.f32(float %1, i32 2) %2 = call i16 @llvm.convert.to.fp16.f32(float %powi) store i16 %2, i16* @g, align 2 ret void } define void @fpowi_var(i32 %var) { ; MIPS32-LABEL: fpowi_var: ; MIPS32: # %bb.0: # %entry ; MIPS32-NEXT: lui $2, %hi(_gp_disp) ; MIPS32-NEXT: addiu $2, $2, %lo(_gp_disp) ; MIPS32-NEXT: addiu $sp, $sp, -24 ; MIPS32-NEXT: .cfi_def_cfa_offset 24 ; MIPS32-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; MIPS32-NEXT: sw $16, 16($sp) # 4-byte Folded Spill ; MIPS32-NEXT: .cfi_offset 31, -4 ; MIPS32-NEXT: .cfi_offset 16, -8 ; MIPS32-NEXT: addu $gp, $2, $25 ; MIPS32-NEXT: lw $16, %got(g)($gp) ; MIPS32-NEXT: lh $1, 0($16) ; MIPS32-NEXT: fill.h $w0, $1 ; MIPS32-NEXT: fexupr.w $w0, $w0 ; MIPS32-NEXT: copy_s.w $1, $w0[0] ; MIPS32-NEXT: mtc1 $1, $f12 ; MIPS32-NEXT: lw $25, %call16(__powisf2)($gp) ; MIPS32-NEXT: jalr $25 ; MIPS32-NEXT: move $5, $4 ; MIPS32-NEXT: mfc1 $1, $f0 ; MIPS32-NEXT: fill.w $w0, $1 ; MIPS32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS32-NEXT: copy_u.h $1, $w0[0] ; MIPS32-NEXT: sh $1, 0($16) ; MIPS32-NEXT: lw $16, 16($sp) # 4-byte Folded Reload ; MIPS32-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: addiu $sp, $sp, 24 ; ; MIPS64-N32-LABEL: fpowi_var: ; MIPS64-N32: # %bb.0: # %entry ; MIPS64-N32-NEXT: addiu $sp, $sp, -32 ; MIPS64-N32-NEXT: .cfi_def_cfa_offset 32 ; MIPS64-N32-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: sd $16, 8($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: .cfi_offset 31, -8 ; MIPS64-N32-NEXT: .cfi_offset 28, -16 ; MIPS64-N32-NEXT: .cfi_offset 16, -24 ; MIPS64-N32-NEXT: lui $1, %hi(%neg(%gp_rel(fpowi_var))) ; MIPS64-N32-NEXT: addu $1, $1, $25 ; MIPS64-N32-NEXT: addiu $gp, $1, %lo(%neg(%gp_rel(fpowi_var))) ; MIPS64-N32-NEXT: sll $5, $4, 0 ; MIPS64-N32-NEXT: lw $16, %got_disp(g)($gp) ; MIPS64-N32-NEXT: lh $1, 0($16) ; MIPS64-N32-NEXT: fill.h $w0, $1 ; MIPS64-N32-NEXT: fexupr.w $w0, $w0 ; MIPS64-N32-NEXT: copy_s.w $1, $w0[0] ; MIPS64-N32-NEXT: lw $25, %call16(__powisf2)($gp) ; MIPS64-N32-NEXT: jalr $25 ; MIPS64-N32-NEXT: mtc1 $1, $f12 ; MIPS64-N32-NEXT: mfc1 $1, $f0 ; MIPS64-N32-NEXT: fill.w $w0, $1 ; MIPS64-N32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N32-NEXT: copy_u.h $1, $w0[0] ; MIPS64-N32-NEXT: sh $1, 0($16) ; MIPS64-N32-NEXT: ld $16, 8($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: jr $ra ; MIPS64-N32-NEXT: addiu $sp, $sp, 32 ; ; MIPS64-N64-LABEL: fpowi_var: ; MIPS64-N64: # %bb.0: # %entry ; MIPS64-N64-NEXT: daddiu $sp, $sp, -32 ; MIPS64-N64-NEXT: .cfi_def_cfa_offset 32 ; MIPS64-N64-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: sd $16, 8($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: .cfi_offset 31, -8 ; MIPS64-N64-NEXT: .cfi_offset 28, -16 ; MIPS64-N64-NEXT: .cfi_offset 16, -24 ; MIPS64-N64-NEXT: lui $1, %hi(%neg(%gp_rel(fpowi_var))) ; MIPS64-N64-NEXT: daddu $1, $1, $25 ; MIPS64-N64-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(fpowi_var))) ; MIPS64-N64-NEXT: sll $5, $4, 0 ; MIPS64-N64-NEXT: ld $16, %got_disp(g)($gp) ; MIPS64-N64-NEXT: lh $1, 0($16) ; MIPS64-N64-NEXT: fill.h $w0, $1 ; MIPS64-N64-NEXT: fexupr.w $w0, $w0 ; MIPS64-N64-NEXT: copy_s.w $1, $w0[0] ; MIPS64-N64-NEXT: ld $25, %call16(__powisf2)($gp) ; MIPS64-N64-NEXT: jalr $25 ; MIPS64-N64-NEXT: mtc1 $1, $f12 ; MIPS64-N64-NEXT: mfc1 $1, $f0 ; MIPS64-N64-NEXT: fill.w $w0, $1 ; MIPS64-N64-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N64-NEXT: copy_u.h $1, $w0[0] ; MIPS64-N64-NEXT: sh $1, 0($16) ; MIPS64-N64-NEXT: ld $16, 8($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: jr $ra ; MIPS64-N64-NEXT: daddiu $sp, $sp, 32 entry: %0 = load i16, i16* @g, align 2 %1 = call float @llvm.convert.from.fp16.f32(i16 %0) %powi = call float @llvm.powi.f32(float %1, i32 %var) %2 = call i16 @llvm.convert.to.fp16.f32(float %powi) store i16 %2, i16* @g, align 2 ret void } declare float @llvm.pow.f32(float %Val, float %power) define void @fpow(float %var) { ; MIPS32-LABEL: fpow: ; MIPS32: # %bb.0: # %entry ; MIPS32-NEXT: lui $2, %hi(_gp_disp) ; MIPS32-NEXT: addiu $2, $2, %lo(_gp_disp) ; MIPS32-NEXT: addiu $sp, $sp, -24 ; MIPS32-NEXT: .cfi_def_cfa_offset 24 ; MIPS32-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; MIPS32-NEXT: sw $16, 16($sp) # 4-byte Folded Spill ; MIPS32-NEXT: .cfi_offset 31, -4 ; MIPS32-NEXT: .cfi_offset 16, -8 ; MIPS32-NEXT: addu $gp, $2, $25 ; MIPS32-NEXT: mov.s $f14, $f12 ; MIPS32-NEXT: lw $16, %got(g)($gp) ; MIPS32-NEXT: lh $1, 0($16) ; MIPS32-NEXT: fill.h $w0, $1 ; MIPS32-NEXT: fexupr.w $w0, $w0 ; MIPS32-NEXT: copy_s.w $1, $w0[0] ; MIPS32-NEXT: lw $25, %call16(powf)($gp) ; MIPS32-NEXT: jalr $25 ; MIPS32-NEXT: mtc1 $1, $f12 ; MIPS32-NEXT: mfc1 $1, $f0 ; MIPS32-NEXT: fill.w $w0, $1 ; MIPS32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS32-NEXT: copy_u.h $1, $w0[0] ; MIPS32-NEXT: sh $1, 0($16) ; MIPS32-NEXT: lw $16, 16($sp) # 4-byte Folded Reload ; MIPS32-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: addiu $sp, $sp, 24 ; ; MIPS64-N32-LABEL: fpow: ; MIPS64-N32: # %bb.0: # %entry ; MIPS64-N32-NEXT: addiu $sp, $sp, -32 ; MIPS64-N32-NEXT: .cfi_def_cfa_offset 32 ; MIPS64-N32-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: sd $16, 8($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: .cfi_offset 31, -8 ; MIPS64-N32-NEXT: .cfi_offset 28, -16 ; MIPS64-N32-NEXT: .cfi_offset 16, -24 ; MIPS64-N32-NEXT: lui $1, %hi(%neg(%gp_rel(fpow))) ; MIPS64-N32-NEXT: addu $1, $1, $25 ; MIPS64-N32-NEXT: addiu $gp, $1, %lo(%neg(%gp_rel(fpow))) ; MIPS64-N32-NEXT: mov.s $f13, $f12 ; MIPS64-N32-NEXT: lw $16, %got_disp(g)($gp) ; MIPS64-N32-NEXT: lh $1, 0($16) ; MIPS64-N32-NEXT: fill.h $w0, $1 ; MIPS64-N32-NEXT: fexupr.w $w0, $w0 ; MIPS64-N32-NEXT: copy_s.w $1, $w0[0] ; MIPS64-N32-NEXT: lw $25, %call16(powf)($gp) ; MIPS64-N32-NEXT: jalr $25 ; MIPS64-N32-NEXT: mtc1 $1, $f12 ; MIPS64-N32-NEXT: mfc1 $1, $f0 ; MIPS64-N32-NEXT: fill.w $w0, $1 ; MIPS64-N32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N32-NEXT: copy_u.h $1, $w0[0] ; MIPS64-N32-NEXT: sh $1, 0($16) ; MIPS64-N32-NEXT: ld $16, 8($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: jr $ra ; MIPS64-N32-NEXT: addiu $sp, $sp, 32 ; ; MIPS64-N64-LABEL: fpow: ; MIPS64-N64: # %bb.0: # %entry ; MIPS64-N64-NEXT: daddiu $sp, $sp, -32 ; MIPS64-N64-NEXT: .cfi_def_cfa_offset 32 ; MIPS64-N64-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: sd $16, 8($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: .cfi_offset 31, -8 ; MIPS64-N64-NEXT: .cfi_offset 28, -16 ; MIPS64-N64-NEXT: .cfi_offset 16, -24 ; MIPS64-N64-NEXT: lui $1, %hi(%neg(%gp_rel(fpow))) ; MIPS64-N64-NEXT: daddu $1, $1, $25 ; MIPS64-N64-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(fpow))) ; MIPS64-N64-NEXT: mov.s $f13, $f12 ; MIPS64-N64-NEXT: ld $16, %got_disp(g)($gp) ; MIPS64-N64-NEXT: lh $1, 0($16) ; MIPS64-N64-NEXT: fill.h $w0, $1 ; MIPS64-N64-NEXT: fexupr.w $w0, $w0 ; MIPS64-N64-NEXT: copy_s.w $1, $w0[0] ; MIPS64-N64-NEXT: ld $25, %call16(powf)($gp) ; MIPS64-N64-NEXT: jalr $25 ; MIPS64-N64-NEXT: mtc1 $1, $f12 ; MIPS64-N64-NEXT: mfc1 $1, $f0 ; MIPS64-N64-NEXT: fill.w $w0, $1 ; MIPS64-N64-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N64-NEXT: copy_u.h $1, $w0[0] ; MIPS64-N64-NEXT: sh $1, 0($16) ; MIPS64-N64-NEXT: ld $16, 8($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: jr $ra ; MIPS64-N64-NEXT: daddiu $sp, $sp, 32 entry: %0 = load i16, i16* @g, align 2 %1 = call float @llvm.convert.from.fp16.f32(i16 %0) %powi = call float @llvm.pow.f32(float %1, float %var) %2 = call i16 @llvm.convert.to.fp16.f32(float %powi) store i16 %2, i16* @g, align 2 ret void } declare float @llvm.log2.f32(float %Val) define void @flog2() { ; MIPS32-LABEL: flog2: ; MIPS32: # %bb.0: # %entry ; MIPS32-NEXT: lui $2, %hi(_gp_disp) ; MIPS32-NEXT: addiu $2, $2, %lo(_gp_disp) ; MIPS32-NEXT: addiu $sp, $sp, -24 ; MIPS32-NEXT: .cfi_def_cfa_offset 24 ; MIPS32-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; MIPS32-NEXT: sw $16, 16($sp) # 4-byte Folded Spill ; MIPS32-NEXT: .cfi_offset 31, -4 ; MIPS32-NEXT: .cfi_offset 16, -8 ; MIPS32-NEXT: addu $gp, $2, $25 ; MIPS32-NEXT: lw $16, %got(g)($gp) ; MIPS32-NEXT: lh $1, 0($16) ; MIPS32-NEXT: fill.h $w0, $1 ; MIPS32-NEXT: fexupr.w $w0, $w0 ; MIPS32-NEXT: copy_s.w $1, $w0[0] ; MIPS32-NEXT: lw $25, %call16(log2f)($gp) ; MIPS32-NEXT: jalr $25 ; MIPS32-NEXT: mtc1 $1, $f12 ; MIPS32-NEXT: mfc1 $1, $f0 ; MIPS32-NEXT: fill.w $w0, $1 ; MIPS32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS32-NEXT: copy_u.h $1, $w0[0] ; MIPS32-NEXT: sh $1, 0($16) ; MIPS32-NEXT: lw $16, 16($sp) # 4-byte Folded Reload ; MIPS32-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: addiu $sp, $sp, 24 ; ; MIPS64-N32-LABEL: flog2: ; MIPS64-N32: # %bb.0: # %entry ; MIPS64-N32-NEXT: addiu $sp, $sp, -32 ; MIPS64-N32-NEXT: .cfi_def_cfa_offset 32 ; MIPS64-N32-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: sd $16, 8($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: .cfi_offset 31, -8 ; MIPS64-N32-NEXT: .cfi_offset 28, -16 ; MIPS64-N32-NEXT: .cfi_offset 16, -24 ; MIPS64-N32-NEXT: lui $1, %hi(%neg(%gp_rel(flog2))) ; MIPS64-N32-NEXT: addu $1, $1, $25 ; MIPS64-N32-NEXT: addiu $gp, $1, %lo(%neg(%gp_rel(flog2))) ; MIPS64-N32-NEXT: lw $16, %got_disp(g)($gp) ; MIPS64-N32-NEXT: lh $1, 0($16) ; MIPS64-N32-NEXT: fill.h $w0, $1 ; MIPS64-N32-NEXT: fexupr.w $w0, $w0 ; MIPS64-N32-NEXT: copy_s.w $1, $w0[0] ; MIPS64-N32-NEXT: lw $25, %call16(log2f)($gp) ; MIPS64-N32-NEXT: jalr $25 ; MIPS64-N32-NEXT: mtc1 $1, $f12 ; MIPS64-N32-NEXT: mfc1 $1, $f0 ; MIPS64-N32-NEXT: fill.w $w0, $1 ; MIPS64-N32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N32-NEXT: copy_u.h $1, $w0[0] ; MIPS64-N32-NEXT: sh $1, 0($16) ; MIPS64-N32-NEXT: ld $16, 8($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: jr $ra ; MIPS64-N32-NEXT: addiu $sp, $sp, 32 ; ; MIPS64-N64-LABEL: flog2: ; MIPS64-N64: # %bb.0: # %entry ; MIPS64-N64-NEXT: daddiu $sp, $sp, -32 ; MIPS64-N64-NEXT: .cfi_def_cfa_offset 32 ; MIPS64-N64-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: sd $16, 8($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: .cfi_offset 31, -8 ; MIPS64-N64-NEXT: .cfi_offset 28, -16 ; MIPS64-N64-NEXT: .cfi_offset 16, -24 ; MIPS64-N64-NEXT: lui $1, %hi(%neg(%gp_rel(flog2))) ; MIPS64-N64-NEXT: daddu $1, $1, $25 ; MIPS64-N64-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(flog2))) ; MIPS64-N64-NEXT: ld $16, %got_disp(g)($gp) ; MIPS64-N64-NEXT: lh $1, 0($16) ; MIPS64-N64-NEXT: fill.h $w0, $1 ; MIPS64-N64-NEXT: fexupr.w $w0, $w0 ; MIPS64-N64-NEXT: copy_s.w $1, $w0[0] ; MIPS64-N64-NEXT: ld $25, %call16(log2f)($gp) ; MIPS64-N64-NEXT: jalr $25 ; MIPS64-N64-NEXT: mtc1 $1, $f12 ; MIPS64-N64-NEXT: mfc1 $1, $f0 ; MIPS64-N64-NEXT: fill.w $w0, $1 ; MIPS64-N64-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N64-NEXT: copy_u.h $1, $w0[0] ; MIPS64-N64-NEXT: sh $1, 0($16) ; MIPS64-N64-NEXT: ld $16, 8($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: jr $ra ; MIPS64-N64-NEXT: daddiu $sp, $sp, 32 entry: %0 = load i16, i16* @g, align 2 %1 = call float @llvm.convert.from.fp16.f32(i16 %0) %log2 = call float @llvm.log2.f32(float %1) %2 = call i16 @llvm.convert.to.fp16.f32(float %log2) store i16 %2, i16* @g, align 2 ret void } declare float @llvm.log10.f32(float %Val) define void @flog10() { ; MIPS32-LABEL: flog10: ; MIPS32: # %bb.0: # %entry ; MIPS32-NEXT: lui $2, %hi(_gp_disp) ; MIPS32-NEXT: addiu $2, $2, %lo(_gp_disp) ; MIPS32-NEXT: addiu $sp, $sp, -24 ; MIPS32-NEXT: .cfi_def_cfa_offset 24 ; MIPS32-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; MIPS32-NEXT: sw $16, 16($sp) # 4-byte Folded Spill ; MIPS32-NEXT: .cfi_offset 31, -4 ; MIPS32-NEXT: .cfi_offset 16, -8 ; MIPS32-NEXT: addu $gp, $2, $25 ; MIPS32-NEXT: lw $16, %got(g)($gp) ; MIPS32-NEXT: lh $1, 0($16) ; MIPS32-NEXT: fill.h $w0, $1 ; MIPS32-NEXT: fexupr.w $w0, $w0 ; MIPS32-NEXT: copy_s.w $1, $w0[0] ; MIPS32-NEXT: lw $25, %call16(log10f)($gp) ; MIPS32-NEXT: jalr $25 ; MIPS32-NEXT: mtc1 $1, $f12 ; MIPS32-NEXT: mfc1 $1, $f0 ; MIPS32-NEXT: fill.w $w0, $1 ; MIPS32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS32-NEXT: copy_u.h $1, $w0[0] ; MIPS32-NEXT: sh $1, 0($16) ; MIPS32-NEXT: lw $16, 16($sp) # 4-byte Folded Reload ; MIPS32-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: addiu $sp, $sp, 24 ; ; MIPS64-N32-LABEL: flog10: ; MIPS64-N32: # %bb.0: # %entry ; MIPS64-N32-NEXT: addiu $sp, $sp, -32 ; MIPS64-N32-NEXT: .cfi_def_cfa_offset 32 ; MIPS64-N32-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: sd $16, 8($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: .cfi_offset 31, -8 ; MIPS64-N32-NEXT: .cfi_offset 28, -16 ; MIPS64-N32-NEXT: .cfi_offset 16, -24 ; MIPS64-N32-NEXT: lui $1, %hi(%neg(%gp_rel(flog10))) ; MIPS64-N32-NEXT: addu $1, $1, $25 ; MIPS64-N32-NEXT: addiu $gp, $1, %lo(%neg(%gp_rel(flog10))) ; MIPS64-N32-NEXT: lw $16, %got_disp(g)($gp) ; MIPS64-N32-NEXT: lh $1, 0($16) ; MIPS64-N32-NEXT: fill.h $w0, $1 ; MIPS64-N32-NEXT: fexupr.w $w0, $w0 ; MIPS64-N32-NEXT: copy_s.w $1, $w0[0] ; MIPS64-N32-NEXT: lw $25, %call16(log10f)($gp) ; MIPS64-N32-NEXT: jalr $25 ; MIPS64-N32-NEXT: mtc1 $1, $f12 ; MIPS64-N32-NEXT: mfc1 $1, $f0 ; MIPS64-N32-NEXT: fill.w $w0, $1 ; MIPS64-N32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N32-NEXT: copy_u.h $1, $w0[0] ; MIPS64-N32-NEXT: sh $1, 0($16) ; MIPS64-N32-NEXT: ld $16, 8($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: jr $ra ; MIPS64-N32-NEXT: addiu $sp, $sp, 32 ; ; MIPS64-N64-LABEL: flog10: ; MIPS64-N64: # %bb.0: # %entry ; MIPS64-N64-NEXT: daddiu $sp, $sp, -32 ; MIPS64-N64-NEXT: .cfi_def_cfa_offset 32 ; MIPS64-N64-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: sd $16, 8($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: .cfi_offset 31, -8 ; MIPS64-N64-NEXT: .cfi_offset 28, -16 ; MIPS64-N64-NEXT: .cfi_offset 16, -24 ; MIPS64-N64-NEXT: lui $1, %hi(%neg(%gp_rel(flog10))) ; MIPS64-N64-NEXT: daddu $1, $1, $25 ; MIPS64-N64-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(flog10))) ; MIPS64-N64-NEXT: ld $16, %got_disp(g)($gp) ; MIPS64-N64-NEXT: lh $1, 0($16) ; MIPS64-N64-NEXT: fill.h $w0, $1 ; MIPS64-N64-NEXT: fexupr.w $w0, $w0 ; MIPS64-N64-NEXT: copy_s.w $1, $w0[0] ; MIPS64-N64-NEXT: ld $25, %call16(log10f)($gp) ; MIPS64-N64-NEXT: jalr $25 ; MIPS64-N64-NEXT: mtc1 $1, $f12 ; MIPS64-N64-NEXT: mfc1 $1, $f0 ; MIPS64-N64-NEXT: fill.w $w0, $1 ; MIPS64-N64-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N64-NEXT: copy_u.h $1, $w0[0] ; MIPS64-N64-NEXT: sh $1, 0($16) ; MIPS64-N64-NEXT: ld $16, 8($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: jr $ra ; MIPS64-N64-NEXT: daddiu $sp, $sp, 32 entry: %0 = load i16, i16* @g, align 2 %1 = call float @llvm.convert.from.fp16.f32(i16 %0) %log10 = call float @llvm.log10.f32(float %1) %2 = call i16 @llvm.convert.to.fp16.f32(float %log10) store i16 %2, i16* @g, align 2 ret void } declare float @llvm.sqrt.f32(float %Val) define void @fsqrt() { ; MIPS32-LABEL: fsqrt: ; MIPS32: # %bb.0: # %entry ; MIPS32-NEXT: lui $2, %hi(_gp_disp) ; MIPS32-NEXT: addiu $2, $2, %lo(_gp_disp) ; MIPS32-NEXT: addu $1, $2, $25 ; MIPS32-NEXT: lw $1, %got(g)($1) ; MIPS32-NEXT: lh $2, 0($1) ; MIPS32-NEXT: fill.h $w0, $2 ; MIPS32-NEXT: fexupr.w $w0, $w0 ; MIPS32-NEXT: copy_s.w $2, $w0[0] ; MIPS32-NEXT: mtc1 $2, $f0 ; MIPS32-NEXT: sqrt.s $f0, $f0 ; MIPS32-NEXT: mfc1 $2, $f0 ; MIPS32-NEXT: fill.w $w0, $2 ; MIPS32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS32-NEXT: copy_u.h $2, $w0[0] ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: sh $2, 0($1) ; ; MIPS64-N32-LABEL: fsqrt: ; MIPS64-N32: # %bb.0: # %entry ; MIPS64-N32-NEXT: lui $1, %hi(%neg(%gp_rel(fsqrt))) ; MIPS64-N32-NEXT: addu $1, $1, $25 ; MIPS64-N32-NEXT: addiu $1, $1, %lo(%neg(%gp_rel(fsqrt))) ; MIPS64-N32-NEXT: lw $1, %got_disp(g)($1) ; MIPS64-N32-NEXT: lh $2, 0($1) ; MIPS64-N32-NEXT: fill.h $w0, $2 ; MIPS64-N32-NEXT: fexupr.w $w0, $w0 ; MIPS64-N32-NEXT: copy_s.w $2, $w0[0] ; MIPS64-N32-NEXT: mtc1 $2, $f0 ; MIPS64-N32-NEXT: sqrt.s $f0, $f0 ; MIPS64-N32-NEXT: mfc1 $2, $f0 ; MIPS64-N32-NEXT: fill.w $w0, $2 ; MIPS64-N32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N32-NEXT: copy_u.h $2, $w0[0] ; MIPS64-N32-NEXT: jr $ra ; MIPS64-N32-NEXT: sh $2, 0($1) ; ; MIPS64-N64-LABEL: fsqrt: ; MIPS64-N64: # %bb.0: # %entry ; MIPS64-N64-NEXT: lui $1, %hi(%neg(%gp_rel(fsqrt))) ; MIPS64-N64-NEXT: daddu $1, $1, $25 ; MIPS64-N64-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(fsqrt))) ; MIPS64-N64-NEXT: ld $1, %got_disp(g)($1) ; MIPS64-N64-NEXT: lh $2, 0($1) ; MIPS64-N64-NEXT: fill.h $w0, $2 ; MIPS64-N64-NEXT: fexupr.w $w0, $w0 ; MIPS64-N64-NEXT: copy_s.w $2, $w0[0] ; MIPS64-N64-NEXT: mtc1 $2, $f0 ; MIPS64-N64-NEXT: sqrt.s $f0, $f0 ; MIPS64-N64-NEXT: mfc1 $2, $f0 ; MIPS64-N64-NEXT: fill.w $w0, $2 ; MIPS64-N64-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N64-NEXT: copy_u.h $2, $w0[0] ; MIPS64-N64-NEXT: jr $ra ; MIPS64-N64-NEXT: sh $2, 0($1) entry: %0 = load i16, i16* @g, align 2 %1 = call float @llvm.convert.from.fp16.f32(i16 %0) %sqrt = call float @llvm.sqrt.f32(float %1) %2 = call i16 @llvm.convert.to.fp16.f32(float %sqrt) store i16 %2, i16* @g, align 2 ret void } declare float @llvm.sin.f32(float %Val) define void @fsin() { ; MIPS32-LABEL: fsin: ; MIPS32: # %bb.0: # %entry ; MIPS32-NEXT: lui $2, %hi(_gp_disp) ; MIPS32-NEXT: addiu $2, $2, %lo(_gp_disp) ; MIPS32-NEXT: addiu $sp, $sp, -24 ; MIPS32-NEXT: .cfi_def_cfa_offset 24 ; MIPS32-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; MIPS32-NEXT: sw $16, 16($sp) # 4-byte Folded Spill ; MIPS32-NEXT: .cfi_offset 31, -4 ; MIPS32-NEXT: .cfi_offset 16, -8 ; MIPS32-NEXT: addu $gp, $2, $25 ; MIPS32-NEXT: lw $16, %got(g)($gp) ; MIPS32-NEXT: lh $1, 0($16) ; MIPS32-NEXT: fill.h $w0, $1 ; MIPS32-NEXT: fexupr.w $w0, $w0 ; MIPS32-NEXT: copy_s.w $1, $w0[0] ; MIPS32-NEXT: lw $25, %call16(sinf)($gp) ; MIPS32-NEXT: jalr $25 ; MIPS32-NEXT: mtc1 $1, $f12 ; MIPS32-NEXT: mfc1 $1, $f0 ; MIPS32-NEXT: fill.w $w0, $1 ; MIPS32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS32-NEXT: copy_u.h $1, $w0[0] ; MIPS32-NEXT: sh $1, 0($16) ; MIPS32-NEXT: lw $16, 16($sp) # 4-byte Folded Reload ; MIPS32-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: addiu $sp, $sp, 24 ; ; MIPS64-N32-LABEL: fsin: ; MIPS64-N32: # %bb.0: # %entry ; MIPS64-N32-NEXT: addiu $sp, $sp, -32 ; MIPS64-N32-NEXT: .cfi_def_cfa_offset 32 ; MIPS64-N32-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: sd $16, 8($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: .cfi_offset 31, -8 ; MIPS64-N32-NEXT: .cfi_offset 28, -16 ; MIPS64-N32-NEXT: .cfi_offset 16, -24 ; MIPS64-N32-NEXT: lui $1, %hi(%neg(%gp_rel(fsin))) ; MIPS64-N32-NEXT: addu $1, $1, $25 ; MIPS64-N32-NEXT: addiu $gp, $1, %lo(%neg(%gp_rel(fsin))) ; MIPS64-N32-NEXT: lw $16, %got_disp(g)($gp) ; MIPS64-N32-NEXT: lh $1, 0($16) ; MIPS64-N32-NEXT: fill.h $w0, $1 ; MIPS64-N32-NEXT: fexupr.w $w0, $w0 ; MIPS64-N32-NEXT: copy_s.w $1, $w0[0] ; MIPS64-N32-NEXT: lw $25, %call16(sinf)($gp) ; MIPS64-N32-NEXT: jalr $25 ; MIPS64-N32-NEXT: mtc1 $1, $f12 ; MIPS64-N32-NEXT: mfc1 $1, $f0 ; MIPS64-N32-NEXT: fill.w $w0, $1 ; MIPS64-N32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N32-NEXT: copy_u.h $1, $w0[0] ; MIPS64-N32-NEXT: sh $1, 0($16) ; MIPS64-N32-NEXT: ld $16, 8($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: jr $ra ; MIPS64-N32-NEXT: addiu $sp, $sp, 32 ; ; MIPS64-N64-LABEL: fsin: ; MIPS64-N64: # %bb.0: # %entry ; MIPS64-N64-NEXT: daddiu $sp, $sp, -32 ; MIPS64-N64-NEXT: .cfi_def_cfa_offset 32 ; MIPS64-N64-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: sd $16, 8($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: .cfi_offset 31, -8 ; MIPS64-N64-NEXT: .cfi_offset 28, -16 ; MIPS64-N64-NEXT: .cfi_offset 16, -24 ; MIPS64-N64-NEXT: lui $1, %hi(%neg(%gp_rel(fsin))) ; MIPS64-N64-NEXT: daddu $1, $1, $25 ; MIPS64-N64-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(fsin))) ; MIPS64-N64-NEXT: ld $16, %got_disp(g)($gp) ; MIPS64-N64-NEXT: lh $1, 0($16) ; MIPS64-N64-NEXT: fill.h $w0, $1 ; MIPS64-N64-NEXT: fexupr.w $w0, $w0 ; MIPS64-N64-NEXT: copy_s.w $1, $w0[0] ; MIPS64-N64-NEXT: ld $25, %call16(sinf)($gp) ; MIPS64-N64-NEXT: jalr $25 ; MIPS64-N64-NEXT: mtc1 $1, $f12 ; MIPS64-N64-NEXT: mfc1 $1, $f0 ; MIPS64-N64-NEXT: fill.w $w0, $1 ; MIPS64-N64-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N64-NEXT: copy_u.h $1, $w0[0] ; MIPS64-N64-NEXT: sh $1, 0($16) ; MIPS64-N64-NEXT: ld $16, 8($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: jr $ra ; MIPS64-N64-NEXT: daddiu $sp, $sp, 32 entry: %0 = load i16, i16* @g, align 2 %1 = call float @llvm.convert.from.fp16.f32(i16 %0) %sin = call float @llvm.sin.f32(float %1) %2 = call i16 @llvm.convert.to.fp16.f32(float %sin) store i16 %2, i16* @g, align 2 ret void } declare float @llvm.cos.f32(float %Val) define void @fcos() { ; MIPS32-LABEL: fcos: ; MIPS32: # %bb.0: # %entry ; MIPS32-NEXT: lui $2, %hi(_gp_disp) ; MIPS32-NEXT: addiu $2, $2, %lo(_gp_disp) ; MIPS32-NEXT: addiu $sp, $sp, -24 ; MIPS32-NEXT: .cfi_def_cfa_offset 24 ; MIPS32-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; MIPS32-NEXT: sw $16, 16($sp) # 4-byte Folded Spill ; MIPS32-NEXT: .cfi_offset 31, -4 ; MIPS32-NEXT: .cfi_offset 16, -8 ; MIPS32-NEXT: addu $gp, $2, $25 ; MIPS32-NEXT: lw $16, %got(g)($gp) ; MIPS32-NEXT: lh $1, 0($16) ; MIPS32-NEXT: fill.h $w0, $1 ; MIPS32-NEXT: fexupr.w $w0, $w0 ; MIPS32-NEXT: copy_s.w $1, $w0[0] ; MIPS32-NEXT: lw $25, %call16(cosf)($gp) ; MIPS32-NEXT: jalr $25 ; MIPS32-NEXT: mtc1 $1, $f12 ; MIPS32-NEXT: mfc1 $1, $f0 ; MIPS32-NEXT: fill.w $w0, $1 ; MIPS32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS32-NEXT: copy_u.h $1, $w0[0] ; MIPS32-NEXT: sh $1, 0($16) ; MIPS32-NEXT: lw $16, 16($sp) # 4-byte Folded Reload ; MIPS32-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: addiu $sp, $sp, 24 ; ; MIPS64-N32-LABEL: fcos: ; MIPS64-N32: # %bb.0: # %entry ; MIPS64-N32-NEXT: addiu $sp, $sp, -32 ; MIPS64-N32-NEXT: .cfi_def_cfa_offset 32 ; MIPS64-N32-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: sd $16, 8($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: .cfi_offset 31, -8 ; MIPS64-N32-NEXT: .cfi_offset 28, -16 ; MIPS64-N32-NEXT: .cfi_offset 16, -24 ; MIPS64-N32-NEXT: lui $1, %hi(%neg(%gp_rel(fcos))) ; MIPS64-N32-NEXT: addu $1, $1, $25 ; MIPS64-N32-NEXT: addiu $gp, $1, %lo(%neg(%gp_rel(fcos))) ; MIPS64-N32-NEXT: lw $16, %got_disp(g)($gp) ; MIPS64-N32-NEXT: lh $1, 0($16) ; MIPS64-N32-NEXT: fill.h $w0, $1 ; MIPS64-N32-NEXT: fexupr.w $w0, $w0 ; MIPS64-N32-NEXT: copy_s.w $1, $w0[0] ; MIPS64-N32-NEXT: lw $25, %call16(cosf)($gp) ; MIPS64-N32-NEXT: jalr $25 ; MIPS64-N32-NEXT: mtc1 $1, $f12 ; MIPS64-N32-NEXT: mfc1 $1, $f0 ; MIPS64-N32-NEXT: fill.w $w0, $1 ; MIPS64-N32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N32-NEXT: copy_u.h $1, $w0[0] ; MIPS64-N32-NEXT: sh $1, 0($16) ; MIPS64-N32-NEXT: ld $16, 8($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: jr $ra ; MIPS64-N32-NEXT: addiu $sp, $sp, 32 ; ; MIPS64-N64-LABEL: fcos: ; MIPS64-N64: # %bb.0: # %entry ; MIPS64-N64-NEXT: daddiu $sp, $sp, -32 ; MIPS64-N64-NEXT: .cfi_def_cfa_offset 32 ; MIPS64-N64-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: sd $16, 8($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: .cfi_offset 31, -8 ; MIPS64-N64-NEXT: .cfi_offset 28, -16 ; MIPS64-N64-NEXT: .cfi_offset 16, -24 ; MIPS64-N64-NEXT: lui $1, %hi(%neg(%gp_rel(fcos))) ; MIPS64-N64-NEXT: daddu $1, $1, $25 ; MIPS64-N64-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(fcos))) ; MIPS64-N64-NEXT: ld $16, %got_disp(g)($gp) ; MIPS64-N64-NEXT: lh $1, 0($16) ; MIPS64-N64-NEXT: fill.h $w0, $1 ; MIPS64-N64-NEXT: fexupr.w $w0, $w0 ; MIPS64-N64-NEXT: copy_s.w $1, $w0[0] ; MIPS64-N64-NEXT: ld $25, %call16(cosf)($gp) ; MIPS64-N64-NEXT: jalr $25 ; MIPS64-N64-NEXT: mtc1 $1, $f12 ; MIPS64-N64-NEXT: mfc1 $1, $f0 ; MIPS64-N64-NEXT: fill.w $w0, $1 ; MIPS64-N64-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N64-NEXT: copy_u.h $1, $w0[0] ; MIPS64-N64-NEXT: sh $1, 0($16) ; MIPS64-N64-NEXT: ld $16, 8($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: jr $ra ; MIPS64-N64-NEXT: daddiu $sp, $sp, 32 entry: %0 = load i16, i16* @g, align 2 %1 = call float @llvm.convert.from.fp16.f32(i16 %0) %cos = call float @llvm.cos.f32(float %1) %2 = call i16 @llvm.convert.to.fp16.f32(float %cos) store i16 %2, i16* @g, align 2 ret void } declare float @llvm.exp.f32(float %Val) define void @fexp() { ; MIPS32-LABEL: fexp: ; MIPS32: # %bb.0: # %entry ; MIPS32-NEXT: lui $2, %hi(_gp_disp) ; MIPS32-NEXT: addiu $2, $2, %lo(_gp_disp) ; MIPS32-NEXT: addiu $sp, $sp, -24 ; MIPS32-NEXT: .cfi_def_cfa_offset 24 ; MIPS32-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; MIPS32-NEXT: sw $16, 16($sp) # 4-byte Folded Spill ; MIPS32-NEXT: .cfi_offset 31, -4 ; MIPS32-NEXT: .cfi_offset 16, -8 ; MIPS32-NEXT: addu $gp, $2, $25 ; MIPS32-NEXT: lw $16, %got(g)($gp) ; MIPS32-NEXT: lh $1, 0($16) ; MIPS32-NEXT: fill.h $w0, $1 ; MIPS32-NEXT: fexupr.w $w0, $w0 ; MIPS32-NEXT: copy_s.w $1, $w0[0] ; MIPS32-NEXT: lw $25, %call16(expf)($gp) ; MIPS32-NEXT: jalr $25 ; MIPS32-NEXT: mtc1 $1, $f12 ; MIPS32-NEXT: mfc1 $1, $f0 ; MIPS32-NEXT: fill.w $w0, $1 ; MIPS32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS32-NEXT: copy_u.h $1, $w0[0] ; MIPS32-NEXT: sh $1, 0($16) ; MIPS32-NEXT: lw $16, 16($sp) # 4-byte Folded Reload ; MIPS32-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: addiu $sp, $sp, 24 ; ; MIPS64-N32-LABEL: fexp: ; MIPS64-N32: # %bb.0: # %entry ; MIPS64-N32-NEXT: addiu $sp, $sp, -32 ; MIPS64-N32-NEXT: .cfi_def_cfa_offset 32 ; MIPS64-N32-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: sd $16, 8($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: .cfi_offset 31, -8 ; MIPS64-N32-NEXT: .cfi_offset 28, -16 ; MIPS64-N32-NEXT: .cfi_offset 16, -24 ; MIPS64-N32-NEXT: lui $1, %hi(%neg(%gp_rel(fexp))) ; MIPS64-N32-NEXT: addu $1, $1, $25 ; MIPS64-N32-NEXT: addiu $gp, $1, %lo(%neg(%gp_rel(fexp))) ; MIPS64-N32-NEXT: lw $16, %got_disp(g)($gp) ; MIPS64-N32-NEXT: lh $1, 0($16) ; MIPS64-N32-NEXT: fill.h $w0, $1 ; MIPS64-N32-NEXT: fexupr.w $w0, $w0 ; MIPS64-N32-NEXT: copy_s.w $1, $w0[0] ; MIPS64-N32-NEXT: lw $25, %call16(expf)($gp) ; MIPS64-N32-NEXT: jalr $25 ; MIPS64-N32-NEXT: mtc1 $1, $f12 ; MIPS64-N32-NEXT: mfc1 $1, $f0 ; MIPS64-N32-NEXT: fill.w $w0, $1 ; MIPS64-N32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N32-NEXT: copy_u.h $1, $w0[0] ; MIPS64-N32-NEXT: sh $1, 0($16) ; MIPS64-N32-NEXT: ld $16, 8($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: jr $ra ; MIPS64-N32-NEXT: addiu $sp, $sp, 32 ; ; MIPS64-N64-LABEL: fexp: ; MIPS64-N64: # %bb.0: # %entry ; MIPS64-N64-NEXT: daddiu $sp, $sp, -32 ; MIPS64-N64-NEXT: .cfi_def_cfa_offset 32 ; MIPS64-N64-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: sd $16, 8($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: .cfi_offset 31, -8 ; MIPS64-N64-NEXT: .cfi_offset 28, -16 ; MIPS64-N64-NEXT: .cfi_offset 16, -24 ; MIPS64-N64-NEXT: lui $1, %hi(%neg(%gp_rel(fexp))) ; MIPS64-N64-NEXT: daddu $1, $1, $25 ; MIPS64-N64-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(fexp))) ; MIPS64-N64-NEXT: ld $16, %got_disp(g)($gp) ; MIPS64-N64-NEXT: lh $1, 0($16) ; MIPS64-N64-NEXT: fill.h $w0, $1 ; MIPS64-N64-NEXT: fexupr.w $w0, $w0 ; MIPS64-N64-NEXT: copy_s.w $1, $w0[0] ; MIPS64-N64-NEXT: ld $25, %call16(expf)($gp) ; MIPS64-N64-NEXT: jalr $25 ; MIPS64-N64-NEXT: mtc1 $1, $f12 ; MIPS64-N64-NEXT: mfc1 $1, $f0 ; MIPS64-N64-NEXT: fill.w $w0, $1 ; MIPS64-N64-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N64-NEXT: copy_u.h $1, $w0[0] ; MIPS64-N64-NEXT: sh $1, 0($16) ; MIPS64-N64-NEXT: ld $16, 8($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: jr $ra ; MIPS64-N64-NEXT: daddiu $sp, $sp, 32 entry: %0 = load i16, i16* @g, align 2 %1 = call float @llvm.convert.from.fp16.f32(i16 %0) %exp = call float @llvm.exp.f32(float %1) %2 = call i16 @llvm.convert.to.fp16.f32(float %exp) store i16 %2, i16* @g, align 2 ret void } declare float @llvm.exp2.f32(float %Val) define void @fexp2() { ; MIPS32-LABEL: fexp2: ; MIPS32: # %bb.0: # %entry ; MIPS32-NEXT: lui $2, %hi(_gp_disp) ; MIPS32-NEXT: addiu $2, $2, %lo(_gp_disp) ; MIPS32-NEXT: addiu $sp, $sp, -24 ; MIPS32-NEXT: .cfi_def_cfa_offset 24 ; MIPS32-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; MIPS32-NEXT: sw $16, 16($sp) # 4-byte Folded Spill ; MIPS32-NEXT: .cfi_offset 31, -4 ; MIPS32-NEXT: .cfi_offset 16, -8 ; MIPS32-NEXT: addu $gp, $2, $25 ; MIPS32-NEXT: lw $16, %got(g)($gp) ; MIPS32-NEXT: lh $1, 0($16) ; MIPS32-NEXT: fill.h $w0, $1 ; MIPS32-NEXT: fexupr.w $w0, $w0 ; MIPS32-NEXT: copy_s.w $1, $w0[0] ; MIPS32-NEXT: lw $25, %call16(exp2f)($gp) ; MIPS32-NEXT: jalr $25 ; MIPS32-NEXT: mtc1 $1, $f12 ; MIPS32-NEXT: mfc1 $1, $f0 ; MIPS32-NEXT: fill.w $w0, $1 ; MIPS32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS32-NEXT: copy_u.h $1, $w0[0] ; MIPS32-NEXT: sh $1, 0($16) ; MIPS32-NEXT: lw $16, 16($sp) # 4-byte Folded Reload ; MIPS32-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: addiu $sp, $sp, 24 ; ; MIPS64-N32-LABEL: fexp2: ; MIPS64-N32: # %bb.0: # %entry ; MIPS64-N32-NEXT: addiu $sp, $sp, -32 ; MIPS64-N32-NEXT: .cfi_def_cfa_offset 32 ; MIPS64-N32-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: sd $16, 8($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: .cfi_offset 31, -8 ; MIPS64-N32-NEXT: .cfi_offset 28, -16 ; MIPS64-N32-NEXT: .cfi_offset 16, -24 ; MIPS64-N32-NEXT: lui $1, %hi(%neg(%gp_rel(fexp2))) ; MIPS64-N32-NEXT: addu $1, $1, $25 ; MIPS64-N32-NEXT: addiu $gp, $1, %lo(%neg(%gp_rel(fexp2))) ; MIPS64-N32-NEXT: lw $16, %got_disp(g)($gp) ; MIPS64-N32-NEXT: lh $1, 0($16) ; MIPS64-N32-NEXT: fill.h $w0, $1 ; MIPS64-N32-NEXT: fexupr.w $w0, $w0 ; MIPS64-N32-NEXT: copy_s.w $1, $w0[0] ; MIPS64-N32-NEXT: lw $25, %call16(exp2f)($gp) ; MIPS64-N32-NEXT: jalr $25 ; MIPS64-N32-NEXT: mtc1 $1, $f12 ; MIPS64-N32-NEXT: mfc1 $1, $f0 ; MIPS64-N32-NEXT: fill.w $w0, $1 ; MIPS64-N32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N32-NEXT: copy_u.h $1, $w0[0] ; MIPS64-N32-NEXT: sh $1, 0($16) ; MIPS64-N32-NEXT: ld $16, 8($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: jr $ra ; MIPS64-N32-NEXT: addiu $sp, $sp, 32 ; ; MIPS64-N64-LABEL: fexp2: ; MIPS64-N64: # %bb.0: # %entry ; MIPS64-N64-NEXT: daddiu $sp, $sp, -32 ; MIPS64-N64-NEXT: .cfi_def_cfa_offset 32 ; MIPS64-N64-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: sd $16, 8($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: .cfi_offset 31, -8 ; MIPS64-N64-NEXT: .cfi_offset 28, -16 ; MIPS64-N64-NEXT: .cfi_offset 16, -24 ; MIPS64-N64-NEXT: lui $1, %hi(%neg(%gp_rel(fexp2))) ; MIPS64-N64-NEXT: daddu $1, $1, $25 ; MIPS64-N64-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(fexp2))) ; MIPS64-N64-NEXT: ld $16, %got_disp(g)($gp) ; MIPS64-N64-NEXT: lh $1, 0($16) ; MIPS64-N64-NEXT: fill.h $w0, $1 ; MIPS64-N64-NEXT: fexupr.w $w0, $w0 ; MIPS64-N64-NEXT: copy_s.w $1, $w0[0] ; MIPS64-N64-NEXT: ld $25, %call16(exp2f)($gp) ; MIPS64-N64-NEXT: jalr $25 ; MIPS64-N64-NEXT: mtc1 $1, $f12 ; MIPS64-N64-NEXT: mfc1 $1, $f0 ; MIPS64-N64-NEXT: fill.w $w0, $1 ; MIPS64-N64-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N64-NEXT: copy_u.h $1, $w0[0] ; MIPS64-N64-NEXT: sh $1, 0($16) ; MIPS64-N64-NEXT: ld $16, 8($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: jr $ra ; MIPS64-N64-NEXT: daddiu $sp, $sp, 32 entry: %0 = load i16, i16* @g, align 2 %1 = call float @llvm.convert.from.fp16.f32(i16 %0) %exp2 = call float @llvm.exp2.f32(float %1) %2 = call i16 @llvm.convert.to.fp16.f32(float %exp2) store i16 %2, i16* @g, align 2 ret void } declare float @llvm.fma.f32(float, float, float) define void @ffma(float %b, float %c) { ; MIPS32-LABEL: ffma: ; MIPS32: # %bb.0: # %entry ; MIPS32-NEXT: lui $2, %hi(_gp_disp) ; MIPS32-NEXT: addiu $2, $2, %lo(_gp_disp) ; MIPS32-NEXT: addiu $sp, $sp, -24 ; MIPS32-NEXT: .cfi_def_cfa_offset 24 ; MIPS32-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; MIPS32-NEXT: sw $16, 16($sp) # 4-byte Folded Spill ; MIPS32-NEXT: .cfi_offset 31, -4 ; MIPS32-NEXT: .cfi_offset 16, -8 ; MIPS32-NEXT: addu $gp, $2, $25 ; MIPS32-NEXT: mov.s $f0, $f12 ; MIPS32-NEXT: mfc1 $6, $f14 ; MIPS32-NEXT: lw $16, %got(g)($gp) ; MIPS32-NEXT: lh $1, 0($16) ; MIPS32-NEXT: fill.h $w1, $1 ; MIPS32-NEXT: fexupr.w $w1, $w1 ; MIPS32-NEXT: copy_s.w $1, $w1[0] ; MIPS32-NEXT: mtc1 $1, $f12 ; MIPS32-NEXT: lw $25, %call16(fmaf)($gp) ; MIPS32-NEXT: jalr $25 ; MIPS32-NEXT: mov.s $f14, $f0 ; MIPS32-NEXT: mfc1 $1, $f0 ; MIPS32-NEXT: fill.w $w0, $1 ; MIPS32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS32-NEXT: copy_u.h $1, $w0[0] ; MIPS32-NEXT: sh $1, 0($16) ; MIPS32-NEXT: lw $16, 16($sp) # 4-byte Folded Reload ; MIPS32-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: addiu $sp, $sp, 24 ; ; MIPS64-N32-LABEL: ffma: ; MIPS64-N32: # %bb.0: # %entry ; MIPS64-N32-NEXT: addiu $sp, $sp, -32 ; MIPS64-N32-NEXT: .cfi_def_cfa_offset 32 ; MIPS64-N32-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: sd $16, 8($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: .cfi_offset 31, -8 ; MIPS64-N32-NEXT: .cfi_offset 28, -16 ; MIPS64-N32-NEXT: .cfi_offset 16, -24 ; MIPS64-N32-NEXT: lui $1, %hi(%neg(%gp_rel(ffma))) ; MIPS64-N32-NEXT: addu $1, $1, $25 ; MIPS64-N32-NEXT: addiu $gp, $1, %lo(%neg(%gp_rel(ffma))) ; MIPS64-N32-NEXT: mov.s $f14, $f13 ; MIPS64-N32-NEXT: mov.s $f13, $f12 ; MIPS64-N32-NEXT: lw $16, %got_disp(g)($gp) ; MIPS64-N32-NEXT: lh $1, 0($16) ; MIPS64-N32-NEXT: fill.h $w0, $1 ; MIPS64-N32-NEXT: fexupr.w $w0, $w0 ; MIPS64-N32-NEXT: copy_s.w $1, $w0[0] ; MIPS64-N32-NEXT: lw $25, %call16(fmaf)($gp) ; MIPS64-N32-NEXT: jalr $25 ; MIPS64-N32-NEXT: mtc1 $1, $f12 ; MIPS64-N32-NEXT: mfc1 $1, $f0 ; MIPS64-N32-NEXT: fill.w $w0, $1 ; MIPS64-N32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N32-NEXT: copy_u.h $1, $w0[0] ; MIPS64-N32-NEXT: sh $1, 0($16) ; MIPS64-N32-NEXT: ld $16, 8($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: jr $ra ; MIPS64-N32-NEXT: addiu $sp, $sp, 32 ; ; MIPS64-N64-LABEL: ffma: ; MIPS64-N64: # %bb.0: # %entry ; MIPS64-N64-NEXT: daddiu $sp, $sp, -32 ; MIPS64-N64-NEXT: .cfi_def_cfa_offset 32 ; MIPS64-N64-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: sd $16, 8($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: .cfi_offset 31, -8 ; MIPS64-N64-NEXT: .cfi_offset 28, -16 ; MIPS64-N64-NEXT: .cfi_offset 16, -24 ; MIPS64-N64-NEXT: lui $1, %hi(%neg(%gp_rel(ffma))) ; MIPS64-N64-NEXT: daddu $1, $1, $25 ; MIPS64-N64-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(ffma))) ; MIPS64-N64-NEXT: mov.s $f14, $f13 ; MIPS64-N64-NEXT: mov.s $f13, $f12 ; MIPS64-N64-NEXT: ld $16, %got_disp(g)($gp) ; MIPS64-N64-NEXT: lh $1, 0($16) ; MIPS64-N64-NEXT: fill.h $w0, $1 ; MIPS64-N64-NEXT: fexupr.w $w0, $w0 ; MIPS64-N64-NEXT: copy_s.w $1, $w0[0] ; MIPS64-N64-NEXT: ld $25, %call16(fmaf)($gp) ; MIPS64-N64-NEXT: jalr $25 ; MIPS64-N64-NEXT: mtc1 $1, $f12 ; MIPS64-N64-NEXT: mfc1 $1, $f0 ; MIPS64-N64-NEXT: fill.w $w0, $1 ; MIPS64-N64-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N64-NEXT: copy_u.h $1, $w0[0] ; MIPS64-N64-NEXT: sh $1, 0($16) ; MIPS64-N64-NEXT: ld $16, 8($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: jr $ra ; MIPS64-N64-NEXT: daddiu $sp, $sp, 32 entry: %0 = load i16, i16* @g, align 2 %1 = call float @llvm.convert.from.fp16.f32(i16 %0) %fma = call float @llvm.fma.f32(float %1, float %b, float %c) %2 = call i16 @llvm.convert.to.fp16.f32(float %fma) store i16 %2, i16* @g, align 2 ret void } ; FIXME: For MIPSR6, this should produced the maddf.s instruction. MIPSR5 cannot ; fuse the operation such that the intermediate result is not rounded. declare float @llvm.fmuladd.f32(float, float, float) define void @ffmuladd(float %b, float %c) { ; MIPS32-O32-LABEL: ffmuladd: ; MIPS32-O32: # %bb.0: # %entry ; MIPS32-O32-NEXT: lui $2, %hi(_gp_disp) ; MIPS32-O32-NEXT: addiu $2, $2, %lo(_gp_disp) ; MIPS32-O32-NEXT: addu $1, $2, $25 ; MIPS32-O32-NEXT: lw $1, %got(g)($1) ; MIPS32-O32-NEXT: lh $2, 0($1) ; MIPS32-O32-NEXT: fill.h $w0, $2 ; MIPS32-O32-NEXT: fexupr.w $w0, $w0 ; MIPS32-O32-NEXT: copy_s.w $2, $w0[0] ; MIPS32-O32-NEXT: mtc1 $2, $f0 ; MIPS32-O32-NEXT: madd.s $f0, $f14, $f0, $f12 ; MIPS32-O32-NEXT: mfc1 $2, $f0 ; MIPS32-O32-NEXT: fill.w $w0, $2 ; MIPS32-O32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS32-O32-NEXT: copy_u.h $2, $w0[0] ; MIPS32-O32-NEXT: jr $ra ; MIPS32-O32-NEXT: sh $2, 0($1) ; ; MIPS64R5-N32-LABEL: ffmuladd: ; MIPS64R5-N32: # %bb.0: # %entry ; MIPS64R5-N32-NEXT: lui $1, %hi(%neg(%gp_rel(ffmuladd))) ; MIPS64R5-N32-NEXT: addu $1, $1, $25 ; MIPS64R5-N32-NEXT: addiu $1, $1, %lo(%neg(%gp_rel(ffmuladd))) ; MIPS64R5-N32-NEXT: lw $1, %got_disp(g)($1) ; MIPS64R5-N32-NEXT: lh $2, 0($1) ; MIPS64R5-N32-NEXT: fill.h $w0, $2 ; MIPS64R5-N32-NEXT: fexupr.w $w0, $w0 ; MIPS64R5-N32-NEXT: copy_s.w $2, $w0[0] ; MIPS64R5-N32-NEXT: mtc1 $2, $f0 ; MIPS64R5-N32-NEXT: madd.s $f0, $f13, $f0, $f12 ; MIPS64R5-N32-NEXT: mfc1 $2, $f0 ; MIPS64R5-N32-NEXT: fill.w $w0, $2 ; MIPS64R5-N32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64R5-N32-NEXT: copy_u.h $2, $w0[0] ; MIPS64R5-N32-NEXT: jr $ra ; MIPS64R5-N32-NEXT: sh $2, 0($1) ; ; MIPS64R5-N64-LABEL: ffmuladd: ; MIPS64R5-N64: # %bb.0: # %entry ; MIPS64R5-N64-NEXT: lui $1, %hi(%neg(%gp_rel(ffmuladd))) ; MIPS64R5-N64-NEXT: daddu $1, $1, $25 ; MIPS64R5-N64-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(ffmuladd))) ; MIPS64R5-N64-NEXT: ld $1, %got_disp(g)($1) ; MIPS64R5-N64-NEXT: lh $2, 0($1) ; MIPS64R5-N64-NEXT: fill.h $w0, $2 ; MIPS64R5-N64-NEXT: fexupr.w $w0, $w0 ; MIPS64R5-N64-NEXT: copy_s.w $2, $w0[0] ; MIPS64R5-N64-NEXT: mtc1 $2, $f0 ; MIPS64R5-N64-NEXT: madd.s $f0, $f13, $f0, $f12 ; MIPS64R5-N64-NEXT: mfc1 $2, $f0 ; MIPS64R5-N64-NEXT: fill.w $w0, $2 ; MIPS64R5-N64-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64R5-N64-NEXT: copy_u.h $2, $w0[0] ; MIPS64R5-N64-NEXT: jr $ra ; MIPS64R5-N64-NEXT: sh $2, 0($1) ; ; MIPSR6-O32-LABEL: ffmuladd: ; MIPSR6-O32: # %bb.0: # %entry ; MIPSR6-O32-NEXT: lui $2, %hi(_gp_disp) ; MIPSR6-O32-NEXT: addiu $2, $2, %lo(_gp_disp) ; MIPSR6-O32-NEXT: addu $1, $2, $25 ; MIPSR6-O32-NEXT: lw $1, %got(g)($1) ; MIPSR6-O32-NEXT: lh $2, 0($1) ; MIPSR6-O32-NEXT: fill.h $w0, $2 ; MIPSR6-O32-NEXT: fexupr.w $w0, $w0 ; MIPSR6-O32-NEXT: copy_s.w $2, $w0[0] ; MIPSR6-O32-NEXT: mtc1 $2, $f0 ; MIPSR6-O32-NEXT: mul.s $f0, $f0, $f12 ; MIPSR6-O32-NEXT: add.s $f0, $f0, $f14 ; MIPSR6-O32-NEXT: mfc1 $2, $f0 ; MIPSR6-O32-NEXT: fill.w $w0, $2 ; MIPSR6-O32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPSR6-O32-NEXT: copy_u.h $2, $w0[0] ; MIPSR6-O32-NEXT: jr $ra ; MIPSR6-O32-NEXT: sh $2, 0($1) ; ; MIPSR6-N32-LABEL: ffmuladd: ; MIPSR6-N32: # %bb.0: # %entry ; MIPSR6-N32-NEXT: lui $1, %hi(%neg(%gp_rel(ffmuladd))) ; MIPSR6-N32-NEXT: addu $1, $1, $25 ; MIPSR6-N32-NEXT: addiu $1, $1, %lo(%neg(%gp_rel(ffmuladd))) ; MIPSR6-N32-NEXT: lw $1, %got_disp(g)($1) ; MIPSR6-N32-NEXT: lh $2, 0($1) ; MIPSR6-N32-NEXT: fill.h $w0, $2 ; MIPSR6-N32-NEXT: fexupr.w $w0, $w0 ; MIPSR6-N32-NEXT: copy_s.w $2, $w0[0] ; MIPSR6-N32-NEXT: mtc1 $2, $f0 ; MIPSR6-N32-NEXT: mul.s $f0, $f0, $f12 ; MIPSR6-N32-NEXT: add.s $f0, $f0, $f13 ; MIPSR6-N32-NEXT: mfc1 $2, $f0 ; MIPSR6-N32-NEXT: fill.w $w0, $2 ; MIPSR6-N32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPSR6-N32-NEXT: copy_u.h $2, $w0[0] ; MIPSR6-N32-NEXT: jr $ra ; MIPSR6-N32-NEXT: sh $2, 0($1) ; ; MIPSR6-N64-LABEL: ffmuladd: ; MIPSR6-N64: # %bb.0: # %entry ; MIPSR6-N64-NEXT: lui $1, %hi(%neg(%gp_rel(ffmuladd))) ; MIPSR6-N64-NEXT: daddu $1, $1, $25 ; MIPSR6-N64-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(ffmuladd))) ; MIPSR6-N64-NEXT: ld $1, %got_disp(g)($1) ; MIPSR6-N64-NEXT: lh $2, 0($1) ; MIPSR6-N64-NEXT: fill.h $w0, $2 ; MIPSR6-N64-NEXT: fexupr.w $w0, $w0 ; MIPSR6-N64-NEXT: copy_s.w $2, $w0[0] ; MIPSR6-N64-NEXT: mtc1 $2, $f0 ; MIPSR6-N64-NEXT: mul.s $f0, $f0, $f12 ; MIPSR6-N64-NEXT: add.s $f0, $f0, $f13 ; MIPSR6-N64-NEXT: mfc1 $2, $f0 ; MIPSR6-N64-NEXT: fill.w $w0, $2 ; MIPSR6-N64-NEXT: fexdo.h $w0, $w0, $w0 ; MIPSR6-N64-NEXT: copy_u.h $2, $w0[0] ; MIPSR6-N64-NEXT: jr $ra ; MIPSR6-N64-NEXT: sh $2, 0($1) entry: %0 = load i16, i16* @g, align 2 %1 = call float @llvm.convert.from.fp16.f32(i16 %0) ; MIPS32-N32: madd.s $f[[F1:[0-9]]], $f13, $f[[F0]], $f12 ; MIPS32-N64: madd.s $f[[F1:[0-9]]], $f13, $f[[F0]], $f12 %fmuladd = call float @llvm.fmuladd.f32(float %1, float %b, float %c) %2 = call i16 @llvm.convert.to.fp16.f32(float %fmuladd) store i16 %2, i16* @g, align 2 ret void } declare float @llvm.fabs.f32(float %Val) define void @ffabs() { ; MIPS32-LABEL: ffabs: ; MIPS32: # %bb.0: # %entry ; MIPS32-NEXT: lui $2, %hi(_gp_disp) ; MIPS32-NEXT: addiu $2, $2, %lo(_gp_disp) ; MIPS32-NEXT: addu $1, $2, $25 ; MIPS32-NEXT: lw $1, %got(g)($1) ; MIPS32-NEXT: lh $2, 0($1) ; MIPS32-NEXT: fill.h $w0, $2 ; MIPS32-NEXT: fexupr.w $w0, $w0 ; MIPS32-NEXT: copy_s.w $2, $w0[0] ; MIPS32-NEXT: mtc1 $2, $f0 ; MIPS32-NEXT: abs.s $f0, $f0 ; MIPS32-NEXT: mfc1 $2, $f0 ; MIPS32-NEXT: fill.w $w0, $2 ; MIPS32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS32-NEXT: copy_u.h $2, $w0[0] ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: sh $2, 0($1) ; ; MIPS64-N32-LABEL: ffabs: ; MIPS64-N32: # %bb.0: # %entry ; MIPS64-N32-NEXT: lui $1, %hi(%neg(%gp_rel(ffabs))) ; MIPS64-N32-NEXT: addu $1, $1, $25 ; MIPS64-N32-NEXT: addiu $1, $1, %lo(%neg(%gp_rel(ffabs))) ; MIPS64-N32-NEXT: lw $1, %got_disp(g)($1) ; MIPS64-N32-NEXT: lh $2, 0($1) ; MIPS64-N32-NEXT: fill.h $w0, $2 ; MIPS64-N32-NEXT: fexupr.w $w0, $w0 ; MIPS64-N32-NEXT: copy_s.w $2, $w0[0] ; MIPS64-N32-NEXT: mtc1 $2, $f0 ; MIPS64-N32-NEXT: abs.s $f0, $f0 ; MIPS64-N32-NEXT: mfc1 $2, $f0 ; MIPS64-N32-NEXT: fill.w $w0, $2 ; MIPS64-N32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N32-NEXT: copy_u.h $2, $w0[0] ; MIPS64-N32-NEXT: jr $ra ; MIPS64-N32-NEXT: sh $2, 0($1) ; ; MIPS64-N64-LABEL: ffabs: ; MIPS64-N64: # %bb.0: # %entry ; MIPS64-N64-NEXT: lui $1, %hi(%neg(%gp_rel(ffabs))) ; MIPS64-N64-NEXT: daddu $1, $1, $25 ; MIPS64-N64-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(ffabs))) ; MIPS64-N64-NEXT: ld $1, %got_disp(g)($1) ; MIPS64-N64-NEXT: lh $2, 0($1) ; MIPS64-N64-NEXT: fill.h $w0, $2 ; MIPS64-N64-NEXT: fexupr.w $w0, $w0 ; MIPS64-N64-NEXT: copy_s.w $2, $w0[0] ; MIPS64-N64-NEXT: mtc1 $2, $f0 ; MIPS64-N64-NEXT: abs.s $f0, $f0 ; MIPS64-N64-NEXT: mfc1 $2, $f0 ; MIPS64-N64-NEXT: fill.w $w0, $2 ; MIPS64-N64-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N64-NEXT: copy_u.h $2, $w0[0] ; MIPS64-N64-NEXT: jr $ra ; MIPS64-N64-NEXT: sh $2, 0($1) entry: %0 = load i16, i16* @g, align 2 %1 = call float @llvm.convert.from.fp16.f32(i16 %0) %fabs = call float @llvm.fabs.f32(float %1) %2 = call i16 @llvm.convert.to.fp16.f32(float %fabs) store i16 %2, i16* @g, align 2 ret void } declare float @llvm.minnum.f32(float %Val, float %b) define void @fminnum(float %b) { ; MIPS32-LABEL: fminnum: ; MIPS32: # %bb.0: # %entry ; MIPS32-NEXT: lui $2, %hi(_gp_disp) ; MIPS32-NEXT: addiu $2, $2, %lo(_gp_disp) ; MIPS32-NEXT: addiu $sp, $sp, -24 ; MIPS32-NEXT: .cfi_def_cfa_offset 24 ; MIPS32-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; MIPS32-NEXT: sw $16, 16($sp) # 4-byte Folded Spill ; MIPS32-NEXT: .cfi_offset 31, -4 ; MIPS32-NEXT: .cfi_offset 16, -8 ; MIPS32-NEXT: addu $gp, $2, $25 ; MIPS32-NEXT: mov.s $f14, $f12 ; MIPS32-NEXT: lw $16, %got(g)($gp) ; MIPS32-NEXT: lh $1, 0($16) ; MIPS32-NEXT: fill.h $w0, $1 ; MIPS32-NEXT: fexupr.w $w0, $w0 ; MIPS32-NEXT: copy_s.w $1, $w0[0] ; MIPS32-NEXT: lw $25, %call16(fminf)($gp) ; MIPS32-NEXT: jalr $25 ; MIPS32-NEXT: mtc1 $1, $f12 ; MIPS32-NEXT: mfc1 $1, $f0 ; MIPS32-NEXT: fill.w $w0, $1 ; MIPS32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS32-NEXT: copy_u.h $1, $w0[0] ; MIPS32-NEXT: sh $1, 0($16) ; MIPS32-NEXT: lw $16, 16($sp) # 4-byte Folded Reload ; MIPS32-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: addiu $sp, $sp, 24 ; ; MIPS64-N32-LABEL: fminnum: ; MIPS64-N32: # %bb.0: # %entry ; MIPS64-N32-NEXT: addiu $sp, $sp, -32 ; MIPS64-N32-NEXT: .cfi_def_cfa_offset 32 ; MIPS64-N32-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: sd $16, 8($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: .cfi_offset 31, -8 ; MIPS64-N32-NEXT: .cfi_offset 28, -16 ; MIPS64-N32-NEXT: .cfi_offset 16, -24 ; MIPS64-N32-NEXT: lui $1, %hi(%neg(%gp_rel(fminnum))) ; MIPS64-N32-NEXT: addu $1, $1, $25 ; MIPS64-N32-NEXT: addiu $gp, $1, %lo(%neg(%gp_rel(fminnum))) ; MIPS64-N32-NEXT: mov.s $f13, $f12 ; MIPS64-N32-NEXT: lw $16, %got_disp(g)($gp) ; MIPS64-N32-NEXT: lh $1, 0($16) ; MIPS64-N32-NEXT: fill.h $w0, $1 ; MIPS64-N32-NEXT: fexupr.w $w0, $w0 ; MIPS64-N32-NEXT: copy_s.w $1, $w0[0] ; MIPS64-N32-NEXT: lw $25, %call16(fminf)($gp) ; MIPS64-N32-NEXT: jalr $25 ; MIPS64-N32-NEXT: mtc1 $1, $f12 ; MIPS64-N32-NEXT: mfc1 $1, $f0 ; MIPS64-N32-NEXT: fill.w $w0, $1 ; MIPS64-N32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N32-NEXT: copy_u.h $1, $w0[0] ; MIPS64-N32-NEXT: sh $1, 0($16) ; MIPS64-N32-NEXT: ld $16, 8($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: jr $ra ; MIPS64-N32-NEXT: addiu $sp, $sp, 32 ; ; MIPS64-N64-LABEL: fminnum: ; MIPS64-N64: # %bb.0: # %entry ; MIPS64-N64-NEXT: daddiu $sp, $sp, -32 ; MIPS64-N64-NEXT: .cfi_def_cfa_offset 32 ; MIPS64-N64-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: sd $16, 8($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: .cfi_offset 31, -8 ; MIPS64-N64-NEXT: .cfi_offset 28, -16 ; MIPS64-N64-NEXT: .cfi_offset 16, -24 ; MIPS64-N64-NEXT: lui $1, %hi(%neg(%gp_rel(fminnum))) ; MIPS64-N64-NEXT: daddu $1, $1, $25 ; MIPS64-N64-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(fminnum))) ; MIPS64-N64-NEXT: mov.s $f13, $f12 ; MIPS64-N64-NEXT: ld $16, %got_disp(g)($gp) ; MIPS64-N64-NEXT: lh $1, 0($16) ; MIPS64-N64-NEXT: fill.h $w0, $1 ; MIPS64-N64-NEXT: fexupr.w $w0, $w0 ; MIPS64-N64-NEXT: copy_s.w $1, $w0[0] ; MIPS64-N64-NEXT: ld $25, %call16(fminf)($gp) ; MIPS64-N64-NEXT: jalr $25 ; MIPS64-N64-NEXT: mtc1 $1, $f12 ; MIPS64-N64-NEXT: mfc1 $1, $f0 ; MIPS64-N64-NEXT: fill.w $w0, $1 ; MIPS64-N64-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N64-NEXT: copy_u.h $1, $w0[0] ; MIPS64-N64-NEXT: sh $1, 0($16) ; MIPS64-N64-NEXT: ld $16, 8($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: jr $ra ; MIPS64-N64-NEXT: daddiu $sp, $sp, 32 entry: %0 = load i16, i16* @g, align 2 %1 = call float @llvm.convert.from.fp16.f32(i16 %0) %minnum = call float @llvm.minnum.f32(float %1, float %b) %2 = call i16 @llvm.convert.to.fp16.f32(float %minnum) store i16 %2, i16* @g, align 2 ret void } declare float @llvm.maxnum.f32(float %Val, float %b) define void @fmaxnum(float %b) { ; MIPS32-LABEL: fmaxnum: ; MIPS32: # %bb.0: # %entry ; MIPS32-NEXT: lui $2, %hi(_gp_disp) ; MIPS32-NEXT: addiu $2, $2, %lo(_gp_disp) ; MIPS32-NEXT: addiu $sp, $sp, -24 ; MIPS32-NEXT: .cfi_def_cfa_offset 24 ; MIPS32-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; MIPS32-NEXT: sw $16, 16($sp) # 4-byte Folded Spill ; MIPS32-NEXT: .cfi_offset 31, -4 ; MIPS32-NEXT: .cfi_offset 16, -8 ; MIPS32-NEXT: addu $gp, $2, $25 ; MIPS32-NEXT: mov.s $f14, $f12 ; MIPS32-NEXT: lw $16, %got(g)($gp) ; MIPS32-NEXT: lh $1, 0($16) ; MIPS32-NEXT: fill.h $w0, $1 ; MIPS32-NEXT: fexupr.w $w0, $w0 ; MIPS32-NEXT: copy_s.w $1, $w0[0] ; MIPS32-NEXT: lw $25, %call16(fmaxf)($gp) ; MIPS32-NEXT: jalr $25 ; MIPS32-NEXT: mtc1 $1, $f12 ; MIPS32-NEXT: mfc1 $1, $f0 ; MIPS32-NEXT: fill.w $w0, $1 ; MIPS32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS32-NEXT: copy_u.h $1, $w0[0] ; MIPS32-NEXT: sh $1, 0($16) ; MIPS32-NEXT: lw $16, 16($sp) # 4-byte Folded Reload ; MIPS32-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: addiu $sp, $sp, 24 ; ; MIPS64-N32-LABEL: fmaxnum: ; MIPS64-N32: # %bb.0: # %entry ; MIPS64-N32-NEXT: addiu $sp, $sp, -32 ; MIPS64-N32-NEXT: .cfi_def_cfa_offset 32 ; MIPS64-N32-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: sd $16, 8($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: .cfi_offset 31, -8 ; MIPS64-N32-NEXT: .cfi_offset 28, -16 ; MIPS64-N32-NEXT: .cfi_offset 16, -24 ; MIPS64-N32-NEXT: lui $1, %hi(%neg(%gp_rel(fmaxnum))) ; MIPS64-N32-NEXT: addu $1, $1, $25 ; MIPS64-N32-NEXT: addiu $gp, $1, %lo(%neg(%gp_rel(fmaxnum))) ; MIPS64-N32-NEXT: mov.s $f13, $f12 ; MIPS64-N32-NEXT: lw $16, %got_disp(g)($gp) ; MIPS64-N32-NEXT: lh $1, 0($16) ; MIPS64-N32-NEXT: fill.h $w0, $1 ; MIPS64-N32-NEXT: fexupr.w $w0, $w0 ; MIPS64-N32-NEXT: copy_s.w $1, $w0[0] ; MIPS64-N32-NEXT: lw $25, %call16(fmaxf)($gp) ; MIPS64-N32-NEXT: jalr $25 ; MIPS64-N32-NEXT: mtc1 $1, $f12 ; MIPS64-N32-NEXT: mfc1 $1, $f0 ; MIPS64-N32-NEXT: fill.w $w0, $1 ; MIPS64-N32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N32-NEXT: copy_u.h $1, $w0[0] ; MIPS64-N32-NEXT: sh $1, 0($16) ; MIPS64-N32-NEXT: ld $16, 8($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: jr $ra ; MIPS64-N32-NEXT: addiu $sp, $sp, 32 ; ; MIPS64-N64-LABEL: fmaxnum: ; MIPS64-N64: # %bb.0: # %entry ; MIPS64-N64-NEXT: daddiu $sp, $sp, -32 ; MIPS64-N64-NEXT: .cfi_def_cfa_offset 32 ; MIPS64-N64-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: sd $16, 8($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: .cfi_offset 31, -8 ; MIPS64-N64-NEXT: .cfi_offset 28, -16 ; MIPS64-N64-NEXT: .cfi_offset 16, -24 ; MIPS64-N64-NEXT: lui $1, %hi(%neg(%gp_rel(fmaxnum))) ; MIPS64-N64-NEXT: daddu $1, $1, $25 ; MIPS64-N64-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(fmaxnum))) ; MIPS64-N64-NEXT: mov.s $f13, $f12 ; MIPS64-N64-NEXT: ld $16, %got_disp(g)($gp) ; MIPS64-N64-NEXT: lh $1, 0($16) ; MIPS64-N64-NEXT: fill.h $w0, $1 ; MIPS64-N64-NEXT: fexupr.w $w0, $w0 ; MIPS64-N64-NEXT: copy_s.w $1, $w0[0] ; MIPS64-N64-NEXT: ld $25, %call16(fmaxf)($gp) ; MIPS64-N64-NEXT: jalr $25 ; MIPS64-N64-NEXT: mtc1 $1, $f12 ; MIPS64-N64-NEXT: mfc1 $1, $f0 ; MIPS64-N64-NEXT: fill.w $w0, $1 ; MIPS64-N64-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N64-NEXT: copy_u.h $1, $w0[0] ; MIPS64-N64-NEXT: sh $1, 0($16) ; MIPS64-N64-NEXT: ld $16, 8($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: jr $ra ; MIPS64-N64-NEXT: daddiu $sp, $sp, 32 entry: %0 = load i16, i16* @g, align 2 %1 = call float @llvm.convert.from.fp16.f32(i16 %0) %maxnum = call float @llvm.maxnum.f32(float %1, float %b) %2 = call i16 @llvm.convert.to.fp16.f32(float %maxnum) store i16 %2, i16* @g, align 2 ret void } ; This expansion of fcopysign could be done without converting f16 to float. declare float @llvm.copysign.f32(float %Val, float %b) define void @fcopysign(float %b) { ; MIPS32-LABEL: fcopysign: ; MIPS32: # %bb.0: # %entry ; MIPS32-NEXT: lui $2, %hi(_gp_disp) ; MIPS32-NEXT: addiu $2, $2, %lo(_gp_disp) ; MIPS32-NEXT: addu $1, $2, $25 ; MIPS32-NEXT: lw $1, %got(g)($1) ; MIPS32-NEXT: lh $2, 0($1) ; MIPS32-NEXT: fill.h $w0, $2 ; MIPS32-NEXT: fexupr.w $w0, $w0 ; MIPS32-NEXT: copy_s.w $2, $w0[0] ; MIPS32-NEXT: mfc1 $3, $f12 ; MIPS32-NEXT: ext $3, $3, 31, 1 ; MIPS32-NEXT: ins $2, $3, 31, 1 ; MIPS32-NEXT: fill.w $w0, $2 ; MIPS32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS32-NEXT: copy_u.h $2, $w0[0] ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: sh $2, 0($1) ; ; MIPS64-N32-LABEL: fcopysign: ; MIPS64-N32: # %bb.0: # %entry ; MIPS64-N32-NEXT: lui $1, %hi(%neg(%gp_rel(fcopysign))) ; MIPS64-N32-NEXT: addu $1, $1, $25 ; MIPS64-N32-NEXT: addiu $1, $1, %lo(%neg(%gp_rel(fcopysign))) ; MIPS64-N32-NEXT: lw $1, %got_disp(g)($1) ; MIPS64-N32-NEXT: lh $2, 0($1) ; MIPS64-N32-NEXT: fill.h $w0, $2 ; MIPS64-N32-NEXT: fexupr.w $w0, $w0 ; MIPS64-N32-NEXT: copy_s.w $2, $w0[0] ; MIPS64-N32-NEXT: mfc1 $3, $f12 ; MIPS64-N32-NEXT: ext $3, $3, 31, 1 ; MIPS64-N32-NEXT: ins $2, $3, 31, 1 ; MIPS64-N32-NEXT: fill.w $w0, $2 ; MIPS64-N32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N32-NEXT: copy_u.h $2, $w0[0] ; MIPS64-N32-NEXT: jr $ra ; MIPS64-N32-NEXT: sh $2, 0($1) ; ; MIPS64-N64-LABEL: fcopysign: ; MIPS64-N64: # %bb.0: # %entry ; MIPS64-N64-NEXT: lui $1, %hi(%neg(%gp_rel(fcopysign))) ; MIPS64-N64-NEXT: daddu $1, $1, $25 ; MIPS64-N64-NEXT: daddiu $1, $1, %lo(%neg(%gp_rel(fcopysign))) ; MIPS64-N64-NEXT: ld $1, %got_disp(g)($1) ; MIPS64-N64-NEXT: lh $2, 0($1) ; MIPS64-N64-NEXT: fill.h $w0, $2 ; MIPS64-N64-NEXT: fexupr.w $w0, $w0 ; MIPS64-N64-NEXT: copy_s.w $2, $w0[0] ; MIPS64-N64-NEXT: mfc1 $3, $f12 ; MIPS64-N64-NEXT: ext $3, $3, 31, 1 ; MIPS64-N64-NEXT: ins $2, $3, 31, 1 ; MIPS64-N64-NEXT: fill.w $w0, $2 ; MIPS64-N64-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N64-NEXT: copy_u.h $2, $w0[0] ; MIPS64-N64-NEXT: jr $ra ; MIPS64-N64-NEXT: sh $2, 0($1) entry: %0 = load i16, i16* @g, align 2 %1 = call float @llvm.convert.from.fp16.f32(i16 %0) %copysign = call float @llvm.copysign.f32(float %1, float %b) %2 = call i16 @llvm.convert.to.fp16.f32(float %copysign) store i16 %2, i16* @g, align 2 ret void } declare float @llvm.floor.f32(float %Val) define void @ffloor() { ; MIPS32-LABEL: ffloor: ; MIPS32: # %bb.0: # %entry ; MIPS32-NEXT: lui $2, %hi(_gp_disp) ; MIPS32-NEXT: addiu $2, $2, %lo(_gp_disp) ; MIPS32-NEXT: addiu $sp, $sp, -24 ; MIPS32-NEXT: .cfi_def_cfa_offset 24 ; MIPS32-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; MIPS32-NEXT: sw $16, 16($sp) # 4-byte Folded Spill ; MIPS32-NEXT: .cfi_offset 31, -4 ; MIPS32-NEXT: .cfi_offset 16, -8 ; MIPS32-NEXT: addu $gp, $2, $25 ; MIPS32-NEXT: lw $16, %got(g)($gp) ; MIPS32-NEXT: lh $1, 0($16) ; MIPS32-NEXT: fill.h $w0, $1 ; MIPS32-NEXT: fexupr.w $w0, $w0 ; MIPS32-NEXT: copy_s.w $1, $w0[0] ; MIPS32-NEXT: lw $25, %call16(floorf)($gp) ; MIPS32-NEXT: jalr $25 ; MIPS32-NEXT: mtc1 $1, $f12 ; MIPS32-NEXT: mfc1 $1, $f0 ; MIPS32-NEXT: fill.w $w0, $1 ; MIPS32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS32-NEXT: copy_u.h $1, $w0[0] ; MIPS32-NEXT: sh $1, 0($16) ; MIPS32-NEXT: lw $16, 16($sp) # 4-byte Folded Reload ; MIPS32-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: addiu $sp, $sp, 24 ; ; MIPS64-N32-LABEL: ffloor: ; MIPS64-N32: # %bb.0: # %entry ; MIPS64-N32-NEXT: addiu $sp, $sp, -32 ; MIPS64-N32-NEXT: .cfi_def_cfa_offset 32 ; MIPS64-N32-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: sd $16, 8($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: .cfi_offset 31, -8 ; MIPS64-N32-NEXT: .cfi_offset 28, -16 ; MIPS64-N32-NEXT: .cfi_offset 16, -24 ; MIPS64-N32-NEXT: lui $1, %hi(%neg(%gp_rel(ffloor))) ; MIPS64-N32-NEXT: addu $1, $1, $25 ; MIPS64-N32-NEXT: addiu $gp, $1, %lo(%neg(%gp_rel(ffloor))) ; MIPS64-N32-NEXT: lw $16, %got_disp(g)($gp) ; MIPS64-N32-NEXT: lh $1, 0($16) ; MIPS64-N32-NEXT: fill.h $w0, $1 ; MIPS64-N32-NEXT: fexupr.w $w0, $w0 ; MIPS64-N32-NEXT: copy_s.w $1, $w0[0] ; MIPS64-N32-NEXT: lw $25, %call16(floorf)($gp) ; MIPS64-N32-NEXT: jalr $25 ; MIPS64-N32-NEXT: mtc1 $1, $f12 ; MIPS64-N32-NEXT: mfc1 $1, $f0 ; MIPS64-N32-NEXT: fill.w $w0, $1 ; MIPS64-N32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N32-NEXT: copy_u.h $1, $w0[0] ; MIPS64-N32-NEXT: sh $1, 0($16) ; MIPS64-N32-NEXT: ld $16, 8($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: jr $ra ; MIPS64-N32-NEXT: addiu $sp, $sp, 32 ; ; MIPS64-N64-LABEL: ffloor: ; MIPS64-N64: # %bb.0: # %entry ; MIPS64-N64-NEXT: daddiu $sp, $sp, -32 ; MIPS64-N64-NEXT: .cfi_def_cfa_offset 32 ; MIPS64-N64-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: sd $16, 8($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: .cfi_offset 31, -8 ; MIPS64-N64-NEXT: .cfi_offset 28, -16 ; MIPS64-N64-NEXT: .cfi_offset 16, -24 ; MIPS64-N64-NEXT: lui $1, %hi(%neg(%gp_rel(ffloor))) ; MIPS64-N64-NEXT: daddu $1, $1, $25 ; MIPS64-N64-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(ffloor))) ; MIPS64-N64-NEXT: ld $16, %got_disp(g)($gp) ; MIPS64-N64-NEXT: lh $1, 0($16) ; MIPS64-N64-NEXT: fill.h $w0, $1 ; MIPS64-N64-NEXT: fexupr.w $w0, $w0 ; MIPS64-N64-NEXT: copy_s.w $1, $w0[0] ; MIPS64-N64-NEXT: ld $25, %call16(floorf)($gp) ; MIPS64-N64-NEXT: jalr $25 ; MIPS64-N64-NEXT: mtc1 $1, $f12 ; MIPS64-N64-NEXT: mfc1 $1, $f0 ; MIPS64-N64-NEXT: fill.w $w0, $1 ; MIPS64-N64-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N64-NEXT: copy_u.h $1, $w0[0] ; MIPS64-N64-NEXT: sh $1, 0($16) ; MIPS64-N64-NEXT: ld $16, 8($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: jr $ra ; MIPS64-N64-NEXT: daddiu $sp, $sp, 32 entry: %0 = load i16, i16* @g, align 2 %1 = call float @llvm.convert.from.fp16.f32(i16 %0) %floor = call float @llvm.floor.f32(float %1) %2 = call i16 @llvm.convert.to.fp16.f32(float %floor) store i16 %2, i16* @g, align 2 ret void } declare float @llvm.ceil.f32(float %Val) define void @fceil() { ; MIPS32-LABEL: fceil: ; MIPS32: # %bb.0: # %entry ; MIPS32-NEXT: lui $2, %hi(_gp_disp) ; MIPS32-NEXT: addiu $2, $2, %lo(_gp_disp) ; MIPS32-NEXT: addiu $sp, $sp, -24 ; MIPS32-NEXT: .cfi_def_cfa_offset 24 ; MIPS32-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; MIPS32-NEXT: sw $16, 16($sp) # 4-byte Folded Spill ; MIPS32-NEXT: .cfi_offset 31, -4 ; MIPS32-NEXT: .cfi_offset 16, -8 ; MIPS32-NEXT: addu $gp, $2, $25 ; MIPS32-NEXT: lw $16, %got(g)($gp) ; MIPS32-NEXT: lh $1, 0($16) ; MIPS32-NEXT: fill.h $w0, $1 ; MIPS32-NEXT: fexupr.w $w0, $w0 ; MIPS32-NEXT: copy_s.w $1, $w0[0] ; MIPS32-NEXT: lw $25, %call16(ceilf)($gp) ; MIPS32-NEXT: jalr $25 ; MIPS32-NEXT: mtc1 $1, $f12 ; MIPS32-NEXT: mfc1 $1, $f0 ; MIPS32-NEXT: fill.w $w0, $1 ; MIPS32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS32-NEXT: copy_u.h $1, $w0[0] ; MIPS32-NEXT: sh $1, 0($16) ; MIPS32-NEXT: lw $16, 16($sp) # 4-byte Folded Reload ; MIPS32-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: addiu $sp, $sp, 24 ; ; MIPS64-N32-LABEL: fceil: ; MIPS64-N32: # %bb.0: # %entry ; MIPS64-N32-NEXT: addiu $sp, $sp, -32 ; MIPS64-N32-NEXT: .cfi_def_cfa_offset 32 ; MIPS64-N32-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: sd $16, 8($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: .cfi_offset 31, -8 ; MIPS64-N32-NEXT: .cfi_offset 28, -16 ; MIPS64-N32-NEXT: .cfi_offset 16, -24 ; MIPS64-N32-NEXT: lui $1, %hi(%neg(%gp_rel(fceil))) ; MIPS64-N32-NEXT: addu $1, $1, $25 ; MIPS64-N32-NEXT: addiu $gp, $1, %lo(%neg(%gp_rel(fceil))) ; MIPS64-N32-NEXT: lw $16, %got_disp(g)($gp) ; MIPS64-N32-NEXT: lh $1, 0($16) ; MIPS64-N32-NEXT: fill.h $w0, $1 ; MIPS64-N32-NEXT: fexupr.w $w0, $w0 ; MIPS64-N32-NEXT: copy_s.w $1, $w0[0] ; MIPS64-N32-NEXT: lw $25, %call16(ceilf)($gp) ; MIPS64-N32-NEXT: jalr $25 ; MIPS64-N32-NEXT: mtc1 $1, $f12 ; MIPS64-N32-NEXT: mfc1 $1, $f0 ; MIPS64-N32-NEXT: fill.w $w0, $1 ; MIPS64-N32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N32-NEXT: copy_u.h $1, $w0[0] ; MIPS64-N32-NEXT: sh $1, 0($16) ; MIPS64-N32-NEXT: ld $16, 8($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: jr $ra ; MIPS64-N32-NEXT: addiu $sp, $sp, 32 ; ; MIPS64-N64-LABEL: fceil: ; MIPS64-N64: # %bb.0: # %entry ; MIPS64-N64-NEXT: daddiu $sp, $sp, -32 ; MIPS64-N64-NEXT: .cfi_def_cfa_offset 32 ; MIPS64-N64-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: sd $16, 8($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: .cfi_offset 31, -8 ; MIPS64-N64-NEXT: .cfi_offset 28, -16 ; MIPS64-N64-NEXT: .cfi_offset 16, -24 ; MIPS64-N64-NEXT: lui $1, %hi(%neg(%gp_rel(fceil))) ; MIPS64-N64-NEXT: daddu $1, $1, $25 ; MIPS64-N64-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(fceil))) ; MIPS64-N64-NEXT: ld $16, %got_disp(g)($gp) ; MIPS64-N64-NEXT: lh $1, 0($16) ; MIPS64-N64-NEXT: fill.h $w0, $1 ; MIPS64-N64-NEXT: fexupr.w $w0, $w0 ; MIPS64-N64-NEXT: copy_s.w $1, $w0[0] ; MIPS64-N64-NEXT: ld $25, %call16(ceilf)($gp) ; MIPS64-N64-NEXT: jalr $25 ; MIPS64-N64-NEXT: mtc1 $1, $f12 ; MIPS64-N64-NEXT: mfc1 $1, $f0 ; MIPS64-N64-NEXT: fill.w $w0, $1 ; MIPS64-N64-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N64-NEXT: copy_u.h $1, $w0[0] ; MIPS64-N64-NEXT: sh $1, 0($16) ; MIPS64-N64-NEXT: ld $16, 8($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: jr $ra ; MIPS64-N64-NEXT: daddiu $sp, $sp, 32 entry: %0 = load i16, i16* @g, align 2 %1 = call float @llvm.convert.from.fp16.f32(i16 %0) %ceil = call float @llvm.ceil.f32(float %1) %2 = call i16 @llvm.convert.to.fp16.f32(float %ceil) store i16 %2, i16* @g, align 2 ret void } declare float @llvm.trunc.f32(float %Val) define void @ftrunc() { ; MIPS32-LABEL: ftrunc: ; MIPS32: # %bb.0: # %entry ; MIPS32-NEXT: lui $2, %hi(_gp_disp) ; MIPS32-NEXT: addiu $2, $2, %lo(_gp_disp) ; MIPS32-NEXT: addiu $sp, $sp, -24 ; MIPS32-NEXT: .cfi_def_cfa_offset 24 ; MIPS32-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; MIPS32-NEXT: sw $16, 16($sp) # 4-byte Folded Spill ; MIPS32-NEXT: .cfi_offset 31, -4 ; MIPS32-NEXT: .cfi_offset 16, -8 ; MIPS32-NEXT: addu $gp, $2, $25 ; MIPS32-NEXT: lw $16, %got(g)($gp) ; MIPS32-NEXT: lh $1, 0($16) ; MIPS32-NEXT: fill.h $w0, $1 ; MIPS32-NEXT: fexupr.w $w0, $w0 ; MIPS32-NEXT: copy_s.w $1, $w0[0] ; MIPS32-NEXT: lw $25, %call16(truncf)($gp) ; MIPS32-NEXT: jalr $25 ; MIPS32-NEXT: mtc1 $1, $f12 ; MIPS32-NEXT: mfc1 $1, $f0 ; MIPS32-NEXT: fill.w $w0, $1 ; MIPS32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS32-NEXT: copy_u.h $1, $w0[0] ; MIPS32-NEXT: sh $1, 0($16) ; MIPS32-NEXT: lw $16, 16($sp) # 4-byte Folded Reload ; MIPS32-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: addiu $sp, $sp, 24 ; ; MIPS64-N32-LABEL: ftrunc: ; MIPS64-N32: # %bb.0: # %entry ; MIPS64-N32-NEXT: addiu $sp, $sp, -32 ; MIPS64-N32-NEXT: .cfi_def_cfa_offset 32 ; MIPS64-N32-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: sd $16, 8($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: .cfi_offset 31, -8 ; MIPS64-N32-NEXT: .cfi_offset 28, -16 ; MIPS64-N32-NEXT: .cfi_offset 16, -24 ; MIPS64-N32-NEXT: lui $1, %hi(%neg(%gp_rel(ftrunc))) ; MIPS64-N32-NEXT: addu $1, $1, $25 ; MIPS64-N32-NEXT: addiu $gp, $1, %lo(%neg(%gp_rel(ftrunc))) ; MIPS64-N32-NEXT: lw $16, %got_disp(g)($gp) ; MIPS64-N32-NEXT: lh $1, 0($16) ; MIPS64-N32-NEXT: fill.h $w0, $1 ; MIPS64-N32-NEXT: fexupr.w $w0, $w0 ; MIPS64-N32-NEXT: copy_s.w $1, $w0[0] ; MIPS64-N32-NEXT: lw $25, %call16(truncf)($gp) ; MIPS64-N32-NEXT: jalr $25 ; MIPS64-N32-NEXT: mtc1 $1, $f12 ; MIPS64-N32-NEXT: mfc1 $1, $f0 ; MIPS64-N32-NEXT: fill.w $w0, $1 ; MIPS64-N32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N32-NEXT: copy_u.h $1, $w0[0] ; MIPS64-N32-NEXT: sh $1, 0($16) ; MIPS64-N32-NEXT: ld $16, 8($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: jr $ra ; MIPS64-N32-NEXT: addiu $sp, $sp, 32 ; ; MIPS64-N64-LABEL: ftrunc: ; MIPS64-N64: # %bb.0: # %entry ; MIPS64-N64-NEXT: daddiu $sp, $sp, -32 ; MIPS64-N64-NEXT: .cfi_def_cfa_offset 32 ; MIPS64-N64-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: sd $16, 8($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: .cfi_offset 31, -8 ; MIPS64-N64-NEXT: .cfi_offset 28, -16 ; MIPS64-N64-NEXT: .cfi_offset 16, -24 ; MIPS64-N64-NEXT: lui $1, %hi(%neg(%gp_rel(ftrunc))) ; MIPS64-N64-NEXT: daddu $1, $1, $25 ; MIPS64-N64-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(ftrunc))) ; MIPS64-N64-NEXT: ld $16, %got_disp(g)($gp) ; MIPS64-N64-NEXT: lh $1, 0($16) ; MIPS64-N64-NEXT: fill.h $w0, $1 ; MIPS64-N64-NEXT: fexupr.w $w0, $w0 ; MIPS64-N64-NEXT: copy_s.w $1, $w0[0] ; MIPS64-N64-NEXT: ld $25, %call16(truncf)($gp) ; MIPS64-N64-NEXT: jalr $25 ; MIPS64-N64-NEXT: mtc1 $1, $f12 ; MIPS64-N64-NEXT: mfc1 $1, $f0 ; MIPS64-N64-NEXT: fill.w $w0, $1 ; MIPS64-N64-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N64-NEXT: copy_u.h $1, $w0[0] ; MIPS64-N64-NEXT: sh $1, 0($16) ; MIPS64-N64-NEXT: ld $16, 8($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: jr $ra ; MIPS64-N64-NEXT: daddiu $sp, $sp, 32 entry: %0 = load i16, i16* @g, align 2 %1 = call float @llvm.convert.from.fp16.f32(i16 %0) %trunc = call float @llvm.trunc.f32(float %1) %2 = call i16 @llvm.convert.to.fp16.f32(float %trunc) store i16 %2, i16* @g, align 2 ret void } declare float @llvm.rint.f32(float %Val) define void @frint() { ; MIPS32-LABEL: frint: ; MIPS32: # %bb.0: # %entry ; MIPS32-NEXT: lui $2, %hi(_gp_disp) ; MIPS32-NEXT: addiu $2, $2, %lo(_gp_disp) ; MIPS32-NEXT: addiu $sp, $sp, -24 ; MIPS32-NEXT: .cfi_def_cfa_offset 24 ; MIPS32-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; MIPS32-NEXT: sw $16, 16($sp) # 4-byte Folded Spill ; MIPS32-NEXT: .cfi_offset 31, -4 ; MIPS32-NEXT: .cfi_offset 16, -8 ; MIPS32-NEXT: addu $gp, $2, $25 ; MIPS32-NEXT: lw $16, %got(g)($gp) ; MIPS32-NEXT: lh $1, 0($16) ; MIPS32-NEXT: fill.h $w0, $1 ; MIPS32-NEXT: fexupr.w $w0, $w0 ; MIPS32-NEXT: copy_s.w $1, $w0[0] ; MIPS32-NEXT: lw $25, %call16(rintf)($gp) ; MIPS32-NEXT: jalr $25 ; MIPS32-NEXT: mtc1 $1, $f12 ; MIPS32-NEXT: mfc1 $1, $f0 ; MIPS32-NEXT: fill.w $w0, $1 ; MIPS32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS32-NEXT: copy_u.h $1, $w0[0] ; MIPS32-NEXT: sh $1, 0($16) ; MIPS32-NEXT: lw $16, 16($sp) # 4-byte Folded Reload ; MIPS32-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: addiu $sp, $sp, 24 ; ; MIPS64-N32-LABEL: frint: ; MIPS64-N32: # %bb.0: # %entry ; MIPS64-N32-NEXT: addiu $sp, $sp, -32 ; MIPS64-N32-NEXT: .cfi_def_cfa_offset 32 ; MIPS64-N32-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: sd $16, 8($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: .cfi_offset 31, -8 ; MIPS64-N32-NEXT: .cfi_offset 28, -16 ; MIPS64-N32-NEXT: .cfi_offset 16, -24 ; MIPS64-N32-NEXT: lui $1, %hi(%neg(%gp_rel(frint))) ; MIPS64-N32-NEXT: addu $1, $1, $25 ; MIPS64-N32-NEXT: addiu $gp, $1, %lo(%neg(%gp_rel(frint))) ; MIPS64-N32-NEXT: lw $16, %got_disp(g)($gp) ; MIPS64-N32-NEXT: lh $1, 0($16) ; MIPS64-N32-NEXT: fill.h $w0, $1 ; MIPS64-N32-NEXT: fexupr.w $w0, $w0 ; MIPS64-N32-NEXT: copy_s.w $1, $w0[0] ; MIPS64-N32-NEXT: lw $25, %call16(rintf)($gp) ; MIPS64-N32-NEXT: jalr $25 ; MIPS64-N32-NEXT: mtc1 $1, $f12 ; MIPS64-N32-NEXT: mfc1 $1, $f0 ; MIPS64-N32-NEXT: fill.w $w0, $1 ; MIPS64-N32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N32-NEXT: copy_u.h $1, $w0[0] ; MIPS64-N32-NEXT: sh $1, 0($16) ; MIPS64-N32-NEXT: ld $16, 8($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: jr $ra ; MIPS64-N32-NEXT: addiu $sp, $sp, 32 ; ; MIPS64-N64-LABEL: frint: ; MIPS64-N64: # %bb.0: # %entry ; MIPS64-N64-NEXT: daddiu $sp, $sp, -32 ; MIPS64-N64-NEXT: .cfi_def_cfa_offset 32 ; MIPS64-N64-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: sd $16, 8($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: .cfi_offset 31, -8 ; MIPS64-N64-NEXT: .cfi_offset 28, -16 ; MIPS64-N64-NEXT: .cfi_offset 16, -24 ; MIPS64-N64-NEXT: lui $1, %hi(%neg(%gp_rel(frint))) ; MIPS64-N64-NEXT: daddu $1, $1, $25 ; MIPS64-N64-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(frint))) ; MIPS64-N64-NEXT: ld $16, %got_disp(g)($gp) ; MIPS64-N64-NEXT: lh $1, 0($16) ; MIPS64-N64-NEXT: fill.h $w0, $1 ; MIPS64-N64-NEXT: fexupr.w $w0, $w0 ; MIPS64-N64-NEXT: copy_s.w $1, $w0[0] ; MIPS64-N64-NEXT: ld $25, %call16(rintf)($gp) ; MIPS64-N64-NEXT: jalr $25 ; MIPS64-N64-NEXT: mtc1 $1, $f12 ; MIPS64-N64-NEXT: mfc1 $1, $f0 ; MIPS64-N64-NEXT: fill.w $w0, $1 ; MIPS64-N64-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N64-NEXT: copy_u.h $1, $w0[0] ; MIPS64-N64-NEXT: sh $1, 0($16) ; MIPS64-N64-NEXT: ld $16, 8($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: jr $ra ; MIPS64-N64-NEXT: daddiu $sp, $sp, 32 entry: %0 = load i16, i16* @g, align 2 %1 = call float @llvm.convert.from.fp16.f32(i16 %0) %rint = call float @llvm.rint.f32(float %1) %2 = call i16 @llvm.convert.to.fp16.f32(float %rint) store i16 %2, i16* @g, align 2 ret void } declare float @llvm.nearbyint.f32(float %Val) define void @fnearbyint() { ; MIPS32-LABEL: fnearbyint: ; MIPS32: # %bb.0: # %entry ; MIPS32-NEXT: lui $2, %hi(_gp_disp) ; MIPS32-NEXT: addiu $2, $2, %lo(_gp_disp) ; MIPS32-NEXT: addiu $sp, $sp, -24 ; MIPS32-NEXT: .cfi_def_cfa_offset 24 ; MIPS32-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; MIPS32-NEXT: sw $16, 16($sp) # 4-byte Folded Spill ; MIPS32-NEXT: .cfi_offset 31, -4 ; MIPS32-NEXT: .cfi_offset 16, -8 ; MIPS32-NEXT: addu $gp, $2, $25 ; MIPS32-NEXT: lw $16, %got(g)($gp) ; MIPS32-NEXT: lh $1, 0($16) ; MIPS32-NEXT: fill.h $w0, $1 ; MIPS32-NEXT: fexupr.w $w0, $w0 ; MIPS32-NEXT: copy_s.w $1, $w0[0] ; MIPS32-NEXT: lw $25, %call16(nearbyintf)($gp) ; MIPS32-NEXT: jalr $25 ; MIPS32-NEXT: mtc1 $1, $f12 ; MIPS32-NEXT: mfc1 $1, $f0 ; MIPS32-NEXT: fill.w $w0, $1 ; MIPS32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS32-NEXT: copy_u.h $1, $w0[0] ; MIPS32-NEXT: sh $1, 0($16) ; MIPS32-NEXT: lw $16, 16($sp) # 4-byte Folded Reload ; MIPS32-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: addiu $sp, $sp, 24 ; ; MIPS64-N32-LABEL: fnearbyint: ; MIPS64-N32: # %bb.0: # %entry ; MIPS64-N32-NEXT: addiu $sp, $sp, -32 ; MIPS64-N32-NEXT: .cfi_def_cfa_offset 32 ; MIPS64-N32-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: sd $16, 8($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: .cfi_offset 31, -8 ; MIPS64-N32-NEXT: .cfi_offset 28, -16 ; MIPS64-N32-NEXT: .cfi_offset 16, -24 ; MIPS64-N32-NEXT: lui $1, %hi(%neg(%gp_rel(fnearbyint))) ; MIPS64-N32-NEXT: addu $1, $1, $25 ; MIPS64-N32-NEXT: addiu $gp, $1, %lo(%neg(%gp_rel(fnearbyint))) ; MIPS64-N32-NEXT: lw $16, %got_disp(g)($gp) ; MIPS64-N32-NEXT: lh $1, 0($16) ; MIPS64-N32-NEXT: fill.h $w0, $1 ; MIPS64-N32-NEXT: fexupr.w $w0, $w0 ; MIPS64-N32-NEXT: copy_s.w $1, $w0[0] ; MIPS64-N32-NEXT: lw $25, %call16(nearbyintf)($gp) ; MIPS64-N32-NEXT: jalr $25 ; MIPS64-N32-NEXT: mtc1 $1, $f12 ; MIPS64-N32-NEXT: mfc1 $1, $f0 ; MIPS64-N32-NEXT: fill.w $w0, $1 ; MIPS64-N32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N32-NEXT: copy_u.h $1, $w0[0] ; MIPS64-N32-NEXT: sh $1, 0($16) ; MIPS64-N32-NEXT: ld $16, 8($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: jr $ra ; MIPS64-N32-NEXT: addiu $sp, $sp, 32 ; ; MIPS64-N64-LABEL: fnearbyint: ; MIPS64-N64: # %bb.0: # %entry ; MIPS64-N64-NEXT: daddiu $sp, $sp, -32 ; MIPS64-N64-NEXT: .cfi_def_cfa_offset 32 ; MIPS64-N64-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: sd $16, 8($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: .cfi_offset 31, -8 ; MIPS64-N64-NEXT: .cfi_offset 28, -16 ; MIPS64-N64-NEXT: .cfi_offset 16, -24 ; MIPS64-N64-NEXT: lui $1, %hi(%neg(%gp_rel(fnearbyint))) ; MIPS64-N64-NEXT: daddu $1, $1, $25 ; MIPS64-N64-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(fnearbyint))) ; MIPS64-N64-NEXT: ld $16, %got_disp(g)($gp) ; MIPS64-N64-NEXT: lh $1, 0($16) ; MIPS64-N64-NEXT: fill.h $w0, $1 ; MIPS64-N64-NEXT: fexupr.w $w0, $w0 ; MIPS64-N64-NEXT: copy_s.w $1, $w0[0] ; MIPS64-N64-NEXT: ld $25, %call16(nearbyintf)($gp) ; MIPS64-N64-NEXT: jalr $25 ; MIPS64-N64-NEXT: mtc1 $1, $f12 ; MIPS64-N64-NEXT: mfc1 $1, $f0 ; MIPS64-N64-NEXT: fill.w $w0, $1 ; MIPS64-N64-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N64-NEXT: copy_u.h $1, $w0[0] ; MIPS64-N64-NEXT: sh $1, 0($16) ; MIPS64-N64-NEXT: ld $16, 8($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: jr $ra ; MIPS64-N64-NEXT: daddiu $sp, $sp, 32 entry: %0 = load i16, i16* @g, align 2 %1 = call float @llvm.convert.from.fp16.f32(i16 %0) %nearbyint = call float @llvm.nearbyint.f32(float %1) %2 = call i16 @llvm.convert.to.fp16.f32(float %nearbyint) store i16 %2, i16* @g, align 2 ret void } declare float @llvm.round.f32(float %Val) define void @fround() { ; MIPS32-LABEL: fround: ; MIPS32: # %bb.0: # %entry ; MIPS32-NEXT: lui $2, %hi(_gp_disp) ; MIPS32-NEXT: addiu $2, $2, %lo(_gp_disp) ; MIPS32-NEXT: addiu $sp, $sp, -24 ; MIPS32-NEXT: .cfi_def_cfa_offset 24 ; MIPS32-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; MIPS32-NEXT: sw $16, 16($sp) # 4-byte Folded Spill ; MIPS32-NEXT: .cfi_offset 31, -4 ; MIPS32-NEXT: .cfi_offset 16, -8 ; MIPS32-NEXT: addu $gp, $2, $25 ; MIPS32-NEXT: lw $16, %got(g)($gp) ; MIPS32-NEXT: lh $1, 0($16) ; MIPS32-NEXT: fill.h $w0, $1 ; MIPS32-NEXT: fexupr.w $w0, $w0 ; MIPS32-NEXT: copy_s.w $1, $w0[0] ; MIPS32-NEXT: lw $25, %call16(roundf)($gp) ; MIPS32-NEXT: jalr $25 ; MIPS32-NEXT: mtc1 $1, $f12 ; MIPS32-NEXT: mfc1 $1, $f0 ; MIPS32-NEXT: fill.w $w0, $1 ; MIPS32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS32-NEXT: copy_u.h $1, $w0[0] ; MIPS32-NEXT: sh $1, 0($16) ; MIPS32-NEXT: lw $16, 16($sp) # 4-byte Folded Reload ; MIPS32-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; MIPS32-NEXT: jr $ra ; MIPS32-NEXT: addiu $sp, $sp, 24 ; ; MIPS64-N32-LABEL: fround: ; MIPS64-N32: # %bb.0: # %entry ; MIPS64-N32-NEXT: addiu $sp, $sp, -32 ; MIPS64-N32-NEXT: .cfi_def_cfa_offset 32 ; MIPS64-N32-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: sd $16, 8($sp) # 8-byte Folded Spill ; MIPS64-N32-NEXT: .cfi_offset 31, -8 ; MIPS64-N32-NEXT: .cfi_offset 28, -16 ; MIPS64-N32-NEXT: .cfi_offset 16, -24 ; MIPS64-N32-NEXT: lui $1, %hi(%neg(%gp_rel(fround))) ; MIPS64-N32-NEXT: addu $1, $1, $25 ; MIPS64-N32-NEXT: addiu $gp, $1, %lo(%neg(%gp_rel(fround))) ; MIPS64-N32-NEXT: lw $16, %got_disp(g)($gp) ; MIPS64-N32-NEXT: lh $1, 0($16) ; MIPS64-N32-NEXT: fill.h $w0, $1 ; MIPS64-N32-NEXT: fexupr.w $w0, $w0 ; MIPS64-N32-NEXT: copy_s.w $1, $w0[0] ; MIPS64-N32-NEXT: lw $25, %call16(roundf)($gp) ; MIPS64-N32-NEXT: jalr $25 ; MIPS64-N32-NEXT: mtc1 $1, $f12 ; MIPS64-N32-NEXT: mfc1 $1, $f0 ; MIPS64-N32-NEXT: fill.w $w0, $1 ; MIPS64-N32-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N32-NEXT: copy_u.h $1, $w0[0] ; MIPS64-N32-NEXT: sh $1, 0($16) ; MIPS64-N32-NEXT: ld $16, 8($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64-N32-NEXT: jr $ra ; MIPS64-N32-NEXT: addiu $sp, $sp, 32 ; ; MIPS64-N64-LABEL: fround: ; MIPS64-N64: # %bb.0: # %entry ; MIPS64-N64-NEXT: daddiu $sp, $sp, -32 ; MIPS64-N64-NEXT: .cfi_def_cfa_offset 32 ; MIPS64-N64-NEXT: sd $ra, 24($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: sd $gp, 16($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: sd $16, 8($sp) # 8-byte Folded Spill ; MIPS64-N64-NEXT: .cfi_offset 31, -8 ; MIPS64-N64-NEXT: .cfi_offset 28, -16 ; MIPS64-N64-NEXT: .cfi_offset 16, -24 ; MIPS64-N64-NEXT: lui $1, %hi(%neg(%gp_rel(fround))) ; MIPS64-N64-NEXT: daddu $1, $1, $25 ; MIPS64-N64-NEXT: daddiu $gp, $1, %lo(%neg(%gp_rel(fround))) ; MIPS64-N64-NEXT: ld $16, %got_disp(g)($gp) ; MIPS64-N64-NEXT: lh $1, 0($16) ; MIPS64-N64-NEXT: fill.h $w0, $1 ; MIPS64-N64-NEXT: fexupr.w $w0, $w0 ; MIPS64-N64-NEXT: copy_s.w $1, $w0[0] ; MIPS64-N64-NEXT: ld $25, %call16(roundf)($gp) ; MIPS64-N64-NEXT: jalr $25 ; MIPS64-N64-NEXT: mtc1 $1, $f12 ; MIPS64-N64-NEXT: mfc1 $1, $f0 ; MIPS64-N64-NEXT: fill.w $w0, $1 ; MIPS64-N64-NEXT: fexdo.h $w0, $w0, $w0 ; MIPS64-N64-NEXT: copy_u.h $1, $w0[0] ; MIPS64-N64-NEXT: sh $1, 0($16) ; MIPS64-N64-NEXT: ld $16, 8($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: ld $gp, 16($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: ld $ra, 24($sp) # 8-byte Folded Reload ; MIPS64-N64-NEXT: jr $ra ; MIPS64-N64-NEXT: daddiu $sp, $sp, 32 entry: %0 = load i16, i16* @g, align 2 %1 = call float @llvm.convert.from.fp16.f32(i16 %0) %round = call float @llvm.round.f32(float %1) %2 = call i16 @llvm.convert.to.fp16.f32(float %round) store i16 %2, i16* @g, align 2 ret void } Index: vendor/llvm/dist-release_80/test/CodeGen/Mips/o32_cc_byval.ll =================================================================== --- vendor/llvm/dist-release_80/test/CodeGen/Mips/o32_cc_byval.ll (revision 343793) +++ vendor/llvm/dist-release_80/test/CodeGen/Mips/o32_cc_byval.ll (revision 343794) @@ -1,258 +1,259 @@ ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py -; RUN: llc -mtriple=mipsel-unknown-linux-gnu -relocation-model=pic < %s | FileCheck %s +; RUN: llc -mtriple=mipsel-unknown-linux-gnu -relocation-model=pic \ +; RUN: -mips-jalr-reloc=false < %s | FileCheck %s %0 = type { i8, i16, i32, i64, double, i32, [4 x i8] } %struct.S1 = type { i8, i16, i32, i64, double, i32 } %struct.S2 = type { [4 x i32] } %struct.S3 = type { i8 } @f1.s1 = internal unnamed_addr constant %0 { i8 1, i16 2, i32 3, i64 4, double 5.000000e+00, i32 6, [4 x i8] undef }, align 8 @f1.s2 = internal unnamed_addr constant %struct.S2 { [4 x i32] [i32 7, i32 8, i32 9, i32 10] }, align 4 define void @f1() nounwind { ; CHECK-LABEL: f1: ; CHECK: # %bb.0: # %entry ; CHECK-NEXT: lui $2, %hi(_gp_disp) ; CHECK-NEXT: addiu $2, $2, %lo(_gp_disp) ; CHECK-NEXT: addiu $sp, $sp, -64 ; CHECK-NEXT: sw $ra, 60($sp) # 4-byte Folded Spill ; CHECK-NEXT: sw $18, 56($sp) # 4-byte Folded Spill ; CHECK-NEXT: sw $17, 52($sp) # 4-byte Folded Spill ; CHECK-NEXT: sw $16, 48($sp) # 4-byte Folded Spill ; CHECK-NEXT: addu $16, $2, $25 ; CHECK-NEXT: lw $17, %got(f1.s1)($16) ; CHECK-NEXT: addiu $18, $17, %lo(f1.s1) ; CHECK-NEXT: lw $1, 12($18) ; CHECK-NEXT: lw $2, 16($18) ; CHECK-NEXT: lw $3, 20($18) ; CHECK-NEXT: lw $4, 24($18) ; CHECK-NEXT: lw $5, 28($18) ; CHECK-NEXT: sw $5, 36($sp) ; CHECK-NEXT: sw $4, 32($sp) ; CHECK-NEXT: sw $3, 28($sp) ; CHECK-NEXT: sw $2, 24($sp) ; CHECK-NEXT: sw $1, 20($sp) ; CHECK-NEXT: lw $1, 8($18) ; CHECK-NEXT: sw $1, 16($sp) ; CHECK-NEXT: lw $6, %lo(f1.s1)($17) ; CHECK-NEXT: lw $7, 4($18) ; CHECK-NEXT: lw $1, %got($CPI0_0)($16) ; CHECK-NEXT: lwc1 $f12, %lo($CPI0_0)($1) ; CHECK-NEXT: lw $25, %call16(callee1)($16) ; CHECK-NEXT: jalr $25 ; CHECK-NEXT: move $gp, $16 ; CHECK-NEXT: lw $1, %got(f1.s2)($16) ; CHECK-NEXT: addiu $2, $1, %lo(f1.s2) ; CHECK-NEXT: lw $7, 12($2) ; CHECK-NEXT: lw $6, 8($2) ; CHECK-NEXT: lw $5, 4($2) ; CHECK-NEXT: lw $4, %lo(f1.s2)($1) ; CHECK-NEXT: lw $25, %call16(callee2)($16) ; CHECK-NEXT: jalr $25 ; CHECK-NEXT: move $gp, $16 ; CHECK-NEXT: addiu $1, $zero, 11 ; CHECK-NEXT: lw $2, %got($CPI0_1)($16) ; CHECK-NEXT: lwc1 $f12, %lo($CPI0_1)($2) ; CHECK-NEXT: sb $1, 40($sp) ; CHECK-NEXT: lw $1, 16($18) ; CHECK-NEXT: lw $2, 20($18) ; CHECK-NEXT: lw $3, 24($18) ; CHECK-NEXT: lw $4, 28($18) ; CHECK-NEXT: sw $4, 36($sp) ; CHECK-NEXT: sw $3, 32($sp) ; CHECK-NEXT: sw $2, 28($sp) ; CHECK-NEXT: sw $1, 24($sp) ; CHECK-NEXT: lw $1, 12($18) ; CHECK-NEXT: sw $1, 20($sp) ; CHECK-NEXT: lw $1, 8($18) ; CHECK-NEXT: sw $1, 16($sp) ; CHECK-NEXT: lw $7, 4($18) ; CHECK-NEXT: lw $6, %lo(f1.s1)($17) ; CHECK-NEXT: lbu $5, 40($sp) ; CHECK-NEXT: lw $25, %call16(callee3)($16) ; CHECK-NEXT: jalr $25 ; CHECK-NEXT: move $gp, $16 ; CHECK-NEXT: lw $16, 48($sp) # 4-byte Folded Reload ; CHECK-NEXT: lw $17, 52($sp) # 4-byte Folded Reload ; CHECK-NEXT: lw $18, 56($sp) # 4-byte Folded Reload ; CHECK-NEXT: lw $ra, 60($sp) # 4-byte Folded Reload ; CHECK-NEXT: jr $ra ; CHECK-NEXT: addiu $sp, $sp, 64 entry: %agg.tmp10 = alloca %struct.S3, align 4 call void @callee1(float 2.000000e+01, %struct.S1* byval bitcast (%0* @f1.s1 to %struct.S1*)) nounwind call void @callee2(%struct.S2* byval @f1.s2) nounwind %tmp11 = getelementptr inbounds %struct.S3, %struct.S3* %agg.tmp10, i32 0, i32 0 store i8 11, i8* %tmp11, align 4 call void @callee3(float 2.100000e+01, %struct.S3* byval %agg.tmp10, %struct.S1* byval bitcast (%0* @f1.s1 to %struct.S1*)) nounwind ret void } declare void @callee1(float, %struct.S1* byval) declare void @callee2(%struct.S2* byval) declare void @callee3(float, %struct.S3* byval, %struct.S1* byval) define void @f2(float %f, %struct.S1* nocapture byval %s1) nounwind { ; CHECK-LABEL: f2: ; CHECK: # %bb.0: # %entry ; CHECK-NEXT: lui $2, %hi(_gp_disp) ; CHECK-NEXT: addiu $2, $2, %lo(_gp_disp) ; CHECK-NEXT: addiu $sp, $sp, -48 ; CHECK-NEXT: sw $ra, 44($sp) # 4-byte Folded Spill ; CHECK-NEXT: addu $gp, $2, $25 ; CHECK-NEXT: sw $6, 56($sp) ; CHECK-NEXT: sw $7, 60($sp) ; CHECK-NEXT: lw $4, 80($sp) ; CHECK-NEXT: ldc1 $f0, 72($sp) ; CHECK-NEXT: lw $1, 64($sp) ; CHECK-NEXT: lw $2, 68($sp) ; CHECK-NEXT: lh $3, 58($sp) ; CHECK-NEXT: sll $5, $6, 24 ; CHECK-NEXT: sra $5, $5, 24 ; CHECK-NEXT: swc1 $f12, 36($sp) ; CHECK-NEXT: sw $5, 32($sp) ; CHECK-NEXT: sw $3, 28($sp) ; CHECK-NEXT: sw $2, 20($sp) ; CHECK-NEXT: sw $1, 16($sp) ; CHECK-NEXT: sw $7, 24($sp) ; CHECK-NEXT: mfc1 $6, $f0 ; CHECK-NEXT: lw $25, %call16(callee4)($gp) ; CHECK-NEXT: jalr $25 ; CHECK-NEXT: mfc1 $7, $f1 ; CHECK-NEXT: lw $ra, 44($sp) # 4-byte Folded Reload ; CHECK-NEXT: jr $ra ; CHECK-NEXT: addiu $sp, $sp, 48 entry: %i2 = getelementptr inbounds %struct.S1, %struct.S1* %s1, i32 0, i32 5 %tmp = load i32, i32* %i2, align 4 %d = getelementptr inbounds %struct.S1, %struct.S1* %s1, i32 0, i32 4 %tmp1 = load double, double* %d, align 8 %ll = getelementptr inbounds %struct.S1, %struct.S1* %s1, i32 0, i32 3 %tmp2 = load i64, i64* %ll, align 8 %i = getelementptr inbounds %struct.S1, %struct.S1* %s1, i32 0, i32 2 %tmp3 = load i32, i32* %i, align 4 %s = getelementptr inbounds %struct.S1, %struct.S1* %s1, i32 0, i32 1 %tmp4 = load i16, i16* %s, align 2 %c = getelementptr inbounds %struct.S1, %struct.S1* %s1, i32 0, i32 0 %tmp5 = load i8, i8* %c, align 1 tail call void @callee4(i32 %tmp, double %tmp1, i64 %tmp2, i32 %tmp3, i16 signext %tmp4, i8 signext %tmp5, float %f) nounwind ret void } declare void @callee4(i32, double, i64, i32, i16 signext, i8 signext, float) define void @f3(%struct.S2* nocapture byval %s2) nounwind { ; CHECK-LABEL: f3: ; CHECK: # %bb.0: # %entry ; CHECK-NEXT: lui $2, %hi(_gp_disp) ; CHECK-NEXT: addiu $2, $2, %lo(_gp_disp) ; CHECK-NEXT: addiu $sp, $sp, -48 ; CHECK-NEXT: sw $ra, 44($sp) # 4-byte Folded Spill ; CHECK-NEXT: addu $gp, $2, $25 ; CHECK-NEXT: sw $7, 60($sp) ; CHECK-NEXT: sw $6, 56($sp) ; CHECK-NEXT: sw $5, 52($sp) ; CHECK-NEXT: sw $4, 48($sp) ; CHECK-NEXT: addiu $1, $zero, 3 ; CHECK-NEXT: addiu $2, $zero, 4 ; CHECK-NEXT: addiu $3, $zero, 5 ; CHECK-NEXT: lui $5, 16576 ; CHECK-NEXT: sw $5, 36($sp) ; CHECK-NEXT: sw $3, 32($sp) ; CHECK-NEXT: sw $2, 28($sp) ; CHECK-NEXT: sw $1, 16($sp) ; CHECK-NEXT: sw $7, 24($sp) ; CHECK-NEXT: sw $zero, 20($sp) ; CHECK-NEXT: lw $1, %got($CPI2_0)($gp) ; CHECK-NEXT: ldc1 $f0, %lo($CPI2_0)($1) ; CHECK-NEXT: mfc1 $6, $f0 ; CHECK-NEXT: lw $25, %call16(callee4)($gp) ; CHECK-NEXT: jalr $25 ; CHECK-NEXT: mfc1 $7, $f1 ; CHECK-NEXT: lw $ra, 44($sp) # 4-byte Folded Reload ; CHECK-NEXT: jr $ra ; CHECK-NEXT: addiu $sp, $sp, 48 entry: %arrayidx = getelementptr inbounds %struct.S2, %struct.S2* %s2, i32 0, i32 0, i32 0 %tmp = load i32, i32* %arrayidx, align 4 %arrayidx2 = getelementptr inbounds %struct.S2, %struct.S2* %s2, i32 0, i32 0, i32 3 %tmp3 = load i32, i32* %arrayidx2, align 4 tail call void @callee4(i32 %tmp, double 2.000000e+00, i64 3, i32 %tmp3, i16 signext 4, i8 signext 5, float 6.000000e+00) nounwind ret void } define void @f4(float %f, %struct.S3* nocapture byval %s3, %struct.S1* nocapture byval %s1) nounwind { ; CHECK-LABEL: f4: ; CHECK: # %bb.0: # %entry ; CHECK-NEXT: lui $2, %hi(_gp_disp) ; CHECK-NEXT: addiu $2, $2, %lo(_gp_disp) ; CHECK-NEXT: addiu $sp, $sp, -48 ; CHECK-NEXT: sw $ra, 44($sp) # 4-byte Folded Spill ; CHECK-NEXT: addu $gp, $2, $25 ; CHECK-NEXT: move $4, $7 ; CHECK-NEXT: sw $6, 56($sp) ; CHECK-NEXT: sw $5, 52($sp) ; CHECK-NEXT: sw $7, 60($sp) ; CHECK-NEXT: lw $1, 80($sp) ; CHECK-NEXT: sll $2, $5, 24 ; CHECK-NEXT: sra $2, $2, 24 ; CHECK-NEXT: addiu $3, $zero, 4 ; CHECK-NEXT: lui $5, 16576 ; CHECK-NEXT: sw $5, 36($sp) ; CHECK-NEXT: sw $2, 32($sp) ; CHECK-NEXT: sw $3, 28($sp) ; CHECK-NEXT: sw $1, 24($sp) ; CHECK-NEXT: addiu $1, $zero, 3 ; CHECK-NEXT: sw $1, 16($sp) ; CHECK-NEXT: sw $zero, 20($sp) ; CHECK-NEXT: lw $1, %got($CPI3_0)($gp) ; CHECK-NEXT: ldc1 $f0, %lo($CPI3_0)($1) ; CHECK-NEXT: mfc1 $6, $f0 ; CHECK-NEXT: lw $25, %call16(callee4)($gp) ; CHECK-NEXT: jalr $25 ; CHECK-NEXT: mfc1 $7, $f1 ; CHECK-NEXT: lw $ra, 44($sp) # 4-byte Folded Reload ; CHECK-NEXT: jr $ra ; CHECK-NEXT: addiu $sp, $sp, 48 entry: %i = getelementptr inbounds %struct.S1, %struct.S1* %s1, i32 0, i32 2 %tmp = load i32, i32* %i, align 4 %i2 = getelementptr inbounds %struct.S1, %struct.S1* %s1, i32 0, i32 5 %tmp1 = load i32, i32* %i2, align 4 %c = getelementptr inbounds %struct.S3, %struct.S3* %s3, i32 0, i32 0 %tmp2 = load i8, i8* %c, align 1 tail call void @callee4(i32 %tmp, double 2.000000e+00, i64 3, i32 %tmp1, i16 signext 4, i8 signext %tmp2, float 6.000000e+00) nounwind ret void } %struct.S4 = type { [4 x i32] } define void @f5(i64 %a0, %struct.S4* nocapture byval %a1) nounwind { ; CHECK-LABEL: f5: ; CHECK: # %bb.0: # %entry ; CHECK-NEXT: lui $2, %hi(_gp_disp) ; CHECK-NEXT: addiu $2, $2, %lo(_gp_disp) ; CHECK-NEXT: addiu $sp, $sp, -32 ; CHECK-NEXT: sw $ra, 28($sp) # 4-byte Folded Spill ; CHECK-NEXT: addu $gp, $2, $25 ; CHECK-NEXT: sw $7, 44($sp) ; CHECK-NEXT: sw $6, 40($sp) ; CHECK-NEXT: sw $5, 20($sp) ; CHECK-NEXT: sw $4, 16($sp) ; CHECK-NEXT: lw $7, 52($sp) ; CHECK-NEXT: lw $6, 48($sp) ; CHECK-NEXT: lw $5, 44($sp) ; CHECK-NEXT: lw $25, %call16(f6)($gp) ; CHECK-NEXT: jalr $25 ; CHECK-NEXT: lw $4, 40($sp) ; CHECK-NEXT: lw $ra, 28($sp) # 4-byte Folded Reload ; CHECK-NEXT: jr $ra ; CHECK-NEXT: addiu $sp, $sp, 32 entry: tail call void @f6(%struct.S4* byval %a1, i64 %a0) nounwind ret void } declare void @f6(%struct.S4* nocapture byval, i64) Index: vendor/llvm/dist-release_80/test/CodeGen/Mips/reloc-jalr.ll =================================================================== --- vendor/llvm/dist-release_80/test/CodeGen/Mips/reloc-jalr.ll (nonexistent) +++ vendor/llvm/dist-release_80/test/CodeGen/Mips/reloc-jalr.ll (revision 343794) @@ -0,0 +1,154 @@ +; RUN: llc -mtriple=mips-linux-gnu -relocation-model=pic -mips-tail-calls=1 \ +; RUN: -O2 < %s | \ +; RUN: FileCheck %s -check-prefixes=ALL,JALR-32R2,TAILCALL-32R2 + +; RUN: llc -mtriple=mips64-linux-gnu -relocation-model=pic -mips-tail-calls=1 \ +; RUN: -O2 < %s | \ +; RUN: FileCheck %s -check-prefixes=ALL,JALR-64R2,TAILCALL-64R2 + +; RUN: llc -mtriple=mips-linux-gnu -relocation-model=pic -mips-tail-calls=1 \ +; RUN: -O2 -mcpu=mips32r6 -mips-compact-branches=always < %s | \ +; RUN: FileCheck %s -check-prefixes=ALL,JALR-32R6,TAILCALL-32R6 + +; RUN: llc -mtriple=mips64-linux-gnu -relocation-model=pic -mips-tail-calls=1 \ +; RUN: -O2 -mcpu=mips64r6 -mips-compact-branches=always < %s | \ +; RUN: FileCheck %s -check-prefixes=ALL,JALR-64R6,TAILCALL-64R6 + +; RUN: llc -mtriple=mips-linux-gnu -relocation-model=pic -mips-tail-calls=1 \ +; RUN: -O2 -mcpu=mips32r6 -mips-compact-branches=never < %s | \ +; RUN: FileCheck %s -check-prefixes=ALL,JALR-32R2,TAILCALL-32R2 + +; RUN: llc -mtriple=mips64-linux-gnu -relocation-model=pic -mips-tail-calls=1 \ +; RUN: -O2 -mcpu=mips64r6 -mips-compact-branches=never < %s | \ +; RUN: FileCheck %s -check-prefixes=ALL,JALR-64R2,TAILCALL-64R2 + +; RUN: llc -mtriple=mips-linux-gnu -relocation-model=pic -mips-tail-calls=1 \ +; RUN: -O2 -mattr=+micromips -mcpu=mips32r2 < %s | \ +; RUN: FileCheck %s -check-prefixes=ALL,JALR-MM,TAILCALL-MM + +; RUN: llc -mtriple=mips-linux-gnu -relocation-model=pic -mips-tail-calls=1 \ +; RUN: -O2 -mattr=+micromips -mcpu=mips32r6 < %s | \ +; RUN: FileCheck %s -check-prefixes=ALL,JALR-MM + +; RUN: llc -mtriple=mips-linux-gnu -relocation-model=pic \ +; RUN: -O0 < %s | FileCheck %s -check-prefixes=ALL,JALR-32R2 + +; RUN: llc -mtriple=mips64-linux-gnu -relocation-model=pic \ +; RUN: -O0 < %s | FileCheck %s -check-prefixes=ALL,JALR-64R2 + +; RUN: llc -mtriple=mips-linux-gnu -relocation-model=pic \ +; RUN: -O0 -mcpu=mips32r6 -mips-compact-branches=always < %s | \ +; RUN: FileCheck %s -check-prefixes=ALL,JALR-32R6 + +; RUN: llc -mtriple=mips64-linux-gnu -relocation-model=pic \ +; RUN: -O0 -mcpu=mips64r6 -mips-compact-branches=always < %s | \ +; RUN: FileCheck %s -check-prefixes=ALL,JALR-64R6 + +; RUN: llc -mtriple=mips-linux-gnu -relocation-model=pic \ +; RUN: -O0 -mcpu=mips32r6 -mips-compact-branches=never < %s | \ +; RUN: FileCheck %s -check-prefixes=ALL,JALR-32R2 + +; RUN: llc -mtriple=mips64-linux-gnu -relocation-model=pic \ +; RUN: -O0 -mcpu=mips64r6 -mips-compact-branches=never < %s | \ +; RUN: FileCheck %s -check-prefixes=ALL,JALR-64R2 + +; RUN: llc -mtriple=mips-linux-gnu -relocation-model=pic \ +; RUN: -O0 -mattr=+micromips -mcpu=mips32r2 < %s | \ +; RUN: FileCheck %s -check-prefixes=ALL,JALR-MM + +; RUN: llc -mtriple=mips-linux-gnu -relocation-model=pic \ +; RUN: -O0 -mattr=+micromips -mcpu=mips32r6 < %s | \ +; RUN: FileCheck %s -check-prefixes=ALL,JALR-MM + +; RUN: llc -mtriple=mips-linux-gnu -relocation-model=pic -mips-tail-calls=1 \ +; RUN: -O2 -mips-jalr-reloc=false < %s | \ +; RUN: FileCheck %s -check-prefixes=ALL,NORELOC + +; RUN: llc -mtriple=mips-linux-gnu -relocation-model=static -mips-tail-calls=1 \ +; RUN: -O2 < %s | \ +; RUN: FileCheck %s -check-prefixes=ALL,NORELOC + +; RUN: llc -mtriple=mips-linux-gnu -relocation-model=pic -mips-tail-calls=1 \ +; RUN: -O0 -mips-jalr-reloc=false < %s | \ +; RUN: FileCheck %s -check-prefixes=ALL,NORELOC + +; RUN: llc -mtriple=mips-linux-gnu -relocation-model=static -mips-tail-calls=1 \ +; RUN: -O0 < %s | \ +; RUN: FileCheck %s -check-prefixes=ALL,NORELOC + +; RUN: llc -mtriple=mips64-linux-gnu -relocation-model=pic -mips-tail-calls=1 \ +; RUN: -O2 -mips-jalr-reloc=false < %s | \ +; RUN: FileCheck %s -check-prefixes=ALL,NORELOC + +; RUN: llc -mtriple=mips64-linux-gnu -mips-tail-calls=1 \ +; RUN: -O2 -relocation-model=static < %s | \ +; RUN: FileCheck %s -check-prefixes=ALL,NORELOC + +; RUN: llc -mtriple=mips64-linux-gnu -relocation-model=pic \ +; RUN: -O0 -mips-jalr-reloc=false < %s | \ +; RUN: FileCheck %s -check-prefixes=ALL,NORELOC + +; RUN: llc -mtriple=mips64-linux-gnu -relocation-model=static \ +; RUN: -O0 < %s | \ +; RUN: FileCheck %s -check-prefixes=ALL,NORELOC + +define internal void @foo() noinline { +entry: + ret void +} + +define void @checkCall() { +entry: +; ALL-LABEL: checkCall: + call void @foo() +; JALR-32R2: .reloc ([[TMPLABEL:.*]]), R_MIPS_JALR, foo +; JALR-32R2-NEXT: [[TMPLABEL]]: +; JALR-32R2-NEXT: jalr $25 + +; JALR-64R2: .reloc [[TMPLABEL:.*]], R_MIPS_JALR, foo +; JALR-64R2-NEXT: [[TMPLABEL]]: +; JALR-64R2-NEXT: jalr $25 + +; JALR-MM: .reloc ([[TMPLABEL:.*]]), R_MICROMIPS_JALR, foo +; JALR-MM-NEXT: [[TMPLABEL]]: +; JALR-MM-NEXT: jalr $25 + +; JALR-32R6: .reloc ([[TMPLABEL:.*]]), R_MIPS_JALR, foo +; JALR-32R6-NEXT: [[TMPLABEL]]: +; JALR-32R6-NEXT: jalrc $25 + +; JALR-64R6: .reloc [[TMPLABEL:.*]], R_MIPS_JALR, foo +; JALR-64R6-NEXT: [[TMPLABEL]]: +; JALR-64R6-NEXT: jalrc $25 + +; NORELOC-NOT: R_MIPS_JALR + ret void +} + +define void @checkTailCall() { +entry: +; ALL-LABEL: checkTailCall: + tail call void @foo() +; TAILCALL-32R2: .reloc ([[TMPLABEL:.*]]), R_MIPS_JALR, foo +; TAILCALL-32R2-NEXT: [[TMPLABEL]]: +; TAILCALL-32R2-NEXT: jr $25 + +; TAILCALL-64R2: .reloc [[TMPLABEL:.*]], R_MIPS_JALR, foo +; TAILCALL-64R2-NEXT: [[TMPLABEL]]: +; TAILCALL-64R2-NEXT: jr $25 + +; TAILCALL-MM: .reloc ([[TMPLABEL:.*]]), R_MICROMIPS_JALR, foo +; TAILCALL-MM-NEXT: [[TMPLABEL]]: +; TAILCALL-MM-NEXT: jrc $25 + +; TAILCALL-32R6: .reloc ([[TMPLABEL:.*]]), R_MIPS_JALR, foo +; TAILCALL-32R6-NEXT: [[TMPLABEL]]: +; TAILCALL-32R6-NEXT: jrc $25 + +; TAILCALL-64R6: .reloc [[TMPLABEL:.*]], R_MIPS_JALR, foo +; TAILCALL-64R6-NEXT: [[TMPLABEL]]: +; TAILCALL-64R6-NEXT: jrc $25 + +; NORELOC-NOT: R_MIPS_JALR + ret void +} Index: vendor/llvm/dist-release_80/test/CodeGen/Mips/shrink-wrapping.ll =================================================================== --- vendor/llvm/dist-release_80/test/CodeGen/Mips/shrink-wrapping.ll (revision 343793) +++ vendor/llvm/dist-release_80/test/CodeGen/Mips/shrink-wrapping.ll (revision 343794) @@ -1,387 +1,387 @@ ; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py ; RUN: llc -mtriple=mips-unknown-linux-gnu -enable-shrink-wrap=true \ ; RUN: -relocation-model=static < %s | \ ; RUN: FileCheck %s -check-prefix=SHRINK-WRAP-STATIC ; RUN: llc -mtriple=mips-unknown-linux-gnu -enable-shrink-wrap=false \ ; RUN: -relocation-model=static < %s | \ ; RUN: FileCheck %s -check-prefix=NO-SHRINK-WRAP-STATIC ; RUN: llc -mtriple=mips-unknown-linux-gnu -enable-shrink-wrap=true \ -; RUN: -relocation-model=pic < %s | \ +; RUN: -relocation-model=pic -mips-jalr-reloc=false < %s | \ ; RUN: FileCheck %s -check-prefix=SHRINK-WRAP-PIC ; RUN: llc -mtriple=mips-unknown-linux-gnu -enable-shrink-wrap=false \ -; RUN: -relocation-model=pic < %s | \ +; RUN: -relocation-model=pic -mips-jalr-reloc=false < %s | \ ; RUN: FileCheck %s -check-prefix=NO-SHRINK-WRAP-PIC ; RUN: llc -mtriple=mips64-unknown-linux-gnu -enable-shrink-wrap=true \ ; RUN: -relocation-model=static < %s | \ ; RUN: FileCheck %s -check-prefix=SHRINK-WRAP-64-STATIC ; RUN: llc -mtriple=mips64-unknown-linux-gnu -enable-shrink-wrap=false \ ; RUN: -relocation-model=static < %s | \ ; RUN: FileCheck %s -check-prefix=NO-SHRINK-WRAP-64-STATIC ; RUN: llc -mtriple=mips64-unknown-linux-gnu -enable-shrink-wrap=true \ -; RUN: -relocation-model=pic < %s | \ +; RUN: -relocation-model=pic -mips-jalr-reloc=false < %s | \ ; RUN: FileCheck %s -check-prefix=SHRINK-WRAP-64-PIC ; RUN: llc -mtriple=mips64-unknown-linux-gnu -enable-shrink-wrap=false \ -; RUN: -relocation-model=pic < %s | \ +; RUN: -relocation-model=pic -mips-jalr-reloc=false < %s | \ ; RUN: FileCheck %s -check-prefix=NO-SHRINK-WRAP-64-PIC declare void @f(i32 signext) define i32 @foo(i32 signext %a) { ; SHRINK-WRAP-STATIC-LABEL: foo: ; SHRINK-WRAP-STATIC: # %bb.0: # %entry ; SHRINK-WRAP-STATIC-NEXT: beqz $4, $BB0_2 ; SHRINK-WRAP-STATIC-NEXT: nop ; SHRINK-WRAP-STATIC-NEXT: # %bb.1: # %if.end ; SHRINK-WRAP-STATIC-NEXT: addiu $sp, $sp, -24 ; SHRINK-WRAP-STATIC-NEXT: .cfi_def_cfa_offset 24 ; SHRINK-WRAP-STATIC-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; SHRINK-WRAP-STATIC-NEXT: .cfi_offset 31, -4 ; SHRINK-WRAP-STATIC-NEXT: jal f ; SHRINK-WRAP-STATIC-NEXT: addiu $4, $4, 1 ; SHRINK-WRAP-STATIC-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; SHRINK-WRAP-STATIC-NEXT: addiu $sp, $sp, 24 ; SHRINK-WRAP-STATIC-NEXT: $BB0_2: # %return ; SHRINK-WRAP-STATIC-NEXT: jr $ra ; SHRINK-WRAP-STATIC-NEXT: addiu $2, $zero, 0 ; ; NO-SHRINK-WRAP-STATIC-LABEL: foo: ; NO-SHRINK-WRAP-STATIC: # %bb.0: # %entry ; NO-SHRINK-WRAP-STATIC-NEXT: addiu $sp, $sp, -24 ; NO-SHRINK-WRAP-STATIC-NEXT: .cfi_def_cfa_offset 24 ; NO-SHRINK-WRAP-STATIC-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; NO-SHRINK-WRAP-STATIC-NEXT: .cfi_offset 31, -4 ; NO-SHRINK-WRAP-STATIC-NEXT: beqz $4, $BB0_2 ; NO-SHRINK-WRAP-STATIC-NEXT: nop ; NO-SHRINK-WRAP-STATIC-NEXT: # %bb.1: # %if.end ; NO-SHRINK-WRAP-STATIC-NEXT: jal f ; NO-SHRINK-WRAP-STATIC-NEXT: addiu $4, $4, 1 ; NO-SHRINK-WRAP-STATIC-NEXT: $BB0_2: # %return ; NO-SHRINK-WRAP-STATIC-NEXT: addiu $2, $zero, 0 ; NO-SHRINK-WRAP-STATIC-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; NO-SHRINK-WRAP-STATIC-NEXT: jr $ra ; NO-SHRINK-WRAP-STATIC-NEXT: addiu $sp, $sp, 24 ; ; SHRINK-WRAP-PIC-LABEL: foo: ; SHRINK-WRAP-PIC: # %bb.0: # %entry ; SHRINK-WRAP-PIC-NEXT: lui $2, %hi(_gp_disp) ; SHRINK-WRAP-PIC-NEXT: addiu $2, $2, %lo(_gp_disp) ; SHRINK-WRAP-PIC-NEXT: beqz $4, $BB0_2 ; SHRINK-WRAP-PIC-NEXT: addu $gp, $2, $25 ; SHRINK-WRAP-PIC-NEXT: # %bb.1: # %if.end ; SHRINK-WRAP-PIC-NEXT: addiu $sp, $sp, -24 ; SHRINK-WRAP-PIC-NEXT: .cfi_def_cfa_offset 24 ; SHRINK-WRAP-PIC-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; SHRINK-WRAP-PIC-NEXT: .cfi_offset 31, -4 ; SHRINK-WRAP-PIC-NEXT: lw $25, %call16(f)($gp) ; SHRINK-WRAP-PIC-NEXT: jalr $25 ; SHRINK-WRAP-PIC-NEXT: addiu $4, $4, 1 ; SHRINK-WRAP-PIC-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; SHRINK-WRAP-PIC-NEXT: addiu $sp, $sp, 24 ; SHRINK-WRAP-PIC-NEXT: $BB0_2: # %return ; SHRINK-WRAP-PIC-NEXT: jr $ra ; SHRINK-WRAP-PIC-NEXT: addiu $2, $zero, 0 ; ; NO-SHRINK-WRAP-PIC-LABEL: foo: ; NO-SHRINK-WRAP-PIC: # %bb.0: # %entry ; NO-SHRINK-WRAP-PIC-NEXT: lui $2, %hi(_gp_disp) ; NO-SHRINK-WRAP-PIC-NEXT: addiu $2, $2, %lo(_gp_disp) ; NO-SHRINK-WRAP-PIC-NEXT: addiu $sp, $sp, -24 ; NO-SHRINK-WRAP-PIC-NEXT: .cfi_def_cfa_offset 24 ; NO-SHRINK-WRAP-PIC-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; NO-SHRINK-WRAP-PIC-NEXT: .cfi_offset 31, -4 ; NO-SHRINK-WRAP-PIC-NEXT: beqz $4, $BB0_2 ; NO-SHRINK-WRAP-PIC-NEXT: addu $gp, $2, $25 ; NO-SHRINK-WRAP-PIC-NEXT: # %bb.1: # %if.end ; NO-SHRINK-WRAP-PIC-NEXT: lw $25, %call16(f)($gp) ; NO-SHRINK-WRAP-PIC-NEXT: jalr $25 ; NO-SHRINK-WRAP-PIC-NEXT: addiu $4, $4, 1 ; NO-SHRINK-WRAP-PIC-NEXT: $BB0_2: # %return ; NO-SHRINK-WRAP-PIC-NEXT: addiu $2, $zero, 0 ; NO-SHRINK-WRAP-PIC-NEXT: lw $ra, 20($sp) # 4-byte Folded Reload ; NO-SHRINK-WRAP-PIC-NEXT: jr $ra ; NO-SHRINK-WRAP-PIC-NEXT: addiu $sp, $sp, 24 ; ; SHRINK-WRAP-64-STATIC-LABEL: foo: ; SHRINK-WRAP-64-STATIC: # %bb.0: # %entry ; SHRINK-WRAP-64-STATIC-NEXT: beqz $4, .LBB0_2 ; SHRINK-WRAP-64-STATIC-NEXT: nop ; SHRINK-WRAP-64-STATIC-NEXT: # %bb.1: # %if.end ; SHRINK-WRAP-64-STATIC-NEXT: daddiu $sp, $sp, -16 ; SHRINK-WRAP-64-STATIC-NEXT: .cfi_def_cfa_offset 16 ; SHRINK-WRAP-64-STATIC-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; SHRINK-WRAP-64-STATIC-NEXT: .cfi_offset 31, -8 ; SHRINK-WRAP-64-STATIC-NEXT: jal f ; SHRINK-WRAP-64-STATIC-NEXT: addiu $4, $4, 1 ; SHRINK-WRAP-64-STATIC-NEXT: ld $ra, 8($sp) # 8-byte Folded Reload ; SHRINK-WRAP-64-STATIC-NEXT: daddiu $sp, $sp, 16 ; SHRINK-WRAP-64-STATIC-NEXT: .LBB0_2: # %return ; SHRINK-WRAP-64-STATIC-NEXT: jr $ra ; SHRINK-WRAP-64-STATIC-NEXT: addiu $2, $zero, 0 ; ; NO-SHRINK-WRAP-64-STATIC-LABEL: foo: ; NO-SHRINK-WRAP-64-STATIC: # %bb.0: # %entry ; NO-SHRINK-WRAP-64-STATIC-NEXT: daddiu $sp, $sp, -16 ; NO-SHRINK-WRAP-64-STATIC-NEXT: .cfi_def_cfa_offset 16 ; NO-SHRINK-WRAP-64-STATIC-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; NO-SHRINK-WRAP-64-STATIC-NEXT: .cfi_offset 31, -8 ; NO-SHRINK-WRAP-64-STATIC-NEXT: beqz $4, .LBB0_2 ; NO-SHRINK-WRAP-64-STATIC-NEXT: nop ; NO-SHRINK-WRAP-64-STATIC-NEXT: # %bb.1: # %if.end ; NO-SHRINK-WRAP-64-STATIC-NEXT: jal f ; NO-SHRINK-WRAP-64-STATIC-NEXT: addiu $4, $4, 1 ; NO-SHRINK-WRAP-64-STATIC-NEXT: .LBB0_2: # %return ; NO-SHRINK-WRAP-64-STATIC-NEXT: addiu $2, $zero, 0 ; NO-SHRINK-WRAP-64-STATIC-NEXT: ld $ra, 8($sp) # 8-byte Folded Reload ; NO-SHRINK-WRAP-64-STATIC-NEXT: jr $ra ; NO-SHRINK-WRAP-64-STATIC-NEXT: daddiu $sp, $sp, 16 ; ; SHRINK-WRAP-64-PIC-LABEL: foo: ; SHRINK-WRAP-64-PIC: # %bb.0: # %entry ; SHRINK-WRAP-64-PIC-NEXT: lui $1, %hi(%neg(%gp_rel(foo))) ; SHRINK-WRAP-64-PIC-NEXT: beqz $4, .LBB0_2 ; SHRINK-WRAP-64-PIC-NEXT: daddu $2, $1, $25 ; SHRINK-WRAP-64-PIC-NEXT: # %bb.1: # %if.end ; SHRINK-WRAP-64-PIC-NEXT: daddiu $sp, $sp, -16 ; SHRINK-WRAP-64-PIC-NEXT: .cfi_def_cfa_offset 16 ; SHRINK-WRAP-64-PIC-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; SHRINK-WRAP-64-PIC-NEXT: sd $gp, 0($sp) # 8-byte Folded Spill ; SHRINK-WRAP-64-PIC-NEXT: .cfi_offset 31, -8 ; SHRINK-WRAP-64-PIC-NEXT: .cfi_offset 28, -16 ; SHRINK-WRAP-64-PIC-NEXT: daddiu $gp, $2, %lo(%neg(%gp_rel(foo))) ; SHRINK-WRAP-64-PIC-NEXT: ld $25, %call16(f)($gp) ; SHRINK-WRAP-64-PIC-NEXT: jalr $25 ; SHRINK-WRAP-64-PIC-NEXT: addiu $4, $4, 1 ; SHRINK-WRAP-64-PIC-NEXT: ld $gp, 0($sp) # 8-byte Folded Reload ; SHRINK-WRAP-64-PIC-NEXT: ld $ra, 8($sp) # 8-byte Folded Reload ; SHRINK-WRAP-64-PIC-NEXT: daddiu $sp, $sp, 16 ; SHRINK-WRAP-64-PIC-NEXT: .LBB0_2: # %return ; SHRINK-WRAP-64-PIC-NEXT: jr $ra ; SHRINK-WRAP-64-PIC-NEXT: addiu $2, $zero, 0 ; ; NO-SHRINK-WRAP-64-PIC-LABEL: foo: ; NO-SHRINK-WRAP-64-PIC: # %bb.0: # %entry ; NO-SHRINK-WRAP-64-PIC-NEXT: daddiu $sp, $sp, -16 ; NO-SHRINK-WRAP-64-PIC-NEXT: .cfi_def_cfa_offset 16 ; NO-SHRINK-WRAP-64-PIC-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; NO-SHRINK-WRAP-64-PIC-NEXT: sd $gp, 0($sp) # 8-byte Folded Spill ; NO-SHRINK-WRAP-64-PIC-NEXT: .cfi_offset 31, -8 ; NO-SHRINK-WRAP-64-PIC-NEXT: .cfi_offset 28, -16 ; NO-SHRINK-WRAP-64-PIC-NEXT: lui $1, %hi(%neg(%gp_rel(foo))) ; NO-SHRINK-WRAP-64-PIC-NEXT: beqz $4, .LBB0_2 ; NO-SHRINK-WRAP-64-PIC-NEXT: daddu $2, $1, $25 ; NO-SHRINK-WRAP-64-PIC-NEXT: # %bb.1: # %if.end ; NO-SHRINK-WRAP-64-PIC-NEXT: daddiu $gp, $2, %lo(%neg(%gp_rel(foo))) ; NO-SHRINK-WRAP-64-PIC-NEXT: ld $25, %call16(f)($gp) ; NO-SHRINK-WRAP-64-PIC-NEXT: jalr $25 ; NO-SHRINK-WRAP-64-PIC-NEXT: addiu $4, $4, 1 ; NO-SHRINK-WRAP-64-PIC-NEXT: .LBB0_2: # %return ; NO-SHRINK-WRAP-64-PIC-NEXT: addiu $2, $zero, 0 ; NO-SHRINK-WRAP-64-PIC-NEXT: ld $gp, 0($sp) # 8-byte Folded Reload ; NO-SHRINK-WRAP-64-PIC-NEXT: ld $ra, 8($sp) # 8-byte Folded Reload ; NO-SHRINK-WRAP-64-PIC-NEXT: jr $ra ; NO-SHRINK-WRAP-64-PIC-NEXT: daddiu $sp, $sp, 16 entry: %cmp = icmp eq i32 %a, 0 br i1 %cmp, label %return, label %if.end if.end: %add = add nsw i32 %a, 1 tail call void @f(i32 signext %add) br label %return return: ret i32 0 } ; Test that long branch expansion works correctly with shrink-wrapping enabled. define i32 @foo2(i32 signext %a) { ; SHRINK-WRAP-STATIC-LABEL: foo2: ; SHRINK-WRAP-STATIC: # %bb.0: ; SHRINK-WRAP-STATIC-NEXT: addiu $1, $zero, 4 ; SHRINK-WRAP-STATIC-NEXT: bne $4, $1, $BB1_2 ; SHRINK-WRAP-STATIC-NEXT: nop ; SHRINK-WRAP-STATIC-NEXT: # %bb.1: ; SHRINK-WRAP-STATIC-NEXT: j $BB1_3 ; SHRINK-WRAP-STATIC-NEXT: nop ; SHRINK-WRAP-STATIC-NEXT: $BB1_2: # %if.then ; SHRINK-WRAP-STATIC-NEXT: addiu $sp, $sp, -24 ; SHRINK-WRAP-STATIC-NEXT: .cfi_def_cfa_offset 24 ; SHRINK-WRAP-STATIC-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; SHRINK-WRAP-STATIC-NEXT: .cfi_offset 31, -4 ; SHRINK-WRAP-STATIC-NEXT: #APP ; ; NO-SHRINK-WRAP-STATIC-LABEL: foo2: ; NO-SHRINK-WRAP-STATIC: # %bb.0: ; NO-SHRINK-WRAP-STATIC-NEXT: addiu $sp, $sp, -24 ; NO-SHRINK-WRAP-STATIC-NEXT: .cfi_def_cfa_offset 24 ; NO-SHRINK-WRAP-STATIC-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; NO-SHRINK-WRAP-STATIC-NEXT: .cfi_offset 31, -4 ; NO-SHRINK-WRAP-STATIC-NEXT: addiu $1, $zero, 4 ; NO-SHRINK-WRAP-STATIC-NEXT: bne $4, $1, $BB1_2 ; NO-SHRINK-WRAP-STATIC-NEXT: nop ; NO-SHRINK-WRAP-STATIC-NEXT: # %bb.1: ; NO-SHRINK-WRAP-STATIC-NEXT: j $BB1_3 ; NO-SHRINK-WRAP-STATIC-NEXT: nop ; NO-SHRINK-WRAP-STATIC-NEXT: $BB1_2: # %if.then ; NO-SHRINK-WRAP-STATIC-NEXT: #APP ; ; SHRINK-WRAP-PIC-LABEL: foo2: ; SHRINK-WRAP-PIC: # %bb.0: ; SHRINK-WRAP-PIC-NEXT: lui $2, %hi(_gp_disp) ; SHRINK-WRAP-PIC-NEXT: addiu $2, $2, %lo(_gp_disp) ; SHRINK-WRAP-PIC-NEXT: addiu $1, $zero, 4 ; SHRINK-WRAP-PIC-NEXT: bne $4, $1, $BB1_3 ; SHRINK-WRAP-PIC-NEXT: addu $gp, $2, $25 ; SHRINK-WRAP-PIC-NEXT: # %bb.1: ; SHRINK-WRAP-PIC-NEXT: addiu $sp, $sp, -8 ; SHRINK-WRAP-PIC-NEXT: sw $ra, 0($sp) ; SHRINK-WRAP-PIC-NEXT: lui $1, %hi(($BB1_4)-($BB1_2)) ; SHRINK-WRAP-PIC-NEXT: bal $BB1_2 ; SHRINK-WRAP-PIC-NEXT: addiu $1, $1, %lo(($BB1_4)-($BB1_2)) ; SHRINK-WRAP-PIC-NEXT: $BB1_2: ; SHRINK-WRAP-PIC-NEXT: addu $1, $ra, $1 ; SHRINK-WRAP-PIC-NEXT: lw $ra, 0($sp) ; SHRINK-WRAP-PIC-NEXT: jr $1 ; SHRINK-WRAP-PIC-NEXT: addiu $sp, $sp, 8 ; SHRINK-WRAP-PIC-NEXT: $BB1_3: # %if.then ; SHRINK-WRAP-PIC-NEXT: addiu $sp, $sp, -24 ; SHRINK-WRAP-PIC-NEXT: .cfi_def_cfa_offset 24 ; SHRINK-WRAP-PIC-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; SHRINK-WRAP-PIC-NEXT: .cfi_offset 31, -4 ; SHRINK-WRAP-PIC-NEXT: #APP ; ; NO-SHRINK-WRAP-PIC-LABEL: foo2: ; NO-SHRINK-WRAP-PIC: # %bb.0: ; NO-SHRINK-WRAP-PIC-NEXT: lui $2, %hi(_gp_disp) ; NO-SHRINK-WRAP-PIC-NEXT: addiu $2, $2, %lo(_gp_disp) ; NO-SHRINK-WRAP-PIC-NEXT: addiu $sp, $sp, -24 ; NO-SHRINK-WRAP-PIC-NEXT: .cfi_def_cfa_offset 24 ; NO-SHRINK-WRAP-PIC-NEXT: sw $ra, 20($sp) # 4-byte Folded Spill ; NO-SHRINK-WRAP-PIC-NEXT: .cfi_offset 31, -4 ; NO-SHRINK-WRAP-PIC-NEXT: addiu $1, $zero, 4 ; NO-SHRINK-WRAP-PIC-NEXT: bne $4, $1, $BB1_3 ; NO-SHRINK-WRAP-PIC-NEXT: addu $gp, $2, $25 ; NO-SHRINK-WRAP-PIC-NEXT: # %bb.1: ; NO-SHRINK-WRAP-PIC-NEXT: addiu $sp, $sp, -8 ; NO-SHRINK-WRAP-PIC-NEXT: sw $ra, 0($sp) ; NO-SHRINK-WRAP-PIC-NEXT: lui $1, %hi(($BB1_4)-($BB1_2)) ; NO-SHRINK-WRAP-PIC-NEXT: bal $BB1_2 ; NO-SHRINK-WRAP-PIC-NEXT: addiu $1, $1, %lo(($BB1_4)-($BB1_2)) ; NO-SHRINK-WRAP-PIC-NEXT: $BB1_2: ; NO-SHRINK-WRAP-PIC-NEXT: addu $1, $ra, $1 ; NO-SHRINK-WRAP-PIC-NEXT: lw $ra, 0($sp) ; NO-SHRINK-WRAP-PIC-NEXT: jr $1 ; NO-SHRINK-WRAP-PIC-NEXT: addiu $sp, $sp, 8 ; NO-SHRINK-WRAP-PIC-NEXT: $BB1_3: # %if.then ; NO-SHRINK-WRAP-PIC-NEXT: #APP ; ; SHRINK-WRAP-64-STATIC-LABEL: foo2: ; SHRINK-WRAP-64-STATIC: # %bb.0: ; SHRINK-WRAP-64-STATIC-NEXT: addiu $1, $zero, 4 ; SHRINK-WRAP-64-STATIC-NEXT: bne $4, $1, .LBB1_2 ; SHRINK-WRAP-64-STATIC-NEXT: nop ; SHRINK-WRAP-64-STATIC-NEXT: # %bb.1: ; SHRINK-WRAP-64-STATIC-NEXT: j .LBB1_3 ; SHRINK-WRAP-64-STATIC-NEXT: nop ; SHRINK-WRAP-64-STATIC-NEXT: .LBB1_2: # %if.then ; SHRINK-WRAP-64-STATIC-NEXT: daddiu $sp, $sp, -16 ; SHRINK-WRAP-64-STATIC-NEXT: .cfi_def_cfa_offset 16 ; SHRINK-WRAP-64-STATIC-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; SHRINK-WRAP-64-STATIC-NEXT: .cfi_offset 31, -8 ; SHRINK-WRAP-64-STATIC-NEXT: sll $4, $4, 0 ; SHRINK-WRAP-64-STATIC-NEXT: #APP ; ; NO-SHRINK-WRAP-64-STATIC-LABEL: foo2: ; NO-SHRINK-WRAP-64-STATIC: # %bb.0: ; NO-SHRINK-WRAP-64-STATIC-NEXT: daddiu $sp, $sp, -16 ; NO-SHRINK-WRAP-64-STATIC-NEXT: .cfi_def_cfa_offset 16 ; NO-SHRINK-WRAP-64-STATIC-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; NO-SHRINK-WRAP-64-STATIC-NEXT: .cfi_offset 31, -8 ; NO-SHRINK-WRAP-64-STATIC-NEXT: addiu $1, $zero, 4 ; NO-SHRINK-WRAP-64-STATIC-NEXT: bne $4, $1, .LBB1_2 ; NO-SHRINK-WRAP-64-STATIC-NEXT: nop ; NO-SHRINK-WRAP-64-STATIC-NEXT: # %bb.1: ; NO-SHRINK-WRAP-64-STATIC-NEXT: j .LBB1_3 ; NO-SHRINK-WRAP-64-STATIC-NEXT: nop ; NO-SHRINK-WRAP-64-STATIC-NEXT: .LBB1_2: # %if.then ; NO-SHRINK-WRAP-64-STATIC-NEXT: sll $4, $4, 0 ; NO-SHRINK-WRAP-64-STATIC-NEXT: #APP ; ; SHRINK-WRAP-64-PIC-LABEL: foo2: ; SHRINK-WRAP-64-PIC: # %bb.0: ; SHRINK-WRAP-64-PIC-NEXT: lui $1, %hi(%neg(%gp_rel(foo2))) ; SHRINK-WRAP-64-PIC-NEXT: daddu $2, $1, $25 ; SHRINK-WRAP-64-PIC-NEXT: addiu $1, $zero, 4 ; SHRINK-WRAP-64-PIC-NEXT: bne $4, $1, .LBB1_3 ; SHRINK-WRAP-64-PIC-NEXT: nop ; SHRINK-WRAP-64-PIC-NEXT: # %bb.1: ; SHRINK-WRAP-64-PIC-NEXT: daddiu $sp, $sp, -16 ; SHRINK-WRAP-64-PIC-NEXT: sd $ra, 0($sp) ; SHRINK-WRAP-64-PIC-NEXT: daddiu $1, $zero, %hi(.LBB1_4-.LBB1_2) ; SHRINK-WRAP-64-PIC-NEXT: dsll $1, $1, 16 ; SHRINK-WRAP-64-PIC-NEXT: bal .LBB1_2 ; SHRINK-WRAP-64-PIC-NEXT: daddiu $1, $1, %lo(.LBB1_4-.LBB1_2) ; SHRINK-WRAP-64-PIC-NEXT: .LBB1_2: ; SHRINK-WRAP-64-PIC-NEXT: daddu $1, $ra, $1 ; SHRINK-WRAP-64-PIC-NEXT: ld $ra, 0($sp) ; SHRINK-WRAP-64-PIC-NEXT: jr $1 ; SHRINK-WRAP-64-PIC-NEXT: daddiu $sp, $sp, 16 ; SHRINK-WRAP-64-PIC-NEXT: .LBB1_3: # %if.then ; SHRINK-WRAP-64-PIC-NEXT: daddiu $sp, $sp, -16 ; SHRINK-WRAP-64-PIC-NEXT: .cfi_def_cfa_offset 16 ; SHRINK-WRAP-64-PIC-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; SHRINK-WRAP-64-PIC-NEXT: sd $gp, 0($sp) # 8-byte Folded Spill ; SHRINK-WRAP-64-PIC-NEXT: .cfi_offset 31, -8 ; SHRINK-WRAP-64-PIC-NEXT: .cfi_offset 28, -16 ; SHRINK-WRAP-64-PIC-NEXT: daddiu $gp, $2, %lo(%neg(%gp_rel(foo2))) ; SHRINK-WRAP-64-PIC-NEXT: sll $4, $4, 0 ; SHRINK-WRAP-64-PIC-NEXT: #APP ; ; NO-SHRINK-WRAP-64-PIC-LABEL: foo2: ; NO-SHRINK-WRAP-64-PIC: # %bb.0: ; NO-SHRINK-WRAP-64-PIC-NEXT: daddiu $sp, $sp, -16 ; NO-SHRINK-WRAP-64-PIC-NEXT: .cfi_def_cfa_offset 16 ; NO-SHRINK-WRAP-64-PIC-NEXT: sd $ra, 8($sp) # 8-byte Folded Spill ; NO-SHRINK-WRAP-64-PIC-NEXT: sd $gp, 0($sp) # 8-byte Folded Spill ; NO-SHRINK-WRAP-64-PIC-NEXT: .cfi_offset 31, -8 ; NO-SHRINK-WRAP-64-PIC-NEXT: .cfi_offset 28, -16 ; NO-SHRINK-WRAP-64-PIC-NEXT: lui $1, %hi(%neg(%gp_rel(foo2))) ; NO-SHRINK-WRAP-64-PIC-NEXT: daddu $2, $1, $25 ; NO-SHRINK-WRAP-64-PIC-NEXT: addiu $1, $zero, 4 ; NO-SHRINK-WRAP-64-PIC-NEXT: bne $4, $1, .LBB1_3 ; NO-SHRINK-WRAP-64-PIC-NEXT: nop ; NO-SHRINK-WRAP-64-PIC-NEXT: # %bb.1: ; NO-SHRINK-WRAP-64-PIC-NEXT: daddiu $sp, $sp, -16 ; NO-SHRINK-WRAP-64-PIC-NEXT: sd $ra, 0($sp) ; NO-SHRINK-WRAP-64-PIC-NEXT: daddiu $1, $zero, %hi(.LBB1_4-.LBB1_2) ; NO-SHRINK-WRAP-64-PIC-NEXT: dsll $1, $1, 16 ; NO-SHRINK-WRAP-64-PIC-NEXT: bal .LBB1_2 ; NO-SHRINK-WRAP-64-PIC-NEXT: daddiu $1, $1, %lo(.LBB1_4-.LBB1_2) ; NO-SHRINK-WRAP-64-PIC-NEXT: .LBB1_2: ; NO-SHRINK-WRAP-64-PIC-NEXT: daddu $1, $ra, $1 ; NO-SHRINK-WRAP-64-PIC-NEXT: ld $ra, 0($sp) ; NO-SHRINK-WRAP-64-PIC-NEXT: jr $1 ; NO-SHRINK-WRAP-64-PIC-NEXT: daddiu $sp, $sp, 16 ; NO-SHRINK-WRAP-64-PIC-NEXT: .LBB1_3: # %if.then ; NO-SHRINK-WRAP-64-PIC-NEXT: daddiu $gp, $2, %lo(%neg(%gp_rel(foo2))) ; NO-SHRINK-WRAP-64-PIC-NEXT: sll $4, $4, 0 ; NO-SHRINK-WRAP-64-PIC-NEXT: #APP %1 = icmp ne i32 %a, 4 br i1 %1, label %if.then, label %if.end if.then: call void asm sideeffect ".space 1048576", "~{$1}"() call void @f(i32 signext %a) br label %if.end if.end: ret i32 0 } Index: vendor/llvm/dist-release_80/test/CodeGen/X86/debug-loclists.ll =================================================================== --- vendor/llvm/dist-release_80/test/CodeGen/X86/debug-loclists.ll (revision 343793) +++ vendor/llvm/dist-release_80/test/CodeGen/X86/debug-loclists.ll (revision 343794) @@ -1,142 +1,142 @@ ; RUN: llc -mtriple=x86_64-pc-linux -filetype=obj -o %t < %s ; RUN: llvm-dwarfdump -v %t | FileCheck %s ; CHECK: 0x00000033: DW_TAG_formal_parameter [3] ; CHECK-NEXT: DW_AT_location [DW_FORM_sec_offset] (0x0000000c ; CHECK-NEXT: [0x0000000000000000, 0x0000000000000004): DW_OP_breg5 RDI+0 ; CHECK-NEXT: [0x0000000000000004, 0x0000000000000012): DW_OP_breg3 RBX+0) ; CHECK-NEXT: DW_AT_name [DW_FORM_strx1] (indexed (0000000e) string = "a") ; CHECK-NEXT: DW_AT_decl_file [DW_FORM_data1] ("/home/folder{{\\|\/}}test.cc") ; CHECK-NEXT: DW_AT_decl_line [DW_FORM_data1] (6) ; CHECK-NEXT: DW_AT_type [DW_FORM_ref4] (cu + 0x0040 => {0x00000040} "A") ; CHECK: .debug_loclists contents: -; CHECK-NEXT: 0x00000000: locations list header: length = 0x00000017, version = 0x0005, addr_size = 0x08, seg_size = 0x00, offset_entry_count = 0x00000000 +; CHECK-NEXT: 0x00000000: locations list header: length = 0x00000015, version = 0x0005, addr_size = 0x08, seg_size = 0x00, offset_entry_count = 0x00000000 ; CHECK-NEXT: 0x00000000: ; CHECK-NEXT: [0x0000000000000000, 0x0000000000000004): DW_OP_breg5 RDI+0 ; CHECK-NEXT: [0x0000000000000004, 0x0000000000000012): DW_OP_breg3 RBX+0 ; There is no way to use llvm-dwarfdump atm (2018, october) to verify the DW_LLE_* codes emited, ; because dumper is not yet implements that. Use asm code to do this check instead. ; ; RUN: llc -mtriple=x86_64-pc-linux -filetype=asm < %s -o - | FileCheck %s --check-prefix=ASM ; ASM: .section .debug_loclists,"",@progbits ; ASM-NEXT: .long .Ldebug_loclist_table_end0-.Ldebug_loclist_table_start0 # Length ; ASM-NEXT: .Ldebug_loclist_table_start0: ; ASM-NEXT: .short 5 # Version ; ASM-NEXT: .byte 8 # Address size ; ASM-NEXT: .byte 0 # Segment selector size ; ASM-NEXT: .long 0 # Offset entry count ; ASM-NEXT: .Lloclists_table_base0: ; ASM-NEXT: .Ldebug_loc0: ; ASM-NEXT: .byte 4 # DW_LLE_offset_pair ; ASM-NEXT: .uleb128 .Lfunc_begin0-.Lfunc_begin0 # starting offset ; ASM-NEXT: .uleb128 .Ltmp0-.Lfunc_begin0 # ending offset -; ASM-NEXT: .short 2 # Loc expr size +; ASM-NEXT: .byte 2 # Loc expr size ; ASM-NEXT: .byte 117 # DW_OP_breg5 ; ASM-NEXT: .byte 0 # 0 ; ASM-NEXT: .byte 4 # DW_LLE_offset_pair ; ASM-NEXT: .uleb128 .Ltmp0-.Lfunc_begin0 # starting offset ; ASM-NEXT: .uleb128 .Ltmp1-.Lfunc_begin0 # ending offset -; ASM-NEXT: .short 2 # Loc expr size +; ASM-NEXT: .byte 2 # Loc expr size ; ASM-NEXT: .byte 115 # DW_OP_breg3 ; ASM-NEXT: .byte 0 # 0 ; ASM-NEXT: .byte 0 # DW_LLE_end_of_list ; ASM-NEXT: .Ldebug_loclist_table_end0: ; ModuleID = 'test.cc' source_filename = "test.cc" target datalayout = "e-m:e-i64:64-f80:128-n8:16:32:64-S128" target triple = "x86_64-unknown-linux-gnu" %struct.A = type { i32 (...)** } @_ZTV1A = dso_local unnamed_addr constant { [4 x i8*] } { [4 x i8*] [i8* null, i8* bitcast ({ i8*, i8* }* @_ZTI1A to i8*), i8* bitcast (void (%struct.A*)* @_ZN1A3fooEv to i8*), i8* bitcast (void (%struct.A*)* @_ZN1A3barEv to i8*)] }, align 8 @_ZTVN10__cxxabiv117__class_type_infoE = external dso_local global i8* @_ZTS1A = dso_local constant [3 x i8] c"1A\00", align 1 @_ZTI1A = dso_local constant { i8*, i8* } { i8* bitcast (i8** getelementptr inbounds (i8*, i8** @_ZTVN10__cxxabiv117__class_type_infoE, i64 2) to i8*), i8* getelementptr inbounds ([3 x i8], [3 x i8]* @_ZTS1A, i32 0, i32 0) }, align 8 ; Function Attrs: noinline optnone uwtable define dso_local void @_Z3baz1A(%struct.A* %a) #0 !dbg !7 { entry: call void @llvm.dbg.declare(metadata %struct.A* %a, metadata !23, metadata !DIExpression()), !dbg !24 call void @_ZN1A3fooEv(%struct.A* %a), !dbg !25 call void @_ZN1A3barEv(%struct.A* %a), !dbg !26 ret void, !dbg !27 } ; Function Attrs: nounwind readnone speculatable declare void @llvm.dbg.declare(metadata, metadata, metadata) #1 ; Function Attrs: noinline nounwind optnone uwtable define dso_local void @_ZN1A3fooEv(%struct.A* %this) unnamed_addr #2 align 2 !dbg !28 { entry: %this.addr = alloca %struct.A*, align 8 store %struct.A* %this, %struct.A** %this.addr, align 8 call void @llvm.dbg.declare(metadata %struct.A** %this.addr, metadata !29, metadata !DIExpression()), !dbg !31 %this1 = load %struct.A*, %struct.A** %this.addr, align 8 ret void, !dbg !32 } ; Function Attrs: noinline nounwind optnone uwtable define dso_local void @_ZN1A3barEv(%struct.A* %this) unnamed_addr #2 align 2 !dbg !33 { entry: %this.addr = alloca %struct.A*, align 8 store %struct.A* %this, %struct.A** %this.addr, align 8 call void @llvm.dbg.declare(metadata %struct.A** %this.addr, metadata !34, metadata !DIExpression()), !dbg !35 %this1 = load %struct.A*, %struct.A** %this.addr, align 8 ret void, !dbg !36 } ; Function Attrs: noinline norecurse nounwind optnone uwtable define dso_local i32 @main() #3 !dbg !37 { entry: %retval = alloca i32, align 4 store i32 0, i32* %retval, align 4 ret i32 0, !dbg !38 } !llvm.dbg.cu = !{!0} !llvm.module.flags = !{!3, !4, !5} !llvm.ident = !{!6} !0 = distinct !DICompileUnit(language: DW_LANG_C_plus_plus, file: !1, producer: "clang version 8.0.0 (trunk 344035)", isOptimized: false, runtimeVersion: 0, emissionKind: FullDebug, enums: !2, nameTableKind: None) !1 = !DIFile(filename: "test.cc", directory: "/home/folder", checksumkind: CSK_MD5, checksum: "e0f357ad6dcb791a774a0dae55baf5e7") !2 = !{} !3 = !{i32 2, !"Dwarf Version", i32 5} !4 = !{i32 2, !"Debug Info Version", i32 3} !5 = !{i32 1, !"wchar_size", i32 4} !6 = !{!"clang version 8.0.0 (trunk 344035)"} !7 = distinct !DISubprogram(name: "baz", linkageName: "_Z3baz1A", scope: !1, file: !1, line: 6, type: !8, isLocal: false, isDefinition: true, scopeLine: 6, flags: DIFlagPrototyped, isOptimized: false, unit: !0, retainedNodes: !2) !8 = !DISubroutineType(types: !9) !9 = !{null, !10} !10 = distinct !DICompositeType(tag: DW_TAG_structure_type, name: "A", file: !1, line: 1, size: 64, flags: DIFlagTypePassByReference, elements: !11, vtableHolder: !10, identifier: "_ZTS1A") !11 = !{!12, !18, !22} !12 = !DIDerivedType(tag: DW_TAG_member, name: "_vptr$A", scope: !1, file: !1, baseType: !13, size: 64, flags: DIFlagArtificial) !13 = !DIDerivedType(tag: DW_TAG_pointer_type, baseType: !14, size: 64) !14 = !DIDerivedType(tag: DW_TAG_pointer_type, name: "__vtbl_ptr_type", baseType: !15, size: 64) !15 = !DISubroutineType(types: !16) !16 = !{!17} !17 = !DIBasicType(name: "int", size: 32, encoding: DW_ATE_signed) !18 = !DISubprogram(name: "foo", linkageName: "_ZN1A3fooEv", scope: !10, file: !1, line: 2, type: !19, isLocal: false, isDefinition: false, scopeLine: 2, containingType: !10, virtuality: DW_VIRTUALITY_virtual, virtualIndex: 0, flags: DIFlagPrototyped, isOptimized: false) !19 = !DISubroutineType(types: !20) !20 = !{null, !21} !21 = !DIDerivedType(tag: DW_TAG_pointer_type, baseType: !10, size: 64, flags: DIFlagArtificial | DIFlagObjectPointer) !22 = !DISubprogram(name: "bar", linkageName: "_ZN1A3barEv", scope: !10, file: !1, line: 3, type: !19, isLocal: false, isDefinition: false, scopeLine: 3, containingType: !10, virtuality: DW_VIRTUALITY_virtual, virtualIndex: 1, flags: DIFlagPrototyped, isOptimized: false) !23 = !DILocalVariable(name: "a", arg: 1, scope: !7, file: !1, line: 6, type: !10) !24 = !DILocation(line: 6, column: 19, scope: !7) !25 = !DILocation(line: 7, column: 6, scope: !7) !26 = !DILocation(line: 8, column: 6, scope: !7) !27 = !DILocation(line: 9, column: 1, scope: !7) !28 = distinct !DISubprogram(name: "foo", linkageName: "_ZN1A3fooEv", scope: !10, file: !1, line: 12, type: !19, isLocal: false, isDefinition: true, scopeLine: 12, flags: DIFlagPrototyped, isOptimized: false, unit: !0, declaration: !18, retainedNodes: !2) !29 = !DILocalVariable(name: "this", arg: 1, scope: !28, type: !30, flags: DIFlagArtificial | DIFlagObjectPointer) !30 = !DIDerivedType(tag: DW_TAG_pointer_type, baseType: !10, size: 64) !31 = !DILocation(line: 0, scope: !28) !32 = !DILocation(line: 12, column: 16, scope: !28) !33 = distinct !DISubprogram(name: "bar", linkageName: "_ZN1A3barEv", scope: !10, file: !1, line: 13, type: !19, isLocal: false, isDefinition: true, scopeLine: 13, flags: DIFlagPrototyped, isOptimized: false, unit: !0, declaration: !22, retainedNodes: !2) !34 = !DILocalVariable(name: "this", arg: 1, scope: !33, type: !30, flags: DIFlagArtificial | DIFlagObjectPointer) !35 = !DILocation(line: 0, scope: !33) !36 = !DILocation(line: 13, column: 16, scope: !33) !37 = distinct !DISubprogram(name: "main", scope: !1, file: !1, line: 15, type: !15, isLocal: false, isDefinition: true, scopeLine: 15, flags: DIFlagPrototyped, isOptimized: false, unit: !0, retainedNodes: !2) !38 = !DILocation(line: 16, column: 3, scope: !37) Index: vendor/llvm/dist-release_80/test/CodeGen/X86/discriminate-mem-ops.ll =================================================================== --- vendor/llvm/dist-release_80/test/CodeGen/X86/discriminate-mem-ops.ll (revision 343793) +++ vendor/llvm/dist-release_80/test/CodeGen/X86/discriminate-mem-ops.ll (revision 343794) @@ -1,55 +1,55 @@ -; RUN: llc < %s | FileCheck %s +; RUN: llc -x86-discriminate-memops < %s | FileCheck %s ; ; original source, compiled with -O3 -gmlt -fdebug-info-for-profiling: ; int sum(int* arr, int pos1, int pos2) { ; return arr[pos1] + arr[pos2]; ; } ; ; ModuleID = 'test.cc' source_filename = "test.cc" target datalayout = "e-m:e-i64:64-f80:128-n8:16:32:64-S128" target triple = "x86_64-unknown-linux-gnu" ; Function Attrs: norecurse nounwind readonly uwtable define i32 @sum(i32* %arr, i32 %pos1, i32 %pos2) !dbg !7 { entry: %idxprom = sext i32 %pos1 to i64, !dbg !9 %arrayidx = getelementptr inbounds i32, i32* %arr, i64 %idxprom, !dbg !9 %0 = load i32, i32* %arrayidx, align 4, !dbg !9, !tbaa !10 %idxprom1 = sext i32 %pos2 to i64, !dbg !14 %arrayidx2 = getelementptr inbounds i32, i32* %arr, i64 %idxprom1, !dbg !14 %1 = load i32, i32* %arrayidx2, align 4, !dbg !14, !tbaa !10 %add = add nsw i32 %1, %0, !dbg !15 ret i32 %add, !dbg !16 } attributes #0 = { "target-cpu"="x86-64" } !llvm.dbg.cu = !{!0} !llvm.module.flags = !{!3, !4, !5} !llvm.ident = !{!6} !0 = distinct !DICompileUnit(language: DW_LANG_C_plus_plus, file: !1, isOptimized: true, runtimeVersion: 0, emissionKind: LineTablesOnly, enums: !2, debugInfoForProfiling: true) !1 = !DIFile(filename: "test.cc", directory: "/tmp") !2 = !{} !3 = !{i32 2, !"Dwarf Version", i32 4} !4 = !{i32 2, !"Debug Info Version", i32 3} !5 = !{i32 1, !"wchar_size", i32 4} !6 = !{!"clang version 7.0.0 (trunk 322155) (llvm/trunk 322159)"} !7 = distinct !DISubprogram(name: "sum", linkageName: "sum", scope: !1, file: !1, line: 1, type: !8, isLocal: false, isDefinition: true, scopeLine: 1, flags: DIFlagPrototyped, isOptimized: true, unit: !0) !8 = !DISubroutineType(types: !2) !9 = !DILocation(line: 2, column: 10, scope: !7) !10 = !{!11, !11, i64 0} !11 = !{!"int", !12, i64 0} !12 = !{!"omnipotent char", !13, i64 0} !13 = !{!"Simple C++ TBAA"} !14 = !DILocation(line: 2, column: 22, scope: !7) !15 = !DILocation(line: 2, column: 20, scope: !7) !16 = !DILocation(line: 2, column: 3, scope: !7) ;CHECK-LABEL: sum: ;CHECK: # %bb.0: ;CHECK: movl (%rdi,%rax,4), %eax ;CHECK-NEXT: .loc 1 2 20 discriminator 2 # test.cc:2:20 ;CHECK-NEXT: addl (%rdi,%rcx,4), %eax ;CHECK-NEXT: .loc 1 2 3 # test.cc:2:3 Index: vendor/llvm/dist-release_80/test/CodeGen/X86/insert-prefetch-inline.ll =================================================================== --- vendor/llvm/dist-release_80/test/CodeGen/X86/insert-prefetch-inline.ll (revision 343793) +++ vendor/llvm/dist-release_80/test/CodeGen/X86/insert-prefetch-inline.ll (revision 343794) @@ -1,76 +1,76 @@ -; RUN: llc < %s -prefetch-hints-file=%S/insert-prefetch-inline.afdo | FileCheck %s +; RUN: llc < %s -x86-discriminate-memops -prefetch-hints-file=%S/insert-prefetch-inline.afdo | FileCheck %s ; ; Verify we can insert prefetch instructions in code belonging to inlined ; functions. ; ; ModuleID = 'test.cc' target datalayout = "e-m:e-i64:64-f80:128-n8:16:32:64-S128" target triple = "x86_64-unknown-linux-gnu" ; Function Attrs: norecurse nounwind readonly uwtable define dso_local i32 @sum(i32* nocapture readonly %arr, i32 %pos1, i32 %pos2) local_unnamed_addr #0 !dbg !7 { entry: %idxprom = sext i32 %pos1 to i64, !dbg !10 %arrayidx = getelementptr inbounds i32, i32* %arr, i64 %idxprom, !dbg !10 %0 = load i32, i32* %arrayidx, align 4, !dbg !10, !tbaa !11 %idxprom1 = sext i32 %pos2 to i64, !dbg !15 %arrayidx2 = getelementptr inbounds i32, i32* %arr, i64 %idxprom1, !dbg !15 %1 = load i32, i32* %arrayidx2, align 4, !dbg !15, !tbaa !11 %add = add nsw i32 %1, %0, !dbg !16 ret i32 %add, !dbg !17 } ; "caller" inlines "sum". The associated .afdo file references instructions ; in "caller" that came from "sum"'s inlining. ; ; Function Attrs: norecurse nounwind readonly uwtable define dso_local i32 @caller(i32* nocapture readonly %arr) local_unnamed_addr #0 !dbg !18 { entry: %0 = load i32, i32* %arr, align 4, !dbg !19, !tbaa !11 %arrayidx2.i = getelementptr inbounds i32, i32* %arr, i64 2, !dbg !21 %1 = load i32, i32* %arrayidx2.i, align 4, !dbg !21, !tbaa !11 %add.i = add nsw i32 %1, %0, !dbg !22 ret i32 %add.i, !dbg !23 } attributes #0 = { "target-cpu"="x86-64" } !llvm.dbg.cu = !{!0} !llvm.module.flags = !{!3, !4, !5} !llvm.ident = !{!6} !0 = distinct !DICompileUnit(language: DW_LANG_C_plus_plus, file: !1, producer: "clang version 7.0.0 (trunk 324940) (llvm/trunk 324941)", isOptimized: true, runtimeVersion: 0, emissionKind: LineTablesOnly, enums: !2, debugInfoForProfiling: true) !1 = !DIFile(filename: "test.cc", directory: "/tmp") !2 = !{} !3 = !{i32 2, !"Dwarf Version", i32 4} !4 = !{i32 2, !"Debug Info Version", i32 3} !5 = !{i32 1, !"wchar_size", i32 4} !6 = !{!"clang version 7.0.0 (trunk 324940) (llvm/trunk 324941)"} !7 = distinct !DISubprogram(name: "sum", linkageName: "sum", scope: !8, file: !8, line: 3, type: !9, isLocal: false, isDefinition: true, scopeLine: 3, flags: DIFlagPrototyped, isOptimized: true, unit: !0) !8 = !DIFile(filename: "./test.h", directory: "/tmp") !9 = !DISubroutineType(types: !2) !10 = !DILocation(line: 6, column: 10, scope: !7) !11 = !{!12, !12, i64 0} !12 = !{!"int", !13, i64 0} !13 = !{!"omnipotent char", !14, i64 0} !14 = !{!"Simple C++ TBAA"} !15 = !DILocation(line: 6, column: 22, scope: !7) !16 = !DILocation(line: 6, column: 20, scope: !7) !17 = !DILocation(line: 6, column: 3, scope: !7) !18 = distinct !DISubprogram(name: "caller", linkageName: "caller", scope: !1, file: !1, line: 4, type: !9, isLocal: false, isDefinition: true, scopeLine: 4, flags: DIFlagPrototyped, isOptimized: true, unit: !0) !19 = !DILocation(line: 6, column: 10, scope: !7, inlinedAt: !20) !20 = distinct !DILocation(line: 6, column: 10, scope: !18) !21 = !DILocation(line: 6, column: 22, scope: !7, inlinedAt: !20) !22 = !DILocation(line: 6, column: 20, scope: !7, inlinedAt: !20) !23 = !DILocation(line: 6, column: 3, scope: !18) ; CHECK-LABEL: caller: ; CHECK-LABEL: # %bb.0: ; CHECK-NEXT: .loc 1 6 22 prologue_end ; CHECK-NEXT: prefetchnta 23464(%rdi) ; CHECK-NEXT: movl 8(%rdi), %eax ; CHECK-NEXT: .loc 1 6 20 is_stmt 0 discriminator 2 ; CHECK-NEXT: prefetchnta 8764(%rdi) ; CHECK-NEXT: prefetchnta 64(%rdi) ; CHECK-NEXT: addl (%rdi), %eax Index: vendor/llvm/dist-release_80/test/CodeGen/X86/insert-prefetch-invalid-instr.ll =================================================================== --- vendor/llvm/dist-release_80/test/CodeGen/X86/insert-prefetch-invalid-instr.ll (revision 343793) +++ vendor/llvm/dist-release_80/test/CodeGen/X86/insert-prefetch-invalid-instr.ll (revision 343794) @@ -1,46 +1,46 @@ -; RUN: llc < %s -prefetch-hints-file=%S/insert-prefetch-invalid-instr.afdo | FileCheck %s +; RUN: llc < %s -x86-discriminate-memops -prefetch-hints-file=%S/insert-prefetch-invalid-instr.afdo | FileCheck %s ; ModuleID = 'prefetch.cc' source_filename = "prefetch.cc" target datalayout = "e-m:e-i64:64-f80:128-n8:16:32:64-S128" target triple = "x86_64-unknown-linux-gnu" ; Function Attrs: norecurse nounwind uwtable define dso_local i32 @main() local_unnamed_addr #0 !dbg !7 { entry: tail call void @llvm.prefetch(i8* inttoptr (i64 291 to i8*), i32 0, i32 0, i32 1), !dbg !9 tail call void @llvm.x86.avx512.gatherpf.dpd.512(i8 97, <8 x i32> undef, i8* null, i32 1, i32 2), !dbg !10 ret i32 291, !dbg !11 } ; Function Attrs: inaccessiblemem_or_argmemonly nounwind declare void @llvm.prefetch(i8* nocapture readonly, i32, i32, i32) #1 ; Function Attrs: argmemonly nounwind declare void @llvm.x86.avx512.gatherpf.dpd.512(i8, <8 x i32>, i8*, i32, i32) #2 attributes #0 = {"target-cpu"="x86-64" "target-features"="+avx512pf,+sse4.2,+ssse3"} attributes #1 = { inaccessiblemem_or_argmemonly nounwind } attributes #2 = { argmemonly nounwind } !llvm.dbg.cu = !{!0} !llvm.module.flags = !{!3, !4, !5} !llvm.ident = !{!6} !0 = distinct !DICompileUnit(language: DW_LANG_C_plus_plus, file: !1, isOptimized: true, runtimeVersion: 0, emissionKind: LineTablesOnly, enums: !2, debugInfoForProfiling: true) !1 = !DIFile(filename: "prefetch.cc", directory: "/tmp") !2 = !{} !3 = !{i32 2, !"Dwarf Version", i32 4} !4 = !{i32 2, !"Debug Info Version", i32 3} !5 = !{i32 1, !"wchar_size", i32 4} !6 = !{!"clang version 7.0.0 (trunk 327078) (llvm/trunk 327086)"} !7 = distinct !DISubprogram(name: "main", scope: !1, file: !1, line: 8, type: !8, isLocal: false, isDefinition: true, scopeLine: 8, flags: DIFlagPrototyped, isOptimized: true, unit: !0) !8 = !DISubroutineType(types: !2) !9 = !DILocation(line: 12, column: 3, scope: !7) !10 = !DILocation(line: 14, column: 3, scope: !7) !11 = !DILocation(line: 15, column: 3, scope: !7) ;CHECK-LABEL: main: ;CHECK: # %bb.0: ;CHECK: prefetchnta 291 ;CHECK-NOT: prefetchnta 42(%rax,%ymm0) ;CHECK: vgatherpf1dpd (%rax,%ymm0) {%k1} Index: vendor/llvm/dist-release_80/test/CodeGen/X86/insert-prefetch.ll =================================================================== --- vendor/llvm/dist-release_80/test/CodeGen/X86/insert-prefetch.ll (revision 343793) +++ vendor/llvm/dist-release_80/test/CodeGen/X86/insert-prefetch.ll (revision 343794) @@ -1,101 +1,101 @@ -; RUN: llc < %s -prefetch-hints-file=%S/insert-prefetch.afdo | FileCheck %s -; RUN: llc < %s -prefetch-hints-file=%S/insert-prefetch-other.afdo | FileCheck %s -check-prefix=OTHERS +; RUN: llc < %s -x86-discriminate-memops -prefetch-hints-file=%S/insert-prefetch.afdo | FileCheck %s +; RUN: llc < %s -x86-discriminate-memops -prefetch-hints-file=%S/insert-prefetch-other.afdo | FileCheck %s -check-prefix=OTHERS ; ; original source, compiled with -O3 -gmlt -fdebug-info-for-profiling: ; int sum(int* arr, int pos1, int pos2) { ; return arr[pos1] + arr[pos2]; ; } ; ; NOTE: debug line numbers were adjusted such that the function would start ; at line 15 (an arbitrary number). The sample profile file format uses ; offsets from the start of the symbol instead of file-relative line numbers. ; The .afdo file reflects that - the instructions are offset '1'. ; ; ModuleID = 'test.cc' source_filename = "test.cc" target datalayout = "e-m:e-i64:64-f80:128-n8:16:32:64-S128" target triple = "x86_64-unknown-linux-gnu" define i32 @sum(i32* %arr, i32 %pos1, i32 %pos2) !dbg !35 !prof !37 { entry: %idxprom = sext i32 %pos1 to i64, !dbg !38 %arrayidx = getelementptr inbounds i32, i32* %arr, i64 %idxprom, !dbg !38 %0 = load i32, i32* %arrayidx, align 4, !dbg !38, !tbaa !39 %idxprom1 = sext i32 %pos2 to i64, !dbg !43 %arrayidx2 = getelementptr inbounds i32, i32* %arr, i64 %idxprom1, !dbg !43 %1 = load i32, i32* %arrayidx2, align 4, !dbg !43, !tbaa !39 %add = add nsw i32 %1, %0, !dbg !44 ret i32 %add, !dbg !45 } attributes #0 = { "target-cpu"="x86-64" } !llvm.dbg.cu = !{!0} !llvm.module.flags = !{!3, !4, !5, !6} !llvm.ident = !{!33} !0 = distinct !DICompileUnit(language: DW_LANG_C_plus_plus, file: !1, isOptimized: true, runtimeVersion: 0, emissionKind: LineTablesOnly, enums: !2, debugInfoForProfiling: true) !1 = !DIFile(filename: "test.cc", directory: "/tmp") !2 = !{} !3 = !{i32 2, !"Dwarf Version", i32 4} !4 = !{i32 2, !"Debug Info Version", i32 3} !5 = !{i32 1, !"wchar_size", i32 4} !6 = !{i32 1, !"ProfileSummary", !7} !7 = !{!8, !9, !10, !11, !12, !13, !14, !15} !8 = !{!"ProfileFormat", !"SampleProfile"} !9 = !{!"TotalCount", i64 0} !10 = !{!"MaxCount", i64 0} !11 = !{!"MaxInternalCount", i64 0} !12 = !{!"MaxFunctionCount", i64 0} !13 = !{!"NumCounts", i64 2} !14 = !{!"NumFunctions", i64 1} !15 = !{!"DetailedSummary", !16} !16 = !{!17, !18, !19, !20, !21, !22, !22, !23, !23, !24, !25, !26, !27, !28, !29, !30, !31, !32} !17 = !{i32 10000, i64 0, i32 0} !18 = !{i32 100000, i64 0, i32 0} !19 = !{i32 200000, i64 0, i32 0} !20 = !{i32 300000, i64 0, i32 0} !21 = !{i32 400000, i64 0, i32 0} !22 = !{i32 500000, i64 0, i32 0} !23 = !{i32 600000, i64 0, i32 0} !24 = !{i32 700000, i64 0, i32 0} !25 = !{i32 800000, i64 0, i32 0} !26 = !{i32 900000, i64 0, i32 0} !27 = !{i32 950000, i64 0, i32 0} !28 = !{i32 990000, i64 0, i32 0} !29 = !{i32 999000, i64 0, i32 0} !30 = !{i32 999900, i64 0, i32 0} !31 = !{i32 999990, i64 0, i32 0} !32 = !{i32 999999, i64 0, i32 0} !33 = !{!"clang version 7.0.0 (trunk 322593) (llvm/trunk 322526)"} !35 = distinct !DISubprogram(name: "sum", linkageName: "sum", scope: !1, file: !1, line: 15, type: !36, isLocal: false, isDefinition: true, scopeLine: 15, flags: DIFlagPrototyped, isOptimized: true, unit: !0) !36 = !DISubroutineType(types: !2) !37 = !{!"function_entry_count", i64 -1} !38 = !DILocation(line: 16, column: 10, scope: !35) !39 = !{!40, !40, i64 0} !40 = !{!"int", !41, i64 0} !41 = !{!"omnipotent char", !42, i64 0} !42 = !{!"Simple C++ TBAA"} !43 = !DILocation(line: 16, column: 22, scope: !35) !44 = !DILocation(line: 16, column: 20, scope: !35) !45 = !DILocation(line: 16, column: 3, scope: !35) ;CHECK-LABEL: sum: ;CHECK: # %bb.0: ;CHECK: prefetchnta 42(%rdi,%rax,4) ;CHECK-NEXT: prefetchnta (%rdi,%rax,4) ;CHECK-NEXT: movl (%rdi,%rax,4), %eax ;CHECK-NEXT: .loc 1 16 20 discriminator 2 # test.cc:16:20 ;CHECK-NEXT: prefetchnta -1(%rdi,%rcx,4) ;CHECK-NEXT: addl (%rdi,%rcx,4), %eax ;CHECK-NEXT: .loc 1 16 3 # test.cc:16:3 ;OTHERS-LABEL: sum: ;OTHERS: # %bb.0: ;OTHERS: prefetcht2 42(%rdi,%rax,4) ;OTHERS-NEXT: prefetcht0 (%rdi,%rax,4) ;OTHERS-NEXT: movl (%rdi,%rax,4), %eax ;OTHERS-NEXT: .loc 1 16 20 discriminator 2 # test.cc:16:20 ;OTHERS-NEXT: prefetcht1 -1(%rdi,%rcx,4) ;OTHERS-NEXT: addl (%rdi,%rcx,4), %eax ;OTHERS-NEXT: .loc 1 16 3 # test.cc:16:3 Index: vendor/llvm/dist-release_80/test/DebugInfo/COFF/types-empty-member-fn.ll =================================================================== --- vendor/llvm/dist-release_80/test/DebugInfo/COFF/types-empty-member-fn.ll (nonexistent) +++ vendor/llvm/dist-release_80/test/DebugInfo/COFF/types-empty-member-fn.ll (revision 343794) @@ -0,0 +1,72 @@ +; RUN: llc < %s -filetype=obj | llvm-readobj - -codeview | FileCheck %s + +; ModuleID = 'foo.3a1fbbbh-cgu.0' +source_filename = "foo.3a1fbbbh-cgu.0" +target datalayout = "e-m:w-i64:64-f80:128-n8:16:32:64-S128" +target triple = "x86_64-pc-windows-msvc" + +; Rust source to regenerate: +; $ cat foo.rs +; pub struct Foo; +; impl Foo { +; pub fn foo() {} +; } +; $ rustc foo.rs --crate-type lib -Cdebuginfo=1 --emit=llvm-ir + +; CHECK: CodeViewTypes [ +; CHECK: MemberFunction (0x1006) { +; CHECK-NEXT: TypeLeafKind: LF_MFUNCTION (0x1009) +; CHECK-NEXT: ReturnType: void (0x3) +; CHECK-NEXT: ClassType: foo::Foo (0x1000) +; CHECK-NEXT: ThisType: 0x0 +; CHECK-NEXT: CallingConvention: NearC (0x0) +; CHECK-NEXT: FunctionOptions [ (0x0) +; CHECK-NEXT: ] +; CHECK-NEXT: NumParameters: 0 +; CHECK-NEXT: ArgListType: () (0x1005) +; CHECK-NEXT: ThisAdjustment: 0 +; CHECK-NEXT: } +; CHECK-NEXT: MemberFuncId (0x1007) { +; CHECK-NEXT: TypeLeafKind: LF_MFUNC_ID (0x1602) +; CHECK-NEXT: ClassType: foo::Foo (0x1000) +; CHECK-NEXT: FunctionType: void foo::Foo::() (0x1006) +; CHECK-NEXT: Name: foo +; CHECK-NEXT: } +; CHECK: CodeViewDebugInfo [ +; CHECK: FunctionLineTable [ +; CHECK-NEXT: LinkageName: _ZN3foo3Foo3foo17hc557c2121772885bE +; CHECK-NEXT: Flags: 0x0 +; CHECK-NEXT: CodeSize: 0x1 +; CHECK-NEXT: FilenameSegment [ +; CHECK-NEXT: Filename: D:\rust\foo.rs (0x0) +; CHECK-NEXT: +0x0 [ +; CHECK-NEXT: LineNumberStart: 3 +; CHECK-NEXT: LineNumberEndDelta: 0 +; CHECK-NEXT: IsStatement: No +; CHECK-NEXT: ] +; CHECK-NEXT: ] +; CHECK-NEXT: ] + +; foo::Foo::foo +; Function Attrs: uwtable +define void @_ZN3foo3Foo3foo17hc557c2121772885bE() unnamed_addr #0 !dbg !5 { +start: + ret void, !dbg !10 +} + +attributes #0 = { uwtable "target-cpu"="x86-64" } + +!llvm.dbg.cu = !{!0} +!llvm.module.flags = !{!3, !4} + +!0 = distinct !DICompileUnit(language: DW_LANG_Rust, file: !1, producer: "clang LLVM (rustc version 1.33.0-nightly (8b0f0156e 2019-01-22))", isOptimized: false, runtimeVersion: 0, emissionKind: LineTablesOnly, enums: !2) +!1 = !DIFile(filename: "foo.rs", directory: "D:\5Crust") +!2 = !{} +!3 = !{i32 2, !"CodeView", i32 1} +!4 = !{i32 2, !"Debug Info Version", i32 3} +!5 = distinct !DISubprogram(name: "foo", linkageName: "_ZN3foo3Foo3foo17hc557c2121772885bE", scope: !6, file: !1, line: 3, type: !9, scopeLine: 3, flags: DIFlagPrototyped, spFlags: DISPFlagDefinition, unit: !0, templateParams: !2, retainedNodes: !2) +!6 = !DICompositeType(tag: DW_TAG_structure_type, name: "Foo", scope: !8, file: !7, align: 8, elements: !2, templateParams: !2, identifier: "5105d9fe1a2a3c68518268151b672274") +!7 = !DIFile(filename: "", directory: "") +!8 = !DINamespace(name: "foo", scope: null) +!9 = !DISubroutineType(types: !2) +!10 = !DILocation(line: 3, scope: !5) Index: vendor/llvm/dist-release_80/test/DebugInfo/Mips/dwarfdump-tls.ll =================================================================== --- vendor/llvm/dist-release_80/test/DebugInfo/Mips/dwarfdump-tls.ll (revision 343793) +++ vendor/llvm/dist-release_80/test/DebugInfo/Mips/dwarfdump-tls.ll (revision 343794) @@ -1,21 +1,43 @@ -; RUN: llc -O0 -march=mips -mcpu=mips32r2 -filetype=obj -o=%t-32.o < %s +; RUN: llc -O0 -march=mips -mcpu=mips32r2 -filetype=obj \ +; RUN: -split-dwarf-file=foo.dwo -o=%t-32.o < %s ; RUN: llvm-dwarfdump %t-32.o 2>&1 | FileCheck %s -; RUN: llc -O0 -march=mips64 -mcpu=mips64r2 -filetype=obj -o=%t-64.o < %s +; RUN: llc -O0 -march=mips64 -mcpu=mips64r2 -filetype=obj \ +; RUN: -split-dwarf-file=foo.dwo -o=%t-64.o < %s ; RUN: llvm-dwarfdump %t-64.o 2>&1 | FileCheck %s +; RUN: llc -O0 -march=mips -mcpu=mips32r2 -filetype=asm \ +; RUN: -split-dwarf-file=foo.dwo < %s | FileCheck -check-prefix=ASM32 %s +; RUN: llc -O0 -march=mips64 -mcpu=mips64r2 -filetype=asm \ +; RUN: -split-dwarf-file=foo.dwo < %s | FileCheck -check-prefix=ASM64 %s + @x = thread_local global i32 5, align 4, !dbg !0 ; CHECK-NOT: error: failed to compute relocation: R_MIPS_TLS_DTPREL + +; CHECK: DW_AT_name ("x") +; CHECK-NEXT: DW_AT_type (0x00000025 "int") +; CHECK-NEXT: DW_AT_external (true) +; CHECK-NEXT: DW_AT_decl_file (0x01) +; CHECK-NEXT: DW_AT_decl_line (1) +; CHECK-NEXT: DW_AT_location (DW_OP_GNU_const_index 0x0, {{DW_OP_GNU_push_tls_address|DW_OP_form_tls_address}}) + +; ASM32: .section .debug_addr +; ASM32-NEXT: $addr_table_base0: +; ASM32-NEXT: .4byte x+32768 + +; ASM64: .section .debug_addr +; ASM64-NEXT: .Laddr_table_base0: +; ASM64-NEXT: .8byte x+32768 !llvm.dbg.cu = !{!2} !llvm.module.flags = !{!7, !8} !0 = !DIGlobalVariableExpression(var: !1, expr: !DIExpression()) !1 = distinct !DIGlobalVariable(name: "x", scope: !2, file: !3, line: 1, type: !6, isLocal: false, isDefinition: true) !2 = distinct !DICompileUnit(language: DW_LANG_C99, file: !3, producer: "clang version 4.0.0", isOptimized: false, runtimeVersion: 0, emissionKind: FullDebug, enums: !4, globals: !5) !3 = !DIFile(filename: "tls.c", directory: "/tmp") !4 = !{} !5 = !{!0} !6 = !DIBasicType(name: "int", size: 32, encoding: DW_ATE_signed) !7 = !{i32 2, !"Dwarf Version", i32 4} !8 = !{i32 2, !"Debug Info Version", i32 3} Index: vendor/llvm/dist-release_80/test/DebugInfo/X86/dwarfdump-debug-loclists.test =================================================================== --- vendor/llvm/dist-release_80/test/DebugInfo/X86/dwarfdump-debug-loclists.test (revision 343793) +++ vendor/llvm/dist-release_80/test/DebugInfo/X86/dwarfdump-debug-loclists.test (revision 343794) @@ -1,167 +1,167 @@ # RUN: llvm-mc %s -filetype obj -triple x86_64-pc-linux -o %t.o # RUN: llvm-dwarfdump -v %t.o | FileCheck %s # CHECK: .debug_info # CHECK: DW_AT_name{{.*}}"stub" # CHECK: DW_AT_location [DW_FORM_sec_offset] (0x0000000c # CHECK-NEXT: [0x0000000000000010, 0x0000000000000020): DW_OP_breg5 RDI+0 # CHECK-NEXT: [0x0000000000000530, 0x0000000000000540): DW_OP_breg6 RBP-8, DW_OP_deref # CHECK-NEXT: [0x0000000000000700, 0x0000000000000710): DW_OP_breg5 RDI+0 # CHECK: .debug_loclists contents: -# CHECK-NEXT: 0x00000000: locations list header: length = 0x0000002f, version = 0x0005, addr_size = 0x08, seg_size = 0x00, offset_entry_count = 0x00000000 +# CHECK-NEXT: 0x00000000: locations list header: length = 0x0000002c, version = 0x0005, addr_size = 0x08, seg_size = 0x00, offset_entry_count = 0x00000000 # CHECK-NEXT: 0x00000000: # CHECK-NEXT: [0x0000000000000000, 0x0000000000000010): DW_OP_breg5 RDI+0 # CHECK-NEXT: [0x0000000000000530, 0x0000000000000540): DW_OP_breg6 RBP-8, DW_OP_deref # CHECK-NEXT: [0x0000000000000700, 0x0000000000000710): DW_OP_breg5 RDI+0 .section .debug_str,"MS",@progbits,1 .asciz "stub" .section .debug_str_offsets,"",@progbits .long 68 .short 5 .short 0 .Lstr_offsets_base0: .zero 64 .section .debug_loclists,"",@progbits .long .Ldebug_loclist_table_end0-.Ldebug_loclist_table_start0 .Ldebug_loclist_table_start0: .short 5 # Version. .byte 8 # Address size. .byte 0 # Segmen selector size. .long 0 # Offset entry count. .Lloclists_table_base0: .Ldebug_loc0: .byte 4 # DW_LLE_offset_pair .uleb128 0x0 # starting offset .uleb128 0x10 # ending offset - .short 2 # Loc expr size + .byte 2 # Loc expr size .byte 117 # DW_OP_breg5 .byte 0 # 0 .byte 6 # DW_LLE_base_address .quad 0x500 # Some address .byte 4 # DW_LLE_offset_pair .uleb128 0x30 # starting offset .uleb128 0x40 # ending offset - .short 3 # Loc expr size + .byte 3 # Loc expr size .byte 118 # DW_OP_breg6 .byte 120 # -8 .byte 6 # DW_OP_deref .byte 8 # DW_LLE_start_length .quad 0x700 # Some address .uleb128 0x10 # length - .short 2 # Loc expr size + .byte 2 # Loc expr size .byte 117 # DW_OP_breg5 .byte 0 # 0 .byte 0 # DW_LLE_end_of_list .Ldebug_loclist_table_end0: .section .debug_abbrev,"",@progbits .byte 1 # Abbreviation Code .byte 17 # DW_TAG_compile_unit .byte 1 # DW_CHILDREN_yes .byte 37 # DW_AT_producer .byte 37 # DW_FORM_strx1 .byte 19 # DW_AT_language .byte 5 # DW_FORM_data2 .byte 3 # DW_AT_name .byte 37 # DW_FORM_strx1 .byte 114 # DW_AT_str_offsets_base .byte 23 # DW_FORM_sec_offset .byte 16 # DW_AT_stmt_list .byte 23 # DW_FORM_sec_offset .byte 27 # DW_AT_comp_dir .byte 37 # DW_FORM_strx1 .byte 17 # DW_AT_low_pc .byte 1 # DW_FORM_addr .byte 18 # DW_AT_high_pc .byte 6 # DW_FORM_data4 .ascii "\214\001" # DW_AT_loclists_base .byte 23 # DW_FORM_sec_offset .byte 0 # EOM(1) .byte 0 # EOM(2) .byte 2 # Abbreviation Code .byte 46 # DW_TAG_subprogram .byte 1 # DW_CHILDREN_yes .byte 17 # DW_AT_low_pc .byte 1 # DW_FORM_addr .byte 18 # DW_AT_high_pc .byte 6 # DW_FORM_data4 .byte 64 # DW_AT_frame_base .byte 24 # DW_FORM_exprloc .byte 110 # DW_AT_linkage_name .byte 37 # DW_FORM_strx1 .byte 3 # DW_AT_name .byte 37 # DW_FORM_strx1 .byte 58 # DW_AT_decl_file .byte 11 # DW_FORM_data1 .byte 59 # DW_AT_decl_line .byte 11 # DW_FORM_data1 .byte 63 # DW_AT_external .byte 25 # DW_FORM_flag_present .byte 0 # EOM(1) .byte 0 # EOM(2) .byte 3 # Abbreviation Code .byte 52 # DW_TAG_variable .byte 0 # DW_CHILDREN_no .byte 2 # DW_AT_location .byte 23 # DW_FORM_sec_offset .byte 3 # DW_AT_name .byte 37 # DW_FORM_strx1 .byte 58 # DW_AT_decl_file .byte 11 # DW_FORM_data1 .byte 59 # DW_AT_decl_line .byte 11 # DW_FORM_data1 .byte 73 # DW_AT_type .byte 19 # DW_FORM_ref4 .byte 0 # EOM(1) .byte 0 # EOM(2) .byte 0 # EOM(3) .section .debug_info,"",@progbits .Lcu_begin0: .long 70 # Length of Unit .short 5 # DWARF version number .byte 1 # DWARF Unit Type .byte 8 # Address Size (in bytes) .long .debug_abbrev # Offset Into Abbrev. Section .byte 1 # Abbrev [1] 0xc:0xef DW_TAG_compile_unit .byte 0 # DW_AT_producer .short 4 # DW_AT_language .byte 1 # DW_AT_name .long .Lstr_offsets_base0 # DW_AT_str_offsets_base .long .Lline_table_start0 # DW_AT_stmt_list .byte 2 # DW_AT_comp_dir .quad 0x10 # DW_AT_low_pc .long 0 # DW_AT_high_pc .long .Lloclists_table_base0 # DW_AT_loclists_base .byte 2 # Abbrev [2] 0x2a:0x20 DW_TAG_subprogram .quad 0 # DW_AT_low_pc .long 0 # DW_AT_high_pc .byte 1 # DW_AT_frame_base .byte 86 .byte 11 # DW_AT_linkage_name .byte 12 # DW_AT_name .byte 1 # DW_AT_decl_file .byte 6 # DW_AT_decl_line # DW_AT_external .byte 3 # Abbrev [3] 0x40:0xb DW_TAG_variable .long .Ldebug_loc0 # DW_AT_location .byte 7 # DW_AT_name .byte 1 # DW_AT_decl_file .byte 6 # DW_AT_decl_line .long 76 # DW_AT_type .byte 0 # End Of Children Mark .byte 0 # End Of Children Mark .byte 0 # End Of Children Mark .section .debug_line,"",@progbits .Lline_table_start0: Index: vendor/llvm/dist-release_80/test/Transforms/FunctionImport/Inputs/comdat.ll =================================================================== --- vendor/llvm/dist-release_80/test/Transforms/FunctionImport/Inputs/comdat.ll (nonexistent) +++ vendor/llvm/dist-release_80/test/Transforms/FunctionImport/Inputs/comdat.ll (revision 343794) @@ -0,0 +1,10 @@ +target datalayout = "e-m:w-i64:64-f80:128-n8:16:32:64-S128" +target triple = "x86_64-pc-windows-msvc19.0.24215" + +define void @main() { +entry: + call i8* @lwt_fun() + ret void +} + +declare i8* @lwt_fun() Index: vendor/llvm/dist-release_80/test/Transforms/FunctionImport/comdat.ll =================================================================== --- vendor/llvm/dist-release_80/test/Transforms/FunctionImport/comdat.ll (nonexistent) +++ vendor/llvm/dist-release_80/test/Transforms/FunctionImport/comdat.ll (revision 343794) @@ -0,0 +1,32 @@ +; Test to ensure that comdat is renamed consistently when comdat leader is +; promoted and renamed due to an import. Required by COFF. + +; REQUIRES: x86-registered-target + +; RUN: opt -thinlto-bc -o %t1.bc %s +; RUN: opt -thinlto-bc -o %t2.bc %S/Inputs/comdat.ll +; RUN: llvm-lto2 run -save-temps -o %t3 %t1.bc %t2.bc \ +; RUN: -r %t1.bc,lwt_fun,plx \ +; RUN: -r %t2.bc,main,plx \ +; RUN: -r %t2.bc,lwt_fun, +; RUN: llvm-dis -o - %t3.1.3.import.bc | FileCheck %s + +target datalayout = "e-m:w-i64:64-f80:128-n8:16:32:64-S128" +target triple = "x86_64-pc-windows-msvc19.0.24215" + +; CHECK: $lwt.llvm.[[HASH:[0-9]+]] = comdat any +$lwt = comdat any + +; CHECK: @lwt_aliasee = private unnamed_addr global {{.*}}, comdat($lwt.llvm.[[HASH]]) +@lwt_aliasee = private unnamed_addr global [1 x i8*] [i8* null], comdat($lwt) + +; CHECK: @lwt.llvm.[[HASH]] = hidden unnamed_addr alias +@lwt = internal unnamed_addr alias [1 x i8*], [1 x i8*]* @lwt_aliasee + +; Below function should get imported into other module, resulting in @lwt being +; promoted and renamed. +define i8* @lwt_fun() { + %1 = getelementptr inbounds [1 x i8*], [1 x i8*]* @lwt, i32 0, i32 0 + %2 = load i8*, i8** %1 + ret i8* %2 +} Index: vendor/llvm/dist-release_80/test/Transforms/LoopTransformWarning/enable_and_isvectorized.ll =================================================================== --- vendor/llvm/dist-release_80/test/Transforms/LoopTransformWarning/enable_and_isvectorized.ll (nonexistent) +++ vendor/llvm/dist-release_80/test/Transforms/LoopTransformWarning/enable_and_isvectorized.ll (revision 343794) @@ -0,0 +1,33 @@ +; RUN: opt -transform-warning -disable-output < %s 2>&1 | FileCheck -allow-empty %s +; +; llvm.org/PR40546 +; Do not warn about about leftover llvm.loop.vectorize.enable for already +; vectorized loops. + +target datalayout = "e-m:w-i64:64-f80:128-n8:16:32:64-S128" + +define void @test(i32 %n) { +entry: + %cmp = icmp eq i32 %n, 0 + br i1 %cmp, label %simd.if.end, label %omp.inner.for.body.preheader + +omp.inner.for.body.preheader: + %wide.trip.count = zext i32 %n to i64 + br label %omp.inner.for.body + +omp.inner.for.body: + %indvars.iv = phi i64 [ 0, %omp.inner.for.body.preheader ], [ %indvars.iv.next, %omp.inner.for.body ] + %indvars.iv.next = add nuw nsw i64 %indvars.iv, 1 + %exitcond = icmp eq i64 %indvars.iv.next, %wide.trip.count + br i1 %exitcond, label %simd.if.end, label %omp.inner.for.body, !llvm.loop !0 + +simd.if.end: + ret void +} + +!0 = distinct !{!0, !1, !2} +!1 = !{!"llvm.loop.vectorize.enable", i1 true} +!2 = !{!"llvm.loop.isvectorized"} + + +; CHECK-NOT: loop not vectorized Index: vendor/llvm/dist-release_80/test/Transforms/LoopVectorize/no_switch_disable_vectorization.ll =================================================================== --- vendor/llvm/dist-release_80/test/Transforms/LoopVectorize/no_switch_disable_vectorization.ll (nonexistent) +++ vendor/llvm/dist-release_80/test/Transforms/LoopVectorize/no_switch_disable_vectorization.ll (revision 343794) @@ -0,0 +1,95 @@ +; RUN: opt < %s -loop-vectorize -force-vector-width=4 -transform-warning -S 2>&1 | FileCheck %s +; RUN: opt < %s -loop-vectorize -force-vector-width=1 -transform-warning -S 2>&1 | FileCheck %s -check-prefix=NOANALYSIS +; RUN: opt < %s -loop-vectorize -force-vector-width=4 -transform-warning -pass-remarks-missed='loop-vectorize' -S 2>&1 | FileCheck %s -check-prefix=MOREINFO + +; This test is a copy of no_switch.ll, with the "llvm.loop.vectorize.enable" metadata set to false. +; It tests that vectorization is explicitly disabled and no warnings are emitted. + +; CHECK-NOT: remark: source.cpp:4:5: loop not vectorized: loop contains a switch statement +; CHECK-NOT: warning: source.cpp:4:5: loop not vectorized: the optimizer was unable to perform the requested transformation; the transformation might be disabled or specified as part of an unsupported transformation ordering + +; NOANALYSIS-NOT: remark: {{.*}} +; NOANALYSIS-NOT: warning: source.cpp:4:5: loop not vectorized: the optimizer was unable to perform the requested transformation; the transformation might be disabled or specified as part of an unsupported transformation ordering + +; MOREINFO: remark: source.cpp:4:5: loop not vectorized: vectorization is explicitly disabled +; MOREINFO-NOT: warning: source.cpp:4:5: loop not vectorized: the optimizer was unable to perform the requested transformation; the transformation might be disabled or specified as part of an unsupported transformation ordering + +; CHECK: _Z11test_switchPii +; CHECK-NOT: x i32> +; CHECK: ret + +target datalayout = "e-m:o-i64:64-f80:128-n8:16:32:64-S128" + +; Function Attrs: nounwind optsize ssp uwtable +define void @_Z11test_switchPii(i32* nocapture %A, i32 %Length) #0 !dbg !4 { +entry: + %cmp18 = icmp sgt i32 %Length, 0, !dbg !10 + br i1 %cmp18, label %for.body.preheader, label %for.end, !dbg !10, !llvm.loop !12 + +for.body.preheader: ; preds = %entry + br label %for.body, !dbg !14 + +for.body: ; preds = %for.body.preheader, %for.inc + %indvars.iv = phi i64 [ %indvars.iv.next, %for.inc ], [ 0, %for.body.preheader ] + %arrayidx = getelementptr inbounds i32, i32* %A, i64 %indvars.iv, !dbg !14 + %0 = load i32, i32* %arrayidx, align 4, !dbg !14, !tbaa !16 + switch i32 %0, label %for.inc [ + i32 0, label %sw.bb + i32 1, label %sw.bb3 + ], !dbg !14 + +sw.bb: ; preds = %for.body + %1 = trunc i64 %indvars.iv to i32, !dbg !20 + %mul = shl nsw i32 %1, 1, !dbg !20 + br label %for.inc, !dbg !22 + +sw.bb3: ; preds = %for.body + %2 = trunc i64 %indvars.iv to i32, !dbg !23 + store i32 %2, i32* %arrayidx, align 4, !dbg !23, !tbaa !16 + br label %for.inc, !dbg !23 + +for.inc: ; preds = %sw.bb3, %for.body, %sw.bb + %storemerge = phi i32 [ %mul, %sw.bb ], [ 0, %for.body ], [ 0, %sw.bb3 ] + store i32 %storemerge, i32* %arrayidx, align 4, !dbg !20, !tbaa !16 + %indvars.iv.next = add nuw nsw i64 %indvars.iv, 1, !dbg !10 + %lftr.wideiv = trunc i64 %indvars.iv.next to i32, !dbg !10 + %exitcond = icmp eq i32 %lftr.wideiv, %Length, !dbg !10 + br i1 %exitcond, label %for.end.loopexit, label %for.body, !dbg !10, !llvm.loop !12 + +for.end.loopexit: ; preds = %for.inc + br label %for.end + +for.end: ; preds = %for.end.loopexit, %entry + ret void, !dbg !24 +} + +attributes #0 = { nounwind } + +!llvm.dbg.cu = !{!0} +!llvm.module.flags = !{!7, !8} +!llvm.ident = !{!9} + +!0 = distinct !DICompileUnit(language: DW_LANG_C_plus_plus, producer: "clang version 3.5.0", isOptimized: true, runtimeVersion: 6, emissionKind: LineTablesOnly, file: !1, enums: !2, retainedTypes: !2, globals: !2, imports: !2) +!1 = !DIFile(filename: "source.cpp", directory: ".") +!2 = !{} +!4 = distinct !DISubprogram(name: "test_switch", line: 1, isLocal: false, isDefinition: true, virtualIndex: 6, flags: DIFlagPrototyped, isOptimized: true, unit: !0, scopeLine: 1, file: !1, scope: !5, type: !6, retainedNodes: !2) +!5 = !DIFile(filename: "source.cpp", directory: ".") +!6 = !DISubroutineType(types: !2) +!7 = !{i32 2, !"Dwarf Version", i32 2} +!8 = !{i32 2, !"Debug Info Version", i32 3} +!9 = !{!"clang version 3.5.0"} +!10 = !DILocation(line: 3, column: 8, scope: !11) +!11 = distinct !DILexicalBlock(line: 3, column: 3, file: !1, scope: !4) +!12 = !{!12, !13, !13} +!13 = !{!"llvm.loop.vectorize.enable", i1 false} +!14 = !DILocation(line: 4, column: 5, scope: !15) +!15 = distinct !DILexicalBlock(line: 3, column: 36, file: !1, scope: !11) +!16 = !{!17, !17, i64 0} +!17 = !{!"int", !18, i64 0} +!18 = !{!"omnipotent char", !19, i64 0} +!19 = !{!"Simple C/C++ TBAA"} +!20 = !DILocation(line: 6, column: 7, scope: !21) +!21 = distinct !DILexicalBlock(line: 4, column: 18, file: !1, scope: !15) +!22 = !DILocation(line: 7, column: 5, scope: !21) +!23 = !DILocation(line: 9, column: 7, scope: !21) +!24 = !DILocation(line: 14, column: 1, scope: !4) Index: vendor/llvm/dist-release_80/test/tools/llvm-dwarfdump/X86/debug_loclists_startx_length.s =================================================================== --- vendor/llvm/dist-release_80/test/tools/llvm-dwarfdump/X86/debug_loclists_startx_length.s (revision 343793) +++ vendor/llvm/dist-release_80/test/tools/llvm-dwarfdump/X86/debug_loclists_startx_length.s (revision 343794) @@ -1,27 +1,27 @@ # RUN: llvm-mc %s -filetype obj -triple x86_64-pc-linux -o %t.o # RUN: llvm-dwarfdump -v %t.o | FileCheck %s # DW_LLE_startx_length has different `length` encoding in pre-DWARF 5 # and final DWARF 5 versions. This test checks we are able to parse # the final version which uses ULEB128 and not the U32. # CHECK: .debug_loclists contents: -# CHECK-NEXT: 0x00000000: locations list header: length = 0x0000000f, version = 0x0005, addr_size = 0x08, seg_size = 0x00, offset_entry_count = 0x00000000 +# CHECK-NEXT: 0x00000000: locations list header: length = 0x0000000e, version = 0x0005, addr_size = 0x08, seg_size = 0x00, offset_entry_count = 0x00000000 # CHECK-NEXT: 0x00000000: # CHECK-NEXT: Addr idx 1 (w/ length 16): DW_OP_reg5 RDI .section .debug_loclists,"",@progbits .long .Ldebug_loclist_table_end0-.Ldebug_loclist_table_start0 .Ldebug_loclist_table_start0: .short 5 # Version. .byte 8 # Address size. .byte 0 # Segmen selector size. .long 0 # Offset entry count. .byte 3 # DW_LLE_startx_length .byte 0x01 # Index .uleb128 0x10 # Length - .short 1 # Loc expr size + .byte 1 # Loc expr size .byte 85 # DW_OP_reg5 .byte 0 # DW_LLE_end_of_list .Ldebug_loclist_table_end0: