From dbdef45559c878d1dfc1b559cd497e4bc07d2569 Mon Sep 17 00:00:00 2001 From: Daniel Nagel Date: Mon, 1 Sep 2014 11:37:44 +0200 Subject: [PATCH 001/105] Detect and use SDL2 with help of this Dolphin CMake module: https://github.com/dolphin-emu/dolphin/blob/master/CMakeTests/FindSDL2.cmake --- CMakeLists.txt | 11 +-- CMakeTests/FindSDL2.cmake | 180 ++++++++++++++++++++++++++++++++++++++ 2 files changed, 186 insertions(+), 5 deletions(-) create mode 100644 CMakeTests/FindSDL2.cmake diff --git a/CMakeLists.txt b/CMakeLists.txt index 95d9464074..a78a187e7b 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -2,6 +2,7 @@ cmake_minimum_required(VERSION 2.8.8) project(PPSSPP) enable_language(ASM) +set(CMAKE_MODULE_PATH ${CMAKE_MODULE_PATH} ${CMAKE_SOURCE_DIR}/CMakeTests) add_definitions(-DPPSSPP) @@ -126,7 +127,7 @@ if(MAEMO) endif() if (NOT BLACKBERRY AND NOT ANDROID AND NOT IOS) - include(FindSDL) + include(FindSDL2) endif() include(FindThreads) @@ -729,15 +730,15 @@ elseif(BLACKBERRY) set(nativeExtra ${nativeExtra} native/base/BlackberryMain.cpp native/base/BlackberryDisplay.cpp) set(nativeExtraLibs ${nativeExtraLibs} OpenAL bps screen socket EGL) set(TargetBin PPSSPPBlackberry) -elseif(SDL_FOUND) +elseif(SDL2_FOUND) set(TargetBin PPSSPPSDL) # Require SDL - include_directories(${SDL_INCLUDE_DIR}) + include_directories(${SDL2_INCLUDE_DIR}) set(nativeExtra ${nativeExtra} SDL/SDLJoystick.h SDL/SDLJoystick.cpp native/base/PCMain.cpp) - set(nativeExtraLibs ${nativeExtraLibs} ${SDL_LIBRARY}) + set(nativeExtraLibs ${nativeExtraLibs} ${SDL2_LIBRARY}) if(APPLE) set(nativeExtra ${nativeExtra} SDL/SDLMain.h SDL/SDLMain.mm) set(nativeExtraLibs ${nativeExtraLibs} ${COCOA_LIBRARY}) @@ -746,7 +747,7 @@ elseif(SDL_FOUND) endif() set(TargetBin PPSSPPSDL) else() - message(FATAL_ERROR "Could not find SDL. Failing.") + message(FATAL_ERROR "Could not find SDL2. Failing.") endif() set(NativeAppSource diff --git a/CMakeTests/FindSDL2.cmake b/CMakeTests/FindSDL2.cmake new file mode 100644 index 0000000000..614426cccf --- /dev/null +++ b/CMakeTests/FindSDL2.cmake @@ -0,0 +1,180 @@ +# Locate SDL2 library +# This module defines +# SDL2_LIBRARY, the name of the library to link against +# SDL2_FOUND, if false, do not try to link to SDL2 +# SDL2_INCLUDE_DIR, where to find SDL.h +# +# This module responds to the the flag: +# SDL2_BUILDING_LIBRARY +# If this is defined, then no SDL2_main will be linked in because +# only applications need main(). +# Otherwise, it is assumed you are building an application and this +# module will attempt to locate and set the the proper link flags +# as part of the returned SDL2_LIBRARY variable. +# +# Don't forget to include SDL2main.h and SDL2main.m your project for the +# OS X framework based version. (Other versions link to -lSDL2main which +# this module will try to find on your behalf.) Also for OS X, this +# module will automatically add the -framework Cocoa on your behalf. +# +# +# Additional Note: If you see an empty SDL2_LIBRARY_TEMP in your configuration +# and no SDL2_LIBRARY, it means CMake did not find your SDL2 library +# (SDL2.dll, libsdl2.so, SDL2.framework, etc). +# Set SDL2_LIBRARY_TEMP to point to your SDL2 library, and configure again. +# Similarly, if you see an empty SDL2MAIN_LIBRARY, you should set this value +# as appropriate. These values are used to generate the final SDL2_LIBRARY +# variable, but when these values are unset, SDL2_LIBRARY does not get created. +# +# +# $SDL2DIR is an environment variable that would +# correspond to the ./configure --prefix=$SDL2DIR +# used in building SDL2. +# l.e.galup 9-20-02 +# +# Modified by Eric Wing. +# Added code to assist with automated building by using environmental variables +# and providing a more controlled/consistent search behavior. +# Added new modifications to recognize OS X frameworks and +# additional Unix paths (FreeBSD, etc). +# Also corrected the header search path to follow "proper" SDL2 guidelines. +# Added a search for SDL2main which is needed by some platforms. +# Added a search for threads which is needed by some platforms. +# Added needed compile switches for MinGW. +# +# On OSX, this will prefer the Framework version (if found) over others. +# People will have to manually change the cache values of +# SDL2_LIBRARY to override this selection or set the CMake environment +# CMAKE_INCLUDE_PATH to modify the search paths. +# +# Note that the header path has changed from SDL2/SDL.h to just SDL.h +# This needed to change because "proper" SDL2 convention +# is #include "SDL.h", not . This is done for portability +# reasons because not all systems place things in SDL2/ (see FreeBSD). +# +# Ported by Johnny Patterson. This is a literal port for SDL2 of the FindSDL.cmake +# module with the minor edit of changing "SDL" to "SDL2" where necessary. This +# was not created for redistribution, and exists temporarily pending official +# SDL2 CMake modules. + +#============================================================================= +# Copyright 2003-2009 Kitware, Inc. +# +# Distributed under the OSI-approved BSD License (the "License"); +# see accompanying file Copyright.txt for details. +# +# This software is distributed WITHOUT ANY WARRANTY; without even the +# implied warranty of MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. +# See the License for more information. +#============================================================================= +# (To distribute this file outside of CMake, substitute the full +# License text for the above reference.) + +FIND_PATH(SDL2_INCLUDE_DIR SDL.h + HINTS + $ENV{SDL2DIR} + PATH_SUFFIXES include/SDL2 include + PATHS + ~/Library/Frameworks + /Library/Frameworks + /usr/local/include/SDL2 + /usr/include/SDL2 + /sw # Fink + /opt/local # DarwinPorts + /opt/csw # Blastwave + /opt +) +#MESSAGE("SDL2_INCLUDE_DIR is ${SDL2_INCLUDE_DIR}") + +FIND_LIBRARY(SDL2_LIBRARY_TEMP + NAMES SDL2 + HINTS + $ENV{SDL2DIR} + PATH_SUFFIXES lib64 lib + PATHS + /sw + /opt/local + /opt/csw + /opt +) + +#MESSAGE("SDL2_LIBRARY_TEMP is ${SDL2_LIBRARY_TEMP}") + +IF(NOT SDL2_BUILDING_LIBRARY) + IF(NOT ${SDL2_INCLUDE_DIR} MATCHES ".framework") + # Non-OS X framework versions expect you to also dynamically link to + # SDL2main. This is mainly for Windows and OS X. Other (Unix) platforms + # seem to provide SDL2main for compatibility even though they don't + # necessarily need it. + FIND_LIBRARY(SDL2MAIN_LIBRARY + NAMES SDL2main + HINTS + $ENV{SDL2DIR} + PATH_SUFFIXES lib64 lib + PATHS + /sw + /opt/local + /opt/csw + /opt + ) + ENDIF(NOT ${SDL2_INCLUDE_DIR} MATCHES ".framework") +ENDIF(NOT SDL2_BUILDING_LIBRARY) + +# SDL2 may require threads on your system. +# The Apple build may not need an explicit flag because one of the +# frameworks may already provide it. +# But for non-OSX systems, I will use the CMake Threads package. +IF(NOT APPLE) + FIND_PACKAGE(Threads) +ENDIF(NOT APPLE) + +# MinGW needs an additional library, mwindows +# It's total link flags should look like -lmingw32 -lSDL2main -lSDL2 -lmwindows +# (Actually on second look, I think it only needs one of the m* libraries.) +IF(MINGW) + SET(MINGW32_LIBRARY mingw32 CACHE STRING "mwindows for MinGW") +ENDIF(MINGW) + +SET(SDL2_FOUND "NO") +IF(SDL2_LIBRARY_TEMP) + # For SDL2main + IF(NOT SDL2_BUILDING_LIBRARY) + IF(SDL2MAIN_LIBRARY) + SET(SDL2_LIBRARY_TEMP ${SDL2MAIN_LIBRARY} ${SDL2_LIBRARY_TEMP}) + ENDIF(SDL2MAIN_LIBRARY) + ENDIF(NOT SDL2_BUILDING_LIBRARY) + + # For OS X, SDL2 uses Cocoa as a backend so it must link to Cocoa. + # CMake doesn't display the -framework Cocoa string in the UI even + # though it actually is there if I modify a pre-used variable. + # I think it has something to do with the CACHE STRING. + # So I use a temporary variable until the end so I can set the + # "real" variable in one-shot. + IF(APPLE) + SET(SDL2_LIBRARY_TEMP ${SDL2_LIBRARY_TEMP} "-framework Cocoa") + ENDIF(APPLE) + + # For threads, as mentioned Apple doesn't need this. + # In fact, there seems to be a problem if I used the Threads package + # and try using this line, so I'm just skipping it entirely for OS X. + IF(NOT APPLE) + SET(SDL2_LIBRARY_TEMP ${SDL2_LIBRARY_TEMP} ${CMAKE_THREAD_LIBS_INIT}) + ENDIF(NOT APPLE) + + # For MinGW library + IF(MINGW) + SET(SDL2_LIBRARY_TEMP ${MINGW32_LIBRARY} ${SDL2_LIBRARY_TEMP}) + ENDIF(MINGW) + + # Set the final string here so the GUI reflects the final state. + SET(SDL2_LIBRARY ${SDL2_LIBRARY_TEMP} CACHE STRING "Where the SDL2 Library can be found") + # Set the temp variable to INTERNAL so it is not seen in the CMake GUI + SET(SDL2_LIBRARY_TEMP "${SDL2_LIBRARY_TEMP}" CACHE INTERNAL "") + + SET(SDL2_FOUND "YES") +ENDIF(SDL2_LIBRARY_TEMP) + +INCLUDE(FindPackageHandleStandardArgs) + +FIND_PACKAGE_HANDLE_STANDARD_ARGS(SDL2 + REQUIRED_VARS SDL2_LIBRARY SDL2_INCLUDE_DIR) From 1d096f7d02e78ab6781aa11395621f80cfd91024 Mon Sep 17 00:00:00 2001 From: Daniel Nagel Date: Mon, 1 Sep 2014 15:47:21 +0200 Subject: [PATCH 002/105] Look for SDL_gamecontroller.h instead of SDL.h to prevent CMake from using SDL1 include dirs --- CMakeTests/FindSDL2.cmake | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/CMakeTests/FindSDL2.cmake b/CMakeTests/FindSDL2.cmake index 614426cccf..ec26ad3961 100644 --- a/CMakeTests/FindSDL2.cmake +++ b/CMakeTests/FindSDL2.cmake @@ -70,7 +70,7 @@ # (To distribute this file outside of CMake, substitute the full # License text for the above reference.) -FIND_PATH(SDL2_INCLUDE_DIR SDL.h +FIND_PATH(SDL2_INCLUDE_DIR SDL_gamecontroller.h HINTS $ENV{SDL2DIR} PATH_SUFFIXES include/SDL2 include From afefac86ad85dad8dbf2c25678f6978c620caffc Mon Sep 17 00:00:00 2001 From: Daniel Nagel Date: Mon, 1 Sep 2014 16:35:19 +0200 Subject: [PATCH 003/105] Update to SDL2 --- SDL/SDLJoystick.cpp | 8 ++------ native | 2 +- 2 files changed, 3 insertions(+), 7 deletions(-) diff --git a/SDL/SDLJoystick.cpp b/SDL/SDLJoystick.cpp index b09ddd8ade..441a5dfb6d 100644 --- a/SDL/SDLJoystick.cpp +++ b/SDL/SDLJoystick.cpp @@ -13,11 +13,7 @@ extern "C" { SDLJoystick::SDLJoystick(bool init_SDL ): thread(NULL), running(true) { if (init_SDL) { - SDL_Init(SDL_INIT_JOYSTICK | SDL_INIT_VIDEO -#ifndef _WIN32 - | SDL_INIT_EVENTTHREAD -#endif - ); + SDL_Init(SDL_INIT_JOYSTICK | SDL_INIT_VIDEO); } fillMapping(); @@ -44,7 +40,7 @@ SDLJoystick::~SDLJoystick(){ } void SDLJoystick::startEventLoop(){ - thread = SDL_CreateThread(SDLJoystickThreadWrapper, static_cast(this)); + thread = SDL_CreateThread(SDLJoystickThreadWrapper, "joystick",static_cast(this)); } void SDLJoystick::ProcessInput(SDL_Event &event){ diff --git a/native b/native index 9a72d5323b..d4014f5f7f 160000 --- a/native +++ b/native @@ -1 +1 @@ -Subproject commit 9a72d5323bfafdaf97aec583eefef9e51e0be594 +Subproject commit d4014f5f7f73f3b3774c498ee5a0145752f9a25b From e5778edcacbe8595a5ff5106f3d8627745ce86f1 Mon Sep 17 00:00:00 2001 From: Daniel Nagel Date: Mon, 1 Sep 2014 20:25:00 +0200 Subject: [PATCH 004/105] Update native submodule --- native | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/native b/native index d4014f5f7f..4a85cbe404 160000 --- a/native +++ b/native @@ -1 +1 @@ -Subproject commit d4014f5f7f73f3b3774c498ee5a0145752f9a25b +Subproject commit 4a85cbe4040e4a33484e8b895a4d84ed1e66479d From 199e6bcd3bd57e173d9704c4016a6759203e3b7b Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Tue, 23 Sep 2014 08:31:29 -0700 Subject: [PATCH 005/105] Avoid crashing when calling an invalid address. We already have a check, let's use it properly. --- GPU/GPUCommon.cpp | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/GPU/GPUCommon.cpp b/GPU/GPUCommon.cpp index 7a186900c9..9d4d875e87 100644 --- a/GPU/GPUCommon.cpp +++ b/GPU/GPUCommon.cpp @@ -756,6 +756,10 @@ void GPUCommon::Execute_Call(u32 op, u32 diff) { // Saint Seiya needs correct support for relative calls. const u32 retval = currentList->pc + 4; const u32 target = gstate_c.getRelativeAddress(op & 0x00FFFFFC); + if (!Memory::IsValidAddress(target)) { + ERROR_LOG_REPORT(G3D, "CALL to illegal address %08x - ignoring! data=%06x", target, op & 0x00FFFFFF); + return; + } // Bone matrix optimization - many games will CALL a bone matrix (!). if ((Memory::ReadUnchecked_U32(target) >> 24) == GE_CMD_BONEMATRIXDATA) { @@ -770,8 +774,6 @@ void GPUCommon::Execute_Call(u32 op, u32 diff) { if (currentList->stackptr == ARRAY_SIZE(currentList->stack)) { ERROR_LOG_REPORT(G3D, "CALL: Stack full!"); - } else if (!Memory::IsValidAddress(target)) { - ERROR_LOG_REPORT(G3D, "CALL to illegal address %08x - ignoring! data=%06x", target, op & 0x00FFFFFF); } else { auto &stackEntry = currentList->stack[currentList->stackptr++]; stackEntry.pc = retval; From 70705d4a9df3556a0fcbfc50aaf78986831c81ba Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Sun, 21 Sep 2014 19:59:51 -0700 Subject: [PATCH 006/105] Remove incorrect atrac decode ptr nullcheck. Already shown in decode test to be valid. --- Core/HLE/sceAtrac.cpp | 6 ++---- 1 file changed, 2 insertions(+), 4 deletions(-) diff --git a/Core/HLE/sceAtrac.cpp b/Core/HLE/sceAtrac.cpp index 21392edece..ea330a558d 100644 --- a/Core/HLE/sceAtrac.cpp +++ b/Core/HLE/sceAtrac.cpp @@ -695,10 +695,8 @@ u32 _AtracDecodeData(int atracID, u8* outbuf, u32 *SamplesNum, u32* finish, int u32 sceAtracDecodeData(int atracID, u32 outAddr, u32 numSamplesAddr, u32 finishFlagAddr, u32 remainAddr) { int ret = -1; - if (!Memory::IsValidAddress(outAddr)) { - ERROR_LOG(ME, "%08x=sceAtracDecodeData(%i, %08x, ...): Bad out addr, skipping", ret, atracID, outAddr); - return SCE_KERNEL_ERROR_ILLEGAL_ADDR; - } + + // Note that outAddr being null is completely valid here, used to skip data. u32 numSamples = 0; u32 finish = 0; From ac1fcdb26994eb960e01ebb196a298aba0202d25 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Sun, 21 Sep 2014 19:57:27 -0700 Subject: [PATCH 007/105] Skip samples in the first chunk of atrac output. This seems to be what the PSP actually does, although not sure. The first result is always smaller by this amount (numerous atrac files tested.) --- Core/HLE/sceAtrac.cpp | 85 ++++++++++++++++++++++++++++++++----------- 1 file changed, 64 insertions(+), 21 deletions(-) diff --git a/Core/HLE/sceAtrac.cpp b/Core/HLE/sceAtrac.cpp index ea330a558d..6ed91460c8 100644 --- a/Core/HLE/sceAtrac.cpp +++ b/Core/HLE/sceAtrac.cpp @@ -15,6 +15,7 @@ // Official git repository and contact information can be found at // https://github.com/hrydgard/ppsspp and http://www.ppsspp.org/. +#include #include "Core/HLE/HLE.h" #include "Core/HLE/FunctionWrappers.h" @@ -27,12 +28,22 @@ #include "Core/HW/BufferQueue.h" #include "Common/ChunkFile.h" -#include "sceKernel.h" -#include "sceUtility.h" -#include "sceKernelMemory.h" -#include "sceAtrac.h" +#include "Core/HLE/sceKernel.h" +#include "Core/HLE/sceUtility.h" +#include "Core/HLE/sceKernelMemory.h" +#include "Core/HLE/sceAtrac.h" -#include +// Notes about sceAtrac buffer management +// +// sceAtrac decodes from a buffer the game fills, where this buffer is one of: +// * Not yet initialized (state NO DATA = 1) +// * The entire size of the audio data, and filled with audio data (state ALL DATA LOADED = 2) +// * The entire size, but only partially filled so far (state HALFWAY BUFFER = 3) +// * Smaller than the audio, sliding without any loop (state STREAMED WITHOUT LOOP = 4) +// * Smaller than the audio, sliding with a loop at the end (state STREAMED WITH LOOP AT END = 5) +// * Smaller with a second buffer to help with a loop in the middle (state STREAMED WITH SECOND BUF = 6) +// * Not managed, decoding using "low level" manual looping etc. (LOW LEVEL = 8) +// * Not managed, reseved externally - possibly by sceSas - through low level (RESERVED = 16) #define ATRAC_ERROR_API_FAIL 0x80630002 #define ATRAC_ERROR_NO_ATRACID 0x80630003 @@ -108,7 +119,7 @@ struct AtracLoopInfo { struct Atrac { Atrac() : atracID(-1), data_buf(0), decodePos(0), decodeEnd(0), atracChannels(0), atracOutputChannels(2), atracBitrate(64), atracBytesPerFrame(0), atracBufSize(0), - currentSample(0), endSample(0), firstSampleoffset(0), + currentSample(0), endSample(0), firstSampleoffset(0), dataOff(0), loopinfoNum(0), loopStartSample(-1), loopEndSample(-1), loopNum(0), failedDecode(false), resetBuffer(false), codecType(0) { memset(&first, 0, sizeof(first)); @@ -142,7 +153,7 @@ struct Atrac { } void DoState(PointerWrap &p) { - auto s = p.Section("Atrac", 1 , 2); + auto s = p.Section("Atrac", 1, 3); if (!s) return; @@ -157,6 +168,11 @@ struct Atrac { p.Do(currentSample); p.Do(endSample); p.Do(firstSampleoffset); + if (s >= 3) { + p.Do(dataOff); + } else { + dataOff = firstSampleoffset; + } u32 has_data_buf = data_buf != NULL; p.Do(has_data_buf); @@ -229,8 +245,9 @@ struct Atrac { int currentSample; int endSample; - // Offset of the first sample in the input buffer int firstSampleoffset; + // Offset of the first sample in the input buffer + int dataOff; std::vector loopinfo; int loopinfoNum; @@ -421,7 +438,7 @@ int Atrac::Analyze() { first.filesize = Memory::Read_U32(first.addr + 4) + 8; u32 offset = 12; - int atracSampleoffset = 0; + firstSampleoffset = 0; this->decodeEnd = first.filesize; bool bfoundData = false; @@ -454,7 +471,7 @@ int Atrac::Analyze() { { if (chunkSize >= 8) { endSample = Memory::Read_U32(first.addr + offset); - atracSampleoffset = Memory::Read_U32(first.addr + offset + 4); + firstSampleoffset = Memory::Read_U32(first.addr + offset + 4); } } break; @@ -470,8 +487,8 @@ int Atrac::Analyze() { for (int i = 0; i < loopinfoNum; i++, loopinfoAddr += 24) { loopinfo[i].cuePointID = Memory::Read_U32(loopinfoAddr); loopinfo[i].type = Memory::Read_U32(loopinfoAddr + 4); - loopinfo[i].startSample = Memory::Read_U32(loopinfoAddr + 8) - atracSampleoffset; - loopinfo[i].endSample = Memory::Read_U32(loopinfoAddr + 12) - atracSampleoffset; + loopinfo[i].startSample = Memory::Read_U32(loopinfoAddr + 8) - firstSampleoffset; + loopinfo[i].endSample = Memory::Read_U32(loopinfoAddr + 12) - firstSampleoffset; loopinfo[i].fraction = Memory::Read_U32(loopinfoAddr + 16); loopinfo[i].playCount = Memory::Read_U32(loopinfoAddr + 20); @@ -484,7 +501,7 @@ int Atrac::Analyze() { case DATA_CHUNK_MAGIC: { bfoundData = true; - firstSampleoffset = offset; + dataOff = offset; } break; } @@ -603,6 +620,15 @@ u32 _AtracDecodeData(int atracID, u8* outbuf, u32 *SamplesNum, u32* finish, int // TODO: This isn't at all right, but at least it makes the music "last" some time. u32 numSamples = 0; u32 atracSamplesPerFrame = (atrac->codecType == PSP_MODE_AT_3_PLUS ? ATRAC3PLUS_MAX_SAMPLES : ATRAC3_MAX_SAMPLES); + + int skipSamples = 0; + if (atrac->currentSample == 0) { + // Some kind of header size? + u32 firstOffsetExtra = atrac->codecType == PSP_CODEC_AT3PLUS ? 368 : 69; + // It seems like the PSP aligns the sample position to 0x800...? + skipSamples = atrac->firstSampleoffset + firstOffsetExtra; + } + #ifdef USE_FFMPEG if (!atrac->failedDecode && (atrac->codecType == PSP_MODE_AT_3 || atrac->codecType == PSP_MODE_AT_3_PLUS) && atrac->pCodecCtx) { int forceseekSample = atrac->currentSample * 2 > atrac->endSample ? 0 : atrac->endSample; @@ -643,14 +669,31 @@ u32 _AtracDecodeData(int atracID, u8* outbuf, u32 *SamplesNum, u32* finish, int if (got_frame) { // got a frame // Use a small buffer and keep overwriting it with file data constantly - atrac->first.writableBytes += atrac->atracBytesPerFrame; - int decoded = av_samples_get_buffer_size(NULL, atrac->pFrame->channels, - atrac->pFrame->nb_samples, (AVSampleFormat)atrac->pFrame->format, 1); - u8 *out = outbuf; - if (out != NULL) { - numSamples = atrac->pFrame->nb_samples; - avret = swr_convert(atrac->pSwrCtx, &out, atrac->pFrame->nb_samples, - (const u8 **)atrac->pFrame->extended_data, atrac->pFrame->nb_samples); + atrac->first.writableBytes += atrac->atracBytesPerFrame; + int skipped = std::min(skipSamples, atrac->pFrame->nb_samples); + skipSamples -= skipped; + numSamples = atrac->pFrame->nb_samples - skipped; + atrac->currentSample += skipped; + + if (skipped > 0 && numSamples == 0) { + // Wait for the next one. + got_frame = 0; + } + + if (outbuf != NULL && numSamples != 0) { + int inbufOffset = 0; + if (skipped != 0) { + AVSampleFormat fmt = (AVSampleFormat)atrac->pFrame->format; + // We want the offset per channel. + inbufOffset = av_samples_get_buffer_size(NULL, 1, skipped, fmt, 1); + } + + u8 *out = outbuf; + const u8 *inbuf[2] = { + atrac->pFrame->extended_data[0] + inbufOffset, + atrac->pFrame->extended_data[1] + inbufOffset, + }; + avret = swr_convert(atrac->pSwrCtx, &out, numSamples, inbuf, numSamples); if (avret < 0) { ERROR_LOG(ME, "swr_convert: Error while converting %d", avret); } From 6b6bf3f8e6eecac8e3d73edae0de58eb873ee705 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Sun, 21 Sep 2014 20:11:05 -0700 Subject: [PATCH 008/105] Correct the dataOff member of atrac context. --- Core/HLE/sceAtrac.cpp | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/Core/HLE/sceAtrac.cpp b/Core/HLE/sceAtrac.cpp index 6ed91460c8..3a4b7a61bf 100644 --- a/Core/HLE/sceAtrac.cpp +++ b/Core/HLE/sceAtrac.cpp @@ -1663,7 +1663,7 @@ void _AtracGenarateContext(Atrac *atrac, SceAtracId *context) { context->info.samplesPerChan = (atrac->codecType == PSP_MODE_AT_3_PLUS ? ATRAC3PLUS_MAX_SAMPLES : ATRAC3_MAX_SAMPLES); context->info.sampleSize = atrac->atracBytesPerFrame; context->info.numChan = atrac->atracChannels; - context->info.dataOff = atrac->firstSampleoffset; + context->info.dataOff = atrac->dataOff; context->info.endSample = atrac->endSample; context->info.dataEnd = atrac->first.filesize; context->info.curOff = atrac->first.size; @@ -1815,6 +1815,7 @@ int sceAtracLowLevelInitDecoder(int atracID, u32 paramsAddr) { } atrac->firstSampleoffset = headersize; + atrac->dataOff = headersize; atrac->first.size = headersize; atrac->first.filesize = headersize + atrac->atracBytesPerFrame; atrac->data_buf = new u8[atrac->first.filesize]; @@ -1839,6 +1840,7 @@ int sceAtracLowLevelInitDecoder(int atracID, u32 paramsAddr) { } atrac->firstSampleoffset = headersize; + atrac->dataOff = headersize; atrac->first.size = headersize; atrac->first.filesize = headersize + atrac->atracBytesPerFrame; atrac->data_buf = new u8[atrac->first.filesize]; From 4702ae0e4192cd79bacee2506411268fe8aed609 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Mon, 22 Sep 2014 22:30:30 -0700 Subject: [PATCH 009/105] Add breakpoints to most sceAtrac mem access. --- Core/HLE/sceAtrac.cpp | 24 ++++++++++++++++++------ Core/HLE/sceAtrac.h | 4 ++-- Core/HLE/sceSas.cpp | 2 +- Core/HW/SasAudio.cpp | 6 +++--- Core/HW/SasAudio.h | 2 +- 5 files changed, 25 insertions(+), 13 deletions(-) diff --git a/Core/HLE/sceAtrac.cpp b/Core/HLE/sceAtrac.cpp index 3a4b7a61bf..e33321ca9e 100644 --- a/Core/HLE/sceAtrac.cpp +++ b/Core/HLE/sceAtrac.cpp @@ -24,6 +24,7 @@ #include "Core/MemMap.h" #include "Core/Reporting.h" #include "Core/Config.h" +#include "Core/Debugger/Breakpoints.h" #include "Core/HW/MediaEngine.h" #include "Core/HW/BufferQueue.h" #include "Common/ChunkFile.h" @@ -549,12 +550,13 @@ u32 sceAtracGetAtracID(int codecType) { return atracID; } -u32 _AtracAddStreamData(int atracID, u8 *buf, u32 bytesToAdd) { +u32 _AtracAddStreamData(int atracID, u32 bufPtr, u32 bytesToAdd) { Atrac *atrac = getAtrac(atracID); if (!atrac) return 0; int addbytes = std::min(bytesToAdd, atrac->first.filesize - atrac->first.fileoffset); - memcpy(atrac->data_buf + atrac->first.fileoffset, buf, addbytes); + Memory::Memcpy(atrac->data_buf + atrac->first.fileoffset, bufPtr, addbytes); + CBreakPoints::ExecMemCheck(bufPtr, false, addbytes, currentMIPS->pc); atrac->first.size += bytesToAdd; if (atrac->first.size > atrac->first.filesize) atrac->first.size = atrac->first.filesize; @@ -590,6 +592,7 @@ u32 sceAtracAddStreamData(int atracID, u32 bytesToAdd) { if (bytesToAdd > 0) { int addbytes = std::min(bytesToAdd, atrac->first.filesize - atrac->first.fileoffset); Memory::Memcpy(atrac->data_buf + atrac->first.fileoffset, atrac->first.addr + atrac->first.offset, addbytes); + CBreakPoints::ExecMemCheck(atrac->first.addr + atrac->first.offset, false, addbytes, currentMIPS->pc); } atrac->first.size += bytesToAdd; if (atrac->first.size > atrac->first.filesize) @@ -601,7 +604,7 @@ u32 sceAtracAddStreamData(int atracID, u32 bytesToAdd) { return 0; } -u32 _AtracDecodeData(int atracID, u8* outbuf, u32 *SamplesNum, u32* finish, int *remains) { +u32 _AtracDecodeData(int atracID, u8 *outbuf, u32 outbufPtr, u32 *SamplesNum, u32 *finish, int *remains) { Atrac *atrac = getAtrac(atracID); u32 ret = 0; @@ -694,6 +697,10 @@ u32 _AtracDecodeData(int atracID, u8* outbuf, u32 *SamplesNum, u32* finish, int atrac->pFrame->extended_data[1] + inbufOffset, }; avret = swr_convert(atrac->pSwrCtx, &out, numSamples, inbuf, numSamples); + if (outbufPtr != 0) { + u32 outBytes = numSamples * atrac->atracOutputChannels * sizeof(s16); + CBreakPoints::ExecMemCheck(outbufPtr, true, outBytes, currentMIPS->pc); + } if (avret < 0) { ERROR_LOG(ME, "swr_convert: Error while converting %d", avret); } @@ -744,7 +751,7 @@ u32 sceAtracDecodeData(int atracID, u32 outAddr, u32 numSamplesAddr, u32 finishF u32 numSamples = 0; u32 finish = 0; int remains = 0; - ret = _AtracDecodeData(atracID, Memory::GetPointer(outAddr), &numSamples, &finish, &remains); + ret = _AtracDecodeData(atracID, Memory::GetPointer(outAddr), outAddr, &numSamples, &finish, &remains); if (ret != (int)ATRAC_ERROR_BAD_ATRACID && ret != (int)ATRAC_ERROR_NO_DATA) { if (Memory::IsValidAddress(numSamplesAddr)) Memory::Write_U32(numSamples, numSamplesAddr); @@ -1237,7 +1244,9 @@ int _AtracSetData(Atrac *atrac, u32 buffer, u32 bufferSize) { #ifdef USE_FFMPEG atrac->data_buf = new u8[atrac->first.filesize]; - Memory::Memcpy(atrac->data_buf, buffer, std::min(bufferSize, atrac->first.filesize)); + u32 copybytes = std::min(bufferSize, atrac->first.filesize); + Memory::Memcpy(atrac->data_buf, buffer, copybytes); + CBreakPoints::ExecMemCheck(buffer, false, copybytes, currentMIPS->pc); return __AtracSetContext(atrac); #endif // USE_FFMPEG @@ -1248,7 +1257,9 @@ int _AtracSetData(Atrac *atrac, u32 buffer, u32 bufferSize) { WARN_LOG(ME, "This is an atrac3+ stereo audio"); } atrac->data_buf = new u8[atrac->first.filesize]; - Memory::Memcpy(atrac->data_buf, buffer, std::min(bufferSize, atrac->first.filesize)); + u32 copybytes = std::min(bufferSize, atrac->first.filesize); + Memory::Memcpy(atrac->data_buf, buffer, copybytes); + CBreakPoints::ExecMemCheck(buffer, false, copybytes, currentMIPS->pc); return __AtracSetContext(atrac); } @@ -1868,6 +1879,7 @@ int sceAtracLowLevelDecode(int atracID, u32 sourceAddr, u32 sourceBytesConsumedA u32 sourcebytes = atrac->first.writableBytes; if (sourcebytes > 0) { Memory::Memcpy(atrac->data_buf + atrac->first.size, sourceAddr, sourcebytes); + CBreakPoints::ExecMemCheck(sourceAddr, false, sourcebytes, currentMIPS->pc); if (atrac->decodePos >= atrac->first.size) { atrac->decodePos = atrac->first.size; } diff --git a/Core/HLE/sceAtrac.h b/Core/HLE/sceAtrac.h index 7039a02fe2..aa4d2789c3 100644 --- a/Core/HLE/sceAtrac.h +++ b/Core/HLE/sceAtrac.h @@ -66,6 +66,6 @@ typedef struct // provide some decoder interface -u32 _AtracAddStreamData(int atracID, u8 *buf, u32 bytesToAdd); -u32 _AtracDecodeData(int atracID, u8* outbuf, u32 *SamplesNum, u32* finish, int *remains); +u32 _AtracAddStreamData(int atracID, u32 bufPtr, u32 bytesToAdd); +u32 _AtracDecodeData(int atracID, u8* outbuf, u32 outbufPtr, u32 *SamplesNum, u32* finish, int *remains); int _AtracGetIDByContext(u32 contextAddr); \ No newline at end of file diff --git a/Core/HLE/sceSas.cpp b/Core/HLE/sceSas.cpp index 08a4583bda..af5c39208e 100644 --- a/Core/HLE/sceSas.cpp +++ b/Core/HLE/sceSas.cpp @@ -564,7 +564,7 @@ u32 __sceSasConcatenateATRAC3(u32 core, int voiceNum, u32 atrac3DataAddr, int at DEBUG_LOG_REPORT(SCESAS, "__sceSasConcatenateATRAC3(%08x, %i, %08x, %i)", core, voiceNum, atrac3DataAddr, atrac3DataLength); SasVoice &v = sas->voices[voiceNum]; if (Memory::IsValidAddress(atrac3DataAddr)) - v.atrac3.addStreamData(Memory::GetPointer(atrac3DataAddr), atrac3DataLength); + v.atrac3.addStreamData(atrac3DataAddr, atrac3DataLength); return 0; } diff --git a/Core/HW/SasAudio.cpp b/Core/HW/SasAudio.cpp index 6b9629625f..e42137fe47 100644 --- a/Core/HW/SasAudio.cpp +++ b/Core/HW/SasAudio.cpp @@ -188,7 +188,7 @@ int SasAtrac3::getNextSamples(s16* outbuf, int wantedSamples) { u32 numSamples = 0; int remains = 0; static s16 buf[0x800]; - _AtracDecodeData(atracID, (u8*)buf, &numSamples, &finish, &remains); + _AtracDecodeData(atracID, (u8*)buf, 0, &numSamples, &finish, &remains); if (numSamples > 0) sampleQueue->push((u8*)buf, numSamples * sizeof(s16)); else @@ -198,9 +198,9 @@ int SasAtrac3::getNextSamples(s16* outbuf, int wantedSamples) { return finish; } -int SasAtrac3::addStreamData(u8* buf, u32 addbytes) { +int SasAtrac3::addStreamData(u32 bufPtr, u32 addbytes) { if (atracID > 0) { - _AtracAddStreamData(atracID, buf, addbytes); + _AtracAddStreamData(atracID, bufPtr, addbytes); } return 0; } diff --git a/Core/HW/SasAudio.h b/Core/HW/SasAudio.h index ac8daec92a..aa83623fc3 100644 --- a/Core/HW/SasAudio.h +++ b/Core/HW/SasAudio.h @@ -126,7 +126,7 @@ public: ~SasAtrac3() { if (sampleQueue) delete sampleQueue; } int setContext(u32 context); int getNextSamples(s16* outbuf, int wantedSamples); - int addStreamData(u8* buf, u32 addbytes); + int addStreamData(u32 bufPtr, u32 addbytes); void DoState(PointerWrap &p); private: From 0aa7247feaaec564a4233ff502189a97a229d731 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Mon, 22 Sep 2014 23:18:33 -0700 Subject: [PATCH 010/105] Fix seeking in atrac after the start. Not sure the very start is right though, arg. --- Core/HLE/sceAtrac.cpp | 16 +++++++--------- 1 file changed, 7 insertions(+), 9 deletions(-) diff --git a/Core/HLE/sceAtrac.cpp b/Core/HLE/sceAtrac.cpp index e33321ca9e..4c42b6b3c2 100644 --- a/Core/HLE/sceAtrac.cpp +++ b/Core/HLE/sceAtrac.cpp @@ -624,19 +624,17 @@ u32 _AtracDecodeData(int atracID, u8 *outbuf, u32 outbufPtr, u32 *SamplesNum, u3 u32 numSamples = 0; u32 atracSamplesPerFrame = (atrac->codecType == PSP_MODE_AT_3_PLUS ? ATRAC3PLUS_MAX_SAMPLES : ATRAC3_MAX_SAMPLES); - int skipSamples = 0; - if (atrac->currentSample == 0) { - // Some kind of header size? - u32 firstOffsetExtra = atrac->codecType == PSP_CODEC_AT3PLUS ? 368 : 69; - // It seems like the PSP aligns the sample position to 0x800...? - skipSamples = atrac->firstSampleoffset + firstOffsetExtra; - } + // Some kind of header size? + u32 firstOffsetExtra = atrac->codecType == PSP_CODEC_AT3PLUS ? 368 : 69; + // It seems like the PSP aligns the sample position to 0x800...? + int offsetSamples = atrac->firstSampleoffset + firstOffsetExtra; + int skipSamples = atrac->currentSample == 0 ? offsetSamples : 0; #ifdef USE_FFMPEG if (!atrac->failedDecode && (atrac->codecType == PSP_MODE_AT_3 || atrac->codecType == PSP_MODE_AT_3_PLUS) && atrac->pCodecCtx) { int forceseekSample = atrac->currentSample * 2 > atrac->endSample ? 0 : atrac->endSample; atrac->SeekToSample(forceseekSample); - atrac->SeekToSample(atrac->currentSample); + atrac->SeekToSample(atrac->currentSample == 0 ? 0 : atrac->currentSample + offsetSamples); AVPacket packet; av_init_packet(&packet); int got_frame, avret; @@ -652,6 +650,7 @@ u32 _AtracDecodeData(int atracID, u8 *outbuf, u32 outbufPtr, u32 *SamplesNum, u3 if (avret == AVERROR_PATCHWELCOME) { ERROR_LOG(ME, "Unsupported feature in ATRAC audio."); // Let's try the next frame. + // TODO: Or actually, we should return a blank frame and pretend it worked. } else if (avret < 0) { ERROR_LOG(ME, "avcodec_decode_audio4: Error decoding audio %d", avret); av_free_packet(&packet); @@ -676,7 +675,6 @@ u32 _AtracDecodeData(int atracID, u8 *outbuf, u32 outbufPtr, u32 *SamplesNum, u3 int skipped = std::min(skipSamples, atrac->pFrame->nb_samples); skipSamples -= skipped; numSamples = atrac->pFrame->nb_samples - skipped; - atrac->currentSample += skipped; if (skipped > 0 && numSamples == 0) { // Wait for the next one. From 68f4a1e7f7f0826679f2d7afce1f950aca68b644 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Mon, 22 Sep 2014 23:19:03 -0700 Subject: [PATCH 011/105] Return the correct next sample at the beginning. --- Core/HLE/sceAtrac.cpp | 12 ++++++++++-- 1 file changed, 10 insertions(+), 2 deletions(-) diff --git a/Core/HLE/sceAtrac.cpp b/Core/HLE/sceAtrac.cpp index 4c42b6b3c2..c208e1d62e 100644 --- a/Core/HLE/sceAtrac.cpp +++ b/Core/HLE/sceAtrac.cpp @@ -956,10 +956,18 @@ u32 sceAtracGetNextSample(int atracID, u32 outNAddr) { if (atrac->currentSample >= atrac->endSample) { if (Memory::IsValidAddress(outNAddr)) Memory::Write_U32(0, outNAddr); - return ATRAC_ERROR_ALL_DATA_DECODED; + return 0; } else { - u32 numSamples = atrac->endSample - atrac->currentSample; u32 atracSamplesPerFrame = (atrac->codecType == PSP_MODE_AT_3_PLUS ? ATRAC3PLUS_MAX_SAMPLES : ATRAC3_MAX_SAMPLES); + // Some kind of header size? + u32 firstOffsetExtra = atrac->codecType == PSP_CODEC_AT3PLUS ? 368 : 69; + // It seems like the PSP aligns the sample position to 0x800...? + u32 skipSamples = atrac->firstSampleoffset + firstOffsetExtra; + u32 firstSamples = (atracSamplesPerFrame - skipSamples) % atracSamplesPerFrame; + u32 numSamples = atrac->endSample - atrac->currentSample; + if (atrac->currentSample == 0) { + numSamples = firstSamples; + } if (numSamples > atracSamplesPerFrame) numSamples = atracSamplesPerFrame; if (Memory::IsValidAddress(outNAddr)) From fa42426d211159e0b413c97848ad2499047cd542 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Mon, 22 Sep 2014 23:21:08 -0700 Subject: [PATCH 012/105] Clamp the final sample count during decode. Some games depend on / expect this, or else they'll let important data get overwritten. --- Core/HLE/sceAtrac.cpp | 3 +++ 1 file changed, 3 insertions(+) diff --git a/Core/HLE/sceAtrac.cpp b/Core/HLE/sceAtrac.cpp index c208e1d62e..2949ef5897 100644 --- a/Core/HLE/sceAtrac.cpp +++ b/Core/HLE/sceAtrac.cpp @@ -676,6 +676,9 @@ u32 _AtracDecodeData(int atracID, u8 *outbuf, u32 outbufPtr, u32 *SamplesNum, u3 skipSamples -= skipped; numSamples = atrac->pFrame->nb_samples - skipped; + // If we're at the end, clamp to samples we want. It always returns a full chunk. + numSamples = std::min((u32)atrac->endSample - (u32)atrac->currentSample, numSamples); + if (skipped > 0 && numSamples == 0) { // Wait for the next one. got_frame = 0; From e717a87f9fd5e6fbbaea3309c32d8fd326f1f9b7 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Mon, 22 Sep 2014 23:21:08 -0700 Subject: [PATCH 013/105] Add extra frames if we run out of atrac data. We could probably insert frames instead for GHA phase shifting, but this will solve other bugs too, I think. --- Core/HLE/sceAtrac.cpp | 11 ++++++++++- 1 file changed, 10 insertions(+), 1 deletion(-) diff --git a/Core/HLE/sceAtrac.cpp b/Core/HLE/sceAtrac.cpp index 2949ef5897..7e4644adbb 100644 --- a/Core/HLE/sceAtrac.cpp +++ b/Core/HLE/sceAtrac.cpp @@ -637,7 +637,7 @@ u32 _AtracDecodeData(int atracID, u8 *outbuf, u32 outbufPtr, u32 *SamplesNum, u3 atrac->SeekToSample(atrac->currentSample == 0 ? 0 : atrac->currentSample + offsetSamples); AVPacket packet; av_init_packet(&packet); - int got_frame, avret; + int got_frame = 0, avret; while (av_read_frame(atrac->pFormatCtx, &packet) >= 0) { if (packet.stream_index != atrac->audio_stream_index) { av_free_packet(&packet); @@ -713,6 +713,15 @@ u32 _AtracDecodeData(int atracID, u8 *outbuf, u32 outbufPtr, u32 *SamplesNum, u3 break; } } + + if (!got_frame && atrac->currentSample < atrac->endSample) { + // Never got a frame. We may have dropped a GHA frame or otherwise have a bug. + // For now, let's try to provide an extra "frame" if possible so games don't infinite loop. + numSamples = std::min((u32)atrac->endSample - (u32)atrac->currentSample, atracSamplesPerFrame); + u32 outBytes = numSamples * atrac->atracOutputChannels * sizeof(s16); + memset(outbuf, 0, outBytes); + CBreakPoints::ExecMemCheck(outbufPtr, true, outBytes, currentMIPS->pc); + } } #endif // USE_FFMPEG From 0ebe5325d4cc749ca5e9a8db264e8cdfe1f37a24 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Tue, 23 Sep 2014 09:08:40 -0700 Subject: [PATCH 014/105] Correct the end from sceAtracGetSoundSample(). I think it's meant to be the last *valid* sample. --- Core/HLE/sceAtrac.cpp | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/Core/HLE/sceAtrac.cpp b/Core/HLE/sceAtrac.cpp index 7e4644adbb..fe0702220c 100644 --- a/Core/HLE/sceAtrac.cpp +++ b/Core/HLE/sceAtrac.cpp @@ -1039,7 +1039,7 @@ u32 sceAtracGetSoundSample(int atracID, u32 outEndSampleAddr, u32 outLoopStartSa } else { DEBUG_LOG(ME, "sceAtracGetSoundSample(%i, %08x, %08x, %08x)", atracID, outEndSampleAddr, outLoopStartSampleAddr, outLoopEndSampleAddr); if (Memory::IsValidAddress(outEndSampleAddr)) - Memory::Write_U32(atrac->endSample, outEndSampleAddr); + Memory::Write_U32(atrac->endSample - 1, outEndSampleAddr); if (Memory::IsValidAddress(outLoopStartSampleAddr)) Memory::Write_U32(atrac->loopStartSample, outLoopStartSampleAddr); if (Memory::IsValidAddress(outLoopEndSampleAddr)) From c88b66b308f9202b455dbbe06bbad20b4dd008da Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Tue, 23 Sep 2014 21:13:47 -0700 Subject: [PATCH 015/105] d3d9: Emulate some logic ops with blending. This makes Brave Story's intro visible. Also add for GLES2/GLES3, but doesn't seem to work on GLES2. --- Core/HLE/sceAtrac.cpp | 2 +- GPU/Directx9/PixelShaderGeneratorDX9.cpp | 41 +++++++++- GPU/Directx9/StateMappingDX9.cpp | 100 ++++++++++++++++++----- GPU/Directx9/TransformPipelineDX9.h | 3 +- GPU/GLES/FragmentShaderGenerator.cpp | 44 +++++++++- GPU/GLES/StateMapping.cpp | 100 ++++++++++++++++++----- GPU/GLES/TransformPipeline.h | 3 +- 7 files changed, 248 insertions(+), 45 deletions(-) diff --git a/Core/HLE/sceAtrac.cpp b/Core/HLE/sceAtrac.cpp index fe0702220c..dbb76ce768 100644 --- a/Core/HLE/sceAtrac.cpp +++ b/Core/HLE/sceAtrac.cpp @@ -44,7 +44,7 @@ // * Smaller than the audio, sliding with a loop at the end (state STREAMED WITH LOOP AT END = 5) // * Smaller with a second buffer to help with a loop in the middle (state STREAMED WITH SECOND BUF = 6) // * Not managed, decoding using "low level" manual looping etc. (LOW LEVEL = 8) -// * Not managed, reseved externally - possibly by sceSas - through low level (RESERVED = 16) +// * Not managed, reserved externally - possibly by sceSas - through low level (RESERVED = 16) #define ATRAC_ERROR_API_FAIL 0x80630002 #define ATRAC_ERROR_NO_ATRACID 0x80630003 diff --git a/GPU/Directx9/PixelShaderGeneratorDX9.cpp b/GPU/Directx9/PixelShaderGeneratorDX9.cpp index 9b25ea5d82..4d7dd5c342 100644 --- a/GPU/Directx9/PixelShaderGeneratorDX9.cpp +++ b/GPU/Directx9/PixelShaderGeneratorDX9.cpp @@ -376,6 +376,33 @@ static bool CanDoubleSrcBlendMode() { } } +enum LogicOpReplaceType { + LOGICOPTYPE_NORMAL, + LOGICOPTYPE_ONE, + LOGICOPTYPE_INVERT, +}; + +static inline LogicOpReplaceType ReplaceLogicOpType() { + if (gstate.isLogicOpEnabled()) { + switch (gstate.getLogicOp()) { + case GE_LOGIC_COPY_INVERTED: + case GE_LOGIC_AND_INVERTED: + case GE_LOGIC_OR_INVERTED: + case GE_LOGIC_NOR: + case GE_LOGIC_NAND: + case GE_LOGIC_EQUIV: + return LOGICOPTYPE_INVERT; + case GE_LOGIC_INVERTED: + return LOGICOPTYPE_ONE; + case GE_LOGIC_SET: + return LOGICOPTYPE_ONE; + default: + return LOGICOPTYPE_NORMAL; + } + } + return LOGICOPTYPE_NORMAL; +} + // Here we must take all the bits of the gstate that determine what the fragment shader will // look like, and concatenate them together into an ID. void ComputeFragmentShaderIDDX9(FragmentShaderIDDX9 *id) { @@ -448,7 +475,8 @@ void ComputeFragmentShaderIDDX9(FragmentShaderIDDX9 *id) { gpuStats.numNonAlphaTestedDraws++; id0 |= (gstate_c.bgraTexture & 1) << 29; - // 30 and 31 are free. + // 2 bits. + id0 |= ReplaceLogicOpType() << 30; // 3 bits. id1 |= replaceBlend << 0; @@ -790,6 +818,17 @@ void GenerateFragmentShaderDX9(char *buffer) { break; } + switch (ReplaceLogicOpType()) { + case LOGICOPTYPE_ONE: + WRITE(p, " v.rgb = float3(1.0, 1.0, 1.0);\n"); + break; + case LOGICOPTYPE_INVERT: + WRITE(p, " v.rgb = float3(1.0, 1.0, 1.0) - v.rgb;\n"); + break; + case LOGICOPTYPE_NORMAL: + break; + } + WRITE(p, " return v;\n"); WRITE(p, "}\n"); } diff --git a/GPU/Directx9/StateMappingDX9.cpp b/GPU/Directx9/StateMappingDX9.cpp index 8c6dd6e964..8d31ded210 100644 --- a/GPU/Directx9/StateMappingDX9.cpp +++ b/GPU/Directx9/StateMappingDX9.cpp @@ -162,15 +162,76 @@ inline void TransformDrawEngineDX9::ResetShaderBlending() { } } -void TransformDrawEngineDX9::ApplyStencilReplaceOnly() { +void TransformDrawEngineDX9::ApplyStencilReplaceAndLogicOp(ReplaceAlphaType replaceAlphaWithStencil) { + StencilValueType stencilType = STENCIL_VALUE_KEEP; + if (replaceAlphaWithStencil == REPLACE_ALPHA_YES) { + stencilType = ReplaceAlphaWithStencilType(); + } + + // Normally, we would add src + 0, but the logic op may have us do differently. + D3DBLEND srcBlend = D3DBLEND_ONE; + D3DBLEND dstBlend = D3DBLEND_ZERO; + D3DBLENDOP blendOp = D3DBLENDOP_ADD; + if (gstate.isLogicOpEnabled()) { + switch (gstate.getLogicOp()) + { + case GE_LOGIC_CLEAR: + srcBlend = D3DBLEND_ZERO; + break; + case GE_LOGIC_AND: + case GE_LOGIC_AND_REVERSE: + WARN_LOG_REPORT_ONCE(d3dLogicOpAnd, G3D, "Unsupported AND logic op: %x", gstate.getLogicOp()); + break; + case GE_LOGIC_COPY: + // This is the same as off. + break; + case GE_LOGIC_COPY_INVERTED: + // Handled in the shader. + break; + case GE_LOGIC_AND_INVERTED: + case GE_LOGIC_NOR: + case GE_LOGIC_NAND: + case GE_LOGIC_EQUIV: + // Handled in the shader. + WARN_LOG_REPORT_ONCE(d3dLogicOpAndInverted, G3D, "Attempted invert for logic op: %x", gstate.getLogicOp()); + break; + case GE_LOGIC_INVERTED: + srcBlend = D3DBLEND_ONE; + dstBlend = D3DBLEND_ONE; + blendOp = D3DBLENDOP_SUBTRACT; + WARN_LOG_REPORT_ONCE(d3dLogicOpInverted, G3D, "Attempted inverse for logic op: %x", gstate.getLogicOp()); + break; + case GE_LOGIC_NOOP: + srcBlend = D3DBLEND_ZERO; + dstBlend = D3DBLEND_ONE; + break; + case GE_LOGIC_XOR: + WARN_LOG_REPORT_ONCE(d3dLogicOpOrXor, G3D, "Unsupported XOR logic op: %x", gstate.getLogicOp()); + break; + case GE_LOGIC_OR: + case GE_LOGIC_OR_INVERTED: + // Inverted in shader. + dstBlend = D3DBLEND_ONE; + WARN_LOG_REPORT_ONCE(d3dLogicOpOr, G3D, "Attempted or for logic op: %x", gstate.getLogicOp()); + break; + case GE_LOGIC_OR_REVERSE: + WARN_LOG_REPORT_ONCE(d3dLogicOpOrReverse, G3D, "Unsupported OR REVERSE logic op: %x", gstate.getLogicOp()); + break; + case GE_LOGIC_SET: + dstBlend = D3DBLEND_ONE; + WARN_LOG_REPORT_ONCE(d3dLogicOpSet, G3D, "Attempted set for logic op: %x", gstate.getLogicOp()); + break; + } + } + // We're not blending, but we may still want to blend for stencil. // This is only useful for INCR/DECR/INVERT. Others can write directly. - switch (ReplaceAlphaWithStencilType()) { + switch (stencilType) { case STENCIL_VALUE_INCR_4: case STENCIL_VALUE_INCR_8: // We'll add the incremented value output by the shader. - dxstate.blendFunc.set(D3DBLEND_ONE, D3DBLEND_ZERO, D3DBLEND_ONE, D3DBLEND_ONE); - dxstate.blendEquation.set(D3DBLENDOP_ADD, D3DBLENDOP_ADD); + dxstate.blendFunc.set(srcBlend, dstBlend, D3DBLEND_ONE, D3DBLEND_ONE); + dxstate.blendEquation.set(blendOp, D3DBLENDOP_ADD); dxstate.blend.enable(); dxstate.blendSeparate.enable(); break; @@ -178,22 +239,29 @@ void TransformDrawEngineDX9::ApplyStencilReplaceOnly() { case STENCIL_VALUE_DECR_4: case STENCIL_VALUE_DECR_8: // We'll subtract the incremented value output by the shader. - dxstate.blendFunc.set(D3DBLEND_ONE, D3DBLEND_ZERO, D3DBLEND_ONE, D3DBLEND_ONE); - dxstate.blendEquation.set(D3DBLENDOP_ADD, D3DBLENDOP_SUBTRACT); + dxstate.blendFunc.set(srcBlend, dstBlend, D3DBLEND_ONE, D3DBLEND_ONE); + dxstate.blendEquation.set(blendOp, D3DBLENDOP_SUBTRACT); dxstate.blend.enable(); dxstate.blendSeparate.enable(); break; case STENCIL_VALUE_INVERT: // The shader will output one, and reverse subtracting will essentially invert. - dxstate.blendFunc.set(D3DBLEND_ONE, D3DBLEND_ZERO, D3DBLEND_ONE, D3DBLEND_ONE); - dxstate.blendEquation.set(D3DBLENDOP_ADD, D3DBLENDOP_REVSUBTRACT); + dxstate.blendFunc.set(srcBlend, dstBlend, D3DBLEND_ONE, D3DBLEND_ONE); + dxstate.blendEquation.set(blendOp, D3DBLENDOP_REVSUBTRACT); dxstate.blend.enable(); dxstate.blendSeparate.enable(); break; default: - dxstate.blend.disable(); + if (srcBlend == D3DBLEND_ONE && dstBlend == D3DBLEND_ZERO && blendOp == D3DBLENDOP_ADD) { + dxstate.blend.disable(); + } else { + dxstate.blendFunc.set(srcBlend, dstBlend, D3DBLEND_ONE, D3DBLEND_ZERO); + dxstate.blendEquation.set(blendOp, D3DBLENDOP_ADD); + dxstate.blend.enable(); + dxstate.blendSeparate.enable(); + } break; } } @@ -206,6 +274,7 @@ void TransformDrawEngineDX9::ApplyBlendState() { // These may clip incorrectly, so we avoid unfortunately. // * Direct3D only has one arbitrary fixed color. We premultiply the other in the shader. // * The written output alpha should actually be the stencil value. Alpha is not written. + // * We try to apply logical operations through blending. // // If we can't apply blending, we make a copy of the framebuffer and do it manually. @@ -220,22 +289,13 @@ void TransformDrawEngineDX9::ApplyBlendState() { case REPLACE_BLEND_NO: ResetShaderBlending(); // We may still want to do something about stencil -> alpha. - if (replaceAlphaWithStencil == REPLACE_ALPHA_YES) { - ApplyStencilReplaceOnly(); - } else { - dxstate.blend.disable(); - } + ApplyStencilReplaceAndLogicOp(replaceAlphaWithStencil); return; case REPLACE_BLEND_COPY_FBO: if (ApplyShaderBlending()) { // We may still want to do something about stencil -> alpha. - if (replaceAlphaWithStencil == REPLACE_ALPHA_YES) { - ApplyStencilReplaceOnly(); - } else { - // None of the below logic is interesting, we're gonna do it entirely in the shader. - dxstate.blend.disable(); - } + ApplyStencilReplaceAndLogicOp(replaceAlphaWithStencil); return; } // Until next time, force it off. diff --git a/GPU/Directx9/TransformPipelineDX9.h b/GPU/Directx9/TransformPipelineDX9.h index 74a2db807a..45d80ceb3a 100644 --- a/GPU/Directx9/TransformPipelineDX9.h +++ b/GPU/Directx9/TransformPipelineDX9.h @@ -25,6 +25,7 @@ #include "GPU/Common/IndexGenerator.h" #include "GPU/Common/VertexDecoderCommon.h" #include "GPU/Common/DrawEngineCommon.h" +#include "GPU/Directx9/PixelShaderGeneratorDX9.h" struct DecVtxFormat; @@ -183,7 +184,7 @@ private: void ApplyDrawState(int prim); void ApplyDrawStateLate(); void ApplyBlendState(); - void ApplyStencilReplaceOnly(); + void ApplyStencilReplaceAndLogicOp(ReplaceAlphaType replaceAlphaWithStencil); bool ApplyShaderBlending(); inline void ResetShaderBlending(); diff --git a/GPU/GLES/FragmentShaderGenerator.cpp b/GPU/GLES/FragmentShaderGenerator.cpp index ab49f23ebe..793c3f58b2 100644 --- a/GPU/GLES/FragmentShaderGenerator.cpp +++ b/GPU/GLES/FragmentShaderGenerator.cpp @@ -354,6 +354,35 @@ ReplaceBlendType ReplaceBlendWithShader() { } } +enum LogicOpReplaceType { + LOGICOPTYPE_NORMAL, + LOGICOPTYPE_ONE, + LOGICOPTYPE_INVERT, +}; + +static inline LogicOpReplaceType ReplaceLogicOpType() { +#if defined(USING_GLES2) + if (gstate.isLogicOpEnabled()) { + switch (gstate.getLogicOp()) { + case GE_LOGIC_COPY_INVERTED: + case GE_LOGIC_AND_INVERTED: + case GE_LOGIC_OR_INVERTED: + case GE_LOGIC_NOR: + case GE_LOGIC_NAND: + case GE_LOGIC_EQUIV: + return LOGICOPTYPE_INVERT; + case GE_LOGIC_INVERTED: + return LOGICOPTYPE_ONE; + case GE_LOGIC_SET: + return LOGICOPTYPE_ONE; + default: + return LOGICOPTYPE_NORMAL; + } + } +#endif + return LOGICOPTYPE_NORMAL; +} + // Here we must take all the bits of the gstate that determine what the fragment shader will // look like, and concatenate them together into an ID. void ComputeFragmentShaderID(FragmentShaderID *id) { @@ -425,7 +454,9 @@ void ComputeFragmentShaderID(FragmentShaderID *id) { else gpuStats.numNonAlphaTestedDraws++; - // 29 - 31 are free. + // 29 is free. + // 2 bits. + id0 |= ReplaceLogicOpType() << 30; // 3 bits. id1 |= replaceBlend << 0; @@ -1007,6 +1038,17 @@ void GenerateFragmentShader(char *buffer) { break; } + switch (ReplaceLogicOpType()) { + case LOGICOPTYPE_ONE: + WRITE(p, " %s.rgb = vec3(1.0, 1.0, 1.0);\n", fragColor0); + break; + case LOGICOPTYPE_INVERT: + WRITE(p, " %s.rgb = vec3(1.0, 1.0, 1.0) - %s.rgb;\n", fragColor0, fragColor0); + break; + case LOGICOPTYPE_NORMAL: + break; + } + #ifdef DEBUG_SHADER if (doTexture) { WRITE(p, " %s = texture2D(tex, v_texcoord.xy);\n", fragColor0); diff --git a/GPU/GLES/StateMapping.cpp b/GPU/GLES/StateMapping.cpp index 1c4c0b214e..8a3f11a758 100644 --- a/GPU/GLES/StateMapping.cpp +++ b/GPU/GLES/StateMapping.cpp @@ -208,35 +208,104 @@ inline void TransformDrawEngine::ResetShaderBlending() { } } -void TransformDrawEngine::ApplyStencilReplaceOnly() { +void TransformDrawEngine::ApplyStencilReplaceAndLogicOp(ReplaceAlphaType replaceAlphaWithStencil) { + StencilValueType stencilType = STENCIL_VALUE_KEEP; + if (replaceAlphaWithStencil == REPLACE_ALPHA_YES) { + stencilType = ReplaceAlphaWithStencilType(); + } + + // Normally, we would add src + 0, but the logic op may have us do differently. + GLenum srcBlend = GL_ONE; + GLenum dstBlend = GL_ZERO; + GLenum blendOp = GL_FUNC_ADD; +#if defined(USING_GLES2) + if (gstate.isLogicOpEnabled()) { + switch (gstate.getLogicOp()) + { + case GE_LOGIC_CLEAR: + srcBlend = GL_ZERO; + break; + case GE_LOGIC_AND: + case GE_LOGIC_AND_REVERSE: + WARN_LOG_REPORT_ONCE(d3dLogicOpAnd, G3D, "Unsupported AND logic op: %x", gstate.getLogicOp()); + break; + case GE_LOGIC_COPY: + // This is the same as off. + break; + case GE_LOGIC_COPY_INVERTED: + // Handled in the shader. + break; + case GE_LOGIC_AND_INVERTED: + case GE_LOGIC_NOR: + case GE_LOGIC_NAND: + case GE_LOGIC_EQUIV: + // Handled in the shader. + WARN_LOG_REPORT_ONCE(d3dLogicOpAndInverted, G3D, "Attempted invert for logic op: %x", gstate.getLogicOp()); + break; + case GE_LOGIC_INVERTED: + srcBlend = GL_ONE; + dstBlend = GL_ONE; + blendOp = GL_FUNC_SUBTRACT; + WARN_LOG_REPORT_ONCE(d3dLogicOpInverted, G3D, "Attempted inverse for logic op: %x", gstate.getLogicOp()); + break; + case GE_LOGIC_NOOP: + srcBlend = GL_ZERO; + dstBlend = GL_ONE; + break; + case GE_LOGIC_XOR: + WARN_LOG_REPORT_ONCE(d3dLogicOpOrXor, G3D, "Unsupported XOR logic op: %x", gstate.getLogicOp()); + break; + case GE_LOGIC_OR: + case GE_LOGIC_OR_INVERTED: + // Inverted in shader. + dstBlend = GL_ONE; + WARN_LOG_REPORT_ONCE(d3dLogicOpOr, G3D, "Attempted or for logic op: %x", gstate.getLogicOp()); + break; + case GE_LOGIC_OR_REVERSE: + WARN_LOG_REPORT_ONCE(d3dLogicOpOrReverse, G3D, "Unsupported OR REVERSE logic op: %x", gstate.getLogicOp()); + break; + case GE_LOGIC_SET: + dstBlend = GL_ONE; + WARN_LOG_REPORT_ONCE(d3dLogicOpSet, G3D, "Attempted set for logic op: %x", gstate.getLogicOp()); + break; + } + } +#endif + // We're not blending, but we may still want to blend for stencil. // This is only useful for INCR/DECR/INVERT. Others can write directly. - switch (ReplaceAlphaWithStencilType()) { + switch (stencilType) { case STENCIL_VALUE_INCR_4: case STENCIL_VALUE_INCR_8: // We'll add the incremented value output by the shader. - glstate.blendFuncSeparate.set(GL_ONE, GL_ZERO, GL_ONE, GL_ONE); - glstate.blendEquationSeparate.set(GL_FUNC_ADD, GL_FUNC_ADD); + glstate.blendFuncSeparate.set(srcBlend, dstBlend, GL_ONE, GL_ONE); + glstate.blendEquationSeparate.set(blendOp, GL_FUNC_ADD); glstate.blend.enable(); break; case STENCIL_VALUE_DECR_4: case STENCIL_VALUE_DECR_8: // We'll subtract the incremented value output by the shader. - glstate.blendFuncSeparate.set(GL_ONE, GL_ZERO, GL_ONE, GL_ONE); - glstate.blendEquationSeparate.set(GL_FUNC_ADD, GL_FUNC_SUBTRACT); + glstate.blendFuncSeparate.set(srcBlend, dstBlend, GL_ONE, GL_ONE); + glstate.blendEquationSeparate.set(blendOp, GL_FUNC_SUBTRACT); glstate.blend.enable(); break; case STENCIL_VALUE_INVERT: // The shader will output one, and reverse subtracting will essentially invert. - glstate.blendFuncSeparate.set(GL_ONE, GL_ZERO, GL_ONE, GL_ONE); - glstate.blendEquationSeparate.set(GL_FUNC_ADD, GL_FUNC_REVERSE_SUBTRACT); + glstate.blendFuncSeparate.set(srcBlend, dstBlend, GL_ONE, GL_ONE); + glstate.blendEquationSeparate.set(blendOp, GL_FUNC_REVERSE_SUBTRACT); glstate.blend.enable(); break; default: - glstate.blend.disable(); + if (srcBlend == GL_ONE && dstBlend == GL_ZERO && blendOp == GL_FUNC_ADD) { + glstate.blend.disable(); + } else { + glstate.blendFuncSeparate.set(srcBlend, dstBlend, GL_ONE, GL_ZERO); + glstate.blendEquationSeparate.set(blendOp, GL_FUNC_ADD); + glstate.blend.enable(); + } break; } } @@ -261,22 +330,13 @@ void TransformDrawEngine::ApplyBlendState() { case REPLACE_BLEND_NO: ResetShaderBlending(); // We may still want to do something about stencil -> alpha. - if (replaceAlphaWithStencil == REPLACE_ALPHA_YES) { - ApplyStencilReplaceOnly(); - } else { - glstate.blend.disable(); - } + ApplyStencilReplaceAndLogicOp(replaceAlphaWithStencil); return; case REPLACE_BLEND_COPY_FBO: if (ApplyShaderBlending()) { // We may still want to do something about stencil -> alpha. - if (replaceAlphaWithStencil == REPLACE_ALPHA_YES) { - ApplyStencilReplaceOnly(); - } else { - // None of the below logic is interesting, we're gonna do it entirely in the shader. - glstate.blend.disable(); - } + ApplyStencilReplaceAndLogicOp(replaceAlphaWithStencil); return; } // Until next time, force it off. diff --git a/GPU/GLES/TransformPipeline.h b/GPU/GLES/TransformPipeline.h index e8bcab3c75..1b9d16bfc5 100644 --- a/GPU/GLES/TransformPipeline.h +++ b/GPU/GLES/TransformPipeline.h @@ -23,6 +23,7 @@ #include "GPU/Common/IndexGenerator.h" #include "GPU/Common/VertexDecoderCommon.h" #include "GPU/Common/DrawEngineCommon.h" +#include "GPU/GLES/FragmentShaderGenerator.h" #include "gfx/gl_common.h" #include "gfx/gl_lost_manager.h" @@ -180,7 +181,7 @@ private: void ApplyDrawState(int prim); void ApplyDrawStateLate(); void ApplyBlendState(); - void ApplyStencilReplaceOnly(); + void ApplyStencilReplaceAndLogicOp(ReplaceAlphaType replaceAlphaWithStencil); bool ApplyShaderBlending(); inline void ResetShaderBlending(); GLuint AllocateBuffer(); From 81592e9cf509a064d0c39214798fb260e2ebfba1 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Tue, 23 Sep 2014 23:51:58 -0700 Subject: [PATCH 016/105] gles: Avoid pow(<= 0, 0) entirely, undefined. Should work on more driver versions. Fixes #6941. --- GPU/GLES/VertexShaderGenerator.cpp | 13 +++++++------ 1 file changed, 7 insertions(+), 6 deletions(-) diff --git a/GPU/GLES/VertexShaderGenerator.cpp b/GPU/GLES/VertexShaderGenerator.cpp index ae444a2ff2..4a02c2262d 100644 --- a/GPU/GLES/VertexShaderGenerator.cpp +++ b/GPU/GLES/VertexShaderGenerator.cpp @@ -521,14 +521,15 @@ void GenerateVertexShader(int prim, u32 vertType, char *buffer, bool useHWTransf bool doSpecular = gstate.isUsingSpecularLight(i); bool poweredDiffuse = gstate.isUsingPoweredDiffuseLight(i); + WRITE(p, " mediump float dot%i = max(dot(toLight, worldnormal), 0.0);\n", i); if (poweredDiffuse) { - WRITE(p, " mediump float dot%i = pow(dot(toLight, worldnormal), u_matspecular.a);\n", i); - // Ugly NaN check. pow(0.0, 0.0) may be undefined, but PSP seems to treat it as 1.0. + // pow(0.0, 0.0) may be undefined, but the PSP seems to treat it as 1.0. // Seen in Tales of the World: Radiant Mythology (#2424.) - WRITE(p, " if (!(dot%i < 1.0) && !(dot%i > 0.0))\n", i, i); + WRITE(p, " if (dot%i == 0.0 && u_matspecular.a == 0.0) {\n", i); WRITE(p, " dot%i = 1.0;\n", i); - } else { - WRITE(p, " mediump float dot%i = dot(toLight, worldnormal);\n", i); + WRITE(p, " } else {\n"); + WRITE(p, " dot%i = pow(dot%i, u_matspecular.a);\n", i, i); + WRITE(p, " }\n"); } const char *timesLightScale = " * lightScale"; @@ -555,7 +556,7 @@ void GenerateVertexShader(int prim, u32 vertType, char *buffer, bool useHWTransf break; } - WRITE(p, " diffuse = (u_lightdiffuse%i * %s) * max(dot%i, 0.0);\n", i, diffuseStr, i); + WRITE(p, " diffuse = (u_lightdiffuse%i * %s) * dot%i;\n", i, diffuseStr, i); if (doSpecular) { WRITE(p, " dot%i = dot(normalize(toLight + vec3(0.0, 0.0, 1.0)), worldnormal);\n", i); WRITE(p, " if (dot%i > 0.0)\n", i); From d1e992736b81bd1b32663396098d80d3c982a352 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Wed, 24 Sep 2014 23:09:09 -0700 Subject: [PATCH 017/105] Simplify the viewport code a bit. --- GPU/Directx9/ShaderManagerDX9.cpp | 2 +- GPU/Directx9/StateMappingDX9.cpp | 26 +++++++++++--------------- GPU/GLES/StateMapping.cpp | 29 ++++++++++++++--------------- 3 files changed, 26 insertions(+), 31 deletions(-) diff --git a/GPU/Directx9/ShaderManagerDX9.cpp b/GPU/Directx9/ShaderManagerDX9.cpp index ce1b4d2eaf..9cda59bba7 100644 --- a/GPU/Directx9/ShaderManagerDX9.cpp +++ b/GPU/Directx9/ShaderManagerDX9.cpp @@ -291,7 +291,7 @@ void ShaderManagerDX9::VSUpdateUniforms(int dirtyUniforms) { memcpy(&flippedMatrix, gstate.projMatrix, 16 * sizeof(float)); const bool invertedY = gstate_c.vpHeight < 0; - if (invertedY) { + if (!invertedY) { flippedMatrix[5] = -flippedMatrix[5]; flippedMatrix[13] = -flippedMatrix[13]; } diff --git a/GPU/Directx9/StateMappingDX9.cpp b/GPU/Directx9/StateMappingDX9.cpp index 8d31ded210..4de3ee9a0f 100644 --- a/GPU/Directx9/StateMappingDX9.cpp +++ b/GPU/Directx9/StateMappingDX9.cpp @@ -708,21 +708,21 @@ void TransformDrawEngineDX9::ApplyDrawState(int prim) { 0.f, 1.f); } else { // These we can turn into a glViewport call, offset by offsetX and offsetY. Math after. - float vpXa = getFloat24(gstate.viewportx1); - float vpXb = getFloat24(gstate.viewportx2); - float vpYa = getFloat24(gstate.viewporty1); - float vpYb = getFloat24(gstate.viewporty2); + float vpXScale = getFloat24(gstate.viewportx1); + float vpXCenter = getFloat24(gstate.viewportx2); + float vpYScale = getFloat24(gstate.viewporty1); + float vpYCenter = getFloat24(gstate.viewporty2); // The viewport transform appears to go like this: - // Xscreen = -offsetX + vpXb + vpXa * Xview - // Yscreen = -offsetY + vpYb + vpYa * Yview - // Zscreen = vpZb + vpZa * Zview + // Xscreen = -offsetX + vpXCenter + vpXScale * Xview + // Yscreen = -offsetY + vpYCenter + vpYScale * Yview + // Zscreen = vpZCenter + vpZScale * Zview // This means that to get the analogue glViewport we must: - float vpX0 = vpXb - offsetX - vpXa; - float vpY0 = vpYb - offsetY + vpYa; // Need to account for sign of Y - gstate_c.vpWidth = vpXa * 2.0f; - gstate_c.vpHeight = -vpYa * 2.0f; + float vpX0 = vpXCenter - offsetX - fabsf(vpXScale); + float vpY0 = vpYCenter - offsetY - fabsf(vpYScale); // Need to account for sign of Y + gstate_c.vpWidth = vpXScale * 2.0f; + gstate_c.vpHeight = vpYScale * 2.0f; float vpWidth = fabsf(gstate_c.vpWidth); float vpHeight = fabsf(gstate_c.vpHeight); @@ -731,10 +731,6 @@ void TransformDrawEngineDX9::ApplyDrawState(int prim) { vpY0 *= renderHeightFactor; vpWidth *= renderWidthFactor; vpHeight *= renderHeightFactor; - - vpX0 = (vpXb - offsetX - fabsf(vpXa)) * renderWidthFactor; - // Flip vpY0 to match the OpenGL coordinate system. - vpY0 = (framebufferManager_->GetTargetHeight() - (vpYb - offsetY + fabsf(vpYa))) * renderHeightFactor; // shaderManager_->DirtyUniform(DIRTY_PROJMATRIX); diff --git a/GPU/GLES/StateMapping.cpp b/GPU/GLES/StateMapping.cpp index 8a3f11a758..4542266302 100644 --- a/GPU/GLES/StateMapping.cpp +++ b/GPU/GLES/StateMapping.cpp @@ -772,21 +772,21 @@ void TransformDrawEngine::ApplyDrawState(int prim) { glstate.depthRange.set(0.0f, 1.0f); } else { // These we can turn into a glViewport call, offset by offsetX and offsetY. Math after. - float vpXa = getFloat24(gstate.viewportx1); - float vpXb = getFloat24(gstate.viewportx2); - float vpYa = getFloat24(gstate.viewporty1); - float vpYb = getFloat24(gstate.viewporty2); + float vpXScale = getFloat24(gstate.viewportx1); + float vpXCenter = getFloat24(gstate.viewportx2); + float vpYScale = getFloat24(gstate.viewporty1); + float vpYCenter = getFloat24(gstate.viewporty2); - // The viewport transform appears to go like this: - // Xscreen = -offsetX + vpXb + vpXa * Xview - // Yscreen = -offsetY + vpYb + vpYa * Yview - // Zscreen = vpZb + vpZa * Zview + // The viewport transform appears to go like this: + // Xscreen = -offsetX + vpXCenter + vpXScale * Xview + // Yscreen = -offsetY + vpYCenter + vpYScale * Yview + // Zscreen = vpZCenter + vpZScale * Zview // This means that to get the analogue glViewport we must: - float vpX0 = vpXb - offsetX - vpXa; - float vpY0 = vpYb - offsetY + vpYa; // Need to account for sign of Y - gstate_c.vpWidth = vpXa * 2.0f; - gstate_c.vpHeight = -vpYa * 2.0f; + float vpX0 = vpXCenter - offsetX - fabsf(vpXScale); + float vpY0 = vpYCenter - offsetY + fabsf(vpYScale); + gstate_c.vpWidth = vpXScale * 2.0f; + gstate_c.vpHeight = -vpYScale * 2.0f; float vpWidth = fabsf(gstate_c.vpWidth); float vpHeight = fabsf(gstate_c.vpHeight); @@ -796,10 +796,9 @@ void TransformDrawEngine::ApplyDrawState(int prim) { vpWidth *= renderWidthFactor; vpHeight *= renderHeightFactor; - vpX0 = (vpXb - offsetX - fabsf(vpXa)) * renderWidthFactor; // Flip vpY0 to match the OpenGL coordinate system. - vpY0 = renderHeight - (vpYb - offsetY + fabsf(vpYa)) * renderHeightFactor; - + vpY0 = renderHeight - vpY0; + glstate.viewport.set(vpX0 + renderX, vpY0 + renderY, vpWidth, vpHeight); // Sadly, as glViewport takes integers, we will not be able to support sub pixel offsets this way. But meh. // shaderManager_->DirtyUniform(DIRTY_PROJMATRIX); From cee2827172d06d18b09e6482703a9d86d88cac39 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Wed, 24 Sep 2014 23:10:13 -0700 Subject: [PATCH 018/105] Normalize newlines, no code changes. We really shouldn't let mixed newlines creep into the codebase. They're annoying. --- GPU/Common/DrawEngineCommon.cpp | 206 ++++++++++++++++---------------- GPU/Common/DrawEngineCommon.h | 4 +- 2 files changed, 105 insertions(+), 105 deletions(-) diff --git a/GPU/Common/DrawEngineCommon.cpp b/GPU/Common/DrawEngineCommon.cpp index c4ebf1dbe7..982a9c5246 100644 --- a/GPU/Common/DrawEngineCommon.cpp +++ b/GPU/Common/DrawEngineCommon.cpp @@ -238,106 +238,106 @@ bool DrawEngineCommon::GetCurrentSimpleVertices(int count, std::vectorDecodeVerts(bufPtr, inPtr, lowerBound, upperBound); - - // OK, morphing eliminated but bones still remain to be taken care of. - // Let's do a partial software transform where we only do skinning. - - VertexReader reader(bufPtr, dec->GetDecVtxFmt(), vertType); - - SimpleVertex *sverts = (SimpleVertex *)outPtr; - - const u8 defaultColor[4] = { - (u8)gstate.getMaterialAmbientR(), - (u8)gstate.getMaterialAmbientG(), - (u8)gstate.getMaterialAmbientB(), - (u8)gstate.getMaterialAmbientA(), - }; - - // Let's have two separate loops, one for non skinning and one for skinning. - if (!g_Config.bSoftwareSkinning && (vertType & GE_VTYPE_WEIGHT_MASK) != GE_VTYPE_WEIGHT_NONE) { - int numBoneWeights = vertTypeGetNumBoneWeights(vertType); - for (int i = lowerBound; i <= upperBound; i++) { - reader.Goto(i); - SimpleVertex &sv = sverts[i]; - if (vertType & GE_VTYPE_TC_MASK) { - reader.ReadUV(sv.uv); - } - - if (vertType & GE_VTYPE_COL_MASK) { - reader.ReadColor0_8888(sv.color); - } else { - memcpy(sv.color, defaultColor, 4); - } - - float nrm[3], pos[3]; - float bnrm[3], bpos[3]; - - if (vertType & GE_VTYPE_NRM_MASK) { - // Normals are generated during tesselation anyway, not sure if any need to supply - reader.ReadNrm(nrm); - } else { - nrm[0] = 0; - nrm[1] = 0; - nrm[2] = 1.0f; - } - reader.ReadPos(pos); - - // Apply skinning transform directly - float weights[8]; - reader.ReadWeights(weights); - // Skinning - Vec3Packedf psum(0, 0, 0); - Vec3Packedf nsum(0, 0, 0); - for (int w = 0; w < numBoneWeights; w++) { - if (weights[w] != 0.0f) { - Vec3ByMatrix43(bpos, pos, gstate.boneMatrix + w * 12); - Vec3Packedf tpos(bpos); - psum += tpos * weights[w]; - - Norm3ByMatrix43(bnrm, nrm, gstate.boneMatrix + w * 12); - Vec3Packedf tnorm(bnrm); - nsum += tnorm * weights[w]; - } - } - sv.pos = psum; - sv.nrm = nsum; - } - } else { - for (int i = lowerBound; i <= upperBound; i++) { - reader.Goto(i); - SimpleVertex &sv = sverts[i]; - if (vertType & GE_VTYPE_TC_MASK) { - reader.ReadUV(sv.uv); - } else { - sv.uv[0] = 0; // This will get filled in during tesselation - sv.uv[1] = 0; - } - if (vertType & GE_VTYPE_COL_MASK) { - reader.ReadColor0_8888(sv.color); - } else { - memcpy(sv.color, defaultColor, 4); - } - if (vertType & GE_VTYPE_NRM_MASK) { - // Normals are generated during tesselation anyway, not sure if any need to supply - reader.ReadNrm((float *)&sv.nrm); - } else { - sv.nrm.x = 0; - sv.nrm.y = 0; - sv.nrm.z = 1.0f; - } - reader.ReadPos((float *)&sv.pos); - } - } - - // Okay, there we are! Return the new type (but keep the index bits) - return GE_VTYPE_TC_FLOAT | GE_VTYPE_COL_8888 | GE_VTYPE_NRM_FLOAT | GE_VTYPE_POS_FLOAT | (vertType & (GE_VTYPE_IDX_MASK | GE_VTYPE_THROUGH)); -} + +// This normalizes a set of vertices in any format to SimpleVertex format, by processing away morphing AND skinning. +// The rest of the transform pipeline like lighting will go as normal, either hardware or software. +// The implementation is initially a bit inefficient but shouldn't be a big deal. +// An intermediate buffer of not-easy-to-predict size is stored at bufPtr. +u32 DrawEngineCommon::NormalizeVertices(u8 *outPtr, u8 *bufPtr, const u8 *inPtr, VertexDecoder *dec, int lowerBound, int upperBound, u32 vertType) { + // First, decode the vertices into a GPU compatible format. This step can be eliminated but will need a separate + // implementation of the vertex decoder. + dec->DecodeVerts(bufPtr, inPtr, lowerBound, upperBound); + + // OK, morphing eliminated but bones still remain to be taken care of. + // Let's do a partial software transform where we only do skinning. + + VertexReader reader(bufPtr, dec->GetDecVtxFmt(), vertType); + + SimpleVertex *sverts = (SimpleVertex *)outPtr; + + const u8 defaultColor[4] = { + (u8)gstate.getMaterialAmbientR(), + (u8)gstate.getMaterialAmbientG(), + (u8)gstate.getMaterialAmbientB(), + (u8)gstate.getMaterialAmbientA(), + }; + + // Let's have two separate loops, one for non skinning and one for skinning. + if (!g_Config.bSoftwareSkinning && (vertType & GE_VTYPE_WEIGHT_MASK) != GE_VTYPE_WEIGHT_NONE) { + int numBoneWeights = vertTypeGetNumBoneWeights(vertType); + for (int i = lowerBound; i <= upperBound; i++) { + reader.Goto(i); + SimpleVertex &sv = sverts[i]; + if (vertType & GE_VTYPE_TC_MASK) { + reader.ReadUV(sv.uv); + } + + if (vertType & GE_VTYPE_COL_MASK) { + reader.ReadColor0_8888(sv.color); + } else { + memcpy(sv.color, defaultColor, 4); + } + + float nrm[3], pos[3]; + float bnrm[3], bpos[3]; + + if (vertType & GE_VTYPE_NRM_MASK) { + // Normals are generated during tesselation anyway, not sure if any need to supply + reader.ReadNrm(nrm); + } else { + nrm[0] = 0; + nrm[1] = 0; + nrm[2] = 1.0f; + } + reader.ReadPos(pos); + + // Apply skinning transform directly + float weights[8]; + reader.ReadWeights(weights); + // Skinning + Vec3Packedf psum(0, 0, 0); + Vec3Packedf nsum(0, 0, 0); + for (int w = 0; w < numBoneWeights; w++) { + if (weights[w] != 0.0f) { + Vec3ByMatrix43(bpos, pos, gstate.boneMatrix + w * 12); + Vec3Packedf tpos(bpos); + psum += tpos * weights[w]; + + Norm3ByMatrix43(bnrm, nrm, gstate.boneMatrix + w * 12); + Vec3Packedf tnorm(bnrm); + nsum += tnorm * weights[w]; + } + } + sv.pos = psum; + sv.nrm = nsum; + } + } else { + for (int i = lowerBound; i <= upperBound; i++) { + reader.Goto(i); + SimpleVertex &sv = sverts[i]; + if (vertType & GE_VTYPE_TC_MASK) { + reader.ReadUV(sv.uv); + } else { + sv.uv[0] = 0; // This will get filled in during tesselation + sv.uv[1] = 0; + } + if (vertType & GE_VTYPE_COL_MASK) { + reader.ReadColor0_8888(sv.color); + } else { + memcpy(sv.color, defaultColor, 4); + } + if (vertType & GE_VTYPE_NRM_MASK) { + // Normals are generated during tesselation anyway, not sure if any need to supply + reader.ReadNrm((float *)&sv.nrm); + } else { + sv.nrm.x = 0; + sv.nrm.y = 0; + sv.nrm.z = 1.0f; + } + reader.ReadPos((float *)&sv.pos); + } + } + + // Okay, there we are! Return the new type (but keep the index bits) + return GE_VTYPE_TC_FLOAT | GE_VTYPE_COL_8888 | GE_VTYPE_NRM_FLOAT | GE_VTYPE_POS_FLOAT | (vertType & (GE_VTYPE_IDX_MASK | GE_VTYPE_THROUGH)); +} diff --git a/GPU/Common/DrawEngineCommon.h b/GPU/Common/DrawEngineCommon.h index 19ab1cc333..28cd21b15f 100644 --- a/GPU/Common/DrawEngineCommon.h +++ b/GPU/Common/DrawEngineCommon.h @@ -35,8 +35,8 @@ public: virtual u32 NormalizeVertices(u8 *outPtr, u8 *bufPtr, const u8 *inPtr, int lowerBound, int upperBound, u32 vertType) = 0; bool GetCurrentSimpleVertices(int count, std::vector &vertices, std::vector &indices); - - static u32 NormalizeVertices(u8 *outPtr, u8 *bufPtr, const u8 *inPtr, VertexDecoder *dec, int lowerBound, int upperBound, u32 vertType); + + static u32 NormalizeVertices(u8 *outPtr, u8 *bufPtr, const u8 *inPtr, VertexDecoder *dec, int lowerBound, int upperBound, u32 vertType); protected: // Vertex collector buffers From 67a54504c782f980f5facead09e17f1bdbc0cea3 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Wed, 24 Sep 2014 23:11:47 -0700 Subject: [PATCH 019/105] Oops, left the comment in the wrong one. --- GPU/Directx9/StateMappingDX9.cpp | 2 +- GPU/GLES/StateMapping.cpp | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/GPU/Directx9/StateMappingDX9.cpp b/GPU/Directx9/StateMappingDX9.cpp index 4de3ee9a0f..c4523cc678 100644 --- a/GPU/Directx9/StateMappingDX9.cpp +++ b/GPU/Directx9/StateMappingDX9.cpp @@ -720,7 +720,7 @@ void TransformDrawEngineDX9::ApplyDrawState(int prim) { // This means that to get the analogue glViewport we must: float vpX0 = vpXCenter - offsetX - fabsf(vpXScale); - float vpY0 = vpYCenter - offsetY - fabsf(vpYScale); // Need to account for sign of Y + float vpY0 = vpYCenter - offsetY - fabsf(vpYScale); gstate_c.vpWidth = vpXScale * 2.0f; gstate_c.vpHeight = vpYScale * 2.0f; diff --git a/GPU/GLES/StateMapping.cpp b/GPU/GLES/StateMapping.cpp index 4542266302..a826854cc5 100644 --- a/GPU/GLES/StateMapping.cpp +++ b/GPU/GLES/StateMapping.cpp @@ -784,7 +784,7 @@ void TransformDrawEngine::ApplyDrawState(int prim) { // This means that to get the analogue glViewport we must: float vpX0 = vpXCenter - offsetX - fabsf(vpXScale); - float vpY0 = vpYCenter - offsetY + fabsf(vpYScale); + float vpY0 = vpYCenter - offsetY + fabsf(vpYScale); // Need to account for sign of Y gstate_c.vpWidth = vpXScale * 2.0f; gstate_c.vpHeight = -vpYScale * 2.0f; From ea9a0182e43e3c1a33ea26d73b30fc4ebc29eea3 Mon Sep 17 00:00:00 2001 From: daniel229 Date: Fri, 26 Sep 2014 16:55:37 +0800 Subject: [PATCH 020/105] Replace download frame in Sora no kiseki FC --- Core/HLE/ReplaceTables.cpp | 10 ++++++++++ Core/MIPS/MIPSAnalyst.cpp | 1 + 2 files changed, 11 insertions(+) diff --git a/Core/HLE/ReplaceTables.cpp b/Core/HLE/ReplaceTables.cpp index 6c435a9aa8..eb2bba9b5e 100644 --- a/Core/HLE/ReplaceTables.cpp +++ b/Core/HLE/ReplaceTables.cpp @@ -690,6 +690,15 @@ static int Hook_kagaku_no_ensemble_download_frame() { return 0; } +static int Hook_soranokiseki_fc_download_frame() { + const u32 fb_address = currentMIPS->r[MIPS_REG_A2]; + if (Memory::IsVRAMAddress(fb_address)) { + gpu->PerformMemoryDownload(fb_address, 0x00044000); + CBreakPoints::ExecMemCheck(fb_address, true, 0x00044000, currentMIPS->pc); + } + return 0; +} + // Can either replace with C functions or functions emitted in Asm/ArmAsm. static const ReplacementTableEntry entries[] = { // TODO: I think some games can be helped quite a bit by implementing the @@ -747,6 +756,7 @@ static const ReplacementTableEntry entries[] = { { "suikoden1_and_2_download_frame_2", &Hook_suikoden1_and_2_download_frame_2, 0, REPFLAG_HOOKENTER, 0x48 }, { "rezel_cross_download_frame", &Hook_rezel_cross_download_frame, 0, REPFLAG_HOOKENTER, 0x54 }, { "kagaku_no_ensemble_download_frame", &Hook_kagaku_no_ensemble_download_frame, 0, REPFLAG_HOOKENTER, 0x38 }, + { "soranokiseki_fc_download_frame", &Hook_soranokiseki_fc_download_frame, 0, REPFLAG_HOOKENTER, 0x180 }, {} }; diff --git a/Core/MIPS/MIPSAnalyst.cpp b/Core/MIPS/MIPSAnalyst.cpp index e176219cdf..839108f3d4 100644 --- a/Core/MIPS/MIPSAnalyst.cpp +++ b/Core/MIPS/MIPSAnalyst.cpp @@ -344,6 +344,7 @@ static const HardHashTableEntry hardcodedHashes[] = { { 0xb8bd1f0e02e9ad87, 156, "dl_write_light_dir", }, { 0xb8cfaeebfeb2de20, 7548, "_vfprintf_r", }, { 0xb97f352e85661af6, 32, "finitef", }, + { 0xba76a8e853426baa, 544, "soranokiseki_fc_download_frame", }, // Sora no kiseki FC { 0xbb3c6592ed319ba4, 132, "dl_write_fog_params", }, { 0xbb7d7c93e4c08577, 124, "__truncdfsf2", }, { 0xbdf54d66079afb96, 200, "dl_write_bone_matrix_3", }, From 4de7e893309406eb627ad010cc45407ed43373e2 Mon Sep 17 00:00:00 2001 From: daniel229 Date: Fri, 26 Sep 2014 17:13:01 +0800 Subject: [PATCH 021/105] Replace download frame in Sora no kiseki SC,and a comment for Sora no kiseki 3rd --- Core/HLE/ReplaceTables.cpp | 22 ++++++++++++++++++++++ Core/MIPS/MIPSAnalyst.cpp | 3 ++- 2 files changed, 24 insertions(+), 1 deletion(-) diff --git a/Core/HLE/ReplaceTables.cpp b/Core/HLE/ReplaceTables.cpp index eb2bba9b5e..873d17fbd5 100644 --- a/Core/HLE/ReplaceTables.cpp +++ b/Core/HLE/ReplaceTables.cpp @@ -699,6 +699,27 @@ static int Hook_soranokiseki_fc_download_frame() { return 0; } +static int Hook_soranokiseki_sc_download_frame() { + u32 fb_infoaddr; + if (!GetMIPSStaticAddress(fb_infoaddr, 0x28, 0x2C)) { + return 0; + } + const u32 fb_info = Memory::Read_U32(fb_infoaddr); + const MIPSOpcode fb_index_load = Memory::Read_Instruction(currentMIPS->pc + 0x34, true); + if (fb_index_load != MIPS_MAKE_LW(MIPS_GET_RT(fb_index_load), MIPS_GET_RS(fb_index_load), fb_index_load & 0xffff)) { + return 0; + } + const int fb_index_offset = (s16)(fb_index_load & 0xffff); + const u32 fb_index = (Memory::Read_U32(fb_info + fb_index_offset) + 1) & 1; + const u32 fb_address = 0x4000000 + (0x44000 * fb_index); + const u32 dest_address = currentMIPS->r[MIPS_REG_A1]; + if (Memory::IsRAMAddress(dest_address)) { + gpu->PerformMemoryDownload(fb_address, 0x00044000); + CBreakPoints::ExecMemCheck(fb_address, true, 0x00044000, currentMIPS->pc); + } + return 0; +} + // Can either replace with C functions or functions emitted in Asm/ArmAsm. static const ReplacementTableEntry entries[] = { // TODO: I think some games can be helped quite a bit by implementing the @@ -757,6 +778,7 @@ static const ReplacementTableEntry entries[] = { { "rezel_cross_download_frame", &Hook_rezel_cross_download_frame, 0, REPFLAG_HOOKENTER, 0x54 }, { "kagaku_no_ensemble_download_frame", &Hook_kagaku_no_ensemble_download_frame, 0, REPFLAG_HOOKENTER, 0x38 }, { "soranokiseki_fc_download_frame", &Hook_soranokiseki_fc_download_frame, 0, REPFLAG_HOOKENTER, 0x180 }, + { "soranokiseki_sc_download_frame", &Hook_soranokiseki_sc_download_frame, 0, REPFLAG_HOOKENTER, }, {} }; diff --git a/Core/MIPS/MIPSAnalyst.cpp b/Core/MIPS/MIPSAnalyst.cpp index 839108f3d4..fb90522bc4 100644 --- a/Core/MIPS/MIPSAnalyst.cpp +++ b/Core/MIPS/MIPSAnalyst.cpp @@ -326,7 +326,7 @@ static const HardHashTableEntry hardcodedHashes[] = { { 0xafb2c7e56c04c8e9, 48, "vtfm_q", }, { 0xafc9968e7d246a5e, 1588, "atan", }, { 0xafcb7dfbc4d72588, 44, "vector_transform_3x4", }, - { 0xb07f9d82d79deea9, 536, "brandish_download_frame", }, // Brandish + { 0xb07f9d82d79deea9, 536, "brandish_download_frame", }, // Brandish, and Sora no kiseki 3rd { 0xb0db731f27d3aa1b, 40, "vmax_s", }, { 0xb0ef265e87899f0a, 32, "vector_divide_t_s", }, { 0xb183a37baa12607b, 32, "vscl_t", }, @@ -354,6 +354,7 @@ static const HardHashTableEntry hardcodedHashes[] = { { 0xbfa8c16038b7753d, 868, "sakurasou_download_frame", }, // Sakurasou No Pet Na Kanojo { 0xc062f2545ef5dc39, 1076, "kirameki_school_life_download_frame", },// Kirameki School Life SP,and Boku wa Tomodati ga Sukunai { 0xc0feb88cc04a1dc7, 48, "vector_negate_t", }, + { 0xc1220040b0599a75, 472, "soranokiseki_sc_download_frame", }, // Sora no kiseki SC { 0xc1f34599d0b9146b, 116, "__subdf3", }, { 0xc3089f66ee6f0a24, 464, "growlanser_create_saveicon", }, // Growlanswer IV { 0xc319f0d107dd2f45, 888, "__muldf3", }, From 7b4b0eb0a0a6336acda6556ee3a6b34818009c99 Mon Sep 17 00:00:00 2001 From: chinhodado Date: Fri, 26 Sep 2014 15:55:21 -0400 Subject: [PATCH 022/105] Use E_FAIL instead of -1 and clean up --- GPU/Directx9/helper/global.cpp | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/GPU/Directx9/helper/global.cpp b/GPU/Directx9/helper/global.cpp index c9a4abcc4c..da82f0303e 100644 --- a/GPU/Directx9/helper/global.cpp +++ b/GPU/Directx9/helper/global.cpp @@ -63,7 +63,7 @@ bool CompilePixelShader(const char *code, LPDIRECT3DPIXELSHADER9 *pShader, LPD3D ID3DXBuffer* pShaderCode = NULL; ID3DXBuffer* pErrorMsg = NULL; - HRESULT hr = -1; + HRESULT hr = E_FAIL; // Compile pixel shader. hr = dyn_D3DXCompileShader(code, @@ -103,7 +103,7 @@ bool CompileVertexShader(const char *code, LPDIRECT3DVERTEXSHADER9 *pShader, LPD ID3DXBuffer* pShaderCode = NULL; ID3DXBuffer* pErrorMsg = NULL; - HRESULT hr = -1; + HRESULT hr = E_FAIL; // Compile pixel shader. hr = dyn_D3DXCompileShader(code, @@ -141,7 +141,6 @@ bool CompileVertexShader(const char *code, LPDIRECT3DVERTEXSHADER9 *pShader, LPD void CompileShaders() { std::string errorMsg; - HRESULT hr = -1; if (!CompileVertexShader(vscode, &pFramebufferVertexShader, NULL, errorMsg)) { OutputDebugStringA(errorMsg.c_str()); From b97af10a6d0160fde332f5c32222e46567d728b3 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Fri, 26 Sep 2014 08:40:55 -0700 Subject: [PATCH 023/105] d3d9: Show error when default shaders fail. --- GPU/Directx9/helper/global.cpp | 16 ++++++++++------ GPU/Directx9/helper/global.h | 2 +- Windows/D3D9Base.cpp | 22 +++++++++++++++------- 3 files changed, 26 insertions(+), 14 deletions(-) diff --git a/GPU/Directx9/helper/global.cpp b/GPU/Directx9/helper/global.cpp index da82f0303e..dbe08ff789 100644 --- a/GPU/Directx9/helper/global.cpp +++ b/GPU/Directx9/helper/global.cpp @@ -139,22 +139,25 @@ bool CompileVertexShader(const char *code, LPDIRECT3DVERTEXSHADER9 *pShader, LPD return true; } -void CompileShaders() { - std::string errorMsg; - +bool CompileShaders(std::string &errorMsg) { if (!CompileVertexShader(vscode, &pFramebufferVertexShader, NULL, errorMsg)) { OutputDebugStringA(errorMsg.c_str()); - DebugBreak(); + return false; } if (!CompilePixelShader(pscode, &pFramebufferPixelShader, NULL, errorMsg)) { OutputDebugStringA(errorMsg.c_str()); - DebugBreak(); + if (pFramebufferVertexShader) { + pFramebufferVertexShader->Release(); + } + return false; } pD3Ddevice->CreateVertexDeclaration(VertexElements, &pFramebufferVertexDecl); pD3Ddevice->SetVertexDeclaration(pFramebufferVertexDecl); pD3Ddevice->CreateVertexDeclaration(SoftTransVertexElements, &pSoftVertexDecl); + + return true; } void DestroyShaders() { @@ -207,7 +210,8 @@ void DirectxInit(HWND window) { pD3Ddevice->SetRingBufferParameters( &d3dr ); #endif - CompileShaders(); + std::string errorMessage; + CompileShaders(errorMessage); fbo_init(pD3D); } diff --git a/GPU/Directx9/helper/global.h b/GPU/Directx9/helper/global.h index 6f44f2864b..4eebbca15a 100644 --- a/GPU/Directx9/helper/global.h +++ b/GPU/Directx9/helper/global.h @@ -25,7 +25,7 @@ extern LPDIRECT3DPIXELSHADER9 pFramebufferPixelShader; // Pixel Shader extern IDirect3DVertexDeclaration9* pFramebufferVertexDecl; extern IDirect3DVertexDeclaration9* pSoftVertexDecl; -void CompileShaders(); +bool CompileShaders(std::string &errorMessage); bool CompilePixelShader(const char *code, LPDIRECT3DPIXELSHADER9 *pShader, ID3DXConstantTable **pShaderTable, std::string &errorMessage); bool CompileVertexShader(const char *code, LPDIRECT3DVERTEXSHADER9 *pShader, ID3DXConstantTable **pShaderTable, std::string &errorMessage); void DestroyShaders(); diff --git a/Windows/D3D9Base.cpp b/Windows/D3D9Base.cpp index 069f643f9d..976106f62a 100644 --- a/Windows/D3D9Base.cpp +++ b/Windows/D3D9Base.cpp @@ -153,11 +153,7 @@ bool D3D9_Init(HWND hWnd, bool windowed, std::string *error_message) { if (FAILED(hr)) { *error_message = "Failed to create D3D device"; - if (has9Ex) { - d3dEx->Release(); - } else { - d3d->Release(); - } + d3d->Release(); return false; } @@ -167,7 +163,18 @@ bool D3D9_Init(HWND hWnd, bool windowed, std::string *error_message) { LoadD3DX9Dynamic(); - DX9::CompileShaders(); + if (!DX9::CompileShaders(*error_message)) { + *error_message = "Unable to compile shaders: " + *error_message; + device->EndScene(); + device->Release(); + d3d->Release(); + DX9::pD3Ddevice = nullptr; + DX9::pD3DdeviceEx = nullptr; + device = nullptr; + UnloadD3DXDynamic(); + return false; + } + DX9::fbo_init(d3d); if (deviceEx && IsWin7OrLater()) { @@ -182,12 +189,13 @@ void D3D9_Resize(HWND window) { // TODO! } -void D3D9_Shutdown() { +void D3D9_Shutdown() { DX9::DestroyShaders(); DX9::fbo_shutdown(); device->EndScene(); device->Release(); d3d->Release(); + UnloadD3DXDynamic(); DX9::pD3Ddevice = nullptr; DX9::pD3DdeviceEx = nullptr; device = nullptr; From 358462a7f44fe719ace83177b88d342c0a41c50e Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Fri, 26 Sep 2014 09:06:55 -0700 Subject: [PATCH 024/105] Add a .gitattributes to normalize newlines. No code changes. --- .gitattributes | 14 + Core/HLE/sceAudiocodec.cpp | 424 ++++++++++---------- GPU/Common/DrawEngineCommon.cpp | 686 ++++++++++++++++---------------- GPU/Common/DrawEngineCommon.h | 88 ++-- Windows/D3D9Base.cpp | 406 +++++++++---------- Windows/resource.h | 672 +++++++++++++++---------------- 6 files changed, 1152 insertions(+), 1138 deletions(-) create mode 100644 .gitattributes diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000000..531cac20f9 --- /dev/null +++ b/.gitattributes @@ -0,0 +1,14 @@ +*.cpp text diff=cpp +*.h text diff=cpp +*.mm text diff=objc +*.m text diff=objc +*.java text diff=java +*.sh text eol=lf +*.vcproj text eol=crlf +*.vcproj.filters text eol=crlf +*.sln text eol=crlf +*.properties text +*.xml text + +# To avoid mucking up the utf-8 characters. +Core/Dialog/PSPOskDialog.cpp binary \ No newline at end of file diff --git a/Core/HLE/sceAudiocodec.cpp b/Core/HLE/sceAudiocodec.cpp index e5ffd49275..3cca598605 100644 --- a/Core/HLE/sceAudiocodec.cpp +++ b/Core/HLE/sceAudiocodec.cpp @@ -1,212 +1,212 @@ -// Copyright (c) 2012- PPSSPP Project. - -// This program is free software: you can redistribute it and/or modify -// it under the terms of the GNU General Public License as published by -// the Free Software Foundation, version 2.0 or later versions. - -// This program is distributed in the hope that it will be useful, -// but WITHOUT ANY WARRANTY; without even the implied warranty of -// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -// GNU General Public License 2.0 for more details. - -// A copy of the GPL 2.0 should have been included with the program. -// If not, see http://www.gnu.org/licenses/ - -// Official git repository and contact information can be found at -// https://github.com/hrydgard/ppsspp and http://www.ppsspp.org/. - -#include "Core/HLE/HLE.h" -#include "Core/HLE/FunctionWrappers.h" -#include "Core/HLE/sceAudiocodec.h" -#include "Core/MemMap.h" -#include "Core/Reporting.h" -#include "Core/HW/SimpleAudioDec.h" -#include "Common/ChunkFile.h" - -// Following kaien_fr's sample code https://github.com/hrydgard/ppsspp/issues/5620#issuecomment-37086024 -// Should probably store the EDRAM get/release status somewhere within here, etc. -struct AudioCodecContext { - u32_le unknown[6]; - u32_le inDataPtr; // 6 - u32_le inDataSize; // 7 - u32_le outDataPtr; // 8 - u32_le audioSamplesPerFrame; // 9 - u32_le inDataSizeAgain; // 10 ?? -}; - -// audioList is to store current playing audios. -static std::map audioList; - -static bool oldStateLoaded = false; - -// find the audio decoder for corresponding ctxPtr in audioList -static SimpleAudio *findDecoder(u32 ctxPtr) { - auto it = audioList.find(ctxPtr); - if (it != audioList.end()) { - return it->second; - } - return NULL; -} - -// remove decoder from audioList -static bool removeDecoder(u32 ctxPtr) { - auto it = audioList.find(ctxPtr); - if (it != audioList.end()) { - delete it->second; - audioList.erase(it); - return true; - } - return false; -} - -static void clearDecoders() { - for (auto it = audioList.begin(), end = audioList.end(); it != end; it++) { - delete it->second; - } - audioList.clear(); -} - -void __AudioCodecInit() { - oldStateLoaded = false; -} - -void __AudioCodecShutdown() { - // We need to kill off any still opened codecs to not leak memory. - clearDecoders(); -} - -int sceAudiocodecInit(u32 ctxPtr, int codec) { - if (IsValidCodec(codec)) { - // Create audio decoder for given audio codec and push it into AudioList - if (removeDecoder(ctxPtr)) { - WARN_LOG_REPORT(HLE, "sceAudiocodecInit(%08x, %d): replacing existing context", ctxPtr, codec); - } - auto decoder = new SimpleAudio(codec); - decoder->SetCtxPtr(ctxPtr); - audioList[ctxPtr] = decoder; - INFO_LOG(ME, "sceAudiocodecInit(%08x, %i (%s))", ctxPtr, codec, GetCodecName(codec)); - DEBUG_LOG(ME, "Number of playing sceAudioCodec audios : %d", (int)audioList.size()); - return 0; - } - ERROR_LOG_REPORT(ME, "sceAudiocodecInit(%08x, %i (%s)): Unknown audio codec %i", ctxPtr, codec, GetCodecName(codec), codec); - return 0; -} - -int sceAudiocodecDecode(u32 ctxPtr, int codec) { - if (!ctxPtr){ - ERROR_LOG_REPORT(ME, "sceAudiocodecDecode(%08x, %i (%s)) got NULL pointer", ctxPtr, codec, GetCodecName(codec)); - return -1; - } - - if (IsValidCodec(codec)){ - // Use SimpleAudioDec to decode audio - auto ctx = PSPPointer::Create(ctxPtr); // On stack, no need to allocate. - int outbytes = 0; - // find a decoder in audioList - auto decoder = findDecoder(ctxPtr); - - if (!decoder && oldStateLoaded) { - // We must have loaded an old state that did not have sceAudiocodec information. - // Fake it by creating the desired context. - decoder = new SimpleAudio(codec); - decoder->SetCtxPtr(ctxPtr); - audioList[ctxPtr] = decoder; - } - - if (decoder != NULL) { - // Decode audio - decoder->Decode(Memory::GetPointer(ctx->inDataPtr), ctx->inDataSize, Memory::GetPointer(ctx->outDataPtr), &outbytes); - } - DEBUG_LOG(ME, "sceAudiocodecDec(%08x, %i (%s))", ctxPtr, codec, GetCodecName(codec)); - return 0; - } - ERROR_LOG_REPORT(ME, "UNIMPL sceAudiocodecDecode(%08x, %i (%s))", ctxPtr, codec, GetCodecName(codec)); - return 0; -} - -int sceAudiocodecGetInfo(u32 ctxPtr, int codec) { - ERROR_LOG_REPORT(ME, "UNIMPL sceAudiocodecGetInfo(%08x, %i (%s))", ctxPtr, codec, GetCodecName(codec)); - return 0; -} - -int sceAudiocodecCheckNeedMem(u32 ctxPtr, int codec) { - WARN_LOG(ME, "UNIMPL sceAudiocodecCheckNeedMem(%08x, %i (%s))", ctxPtr, codec, GetCodecName(codec)); - return 0; -} - -int sceAudiocodecGetEDRAM(u32 ctxPtr, int codec) { - WARN_LOG(ME, "UNIMPL sceAudiocodecGetEDRAM(%08x, %i (%s))", ctxPtr, codec, GetCodecName(codec)); - return 0; -} - -int sceAudiocodecReleaseEDRAM(u32 ctxPtr, int id) { - if (removeDecoder(ctxPtr)){ - INFO_LOG(ME, "sceAudiocodecReleaseEDRAM(%08x, %i)", ctxPtr, id); - return 0; - } - WARN_LOG(ME, "UNIMPL sceAudiocodecReleaseEDRAM(%08x, %i)", ctxPtr, id); - return 0; -} - -const HLEFunction sceAudiocodec[] = { - { 0x70A703F8, WrapI_UI, "sceAudiocodecDecode" }, - { 0x5B37EB1D, WrapI_UI, "sceAudiocodecInit" }, - { 0x8ACA11D5, WrapI_UI, "sceAudiocodecGetInfo" }, - { 0x3A20A200, WrapI_UI, "sceAudiocodecGetEDRAM" }, - { 0x29681260, WrapI_UI, "sceAudiocodecReleaseEDRAM" }, - { 0x9D3F790C, WrapI_UI, "sceAudiocodecCheckNeedMem" }, - { 0x59176a0f, 0, "sceAudiocodec_59176A0F" }, -}; - -void Register_sceAudiocodec() -{ - RegisterModule("sceAudiocodec", ARRAY_SIZE(sceAudiocodec), sceAudiocodec); -} - -void __sceAudiocodecDoState(PointerWrap &p){ - auto s = p.Section("AudioList", 0, 2); - if (!s) { - oldStateLoaded = true; - return; - } - - int count = (int)audioList.size(); - p.Do(count); - - if (count > 0) { - if (p.mode == PointerWrap::MODE_READ) { - clearDecoders(); - - // loadstate if audioList is nonempty - auto codec_ = new int[count]; - auto ctxPtr_ = new u32[count]; - p.DoArray(codec_, s >= 2 ? count : (int)ARRAY_SIZE(codec_)); - p.DoArray(ctxPtr_, s >= 2 ? count : (int)ARRAY_SIZE(ctxPtr_)); - for (int i = 0; i < count; i++) { - auto decoder = new SimpleAudio(codec_[i]); - decoder->SetCtxPtr(ctxPtr_[i]); - audioList[ctxPtr_[i]] = decoder; - } - delete[] codec_; - delete[] ctxPtr_; - } - else - { - // savestate if audioList is nonempty - // Some of this is only necessary in Write but won't really hurt Measure. - auto codec_ = new int[count]; - auto ctxPtr_ = new u32[count]; - int i = 0; - for (auto it = audioList.begin(), end = audioList.end(); it != end; it++) { - const SimpleAudio *decoder = it->second; - codec_[i] = decoder->GetAudioType(); - ctxPtr_[i] = decoder->GetCtxPtr(); - i++; - } - p.DoArray(codec_, count); - p.DoArray(ctxPtr_, count); - delete[] codec_; - delete[] ctxPtr_; - } - } -} +// Copyright (c) 2012- PPSSPP Project. + +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU General Public License as published by +// the Free Software Foundation, version 2.0 or later versions. + +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU General Public License 2.0 for more details. + +// A copy of the GPL 2.0 should have been included with the program. +// If not, see http://www.gnu.org/licenses/ + +// Official git repository and contact information can be found at +// https://github.com/hrydgard/ppsspp and http://www.ppsspp.org/. + +#include "Core/HLE/HLE.h" +#include "Core/HLE/FunctionWrappers.h" +#include "Core/HLE/sceAudiocodec.h" +#include "Core/MemMap.h" +#include "Core/Reporting.h" +#include "Core/HW/SimpleAudioDec.h" +#include "Common/ChunkFile.h" + +// Following kaien_fr's sample code https://github.com/hrydgard/ppsspp/issues/5620#issuecomment-37086024 +// Should probably store the EDRAM get/release status somewhere within here, etc. +struct AudioCodecContext { + u32_le unknown[6]; + u32_le inDataPtr; // 6 + u32_le inDataSize; // 7 + u32_le outDataPtr; // 8 + u32_le audioSamplesPerFrame; // 9 + u32_le inDataSizeAgain; // 10 ?? +}; + +// audioList is to store current playing audios. +static std::map audioList; + +static bool oldStateLoaded = false; + +// find the audio decoder for corresponding ctxPtr in audioList +static SimpleAudio *findDecoder(u32 ctxPtr) { + auto it = audioList.find(ctxPtr); + if (it != audioList.end()) { + return it->second; + } + return NULL; +} + +// remove decoder from audioList +static bool removeDecoder(u32 ctxPtr) { + auto it = audioList.find(ctxPtr); + if (it != audioList.end()) { + delete it->second; + audioList.erase(it); + return true; + } + return false; +} + +static void clearDecoders() { + for (auto it = audioList.begin(), end = audioList.end(); it != end; it++) { + delete it->second; + } + audioList.clear(); +} + +void __AudioCodecInit() { + oldStateLoaded = false; +} + +void __AudioCodecShutdown() { + // We need to kill off any still opened codecs to not leak memory. + clearDecoders(); +} + +int sceAudiocodecInit(u32 ctxPtr, int codec) { + if (IsValidCodec(codec)) { + // Create audio decoder for given audio codec and push it into AudioList + if (removeDecoder(ctxPtr)) { + WARN_LOG_REPORT(HLE, "sceAudiocodecInit(%08x, %d): replacing existing context", ctxPtr, codec); + } + auto decoder = new SimpleAudio(codec); + decoder->SetCtxPtr(ctxPtr); + audioList[ctxPtr] = decoder; + INFO_LOG(ME, "sceAudiocodecInit(%08x, %i (%s))", ctxPtr, codec, GetCodecName(codec)); + DEBUG_LOG(ME, "Number of playing sceAudioCodec audios : %d", (int)audioList.size()); + return 0; + } + ERROR_LOG_REPORT(ME, "sceAudiocodecInit(%08x, %i (%s)): Unknown audio codec %i", ctxPtr, codec, GetCodecName(codec), codec); + return 0; +} + +int sceAudiocodecDecode(u32 ctxPtr, int codec) { + if (!ctxPtr){ + ERROR_LOG_REPORT(ME, "sceAudiocodecDecode(%08x, %i (%s)) got NULL pointer", ctxPtr, codec, GetCodecName(codec)); + return -1; + } + + if (IsValidCodec(codec)){ + // Use SimpleAudioDec to decode audio + auto ctx = PSPPointer::Create(ctxPtr); // On stack, no need to allocate. + int outbytes = 0; + // find a decoder in audioList + auto decoder = findDecoder(ctxPtr); + + if (!decoder && oldStateLoaded) { + // We must have loaded an old state that did not have sceAudiocodec information. + // Fake it by creating the desired context. + decoder = new SimpleAudio(codec); + decoder->SetCtxPtr(ctxPtr); + audioList[ctxPtr] = decoder; + } + + if (decoder != NULL) { + // Decode audio + decoder->Decode(Memory::GetPointer(ctx->inDataPtr), ctx->inDataSize, Memory::GetPointer(ctx->outDataPtr), &outbytes); + } + DEBUG_LOG(ME, "sceAudiocodecDec(%08x, %i (%s))", ctxPtr, codec, GetCodecName(codec)); + return 0; + } + ERROR_LOG_REPORT(ME, "UNIMPL sceAudiocodecDecode(%08x, %i (%s))", ctxPtr, codec, GetCodecName(codec)); + return 0; +} + +int sceAudiocodecGetInfo(u32 ctxPtr, int codec) { + ERROR_LOG_REPORT(ME, "UNIMPL sceAudiocodecGetInfo(%08x, %i (%s))", ctxPtr, codec, GetCodecName(codec)); + return 0; +} + +int sceAudiocodecCheckNeedMem(u32 ctxPtr, int codec) { + WARN_LOG(ME, "UNIMPL sceAudiocodecCheckNeedMem(%08x, %i (%s))", ctxPtr, codec, GetCodecName(codec)); + return 0; +} + +int sceAudiocodecGetEDRAM(u32 ctxPtr, int codec) { + WARN_LOG(ME, "UNIMPL sceAudiocodecGetEDRAM(%08x, %i (%s))", ctxPtr, codec, GetCodecName(codec)); + return 0; +} + +int sceAudiocodecReleaseEDRAM(u32 ctxPtr, int id) { + if (removeDecoder(ctxPtr)){ + INFO_LOG(ME, "sceAudiocodecReleaseEDRAM(%08x, %i)", ctxPtr, id); + return 0; + } + WARN_LOG(ME, "UNIMPL sceAudiocodecReleaseEDRAM(%08x, %i)", ctxPtr, id); + return 0; +} + +const HLEFunction sceAudiocodec[] = { + { 0x70A703F8, WrapI_UI, "sceAudiocodecDecode" }, + { 0x5B37EB1D, WrapI_UI, "sceAudiocodecInit" }, + { 0x8ACA11D5, WrapI_UI, "sceAudiocodecGetInfo" }, + { 0x3A20A200, WrapI_UI, "sceAudiocodecGetEDRAM" }, + { 0x29681260, WrapI_UI, "sceAudiocodecReleaseEDRAM" }, + { 0x9D3F790C, WrapI_UI, "sceAudiocodecCheckNeedMem" }, + { 0x59176a0f, 0, "sceAudiocodec_59176A0F" }, +}; + +void Register_sceAudiocodec() +{ + RegisterModule("sceAudiocodec", ARRAY_SIZE(sceAudiocodec), sceAudiocodec); +} + +void __sceAudiocodecDoState(PointerWrap &p){ + auto s = p.Section("AudioList", 0, 2); + if (!s) { + oldStateLoaded = true; + return; + } + + int count = (int)audioList.size(); + p.Do(count); + + if (count > 0) { + if (p.mode == PointerWrap::MODE_READ) { + clearDecoders(); + + // loadstate if audioList is nonempty + auto codec_ = new int[count]; + auto ctxPtr_ = new u32[count]; + p.DoArray(codec_, s >= 2 ? count : (int)ARRAY_SIZE(codec_)); + p.DoArray(ctxPtr_, s >= 2 ? count : (int)ARRAY_SIZE(ctxPtr_)); + for (int i = 0; i < count; i++) { + auto decoder = new SimpleAudio(codec_[i]); + decoder->SetCtxPtr(ctxPtr_[i]); + audioList[ctxPtr_[i]] = decoder; + } + delete[] codec_; + delete[] ctxPtr_; + } + else + { + // savestate if audioList is nonempty + // Some of this is only necessary in Write but won't really hurt Measure. + auto codec_ = new int[count]; + auto ctxPtr_ = new u32[count]; + int i = 0; + for (auto it = audioList.begin(), end = audioList.end(); it != end; it++) { + const SimpleAudio *decoder = it->second; + codec_[i] = decoder->GetAudioType(); + ctxPtr_[i] = decoder->GetCtxPtr(); + i++; + } + p.DoArray(codec_, count); + p.DoArray(ctxPtr_, count); + delete[] codec_; + delete[] ctxPtr_; + } + } +} diff --git a/GPU/Common/DrawEngineCommon.cpp b/GPU/Common/DrawEngineCommon.cpp index 982a9c5246..611c3df841 100644 --- a/GPU/Common/DrawEngineCommon.cpp +++ b/GPU/Common/DrawEngineCommon.cpp @@ -1,343 +1,343 @@ -// Copyright (c) 2013- PPSSPP Project. - -// This program is free software: you can redistribute it and/or modify -// it under the terms of the GNU General Public License as published by -// the Free Software Foundation, version 2.0 or later versions. - -// This program is distributed in the hope that it will be useful, -// but WITHOUT ANY WARRANTY; without even the implied warranty of -// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -// GNU General Public License 2.0 for more details. - -// A copy of the GPL 2.0 should have been included with the program. -// If not, see http://www.gnu.org/licenses/ - -// Official git repository and contact information can be found at -// https://github.com/hrydgard/ppsspp and http://www.ppsspp.org/. - -#include "GPU/Common/DrawEngineCommon.h" -#include "GPU/Common/SplineCommon.h" -#include "GPU/Common/VertexDecoderCommon.h" -#include "GPU/ge_constants.h" -#include "GPU/GPUState.h" - -#include "Core/Config.h" - -#include - -DrawEngineCommon::~DrawEngineCommon() { } - -struct Plane { - float x, y, z, w; - void Set(float _x, float _y, float _z, float _w) { x = _x; y = _y; z = _z; w = _w; } - float Test(float f[3]) const { return x * f[0] + y * f[1] + z * f[2] + w; } -}; - -static void PlanesFromMatrix(float mtx[16], Plane planes[6]) { - planes[0].Set(mtx[3]-mtx[0], mtx[7]-mtx[4], mtx[11]-mtx[8], mtx[15]-mtx[12]); // Right - planes[1].Set(mtx[3]+mtx[0], mtx[7]+mtx[4], mtx[11]+mtx[8], mtx[15]+mtx[12]); // Left - planes[2].Set(mtx[3]+mtx[1], mtx[7]+mtx[5], mtx[11]+mtx[9], mtx[15]+mtx[13]); // Bottom - planes[3].Set(mtx[3]-mtx[1], mtx[7]-mtx[5], mtx[11]-mtx[9], mtx[15]-mtx[13]); // Top - planes[4].Set(mtx[3]+mtx[2], mtx[7]+mtx[6], mtx[11]+mtx[10], mtx[15]+mtx[14]); // Near - planes[5].Set(mtx[3]-mtx[2], mtx[7]-mtx[6], mtx[11]-mtx[10], mtx[15]-mtx[14]); // Far -} - -static Vec3f ClipToScreen(const Vec4f& coords) { - // TODO: Check for invalid parameters (x2 < x1, etc) - float vpx1 = getFloat24(gstate.viewportx1); - float vpx2 = getFloat24(gstate.viewportx2); - float vpy1 = getFloat24(gstate.viewporty1); - float vpy2 = getFloat24(gstate.viewporty2); - float vpz1 = getFloat24(gstate.viewportz1); - float vpz2 = getFloat24(gstate.viewportz2); - - float retx = coords.x * vpx1 / coords.w + vpx2; - float rety = coords.y * vpy1 / coords.w + vpy2; - float retz = coords.z * vpz1 / coords.w + vpz2; - - // 16 = 0xFFFF / 4095.9375 - return Vec3f(retx * 16, rety * 16, retz); -} - -static Vec3f ScreenToDrawing(const Vec3f& coords) { - Vec3f ret; - ret.x = (coords.x - gstate.getOffsetX16()) * (1.0f / 16.0f); - ret.y = (coords.y - gstate.getOffsetY16()) * (1.0f / 16.0f); - ret.z = coords.z; - return ret; -} - -// This code is HIGHLY unoptimized! -// -// It does the simplest and safest test possible: If all points of a bbox is outside a single of -// our clipping planes, we reject the box. Tighter bounds would be desirable but would take more calculations. -bool DrawEngineCommon::TestBoundingBox(void* control_points, int vertexCount, u32 vertType) { - SimpleVertex *corners = (SimpleVertex *)(decoded + 65536 * 12); - float *verts = (float *)(decoded + 65536 * 18); - - // Try to skip NormalizeVertices if it's pure positions. No need to bother with a vertex decoder - // and a large vertex format. - if ((vertType & 0xFFFFFF) == GE_VTYPE_POS_FLOAT) { - // memcpy(verts, control_points, 12 * vertexCount); - verts = (float *)control_points; - } else if ((vertType & 0xFFFFFF) == GE_VTYPE_POS_8BIT) { - const s8 *vtx = (const s8 *)control_points; - for (int i = 0; i < vertexCount * 3; i++) { - verts[i] = vtx[i] * (1.0f / 128.0f); - } - } else if ((vertType & 0xFFFFFF) == GE_VTYPE_POS_16BIT) { - const s16 *vtx = (const s16*)control_points; - for (int i = 0; i < vertexCount * 3; i++) { - verts[i] = vtx[i] * (1.0f / 32768.0f); - } - } else { - // Simplify away bones and morph before proceeding - u8 *temp_buffer = decoded + 65536 * 24; - NormalizeVertices((u8 *)corners, temp_buffer, (u8 *)control_points, 0, vertexCount, vertType); - // Special case for float positions only. - const float *ctrl = (const float *)control_points; - for (int i = 0; i < vertexCount; i++) { - verts[i * 3] = corners[i].pos.x; - verts[i * 3 + 1] = corners[i].pos.y; - verts[i * 3 + 2] = corners[i].pos.z; - } - } - - Plane planes[6]; - - float world[16]; - float view[16]; - float worldview[16]; - float worldviewproj[16]; - ConvertMatrix4x3To4x4(world, gstate.worldMatrix); - ConvertMatrix4x3To4x4(view, gstate.viewMatrix); - Matrix4ByMatrix4(worldview, world, view); - Matrix4ByMatrix4(worldviewproj, worldview, gstate.projMatrix); - PlanesFromMatrix(worldviewproj, planes); - for (int plane = 0; plane < 6; plane++) { - int inside = 0; - int out = 0; - for (int i = 0; i < vertexCount; i++) { - // Here we can test against the frustum planes! - float value = planes[plane].Test(verts + i * 3); - if (value < 0) - out++; - else - inside++; - } - - if (inside == 0) { - // All out - return false; - } - - // Any out. For testing that the planes are in the right locations. - // if (out != 0) return false; - } - - return true; -} - -// TODO: This probably is not the best interface. -bool DrawEngineCommon::GetCurrentSimpleVertices(int count, std::vector &vertices, std::vector &indices) { - // This is always for the current vertices. - u16 indexLowerBound = 0; - u16 indexUpperBound = count - 1; - - bool savedVertexFullAlpha = gstate_c.vertexFullAlpha; - - if ((gstate.vertType & GE_VTYPE_IDX_MASK) != GE_VTYPE_IDX_NONE) { - const u8 *inds = Memory::GetPointer(gstate_c.indexAddr); - const u16 *inds16 = (const u16 *)inds; - - if (inds) { - GetIndexBounds(inds, count, gstate.vertType, &indexLowerBound, &indexUpperBound); - indices.resize(count); - switch (gstate.vertType & GE_VTYPE_IDX_MASK) { - case GE_VTYPE_IDX_16BIT: - for (int i = 0; i < count; ++i) { - indices[i] = inds16[i]; - } - break; - case GE_VTYPE_IDX_8BIT: - for (int i = 0; i < count; ++i) { - indices[i] = inds[i]; - } - break; - default: - return false; - } - } else { - indices.clear(); - } - } else { - indices.clear(); - } - - static std::vector temp_buffer; - static std::vector simpleVertices; - temp_buffer.resize(std::max((int)indexUpperBound, 8192) * 128 / sizeof(u32)); - simpleVertices.resize(indexUpperBound + 1); - NormalizeVertices((u8 *)(&simpleVertices[0]), (u8 *)(&temp_buffer[0]), Memory::GetPointer(gstate_c.vertexAddr), indexLowerBound, indexUpperBound, gstate.vertType); - - float world[16]; - float view[16]; - float worldview[16]; - float worldviewproj[16]; - ConvertMatrix4x3To4x4(world, gstate.worldMatrix); - ConvertMatrix4x3To4x4(view, gstate.viewMatrix); - Matrix4ByMatrix4(worldview, world, view); - Matrix4ByMatrix4(worldviewproj, worldview, gstate.projMatrix); - - vertices.resize(indexUpperBound + 1); - for (int i = indexLowerBound; i <= indexUpperBound; ++i) { - const SimpleVertex &vert = simpleVertices[i]; - - if (gstate.isModeThrough()) { - if (gstate.vertType & GE_VTYPE_TC_MASK) { - vertices[i].u = vert.uv[0]; - vertices[i].v = vert.uv[1]; - } else { - vertices[i].u = 0.0f; - vertices[i].v = 0.0f; - } - vertices[i].x = vert.pos.x; - vertices[i].y = vert.pos.y; - vertices[i].z = vert.pos.z; - if (gstate.vertType & GE_VTYPE_COL_MASK) { - memcpy(vertices[i].c, vert.color, sizeof(vertices[i].c)); - } else { - memset(vertices[i].c, 0, sizeof(vertices[i].c)); - } - } else { - float clipPos[4]; - Vec3ByMatrix44(clipPos, vert.pos.AsArray(), worldviewproj); - Vec3f screenPos = ClipToScreen(clipPos); - Vec3f drawPos = ScreenToDrawing(screenPos); - - if (gstate.vertType & GE_VTYPE_TC_MASK) { - vertices[i].u = vert.uv[0]; - vertices[i].v = vert.uv[1]; - } else { - vertices[i].u = 0.0f; - vertices[i].v = 0.0f; - } - vertices[i].x = drawPos.x; - vertices[i].y = drawPos.y; - vertices[i].z = drawPos.z; - if (gstate.vertType & GE_VTYPE_COL_MASK) { - memcpy(vertices[i].c, vert.color, sizeof(vertices[i].c)); - } else { - memset(vertices[i].c, 0, sizeof(vertices[i].c)); - } - } - } - - gstate_c.vertexFullAlpha = savedVertexFullAlpha; - - return true; -} - - -// This normalizes a set of vertices in any format to SimpleVertex format, by processing away morphing AND skinning. -// The rest of the transform pipeline like lighting will go as normal, either hardware or software. -// The implementation is initially a bit inefficient but shouldn't be a big deal. -// An intermediate buffer of not-easy-to-predict size is stored at bufPtr. -u32 DrawEngineCommon::NormalizeVertices(u8 *outPtr, u8 *bufPtr, const u8 *inPtr, VertexDecoder *dec, int lowerBound, int upperBound, u32 vertType) { - // First, decode the vertices into a GPU compatible format. This step can be eliminated but will need a separate - // implementation of the vertex decoder. - dec->DecodeVerts(bufPtr, inPtr, lowerBound, upperBound); - - // OK, morphing eliminated but bones still remain to be taken care of. - // Let's do a partial software transform where we only do skinning. - - VertexReader reader(bufPtr, dec->GetDecVtxFmt(), vertType); - - SimpleVertex *sverts = (SimpleVertex *)outPtr; - - const u8 defaultColor[4] = { - (u8)gstate.getMaterialAmbientR(), - (u8)gstate.getMaterialAmbientG(), - (u8)gstate.getMaterialAmbientB(), - (u8)gstate.getMaterialAmbientA(), - }; - - // Let's have two separate loops, one for non skinning and one for skinning. - if (!g_Config.bSoftwareSkinning && (vertType & GE_VTYPE_WEIGHT_MASK) != GE_VTYPE_WEIGHT_NONE) { - int numBoneWeights = vertTypeGetNumBoneWeights(vertType); - for (int i = lowerBound; i <= upperBound; i++) { - reader.Goto(i); - SimpleVertex &sv = sverts[i]; - if (vertType & GE_VTYPE_TC_MASK) { - reader.ReadUV(sv.uv); - } - - if (vertType & GE_VTYPE_COL_MASK) { - reader.ReadColor0_8888(sv.color); - } else { - memcpy(sv.color, defaultColor, 4); - } - - float nrm[3], pos[3]; - float bnrm[3], bpos[3]; - - if (vertType & GE_VTYPE_NRM_MASK) { - // Normals are generated during tesselation anyway, not sure if any need to supply - reader.ReadNrm(nrm); - } else { - nrm[0] = 0; - nrm[1] = 0; - nrm[2] = 1.0f; - } - reader.ReadPos(pos); - - // Apply skinning transform directly - float weights[8]; - reader.ReadWeights(weights); - // Skinning - Vec3Packedf psum(0, 0, 0); - Vec3Packedf nsum(0, 0, 0); - for (int w = 0; w < numBoneWeights; w++) { - if (weights[w] != 0.0f) { - Vec3ByMatrix43(bpos, pos, gstate.boneMatrix + w * 12); - Vec3Packedf tpos(bpos); - psum += tpos * weights[w]; - - Norm3ByMatrix43(bnrm, nrm, gstate.boneMatrix + w * 12); - Vec3Packedf tnorm(bnrm); - nsum += tnorm * weights[w]; - } - } - sv.pos = psum; - sv.nrm = nsum; - } - } else { - for (int i = lowerBound; i <= upperBound; i++) { - reader.Goto(i); - SimpleVertex &sv = sverts[i]; - if (vertType & GE_VTYPE_TC_MASK) { - reader.ReadUV(sv.uv); - } else { - sv.uv[0] = 0; // This will get filled in during tesselation - sv.uv[1] = 0; - } - if (vertType & GE_VTYPE_COL_MASK) { - reader.ReadColor0_8888(sv.color); - } else { - memcpy(sv.color, defaultColor, 4); - } - if (vertType & GE_VTYPE_NRM_MASK) { - // Normals are generated during tesselation anyway, not sure if any need to supply - reader.ReadNrm((float *)&sv.nrm); - } else { - sv.nrm.x = 0; - sv.nrm.y = 0; - sv.nrm.z = 1.0f; - } - reader.ReadPos((float *)&sv.pos); - } - } - - // Okay, there we are! Return the new type (but keep the index bits) - return GE_VTYPE_TC_FLOAT | GE_VTYPE_COL_8888 | GE_VTYPE_NRM_FLOAT | GE_VTYPE_POS_FLOAT | (vertType & (GE_VTYPE_IDX_MASK | GE_VTYPE_THROUGH)); -} +// Copyright (c) 2013- PPSSPP Project. + +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU General Public License as published by +// the Free Software Foundation, version 2.0 or later versions. + +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU General Public License 2.0 for more details. + +// A copy of the GPL 2.0 should have been included with the program. +// If not, see http://www.gnu.org/licenses/ + +// Official git repository and contact information can be found at +// https://github.com/hrydgard/ppsspp and http://www.ppsspp.org/. + +#include "GPU/Common/DrawEngineCommon.h" +#include "GPU/Common/SplineCommon.h" +#include "GPU/Common/VertexDecoderCommon.h" +#include "GPU/ge_constants.h" +#include "GPU/GPUState.h" + +#include "Core/Config.h" + +#include + +DrawEngineCommon::~DrawEngineCommon() { } + +struct Plane { + float x, y, z, w; + void Set(float _x, float _y, float _z, float _w) { x = _x; y = _y; z = _z; w = _w; } + float Test(float f[3]) const { return x * f[0] + y * f[1] + z * f[2] + w; } +}; + +static void PlanesFromMatrix(float mtx[16], Plane planes[6]) { + planes[0].Set(mtx[3]-mtx[0], mtx[7]-mtx[4], mtx[11]-mtx[8], mtx[15]-mtx[12]); // Right + planes[1].Set(mtx[3]+mtx[0], mtx[7]+mtx[4], mtx[11]+mtx[8], mtx[15]+mtx[12]); // Left + planes[2].Set(mtx[3]+mtx[1], mtx[7]+mtx[5], mtx[11]+mtx[9], mtx[15]+mtx[13]); // Bottom + planes[3].Set(mtx[3]-mtx[1], mtx[7]-mtx[5], mtx[11]-mtx[9], mtx[15]-mtx[13]); // Top + planes[4].Set(mtx[3]+mtx[2], mtx[7]+mtx[6], mtx[11]+mtx[10], mtx[15]+mtx[14]); // Near + planes[5].Set(mtx[3]-mtx[2], mtx[7]-mtx[6], mtx[11]-mtx[10], mtx[15]-mtx[14]); // Far +} + +static Vec3f ClipToScreen(const Vec4f& coords) { + // TODO: Check for invalid parameters (x2 < x1, etc) + float vpx1 = getFloat24(gstate.viewportx1); + float vpx2 = getFloat24(gstate.viewportx2); + float vpy1 = getFloat24(gstate.viewporty1); + float vpy2 = getFloat24(gstate.viewporty2); + float vpz1 = getFloat24(gstate.viewportz1); + float vpz2 = getFloat24(gstate.viewportz2); + + float retx = coords.x * vpx1 / coords.w + vpx2; + float rety = coords.y * vpy1 / coords.w + vpy2; + float retz = coords.z * vpz1 / coords.w + vpz2; + + // 16 = 0xFFFF / 4095.9375 + return Vec3f(retx * 16, rety * 16, retz); +} + +static Vec3f ScreenToDrawing(const Vec3f& coords) { + Vec3f ret; + ret.x = (coords.x - gstate.getOffsetX16()) * (1.0f / 16.0f); + ret.y = (coords.y - gstate.getOffsetY16()) * (1.0f / 16.0f); + ret.z = coords.z; + return ret; +} + +// This code is HIGHLY unoptimized! +// +// It does the simplest and safest test possible: If all points of a bbox is outside a single of +// our clipping planes, we reject the box. Tighter bounds would be desirable but would take more calculations. +bool DrawEngineCommon::TestBoundingBox(void* control_points, int vertexCount, u32 vertType) { + SimpleVertex *corners = (SimpleVertex *)(decoded + 65536 * 12); + float *verts = (float *)(decoded + 65536 * 18); + + // Try to skip NormalizeVertices if it's pure positions. No need to bother with a vertex decoder + // and a large vertex format. + if ((vertType & 0xFFFFFF) == GE_VTYPE_POS_FLOAT) { + // memcpy(verts, control_points, 12 * vertexCount); + verts = (float *)control_points; + } else if ((vertType & 0xFFFFFF) == GE_VTYPE_POS_8BIT) { + const s8 *vtx = (const s8 *)control_points; + for (int i = 0; i < vertexCount * 3; i++) { + verts[i] = vtx[i] * (1.0f / 128.0f); + } + } else if ((vertType & 0xFFFFFF) == GE_VTYPE_POS_16BIT) { + const s16 *vtx = (const s16*)control_points; + for (int i = 0; i < vertexCount * 3; i++) { + verts[i] = vtx[i] * (1.0f / 32768.0f); + } + } else { + // Simplify away bones and morph before proceeding + u8 *temp_buffer = decoded + 65536 * 24; + NormalizeVertices((u8 *)corners, temp_buffer, (u8 *)control_points, 0, vertexCount, vertType); + // Special case for float positions only. + const float *ctrl = (const float *)control_points; + for (int i = 0; i < vertexCount; i++) { + verts[i * 3] = corners[i].pos.x; + verts[i * 3 + 1] = corners[i].pos.y; + verts[i * 3 + 2] = corners[i].pos.z; + } + } + + Plane planes[6]; + + float world[16]; + float view[16]; + float worldview[16]; + float worldviewproj[16]; + ConvertMatrix4x3To4x4(world, gstate.worldMatrix); + ConvertMatrix4x3To4x4(view, gstate.viewMatrix); + Matrix4ByMatrix4(worldview, world, view); + Matrix4ByMatrix4(worldviewproj, worldview, gstate.projMatrix); + PlanesFromMatrix(worldviewproj, planes); + for (int plane = 0; plane < 6; plane++) { + int inside = 0; + int out = 0; + for (int i = 0; i < vertexCount; i++) { + // Here we can test against the frustum planes! + float value = planes[plane].Test(verts + i * 3); + if (value < 0) + out++; + else + inside++; + } + + if (inside == 0) { + // All out + return false; + } + + // Any out. For testing that the planes are in the right locations. + // if (out != 0) return false; + } + + return true; +} + +// TODO: This probably is not the best interface. +bool DrawEngineCommon::GetCurrentSimpleVertices(int count, std::vector &vertices, std::vector &indices) { + // This is always for the current vertices. + u16 indexLowerBound = 0; + u16 indexUpperBound = count - 1; + + bool savedVertexFullAlpha = gstate_c.vertexFullAlpha; + + if ((gstate.vertType & GE_VTYPE_IDX_MASK) != GE_VTYPE_IDX_NONE) { + const u8 *inds = Memory::GetPointer(gstate_c.indexAddr); + const u16 *inds16 = (const u16 *)inds; + + if (inds) { + GetIndexBounds(inds, count, gstate.vertType, &indexLowerBound, &indexUpperBound); + indices.resize(count); + switch (gstate.vertType & GE_VTYPE_IDX_MASK) { + case GE_VTYPE_IDX_16BIT: + for (int i = 0; i < count; ++i) { + indices[i] = inds16[i]; + } + break; + case GE_VTYPE_IDX_8BIT: + for (int i = 0; i < count; ++i) { + indices[i] = inds[i]; + } + break; + default: + return false; + } + } else { + indices.clear(); + } + } else { + indices.clear(); + } + + static std::vector temp_buffer; + static std::vector simpleVertices; + temp_buffer.resize(std::max((int)indexUpperBound, 8192) * 128 / sizeof(u32)); + simpleVertices.resize(indexUpperBound + 1); + NormalizeVertices((u8 *)(&simpleVertices[0]), (u8 *)(&temp_buffer[0]), Memory::GetPointer(gstate_c.vertexAddr), indexLowerBound, indexUpperBound, gstate.vertType); + + float world[16]; + float view[16]; + float worldview[16]; + float worldviewproj[16]; + ConvertMatrix4x3To4x4(world, gstate.worldMatrix); + ConvertMatrix4x3To4x4(view, gstate.viewMatrix); + Matrix4ByMatrix4(worldview, world, view); + Matrix4ByMatrix4(worldviewproj, worldview, gstate.projMatrix); + + vertices.resize(indexUpperBound + 1); + for (int i = indexLowerBound; i <= indexUpperBound; ++i) { + const SimpleVertex &vert = simpleVertices[i]; + + if (gstate.isModeThrough()) { + if (gstate.vertType & GE_VTYPE_TC_MASK) { + vertices[i].u = vert.uv[0]; + vertices[i].v = vert.uv[1]; + } else { + vertices[i].u = 0.0f; + vertices[i].v = 0.0f; + } + vertices[i].x = vert.pos.x; + vertices[i].y = vert.pos.y; + vertices[i].z = vert.pos.z; + if (gstate.vertType & GE_VTYPE_COL_MASK) { + memcpy(vertices[i].c, vert.color, sizeof(vertices[i].c)); + } else { + memset(vertices[i].c, 0, sizeof(vertices[i].c)); + } + } else { + float clipPos[4]; + Vec3ByMatrix44(clipPos, vert.pos.AsArray(), worldviewproj); + Vec3f screenPos = ClipToScreen(clipPos); + Vec3f drawPos = ScreenToDrawing(screenPos); + + if (gstate.vertType & GE_VTYPE_TC_MASK) { + vertices[i].u = vert.uv[0]; + vertices[i].v = vert.uv[1]; + } else { + vertices[i].u = 0.0f; + vertices[i].v = 0.0f; + } + vertices[i].x = drawPos.x; + vertices[i].y = drawPos.y; + vertices[i].z = drawPos.z; + if (gstate.vertType & GE_VTYPE_COL_MASK) { + memcpy(vertices[i].c, vert.color, sizeof(vertices[i].c)); + } else { + memset(vertices[i].c, 0, sizeof(vertices[i].c)); + } + } + } + + gstate_c.vertexFullAlpha = savedVertexFullAlpha; + + return true; +} + + +// This normalizes a set of vertices in any format to SimpleVertex format, by processing away morphing AND skinning. +// The rest of the transform pipeline like lighting will go as normal, either hardware or software. +// The implementation is initially a bit inefficient but shouldn't be a big deal. +// An intermediate buffer of not-easy-to-predict size is stored at bufPtr. +u32 DrawEngineCommon::NormalizeVertices(u8 *outPtr, u8 *bufPtr, const u8 *inPtr, VertexDecoder *dec, int lowerBound, int upperBound, u32 vertType) { + // First, decode the vertices into a GPU compatible format. This step can be eliminated but will need a separate + // implementation of the vertex decoder. + dec->DecodeVerts(bufPtr, inPtr, lowerBound, upperBound); + + // OK, morphing eliminated but bones still remain to be taken care of. + // Let's do a partial software transform where we only do skinning. + + VertexReader reader(bufPtr, dec->GetDecVtxFmt(), vertType); + + SimpleVertex *sverts = (SimpleVertex *)outPtr; + + const u8 defaultColor[4] = { + (u8)gstate.getMaterialAmbientR(), + (u8)gstate.getMaterialAmbientG(), + (u8)gstate.getMaterialAmbientB(), + (u8)gstate.getMaterialAmbientA(), + }; + + // Let's have two separate loops, one for non skinning and one for skinning. + if (!g_Config.bSoftwareSkinning && (vertType & GE_VTYPE_WEIGHT_MASK) != GE_VTYPE_WEIGHT_NONE) { + int numBoneWeights = vertTypeGetNumBoneWeights(vertType); + for (int i = lowerBound; i <= upperBound; i++) { + reader.Goto(i); + SimpleVertex &sv = sverts[i]; + if (vertType & GE_VTYPE_TC_MASK) { + reader.ReadUV(sv.uv); + } + + if (vertType & GE_VTYPE_COL_MASK) { + reader.ReadColor0_8888(sv.color); + } else { + memcpy(sv.color, defaultColor, 4); + } + + float nrm[3], pos[3]; + float bnrm[3], bpos[3]; + + if (vertType & GE_VTYPE_NRM_MASK) { + // Normals are generated during tesselation anyway, not sure if any need to supply + reader.ReadNrm(nrm); + } else { + nrm[0] = 0; + nrm[1] = 0; + nrm[2] = 1.0f; + } + reader.ReadPos(pos); + + // Apply skinning transform directly + float weights[8]; + reader.ReadWeights(weights); + // Skinning + Vec3Packedf psum(0, 0, 0); + Vec3Packedf nsum(0, 0, 0); + for (int w = 0; w < numBoneWeights; w++) { + if (weights[w] != 0.0f) { + Vec3ByMatrix43(bpos, pos, gstate.boneMatrix + w * 12); + Vec3Packedf tpos(bpos); + psum += tpos * weights[w]; + + Norm3ByMatrix43(bnrm, nrm, gstate.boneMatrix + w * 12); + Vec3Packedf tnorm(bnrm); + nsum += tnorm * weights[w]; + } + } + sv.pos = psum; + sv.nrm = nsum; + } + } else { + for (int i = lowerBound; i <= upperBound; i++) { + reader.Goto(i); + SimpleVertex &sv = sverts[i]; + if (vertType & GE_VTYPE_TC_MASK) { + reader.ReadUV(sv.uv); + } else { + sv.uv[0] = 0; // This will get filled in during tesselation + sv.uv[1] = 0; + } + if (vertType & GE_VTYPE_COL_MASK) { + reader.ReadColor0_8888(sv.color); + } else { + memcpy(sv.color, defaultColor, 4); + } + if (vertType & GE_VTYPE_NRM_MASK) { + // Normals are generated during tesselation anyway, not sure if any need to supply + reader.ReadNrm((float *)&sv.nrm); + } else { + sv.nrm.x = 0; + sv.nrm.y = 0; + sv.nrm.z = 1.0f; + } + reader.ReadPos((float *)&sv.pos); + } + } + + // Okay, there we are! Return the new type (but keep the index bits) + return GE_VTYPE_TC_FLOAT | GE_VTYPE_COL_8888 | GE_VTYPE_NRM_FLOAT | GE_VTYPE_POS_FLOAT | (vertType & (GE_VTYPE_IDX_MASK | GE_VTYPE_THROUGH)); +} diff --git a/GPU/Common/DrawEngineCommon.h b/GPU/Common/DrawEngineCommon.h index 28cd21b15f..5959f3c324 100644 --- a/GPU/Common/DrawEngineCommon.h +++ b/GPU/Common/DrawEngineCommon.h @@ -1,45 +1,45 @@ -// Copyright (c) 2013- PPSSPP Project. - -// This program is free software: you can redistribute it and/or modify -// it under the terms of the GNU General Public License as published by -// the Free Software Foundation, version 2.0 or later versions. - -// This program is distributed in the hope that it will be useful, -// but WITHOUT ANY WARRANTY; without even the implied warranty of -// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -// GNU General Public License 2.0 for more details. - -// A copy of the GPL 2.0 should have been included with the program. -// If not, see http://www.gnu.org/licenses/ - -// Official git repository and contact information can be found at -// https://github.com/hrydgard/ppsspp and http://www.ppsspp.org/. - -#pragma once - -#include - -#include "Common/CommonTypes.h" - -#include "GPU/Common/GPUDebugInterface.h" - -class VertexDecoder; - -class DrawEngineCommon { -public: - virtual ~DrawEngineCommon(); - - bool TestBoundingBox(void* control_points, int vertexCount, u32 vertType); - - // TODO: This can be shared once the decoder cache / etc. are. - virtual u32 NormalizeVertices(u8 *outPtr, u8 *bufPtr, const u8 *inPtr, int lowerBound, int upperBound, u32 vertType) = 0; - - bool GetCurrentSimpleVertices(int count, std::vector &vertices, std::vector &indices); - - static u32 NormalizeVertices(u8 *outPtr, u8 *bufPtr, const u8 *inPtr, VertexDecoder *dec, int lowerBound, int upperBound, u32 vertType); - -protected: - // Vertex collector buffers - u8 *decoded; - u16 *decIndex; +// Copyright (c) 2013- PPSSPP Project. + +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU General Public License as published by +// the Free Software Foundation, version 2.0 or later versions. + +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU General Public License 2.0 for more details. + +// A copy of the GPL 2.0 should have been included with the program. +// If not, see http://www.gnu.org/licenses/ + +// Official git repository and contact information can be found at +// https://github.com/hrydgard/ppsspp and http://www.ppsspp.org/. + +#pragma once + +#include + +#include "Common/CommonTypes.h" + +#include "GPU/Common/GPUDebugInterface.h" + +class VertexDecoder; + +class DrawEngineCommon { +public: + virtual ~DrawEngineCommon(); + + bool TestBoundingBox(void* control_points, int vertexCount, u32 vertType); + + // TODO: This can be shared once the decoder cache / etc. are. + virtual u32 NormalizeVertices(u8 *outPtr, u8 *bufPtr, const u8 *inPtr, int lowerBound, int upperBound, u32 vertType) = 0; + + bool GetCurrentSimpleVertices(int count, std::vector &vertices, std::vector &indices); + + static u32 NormalizeVertices(u8 *outPtr, u8 *bufPtr, const u8 *inPtr, VertexDecoder *dec, int lowerBound, int upperBound, u32 vertType); + +protected: + // Vertex collector buffers + u8 *decoded; + u16 *decIndex; }; \ No newline at end of file diff --git a/Windows/D3D9Base.cpp b/Windows/D3D9Base.cpp index 976106f62a..9faef80110 100644 --- a/Windows/D3D9Base.cpp +++ b/Windows/D3D9Base.cpp @@ -1,203 +1,203 @@ -#include "Common/CommonWindows.h" -#include - -#include "GPU/Directx9/helper/global.h" -#include "GPU/Directx9/helper/fbo.h" - -#include "base/logging.h" -#include "util/text/utf8.h" -#include "i18n/i18n.h" - -#include "Core/Config.h" -#include "Windows/D3D9Base.h" -#include "thin3d/thin3d.h" -#include "thin3d/d3dx9_loader.h" - -static bool has9Ex = false; -static LPDIRECT3D9 d3d; -static LPDIRECT3D9EX d3dEx; -static int adapterId; -static LPDIRECT3DDEVICE9 device; -static LPDIRECT3DDEVICE9EX deviceEx; -static HDC hDC; // Private GDI Device Context -static HGLRC hRC; // Permanent Rendering Context -static HWND hWnd; // Holds Our Window Handle - -static int xres, yres; - -// TODO: Make config? -static bool enableGLDebug = true; - -void D3D9_SwapBuffers() { - if (has9Ex) { - deviceEx->EndScene(); - deviceEx->PresentEx(NULL, NULL, NULL, NULL, 0); - deviceEx->BeginScene(); - } else { - device->EndScene(); - device->Present(NULL, NULL, NULL, NULL); - device->BeginScene(); - } -} - -Thin3DContext *D3D9_CreateThin3DContext() { - return T3DCreateDX9Context(d3d, d3dEx, adapterId, device, deviceEx); -} - -typedef HRESULT (*DIRECT3DCREATE9EX)(UINT, IDirect3D9Ex**); - -bool IsWin7OrLater() { - DWORD version = GetVersion(); - DWORD major = (DWORD)(LOBYTE(LOWORD(version))); - DWORD minor = (DWORD)(HIBYTE(LOWORD(version))); - - return (major > 6) || ((major == 6) && (minor >= 1)); -} - -bool D3D9_Init(HWND hWnd, bool windowed, std::string *error_message) { - DIRECT3DCREATE9EX g_pfnCreate9ex; - - HMODULE hD3D9 = LoadLibrary(TEXT("d3d9.dll")); - - if (!hD3D9) { - ELOG("Missing d3d9.dll"); - *error_message = "D3D9.dll missing"; - return false; - } - - g_pfnCreate9ex = (DIRECT3DCREATE9EX)GetProcAddress(hD3D9, "Direct3DCreate9Ex"); - has9Ex = (g_pfnCreate9ex != NULL); - - if (has9Ex) { - HRESULT result = g_pfnCreate9ex(D3D_SDK_VERSION, &d3dEx); - d3d = d3dEx; - if (FAILED(result)) { - *error_message = "D3D9Ex available but context creation failed"; - return false; - } - } else { - d3d = Direct3DCreate9(D3D_SDK_VERSION); - if (!d3d) { - *error_message = "Failed to create D3D9 context"; - return false; - } - } - FreeLibrary(hD3D9); - - D3DCAPS9 d3dCaps; - - D3DDISPLAYMODE d3ddm; - if (FAILED(d3d->GetAdapterDisplayMode(D3DADAPTER_DEFAULT, &d3ddm))) { - *error_message = "GetAdapterDisplayMode failed"; - d3d->Release(); - return false; - } - - adapterId = D3DADAPTER_DEFAULT; - if (FAILED(d3d->GetDeviceCaps(adapterId, D3DDEVTYPE_HAL, &d3dCaps))) { - *error_message = "GetDeviceCaps failed (???)"; - d3d->Release(); - return false; - } - - HRESULT hr; - if (FAILED(hr = d3d->CheckDeviceFormat(D3DADAPTER_DEFAULT, - D3DDEVTYPE_HAL, - d3ddm.Format, - D3DUSAGE_DEPTHSTENCIL, - D3DRTYPE_SURFACE, - D3DFMT_D24S8))) { - if (hr == D3DERR_NOTAVAILABLE) { - *error_message = "D24S8 depth/stencil not available"; - d3d->Release(); - return false; - } - } - - DWORD dwBehaviorFlags = D3DCREATE_MULTITHREADED | D3DCREATE_FPU_PRESERVE; - if (d3dCaps.VertexProcessingCaps != 0) - dwBehaviorFlags |= D3DCREATE_HARDWARE_VERTEXPROCESSING; - else - dwBehaviorFlags |= D3DCREATE_SOFTWARE_VERTEXPROCESSING; - - RECT rc; - GetClientRect(hWnd, &rc); - int xres = rc.right - rc.left; - int yres = rc.bottom - rc.top; - - D3DPRESENT_PARAMETERS pp; - memset(&pp, 0, sizeof(pp)); - pp.BackBufferWidth = xres; - pp.BackBufferHeight = yres; - pp.BackBufferFormat = d3ddm.Format; - pp.MultiSampleType = D3DMULTISAMPLE_NONE; - pp.SwapEffect = D3DSWAPEFFECT_DISCARD; - pp.Windowed = windowed; - pp.hDeviceWindow = hWnd; - pp.EnableAutoDepthStencil = true; - pp.AutoDepthStencilFormat = D3DFMT_D24S8; - pp.PresentationInterval = (g_Config.bVSync) ? D3DPRESENT_INTERVAL_ONE : D3DPRESENT_INTERVAL_IMMEDIATE; - - if (has9Ex) { - if (windowed && IsWin7OrLater()) { - // This new flip mode gives higher performance. - // TODO: This makes it slower? - //pp.BackBufferCount = 2; - //pp.SwapEffect = D3DSWAPEFFECT_FLIPEX; - } - hr = d3dEx->CreateDeviceEx(adapterId, D3DDEVTYPE_HAL, hWnd, dwBehaviorFlags, &pp, NULL, &deviceEx); - device = deviceEx; - } else { - hr = d3d->CreateDevice(adapterId, D3DDEVTYPE_HAL, hWnd, dwBehaviorFlags, &pp, &device); - } - - if (FAILED(hr)) { - *error_message = "Failed to create D3D device"; - d3d->Release(); - return false; - } - - device->BeginScene(); - DX9::pD3Ddevice = device; - DX9::pD3DdeviceEx = deviceEx; - - LoadD3DX9Dynamic(); - - if (!DX9::CompileShaders(*error_message)) { - *error_message = "Unable to compile shaders: " + *error_message; - device->EndScene(); - device->Release(); - d3d->Release(); - DX9::pD3Ddevice = nullptr; - DX9::pD3DdeviceEx = nullptr; - device = nullptr; - UnloadD3DXDynamic(); - return false; - } - - DX9::fbo_init(d3d); - - if (deviceEx && IsWin7OrLater()) { - // TODO: This makes it slower? - //deviceEx->SetMaximumFrameLatency(1); - } - - return true; -} - -void D3D9_Resize(HWND window) { - // TODO! -} - -void D3D9_Shutdown() { - DX9::DestroyShaders(); - DX9::fbo_shutdown(); - device->EndScene(); - device->Release(); - d3d->Release(); - UnloadD3DXDynamic(); - DX9::pD3Ddevice = nullptr; - DX9::pD3DdeviceEx = nullptr; - device = nullptr; - hWnd = nullptr; -} +#include "Common/CommonWindows.h" +#include + +#include "GPU/Directx9/helper/global.h" +#include "GPU/Directx9/helper/fbo.h" + +#include "base/logging.h" +#include "util/text/utf8.h" +#include "i18n/i18n.h" + +#include "Core/Config.h" +#include "Windows/D3D9Base.h" +#include "thin3d/thin3d.h" +#include "thin3d/d3dx9_loader.h" + +static bool has9Ex = false; +static LPDIRECT3D9 d3d; +static LPDIRECT3D9EX d3dEx; +static int adapterId; +static LPDIRECT3DDEVICE9 device; +static LPDIRECT3DDEVICE9EX deviceEx; +static HDC hDC; // Private GDI Device Context +static HGLRC hRC; // Permanent Rendering Context +static HWND hWnd; // Holds Our Window Handle + +static int xres, yres; + +// TODO: Make config? +static bool enableGLDebug = true; + +void D3D9_SwapBuffers() { + if (has9Ex) { + deviceEx->EndScene(); + deviceEx->PresentEx(NULL, NULL, NULL, NULL, 0); + deviceEx->BeginScene(); + } else { + device->EndScene(); + device->Present(NULL, NULL, NULL, NULL); + device->BeginScene(); + } +} + +Thin3DContext *D3D9_CreateThin3DContext() { + return T3DCreateDX9Context(d3d, d3dEx, adapterId, device, deviceEx); +} + +typedef HRESULT (*DIRECT3DCREATE9EX)(UINT, IDirect3D9Ex**); + +bool IsWin7OrLater() { + DWORD version = GetVersion(); + DWORD major = (DWORD)(LOBYTE(LOWORD(version))); + DWORD minor = (DWORD)(HIBYTE(LOWORD(version))); + + return (major > 6) || ((major == 6) && (minor >= 1)); +} + +bool D3D9_Init(HWND hWnd, bool windowed, std::string *error_message) { + DIRECT3DCREATE9EX g_pfnCreate9ex; + + HMODULE hD3D9 = LoadLibrary(TEXT("d3d9.dll")); + + if (!hD3D9) { + ELOG("Missing d3d9.dll"); + *error_message = "D3D9.dll missing"; + return false; + } + + g_pfnCreate9ex = (DIRECT3DCREATE9EX)GetProcAddress(hD3D9, "Direct3DCreate9Ex"); + has9Ex = (g_pfnCreate9ex != NULL); + + if (has9Ex) { + HRESULT result = g_pfnCreate9ex(D3D_SDK_VERSION, &d3dEx); + d3d = d3dEx; + if (FAILED(result)) { + *error_message = "D3D9Ex available but context creation failed"; + return false; + } + } else { + d3d = Direct3DCreate9(D3D_SDK_VERSION); + if (!d3d) { + *error_message = "Failed to create D3D9 context"; + return false; + } + } + FreeLibrary(hD3D9); + + D3DCAPS9 d3dCaps; + + D3DDISPLAYMODE d3ddm; + if (FAILED(d3d->GetAdapterDisplayMode(D3DADAPTER_DEFAULT, &d3ddm))) { + *error_message = "GetAdapterDisplayMode failed"; + d3d->Release(); + return false; + } + + adapterId = D3DADAPTER_DEFAULT; + if (FAILED(d3d->GetDeviceCaps(adapterId, D3DDEVTYPE_HAL, &d3dCaps))) { + *error_message = "GetDeviceCaps failed (???)"; + d3d->Release(); + return false; + } + + HRESULT hr; + if (FAILED(hr = d3d->CheckDeviceFormat(D3DADAPTER_DEFAULT, + D3DDEVTYPE_HAL, + d3ddm.Format, + D3DUSAGE_DEPTHSTENCIL, + D3DRTYPE_SURFACE, + D3DFMT_D24S8))) { + if (hr == D3DERR_NOTAVAILABLE) { + *error_message = "D24S8 depth/stencil not available"; + d3d->Release(); + return false; + } + } + + DWORD dwBehaviorFlags = D3DCREATE_MULTITHREADED | D3DCREATE_FPU_PRESERVE; + if (d3dCaps.VertexProcessingCaps != 0) + dwBehaviorFlags |= D3DCREATE_HARDWARE_VERTEXPROCESSING; + else + dwBehaviorFlags |= D3DCREATE_SOFTWARE_VERTEXPROCESSING; + + RECT rc; + GetClientRect(hWnd, &rc); + int xres = rc.right - rc.left; + int yres = rc.bottom - rc.top; + + D3DPRESENT_PARAMETERS pp; + memset(&pp, 0, sizeof(pp)); + pp.BackBufferWidth = xres; + pp.BackBufferHeight = yres; + pp.BackBufferFormat = d3ddm.Format; + pp.MultiSampleType = D3DMULTISAMPLE_NONE; + pp.SwapEffect = D3DSWAPEFFECT_DISCARD; + pp.Windowed = windowed; + pp.hDeviceWindow = hWnd; + pp.EnableAutoDepthStencil = true; + pp.AutoDepthStencilFormat = D3DFMT_D24S8; + pp.PresentationInterval = (g_Config.bVSync) ? D3DPRESENT_INTERVAL_ONE : D3DPRESENT_INTERVAL_IMMEDIATE; + + if (has9Ex) { + if (windowed && IsWin7OrLater()) { + // This new flip mode gives higher performance. + // TODO: This makes it slower? + //pp.BackBufferCount = 2; + //pp.SwapEffect = D3DSWAPEFFECT_FLIPEX; + } + hr = d3dEx->CreateDeviceEx(adapterId, D3DDEVTYPE_HAL, hWnd, dwBehaviorFlags, &pp, NULL, &deviceEx); + device = deviceEx; + } else { + hr = d3d->CreateDevice(adapterId, D3DDEVTYPE_HAL, hWnd, dwBehaviorFlags, &pp, &device); + } + + if (FAILED(hr)) { + *error_message = "Failed to create D3D device"; + d3d->Release(); + return false; + } + + device->BeginScene(); + DX9::pD3Ddevice = device; + DX9::pD3DdeviceEx = deviceEx; + + LoadD3DX9Dynamic(); + + if (!DX9::CompileShaders(*error_message)) { + *error_message = "Unable to compile shaders: " + *error_message; + device->EndScene(); + device->Release(); + d3d->Release(); + DX9::pD3Ddevice = nullptr; + DX9::pD3DdeviceEx = nullptr; + device = nullptr; + UnloadD3DXDynamic(); + return false; + } + + DX9::fbo_init(d3d); + + if (deviceEx && IsWin7OrLater()) { + // TODO: This makes it slower? + //deviceEx->SetMaximumFrameLatency(1); + } + + return true; +} + +void D3D9_Resize(HWND window) { + // TODO! +} + +void D3D9_Shutdown() { + DX9::DestroyShaders(); + DX9::fbo_shutdown(); + device->EndScene(); + device->Release(); + d3d->Release(); + UnloadD3DXDynamic(); + DX9::pD3Ddevice = nullptr; + DX9::pD3DdeviceEx = nullptr; + device = nullptr; + hWnd = nullptr; +} diff --git a/Windows/resource.h b/Windows/resource.h index 3ab6aff6c0..c5f702807d 100644 --- a/Windows/resource.h +++ b/Windows/resource.h @@ -1,336 +1,336 @@ -// Used by ppsspp.rc -// - -#define VS_USER_DEFINED 100 -#define IDR_MENU1 101 -#define IDD_DISASM 102 -#define IDC_FUNCTIONLIST 103 -#define IDC_DISASMVIEW 104 -#define IDC_GOTOPC 105 -#define IDC_LEFTTABS 106 -#define IDC_RAM 107 -#define IDC_STEPOVER 108 -#define IDC_TABDATATYPE 109 -#define IDC_CALLSTACK 110 -#define ID_MEMVIEW_GOTOINDISASM 112 -#define ID_DISASM_DYNARECRESULTS 113 -#define IDI_PPSSPP 115 -#define IDD_CONFIG 116 -#define IDI_STOPDISABLE 118 -#define ID_DEBUG_DISASSEMBLY 119 -#define ID_DEBUG_REGISTERS 120 -#define WHEEL_DELTA 120 -#define ID_DEBUG_LOG 121 -#define ID_DEBUG_BREAKPOINTS 122 -#define ID_FILE_LOADSTATEFILE 126 -#define ID_FILE_SAVESTATEFILE 127 -#define ID_EMULATION_RESET 130 -#define IDD_ABOUTBOX 133 -#define ID_DEBUG_LOADMAPFILE 134 -#define ID_CONFIG_RESOLUTION 141 -#define ID_OPTIONS_FULLSCREEN 154 -#define ID_OPTIONS_SETTINGS 155 -#define ID_OPTIONS_SHOWERRORS 158 -#define ID_PLUGINS_LOADDEFAULTPLUGINS 159 -#define IDD_MEMORY 160 -#define ID_DEBUG_MEMORYVIEW 161 -#define IDR_ACCELS 162 -#define ID_FILE_BOOTDVD 166 -#define ID_OPTIONS_ENABLEFRAMEBUFFER 167 -#define IDR_POPUPMENUS 169 -#define ID_DEBUG_MEMORYCHECKS 173 -#define IDD_DIALOG2 186 -#define IDD_MEMORYSEARCH 187 -#define ID_DEBUG_MEMORYSEARCH 188 -#define ID_DEBUG_EXPERIMENT 189 -#define IDR_MENU2 190 -#define ID_DISASM_GOTOINMEMORYVIEW 197 -#define ID_DISASM_TOGGLEBREAKPOINT 198 -#define ID_MEMVIEW_DUMP 199 -#define ID_OPTIONS_LOGGPFIFO 200 -#define ID_VIEW_TOOLBAR202 202 -#define ID_VIEW_STATUSBAR 203 -#define ID_HELP_INDEX204 204 -#define ID_HELP_ 206 -#define ID_HELP_HOMEPAGE 208 -#define ID_DEBUG_COMPILESIGNATUREFILE 209 -#define ID_DEBUG_USESIGNATUREFILE 210 -#define ID_DEBUG_UNLOADALLSYMBOLS 211 -#define ID_DEBUG_RESETSYMBOLTABLE 212 -#define IDI_STOP 223 -#define IDD_INPUTBOX 226 -#define IDD_VFPU 231 -#define IDD_BREAKPOINT 233 -#define ID_FILE_LOAD_DIR 234 -#define IDR_DEBUGACCELS 237 -#define ID_DEBUG_DISPLAYMEMVIEW 238 -#define ID_DEBUG_DISPLAYBREAKPOINTLIST 239 -#define ID_DEBUG_DISPLAYTHREADLIST 240 -#define ID_DEBUG_DISPLAYSTACKFRAMELIST 241 -#define ID_DEBUG_ADDBREAKPOINT 242 -#define ID_DEBUG_STEPOVER 243 -#define ID_DEBUG_STEPINTO 244 -#define ID_DEBUG_RUNTOLINE 245 -#define ID_DEBUG_STEPOUT 246 -#define ID_DEBUG_DSIPLAYREGISTERLIST 247 -#define ID_DEBUG_DSIPLAYFUNCTIONLIST 248 -#define ID_MEMVIEW_COPYADDRESS 249 -#define IDD_GEDEBUGGER 250 -#define IDD_TABDISPLAYLISTS 251 -#define IDD_GEDBG_TAB_VALUES 252 -#define IDD_DUMPMEMORY 253 -#define IDD_GEDBG_TAB_VERTICES 254 -#define IDD_GEDBG_TAB_MATRICES 255 - -#define IDC_STOPGO 1001 -#define IDC_ADDRESS 1002 -#define IDC_DEBUG_COUNT 1003 -#define IDC_MEMORY 1006 -#define IDC_SH4REGISTERS 1007 -#define IDC_REGISTERS 1007 -#define IDC_BREAKPOINTS 1008 -#define IDC_STEP 1009 -#define IDC_VERSION 1010 -#define IDC_UP 1014 -#define IDC_DOWN 1015 -#define IDC_BREAKPOINTS_LIST 1015 -#define IDC_ADD 1016 -#define IDC_BREAKPOINT_EDIT 1017 -#define IDC_REMOVE 1018 -#define IDC_REMOVE_ALL 1019 -#define IDC_REGISTER_TAB 1019 -#define IDC_HIDE 1020 -#define IDC_TOGGLEBREAKPOINT 1049 -#define IDC_STAGE 1059 -#define IDC_MEMVIEW 1069 -#define IDC_GOTOLR 1070 -#define IDC_GOTOINT 1071 -#define IDC_MEMSORT 1073 -#define IDC_ALLFUNCTIONS 1075 -#define IDC_RESULTS 1093 -#define IDC_SYMBOLS 1097 -#define IDC_X86ASM 1098 -#define IDC_INPUTBOX 1098 -#define IDC_MODENORMAL 1099 -#define IDC_MODESYMBOLS 1100 -#define IDC_LOG_SHOW 1101 -#define IDC_UPDATELOG 1108 -#define IDC_SETPC 1118 -#define IDC_UPDATEMISC 1134 -#define IDC_CODEADDRESS 1135 -#define IDC_BLOCKNUMBER 1136 -#define IDC_PREVBLOCK 1138 -#define IDC_REGIONS 1142 -#define IDC_REGLIST 1146 -#define IDC_VALUENAME 1148 -#define IDC_FILELIST 1150 -#define IDC_BROWSE 1159 -#define IDC_SHOWVFPU 1161 -#define IDC_LISTCONTROLS 1162 -#define IDC_FORCE_INPUT_DEVICE 1163 -#define IDC_BREAKPOINTLIST 1164 -#define IDC_DEBUGMEMVIEW 1165 -#define IDC_BREAKPOINT_OK 1166 -#define IDC_BREAKPOINT_CANCEL 1167 -#define IDC_BREAKPOINT_ADDRESS 1168 -#define IDC_BREAKPOINT_SIZE 1169 -#define IDC_BREAKPOINT_CONDITION 1170 -#define IDC_BREAKPOINT_EXECUTE 1171 -#define IDC_BREAKPOINT_MEMORY 1172 -#define IDC_BREAKPOINT_READ 1173 -#define IDC_BREAKPOINT_WRITE 1174 -#define IDC_BREAKPOINT_ENABLED 1175 -#define IDC_BREAKPOINT_LOG 1176 -#define IDC_BREAKPOINT_ONCHANGE 1177 -#define IDC_THREADLIST 1178 -#define IDC_THREADNAME 1179 -#define IDC_DISASMSTATUSBAR 1180 -#define IDC_STACKFRAMES 1181 -#define IDC_GEDBG_VALUES 1182 -#define IDC_DUMP_USERMEMORY 1183 -#define IDC_DUMP_VRAM 1184 -#define IDC_DUMP_SCRATCHPAD 1185 -#define IDC_DUMP_CUSTOMRANGE 1186 -#define IDC_DUMP_STARTADDRESS 1187 -#define IDC_DUMP_SIZE 1188 -#define IDC_DUMP_FILENAME 1189 -#define IDC_DUMP_BROWSEFILENAME 1190 -#define IDC_GEDBG_FRAMEBUFADDR 1191 -#define IDC_GEDBG_TEXADDR 1192 -#define IDC_GEDBG_FBTABS 1193 -#define IDC_GEDBG_VERTICES 1194 -#define IDC_GEDBG_RAWVERTS 1195 -#define IDC_GEDBG_MATRICES 1196 - -#define ID_SHADERS_BASE 5000 - -#define ID_FILE_EXIT 40000 -#define ID_DEBUG_SAVEMAPFILE 40001 -#define ID_DISASM_ADDHLE 40002 -#define ID_FUNCLIST_KILLFUNCTION 40003 -#define ID_DISASM_RUNTOHERE 40004 -#define ID_MEMVIEW_COPYVALUE_8 40005 -#define ID_DISASM_COPYINSTRUCTIONDISASM 40006 -#define ID_DISASM_COPYINSTRUCTIONHEX 40007 -#define ID_EMULATION_SPEEDLIMIT 40008 -#define ID_TOGGLE_PAUSE 40009 -#define ID_EMULATION_STOP 40010 -#define ID_FILE_LOAD 40011 -#define ID_HELP_ABOUT 40012 -#define ID_DISASM_FOLLOWBRANCH 40013 -#define ID_DEBUG_IGNOREILLEGALREADS 40014 -#define ID_DISASM_COPYADDRESS 40015 -#define ID_REGLIST_GOTOINMEMORYVIEW 40016 -#define ID_REGLIST_COPYVALUE 40017 -#define ID_REGLIST_COPY 40018 -#define ID_REGLIST_GOTOINDISASM 40019 -#define ID_REGLIST_CHANGE 40020 -#define ID_DISASM_RENAMEFUNCTION 40021 -#define ID_DISASM_SETPCTOHERE 40022 -#define ID_HELP_OPENWEBSITE 40023 -#define ID_OPTIONS_SCREENAUTO 40024 -#define ID_OPTIONS_SCREEN1X 40025 -#define ID_OPTIONS_SCREEN2X 40026 -#define ID_OPTIONS_SCREEN3X 40027 -#define ID_OPTIONS_SCREEN4X 40028 -#define ID_OPTIONS_SCREEN5X 40029 -#define ID_OPTIONS_HARDWARETRANSFORM 40030 -#define IDC_STEPHLE 40032 -#define ID_OPTIONS_LINEARFILTERING 40033 -#define ID_FILE_QUICKSAVESTATE 40034 -#define ID_FILE_QUICKLOADSTATE 40035 -#define ID_FILE_QUICKSAVESTATE_HC 40036 -#define ID_FILE_QUICKLOADSTATE_HC 40037 -#define ID_OPTIONS_CONTROLS 40038 -#define ID_DEBUG_RUNONLOAD 40039 -#define ID_DEBUG_DUMPNEXTFRAME 40040 -#define ID_OPTIONS_VERTEXCACHE 40041 -#define ID_OPTIONS_SHOWFPS 40042 -#define ID_OPTIONS_STRETCHDISPLAY 40043 -#define ID_OPTIONS_FRAMESKIP 40044 -#define IDC_MEMCHECK 40045 -#define ID_FILE_MEMSTICK 40046 -#define ID_FILE_LOAD_MEMSTICK 40047 -#define ID_EMULATION_SOUND 40048 -#define ID_OPTIONS_MIPMAP 40049 -#define ID_TEXTURESCALING_OFF 40050 -#define ID_TEXTURESCALING_XBRZ 40051 -#define ID_TEXTURESCALING_HYBRID 40052 -#define ID_TEXTURESCALING_2X 40053 -#define ID_TEXTURESCALING_3X 40054 -#define ID_TEXTURESCALING_4X 40055 -#define ID_TEXTURESCALING_5X 40056 -#define ID_TEXTURESCALING_DEPOSTERIZE 40057 -#define ID_TEXTURESCALING_BICUBIC 40058 -#define ID_TEXTURESCALING_HYBRID_BICUBIC 40059 -#define IDB_IMAGE_PSP 40060 -#define IDC_STATIC_IMAGE_PSP 40061 -#define ID_CONTROLS_KEY_DISABLE 40062 -#define ID_OPTIONS_TOPMOST 40063 -#define ID_HELP_OPENFORUM 40064 -#define ID_OPTIONS_VSYNC 40065 -#define ID_DEBUG_TAKESCREENSHOT 40066 -#define ID_OPTIONS_TEXTUREFILTERING_AUTO 40067 -#define ID_OPTIONS_NEARESTFILTERING 40068 -#define ID_DISASM_DISASSEMBLETOFILE 40069 -#define ID_OPTIONS_LINEARFILTERING_CG 40070 -#define ID_DISASM_DISABLEBREAKPOINT 40071 -#define ID_DISASM_THREAD_FORCERUN 40072 -#define ID_DISASM_THREAD_KILL 40073 -#define ID_FILE_SAVESTATE_NEXT_SLOT 40074 -#define ID_FILE_SAVESTATE_NEXT_SLOT_HC 40075 -#define ID_OPTIONS_READFBOTOMEMORYGPU 40076 -#define ID_OPTIONS_READFBOTOMEMORYCPU 40077 -#define ID_OPTIONS_NONBUFFEREDRENDERING 40078 -#define ID_OPTIONS_FRAMESKIP_0 40079 -#define ID_OPTIONS_FRAMESKIP_1 40080 -#define ID_OPTIONS_FRAMESKIP_2 40081 -#define ID_OPTIONS_FRAMESKIP_3 40082 -#define ID_OPTIONS_FRAMESKIP_4 40083 -#define ID_OPTIONS_FRAMESKIP_5 40084 -#define ID_OPTIONS_FRAMESKIP_6 40085 -#define ID_OPTIONS_FRAMESKIP_7 40086 -#define ID_OPTIONS_FRAMESKIP_8 40087 -#define ID_OPTIONS_FRAMESKIP_AUTO 40088 -#define ID_OPTIONS_FRAMESKIPDUMMY 40089 -#define ID_OPTIONS_RESOLUTIONDUMMY 40090 -#define ID_DISASM_ASSEMBLE 40091 -#define ID_DISASM_ADDNEWBREAKPOINT 40092 -#define ID_DISASM_EDITBREAKPOINT 40093 -#define ID_EMULATION_CHEATS 40096 -#define ID_HELP_CHINESE_FORUM 40097 -#define ID_OPTIONS_MORE_SETTINGS 40098 -#define ID_FILE_SAVESTATE_SLOT_1 40099 -#define ID_FILE_SAVESTATE_SLOT_2 40100 -#define ID_FILE_SAVESTATE_SLOT_3 40101 -#define ID_FILE_SAVESTATE_SLOT_4 40102 -#define ID_FILE_SAVESTATE_SLOT_5 40103 -#define ID_OPTIONS_WINDOW1X 40104 -#define ID_OPTIONS_WINDOW2X 40105 -#define ID_OPTIONS_WINDOW3X 40106 -#define ID_OPTIONS_WINDOW4X 40107 -#define ID_OPTIONS_WINDOW5X 40108 -#define ID_OPTIONS_BUFFEREDRENDERING 40109 -#define ID_DEBUG_SHOWDEBUGSTATISTICS 40110 -#define ID_OPTIONS_SCREEN6X 40111 -#define ID_OPTIONS_SCREEN7X 40112 -#define ID_OPTIONS_SCREEN8X 40113 -#define ID_OPTIONS_SCREEN9X 40114 -#define ID_OPTIONS_SCREEN10X 40115 -#define ID_DEBUG_GEDEBUGGER 40116 -#define IDC_GEDBG_STEPDRAW 40117 -#define IDC_GEDBG_RESUME 40118 -#define IDC_GEDBG_FRAME 40119 -#define IDC_GEDBG_MAINTAB 40120 -#define IDC_GEDBG_TEX 40121 -#define IDC_GEDBG_STEP 40122 -#define IDC_GEDBG_LISTS_ALLLISTS 40123 -#define IDC_GEDBG_LISTS_STACK 40124 -#define IDC_GEDBG_LISTS_SELECTEDLIST 40125 -#define ID_OPTIONS_FXAA 40126 -#define IDC_DEBUG_BOTTOMTABS 40127 -#define ID_DEBUG_HIDEBOTTOMTABS 40128 -#define ID_DEBUG_TOGGLEBOTTOMTABTITLES 40129 -#define ID_GEDBG_SETSTALLADDR 40130 -#define ID_GEDBG_GOTOPC 40131 -#define ID_GEDBG_GOTOADDR 40132 -#define IDC_GEDBG_STEPTEX 40133 -#define IDC_GEDBG_STEPFRAME 40134 -#define IDC_GEDBG_BREAKTEX 40135 -#define ID_OPTIONS_PAUSE_FOCUS 40136 -#define ID_TEXTURESCALING_AUTO 40137 -#define IDC_GEDBG_STEPPRIM 40138 -#define ID_DISASM_ADDFUNCTION 40139 -#define ID_DISASM_REMOVEFUNCTION 40140 -#define ID_OPTIONS_LANGUAGE 40141 -#define ID_MEMVIEW_COPYVALUE_16 40142 -#define ID_MEMVIEW_COPYVALUE_32 40143 -#define ID_EMULATION_SWITCH_UMD 40144 -#define ID_DEBUG_EXTRACTFILE 40145 -#define ID_OPTIONS_IGNOREWINKEY 40146 -#define IDC_MODULELIST 40147 -#define IDC_GEDBG_TEXLEVELDOWN 40148 -#define IDC_GEDBG_TEXLEVELUP 40149 -#define ID_DEBUG_LOADSYMFILE 40150 -#define ID_DEBUG_SAVESYMFILE 40151 -#define ID_OPTIONS_BUFLINEARFILTER 40152 -#define ID_OPTIONS_BUFNEARESTFILTER 40153 -#define ID_OPTIONS_DIRECTX 40154 -#define ID_OPTIONS_OPENGL 40155 - -// Dummy option to let the buffered rendering hotkey cycle through all the options. -#define ID_OPTIONS_BUFFEREDRENDERINGDUMMY 40500 -#define IDC_STEPOUT 40501 -#define ID_HELP_BUYGOLD 40502 - -#define IDC_STATIC -1 - -// Next default values for new objects -#ifdef APSTUDIO_INVOKED -#ifndef APSTUDIO_READONLY_SYMBOLS -#define _APS_NEXT_RESOURCE_VALUE 256 -#define _APS_NEXT_COMMAND_VALUE 40152 -#define _APS_NEXT_CONTROL_VALUE 1197 -#define _APS_NEXT_SYMED_VALUE 101 -#endif -#endif +// Used by ppsspp.rc +// + +#define VS_USER_DEFINED 100 +#define IDR_MENU1 101 +#define IDD_DISASM 102 +#define IDC_FUNCTIONLIST 103 +#define IDC_DISASMVIEW 104 +#define IDC_GOTOPC 105 +#define IDC_LEFTTABS 106 +#define IDC_RAM 107 +#define IDC_STEPOVER 108 +#define IDC_TABDATATYPE 109 +#define IDC_CALLSTACK 110 +#define ID_MEMVIEW_GOTOINDISASM 112 +#define ID_DISASM_DYNARECRESULTS 113 +#define IDI_PPSSPP 115 +#define IDD_CONFIG 116 +#define IDI_STOPDISABLE 118 +#define ID_DEBUG_DISASSEMBLY 119 +#define ID_DEBUG_REGISTERS 120 +#define WHEEL_DELTA 120 +#define ID_DEBUG_LOG 121 +#define ID_DEBUG_BREAKPOINTS 122 +#define ID_FILE_LOADSTATEFILE 126 +#define ID_FILE_SAVESTATEFILE 127 +#define ID_EMULATION_RESET 130 +#define IDD_ABOUTBOX 133 +#define ID_DEBUG_LOADMAPFILE 134 +#define ID_CONFIG_RESOLUTION 141 +#define ID_OPTIONS_FULLSCREEN 154 +#define ID_OPTIONS_SETTINGS 155 +#define ID_OPTIONS_SHOWERRORS 158 +#define ID_PLUGINS_LOADDEFAULTPLUGINS 159 +#define IDD_MEMORY 160 +#define ID_DEBUG_MEMORYVIEW 161 +#define IDR_ACCELS 162 +#define ID_FILE_BOOTDVD 166 +#define ID_OPTIONS_ENABLEFRAMEBUFFER 167 +#define IDR_POPUPMENUS 169 +#define ID_DEBUG_MEMORYCHECKS 173 +#define IDD_DIALOG2 186 +#define IDD_MEMORYSEARCH 187 +#define ID_DEBUG_MEMORYSEARCH 188 +#define ID_DEBUG_EXPERIMENT 189 +#define IDR_MENU2 190 +#define ID_DISASM_GOTOINMEMORYVIEW 197 +#define ID_DISASM_TOGGLEBREAKPOINT 198 +#define ID_MEMVIEW_DUMP 199 +#define ID_OPTIONS_LOGGPFIFO 200 +#define ID_VIEW_TOOLBAR202 202 +#define ID_VIEW_STATUSBAR 203 +#define ID_HELP_INDEX204 204 +#define ID_HELP_ 206 +#define ID_HELP_HOMEPAGE 208 +#define ID_DEBUG_COMPILESIGNATUREFILE 209 +#define ID_DEBUG_USESIGNATUREFILE 210 +#define ID_DEBUG_UNLOADALLSYMBOLS 211 +#define ID_DEBUG_RESETSYMBOLTABLE 212 +#define IDI_STOP 223 +#define IDD_INPUTBOX 226 +#define IDD_VFPU 231 +#define IDD_BREAKPOINT 233 +#define ID_FILE_LOAD_DIR 234 +#define IDR_DEBUGACCELS 237 +#define ID_DEBUG_DISPLAYMEMVIEW 238 +#define ID_DEBUG_DISPLAYBREAKPOINTLIST 239 +#define ID_DEBUG_DISPLAYTHREADLIST 240 +#define ID_DEBUG_DISPLAYSTACKFRAMELIST 241 +#define ID_DEBUG_ADDBREAKPOINT 242 +#define ID_DEBUG_STEPOVER 243 +#define ID_DEBUG_STEPINTO 244 +#define ID_DEBUG_RUNTOLINE 245 +#define ID_DEBUG_STEPOUT 246 +#define ID_DEBUG_DSIPLAYREGISTERLIST 247 +#define ID_DEBUG_DSIPLAYFUNCTIONLIST 248 +#define ID_MEMVIEW_COPYADDRESS 249 +#define IDD_GEDEBUGGER 250 +#define IDD_TABDISPLAYLISTS 251 +#define IDD_GEDBG_TAB_VALUES 252 +#define IDD_DUMPMEMORY 253 +#define IDD_GEDBG_TAB_VERTICES 254 +#define IDD_GEDBG_TAB_MATRICES 255 + +#define IDC_STOPGO 1001 +#define IDC_ADDRESS 1002 +#define IDC_DEBUG_COUNT 1003 +#define IDC_MEMORY 1006 +#define IDC_SH4REGISTERS 1007 +#define IDC_REGISTERS 1007 +#define IDC_BREAKPOINTS 1008 +#define IDC_STEP 1009 +#define IDC_VERSION 1010 +#define IDC_UP 1014 +#define IDC_DOWN 1015 +#define IDC_BREAKPOINTS_LIST 1015 +#define IDC_ADD 1016 +#define IDC_BREAKPOINT_EDIT 1017 +#define IDC_REMOVE 1018 +#define IDC_REMOVE_ALL 1019 +#define IDC_REGISTER_TAB 1019 +#define IDC_HIDE 1020 +#define IDC_TOGGLEBREAKPOINT 1049 +#define IDC_STAGE 1059 +#define IDC_MEMVIEW 1069 +#define IDC_GOTOLR 1070 +#define IDC_GOTOINT 1071 +#define IDC_MEMSORT 1073 +#define IDC_ALLFUNCTIONS 1075 +#define IDC_RESULTS 1093 +#define IDC_SYMBOLS 1097 +#define IDC_X86ASM 1098 +#define IDC_INPUTBOX 1098 +#define IDC_MODENORMAL 1099 +#define IDC_MODESYMBOLS 1100 +#define IDC_LOG_SHOW 1101 +#define IDC_UPDATELOG 1108 +#define IDC_SETPC 1118 +#define IDC_UPDATEMISC 1134 +#define IDC_CODEADDRESS 1135 +#define IDC_BLOCKNUMBER 1136 +#define IDC_PREVBLOCK 1138 +#define IDC_REGIONS 1142 +#define IDC_REGLIST 1146 +#define IDC_VALUENAME 1148 +#define IDC_FILELIST 1150 +#define IDC_BROWSE 1159 +#define IDC_SHOWVFPU 1161 +#define IDC_LISTCONTROLS 1162 +#define IDC_FORCE_INPUT_DEVICE 1163 +#define IDC_BREAKPOINTLIST 1164 +#define IDC_DEBUGMEMVIEW 1165 +#define IDC_BREAKPOINT_OK 1166 +#define IDC_BREAKPOINT_CANCEL 1167 +#define IDC_BREAKPOINT_ADDRESS 1168 +#define IDC_BREAKPOINT_SIZE 1169 +#define IDC_BREAKPOINT_CONDITION 1170 +#define IDC_BREAKPOINT_EXECUTE 1171 +#define IDC_BREAKPOINT_MEMORY 1172 +#define IDC_BREAKPOINT_READ 1173 +#define IDC_BREAKPOINT_WRITE 1174 +#define IDC_BREAKPOINT_ENABLED 1175 +#define IDC_BREAKPOINT_LOG 1176 +#define IDC_BREAKPOINT_ONCHANGE 1177 +#define IDC_THREADLIST 1178 +#define IDC_THREADNAME 1179 +#define IDC_DISASMSTATUSBAR 1180 +#define IDC_STACKFRAMES 1181 +#define IDC_GEDBG_VALUES 1182 +#define IDC_DUMP_USERMEMORY 1183 +#define IDC_DUMP_VRAM 1184 +#define IDC_DUMP_SCRATCHPAD 1185 +#define IDC_DUMP_CUSTOMRANGE 1186 +#define IDC_DUMP_STARTADDRESS 1187 +#define IDC_DUMP_SIZE 1188 +#define IDC_DUMP_FILENAME 1189 +#define IDC_DUMP_BROWSEFILENAME 1190 +#define IDC_GEDBG_FRAMEBUFADDR 1191 +#define IDC_GEDBG_TEXADDR 1192 +#define IDC_GEDBG_FBTABS 1193 +#define IDC_GEDBG_VERTICES 1194 +#define IDC_GEDBG_RAWVERTS 1195 +#define IDC_GEDBG_MATRICES 1196 + +#define ID_SHADERS_BASE 5000 + +#define ID_FILE_EXIT 40000 +#define ID_DEBUG_SAVEMAPFILE 40001 +#define ID_DISASM_ADDHLE 40002 +#define ID_FUNCLIST_KILLFUNCTION 40003 +#define ID_DISASM_RUNTOHERE 40004 +#define ID_MEMVIEW_COPYVALUE_8 40005 +#define ID_DISASM_COPYINSTRUCTIONDISASM 40006 +#define ID_DISASM_COPYINSTRUCTIONHEX 40007 +#define ID_EMULATION_SPEEDLIMIT 40008 +#define ID_TOGGLE_PAUSE 40009 +#define ID_EMULATION_STOP 40010 +#define ID_FILE_LOAD 40011 +#define ID_HELP_ABOUT 40012 +#define ID_DISASM_FOLLOWBRANCH 40013 +#define ID_DEBUG_IGNOREILLEGALREADS 40014 +#define ID_DISASM_COPYADDRESS 40015 +#define ID_REGLIST_GOTOINMEMORYVIEW 40016 +#define ID_REGLIST_COPYVALUE 40017 +#define ID_REGLIST_COPY 40018 +#define ID_REGLIST_GOTOINDISASM 40019 +#define ID_REGLIST_CHANGE 40020 +#define ID_DISASM_RENAMEFUNCTION 40021 +#define ID_DISASM_SETPCTOHERE 40022 +#define ID_HELP_OPENWEBSITE 40023 +#define ID_OPTIONS_SCREENAUTO 40024 +#define ID_OPTIONS_SCREEN1X 40025 +#define ID_OPTIONS_SCREEN2X 40026 +#define ID_OPTIONS_SCREEN3X 40027 +#define ID_OPTIONS_SCREEN4X 40028 +#define ID_OPTIONS_SCREEN5X 40029 +#define ID_OPTIONS_HARDWARETRANSFORM 40030 +#define IDC_STEPHLE 40032 +#define ID_OPTIONS_LINEARFILTERING 40033 +#define ID_FILE_QUICKSAVESTATE 40034 +#define ID_FILE_QUICKLOADSTATE 40035 +#define ID_FILE_QUICKSAVESTATE_HC 40036 +#define ID_FILE_QUICKLOADSTATE_HC 40037 +#define ID_OPTIONS_CONTROLS 40038 +#define ID_DEBUG_RUNONLOAD 40039 +#define ID_DEBUG_DUMPNEXTFRAME 40040 +#define ID_OPTIONS_VERTEXCACHE 40041 +#define ID_OPTIONS_SHOWFPS 40042 +#define ID_OPTIONS_STRETCHDISPLAY 40043 +#define ID_OPTIONS_FRAMESKIP 40044 +#define IDC_MEMCHECK 40045 +#define ID_FILE_MEMSTICK 40046 +#define ID_FILE_LOAD_MEMSTICK 40047 +#define ID_EMULATION_SOUND 40048 +#define ID_OPTIONS_MIPMAP 40049 +#define ID_TEXTURESCALING_OFF 40050 +#define ID_TEXTURESCALING_XBRZ 40051 +#define ID_TEXTURESCALING_HYBRID 40052 +#define ID_TEXTURESCALING_2X 40053 +#define ID_TEXTURESCALING_3X 40054 +#define ID_TEXTURESCALING_4X 40055 +#define ID_TEXTURESCALING_5X 40056 +#define ID_TEXTURESCALING_DEPOSTERIZE 40057 +#define ID_TEXTURESCALING_BICUBIC 40058 +#define ID_TEXTURESCALING_HYBRID_BICUBIC 40059 +#define IDB_IMAGE_PSP 40060 +#define IDC_STATIC_IMAGE_PSP 40061 +#define ID_CONTROLS_KEY_DISABLE 40062 +#define ID_OPTIONS_TOPMOST 40063 +#define ID_HELP_OPENFORUM 40064 +#define ID_OPTIONS_VSYNC 40065 +#define ID_DEBUG_TAKESCREENSHOT 40066 +#define ID_OPTIONS_TEXTUREFILTERING_AUTO 40067 +#define ID_OPTIONS_NEARESTFILTERING 40068 +#define ID_DISASM_DISASSEMBLETOFILE 40069 +#define ID_OPTIONS_LINEARFILTERING_CG 40070 +#define ID_DISASM_DISABLEBREAKPOINT 40071 +#define ID_DISASM_THREAD_FORCERUN 40072 +#define ID_DISASM_THREAD_KILL 40073 +#define ID_FILE_SAVESTATE_NEXT_SLOT 40074 +#define ID_FILE_SAVESTATE_NEXT_SLOT_HC 40075 +#define ID_OPTIONS_READFBOTOMEMORYGPU 40076 +#define ID_OPTIONS_READFBOTOMEMORYCPU 40077 +#define ID_OPTIONS_NONBUFFEREDRENDERING 40078 +#define ID_OPTIONS_FRAMESKIP_0 40079 +#define ID_OPTIONS_FRAMESKIP_1 40080 +#define ID_OPTIONS_FRAMESKIP_2 40081 +#define ID_OPTIONS_FRAMESKIP_3 40082 +#define ID_OPTIONS_FRAMESKIP_4 40083 +#define ID_OPTIONS_FRAMESKIP_5 40084 +#define ID_OPTIONS_FRAMESKIP_6 40085 +#define ID_OPTIONS_FRAMESKIP_7 40086 +#define ID_OPTIONS_FRAMESKIP_8 40087 +#define ID_OPTIONS_FRAMESKIP_AUTO 40088 +#define ID_OPTIONS_FRAMESKIPDUMMY 40089 +#define ID_OPTIONS_RESOLUTIONDUMMY 40090 +#define ID_DISASM_ASSEMBLE 40091 +#define ID_DISASM_ADDNEWBREAKPOINT 40092 +#define ID_DISASM_EDITBREAKPOINT 40093 +#define ID_EMULATION_CHEATS 40096 +#define ID_HELP_CHINESE_FORUM 40097 +#define ID_OPTIONS_MORE_SETTINGS 40098 +#define ID_FILE_SAVESTATE_SLOT_1 40099 +#define ID_FILE_SAVESTATE_SLOT_2 40100 +#define ID_FILE_SAVESTATE_SLOT_3 40101 +#define ID_FILE_SAVESTATE_SLOT_4 40102 +#define ID_FILE_SAVESTATE_SLOT_5 40103 +#define ID_OPTIONS_WINDOW1X 40104 +#define ID_OPTIONS_WINDOW2X 40105 +#define ID_OPTIONS_WINDOW3X 40106 +#define ID_OPTIONS_WINDOW4X 40107 +#define ID_OPTIONS_WINDOW5X 40108 +#define ID_OPTIONS_BUFFEREDRENDERING 40109 +#define ID_DEBUG_SHOWDEBUGSTATISTICS 40110 +#define ID_OPTIONS_SCREEN6X 40111 +#define ID_OPTIONS_SCREEN7X 40112 +#define ID_OPTIONS_SCREEN8X 40113 +#define ID_OPTIONS_SCREEN9X 40114 +#define ID_OPTIONS_SCREEN10X 40115 +#define ID_DEBUG_GEDEBUGGER 40116 +#define IDC_GEDBG_STEPDRAW 40117 +#define IDC_GEDBG_RESUME 40118 +#define IDC_GEDBG_FRAME 40119 +#define IDC_GEDBG_MAINTAB 40120 +#define IDC_GEDBG_TEX 40121 +#define IDC_GEDBG_STEP 40122 +#define IDC_GEDBG_LISTS_ALLLISTS 40123 +#define IDC_GEDBG_LISTS_STACK 40124 +#define IDC_GEDBG_LISTS_SELECTEDLIST 40125 +#define ID_OPTIONS_FXAA 40126 +#define IDC_DEBUG_BOTTOMTABS 40127 +#define ID_DEBUG_HIDEBOTTOMTABS 40128 +#define ID_DEBUG_TOGGLEBOTTOMTABTITLES 40129 +#define ID_GEDBG_SETSTALLADDR 40130 +#define ID_GEDBG_GOTOPC 40131 +#define ID_GEDBG_GOTOADDR 40132 +#define IDC_GEDBG_STEPTEX 40133 +#define IDC_GEDBG_STEPFRAME 40134 +#define IDC_GEDBG_BREAKTEX 40135 +#define ID_OPTIONS_PAUSE_FOCUS 40136 +#define ID_TEXTURESCALING_AUTO 40137 +#define IDC_GEDBG_STEPPRIM 40138 +#define ID_DISASM_ADDFUNCTION 40139 +#define ID_DISASM_REMOVEFUNCTION 40140 +#define ID_OPTIONS_LANGUAGE 40141 +#define ID_MEMVIEW_COPYVALUE_16 40142 +#define ID_MEMVIEW_COPYVALUE_32 40143 +#define ID_EMULATION_SWITCH_UMD 40144 +#define ID_DEBUG_EXTRACTFILE 40145 +#define ID_OPTIONS_IGNOREWINKEY 40146 +#define IDC_MODULELIST 40147 +#define IDC_GEDBG_TEXLEVELDOWN 40148 +#define IDC_GEDBG_TEXLEVELUP 40149 +#define ID_DEBUG_LOADSYMFILE 40150 +#define ID_DEBUG_SAVESYMFILE 40151 +#define ID_OPTIONS_BUFLINEARFILTER 40152 +#define ID_OPTIONS_BUFNEARESTFILTER 40153 +#define ID_OPTIONS_DIRECTX 40154 +#define ID_OPTIONS_OPENGL 40155 + +// Dummy option to let the buffered rendering hotkey cycle through all the options. +#define ID_OPTIONS_BUFFEREDRENDERINGDUMMY 40500 +#define IDC_STEPOUT 40501 +#define ID_HELP_BUYGOLD 40502 + +#define IDC_STATIC -1 + +// Next default values for new objects +#ifdef APSTUDIO_INVOKED +#ifndef APSTUDIO_READONLY_SYMBOLS +#define _APS_NEXT_RESOURCE_VALUE 256 +#define _APS_NEXT_COMMAND_VALUE 40152 +#define _APS_NEXT_CONTROL_VALUE 1197 +#define _APS_NEXT_SYMED_VALUE 101 +#endif +#endif From 0039eab8782a483901e7085c79f40f02c2b95d94 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Fri, 26 Sep 2014 21:16:56 -0700 Subject: [PATCH 025/105] Offer to toggle graphics backends on any failure. So, even if Direct3D 9 fails. --- Windows/EmuThread.cpp | 34 +++++++++++++++++++++++++++++++--- Windows/OpenGLBase.cpp | 2 +- 2 files changed, 32 insertions(+), 4 deletions(-) diff --git a/Windows/EmuThread.cpp b/Windows/EmuThread.cpp index e1f70f4fd9..27ef8b92c4 100644 --- a/Windows/EmuThread.cpp +++ b/Windows/EmuThread.cpp @@ -3,11 +3,13 @@ #include "base/timeutil.h" #include "base/NativeApp.h" #include "base/mutex.h" +#include "i18n/i18n.h" #include "util/text/utf8.h" #include "Common/Log.h" #include "Common/StringUtils.h" #include "../Globals.h" #include "Windows/EmuThread.h" +#include "Windows/W32Util/Misc.h" #include "Windows/WndMainWindow.h" #include "Windows/resource.h" #include "Core/Reporting.h" @@ -126,12 +128,38 @@ unsigned int WINAPI TheThread(void *) std::string error_string; if (!host->InitGraphics(&error_string)) { + I18NCategory *err = GetI18NCategory("Error"); Reporting::ReportMessage("Graphics init error: %s", error_string.c_str()); - std::string full_error = StringFromFormat( "Failed initializing OpenGL. Try upgrading your graphics drivers.\n\nError message:\n\n%s", error_string.c_str()); - MessageBox(0, ConvertUTF8ToWString(full_error).c_str(), L"OpenGL Error", MB_OK | MB_ICONERROR); + + const char *defaultErrorOpenGL = "Failed initializing graphics. Try upgrading your graphics drivers.\n\nWould you like to try switching to DirectX 9?\n\nError message:"; + const char *defaultErrorDirect3D9 = "Failed initializing graphics. Try upgrading your graphics drivers.\n\nWould you like to try switching to OpenGL?\n\nError message:"; + const char *genericError; + int nextBackend = GPU_BACKEND_DIRECT3D9; + switch (g_Config.iGPUBackend) { + case GPU_BACKEND_DIRECT3D9: + nextBackend = GPU_BACKEND_OPENGL; + genericError = err->T("GenericDirect3D9Error", defaultErrorDirect3D9); + break; + case GPU_BACKEND_OPENGL: + default: + nextBackend = GPU_BACKEND_DIRECT3D9; + genericError = err->T("GenericOpenGLError", defaultErrorOpenGL); + break; + } + std::string full_error = StringFromFormat("%s\n\n%s", genericError, error_string.c_str()); + std::wstring title = ConvertUTF8ToWString(err->T("GenericGraphicsError", "Graphics Error")); + bool yes = IDYES == MessageBox(0, ConvertUTF8ToWString(full_error).c_str(), title.c_str(), MB_ICONERROR | MB_YESNO); ERROR_LOG(BOOT, full_error.c_str()); - // No safe way out without OpenGL. + if (yes) { + // Change the config to the alternative and restart. + g_Config.iGPUBackend = nextBackend; + g_Config.Save(); + + W32Util::ExitAndRestart(); + } + + // No safe way out without graphics. ExitProcess(1); } diff --git a/Windows/OpenGLBase.cpp b/Windows/OpenGLBase.cpp index ca77511688..97a97506b6 100644 --- a/Windows/OpenGLBase.cpp +++ b/Windows/OpenGLBase.cpp @@ -162,7 +162,7 @@ bool GL_Init(HWND window, std::string *error_message) { } // Avoid further error messages. Let's just bail, it's safe, and we can't continue. - ExitProcess(0); + ExitProcess(1); } if (GLEW_OK != glewInit()) { From 32fc4c76765a9a095e2bf99d37592198a4c3d463 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Fri, 26 Sep 2014 21:32:22 -0700 Subject: [PATCH 026/105] d3d9: Try harder to get a shader compile error. --- Common/CommonFuncs.h | 3 ++- Common/Misc.cpp | 20 ++++++++++++++++---- GPU/Directx9/helper/global.cpp | 33 +++++++++++++++++---------------- 3 files changed, 35 insertions(+), 21 deletions(-) diff --git a/Common/CommonFuncs.h b/Common/CommonFuncs.h index d9f9d219e5..049aa8c42b 100644 --- a/Common/CommonFuncs.h +++ b/Common/CommonFuncs.h @@ -98,5 +98,6 @@ extern "C" { // Call directly after the command or use the error num. // This function might change the error code. // Defined in Misc.cpp. -const char* GetLastErrorMsg(); +const char *GetLastErrorMsg(); +const char *GetStringErrorMsg(int errCode); diff --git a/Common/Misc.cpp b/Common/Misc.cpp index 3750bf19a7..f5393cc1f9 100644 --- a/Common/Misc.cpp +++ b/Common/Misc.cpp @@ -30,25 +30,37 @@ // Generic function to get last error message. // Call directly after the command or use the error num. // This function might change the error code. -const char* GetLastErrorMsg() +const char *GetLastErrorMsg() { +#ifndef _XBOX +#ifdef _WIN32 + return GetStringErrorMsg(GetLastError()); +#else + return GetStringErrorMsg(errno); +#endif +#else + return "GetLastErrorMsg"; +#endif +} + +const char *GetStringErrorMsg(int errCode) { static const size_t buff_size = 255; #ifndef _XBOX #ifdef _WIN32 static __declspec(thread) char err_str[buff_size] = {}; - FormatMessageA(FORMAT_MESSAGE_FROM_SYSTEM, NULL, GetLastError(), + FormatMessageA(FORMAT_MESSAGE_FROM_SYSTEM | FORMAT_MESSAGE_IGNORE_INSERTS, NULL, errCode, MAKELANGID(LANG_NEUTRAL, SUBLANG_DEFAULT), err_str, buff_size, NULL); #else static __thread char err_str[buff_size] = {}; // Thread safe (XSI-compliant) - strerror_r(errno, err_str, buff_size); + strerror_r(errCode, err_str, buff_size); #endif return err_str; #else - return "GetLastErrorMsg"; + return "GetStringErrorMsg"; #endif } diff --git a/GPU/Directx9/helper/global.cpp b/GPU/Directx9/helper/global.cpp index dbe08ff789..93940164ab 100644 --- a/GPU/Directx9/helper/global.cpp +++ b/GPU/Directx9/helper/global.cpp @@ -1,6 +1,7 @@ #include "global.h" #include "fbo.h" #include "thin3d/d3dx9_loader.h" +#include "Common/CommonFuncs.h" namespace DX9 { @@ -60,16 +61,14 @@ LPDIRECT3DVERTEXSHADER9 pFramebufferVertexShader = NULL; // Vertex Shader LPDIRECT3DPIXELSHADER9 pFramebufferPixelShader = NULL; // Pixel Shader bool CompilePixelShader(const char *code, LPDIRECT3DPIXELSHADER9 *pShader, LPD3DXCONSTANTTABLE *pShaderTable, std::string &errorMessage) { - ID3DXBuffer* pShaderCode = NULL; - ID3DXBuffer* pErrorMsg = NULL; - - HRESULT hr = E_FAIL; + ID3DXBuffer *pShaderCode = nullptr; + ID3DXBuffer *pErrorMsg = nullptr; // Compile pixel shader. - hr = dyn_D3DXCompileShader(code, + HRESULT hr = dyn_D3DXCompileShader(code, (UINT)strlen(code), - NULL, - NULL, + nullptr, + nullptr, "main", "ps_2_0", 0, @@ -80,11 +79,13 @@ bool CompilePixelShader(const char *code, LPDIRECT3DPIXELSHADER9 *pShader, LPD3D if (pErrorMsg) { errorMessage = (CHAR *)pErrorMsg->GetBufferPointer(); pErrorMsg->Release(); + } else if (FAILED(hr)) { + errorMessage = GetStringErrorMsg(hr); } else { errorMessage = ""; } - if (FAILED(hr)) { + if (FAILED(hr) || !pShaderCode) { if (pShaderCode) pShaderCode->Release(); return false; @@ -100,16 +101,14 @@ bool CompilePixelShader(const char *code, LPDIRECT3DPIXELSHADER9 *pShader, LPD3D } bool CompileVertexShader(const char *code, LPDIRECT3DVERTEXSHADER9 *pShader, LPD3DXCONSTANTTABLE *pShaderTable, std::string &errorMessage) { - ID3DXBuffer* pShaderCode = NULL; - ID3DXBuffer* pErrorMsg = NULL; - - HRESULT hr = E_FAIL; + ID3DXBuffer *pShaderCode = nullptr; + ID3DXBuffer *pErrorMsg = nullptr; // Compile pixel shader. - hr = dyn_D3DXCompileShader(code, + HRESULT hr = dyn_D3DXCompileShader(code, (UINT)strlen(code), - NULL, - NULL, + nullptr, + nullptr, "main", "vs_2_0", 0, @@ -120,11 +119,13 @@ bool CompileVertexShader(const char *code, LPDIRECT3DVERTEXSHADER9 *pShader, LPD if (pErrorMsg) { errorMessage = (CHAR *)pErrorMsg->GetBufferPointer(); pErrorMsg->Release(); + } else if (FAILED(hr)) { + errorMessage = GetStringErrorMsg(hr); } else { errorMessage = ""; } - if (FAILED(hr)) { + if (FAILED(hr) || !pShaderCode) { if (pShaderCode) pShaderCode->Release(); return false; From aad301a97ab703f53e0f3cf2966168d0484376ba Mon Sep 17 00:00:00 2001 From: daniel229 Date: Sat, 27 Sep 2014 14:00:37 +0800 Subject: [PATCH 027/105] Replace download frame in Boku no Natsuyasumi 2 and 4 --- Core/HLE/ReplaceTables.cpp | 10 ++++++++++ Core/MIPS/MIPSAnalyst.cpp | 1 + 2 files changed, 11 insertions(+) diff --git a/Core/HLE/ReplaceTables.cpp b/Core/HLE/ReplaceTables.cpp index 873d17fbd5..6d2844fd2c 100644 --- a/Core/HLE/ReplaceTables.cpp +++ b/Core/HLE/ReplaceTables.cpp @@ -720,6 +720,15 @@ static int Hook_soranokiseki_sc_download_frame() { return 0; } +static int Hook_bokunonatsuyasumi4_download_frame() { + const u32 fb_address = currentMIPS->r[MIPS_REG_A3]; + if (Memory::IsVRAMAddress(fb_address)) { + gpu->PerformMemoryDownload(fb_address, 0x00044000); + CBreakPoints::ExecMemCheck(fb_address, true, 0x00044000, currentMIPS->pc); + } + return 0; +} + // Can either replace with C functions or functions emitted in Asm/ArmAsm. static const ReplacementTableEntry entries[] = { // TODO: I think some games can be helped quite a bit by implementing the @@ -779,6 +788,7 @@ static const ReplacementTableEntry entries[] = { { "kagaku_no_ensemble_download_frame", &Hook_kagaku_no_ensemble_download_frame, 0, REPFLAG_HOOKENTER, 0x38 }, { "soranokiseki_fc_download_frame", &Hook_soranokiseki_fc_download_frame, 0, REPFLAG_HOOKENTER, 0x180 }, { "soranokiseki_sc_download_frame", &Hook_soranokiseki_sc_download_frame, 0, REPFLAG_HOOKENTER, }, + { "bokunonatsuyasumi4_download_frame", &Hook_bokunonatsuyasumi4_download_frame, 0, REPFLAG_HOOKENTER, 0x8C }, {} }; diff --git a/Core/MIPS/MIPSAnalyst.cpp b/Core/MIPS/MIPSAnalyst.cpp index fb90522bc4..693260f3b9 100644 --- a/Core/MIPS/MIPSAnalyst.cpp +++ b/Core/MIPS/MIPSAnalyst.cpp @@ -391,6 +391,7 @@ static const HardHashTableEntry hardcodedHashes[] = { { 0xddfa5a85937aa581, 32, "vdot_q", }, { 0xe0214719d8a0aa4e, 104, "strstr", }, { 0xe029f0699ca3a886, 76, "matrix300_transform_by", }, + { 0xe086d5c9ce89148f, 212, "bokunonatsuyasumi4_download_frame", }, // Boku no Natsuyasumi 2 and 4, { 0xe093c2b0194d52b3, 820, "ff1_battle_effect", }, // Final Fantasy 1 { 0xe1107cf3892724a0, 460, "_memalign_r", }, { 0xe1724e6e29209d97, 24, "vector_length_t_2", }, From 1d4bd6c6951de94268294f069cf7dfea0b3aba80 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Fri, 26 Sep 2014 23:44:04 -0700 Subject: [PATCH 028/105] Add a delay for creating fontlibs and fonts. Matches tests, low bound on the delay. --- Core/HLE/sceFont.cpp | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/Core/HLE/sceFont.cpp b/Core/HLE/sceFont.cpp index 0703e2175c..a7f51e86d3 100644 --- a/Core/HLE/sceFont.cpp +++ b/Core/HLE/sceFont.cpp @@ -765,7 +765,7 @@ u32 sceFontNewLib(u32 paramPtr, u32 errorCodePtr) { // The game should never see this value, the return value is replaced // by the action. Except if we disable the alloc, in this case we return // the handle correctly here. - return newLib->handle(); + return hleDelayResult(newLib->handle(), "new fontlib", 30000); } int sceFontDoneLib(u32 fontLibHandle) { @@ -801,7 +801,7 @@ u32 sceFontOpen(u32 libHandle, u32 index, u32 mode, u32 errorCodePtr) { LoadedFont *font = fontLib->OpenFont(internalFonts[index], openMode, *errorCode); if (font) { *errorCode = 0; - return font->Handle(); + return hleDelayResult(font->Handle(), "font open", 10000); } else { return 0; } From ad191cdd3a1539d2cf017f68f6de134c3aee9e52 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Fri, 26 Sep 2014 23:44:36 -0700 Subject: [PATCH 029/105] Correct error codes in sceFontOpenUserMemory(). --- Core/HLE/sceFont.cpp | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/Core/HLE/sceFont.cpp b/Core/HLE/sceFont.cpp index a7f51e86d3..7f5e8a886d 100644 --- a/Core/HLE/sceFont.cpp +++ b/Core/HLE/sceFont.cpp @@ -823,6 +823,11 @@ u32 sceFontOpenUserMemory(u32 libHandle, u32 memoryFontAddrPtr, u32 memoryFontLe FontLib *fontLib = GetFontLib(libHandle); if (!fontLib) { ERROR_LOG_REPORT(SCEFONT, "sceFontOpenUserMemory(%08x, %08x, %08x, %08x): bad font lib", libHandle, memoryFontAddrPtr, memoryFontLength, errorCodePtr); + *errorCode = ERROR_FONT_INVALID_LIBID; + return 0; + } + if (memoryFontLength == 0) { + ERROR_LOG_REPORT(SCEFONT, "sceFontOpenUserMemory(%08x, %08x, %08x, %08x): invalid size", libHandle, memoryFontAddrPtr, memoryFontLength, errorCodePtr); *errorCode = ERROR_FONT_INVALID_PARAMETER; return 0; } From 00491bb33b03cce0a82782beb6fdd52406c7d0ab Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Sat, 27 Sep 2014 00:13:11 -0700 Subject: [PATCH 030/105] Process msgdialog abort on Update(). Matches tests. --- Core/Dialog/PSPMsgDialog.cpp | 5 +++-- Core/Dialog/PSPMsgDialog.h | 3 ++- 2 files changed, 5 insertions(+), 3 deletions(-) diff --git a/Core/Dialog/PSPMsgDialog.cpp b/Core/Dialog/PSPMsgDialog.cpp index a31706ab0b..c2322b9e84 100755 --- a/Core/Dialog/PSPMsgDialog.cpp +++ b/Core/Dialog/PSPMsgDialog.cpp @@ -218,7 +218,7 @@ int PSPMsgDialog::Update(int animSpeed) { return SCE_ERROR_UTILITY_INVALID_STATUS; } - if ((flag & DS_ERROR)) { + if (flag & (DS_ERROR | DS_ABORT)) { ChangeStatus(SCE_UTILITY_STATUS_FINISHED, 0); } else { UpdateButtons(); @@ -290,7 +290,8 @@ int PSPMsgDialog::Abort() { if (GetStatus() != SCE_UTILITY_STATUS_RUNNING) { return SCE_ERROR_UTILITY_INVALID_STATUS; } else { - ChangeStatus(SCE_UTILITY_STATUS_FINISHED, 0); + // Status is not actually changed until Update(). + flag |= DS_ABORT; return 0; } } diff --git a/Core/Dialog/PSPMsgDialog.h b/Core/Dialog/PSPMsgDialog.h index 43df9f477a..a2f30a797d 100644 --- a/Core/Dialog/PSPMsgDialog.h +++ b/Core/Dialog/PSPMsgDialog.h @@ -85,7 +85,8 @@ private : DS_VALIDBUTTON = 0x20, DS_CANCELBUTTON = 0x40, DS_NOSOUND = 0x80, - DS_ERROR = 0x100 + DS_ERROR = 0x100, + DS_ABORT = 0x200, }; u32 flag; From 2c99baf29508c3d5f189384aa41408d91a0b958a Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Sat, 27 Sep 2014 00:13:27 -0700 Subject: [PATCH 031/105] Delay sceUtilityMsgDialogUpdate() per tests. This is an approximate value, but it should be close. --- Core/HLE/sceUtility.cpp | 2 ++ 1 file changed, 2 insertions(+) diff --git a/Core/HLE/sceUtility.cpp b/Core/HLE/sceUtility.cpp index e1589e4596..a6c4dbb6ac 100644 --- a/Core/HLE/sceUtility.cpp +++ b/Core/HLE/sceUtility.cpp @@ -305,6 +305,8 @@ int sceUtilityMsgDialogUpdate(int animSpeed) int ret = msgDialog.Update(animSpeed); DEBUG_LOG(SCEUTILITY,"%08x=sceUtilityMsgDialogUpdate(%i)", ret, animSpeed); + if (ret >= 0) + return hleDelayResult(ret, "msgdialog update", 800); return ret; } From 42fe8ee32e45c1f94cbb61ba66bfdbee425fc6a4 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Sat, 27 Sep 2014 09:04:24 -0700 Subject: [PATCH 032/105] Convert FormatMessage() to utf-8 to fix locale. Otherwise we get non-utf-8 garbage if the user isn't in a Latin codepage. --- Common/Misc.cpp | 12 ++++++++---- 1 file changed, 8 insertions(+), 4 deletions(-) diff --git a/Common/Misc.cpp b/Common/Misc.cpp index f5393cc1f9..5f80745b83 100644 --- a/Common/Misc.cpp +++ b/Common/Misc.cpp @@ -15,6 +15,7 @@ // Official SVN repository and contact information can be found at // http://code.google.com/p/dolphin-emu/ +#include "util/text/utf8.h" #include "Common.h" #include @@ -44,14 +45,17 @@ const char *GetLastErrorMsg() } const char *GetStringErrorMsg(int errCode) { - static const size_t buff_size = 255; + static const size_t buff_size = 1023; #ifndef _XBOX #ifdef _WIN32 - static __declspec(thread) char err_str[buff_size] = {}; + static __declspec(thread) wchar_t err_strw[buff_size] = {}; - FormatMessageA(FORMAT_MESSAGE_FROM_SYSTEM | FORMAT_MESSAGE_IGNORE_INSERTS, NULL, errCode, + FormatMessageW(FORMAT_MESSAGE_FROM_SYSTEM | FORMAT_MESSAGE_IGNORE_INSERTS, NULL, errCode, MAKELANGID(LANG_NEUTRAL, SUBLANG_DEFAULT), - err_str, buff_size, NULL); + err_strw, buff_size, NULL); + + static __declspec(thread) char err_str[buff_size] = {}; + snprintf(err_str, buff_size, ConvertWStringToUTF8(err_strw).c_str()); #else static __thread char err_str[buff_size] = {}; From e4792116a79240f00003c196d8e9fd45c2b64b10 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Mon, 10 Feb 2014 01:24:40 -0800 Subject: [PATCH 033/105] Initial attempt at a compat report screen. --- CMakeLists.txt | 1 + Core/Reporting.cpp | 9 +-- UI/MainScreen.cpp | 13 +++++ UI/MainScreen.h | 1 + UI/ReportScreen.cpp | 124 +++++++++++++++++++++++++++++++++++++++++ UI/ReportScreen.h | 37 ++++++++++++ UI/UI.vcxproj | 2 + UI/UI.vcxproj.filters | 7 +++ android/jni/Android.mk | 1 + 9 files changed, 191 insertions(+), 4 deletions(-) create mode 100644 UI/ReportScreen.cpp create mode 100644 UI/ReportScreen.h diff --git a/CMakeLists.txt b/CMakeLists.txt index 45b0837097..23fcbdf3fd 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -767,6 +767,7 @@ set(NativeAppSource UI/GamepadEmu.cpp UI/OnScreenDisplay.cpp UI/ControlMappingScreen.cpp + UI/ReportScreen.cpp UI/Store.cpp UI/CwCheatScreen.cpp UI/InstallZipScreen.cpp diff --git a/Core/Reporting.cpp b/Core/Reporting.cpp index 762ae893e7..5de5c56560 100644 --- a/Core/Reporting.cpp +++ b/Core/Reporting.cpp @@ -140,7 +140,7 @@ namespace Reporting return ++spamProtectionCount >= SPAM_LIMIT; } - bool SendReportRequest(const char *uri, const std::string &data, Buffer *output = NULL) + bool SendReportRequest(const char *uri, const std::string &data, const std::string &mimeType, Buffer *output = NULL) { bool result = false; net::AutoInit netInit; @@ -153,7 +153,7 @@ namespace Reporting if (http.Resolve(ServerHostname(), ServerPort())) { http.Connect(); - http.POST("/report/message", data, "application/x-www-form-urlencoded", output); + http.POST("/report/message", data, mimeType, output); http.Disconnect(); result = true; } @@ -289,7 +289,7 @@ namespace Reporting { Payload &payload = payloadBuffer[pos]; - UrlEncoder postdata; + MultipartFormDataEncoder postdata; AddSystemInfo(postdata); AddGameInfo(postdata); AddConfigInfo(postdata); @@ -303,7 +303,8 @@ namespace Reporting payload.string1.clear(); payload.string2.clear(); - SendReportRequest("/report/message", postdata.ToString()); + postdata.Finish(); + SendReportRequest("/report/message", postdata.ToString(), postdata.GetMimeType()); break; } diff --git a/UI/MainScreen.cpp b/UI/MainScreen.cpp index 45a93846be..81ba581525 100644 --- a/UI/MainScreen.cpp +++ b/UI/MainScreen.cpp @@ -44,6 +44,7 @@ #include "UI/CwCheatScreen.h" #include "UI/MiscScreens.h" #include "UI/ControlMappingScreen.h" +#include "UI/ReportScreen.h" #include "UI/Store.h" #include "UI/ui_atlas.h" #include "Core/Config.h" @@ -1176,6 +1177,13 @@ void GamePauseScreen::CreateViews() { if (g_Config.bEnableCheats) { rightColumnItems->Add(new Choice(i->T("Cheats")))->OnClick.Handle(this, &GamePauseScreen::OnCwCheat); } +#if 0 + // TODO, also might be nice to show overall compat rating here? + if (Reporting::IsEnabled()) { + I18NCategory *rp = GetI18NCategory("Reporting"); + rightColumnItems->Add(new Choice(rp->T("ReportButton", "Report Feedback")))->OnClick.Handle(this, &GamePauseScreen::OnReportFeedback); + } +#endif rightColumnItems->Add(new Spacer(25.0)); rightColumnItems->Add(new Choice(i->T("Exit to menu")))->OnClick.Handle(this, &GamePauseScreen::OnExitToMenu); @@ -1207,6 +1215,11 @@ UI::EventReturn GamePauseScreen::OnExitToMenu(UI::EventParams &e) { return UI::EVENT_DONE; } +UI::EventReturn GamePauseScreen::OnReportFeedback(UI::EventParams &e) { + screenManager()->push(new ReportScreen(gamePath_)); + return UI::EVENT_DONE; +} + UI::EventReturn GamePauseScreen::OnLoadState(UI::EventParams &e) { SaveState::LoadSlot(saveSlots_->GetSelection(), 0, 0); diff --git a/UI/MainScreen.h b/UI/MainScreen.h index d49bf6e519..cd586af0df 100644 --- a/UI/MainScreen.h +++ b/UI/MainScreen.h @@ -92,6 +92,7 @@ private: UI::EventReturn OnMainSettings(UI::EventParams &e); UI::EventReturn OnGameSettings(UI::EventParams &e); UI::EventReturn OnExitToMenu(UI::EventParams &e); + UI::EventReturn OnReportFeedback(UI::EventParams &e); UI::EventReturn OnSaveState(UI::EventParams &e); UI::EventReturn OnLoadState(UI::EventParams &e); diff --git a/UI/ReportScreen.cpp b/UI/ReportScreen.cpp new file mode 100644 index 0000000000..9c698a0c37 --- /dev/null +++ b/UI/ReportScreen.cpp @@ -0,0 +1,124 @@ +// Copyright (c) 2014- PPSSPP Project. + +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU General Public License as published by +// the Free Software Foundation, version 2.0 or later versions. + +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU General Public License 2.0 for more details. + +// A copy of the GPL 2.0 should have been included with the program. +// If not, see http://www.gnu.org/licenses/ + +// Official git repository and contact information can be found at +// https://github.com/hrydgard/ppsspp and http://www.ppsspp.org/. + +#include +#include "i18n/i18n.h" +#include "gfx_es2/draw_buffer.h" +#include "ui/ui_context.h" +#include "UI/ReportScreen.h" + +#include "Core/Reporting.h" +#include "Common/Log.h" + +using namespace UI; + +class RatingChoice : public LinearLayout { +public: + RatingChoice(const char *captionKey, int *value, LayoutParams *layoutParams = 0); + + Event OnChoice; + +private: + void AddChoice(int i, const std::string &title); + EventReturn OnChoiceClick(EventParams &e); + + LinearLayout *group_; + int *value_; +}; + +RatingChoice::RatingChoice(const char *captionKey, int *value, LayoutParams *layoutParams) + : LinearLayout(ORIENT_VERTICAL, layoutParams), value_(value) { + SetSpacing(-8.0f); + + I18NCategory *rp = GetI18NCategory("Reporting"); + group_ = new LinearLayout(ORIENT_HORIZONTAL); + Add(new InfoItem(rp->T(captionKey), "")); + Add(group_); + + group_->SetSpacing(0.0f); + AddChoice(0, rp->T("Bad")); + AddChoice(1, rp->T("OK")); + AddChoice(2, rp->T("Great")); +} + +void RatingChoice::AddChoice(int i, const std::string &title) { + auto c = group_->Add(new StickyChoice(title, "")); + c->OnClick.Handle(this, &RatingChoice::OnChoiceClick); + if (*value_ == i) + c->Press(); +} + +EventReturn RatingChoice::OnChoiceClick(EventParams &e) { + // Unstick the other choices that weren't clicked. + for (int i = 0; i < 3; i++) { + auto v = group_->GetViewByIndex(i); + if (v != e.v) { + static_cast(v)->Release(); + } else { + *value_ = i; + } + } + + EventParams e2; + e2.v = e.v; + e2.a = *value_; + // Dispatch immediately (we're already on the UI thread as we're in an event handler). + return OnChoice.Dispatch(e2); +} + +ReportScreen::ReportScreen(const std::string &gamePath) + : UIScreenWithGameBackground(gamePath), graphics_(-1), speed_(-1), gameplay_(-1) { +} + +EventReturn ReportScreen::HandleChoice(EventParams &e) { + submit_->SetEnabled(graphics_ >= 0 && speed_ >= 0 && gameplay_ >= 0); + return EVENT_DONE; +} + +void ReportScreen::CreateViews() { + I18NCategory *rp = GetI18NCategory("Reporting"); + I18NCategory *d = GetI18NCategory("Dialog"); + Margins actionMenuMargins(0, 100, 15, 0); + ViewGroup *leftColumn = new AnchorLayout(new LinearLayoutParams(1.0f)); + ViewGroup *leftColumnItems = new LinearLayout(ORIENT_VERTICAL); + ViewGroup *rightColumn = new ScrollView(ORIENT_VERTICAL, new LinearLayoutParams(300, FILL_PARENT, actionMenuMargins)); + LinearLayout *rightColumnItems = new LinearLayout(ORIENT_VERTICAL); + + leftColumnItems->Add(new InfoItem(rp->T("FeedbackDesc", "How's the emulation? Let us and the community know!"), "")); + + // TODO: screenshot + leftColumnItems->Add(new RatingChoice("Graphics", &graphics_))->OnChoice.Handle(this, &ReportScreen::HandleChoice); + leftColumnItems->Add(new RatingChoice("Speed", &speed_))->OnChoice.Handle(this, &ReportScreen::HandleChoice); + leftColumnItems->Add(new RatingChoice("Gameplay", &gameplay_))->OnChoice.Handle(this, &ReportScreen::HandleChoice); + + rightColumnItems->SetSpacing(0.0f); + // TODO: Handle. + rightColumnItems->Add(new Choice(rp->T("Open Browser"))); + submit_ = new Choice(rp->T("Submit Feedback")); + // TODO: Handle. + rightColumnItems->Add(submit_); + submit_->SetEnabled(graphics_ >= 0 && speed_ >= 0 && gameplay_ >= 0); + + root_ = new LinearLayout(ORIENT_HORIZONTAL); + root_->Add(leftColumn); + root_->Add(rightColumn); + + leftColumn->Add(leftColumnItems); + leftColumn->Add(new Choice(d->T("Back"), "", false, new AnchorLayoutParams(150, WRAP_CONTENT, 10, NONE, NONE, 10)))->OnClick.Handle(this, &UIScreen::OnBack); + + rightColumn->Add(rightColumnItems); +} \ No newline at end of file diff --git a/UI/ReportScreen.h b/UI/ReportScreen.h new file mode 100644 index 0000000000..75d1cef1ae --- /dev/null +++ b/UI/ReportScreen.h @@ -0,0 +1,37 @@ +// Copyright (c) 2014- PPSSPP Project. + +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU General Public License as published by +// the Free Software Foundation, version 2.0 or later versions. + +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU General Public License 2.0 for more details. + +// A copy of the GPL 2.0 should have been included with the program. +// If not, see http://www.gnu.org/licenses/ + +// Official git repository and contact information can be found at +// https://github.com/hrydgard/ppsspp and http://www.ppsspp.org/. + +#pragma once + +#include "base/functional.h" +#include "ui/ui_screen.h" +#include "ui/viewgroup.h" +#include "UI/MiscScreens.h" + +class ReportScreen : public UIScreenWithGameBackground { +public: + ReportScreen(const std::string &gamePath); + +protected: + UI::EventReturn HandleChoice(UI::EventParams &e); + virtual void CreateViews(); + + UI::Choice *submit_; + int graphics_; + int speed_; + int gameplay_; +}; diff --git a/UI/UI.vcxproj b/UI/UI.vcxproj index 9202890720..5589c1917e 100644 --- a/UI/UI.vcxproj +++ b/UI/UI.vcxproj @@ -32,6 +32,7 @@ + @@ -53,6 +54,7 @@ + diff --git a/UI/UI.vcxproj.filters b/UI/UI.vcxproj.filters index 8a3740e483..c118f00537 100644 --- a/UI/UI.vcxproj.filters +++ b/UI/UI.vcxproj.filters @@ -47,6 +47,9 @@ + + Screens + @@ -94,6 +97,10 @@ + + Screens + + diff --git a/android/jni/Android.mk b/android/jni/Android.mk index f1b1b6a762..9c7fdddc26 100644 --- a/android/jni/Android.mk +++ b/android/jni/Android.mk @@ -289,6 +289,7 @@ LOCAL_SRC_FILES := \ $(SRC)/UI/EmuScreen.cpp \ $(SRC)/UI/MainScreen.cpp \ $(SRC)/UI/MiscScreens.cpp \ + $(SRC)/UI/ReportScreen.cpp \ $(SRC)/UI/Store.cpp \ $(SRC)/UI/GamepadEmu.cpp \ $(SRC)/UI/GameInfoCache.cpp \ From d068622b4e448f1874c1f1fa68b311d6bbf61e0c Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Sat, 27 Sep 2014 14:59:37 -0700 Subject: [PATCH 034/105] Add selector for compatibility rating. Might just kill gameplay? --- UI/ReportScreen.cpp | 62 +++++++++++++++++++++++++++++++++++++-------- UI/ReportScreen.h | 1 + 2 files changed, 53 insertions(+), 10 deletions(-) diff --git a/UI/ReportScreen.cpp b/UI/ReportScreen.cpp index 9c698a0c37..71dfe64c05 100644 --- a/UI/ReportScreen.cpp +++ b/UI/ReportScreen.cpp @@ -32,11 +32,18 @@ public: Event OnChoice; -private: +protected: + virtual void SetupChoices(); + virtual int TotalChoices() { + return 3; + } void AddChoice(int i, const std::string &title); - EventReturn OnChoiceClick(EventParams &e); LinearLayout *group_; + +private: + EventReturn OnChoiceClick(EventParams &e); + int *value_; }; @@ -50,6 +57,11 @@ RatingChoice::RatingChoice(const char *captionKey, int *value, LayoutParams *lay Add(group_); group_->SetSpacing(0.0f); + SetupChoices(); +} + +void RatingChoice::SetupChoices() { + I18NCategory *rp = GetI18NCategory("Reporting"); AddChoice(0, rp->T("Bad")); AddChoice(1, rp->T("OK")); AddChoice(2, rp->T("Great")); @@ -64,7 +76,8 @@ void RatingChoice::AddChoice(int i, const std::string &title) { EventReturn RatingChoice::OnChoiceClick(EventParams &e) { // Unstick the other choices that weren't clicked. - for (int i = 0; i < 3; i++) { + int total = TotalChoices(); + for (int i = 0; i < total; i++) { auto v = group_->GetViewByIndex(i); if (v != e.v) { static_cast(v)->Release(); @@ -80,12 +93,38 @@ EventReturn RatingChoice::OnChoiceClick(EventParams &e) { return OnChoice.Dispatch(e2); } +class CompatRatingChoice : public RatingChoice { +public: + CompatRatingChoice(const char *captionKey, int *value, LayoutParams *layoutParams = 0); + +protected: + virtual void SetupChoices() override; + virtual int TotalChoices() override { + return 5; + } +}; + +CompatRatingChoice::CompatRatingChoice(const char *captionKey, int *value, LayoutParams *layoutParams) + : RatingChoice(captionKey, value, layoutParams) { + SetupChoices(); +} + +void CompatRatingChoice::SetupChoices() { + I18NCategory *rp = GetI18NCategory("Reporting"); + group_->Clear(); + AddChoice(1, rp->T("Perfect")); + AddChoice(2, rp->T("Plays")); + AddChoice(3, rp->T("In-game")); + AddChoice(4, rp->T("Menu/Intro")); + AddChoice(5, rp->T("Nothing")); +} + ReportScreen::ReportScreen(const std::string &gamePath) - : UIScreenWithGameBackground(gamePath), graphics_(-1), speed_(-1), gameplay_(-1) { + : UIScreenWithGameBackground(gamePath), overall_(-1), graphics_(-1), speed_(-1), gameplay_(-1) { } EventReturn ReportScreen::HandleChoice(EventParams &e) { - submit_->SetEnabled(graphics_ >= 0 && speed_ >= 0 && gameplay_ >= 0); + submit_->SetEnabled(overall_ >= 0 && graphics_ >= 0 && speed_ >= 0 && gameplay_ >= 0); return EVENT_DONE; } @@ -93,14 +132,15 @@ void ReportScreen::CreateViews() { I18NCategory *rp = GetI18NCategory("Reporting"); I18NCategory *d = GetI18NCategory("Dialog"); Margins actionMenuMargins(0, 100, 15, 0); - ViewGroup *leftColumn = new AnchorLayout(new LinearLayoutParams(1.0f)); - ViewGroup *leftColumnItems = new LinearLayout(ORIENT_VERTICAL); + ViewGroup *leftColumn = new ScrollView(ORIENT_VERTICAL, new LinearLayoutParams(WRAP_CONTENT, FILL_PARENT, 0.4f)); + LinearLayout *leftColumnItems = new LinearLayout(ORIENT_VERTICAL, new LayoutParams(WRAP_CONTENT, FILL_PARENT)); ViewGroup *rightColumn = new ScrollView(ORIENT_VERTICAL, new LinearLayoutParams(300, FILL_PARENT, actionMenuMargins)); LinearLayout *rightColumnItems = new LinearLayout(ORIENT_VERTICAL); leftColumnItems->Add(new InfoItem(rp->T("FeedbackDesc", "How's the emulation? Let us and the community know!"), "")); // TODO: screenshot + leftColumnItems->Add(new CompatRatingChoice("Overall", &overall_))->OnChoice.Handle(this, &ReportScreen::HandleChoice); leftColumnItems->Add(new RatingChoice("Graphics", &graphics_))->OnChoice.Handle(this, &ReportScreen::HandleChoice); leftColumnItems->Add(new RatingChoice("Speed", &speed_))->OnChoice.Handle(this, &ReportScreen::HandleChoice); leftColumnItems->Add(new RatingChoice("Gameplay", &gameplay_))->OnChoice.Handle(this, &ReportScreen::HandleChoice); @@ -111,14 +151,16 @@ void ReportScreen::CreateViews() { submit_ = new Choice(rp->T("Submit Feedback")); // TODO: Handle. rightColumnItems->Add(submit_); - submit_->SetEnabled(graphics_ >= 0 && speed_ >= 0 && gameplay_ >= 0); + submit_->SetEnabled(overall_ >= 0 && graphics_ >= 0 && speed_ >= 0 && gameplay_ >= 0); - root_ = new LinearLayout(ORIENT_HORIZONTAL); + rightColumnItems->Add(new Spacer(25.0)); + rightColumnItems->Add(new Choice(d->T("Back"), "", false, new AnchorLayoutParams(150, WRAP_CONTENT, 10, NONE, NONE, 10)))->OnClick.Handle(this, &UIScreen::OnBack); + + root_ = new LinearLayout(ORIENT_HORIZONTAL, new LinearLayoutParams(FILL_PARENT, FILL_PARENT, 1.0f)); root_->Add(leftColumn); root_->Add(rightColumn); leftColumn->Add(leftColumnItems); - leftColumn->Add(new Choice(d->T("Back"), "", false, new AnchorLayoutParams(150, WRAP_CONTENT, 10, NONE, NONE, 10)))->OnClick.Handle(this, &UIScreen::OnBack); rightColumn->Add(rightColumnItems); } \ No newline at end of file diff --git a/UI/ReportScreen.h b/UI/ReportScreen.h index 75d1cef1ae..8484d7844b 100644 --- a/UI/ReportScreen.h +++ b/UI/ReportScreen.h @@ -31,6 +31,7 @@ protected: virtual void CreateViews(); UI::Choice *submit_; + int overall_; int graphics_; int speed_; int gameplay_; From af822b1647ae06e1e323cd3820fc56213655bd25 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Sat, 27 Sep 2014 15:37:53 -0700 Subject: [PATCH 035/105] Add actual reporting of compatibility. --- Core/Reporting.cpp | 37 +++++++++++++++++++++++++++++++++++-- Core/Reporting.h | 3 +++ UI/MainScreen.cpp | 3 +-- UI/ReportScreen.cpp | 41 ++++++++++++++++++++++++++++++----------- UI/ReportScreen.h | 3 +++ 5 files changed, 72 insertions(+), 15 deletions(-) diff --git a/Core/Reporting.cpp b/Core/Reporting.cpp index 5de5c56560..c7e2b05f32 100644 --- a/Core/Reporting.cpp +++ b/Core/Reporting.cpp @@ -34,6 +34,7 @@ #include "net/resolve.h" #include "net/url.h" +#include "base/stringutil.h" #include "base/buffer.h" #include "thread/thread.h" #include "file/zip_read.h" @@ -62,6 +63,7 @@ namespace Reporting enum RequestType { MESSAGE, + COMPAT, }; struct Payload @@ -69,6 +71,9 @@ namespace Reporting RequestType type; std::string string1; std::string string2; + int int1; + int int2; + int int3; }; static Payload payloadBuffer[PAYLOAD_BUFFER_SIZE]; static int payloadBufferPos = 0; @@ -153,7 +158,7 @@ namespace Reporting if (http.Resolve(ServerHostname(), ServerPort())) { http.Connect(); - http.POST("/report/message", data, mimeType, output); + http.POST(uri, data, mimeType, output); http.Disconnect(); result = true; } @@ -289,7 +294,7 @@ namespace Reporting { Payload &payload = payloadBuffer[pos]; - MultipartFormDataEncoder postdata; + UrlEncoder postdata; AddSystemInfo(postdata); AddGameInfo(postdata); AddConfigInfo(postdata); @@ -306,6 +311,17 @@ namespace Reporting postdata.Finish(); SendReportRequest("/report/message", postdata.ToString(), postdata.GetMimeType()); break; + + case COMPAT: + postdata.Add("compat", payload.string1); + postdata.Add("graphics", StringFromFormat("%d", payload.int1)); + postdata.Add("speed", StringFromFormat("%d", payload.int2)); + postdata.Add("gameplay", StringFromFormat("%d", payload.int3)); + payload.string1.clear(); + + postdata.Finish(); + SendReportRequest("/report/compat", postdata.ToString(), postdata.GetMimeType()); + break; } return 0; @@ -387,4 +403,21 @@ namespace Reporting th.detach(); } + void ReportCompatibility(const char *compat, int graphics, int speed, int gameplay) + { + if (!IsEnabled()) + return; + + int pos = payloadBufferPos++ % PAYLOAD_BUFFER_SIZE; + Payload &payload = payloadBuffer[pos]; + payload.type = COMPAT; + payload.string1 = compat; + payload.int1 = graphics; + payload.int2 = speed; + payload.int3 = gameplay; + + std::thread th(Process, pos); + th.detach(); + } + } diff --git a/Core/Reporting.h b/Core/Reporting.h index 95c6aca4e7..a59ad9d7be 100644 --- a/Core/Reporting.h +++ b/Core/Reporting.h @@ -65,6 +65,9 @@ namespace Reporting // Report a message string, using the format string as a key. void ReportMessage(const char *message, ...); + // Report the compatibility of the current game / configuration. + void ReportCompatibility(const char *compat, int graphics, int speed, int gameplay); + // Returns true if that identifier has not been logged yet. bool ShouldLogOnce(const char *identifier); } \ No newline at end of file diff --git a/UI/MainScreen.cpp b/UI/MainScreen.cpp index 81ba581525..510021cf0a 100644 --- a/UI/MainScreen.cpp +++ b/UI/MainScreen.cpp @@ -1177,13 +1177,12 @@ void GamePauseScreen::CreateViews() { if (g_Config.bEnableCheats) { rightColumnItems->Add(new Choice(i->T("Cheats")))->OnClick.Handle(this, &GamePauseScreen::OnCwCheat); } -#if 0 // TODO, also might be nice to show overall compat rating here? + // Based on their platform or even cpu/gpu/config. Would add an API for it. if (Reporting::IsEnabled()) { I18NCategory *rp = GetI18NCategory("Reporting"); rightColumnItems->Add(new Choice(rp->T("ReportButton", "Report Feedback")))->OnClick.Handle(this, &GamePauseScreen::OnReportFeedback); } -#endif rightColumnItems->Add(new Spacer(25.0)); rightColumnItems->Add(new Choice(i->T("Exit to menu")))->OnClick.Handle(this, &GamePauseScreen::OnExitToMenu); diff --git a/UI/ReportScreen.cpp b/UI/ReportScreen.cpp index 71dfe64c05..bd3a3db497 100644 --- a/UI/ReportScreen.cpp +++ b/UI/ReportScreen.cpp @@ -90,7 +90,8 @@ EventReturn RatingChoice::OnChoiceClick(EventParams &e) { e2.v = e.v; e2.a = *value_; // Dispatch immediately (we're already on the UI thread as we're in an event handler). - return OnChoice.Dispatch(e2); + OnChoice.Dispatch(e2); + return EVENT_DONE; } class CompatRatingChoice : public RatingChoice { @@ -112,11 +113,11 @@ CompatRatingChoice::CompatRatingChoice(const char *captionKey, int *value, Layou void CompatRatingChoice::SetupChoices() { I18NCategory *rp = GetI18NCategory("Reporting"); group_->Clear(); - AddChoice(1, rp->T("Perfect")); - AddChoice(2, rp->T("Plays")); - AddChoice(3, rp->T("In-game")); - AddChoice(4, rp->T("Menu/Intro")); - AddChoice(5, rp->T("Nothing")); + AddChoice(0, rp->T("Perfect")); + AddChoice(1, rp->T("Plays")); + AddChoice(2, rp->T("In-game")); + AddChoice(3, rp->T("Menu/Intro")); + AddChoice(4, rp->T("Nothing")); } ReportScreen::ReportScreen(const std::string &gamePath) @@ -146,11 +147,9 @@ void ReportScreen::CreateViews() { leftColumnItems->Add(new RatingChoice("Gameplay", &gameplay_))->OnChoice.Handle(this, &ReportScreen::HandleChoice); rightColumnItems->SetSpacing(0.0f); - // TODO: Handle. - rightColumnItems->Add(new Choice(rp->T("Open Browser"))); + rightColumnItems->Add(new Choice(rp->T("Open Browser")))->OnClick.Handle(this, &ReportScreen::HandleBrowser); submit_ = new Choice(rp->T("Submit Feedback")); - // TODO: Handle. - rightColumnItems->Add(submit_); + rightColumnItems->Add(submit_)->OnClick.Handle(this, &ReportScreen::HandleSubmit); submit_->SetEnabled(overall_ >= 0 && graphics_ >= 0 && speed_ >= 0 && gameplay_ >= 0); rightColumnItems->Add(new Spacer(25.0)); @@ -161,6 +160,26 @@ void ReportScreen::CreateViews() { root_->Add(rightColumn); leftColumn->Add(leftColumnItems); - rightColumn->Add(rightColumnItems); +} + +EventReturn ReportScreen::HandleSubmit(EventParams &e) { + const char *compat; + switch (overall_) { + case 0: compat = "perfect"; break; + case 1: compat = "playable"; break; + case 2: compat = "ingame"; break; + case 3: compat = "menu"; break; + case 4: compat = "none"; break; + default: compat = "unknown"; break; + } + + Reporting::ReportCompatibility(compat, graphics_ + 1, speed_ + 1, gameplay_ + 1); + screenManager()->finishDialog(this, DR_OK); + return EVENT_DONE; +} + +EventReturn ReportScreen::HandleBrowser(EventParams &e) { + LaunchBrowser("http://report.ppsspp.org/"); + return EVENT_DONE; } \ No newline at end of file diff --git a/UI/ReportScreen.h b/UI/ReportScreen.h index 8484d7844b..affc643a40 100644 --- a/UI/ReportScreen.h +++ b/UI/ReportScreen.h @@ -28,6 +28,9 @@ public: protected: UI::EventReturn HandleChoice(UI::EventParams &e); + UI::EventReturn HandleSubmit(UI::EventParams &e); + UI::EventReturn HandleBrowser(UI::EventParams &e); + virtual void CreateViews(); UI::Choice *submit_; From eded13a21d30502f04508f60f20c14d27e311cf9 Mon Sep 17 00:00:00 2001 From: sum2012 Date: Sun, 28 Sep 2014 15:55:16 +0800 Subject: [PATCH 036/105] Add more information for directx 9 error --- Windows/EmuThread.cpp | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/Windows/EmuThread.cpp b/Windows/EmuThread.cpp index 27ef8b92c4..833384b278 100644 --- a/Windows/EmuThread.cpp +++ b/Windows/EmuThread.cpp @@ -132,7 +132,7 @@ unsigned int WINAPI TheThread(void *) Reporting::ReportMessage("Graphics init error: %s", error_string.c_str()); const char *defaultErrorOpenGL = "Failed initializing graphics. Try upgrading your graphics drivers.\n\nWould you like to try switching to DirectX 9?\n\nError message:"; - const char *defaultErrorDirect3D9 = "Failed initializing graphics. Try upgrading your graphics drivers.\n\nWould you like to try switching to OpenGL?\n\nError message:"; + const char *defaultErrorDirect3D9 = "Failed initializing graphics. Try upgrading your graphics drivers and directx 9 runtime.\n\nWould you like to try switching to OpenGL?\n\nError message:"; const char *genericError; int nextBackend = GPU_BACKEND_DIRECT3D9; switch (g_Config.iGPUBackend) { From feada0ee65f02f853aa400120ba11fd6d953e997 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Sun, 28 Sep 2014 15:13:10 -0700 Subject: [PATCH 037/105] Adjust some logging. Report logic op more correctly, cleanup an incorrect log. --- Core/HLE/sceKernelThread.cpp | 2 +- GPU/Directx9/GPU_DX9.cpp | 6 ------ GPU/Directx9/StateMappingDX9.cpp | 4 ++++ 3 files changed, 5 insertions(+), 7 deletions(-) diff --git a/Core/HLE/sceKernelThread.cpp b/Core/HLE/sceKernelThread.cpp index dd6d1e04b9..fa22c381a2 100644 --- a/Core/HLE/sceKernelThread.cpp +++ b/Core/HLE/sceKernelThread.cpp @@ -2559,7 +2559,7 @@ int sceKernelChangeCurrentThreadAttr(u32 clearAttr, u32 setAttr) // Seems like this is the only allowed attribute? if ((clearAttr & ~PSP_THREAD_ATTR_VFPU) != 0 || (setAttr & ~PSP_THREAD_ATTR_VFPU) != 0) { - ERROR_LOG_REPORT(SCEKERNEL, "0 = sceKernelChangeCurrentThreadAttr(clear = %08x, set = %08x): invalid attr", clearAttr, setAttr); + ERROR_LOG_REPORT(SCEKERNEL, "sceKernelChangeCurrentThreadAttr(clear = %08x, set = %08x): invalid attr", clearAttr, setAttr); return SCE_KERNEL_ERROR_ILLEGAL_ATTR; } diff --git a/GPU/Directx9/GPU_DX9.cpp b/GPU/Directx9/GPU_DX9.cpp index 02bd8b1faa..c2bace2de4 100644 --- a/GPU/Directx9/GPU_DX9.cpp +++ b/GPU/Directx9/GPU_DX9.cpp @@ -1658,13 +1658,7 @@ void DIRECTX9_GPU::Execute_Generic(u32 op, u32 diff) { break; case GE_CMD_LOGICOPENABLE: - if (data != 0) - ERROR_LOG_REPORT_ONCE(logicOpEnable, G3D, "Unsupported logic op enabled: %x", data); - break; - case GE_CMD_LOGICOP: - if (data != 0) - ERROR_LOG_REPORT_ONCE(logicOp, G3D, "Unsupported logic op: %06x", data); break; case GE_CMD_ANTIALIASENABLE: diff --git a/GPU/Directx9/StateMappingDX9.cpp b/GPU/Directx9/StateMappingDX9.cpp index c4523cc678..6c702a80ee 100644 --- a/GPU/Directx9/StateMappingDX9.cpp +++ b/GPU/Directx9/StateMappingDX9.cpp @@ -313,6 +313,10 @@ void TransformDrawEngineDX9::ApplyBlendState() { break; } + if (gstate.isLogicOpEnabled() && gstate.getLogicOp() != GE_LOGIC_COPY) { + WARN_LOG_REPORT_ONCE(logicOpBlend, G3D, "Logic op and blend enabled, unsupported."); + } + dxstate.blend.enable(); dxstate.blendSeparate.enable(); ResetShaderBlending(); From 58afdfac60135ed74cefa4f1a90d8d443c6a7fda Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Sat, 27 Sep 2014 20:24:14 -0700 Subject: [PATCH 038/105] Return an error for MOut on a stereo stream. It seems like it won't downmix, it returns an error. --- Core/HLE/sceAtrac.cpp | 80 ++++++++++++++++++++++++++++++++++--------- 1 file changed, 63 insertions(+), 17 deletions(-) diff --git a/Core/HLE/sceAtrac.cpp b/Core/HLE/sceAtrac.cpp index dbb76ce768..d7da4f5238 100644 --- a/Core/HLE/sceAtrac.cpp +++ b/Core/HLE/sceAtrac.cpp @@ -59,6 +59,7 @@ #define ATRAC_ERROR_INCORRECT_READ_SIZE 0x80630013 #define ATRAC_ERROR_BAD_SAMPLE 0x80630015 #define ATRAC_ERROR_ADD_DATA_IS_TOO_BIG 0x80630018 +#define ATRAC_ERROR_NOT_MONO 0x80630019 #define ATRAC_ERROR_NO_LOOP_INFORMATION 0x80630021 #define ATRAC_ERROR_SECOND_BUFFER_NOT_NEEDED 0x80630022 #define ATRAC_ERROR_BUFFER_IS_EMPTY 0x80630023 @@ -1533,43 +1534,73 @@ int sceAtracIsSecondBufferNeeded(int atracID) { return 0; } -int sceAtracSetMOutHalfwayBuffer(int atracID, u32 MOutHalfBuffer, u32 readSize, u32 MOutHalfBufferSize) { +int sceAtracSetMOutHalfwayBuffer(int atracID, u32 buffer, u32 readSize, u32 bufferSize) { Atrac *atrac = getAtrac(atracID); if (!atrac) { - ERROR_LOG(ME, "sceAtracSetMOutHalfwayBuffer(%i, %08x, %08x, %08x): bad atrac ID", atracID, MOutHalfBuffer, readSize, MOutHalfBufferSize); + ERROR_LOG(ME, "sceAtracSetMOutHalfwayBuffer(%i, %08x, %08x, %08x): bad atrac ID", atracID, buffer, readSize, bufferSize); return ATRAC_ERROR_BAD_ATRACID; } - INFO_LOG(ME, "sceAtracSetMOutHalfwayBuffer(%i, %08x, %08x, %08x)", atracID, MOutHalfBuffer, readSize, MOutHalfBufferSize); - if (readSize > MOutHalfBufferSize) + INFO_LOG(ME, "sceAtracSetMOutHalfwayBuffer(%i, %08x, %08x, %08x)", atracID, buffer, readSize, bufferSize); + if (readSize > bufferSize) return ATRAC_ERROR_INCORRECT_READ_SIZE; int ret = 0; if (atrac != NULL) { - atrac->first.addr = MOutHalfBuffer; + atrac->first.addr = buffer; atrac->first.size = readSize; ret = atrac->Analyze(); if (ret < 0) { - ERROR_LOG_REPORT(ME, "sceAtracSetMOutHalfwayBuffer(%i, %08x, %08x, %08x): bad data", atracID, MOutHalfBuffer, readSize, MOutHalfBufferSize); + ERROR_LOG_REPORT(ME, "sceAtracSetMOutHalfwayBuffer(%i, %08x, %08x, %08x): bad data", atracID, buffer, readSize, bufferSize); return ret; } - atrac->atracOutputChannels = 1; - ret = _AtracSetData(atracID, MOutHalfBuffer, MOutHalfBufferSize); + if (atrac->atracChannels != 1) { + ERROR_LOG_REPORT(ME, "sceAtracSetMOutHalfwayBuffer(%i, %08x, %08x, %08x): not mono data", atracID, buffer, readSize, bufferSize); + ret = ATRAC_ERROR_NOT_MONO; + // It seems it still sets the data. + atrac->atracOutputChannels = 2; + _AtracSetData(atrac, buffer, bufferSize); + return ret; + } else { + atrac->atracOutputChannels = 1; + ret = _AtracSetData(atracID, buffer, bufferSize); + } } return ret; } + u32 sceAtracSetMOutData(int atracID, u32 buffer, u32 bufferSize) { - INFO_LOG(ME, "sceAtracSetMOutData(%i, %08x, %08x)", atracID, buffer, bufferSize); Atrac *atrac = getAtrac(atracID); + if (!atrac) { + ERROR_LOG(ME, "sceAtracSetMOutData(%i, %08x, %08x): bad atrac ID", atracID, buffer, bufferSize); + return ATRAC_ERROR_BAD_ATRACID; + } + + // This doesn't seem to be part of any available libatrac3plus library. + WARN_LOG_REPORT(ME, "sceAtracSetMOutData(%i, %08x, %08x)", atracID, buffer, bufferSize); + // TODO: What is the proper error code here? int ret = 0; if (atrac != NULL) { atrac->first.addr = buffer; atrac->first.size = bufferSize; - // TODO: Error code for bad data (probably yes)? - atrac->Analyze(); - atrac->atracOutputChannels = 1; - ret = _AtracSetData(atracID, buffer, bufferSize); + ret = atrac->Analyze(); + if (ret < 0) { + ERROR_LOG_REPORT(ME, "sceAtracSetMOutData(%i, %08x, %08x): bad data", atracID, buffer, bufferSize); + return ret; + } + if (atrac->atracChannels != 1) { + ERROR_LOG_REPORT(ME, "sceAtracSetMOutData(%i, %08x, %08x): not mono data", atracID, buffer, bufferSize); + ret = ATRAC_ERROR_NOT_MONO; + // It seems it still sets the data. + atrac->atracOutputChannels = 2; + _AtracSetData(atrac, buffer, bufferSize); + // Not sure of the real delay time. + return ret; + } else { + atrac->atracOutputChannels = 1; + ret = _AtracSetData(atracID, buffer, bufferSize); + } } return ret; } @@ -1583,8 +1614,17 @@ int sceAtracSetMOutDataAndGetID(u32 buffer, u32 bufferSize) { Atrac *atrac = new Atrac(); atrac->first.addr = buffer; atrac->first.size = bufferSize; - // TODO: Error code for bad data (probably yes)? - atrac->Analyze(); + int ret = atrac->Analyze(); + if (ret < 0) { + ERROR_LOG_REPORT(ME, "sceAtracSetMOutDataAndGetID(%08x, %08x): bad data", buffer, bufferSize); + delete atrac; + return ret; + } + if (atrac->atracChannels != 1) { + ERROR_LOG_REPORT(ME, "sceAtracSetMOutDataAndGetID(%08x, %08x): not mono data", buffer, bufferSize); + delete atrac; + return ATRAC_ERROR_NOT_MONO; + } atrac->atracOutputChannels = 1; int atracID = createAtrac(atrac, codecType); if (atracID < 0) { @@ -1592,8 +1632,9 @@ int sceAtracSetMOutDataAndGetID(u32 buffer, u32 bufferSize) { delete atrac; return atracID; } - INFO_LOG(ME, "%d=sceAtracSetMOutDataAndGetID(%08x, %08x)", atracID, buffer, bufferSize); - int ret = _AtracSetData(atracID, buffer, bufferSize, true); + // This doesn't seem to be part of any available libatrac3plus library. + WARN_LOG_REPORT(ME, "%d=sceAtracSetMOutDataAndGetID(%08x, %08x)", atracID, buffer, bufferSize); + ret = _AtracSetData(atracID, buffer, bufferSize, true); if (ret < 0) return ret; return atracID; @@ -1618,6 +1659,11 @@ int sceAtracSetMOutHalfwayBufferAndGetID(u32 halfBuffer, u32 readSize, u32 halfB delete atrac; return ret; } + if (atrac->atracChannels != 1) { + ERROR_LOG_REPORT(ME, "sceAtracSetMOutHalfwayBufferAndGetID(%08x, %08x, %08x): not mono data", halfBuffer, readSize, halfBufferSize); + delete atrac; + return ATRAC_ERROR_NOT_MONO; + } atrac->atracOutputChannels = 1; int atracID = createAtrac(atrac, codecType); if (atracID < 0) { From 3856a535032ac0d3d4eff82e9a1060d7495449d7 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Sat, 27 Sep 2014 20:24:50 -0700 Subject: [PATCH 039/105] Handle atrac files with larger "fact" chunks. Ends up with the a separate offset for loops, it seems like. This corrects looping information for these files. --- Core/HLE/sceAtrac.cpp | 15 ++++++++++----- 1 file changed, 10 insertions(+), 5 deletions(-) diff --git a/Core/HLE/sceAtrac.cpp b/Core/HLE/sceAtrac.cpp index d7da4f5238..cd8b78ea11 100644 --- a/Core/HLE/sceAtrac.cpp +++ b/Core/HLE/sceAtrac.cpp @@ -440,6 +440,7 @@ int Atrac::Analyze() { first.filesize = Memory::Read_U32(first.addr + 4) + 8; u32 offset = 12; + int loopFirstSampleOffset = 0; firstSampleoffset = 0; this->decodeEnd = first.filesize; @@ -471,9 +472,13 @@ int Atrac::Analyze() { break; case FACT_CHUNK_MAGIC: { - if (chunkSize >= 8) { - endSample = Memory::Read_U32(first.addr + offset); - firstSampleoffset = Memory::Read_U32(first.addr + offset + 4); + endSample = Memory::Read_U32(first.addr + offset); + firstSampleoffset = Memory::Read_U32(first.addr + offset + 4); + if (chunkSize >= 12) { + // Seems like this indicates it's got a separate offset for loops. + loopFirstSampleOffset = Memory::Read_U32(first.addr + offset + 8); + } else if (chunkSize >= 8) { + loopFirstSampleOffset = firstSampleoffset; } } break; @@ -489,8 +494,8 @@ int Atrac::Analyze() { for (int i = 0; i < loopinfoNum; i++, loopinfoAddr += 24) { loopinfo[i].cuePointID = Memory::Read_U32(loopinfoAddr); loopinfo[i].type = Memory::Read_U32(loopinfoAddr + 4); - loopinfo[i].startSample = Memory::Read_U32(loopinfoAddr + 8) - firstSampleoffset; - loopinfo[i].endSample = Memory::Read_U32(loopinfoAddr + 12) - firstSampleoffset; + loopinfo[i].startSample = Memory::Read_U32(loopinfoAddr + 8) - loopFirstSampleOffset; + loopinfo[i].endSample = Memory::Read_U32(loopinfoAddr + 12) - loopFirstSampleOffset; loopinfo[i].fraction = Memory::Read_U32(loopinfoAddr + 16); loopinfo[i].playCount = Memory::Read_U32(loopinfoAddr + 20); From 2221951cd9803c1c21a46eaba4dd79313a176fcd Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Sat, 27 Sep 2014 20:52:44 -0700 Subject: [PATCH 040/105] Correct atrac looping offset by one frame. --- Core/HLE/sceAtrac.cpp | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/Core/HLE/sceAtrac.cpp b/Core/HLE/sceAtrac.cpp index cd8b78ea11..9bf7151def 100644 --- a/Core/HLE/sceAtrac.cpp +++ b/Core/HLE/sceAtrac.cpp @@ -737,7 +737,7 @@ u32 _AtracDecodeData(int atracID, u8 *outbuf, u32 outbufPtr, u32 *SamplesNum, u3 atrac->decodePos = atrac->getDecodePosBySample(atrac->currentSample); int finishFlag = 0; - if (atrac->loopNum != 0 && (atrac->currentSample + (int)atracSamplesPerFrame > atrac->loopEndSample || + if (atrac->loopNum != 0 && (atrac->currentSample > atrac->loopEndSample || (numSamples == 0 && atrac->first.size >= atrac->first.filesize))) { atrac->currentSample = atrac->loopStartSample; if (atrac->loopNum > 0) From 5cb9bddea9addabc259a9531faa7dfc30752ba68 Mon Sep 17 00:00:00 2001 From: rock88 Date: Thu, 2 Oct 2014 21:05:43 +0700 Subject: [PATCH 041/105] iOS: update few compiler path --- ios/ios.toolchain.cmake | 2 ++ 1 file changed, 2 insertions(+) diff --git a/ios/ios.toolchain.cmake b/ios/ios.toolchain.cmake index 90a13eb2b5..f161f2930e 100644 --- a/ios/ios.toolchain.cmake +++ b/ios/ios.toolchain.cmake @@ -49,6 +49,8 @@ endif (CMAKE_UNAME) include (CMakeForceCompiler) CMAKE_FORCE_C_COMPILER (gcc gcc) CMAKE_FORCE_CXX_COMPILER (g++ g++) +CMAKE_FORCE_C_COMPILER (/usr/bin/clang Apple) +CMAKE_FORCE_CXX_COMPILER (/usr/bin/clang++ Apple) # Skip the platform compiler checks for cross compiling set (CMAKE_CROSSCOMPILING TRUE) From 4dc6e268014d229fa7aa12cd3a5b163fbe15ff9f Mon Sep 17 00:00:00 2001 From: rock88 Date: Thu, 2 Oct 2014 21:49:06 +0700 Subject: [PATCH 042/105] iOS: add LaunchScreen.xib for support iPhone 6 and 6 Plus native screen resolution --- ios/assets/LaunchScreen.xib | 18 ++++++++++++++++++ 1 file changed, 18 insertions(+) create mode 100644 ios/assets/LaunchScreen.xib diff --git a/ios/assets/LaunchScreen.xib b/ios/assets/LaunchScreen.xib new file mode 100644 index 0000000000..70ae06b281 --- /dev/null +++ b/ios/assets/LaunchScreen.xib @@ -0,0 +1,18 @@ + + + + + + + + + + + + + + + + + + From 39716c68b3da13de88fe4b7c5ba068d70c84c9a8 Mon Sep 17 00:00:00 2001 From: rock88 Date: Thu, 2 Oct 2014 21:49:49 +0700 Subject: [PATCH 043/105] iOS: install LaunchScreen.xib --- CMakeLists.txt | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/CMakeLists.txt b/CMakeLists.txt index 45b0837097..7d74a39b7e 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -1498,7 +1498,12 @@ if (TargetBin) set_source_files_properties(${LANG_FILES} PROPERTIES MACOSX_PACKAGE_LOCATION "MacOS/assets/lang") set_source_files_properties(${SHADER_FILES} PROPERTIES MACOSX_PACKAGE_LOCATION "MacOS/assets/shaders") - add_executable(${TargetBin} MACOSX_BUNDLE ${ICON_PATH_ABS} ${NativeAssets} ${SHADER_FILES} ${FLASH0_FILES} ${LANG_FILES} ${NativeAppSource}) + if (IOS) + set(LAUNCH_SCREEN_XIB ${CMAKE_CURRENT_SOURCE_DIR}/ios/assets/LaunchScreen.xib) + set_source_files_properties(${LAUNCH_SCREEN_XIB} PROPERTIES MACOSX_PACKAGE_LOCATION "Resources") + endif() + + add_executable(${TargetBin} MACOSX_BUNDLE ${ICON_PATH_ABS} ${NativeAssets} ${SHADER_FILES} ${FLASH0_FILES} ${LANG_FILES} ${NativeAppSource} ${LAUNCH_SCREEN_XIB}) else() add_executable(${TargetBin} ${NativeAppSource}) endif() From 70fac3a65e3d9979e53d03a1a313b914a9969b10 Mon Sep 17 00:00:00 2001 From: daniel229 Date: Fri, 3 Oct 2014 15:16:12 +0800 Subject: [PATCH 044/105] savedata --- Core/Dialog/PSPSaveDialog.cpp | 12 ++++++------ Core/Dialog/PSPSaveDialog.h | 2 ++ Core/Dialog/SavedataParam.cpp | 31 ++++++++++++------------------- Core/Dialog/SavedataParam.h | 3 ++- 4 files changed, 22 insertions(+), 26 deletions(-) diff --git a/Core/Dialog/PSPSaveDialog.cpp b/Core/Dialog/PSPSaveDialog.cpp index d0e61c4883..c2b83b8969 100755 --- a/Core/Dialog/PSPSaveDialog.cpp +++ b/Core/Dialog/PSPSaveDialog.cpp @@ -1108,14 +1108,14 @@ void PSPSaveDialog::ExecuteNotVisibleIOAction() { break; case SCE_UTILITY_SAVEDATA_TYPE_READDATA: case SCE_UTILITY_SAVEDATA_TYPE_READDATASECURE: - if (param.Load(param.GetPspParam(), GetSelectedSaveDirName(), currentSelectedSave, param.GetPspParam()->mode == SCE_UTILITY_SAVEDATA_TYPE_READDATASECURE)) { - param.GetPspParam()->common.result = 0; - } else if (param.secureCanSkip(param.GetPspParam(),param.GetPspParam()->mode == SCE_UTILITY_SAVEDATA_TYPE_READDATASECURE)) { - // TODO: This makes loading/saving work in some games but also confuses them. Must be wrong in some way. - INFO_LOG(SCEUTILITY,"Has not been saved yet, just skip."); + if (!param.IsSaveDirectoryExist(param.GetPspParam())){ + param.GetPspParam()->common.result = SCE_UTILITY_SAVEDATA_ERROR_RW_NO_DATA; + } else if (!param.IsSfoFileExist(param.GetPspParam())) { + param.GetPspParam()->common.result = SCE_UTILITY_SAVEDATA_ERROR_RW_DATA_BROKEN; + } else if (param.Load(param.GetPspParam(), GetSelectedSaveDirName(), currentSelectedSave, param.GetPspParam()->mode == SCE_UTILITY_SAVEDATA_TYPE_READDATASECURE)) { param.GetPspParam()->common.result = 0; } else { - param.GetPspParam()->common.result = SCE_UTILITY_SAVEDATA_ERROR_RW_NO_DATA; // not sure if correct code + param.GetPspParam()->common.result = SCE_UTILITY_SAVEDATA_ERROR_RW_FILE_NOT_FOUND; } break; default: diff --git a/Core/Dialog/PSPSaveDialog.h b/Core/Dialog/PSPSaveDialog.h index dc39b7b917..3f1f2e8547 100644 --- a/Core/Dialog/PSPSaveDialog.h +++ b/Core/Dialog/PSPSaveDialog.h @@ -33,8 +33,10 @@ #define SCE_UTILITY_SAVEDATA_ERROR_LOAD_INTERNAL (0x8011030b) #define SCE_UTILITY_SAVEDATA_ERROR_RW_NO_MEMSTICK (0x80110321) +#define SCE_UTILITY_SAVEDATA_ERROR_RW_DATA_BROKEN (0x80110326) #define SCE_UTILITY_SAVEDATA_ERROR_RW_NO_DATA (0x80110327) #define SCE_UTILITY_SAVEDATA_ERROR_RW_BAD_PARAMS (0x80110328) +#define SCE_UTILITY_SAVEDATA_ERROR_RW_FILE_NOT_FOUND (0x80110329) #define SCE_UTILITY_SAVEDATA_ERROR_RW_BAD_STATUS (0x8011032c) #define SCE_UTILITY_SAVEDATA_ERROR_SAVE_NO_MS (0x80110381) diff --git a/Core/Dialog/SavedataParam.cpp b/Core/Dialog/SavedataParam.cpp index 760fa3dd83..e83d0dab60 100644 --- a/Core/Dialog/SavedataParam.cpp +++ b/Core/Dialog/SavedataParam.cpp @@ -1106,9 +1106,11 @@ int SavedataParam::GetFilesList(SceUtilitySavedataParam *param) // We need PARAMS.SFO's SAVEDATA_FILE_LIST to determine which entries are secure. PSPFileInfo sfoFileInfo = pspFileSystem.GetFileInfo(dirPath + "/" + SFO_FILENAME); std::set secureFilenames; - // TODO: Error code if not? + if (sfoFileInfo.exists) { secureFilenames = getSecureFileNames(dirPath); + } else { + return SCE_UTILITY_SAVEDATA_ERROR_RW_DATA_BROKEN; } // Does not list directories, nor recurse into them, and ignores files not ALL UPPERCASE. @@ -1657,24 +1659,15 @@ bool SavedataParam::IsInSaveDataList(std::string saveName, int count) { return false; } -bool SavedataParam::secureCanSkip(SceUtilitySavedataParam* param, bool secureMode) { - if (!secureMode) // Only check in secure mode. - return false; - std::string dirPath = savePath + GetGameName(param) + GetSaveName(param); +bool SavedataParam::IsSaveDirectoryExist(SceUtilitySavedataParam* param) { + std::string dirPath = savePath + GetGameName(param) + GetSaveName(param); + PSPFileInfo directoryInfo = pspFileSystem.GetFileInfo(dirPath); + return directoryInfo.exists; +} + +bool SavedataParam::IsSfoFileExist(SceUtilitySavedataParam* param) { + std::string dirPath = savePath + GetGameName(param) + GetSaveName(param); std::string sfoPath = dirPath + "/" + SFO_FILENAME; - std::string secureFileName = GetFileName(param); - std::set secureFileNames; PSPFileInfo sfoInfo = pspFileSystem.GetFileInfo(sfoPath); - // If sfo doesn't exist,shouldn't skip. - if (!sfoInfo.exists) - return false; - - // Get secure file names from PARAM.SFO. - secureFileNames = getSecureFileNames(dirPath); - // Secure file name should be saved in PARAM.SFO - // Cannot find name in PARAM.SFO, could skip. - if (secureFileNames.find(secureFileName) == secureFileNames.end()) - return true; - - return false; + return sfoInfo.exists; } diff --git a/Core/Dialog/SavedataParam.h b/Core/Dialog/SavedataParam.h index 9e5b79f201..6745bd6a20 100644 --- a/Core/Dialog/SavedataParam.h +++ b/Core/Dialog/SavedataParam.h @@ -314,7 +314,8 @@ public: bool GetSize(SceUtilitySavedataParam* param); int GetSaveCryptMode(SceUtilitySavedataParam* param, const std::string &saveDirName); bool IsInSaveDataList(std::string saveName, int count); - bool secureCanSkip(SceUtilitySavedataParam* param, bool secureMode); + bool IsSaveDirectoryExist(SceUtilitySavedataParam* param); + bool IsSfoFileExist(SceUtilitySavedataParam* param); std::string GetGameName(const SceUtilitySavedataParam *param) const; std::string GetSaveName(const SceUtilitySavedataParam *param) const; From b7db78362db88852f99182b400e1a52603f2e259 Mon Sep 17 00:00:00 2001 From: rock88 Date: Fri, 3 Oct 2014 17:50:12 +0700 Subject: [PATCH 045/105] iOS: Add launch xib name to info.plist --- ios/PPSSPP-Info.plist | 2 ++ 1 file changed, 2 insertions(+) diff --git a/ios/PPSSPP-Info.plist b/ios/PPSSPP-Info.plist index 0408e4b079..9f13bf9b8e 100644 --- a/ios/PPSSPP-Info.plist +++ b/ios/PPSSPP-Info.plist @@ -58,5 +58,7 @@ 1 2 + UILaunchStoryboardName + LaunchScreen From 1b520ea673144d6dd30547fa922df40f6a7393d3 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Fri, 3 Oct 2014 07:48:50 -0700 Subject: [PATCH 046/105] Correct first next sample calculation. If it's exactly matching a frame size, we need to return a full frame, rather than 0. Fixes #6967. --- Core/HLE/sceAtrac.cpp | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/Core/HLE/sceAtrac.cpp b/Core/HLE/sceAtrac.cpp index 9bf7151def..1e1f16271a 100644 --- a/Core/HLE/sceAtrac.cpp +++ b/Core/HLE/sceAtrac.cpp @@ -970,10 +970,10 @@ u32 sceAtracGetNextSample(int atracID, u32 outNAddr) { ERROR_LOG(ME, "sceAtracGetNextSample(%i, %08x): no data", atracID, outNAddr); return ATRAC_ERROR_NO_DATA; } else { - DEBUG_LOG(ME, "sceAtracGetNextSample(%i, %08x)", atracID, outNAddr); if (atrac->currentSample >= atrac->endSample) { if (Memory::IsValidAddress(outNAddr)) Memory::Write_U32(0, outNAddr); + DEBUG_LOG(ME, "sceAtracGetNextSample(%i, %08x): 0 samples left", atracID, outNAddr); return 0; } else { u32 atracSamplesPerFrame = (atrac->codecType == PSP_MODE_AT_3_PLUS ? ATRAC3PLUS_MAX_SAMPLES : ATRAC3_MAX_SAMPLES); @@ -983,13 +983,14 @@ u32 sceAtracGetNextSample(int atracID, u32 outNAddr) { u32 skipSamples = atrac->firstSampleoffset + firstOffsetExtra; u32 firstSamples = (atracSamplesPerFrame - skipSamples) % atracSamplesPerFrame; u32 numSamples = atrac->endSample - atrac->currentSample; - if (atrac->currentSample == 0) { + if (atrac->currentSample == 0 && firstSamples != 0) { numSamples = firstSamples; } if (numSamples > atracSamplesPerFrame) numSamples = atracSamplesPerFrame; if (Memory::IsValidAddress(outNAddr)) Memory::Write_U32(numSamples, outNAddr); + DEBUG_LOG(ME, "sceAtracGetNextSample(%i, %08x): %d samples left", atracID, outNAddr, numSamples); } } return 0; From 398646411a1e1f5e03dbc636684e6852e05d8df0 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Fri, 3 Oct 2014 19:48:39 -0700 Subject: [PATCH 047/105] Properly handle atrac packets with multiple frames. This gets us decoding the start of a file and near loops way more correctly. --- Core/HLE/sceAtrac.cpp | 92 ++++++++++++++++++++++++------------------- 1 file changed, 52 insertions(+), 40 deletions(-) diff --git a/Core/HLE/sceAtrac.cpp b/Core/HLE/sceAtrac.cpp index 1e1f16271a..6d47be1c3d 100644 --- a/Core/HLE/sceAtrac.cpp +++ b/Core/HLE/sceAtrac.cpp @@ -127,11 +127,12 @@ struct Atrac { memset(&first, 0, sizeof(first)); memset(&second, 0, sizeof(second)); #ifdef USE_FFMPEG - pFormatCtx = 0; - pAVIOCtx = 0; - pCodecCtx = 0; - pSwrCtx = 0; - pFrame = 0; + pFormatCtx = nullptr; + pAVIOCtx = nullptr; + pCodecCtx = nullptr; + pSwrCtx = nullptr; + pFrame = nullptr; + packet = nullptr; audio_stream_index = 0; #endif // USE_FFMPEG atracContext = 0; @@ -274,6 +275,7 @@ struct Atrac { AVCodecContext *pCodecCtx; SwrContext *pSwrCtx; AVFrame *pFrame; + AVPacket *packet; int audio_stream_index; void ReleaseFFMPEGContext() { @@ -289,17 +291,37 @@ struct Atrac { avcodec_close(pCodecCtx); if (pFormatCtx) avformat_close_input(&pFormatCtx); - pFormatCtx = 0; - pAVIOCtx = 0; - pCodecCtx = 0; - pSwrCtx = 0; - pFrame = 0; + if (packet) + av_free_packet(packet); + delete packet; + pFormatCtx = nullptr; + pAVIOCtx = nullptr; + pCodecCtx = nullptr; + pSwrCtx = nullptr; + pFrame = nullptr; + packet = nullptr; } void SeekToSample(int sample) { s64 seek_pos = (s64)sample; av_seek_frame(pFormatCtx, audio_stream_index, seek_pos, 0); } + + bool FillPacket() { + if (packet->size > 0) { + return true; + } + do { + // This is double-free safe, so we just call it before each read and at the end. + av_free_packet(packet); + if (av_read_frame(pFormatCtx, packet) < 0) { + return false; + } + // We keep reading until we get the right stream index. + } while (packet->stream_index != audio_stream_index); + + return true; + } #endif // USE_FFMPEG }; @@ -641,25 +663,18 @@ u32 _AtracDecodeData(int atracID, u8 *outbuf, u32 outbufPtr, u32 *SamplesNum, u3 int forceseekSample = atrac->currentSample * 2 > atrac->endSample ? 0 : atrac->endSample; atrac->SeekToSample(forceseekSample); atrac->SeekToSample(atrac->currentSample == 0 ? 0 : atrac->currentSample + offsetSamples); - AVPacket packet; - av_init_packet(&packet); int got_frame = 0, avret; - while (av_read_frame(atrac->pFormatCtx, &packet) >= 0) { - if (packet.stream_index != atrac->audio_stream_index) { - av_free_packet(&packet); - continue; - } - + while (atrac->FillPacket()) { got_frame = 0; - int bytes_in_packet = packet.size; - avret = avcodec_decode_audio4(atrac->pCodecCtx, atrac->pFrame, &got_frame, &packet); + avret = avcodec_decode_audio4(atrac->pCodecCtx, atrac->pFrame, &got_frame, atrac->packet); if (avret == AVERROR_PATCHWELCOME) { ERROR_LOG(ME, "Unsupported feature in ATRAC audio."); - // Let's try the next frame. + // Let's try the next packet. + atrac->packet->size = 0; + avret = 0; // TODO: Or actually, we should return a blank frame and pretend it worked. } else if (avret < 0) { ERROR_LOG(ME, "avcodec_decode_audio4: Error decoding audio %d", avret); - av_free_packet(&packet); atrac->failedDecode = true; // No need to free the packet if decode_audio4 fails. // Avoid getting stuck in a loop (Virtua Tennis) @@ -668,11 +683,9 @@ u32 _AtracDecodeData(int atracID, u8 *outbuf, u32 outbufPtr, u32 *SamplesNum, u3 *remains = 0; return ATRAC_ERROR_ALL_DATA_DECODED; } - // FFmpeg seems to return packet.size / 10. - // However, advancing the packet by this causes decode errors. Bug? - if (avret != packet.size && avret != packet.size / 10) { - ERROR_LOG_REPORT_ONCE(multipacket, ME, "WARNING: Remaining data in packet - we currently only decode one frame per packet"); - } + + atrac->packet->size -= avret; + atrac->packet->data += avret; if (got_frame) { // got a frame @@ -713,7 +726,6 @@ u32 _AtracDecodeData(int atracID, u8 *outbuf, u32 outbufPtr, u32 *SamplesNum, u3 } } } - av_free_packet(&packet); if (got_frame) { // We only want one frame per call, let's continue the next time. break; @@ -1239,6 +1251,10 @@ int __AtracSetContext(Atrac *atrac) { // alloc audio frame atrac->pFrame = av_frame_alloc(); + atrac->packet = new AVPacket; + av_init_packet(atrac->packet); + atrac->packet->data = nullptr; + atrac->packet->size = 0; // reinit decodePos, because ffmpeg had changed it. atrac->decodePos = 0; #endif @@ -1962,27 +1978,24 @@ int sceAtracLowLevelDecode(int atracID, u32 sourceAddr, u32 sourceBytesConsumedA atrac->SeekToSample(atrac->currentSample); if (!atrac->failedDecode) { - AVPacket packet; - av_init_packet(&packet); int got_frame, avret; - while (av_read_frame(atrac->pFormatCtx, &packet) >= 0) { - if (packet.stream_index != atrac->audio_stream_index) { - av_free_packet(&packet); - continue; - } - + while (atrac->FillPacket()) { got_frame = 0; - avret = avcodec_decode_audio4(atrac->pCodecCtx, atrac->pFrame, &got_frame, &packet); + avret = avcodec_decode_audio4(atrac->pCodecCtx, atrac->pFrame, &got_frame, atrac->packet); if (avret == AVERROR_PATCHWELCOME) { ERROR_LOG(ME, "Unsupported feature in ATRAC audio."); - // Let's try the next frame. + // Let's try the next packet. + atrac->packet->size = 0; + avret = 0; } else if (avret < 0) { ERROR_LOG(ME, "atracID: %i, avcodec_decode_audio4: Error decoding audio %d", atracID, avret); - av_free_packet(&packet); atrac->failedDecode = true; break; } + atrac->packet->size -= avret; + atrac->packet->data += avret; + if (got_frame) { // got a frame int decoded = av_samples_get_buffer_size(NULL, atrac->pFrame->channels, @@ -1995,7 +2008,6 @@ int sceAtracLowLevelDecode(int atracID, u32 sourceAddr, u32 sourceBytesConsumedA ERROR_LOG(ME, "swr_convert: Error while converting %d", avret); } } - av_free_packet(&packet); if (got_frame) break; } From 24ab84a0fe9ff9214709181cd5088dd71faf7e80 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Fri, 3 Oct 2014 20:04:39 -0700 Subject: [PATCH 048/105] Centralize atrac frame decode logic. --- Core/HLE/sceAtrac.cpp | 87 ++++++++++++++++++++++--------------------- 1 file changed, 45 insertions(+), 42 deletions(-) diff --git a/Core/HLE/sceAtrac.cpp b/Core/HLE/sceAtrac.cpp index 6d47be1c3d..95dadbee5b 100644 --- a/Core/HLE/sceAtrac.cpp +++ b/Core/HLE/sceAtrac.cpp @@ -95,6 +95,12 @@ extern "C" { #endif // USE_FFMPEG +enum AtracDecodeResult { + ATDECODE_FAILED = -1, + ATDECODE_FEEDME = 0, + ATDECODE_GOTFRAME = 1, +}; + struct InputBuffer { u32 addr; u32 size; @@ -322,6 +328,26 @@ struct Atrac { return true; } + + AtracDecodeResult DecodePacket() { + int got_frame = 0; + int bytes_read = avcodec_decode_audio4(pCodecCtx, pFrame, &got_frame, packet); + if (bytes_read == AVERROR_PATCHWELCOME) { + ERROR_LOG(ME, "Unsupported feature in ATRAC audio."); + // Let's try the next packet. + packet->size = 0; + // TODO: Or actually, should we return a blank frame and pretend it worked? + return ATDECODE_FEEDME; + } else if (bytes_read < 0) { + ERROR_LOG_REPORT(ME, "avcodec_decode_audio4: Error decoding audio %d / %08x", bytes_read, bytes_read); + failedDecode = true; + return ATDECODE_FAILED; + } + + packet->size -= bytes_read; + packet->data += bytes_read; + return got_frame ? ATDECODE_GOTFRAME : ATDECODE_FEEDME; + } #endif // USE_FFMPEG }; @@ -663,20 +689,11 @@ u32 _AtracDecodeData(int atracID, u8 *outbuf, u32 outbufPtr, u32 *SamplesNum, u3 int forceseekSample = atrac->currentSample * 2 > atrac->endSample ? 0 : atrac->endSample; atrac->SeekToSample(forceseekSample); atrac->SeekToSample(atrac->currentSample == 0 ? 0 : atrac->currentSample + offsetSamples); - int got_frame = 0, avret; + + AtracDecodeResult res = ATDECODE_FEEDME; while (atrac->FillPacket()) { - got_frame = 0; - avret = avcodec_decode_audio4(atrac->pCodecCtx, atrac->pFrame, &got_frame, atrac->packet); - if (avret == AVERROR_PATCHWELCOME) { - ERROR_LOG(ME, "Unsupported feature in ATRAC audio."); - // Let's try the next packet. - atrac->packet->size = 0; - avret = 0; - // TODO: Or actually, we should return a blank frame and pretend it worked. - } else if (avret < 0) { - ERROR_LOG(ME, "avcodec_decode_audio4: Error decoding audio %d", avret); - atrac->failedDecode = true; - // No need to free the packet if decode_audio4 fails. + res = atrac->DecodePacket(); + if (res == ATDECODE_FAILED) { // Avoid getting stuck in a loop (Virtua Tennis) *SamplesNum = 0; *finish = 1; @@ -684,10 +701,7 @@ u32 _AtracDecodeData(int atracID, u8 *outbuf, u32 outbufPtr, u32 *SamplesNum, u3 return ATRAC_ERROR_ALL_DATA_DECODED; } - atrac->packet->size -= avret; - atrac->packet->data += avret; - - if (got_frame) { + if (res == ATDECODE_GOTFRAME) { // got a frame // Use a small buffer and keep overwriting it with file data constantly atrac->first.writableBytes += atrac->atracBytesPerFrame; @@ -700,7 +714,7 @@ u32 _AtracDecodeData(int atracID, u8 *outbuf, u32 outbufPtr, u32 *SamplesNum, u3 if (skipped > 0 && numSamples == 0) { // Wait for the next one. - got_frame = 0; + res = ATDECODE_FEEDME; } if (outbuf != NULL && numSamples != 0) { @@ -716,7 +730,7 @@ u32 _AtracDecodeData(int atracID, u8 *outbuf, u32 outbufPtr, u32 *SamplesNum, u3 atrac->pFrame->extended_data[0] + inbufOffset, atrac->pFrame->extended_data[1] + inbufOffset, }; - avret = swr_convert(atrac->pSwrCtx, &out, numSamples, inbuf, numSamples); + int avret = swr_convert(atrac->pSwrCtx, &out, numSamples, inbuf, numSamples); if (outbufPtr != 0) { u32 outBytes = numSamples * atrac->atracOutputChannels * sizeof(s16); CBreakPoints::ExecMemCheck(outbufPtr, true, outBytes, currentMIPS->pc); @@ -726,13 +740,13 @@ u32 _AtracDecodeData(int atracID, u8 *outbuf, u32 outbufPtr, u32 *SamplesNum, u3 } } } - if (got_frame) { + if (res == ATDECODE_GOTFRAME) { // We only want one frame per call, let's continue the next time. break; } } - if (!got_frame && atrac->currentSample < atrac->endSample) { + if (res != ATDECODE_GOTFRAME && atrac->currentSample < atrac->endSample) { // Never got a frame. We may have dropped a GHA frame or otherwise have a bug. // For now, let's try to provide an extra "frame" if possible so games don't infinite loop. numSamples = std::min((u32)atrac->endSample - (u32)atrac->currentSample, atracSamplesPerFrame); @@ -1978,37 +1992,26 @@ int sceAtracLowLevelDecode(int atracID, u32 sourceAddr, u32 sourceBytesConsumedA atrac->SeekToSample(atrac->currentSample); if (!atrac->failedDecode) { - int got_frame, avret; + AtracDecodeResult res; while (atrac->FillPacket()) { - got_frame = 0; - avret = avcodec_decode_audio4(atrac->pCodecCtx, atrac->pFrame, &got_frame, atrac->packet); - if (avret == AVERROR_PATCHWELCOME) { - ERROR_LOG(ME, "Unsupported feature in ATRAC audio."); - // Let's try the next packet. - atrac->packet->size = 0; - avret = 0; - } else if (avret < 0) { - ERROR_LOG(ME, "atracID: %i, avcodec_decode_audio4: Error decoding audio %d", atracID, avret); - atrac->failedDecode = true; + res = atrac->DecodePacket(); + if (res == ATDECODE_FAILED) { break; } - atrac->packet->size -= avret; - atrac->packet->data += avret; - - if (got_frame) { + if (res == ATDECODE_GOTFRAME) { // got a frame - int decoded = av_samples_get_buffer_size(NULL, atrac->pFrame->channels, - atrac->pFrame->nb_samples, (AVSampleFormat)atrac->pFrame->format, 1); - u8* out = Memory::GetPointer(samplesAddr); + u8 *out = Memory::GetPointer(samplesAddr); numSamples = atrac->pFrame->nb_samples; - avret = swr_convert(atrac->pSwrCtx, &out, atrac->pFrame->nb_samples, - (const u8**)atrac->pFrame->extended_data, atrac->pFrame->nb_samples); + int avret = swr_convert(atrac->pSwrCtx, &out, numSamples, + (const u8**)atrac->pFrame->extended_data, numSamples); + u32 outBytes = numSamples * atrac->atracOutputChannels * sizeof(s16); + CBreakPoints::ExecMemCheck(samplesAddr, true, outBytes, currentMIPS->pc); if (avret < 0) { ERROR_LOG(ME, "swr_convert: Error while converting %d", avret); } } - if (got_frame) + if (res == ATDECODE_GOTFRAME) break; } } From 2f443f52a6a21bd0af0fdeaa1ba5737980e378df Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Fri, 3 Oct 2014 20:05:08 -0700 Subject: [PATCH 049/105] Ensure we request s16 samples. We won't get these for atrac3+, but we should for atrac3 (seems to be the default anyway, but better to be clear.) --- Core/HLE/sceAtrac.cpp | 1 + 1 file changed, 1 insertion(+) diff --git a/Core/HLE/sceAtrac.cpp b/Core/HLE/sceAtrac.cpp index 95dadbee5b..f41e3715b3 100644 --- a/Core/HLE/sceAtrac.cpp +++ b/Core/HLE/sceAtrac.cpp @@ -1255,6 +1255,7 @@ int __AtracSetContext(Atrac *atrac) { atrac->pCodecCtx->channel_layout = AV_CH_LAYOUT_MONO; // open codec + atrac->pCodecCtx->request_sample_fmt = AV_SAMPLE_FMT_S16; if ((ret = avcodec_open2(atrac->pCodecCtx, pCodec, NULL)) < 0) { ERROR_LOG(ME, "avcodec_open2: Cannot open audio decoder %d", ret); return -1; From f421453bf9e91e5439b4083e82731cbcd5bde809 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Fri, 3 Oct 2014 21:32:07 -0700 Subject: [PATCH 050/105] Align samples even after a loop. This corrects the amount of audio after certain loops, but it doesn't seem to output the right data when this happens. Possibly, seeking isn't doing the right thing and resetting state that shouldn't be reset when a loop happens. Not sure... but it was already wrong before, this just reads the right amount of it. --- Core/HLE/sceAtrac.cpp | 15 +++++++++++++-- 1 file changed, 13 insertions(+), 2 deletions(-) diff --git a/Core/HLE/sceAtrac.cpp b/Core/HLE/sceAtrac.cpp index f41e3715b3..e56dd5a730 100644 --- a/Core/HLE/sceAtrac.cpp +++ b/Core/HLE/sceAtrac.cpp @@ -683,6 +683,12 @@ u32 _AtracDecodeData(int atracID, u8 *outbuf, u32 outbufPtr, u32 *SamplesNum, u3 // It seems like the PSP aligns the sample position to 0x800...? int offsetSamples = atrac->firstSampleoffset + firstOffsetExtra; int skipSamples = atrac->currentSample == 0 ? offsetSamples : 0; + u32 maxSamples = atrac->endSample - atrac->currentSample; + u32 unalignedSamples = (offsetSamples + atrac->currentSample) % atracSamplesPerFrame; + if (unalignedSamples != 0) { + // We're off alignment, possibly due to a loop. Force it back on. + maxSamples = atracSamplesPerFrame - unalignedSamples; + } #ifdef USE_FFMPEG if (!atrac->failedDecode && (atrac->codecType == PSP_MODE_AT_3 || atrac->codecType == PSP_MODE_AT_3_PLUS) && atrac->pCodecCtx) { @@ -710,7 +716,7 @@ u32 _AtracDecodeData(int atracID, u8 *outbuf, u32 outbufPtr, u32 *SamplesNum, u3 numSamples = atrac->pFrame->nb_samples - skipped; // If we're at the end, clamp to samples we want. It always returns a full chunk. - numSamples = std::min((u32)atrac->endSample - (u32)atrac->currentSample, numSamples); + numSamples = std::min(maxSamples, numSamples); if (skipped > 0 && numSamples == 0) { // Wait for the next one. @@ -749,7 +755,7 @@ u32 _AtracDecodeData(int atracID, u8 *outbuf, u32 outbufPtr, u32 *SamplesNum, u3 if (res != ATDECODE_GOTFRAME && atrac->currentSample < atrac->endSample) { // Never got a frame. We may have dropped a GHA frame or otherwise have a bug. // For now, let's try to provide an extra "frame" if possible so games don't infinite loop. - numSamples = std::min((u32)atrac->endSample - (u32)atrac->currentSample, atracSamplesPerFrame); + numSamples = std::min(maxSamples, atracSamplesPerFrame); u32 outBytes = numSamples * atrac->atracOutputChannels * sizeof(s16); memset(outbuf, 0, outBytes); CBreakPoints::ExecMemCheck(outbufPtr, true, outBytes, currentMIPS->pc); @@ -1012,6 +1018,11 @@ u32 sceAtracGetNextSample(int atracID, u32 outNAddr) { if (atrac->currentSample == 0 && firstSamples != 0) { numSamples = firstSamples; } + u32 unalignedSamples = (skipSamples + atrac->currentSample) % atracSamplesPerFrame; + if (unalignedSamples != 0) { + // We're off alignment, possibly due to a loop. Force it back on. + numSamples = atracSamplesPerFrame - unalignedSamples; + } if (numSamples > atracSamplesPerFrame) numSamples = atracSamplesPerFrame; if (Memory::IsValidAddress(outNAddr)) From cf7e280185ee82ec0bf0a36009daed9355daa1cc Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Fri, 3 Oct 2014 22:11:45 -0700 Subject: [PATCH 051/105] Attempt to ensure we don't decode partial frames. --- Core/HLE/sceAtrac.cpp | 27 ++++++++++++++++++++++++++- 1 file changed, 26 insertions(+), 1 deletion(-) diff --git a/Core/HLE/sceAtrac.cpp b/Core/HLE/sceAtrac.cpp index e56dd5a730..a48127710b 100644 --- a/Core/HLE/sceAtrac.cpp +++ b/Core/HLE/sceAtrac.cpp @@ -330,8 +330,33 @@ struct Atrac { } AtracDecodeResult DecodePacket() { + AVPacket tempPacket; + AVPacket *decodePacket = packet; + if (packet->size < (int)atracBytesPerFrame) { + // Whoops, we have a packet that is smaller than a frame. Let's meld a new one. + u32 initialSize = packet->size; + u32 needed = atracBytesPerFrame - initialSize; + av_init_packet(&tempPacket); + av_copy_packet(&tempPacket, packet); + av_grow_packet(&tempPacket, needed); + + // Okay, we're "out of data", let's get more. + packet->size = 0; + + if (FillPacket()) { + if (packet->size >= needed) { + memcpy(tempPacket.data + initialSize, packet->data, needed); + packet->size -= needed; + packet->data += needed; + } + } + } + int got_frame = 0; - int bytes_read = avcodec_decode_audio4(pCodecCtx, pFrame, &got_frame, packet); + int bytes_read = avcodec_decode_audio4(pCodecCtx, pFrame, &got_frame, decodePacket); + if (packet != decodePacket) { + av_free_packet(&tempPacket); + } if (bytes_read == AVERROR_PATCHWELCOME) { ERROR_LOG(ME, "Unsupported feature in ATRAC audio."); // Let's try the next packet. From 72e8e6448fdd80cc6358cef161c192df5128aa12 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Fri, 3 Oct 2014 22:17:25 -0700 Subject: [PATCH 052/105] Don't eat packet data when using a temp packet. --- Core/HLE/sceAtrac.cpp | 10 +++++++--- 1 file changed, 7 insertions(+), 3 deletions(-) diff --git a/Core/HLE/sceAtrac.cpp b/Core/HLE/sceAtrac.cpp index a48127710b..b6b74baa49 100644 --- a/Core/HLE/sceAtrac.cpp +++ b/Core/HLE/sceAtrac.cpp @@ -360,7 +360,9 @@ struct Atrac { if (bytes_read == AVERROR_PATCHWELCOME) { ERROR_LOG(ME, "Unsupported feature in ATRAC audio."); // Let's try the next packet. - packet->size = 0; + if (packet == decodePacket) { + packet->size = 0; + } // TODO: Or actually, should we return a blank frame and pretend it worked? return ATDECODE_FEEDME; } else if (bytes_read < 0) { @@ -369,8 +371,10 @@ struct Atrac { return ATDECODE_FAILED; } - packet->size -= bytes_read; - packet->data += bytes_read; + if (packet == decodePacket) { + packet->size -= bytes_read; + packet->data += bytes_read; + } return got_frame ? ATDECODE_GOTFRAME : ATDECODE_FEEDME; } #endif // USE_FFMPEG From 50e4eded75eef83055305de62db4c24de9763f95 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Fri, 3 Oct 2014 23:08:16 -0700 Subject: [PATCH 053/105] Actually use the temp packet. --- Core/HLE/sceAtrac.cpp | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/Core/HLE/sceAtrac.cpp b/Core/HLE/sceAtrac.cpp index b6b74baa49..d69f858bd6 100644 --- a/Core/HLE/sceAtrac.cpp +++ b/Core/HLE/sceAtrac.cpp @@ -335,7 +335,7 @@ struct Atrac { if (packet->size < (int)atracBytesPerFrame) { // Whoops, we have a packet that is smaller than a frame. Let's meld a new one. u32 initialSize = packet->size; - u32 needed = atracBytesPerFrame - initialSize; + int needed = atracBytesPerFrame - initialSize; av_init_packet(&tempPacket); av_copy_packet(&tempPacket, packet); av_grow_packet(&tempPacket, needed); @@ -350,6 +350,7 @@ struct Atrac { packet->data += needed; } } + decodePacket = &tempPacket; } int got_frame = 0; From adef5bbe59517ae3dc85021b1317da00e26a90f7 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Fri, 3 Oct 2014 23:23:36 -0700 Subject: [PATCH 054/105] Discard packet data when seeking. --- Core/HLE/sceAtrac.cpp | 2 ++ 1 file changed, 2 insertions(+) diff --git a/Core/HLE/sceAtrac.cpp b/Core/HLE/sceAtrac.cpp index d69f858bd6..46554ffbca 100644 --- a/Core/HLE/sceAtrac.cpp +++ b/Core/HLE/sceAtrac.cpp @@ -311,6 +311,8 @@ struct Atrac { void SeekToSample(int sample) { s64 seek_pos = (s64)sample; av_seek_frame(pFormatCtx, audio_stream_index, seek_pos, 0); + // Discard any pending packet data. + packet->size = 0; } bool FillPacket() { From eca659d6f11d26e2010e377ca3f49919702916ac Mon Sep 17 00:00:00 2001 From: sum2012 Date: Sun, 5 Oct 2014 04:31:45 +0800 Subject: [PATCH 055/105] Win32:Prevent crash without ui_atlas.zim Just exit PPSSPP.Work around #6843 --- UI/NativeApp.cpp | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/UI/NativeApp.cpp b/UI/NativeApp.cpp index 3efbdbaf52..55e2550b91 100644 --- a/UI/NativeApp.cpp +++ b/UI/NativeApp.cpp @@ -541,6 +541,10 @@ void NativeInitGraphics() { #endif PanicAlert("Failed to load ui_atlas.zim.\n\nPlace it in the directory \"assets\" under your PPSSPP directory."); ELOG("Failed to load ui_atlas.zim"); +#ifdef _WIN32 + UINT ExitCode = 0; + ExitProcess(ExitCode); +#endif } uiContext = new UIContext(); From d7927009d088151ea02d36f10216e10fa9e40f3e Mon Sep 17 00:00:00 2001 From: daniel229 Date: Sun, 5 Oct 2014 13:39:15 +0800 Subject: [PATCH 056/105] Replace frame download in danganronpa 2 --- Core/HLE/ReplaceTables.cpp | 13 +++++++++++++ Core/MIPS/MIPSAnalyst.cpp | 1 + 2 files changed, 14 insertions(+) diff --git a/Core/HLE/ReplaceTables.cpp b/Core/HLE/ReplaceTables.cpp index 6d2844fd2c..fc04ec7f91 100644 --- a/Core/HLE/ReplaceTables.cpp +++ b/Core/HLE/ReplaceTables.cpp @@ -729,6 +729,18 @@ static int Hook_bokunonatsuyasumi4_download_frame() { return 0; } +static int Hook_danganronpa2_1_download_frame() { + const u32 fb_base = currentMIPS->r[MIPS_REG_V0]; + const u32 fb_offset = currentMIPS->r[MIPS_REG_V1]; + const u32 fb_offset_fix = fb_offset & 0xFFFFFFFC; + const u32 fb_address = fb_base + fb_offset_fix; + if (Memory::IsVRAMAddress(fb_address)) { + gpu->PerformMemoryDownload(fb_address, 0x00088000); + CBreakPoints::ExecMemCheck(fb_address, true, 0x00088000, currentMIPS->pc); + } + return 0; +} + // Can either replace with C functions or functions emitted in Asm/ArmAsm. static const ReplacementTableEntry entries[] = { // TODO: I think some games can be helped quite a bit by implementing the @@ -789,6 +801,7 @@ static const ReplacementTableEntry entries[] = { { "soranokiseki_fc_download_frame", &Hook_soranokiseki_fc_download_frame, 0, REPFLAG_HOOKENTER, 0x180 }, { "soranokiseki_sc_download_frame", &Hook_soranokiseki_sc_download_frame, 0, REPFLAG_HOOKENTER, }, { "bokunonatsuyasumi4_download_frame", &Hook_bokunonatsuyasumi4_download_frame, 0, REPFLAG_HOOKENTER, 0x8C }, + { "danganronpa2_1_download_frame", &Hook_danganronpa2_1_download_frame, 0, REPFLAG_HOOKENTER, 0x68 }, {} }; diff --git a/Core/MIPS/MIPSAnalyst.cpp b/Core/MIPS/MIPSAnalyst.cpp index 693260f3b9..26c13c496f 100644 --- a/Core/MIPS/MIPSAnalyst.cpp +++ b/Core/MIPS/MIPSAnalyst.cpp @@ -327,6 +327,7 @@ static const HardHashTableEntry hardcodedHashes[] = { { 0xafc9968e7d246a5e, 1588, "atan", }, { 0xafcb7dfbc4d72588, 44, "vector_transform_3x4", }, { 0xb07f9d82d79deea9, 536, "brandish_download_frame", }, // Brandish, and Sora no kiseki 3rd + { 0xb09c9bc1343a774c, 456, "danganronpa2_1_download_frame", }, // Danganronpa 2 { 0xb0db731f27d3aa1b, 40, "vmax_s", }, { 0xb0ef265e87899f0a, 32, "vector_divide_t_s", }, { 0xb183a37baa12607b, 32, "vscl_t", }, From a7cf3aeafc70ca2550709d02050e553d66f026e4 Mon Sep 17 00:00:00 2001 From: daniel229 Date: Sun, 5 Oct 2014 13:42:03 +0800 Subject: [PATCH 057/105] Another replace frame download in danganronpa 2 --- Core/HLE/ReplaceTables.cpp | 13 +++++++++++++ Core/MIPS/MIPSAnalyst.cpp | 1 + 2 files changed, 14 insertions(+) diff --git a/Core/HLE/ReplaceTables.cpp b/Core/HLE/ReplaceTables.cpp index fc04ec7f91..d11313ee38 100644 --- a/Core/HLE/ReplaceTables.cpp +++ b/Core/HLE/ReplaceTables.cpp @@ -741,6 +741,18 @@ static int Hook_danganronpa2_1_download_frame() { return 0; } +static int Hook_danganronpa2_2_download_frame() { + const u32 fb_base = currentMIPS->r[MIPS_REG_V0]; + const u32 fb_offset = currentMIPS->r[MIPS_REG_V1]; + const u32 fb_offset_fix = fb_offset & 0xFFFFFFFC; + const u32 fb_address = fb_base + fb_offset_fix; + if (Memory::IsVRAMAddress(fb_address)) { + gpu->PerformMemoryDownload(fb_address, 0x00088000); + CBreakPoints::ExecMemCheck(fb_address, true, 0x00088000, currentMIPS->pc); + } + return 0; +} + // Can either replace with C functions or functions emitted in Asm/ArmAsm. static const ReplacementTableEntry entries[] = { // TODO: I think some games can be helped quite a bit by implementing the @@ -802,6 +814,7 @@ static const ReplacementTableEntry entries[] = { { "soranokiseki_sc_download_frame", &Hook_soranokiseki_sc_download_frame, 0, REPFLAG_HOOKENTER, }, { "bokunonatsuyasumi4_download_frame", &Hook_bokunonatsuyasumi4_download_frame, 0, REPFLAG_HOOKENTER, 0x8C }, { "danganronpa2_1_download_frame", &Hook_danganronpa2_1_download_frame, 0, REPFLAG_HOOKENTER, 0x68 }, + { "danganronpa2_2_download_frame", &Hook_danganronpa2_2_download_frame, 0, REPFLAG_HOOKENTER, 0x94 }, {} }; diff --git a/Core/MIPS/MIPSAnalyst.cpp b/Core/MIPS/MIPSAnalyst.cpp index 26c13c496f..1f027b9ff2 100644 --- a/Core/MIPS/MIPSAnalyst.cpp +++ b/Core/MIPS/MIPSAnalyst.cpp @@ -249,6 +249,7 @@ static const HardHashTableEntry hardcodedHashes[] = { { 0x7245b74db370ae72, 64, "vmmul_q_transp3", }, { 0x7259d52b21814a5a, 40, "vtfm_t_transp", }, { 0x736b34ebc702d873, 104, "vmmul_q_transp", }, + { 0x73a614c08f777d52, 792, "danganronpa2_2_download_frame", }, // Danganronpa 2 { 0x7499a2ce8b60d801, 12, "abs", }, { 0x74ebbe7d341463f3, 72, "dl_write_colortest", }, { 0x755a41f9183bb89a, 60, "vmmul_q", }, From ef1484da65ddf065f07dc0f043e79191756b2cc1 Mon Sep 17 00:00:00 2001 From: daniel229 Date: Sun, 5 Oct 2014 13:44:39 +0800 Subject: [PATCH 058/105] Replace frame download in danganronpa 1 --- Core/HLE/ReplaceTables.cpp | 13 +++++++++++++ Core/MIPS/MIPSAnalyst.cpp | 1 + 2 files changed, 14 insertions(+) diff --git a/Core/HLE/ReplaceTables.cpp b/Core/HLE/ReplaceTables.cpp index d11313ee38..0c27b03f46 100644 --- a/Core/HLE/ReplaceTables.cpp +++ b/Core/HLE/ReplaceTables.cpp @@ -753,6 +753,18 @@ static int Hook_danganronpa2_2_download_frame() { return 0; } +static int Hook_danganronpa1_1_download_frame() { + const u32 fb_base = currentMIPS->r[MIPS_REG_A5]; + const u32 fb_offset = currentMIPS->r[MIPS_REG_V0]; + const u32 fb_offset_fix = fb_offset & 0xFFFFFFFC; + const u32 fb_address = fb_base + fb_offset_fix; + if (Memory::IsVRAMAddress(fb_address)) { + gpu->PerformMemoryDownload(fb_address, 0x00088000); + CBreakPoints::ExecMemCheck(fb_address, true, 0x00088000, currentMIPS->pc); + } + return 0; +} + // Can either replace with C functions or functions emitted in Asm/ArmAsm. static const ReplacementTableEntry entries[] = { // TODO: I think some games can be helped quite a bit by implementing the @@ -815,6 +827,7 @@ static const ReplacementTableEntry entries[] = { { "bokunonatsuyasumi4_download_frame", &Hook_bokunonatsuyasumi4_download_frame, 0, REPFLAG_HOOKENTER, 0x8C }, { "danganronpa2_1_download_frame", &Hook_danganronpa2_1_download_frame, 0, REPFLAG_HOOKENTER, 0x68 }, { "danganronpa2_2_download_frame", &Hook_danganronpa2_2_download_frame, 0, REPFLAG_HOOKENTER, 0x94 }, + { "danganronpa1_1_download_frame", &Hook_danganronpa1_1_download_frame, 0, REPFLAG_HOOKENTER, 0x78 }, {} }; diff --git a/Core/MIPS/MIPSAnalyst.cpp b/Core/MIPS/MIPSAnalyst.cpp index 1f027b9ff2..7f2351413b 100644 --- a/Core/MIPS/MIPSAnalyst.cpp +++ b/Core/MIPS/MIPSAnalyst.cpp @@ -311,6 +311,7 @@ static const HardHashTableEntry hardcodedHashes[] = { { 0xa54967288afe8f26, 600, "ceil", }, { 0xa5ddbbc688e89a4d, 56, "isinf", }, { 0xa662359e30b829e4, 148, "memcmp", }, + { 0xa6a03f0487a911b0, 392, "danganronpa1_1_download_frame", }, // Danganronpa 1 { 0xa8390e65fa087c62, 140, "vtfm_t_q", }, { 0xa85fe8abb88b1c6f, 52, "vector_sub_t", }, { 0xa9194e55cc586557, 268, "memcpy", }, From 5ff098efb9c09dd0f75034472c208839f4e00e46 Mon Sep 17 00:00:00 2001 From: daniel229 Date: Sun, 5 Oct 2014 13:46:47 +0800 Subject: [PATCH 059/105] Another replace frame download in danganronpa 1 --- Core/HLE/ReplaceTables.cpp | 15 +++++++++++++++ Core/MIPS/MIPSAnalyst.cpp | 1 + 2 files changed, 16 insertions(+) diff --git a/Core/HLE/ReplaceTables.cpp b/Core/HLE/ReplaceTables.cpp index 0c27b03f46..426d7e28d7 100644 --- a/Core/HLE/ReplaceTables.cpp +++ b/Core/HLE/ReplaceTables.cpp @@ -765,6 +765,20 @@ static int Hook_danganronpa1_1_download_frame() { return 0; } +static int Hook_danganronpa1_2_download_frame() { + const MIPSOpcode instruction = Memory::Read_Instruction(currentMIPS->pc + 0x8, true); + const int reg_num = instruction >> 11 & 31; + const u32 fb_base = currentMIPS->r[reg_num]; + const u32 fb_offset = currentMIPS->r[MIPS_REG_V0]; + const u32 fb_offset_fix = fb_offset & 0xFFFFFFFC; + const u32 fb_address = fb_base + fb_offset_fix; + if (Memory::IsVRAMAddress(fb_address)) { + gpu->PerformMemoryDownload(fb_address, 0x00088000); + CBreakPoints::ExecMemCheck(fb_address, true, 0x00088000, currentMIPS->pc); + } + return 0; +} + // Can either replace with C functions or functions emitted in Asm/ArmAsm. static const ReplacementTableEntry entries[] = { // TODO: I think some games can be helped quite a bit by implementing the @@ -828,6 +842,7 @@ static const ReplacementTableEntry entries[] = { { "danganronpa2_1_download_frame", &Hook_danganronpa2_1_download_frame, 0, REPFLAG_HOOKENTER, 0x68 }, { "danganronpa2_2_download_frame", &Hook_danganronpa2_2_download_frame, 0, REPFLAG_HOOKENTER, 0x94 }, { "danganronpa1_1_download_frame", &Hook_danganronpa1_1_download_frame, 0, REPFLAG_HOOKENTER, 0x78 }, + { "danganronpa1_2_download_frame", &Hook_danganronpa1_2_download_frame, 0, REPFLAG_HOOKENTER, 0xA8 }, {} }; diff --git a/Core/MIPS/MIPSAnalyst.cpp b/Core/MIPS/MIPSAnalyst.cpp index 7f2351413b..626931a650 100644 --- a/Core/MIPS/MIPSAnalyst.cpp +++ b/Core/MIPS/MIPSAnalyst.cpp @@ -245,6 +245,7 @@ static const HardHashTableEntry hardcodedHashes[] = { { 0x6f101c5c4311c144, 276, "floorf", }, { 0x6f1731f84bbf76c3, 116, "strcmp", }, { 0x6f4e1a1a84df1da0, 68, "dl_write_texmode", }, + { 0x6f7c9109b5b8fa47, 688, "danganronpa1_2_download_frame", }, // Danganronpa 1 { 0x70649c7211f6a8da, 16, "fabsf", }, { 0x7245b74db370ae72, 64, "vmmul_q_transp3", }, { 0x7259d52b21814a5a, 40, "vtfm_t_transp", }, From c9a21ab44d73a2565f151ce16b9fa9f6a3be81a2 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Henrik=20Rydg=C3=A5rd?= Date: Sun, 5 Oct 2014 14:20:30 +0200 Subject: [PATCH 060/105] Add T2 and T3 to our register enum for clarity --- Core/MIPS/MIPS.h | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/Core/MIPS/MIPS.h b/Core/MIPS/MIPS.h index bbe67ce373..c9e7c35cc8 100644 --- a/Core/MIPS/MIPS.h +++ b/Core/MIPS/MIPS.h @@ -38,9 +38,11 @@ enum MIPSGPReg MIPS_REG_A1=5, MIPS_REG_A2=6, MIPS_REG_A3=7, - MIPS_REG_A4=8, // Seems to be N32 register calling convention - there are 8 args instead of 4. + MIPS_REG_A4=8, MIPS_REG_A5=9, + MIPS_REG_T2=10, + MIPS_REG_T3=11, MIPS_REG_T4=12, MIPS_REG_T5=13, MIPS_REG_T6=14, From 5c7f3b6995e3fc4693f6d21ab2e5f2eee1455197 Mon Sep 17 00:00:00 2001 From: Sacha Date: Wed, 24 Sep 2014 10:49:59 +1000 Subject: [PATCH 061/105] Allow creation of window on custom screen set by SDL_VIDEO_FULLSCREEN_HEAD --- Qt/mainwindow.cpp | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/Qt/mainwindow.cpp b/Qt/mainwindow.cpp index fe51934948..9710d86b67 100644 --- a/Qt/mainwindow.cpp +++ b/Qt/mainwindow.cpp @@ -25,6 +25,11 @@ MainWindow::MainWindow(QWidget *parent) : memoryTexWindow(0), displaylistWindow(0) { + // Allow creation on custom screen + int screenNum = QProcessEnvironment::systemEnvironment().value("SDL_VIDEO_FULLSCREEN_HEAD", "0").toInt(); + // Move window to top left coordinate of selected screen + move(qApp->desktop()->screen(screenNum)->pos()); + SetGameTitle(""); emugl = new MainUI(this); From adef4c3b70f66f6419bc9c4a6cf5235e9b55f39b Mon Sep 17 00:00:00 2001 From: Sacha Date: Thu, 9 Oct 2014 04:59:30 +1000 Subject: [PATCH 062/105] Remove unnecessary files from Blackberry build package. --- Blackberry/bar-descriptor.xml | 8 +++++--- native | 2 +- 2 files changed, 6 insertions(+), 4 deletions(-) diff --git a/Blackberry/bar-descriptor.xml b/Blackberry/bar-descriptor.xml index d01e5f4896..509c62e2b9 100644 --- a/Blackberry/bar-descriptor.xml +++ b/Blackberry/bar-descriptor.xml @@ -18,9 +18,11 @@ PPSSPPBlackberry icon-114.png - assets - assets/langregion.ini - assets/shaders + + + + + assets/lang assets/flash0 diff --git a/native b/native index 6ee862b564..d3743d0f39 160000 --- a/native +++ b/native @@ -1 +1 @@ -Subproject commit 6ee862b564bca639659a231937e02dfec0774475 +Subproject commit d3743d0f39554d68f1bb585e2b51892e2adbbdba From 3f3e464dae7d0ab669b29e32c2fb1c221ae42999 Mon Sep 17 00:00:00 2001 From: TwistedUmbrella Date: Wed, 8 Oct 2014 15:59:18 -0400 Subject: [PATCH 063/105] Revert "iOS: Add launch xib name to info.plist" This reverts commit b7db78362db88852f99182b400e1a52603f2e259. --- ios/PPSSPP-Info.plist | 2 -- 1 file changed, 2 deletions(-) diff --git a/ios/PPSSPP-Info.plist b/ios/PPSSPP-Info.plist index 9f13bf9b8e..0408e4b079 100644 --- a/ios/PPSSPP-Info.plist +++ b/ios/PPSSPP-Info.plist @@ -58,7 +58,5 @@ 1 2 - UILaunchStoryboardName - LaunchScreen From f4216483cbdb6aca9eaf231e90d36314e0aaf972 Mon Sep 17 00:00:00 2001 From: TwistedUmbrella Date: Wed, 8 Oct 2014 16:00:01 -0400 Subject: [PATCH 064/105] Revert "iOS: install LaunchScreen.xib" This reverts commit 39716c68b3da13de88fe4b7c5ba068d70c84c9a8. --- CMakeLists.txt | 7 +------ 1 file changed, 1 insertion(+), 6 deletions(-) diff --git a/CMakeLists.txt b/CMakeLists.txt index 8e4add2db5..ed095a5f39 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -1500,12 +1500,7 @@ if (TargetBin) set_source_files_properties(${LANG_FILES} PROPERTIES MACOSX_PACKAGE_LOCATION "MacOS/assets/lang") set_source_files_properties(${SHADER_FILES} PROPERTIES MACOSX_PACKAGE_LOCATION "MacOS/assets/shaders") - if (IOS) - set(LAUNCH_SCREEN_XIB ${CMAKE_CURRENT_SOURCE_DIR}/ios/assets/LaunchScreen.xib) - set_source_files_properties(${LAUNCH_SCREEN_XIB} PROPERTIES MACOSX_PACKAGE_LOCATION "Resources") - endif() - - add_executable(${TargetBin} MACOSX_BUNDLE ${ICON_PATH_ABS} ${NativeAssets} ${SHADER_FILES} ${FLASH0_FILES} ${LANG_FILES} ${NativeAppSource} ${LAUNCH_SCREEN_XIB}) + add_executable(${TargetBin} MACOSX_BUNDLE ${ICON_PATH_ABS} ${NativeAssets} ${SHADER_FILES} ${FLASH0_FILES} ${LANG_FILES} ${NativeAppSource}) else() add_executable(${TargetBin} ${NativeAppSource}) endif() From ea67baa45bbd454f0d2e104566471238d343a3ce Mon Sep 17 00:00:00 2001 From: TwistedUmbrella Date: Wed, 8 Oct 2014 16:00:16 -0400 Subject: [PATCH 065/105] Revert "iOS: add LaunchScreen.xib for support iPhone 6 and 6 Plus native screen resolution" This reverts commit 4dc6e268014d229fa7aa12cd3a5b163fbe15ff9f. --- ios/assets/LaunchScreen.xib | 18 ------------------ 1 file changed, 18 deletions(-) delete mode 100644 ios/assets/LaunchScreen.xib diff --git a/ios/assets/LaunchScreen.xib b/ios/assets/LaunchScreen.xib deleted file mode 100644 index 70ae06b281..0000000000 --- a/ios/assets/LaunchScreen.xib +++ /dev/null @@ -1,18 +0,0 @@ - - - - - - - - - - - - - - - - - - From 31b07d557137aea204efe83dd0359d058eca8c76 Mon Sep 17 00:00:00 2001 From: TwistedUmbrella Date: Wed, 8 Oct 2014 16:00:36 -0400 Subject: [PATCH 066/105] iOS: MacOS compatible post-build command --- CMakeLists.txt | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/CMakeLists.txt b/CMakeLists.txt index ed095a5f39..ec1fe4f293 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -1530,7 +1530,7 @@ if(IOS) ) set(APP_DIR_NAME \${TARGET_BUILD_DIR}/\${FULL_PRODUCT_NAME}) add_custom_command(TARGET PPSSPP POST_BUILD - COMMAND tar -c -C . --exclude .DS_Store --exclude .git -H assets | tar -x -C '${APP_DIR_NAME}' + COMMAND tar -c -C . --exclude .DS_Store --exclude .git -H gnu assets | tar -x -C '${APP_DIR_NAME}' ) # Force Xcode to relink the binary. add_custom_command(TARGET Core PRE_BUILD From a18344e15be82fc9bf6e3731668dd8c6c836e838 Mon Sep 17 00:00:00 2001 From: sergiobenrocha2 Date: Fri, 10 Oct 2014 01:28:35 -0300 Subject: [PATCH 067/105] Fix Qt build in Linux, problem to recognize SDL2. --- Qt/PPSSPP.pro | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/Qt/PPSSPP.pro b/Qt/PPSSPP.pro index d7efc73e98..b27357d982 100644 --- a/Qt/PPSSPP.pro +++ b/Qt/PPSSPP.pro @@ -37,11 +37,11 @@ greaterThan(QT_MAJOR_VERSION,4) { macx|equals(PLATFORM_NAME, "linux") { PRE_TARGETDEPS += $$CONFIG_DIR/libCommon.a $$CONFIG_DIR/libCore.a $$CONFIG_DIR/libGPU.a $$CONFIG_DIR/libNative.a CONFIG += link_pkgconfig - packagesExist(sdl) { + packagesExist(sdl2) { DEFINES += QT_HAS_SDL SOURCES += $$P/SDL/SDLJoystick.cpp HEADERS += $$P/SDL/SDLJoystick.h - PKGCONFIG += sdl + PKGCONFIG += sdl2 macx { LIBS += -F/Library/Frameworks -framework SDL INCLUDEPATH += /Library/Frameworks/SDL.framework/Versions/A/Headers From 36291c5af7a0bc816d199e6291cd49532d71159c Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Henrik=20Rydg=C3=A5rd?= Date: Sat, 11 Oct 2014 14:07:06 +0200 Subject: [PATCH 068/105] Update lang, adding Bulgarian --- assets/langregion.ini | 1 + lang | 2 +- 2 files changed, 2 insertions(+), 1 deletion(-) diff --git a/assets/langregion.ini b/assets/langregion.ini index 8a4db05f40..b41f179597 100644 --- a/assets/langregion.ini +++ b/assets/langregion.ini @@ -36,6 +36,7 @@ fa_IR = "فارسی" ms_MY = "Melayu" da_DK = "Dansk" no_NO = "Norsk" +bg_BG = "български език" [SystemLanguage] ja_JP = "JAPANESE" diff --git a/lang b/lang index d72c79ec7c..9b04681701 160000 --- a/lang +++ b/lang @@ -1 +1 @@ -Subproject commit d72c79ec7c2d092a2b8d1158992ce03a2c7271fd +Subproject commit 9b046817012c0bdaa2df204dbfd2f7d2ea01f41b From 91966824bbb8dc54783d9064ffe8260d81e86dc0 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Henrik=20Rydg=C3=A5rd?= Date: Sat, 11 Oct 2014 15:57:08 +0200 Subject: [PATCH 069/105] minor cleanup: No point in having special functions for ReadFCR/WriteFCR, they're smaller than many other ops.. --- Core/MIPS/MIPS.cpp | 25 ------------------------ Core/MIPS/MIPS.h | 3 --- Core/MIPS/MIPSInt.cpp | 45 +++++++++++++++++++++++++++++++++++-------- 3 files changed, 37 insertions(+), 36 deletions(-) diff --git a/Core/MIPS/MIPS.cpp b/Core/MIPS/MIPS.cpp index 9e8ad38d5b..2277169f80 100644 --- a/Core/MIPS/MIPS.cpp +++ b/Core/MIPS/MIPS.cpp @@ -305,31 +305,6 @@ int MIPSState::RunLoopUntil(u64 globalTicks) { return 1; } -void MIPSState::WriteFCR(int reg, int value) { - if (reg == 31) { - fcr31 = value & 0x0181FFFF; - fpcond = (value >> 23) & 1; - } else { - WARN_LOG_REPORT(CPU, "WriteFCR: Unexpected reg %d (value %08x)", reg, value); - // MessageBox(0, "Invalid FCR","...",0); - } - DEBUG_LOG(CPU, "FCR%i written to, value %08x", reg, value); -} - -u32 MIPSState::ReadFCR(int reg) { - DEBUG_LOG(CPU,"FCR%i read",reg); - if (reg == 31) { - fcr31 = (fcr31 & ~(1<<23)) | ((fpcond & 1)<<23); - return fcr31; - } else if (reg == 0) { - return FCR0_VALUE; - } else { - WARN_LOG_REPORT(CPU, "ReadFCR: Unexpected reg %d", reg); - // MessageBox(0, "Invalid FCR","...",0); - } - return 0; -} - void MIPSState::InvalidateICache(u32 address, int length) { // Only really applies to jit. if (MIPSComp::jit) diff --git a/Core/MIPS/MIPS.h b/Core/MIPS/MIPS.h index c9e7c35cc8..4a69dcd345 100644 --- a/Core/MIPS/MIPS.h +++ b/Core/MIPS/MIPS.h @@ -178,9 +178,6 @@ public: static const u32 FCR0_VALUE = 0x00003351; - void WriteFCR(int reg, int value); - u32 ReadFCR(int reg); - u8 VfpuWriteMask() const { return (vfpuCtrl[VFPU_CTRL_DPREFIX] >> 8) & 0xF; } diff --git a/Core/MIPS/MIPSInt.cpp b/Core/MIPS/MIPSInt.cpp index 45d88f5f43..330f29a054 100644 --- a/Core/MIPS/MIPSInt.cpp +++ b/Core/MIPS/MIPSInt.cpp @@ -527,12 +527,41 @@ namespace MIPSInt int fs = _FS; int rt = _RT; - switch((op>>21)&0x1f) - { - case 0: if (rt != 0) R(rt) = FI(fs); break; //mfc1 - case 2: if (rt != 0) R(rt) = currentMIPS->ReadFCR(fs); break; //cfc1 - case 4: FI(fs) = R(rt); break; //mtc1 - case 6: currentMIPS->WriteFCR(fs, R(rt)); break; //ctc1 + switch ((op>>21)&0x1f) { + case 0: //mfc1 + if (rt != 0) + R(rt) = FI(fs); + break; + + case 2: //cfc1 + if (rt != 0) { + if (fs == 31) { + currentMIPS->fcr31 = (currentMIPS->fcr31 & ~(1<<23)) | ((currentMIPS->fpcond & 1)<<23); + R(rt) = currentMIPS->fcr31; + } else if (fs == 0) { + R(rt) = MIPSState::FCR0_VALUE; + } else { + WARN_LOG_REPORT(CPU, "ReadFCR: Unexpected reg %d", fs); + } + break; + } + + case 4: //mtc1 + FI(fs) = R(rt); + break; + + case 6: //ctc1 + { + u32 value = R(rt); + if (fs == 31) { + currentMIPS->fcr31 = value & 0x0181FFFF; + currentMIPS->fpcond = (value >> 23) & 1; + } else { + WARN_LOG_REPORT(CPU, "WriteFCR: Unexpected reg %d (value %08x)", fs, value); + } + DEBUG_LOG(CPU, "FCR%i written to, value %08x", fs, value); + break; + } default: _dbg_assert_msg_(CPU,0,"Trying to interpret instruction that can't be interpreted"); @@ -818,14 +847,14 @@ namespace MIPSInt static int reported = 0; switch (op & 0x3F) { - case 36: + case 36: // mfic if (!reported) { Reporting::ReportMessage("MFIC instruction hit (%08x) at %08x", op, currentMIPS->pc); WARN_LOG(CPU,"MFIC Disable/Enable Interrupt CPU instruction"); reported = 1; } break; - case 38: + case 38: // mtic if (!reported) { Reporting::ReportMessage("MTIC instruction hit (%08x) at %08x", op, currentMIPS->pc); WARN_LOG(CPU,"MTIC Disable/Enable Interrupt CPU instruction"); From cb6634f54b8c936addcda4132393bfcd8e9ca797 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Sat, 11 Oct 2014 08:42:01 -0700 Subject: [PATCH 070/105] Add udis86 b24baf1 (1.7.3?), not yet used. Linked even on arm to avoid dependency complexity, might skip later. --- CMakeLists.txt | 20 +- Core/Core.vcxproj | 13 + Core/Core.vcxproj.filters | 42 + Qt/Native.pro | 6 + android/jni/Android.mk | 6 + ext/udis86/LICENSE | 22 + ext/udis86/decode.c | 1265 ++++++++ ext/udis86/decode.h | 197 ++ ext/udis86/extern.h | 113 + ext/udis86/itab.c | 5937 +++++++++++++++++++++++++++++++++++++ ext/udis86/itab.h | 935 ++++++ ext/udis86/syn-att.c | 228 ++ ext/udis86/syn-intel.c | 224 ++ ext/udis86/syn.c | 212 ++ ext/udis86/syn.h | 53 + ext/udis86/types.h | 259 ++ ext/udis86/udint.h | 93 + ext/udis86/udis86.c | 456 +++ ext/udis86/udis86.h | 33 + 19 files changed, 10113 insertions(+), 1 deletion(-) create mode 100644 ext/udis86/LICENSE create mode 100644 ext/udis86/decode.c create mode 100644 ext/udis86/decode.h create mode 100644 ext/udis86/extern.h create mode 100644 ext/udis86/itab.c create mode 100644 ext/udis86/itab.h create mode 100644 ext/udis86/syn-att.c create mode 100644 ext/udis86/syn-intel.c create mode 100644 ext/udis86/syn.c create mode 100644 ext/udis86/syn.h create mode 100644 ext/udis86/types.h create mode 100644 ext/udis86/udint.h create mode 100644 ext/udis86/udis86.c create mode 100644 ext/udis86/udis86.h diff --git a/CMakeLists.txt b/CMakeLists.txt index ec1fe4f293..74eb39bd30 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -359,6 +359,23 @@ add_library(snappy STATIC ) include_directories(ext/snappy) +add_library(udis86 STATIC + ext/udis86/decode.c + ext/udis86/decode.h + ext/udis86/extern.h + ext/udis86/itab.c + ext/udis86/itab.h + ext/udis86/syn-att.c + ext/udis86/syn-intel.c + ext/udis86/syn.c + ext/udis86/syn.h + ext/udis86/types.h + ext/udis86/udint.h + ext/udis86/udis86.c + ext/udis86/udis86.h +) +include_directories(ext/udis86) + add_library(vjson STATIC native/ext/vjson/json.cpp native/ext/vjson/json.h @@ -555,6 +572,7 @@ include_directories(ext/cityhash) if (NOT MSVC) # These can be fast even for debug. set_target_properties(snappy PROPERTIES COMPILE_FLAGS "-O3") + set_target_properties(udis86 PROPERTIES COMPILE_FLAGS "-O3") set_target_properties(cityhash PROPERTIES COMPILE_FLAGS "-O3") if(NOT ZLIB_FOUND) set_target_properties(zlib PROPERTIES COMPILE_FLAGS "-O3") @@ -949,7 +967,7 @@ if (LINUX AND NOT ANDROID) SET(RT_LIB rt) endif() -target_link_libraries(native ${LIBZIP_LIBRARY} ${PNG_LIBRARY} rg_etc1 vjson stb_vorbis snappy ${RT_LIB} ${GLEW_LIBRARIES}) +target_link_libraries(native ${LIBZIP_LIBRARY} ${PNG_LIBRARY} rg_etc1 vjson stb_vorbis snappy udis86 ${RT_LIB} ${GLEW_LIBRARIES}) if(ANDROID) target_link_libraries(native log EGL) diff --git a/Core/Core.vcxproj b/Core/Core.vcxproj index 68fbb66c16..3036bdb00a 100644 --- a/Core/Core.vcxproj +++ b/Core/Core.vcxproj @@ -169,6 +169,12 @@ + + + + + + @@ -441,6 +447,13 @@ + + + + + + + diff --git a/Core/Core.vcxproj.filters b/Core/Core.vcxproj.filters index 0b12abd76c..5bcc838c6a 100644 --- a/Core/Core.vcxproj.filters +++ b/Core/Core.vcxproj.filters @@ -55,6 +55,9 @@ {ac0e396a-6e16-4796-a0a0-b8ebf46ea74c} + + {435eb15d-386c-44df-97b4-343a1d6524ec} + @@ -532,6 +535,24 @@ HLE\Libraries + + Ext\udis86 + + + Ext\udis86 + + + Ext\udis86 + + + Ext\udis86 + + + Ext\udis86 + + + Ext\udis86 + @@ -984,6 +1005,27 @@ HLE\Libraries + + Ext\udis86 + + + Ext\udis86 + + + Ext\udis86 + + + Ext\udis86 + + + Ext\udis86 + + + Ext\udis86 + + + Ext\udis86 + diff --git a/Qt/Native.pro b/Qt/Native.pro index 8703641f6b..abccc61a3d 100644 --- a/Qt/Native.pro +++ b/Qt/Native.pro @@ -46,6 +46,12 @@ SOURCES += $$P/ext/snappy/*.cpp HEADERS += $$P/ext/snappy/*.h INCLUDEPATH += $$P/ext/snappy +# udis86 + +SOURCES += $$P/ext/udis86/*.c +HEADERS += $$P/ext/udis86/*.h +INCLUDEPATH += $$P/ext/udis86 + # VJSON SOURCES += $$P/native/ext/vjson/json.cpp \ diff --git a/android/jni/Android.mk b/android/jni/Android.mk index 9c7fdddc26..90df331681 100644 --- a/android/jni/Android.mk +++ b/android/jni/Android.mk @@ -117,6 +117,12 @@ EXEC_AND_LIB_FILES := \ $(SRC)/ext/libkirk/kirk_engine.c \ $(SRC)/ext/snappy/snappy-c.cpp \ $(SRC)/ext/snappy/snappy.cpp \ + $(SRC)/ext/udis86/decode.c \ + $(SRC)/ext/udis86/itab.c \ + $(SRC)/ext/udis86/syn-att.c \ + $(SRC)/ext/udis86/syn-intel.c \ + $(SRC)/ext/udis86/syn.c \ + $(SRC)/ext/udis86/udis86.c \ $(SRC)/ext/xbrz/xbrz.cpp \ $(SRC)/ext/xxhash.c \ $(SRC)/Common/Crypto/md5.cpp \ diff --git a/ext/udis86/LICENSE b/ext/udis86/LICENSE new file mode 100644 index 0000000000..580f35987f --- /dev/null +++ b/ext/udis86/LICENSE @@ -0,0 +1,22 @@ +Copyright (c) 2002-2012, Vivek Thampi +All rights reserved. + +Redistribution and use in source and binary forms, with or without modification, +are permitted provided that the following conditions are met: + +1. Redistributions of source code must retain the above copyright notice, + this list of conditions and the following disclaimer. +2. Redistributions in binary form must reproduce the above copyright notice, + this list of conditions and the following disclaimer in the documentation + and/or other materials provided with the distribution. + +THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND +ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED +WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE +DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE FOR +ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES +(INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; +LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON +ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT +(INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS +SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. diff --git a/ext/udis86/decode.c b/ext/udis86/decode.c new file mode 100644 index 0000000000..83e8e185bc --- /dev/null +++ b/ext/udis86/decode.c @@ -0,0 +1,1265 @@ +/* udis86 - libudis86/decode.c + * + * Copyright (c) 2002-2009 Vivek Thampi + * All rights reserved. + * + * Redistribution and use in source and binary forms, with or without modification, + * are permitted provided that the following conditions are met: + * + * * Redistributions of source code must retain the above copyright notice, + * this list of conditions and the following disclaimer. + * * Redistributions in binary form must reproduce the above copyright notice, + * this list of conditions and the following disclaimer in the documentation + * and/or other materials provided with the distribution. + * + * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND + * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED + * WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE + * DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE FOR + * ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES + * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; + * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON + * ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT + * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + */ +#include "udint.h" +#include "types.h" +#include "extern.h" +#include "decode.h" + +#ifndef __UD_STANDALONE__ +# include +#endif /* __UD_STANDALONE__ */ + +/* The max number of prefixes to an instruction */ +#define MAX_PREFIXES 15 + +/* rex prefix bits */ +#define REX_W(r) ( ( 0xF & ( r ) ) >> 3 ) +#define REX_R(r) ( ( 0x7 & ( r ) ) >> 2 ) +#define REX_X(r) ( ( 0x3 & ( r ) ) >> 1 ) +#define REX_B(r) ( ( 0x1 & ( r ) ) >> 0 ) +#define REX_PFX_MASK(n) ( ( P_REXW(n) << 3 ) | \ + ( P_REXR(n) << 2 ) | \ + ( P_REXX(n) << 1 ) | \ + ( P_REXB(n) << 0 ) ) + +/* scable-index-base bits */ +#define SIB_S(b) ( ( b ) >> 6 ) +#define SIB_I(b) ( ( ( b ) >> 3 ) & 7 ) +#define SIB_B(b) ( ( b ) & 7 ) + +/* modrm bits */ +#define MODRM_REG(b) ( ( ( b ) >> 3 ) & 7 ) +#define MODRM_NNN(b) ( ( ( b ) >> 3 ) & 7 ) +#define MODRM_MOD(b) ( ( ( b ) >> 6 ) & 3 ) +#define MODRM_RM(b) ( ( b ) & 7 ) + +static int decode_ext(struct ud *u, uint16_t ptr); +static int decode_opcode(struct ud *u); + +enum reg_class { /* register classes */ + REGCLASS_GPR, + REGCLASS_MMX, + REGCLASS_CR, + REGCLASS_DB, + REGCLASS_SEG, + REGCLASS_XMM +}; + + /* + * inp_start + * Should be called before each de-code operation. + */ +static void +inp_start(struct ud *u) +{ + u->inp_ctr = 0; +} + +static uint8_t +inp_peek(struct ud *u) +{ + if (u->inp_end == 0) { + if (u->inp_buf != NULL) { + if (u->inp_buf_index < u->inp_buf_size) { + return u->inp_buf[u->inp_buf_index]; + } + } else if (u->inp_peek != UD_EOI) { + return u->inp_peek; + } else { + int c; + if ((c = u->inp_hook(u)) != UD_EOI) { + u->inp_peek = c; + return u->inp_peek; + } + } + } + u->inp_end = 1; + UDERR(u, "byte expected, eoi received\n"); + return 0; +} + +static uint8_t +inp_next(struct ud *u) +{ + if (u->inp_end == 0) { + if (u->inp_buf != NULL) { + if (u->inp_buf_index < u->inp_buf_size) { + u->inp_ctr++; + return (u->inp_curr = u->inp_buf[u->inp_buf_index++]); + } + } else { + int c = u->inp_peek; + if (c != UD_EOI || (c = u->inp_hook(u)) != UD_EOI) { + u->inp_peek = UD_EOI; + u->inp_curr = c; + u->inp_sess[u->inp_ctr++] = u->inp_curr; + return u->inp_curr; + } + } + } + u->inp_end = 1; + UDERR(u, "byte expected, eoi received\n"); + return 0; +} + +static uint8_t +inp_curr(struct ud *u) +{ + return u->inp_curr; +} + + +/* + * inp_uint8 + * int_uint16 + * int_uint32 + * int_uint64 + * Load little-endian values from input + */ +static uint8_t +inp_uint8(struct ud* u) +{ + return inp_next(u); +} + +static uint16_t +inp_uint16(struct ud* u) +{ + uint16_t r, ret; + + ret = inp_next(u); + r = inp_next(u); + return ret | (r << 8); +} + +static uint32_t +inp_uint32(struct ud* u) +{ + uint32_t r, ret; + + ret = inp_next(u); + r = inp_next(u); + ret = ret | (r << 8); + r = inp_next(u); + ret = ret | (r << 16); + r = inp_next(u); + return ret | (r << 24); +} + +static uint64_t +inp_uint64(struct ud* u) +{ + uint64_t r, ret; + + ret = inp_next(u); + r = inp_next(u); + ret = ret | (r << 8); + r = inp_next(u); + ret = ret | (r << 16); + r = inp_next(u); + ret = ret | (r << 24); + r = inp_next(u); + ret = ret | (r << 32); + r = inp_next(u); + ret = ret | (r << 40); + r = inp_next(u); + ret = ret | (r << 48); + r = inp_next(u); + return ret | (r << 56); +} + + +static UD_INLINE int +eff_opr_mode(int dis_mode, int rex_w, int pfx_opr) +{ + if (dis_mode == 64) { + return rex_w ? 64 : (pfx_opr ? 16 : 32); + } else if (dis_mode == 32) { + return pfx_opr ? 16 : 32; + } else { + UD_ASSERT(dis_mode == 16); + return pfx_opr ? 32 : 16; + } +} + + +static UD_INLINE int +eff_adr_mode(int dis_mode, int pfx_adr) +{ + if (dis_mode == 64) { + return pfx_adr ? 32 : 64; + } else if (dis_mode == 32) { + return pfx_adr ? 16 : 32; + } else { + UD_ASSERT(dis_mode == 16); + return pfx_adr ? 32 : 16; + } +} + + +/* + * decode_prefixes + * + * Extracts instruction prefixes. + */ +static int +decode_prefixes(struct ud *u) +{ + int done = 0; + uint8_t curr = 0, last = 0; + UD_RETURN_ON_ERROR(u); + + do { + last = curr; + curr = inp_next(u); + UD_RETURN_ON_ERROR(u); + if (u->inp_ctr == MAX_INSN_LENGTH) { + UD_RETURN_WITH_ERROR(u, "max instruction length"); + } + + switch (curr) + { + case 0x2E: + u->pfx_seg = UD_R_CS; + break; + case 0x36: + u->pfx_seg = UD_R_SS; + break; + case 0x3E: + u->pfx_seg = UD_R_DS; + break; + case 0x26: + u->pfx_seg = UD_R_ES; + break; + case 0x64: + u->pfx_seg = UD_R_FS; + break; + case 0x65: + u->pfx_seg = UD_R_GS; + break; + case 0x67: /* adress-size override prefix */ + u->pfx_adr = 0x67; + break; + case 0xF0: + u->pfx_lock = 0xF0; + break; + case 0x66: + u->pfx_opr = 0x66; + break; + case 0xF2: + u->pfx_str = 0xf2; + break; + case 0xF3: + u->pfx_str = 0xf3; + break; + default: + /* consume if rex */ + done = (u->dis_mode == 64 && (curr & 0xF0) == 0x40) ? 0 : 1; + break; + } + } while (!done); + /* rex prefixes in 64bit mode, must be the last prefix */ + if (u->dis_mode == 64 && (last & 0xF0) == 0x40) { + u->pfx_rex = last; + } + return 0; +} + + +/* + * vex_l, vex_w + * Return the vex.L and vex.W bits + */ +static UD_INLINE uint8_t +vex_l(const struct ud *u) +{ + UD_ASSERT(u->vex_op != 0); + return ((u->vex_op == 0xc4 ? u->vex_b2 : u->vex_b1) >> 2) & 1; +} + +static UD_INLINE uint8_t +vex_w(const struct ud *u) +{ + UD_ASSERT(u->vex_op != 0); + return u->vex_op == 0xc4 ? ((u->vex_b2 >> 7) & 1) : 0; +} + + +static UD_INLINE uint8_t +modrm(struct ud * u) +{ + if ( !u->have_modrm ) { + u->modrm = inp_next( u ); + u->have_modrm = 1; + } + return u->modrm; +} + + +static unsigned int +resolve_operand_size(const struct ud* u, ud_operand_size_t osize) +{ + switch (osize) { + case SZ_V: + return u->opr_mode; + case SZ_Z: + return u->opr_mode == 16 ? 16 : 32; + case SZ_Y: + return u->opr_mode == 16 ? 32 : u->opr_mode; + case SZ_RDQ: + return u->dis_mode == 64 ? 64 : 32; + case SZ_X: + UD_ASSERT(u->vex_op != 0); + return (P_VEXL(u->itab_entry->prefix) && vex_l(u)) ? SZ_QQ : SZ_DQ; + default: + return osize; + } +} + + +static int resolve_mnemonic( struct ud* u ) +{ + /* resolve 3dnow weirdness. */ + if ( u->mnemonic == UD_I3dnow ) { + u->mnemonic = ud_itab[ u->le->table[ inp_curr( u ) ] ].mnemonic; + } + /* SWAPGS is only valid in 64bits mode */ + if ( u->mnemonic == UD_Iswapgs && u->dis_mode != 64 ) { + UDERR(u, "swapgs invalid in 64bits mode\n"); + return -1; + } + + if (u->mnemonic == UD_Ixchg) { + if ((u->operand[0].type == UD_OP_REG && u->operand[0].base == UD_R_AX && + u->operand[1].type == UD_OP_REG && u->operand[1].base == UD_R_AX) || + (u->operand[0].type == UD_OP_REG && u->operand[0].base == UD_R_EAX && + u->operand[1].type == UD_OP_REG && u->operand[1].base == UD_R_EAX)) { + u->operand[0].type = UD_NONE; + u->operand[1].type = UD_NONE; + u->mnemonic = UD_Inop; + } + } + + if (u->mnemonic == UD_Inop && u->pfx_repe) { + u->pfx_repe = 0; + u->mnemonic = UD_Ipause; + } + return 0; +} + + +/* ----------------------------------------------------------------------------- + * decode_a()- Decodes operands of the type seg:offset + * ----------------------------------------------------------------------------- + */ +static void +decode_a(struct ud* u, struct ud_operand *op) +{ + if (u->opr_mode == 16) { + /* seg16:off16 */ + op->type = UD_OP_PTR; + op->size = 32; + op->lval.ptr.off = inp_uint16(u); + op->lval.ptr.seg = inp_uint16(u); + } else { + /* seg16:off32 */ + op->type = UD_OP_PTR; + op->size = 48; + op->lval.ptr.off = inp_uint32(u); + op->lval.ptr.seg = inp_uint16(u); + } +} + +/* ----------------------------------------------------------------------------- + * decode_gpr() - Returns decoded General Purpose Register + * ----------------------------------------------------------------------------- + */ +static enum ud_type +decode_gpr(register struct ud* u, unsigned int s, unsigned char rm) +{ + switch (s) { + case 64: + return UD_R_RAX + rm; + case 32: + return UD_R_EAX + rm; + case 16: + return UD_R_AX + rm; + case 8: + if (u->dis_mode == 64 && u->pfx_rex) { + if (rm >= 4) + return UD_R_SPL + (rm-4); + return UD_R_AL + rm; + } else return UD_R_AL + rm; + case 0: + /* invalid size in case of a decode error */ + UD_ASSERT(u->error); + return UD_NONE; + default: + UD_ASSERT(!"invalid operand size"); + return UD_NONE; + } +} + +static void +decode_reg(struct ud *u, + struct ud_operand *opr, + int type, + int num, + int size) +{ + int reg; + size = resolve_operand_size(u, size); + switch (type) { + case REGCLASS_GPR : reg = decode_gpr(u, size, num); break; + case REGCLASS_MMX : reg = UD_R_MM0 + (num & 7); break; + case REGCLASS_XMM : + reg = num + (size == SZ_QQ ? UD_R_YMM0 : UD_R_XMM0); + break; + case REGCLASS_CR : reg = UD_R_CR0 + num; break; + case REGCLASS_DB : reg = UD_R_DR0 + num; break; + case REGCLASS_SEG : { + /* + * Only 6 segment registers, anything else is an error. + */ + if ((num & 7) > 5) { + UDERR(u, "invalid segment register value\n"); + return; + } else { + reg = UD_R_ES + (num & 7); + } + break; + } + default: + UD_ASSERT(!"invalid register type"); + return; + } + opr->type = UD_OP_REG; + opr->base = reg; + opr->size = size; +} + + +/* + * decode_imm + * + * Decode Immediate values. + */ +static void +decode_imm(struct ud* u, unsigned int size, struct ud_operand *op) +{ + op->size = resolve_operand_size(u, size); + op->type = UD_OP_IMM; + + switch (op->size) { + case 8: op->lval.sbyte = inp_uint8(u); break; + case 16: op->lval.uword = inp_uint16(u); break; + case 32: op->lval.udword = inp_uint32(u); break; + case 64: op->lval.uqword = inp_uint64(u); break; + default: return; + } +} + + +/* + * decode_mem_disp + * + * Decode mem address displacement. + */ +static void +decode_mem_disp(struct ud* u, unsigned int size, struct ud_operand *op) +{ + switch (size) { + case 8: + op->offset = 8; + op->lval.ubyte = inp_uint8(u); + break; + case 16: + op->offset = 16; + op->lval.uword = inp_uint16(u); + break; + case 32: + op->offset = 32; + op->lval.udword = inp_uint32(u); + break; + case 64: + op->offset = 64; + op->lval.uqword = inp_uint64(u); + break; + default: + return; + } +} + + +/* + * decode_modrm_reg + * + * Decodes reg field of mod/rm byte + * + */ +static UD_INLINE void +decode_modrm_reg(struct ud *u, + struct ud_operand *operand, + unsigned int type, + unsigned int size) +{ + uint8_t reg = (REX_R(u->_rex) << 3) | MODRM_REG(modrm(u)); + decode_reg(u, operand, type, reg, size); +} + + +/* + * decode_modrm_rm + * + * Decodes rm field of mod/rm byte + * + */ +static void +decode_modrm_rm(struct ud *u, + struct ud_operand *op, + unsigned char type, /* register type */ + unsigned int size) /* operand size */ + +{ + size_t offset = 0; + unsigned char mod, rm; + + /* get mod, r/m and reg fields */ + mod = MODRM_MOD(modrm(u)); + rm = (REX_B(u->_rex) << 3) | MODRM_RM(modrm(u)); + + /* + * If mod is 11b, then the modrm.rm specifies a register. + * + */ + if (mod == 3) { + decode_reg(u, op, type, rm, size); + return; + } + + /* + * !11b => Memory Address + */ + op->type = UD_OP_MEM; + op->size = resolve_operand_size(u, size); + + if (u->adr_mode == 64) { + op->base = UD_R_RAX + rm; + if (mod == 1) { + offset = 8; + } else if (mod == 2) { + offset = 32; + } else if (mod == 0 && (rm & 7) == 5) { + op->base = UD_R_RIP; + offset = 32; + } else { + offset = 0; + } + /* + * Scale-Index-Base (SIB) + */ + if ((rm & 7) == 4) { + inp_next(u); + + op->base = UD_R_RAX + (SIB_B(inp_curr(u)) | (REX_B(u->_rex) << 3)); + op->index = UD_R_RAX + (SIB_I(inp_curr(u)) | (REX_X(u->_rex) << 3)); + /* special conditions for base reference */ + if (op->index == UD_R_RSP) { + op->index = UD_NONE; + op->scale = UD_NONE; + } else { + op->scale = (1 << SIB_S(inp_curr(u))) & ~1; + } + + if (op->base == UD_R_RBP || op->base == UD_R_R13) { + if (mod == 0) { + op->base = UD_NONE; + } + if (mod == 1) { + offset = 8; + } else { + offset = 32; + } + } + } else { + op->scale = UD_NONE; + op->index = UD_NONE; + } + } else if (u->adr_mode == 32) { + op->base = UD_R_EAX + rm; + if (mod == 1) { + offset = 8; + } else if (mod == 2) { + offset = 32; + } else if (mod == 0 && rm == 5) { + op->base = UD_NONE; + offset = 32; + } else { + offset = 0; + } + + /* Scale-Index-Base (SIB) */ + if ((rm & 7) == 4) { + inp_next(u); + + op->scale = (1 << SIB_S(inp_curr(u))) & ~1; + op->index = UD_R_EAX + (SIB_I(inp_curr(u)) | (REX_X(u->pfx_rex) << 3)); + op->base = UD_R_EAX + (SIB_B(inp_curr(u)) | (REX_B(u->pfx_rex) << 3)); + + if (op->index == UD_R_ESP) { + op->index = UD_NONE; + op->scale = UD_NONE; + } + + /* special condition for base reference */ + if (op->base == UD_R_EBP) { + if (mod == 0) { + op->base = UD_NONE; + } + if (mod == 1) { + offset = 8; + } else { + offset = 32; + } + } + } else { + op->scale = UD_NONE; + op->index = UD_NONE; + } + } else { + const unsigned int bases[] = { UD_R_BX, UD_R_BX, UD_R_BP, UD_R_BP, + UD_R_SI, UD_R_DI, UD_R_BP, UD_R_BX }; + const unsigned int indices[] = { UD_R_SI, UD_R_DI, UD_R_SI, UD_R_DI, + UD_NONE, UD_NONE, UD_NONE, UD_NONE }; + op->base = bases[rm & 7]; + op->index = indices[rm & 7]; + op->scale = UD_NONE; + if (mod == 0 && rm == 6) { + offset = 16; + op->base = UD_NONE; + } else if (mod == 1) { + offset = 8; + } else if (mod == 2) { + offset = 16; + } + } + + if (offset) { + decode_mem_disp(u, offset, op); + } else { + op->offset = 0; + } +} + + +/* + * decode_moffset + * Decode offset-only memory operand + */ +static void +decode_moffset(struct ud *u, unsigned int size, struct ud_operand *opr) +{ + opr->type = UD_OP_MEM; + opr->base = UD_NONE; + opr->index = UD_NONE; + opr->scale = UD_NONE; + opr->size = resolve_operand_size(u, size); + decode_mem_disp(u, u->adr_mode, opr); +} + + +static void +decode_vex_vvvv(struct ud *u, struct ud_operand *opr, unsigned size) +{ + uint8_t vvvv; + UD_ASSERT(u->vex_op != 0); + vvvv = ((u->vex_op == 0xc4 ? u->vex_b2 : u->vex_b1) >> 3) & 0xf; + decode_reg(u, opr, REGCLASS_XMM, (0xf & ~vvvv), size); +} + + +/* + * decode_vex_immreg + * Decode source operand encoded in immediate byte [7:4] + */ +static int +decode_vex_immreg(struct ud *u, struct ud_operand *opr, unsigned size) +{ + uint8_t imm = inp_next(u); + uint8_t mask = u->dis_mode == 64 ? 0xf : 0x7; + UD_RETURN_ON_ERROR(u); + UD_ASSERT(u->vex_op != 0); + decode_reg(u, opr, REGCLASS_XMM, mask & (imm >> 4), size); + return 0; +} + + +/* + * decode_operand + * + * Decodes a single operand. + * Returns the type of the operand (UD_NONE if none) + */ +static int +decode_operand(struct ud *u, + struct ud_operand *operand, + enum ud_operand_code type, + unsigned int size) +{ + operand->type = UD_NONE; + operand->_oprcode = type; + + switch (type) { + case OP_A : + decode_a(u, operand); + break; + case OP_MR: + decode_modrm_rm(u, operand, REGCLASS_GPR, + MODRM_MOD(modrm(u)) == 3 ? + Mx_reg_size(size) : Mx_mem_size(size)); + break; + case OP_F: + u->br_far = 1; + /* intended fall through */ + case OP_M: + if (MODRM_MOD(modrm(u)) == 3) { + UDERR(u, "expected modrm.mod != 3\n"); + } + /* intended fall through */ + case OP_E: + decode_modrm_rm(u, operand, REGCLASS_GPR, size); + break; + case OP_G: + decode_modrm_reg(u, operand, REGCLASS_GPR, size); + break; + case OP_sI: + case OP_I: + decode_imm(u, size, operand); + break; + case OP_I1: + operand->type = UD_OP_CONST; + operand->lval.udword = 1; + break; + case OP_N: + if (MODRM_MOD(modrm(u)) != 3) { + UDERR(u, "expected modrm.mod == 3\n"); + } + /* intended fall through */ + case OP_Q: + decode_modrm_rm(u, operand, REGCLASS_MMX, size); + break; + case OP_P: + decode_modrm_reg(u, operand, REGCLASS_MMX, size); + break; + case OP_U: + if (MODRM_MOD(modrm(u)) != 3) { + UDERR(u, "expected modrm.mod == 3\n"); + } + /* intended fall through */ + case OP_W: + decode_modrm_rm(u, operand, REGCLASS_XMM, size); + break; + case OP_V: + decode_modrm_reg(u, operand, REGCLASS_XMM, size); + break; + case OP_H: + decode_vex_vvvv(u, operand, size); + break; + case OP_MU: + decode_modrm_rm(u, operand, REGCLASS_XMM, + MODRM_MOD(modrm(u)) == 3 ? + Mx_reg_size(size) : Mx_mem_size(size)); + break; + case OP_S: + decode_modrm_reg(u, operand, REGCLASS_SEG, size); + break; + case OP_O: + decode_moffset(u, size, operand); + break; + case OP_R0: + case OP_R1: + case OP_R2: + case OP_R3: + case OP_R4: + case OP_R5: + case OP_R6: + case OP_R7: + decode_reg(u, operand, REGCLASS_GPR, + (REX_B(u->_rex) << 3) | (type - OP_R0), size); + break; + case OP_AL: + case OP_AX: + case OP_eAX: + case OP_rAX: + decode_reg(u, operand, REGCLASS_GPR, 0, size); + break; + case OP_CL: + case OP_CX: + case OP_eCX: + decode_reg(u, operand, REGCLASS_GPR, 1, size); + break; + case OP_DL: + case OP_DX: + case OP_eDX: + decode_reg(u, operand, REGCLASS_GPR, 2, size); + break; + case OP_ES: + case OP_CS: + case OP_DS: + case OP_SS: + case OP_FS: + case OP_GS: + /* in 64bits mode, only fs and gs are allowed */ + if (u->dis_mode == 64) { + if (type != OP_FS && type != OP_GS) { + UDERR(u, "invalid segment register in 64bits\n"); + } + } + operand->type = UD_OP_REG; + operand->base = (type - OP_ES) + UD_R_ES; + operand->size = 16; + break; + case OP_J : + decode_imm(u, size, operand); + operand->type = UD_OP_JIMM; + break ; + case OP_R : + if (MODRM_MOD(modrm(u)) != 3) { + UDERR(u, "expected modrm.mod == 3\n"); + } + decode_modrm_rm(u, operand, REGCLASS_GPR, size); + break; + case OP_C: + decode_modrm_reg(u, operand, REGCLASS_CR, size); + break; + case OP_D: + decode_modrm_reg(u, operand, REGCLASS_DB, size); + break; + case OP_I3 : + operand->type = UD_OP_CONST; + operand->lval.sbyte = 3; + break; + case OP_ST0: + case OP_ST1: + case OP_ST2: + case OP_ST3: + case OP_ST4: + case OP_ST5: + case OP_ST6: + case OP_ST7: + operand->type = UD_OP_REG; + operand->base = (type - OP_ST0) + UD_R_ST0; + operand->size = 80; + break; + case OP_L: + decode_vex_immreg(u, operand, size); + break; + default : + operand->type = UD_NONE; + break; + } + return operand->type; +} + + +/* + * decode_operands + * + * Disassemble upto 3 operands of the current instruction being + * disassembled. By the end of the function, the operand fields + * of the ud structure will have been filled. + */ +static int +decode_operands(struct ud* u) +{ + decode_operand(u, &u->operand[0], + u->itab_entry->operand1.type, + u->itab_entry->operand1.size); + if (u->operand[0].type != UD_NONE) { + decode_operand(u, &u->operand[1], + u->itab_entry->operand2.type, + u->itab_entry->operand2.size); + } + if (u->operand[1].type != UD_NONE) { + decode_operand(u, &u->operand[2], + u->itab_entry->operand3.type, + u->itab_entry->operand3.size); + } + if (u->operand[2].type != UD_NONE) { + decode_operand(u, &u->operand[3], + u->itab_entry->operand4.type, + u->itab_entry->operand4.size); + } + return 0; +} + +/* ----------------------------------------------------------------------------- + * clear_insn() - clear instruction structure + * ----------------------------------------------------------------------------- + */ +static void +clear_insn(register struct ud* u) +{ + u->error = 0; + u->pfx_seg = 0; + u->pfx_opr = 0; + u->pfx_adr = 0; + u->pfx_lock = 0; + u->pfx_repne = 0; + u->pfx_rep = 0; + u->pfx_repe = 0; + u->pfx_rex = 0; + u->pfx_str = 0; + u->mnemonic = UD_Inone; + u->itab_entry = NULL; + u->have_modrm = 0; + u->br_far = 0; + u->vex_op = 0; + u->_rex = 0; + u->operand[0].type = UD_NONE; + u->operand[1].type = UD_NONE; + u->operand[2].type = UD_NONE; + u->operand[3].type = UD_NONE; +} + + +static UD_INLINE int +resolve_pfx_str(struct ud* u) +{ + if (u->pfx_str == 0xf3) { + if (P_STR(u->itab_entry->prefix)) { + u->pfx_rep = 0xf3; + } else { + u->pfx_repe = 0xf3; + } + } else if (u->pfx_str == 0xf2) { + u->pfx_repne = 0xf3; + } + return 0; +} + + +static int +resolve_mode( struct ud* u ) +{ + int default64; + /* if in error state, bail out */ + if ( u->error ) return -1; + + /* propagate prefix effects */ + if ( u->dis_mode == 64 ) { /* set 64bit-mode flags */ + + /* Check validity of instruction m64 */ + if ( P_INV64( u->itab_entry->prefix ) ) { + UDERR(u, "instruction invalid in 64bits\n"); + return -1; + } + + /* compute effective rex based on, + * - vex prefix (if any) + * - rex prefix (if any, and not vex) + * - allowed prefixes specified by the opcode map + */ + if (u->vex_op == 0xc4) { + /* vex has rex.rxb in 1's complement */ + u->_rex = ((~(u->vex_b1 >> 5) & 0x7) /* rex.0rxb */ | + ((u->vex_b2 >> 4) & 0x8) /* rex.w000 */); + } else if (u->vex_op == 0xc5) { + /* vex has rex.r in 1's complement */ + u->_rex = (~(u->vex_b1 >> 5)) & 4; + } else { + UD_ASSERT(u->vex_op == 0); + u->_rex = u->pfx_rex; + } + u->_rex &= REX_PFX_MASK(u->itab_entry->prefix); + + /* whether this instruction has a default operand size of + * 64bit, also hardcoded into the opcode map. + */ + default64 = P_DEF64( u->itab_entry->prefix ); + /* calculate effective operand size */ + if (REX_W(u->_rex)) { + u->opr_mode = 64; + } else if ( u->pfx_opr ) { + u->opr_mode = 16; + } else { + /* unless the default opr size of instruction is 64, + * the effective operand size in the absence of rex.w + * prefix is 32. + */ + u->opr_mode = default64 ? 64 : 32; + } + + /* calculate effective address size */ + u->adr_mode = (u->pfx_adr) ? 32 : 64; + } else if ( u->dis_mode == 32 ) { /* set 32bit-mode flags */ + u->opr_mode = ( u->pfx_opr ) ? 16 : 32; + u->adr_mode = ( u->pfx_adr ) ? 16 : 32; + } else if ( u->dis_mode == 16 ) { /* set 16bit-mode flags */ + u->opr_mode = ( u->pfx_opr ) ? 32 : 16; + u->adr_mode = ( u->pfx_adr ) ? 32 : 16; + } + + return 0; +} + + +static UD_INLINE int +decode_insn(struct ud *u, uint16_t ptr) +{ + UD_ASSERT((ptr & 0x8000) == 0); + u->itab_entry = &ud_itab[ ptr ]; + u->mnemonic = u->itab_entry->mnemonic; + return (resolve_pfx_str(u) == 0 && + resolve_mode(u) == 0 && + decode_operands(u) == 0 && + resolve_mnemonic(u) == 0) ? 0 : -1; +} + + +/* + * decode_3dnow() + * + * Decoding 3dnow is a little tricky because of its strange opcode + * structure. The final opcode disambiguation depends on the last + * byte that comes after the operands have been decoded. Fortunately, + * all 3dnow instructions have the same set of operand types. So we + * go ahead and decode the instruction by picking an arbitrarily chosen + * valid entry in the table, decode the operands, and read the final + * byte to resolve the menmonic. + */ +static UD_INLINE int +decode_3dnow(struct ud* u) +{ + uint16_t ptr; + UD_ASSERT(u->le->type == UD_TAB__OPC_3DNOW); + UD_ASSERT(u->le->table[0xc] != 0); + decode_insn(u, u->le->table[0xc]); + inp_next(u); + if (u->error) { + return -1; + } + ptr = u->le->table[inp_curr(u)]; + UD_ASSERT((ptr & 0x8000) == 0); + u->mnemonic = ud_itab[ptr].mnemonic; + return 0; +} + + +static int +decode_ssepfx(struct ud *u) +{ + uint8_t idx; + uint8_t pfx; + + /* + * String prefixes (f2, f3) take precedence over operand + * size prefix (66). + */ + pfx = u->pfx_str; + if (pfx == 0) { + pfx = u->pfx_opr; + } + idx = ((pfx & 0xf) + 1) / 2; + if (u->le->table[idx] == 0) { + idx = 0; + } + if (idx && u->le->table[idx] != 0) { + /* + * "Consume" the prefix as a part of the opcode, so it is no + * longer exported as an instruction prefix. + */ + u->pfx_str = 0; + if (pfx == 0x66) { + /* + * consume "66" only if it was used for decoding, leaving + * it to be used as an operands size override for some + * simd instructions. + */ + u->pfx_opr = 0; + } + } + return decode_ext(u, u->le->table[idx]); +} + + +static int +decode_vex(struct ud *u) +{ + uint8_t index; + if (u->dis_mode != 64 && MODRM_MOD(inp_peek(u)) != 0x3) { + index = 0; + } else { + u->vex_op = inp_curr(u); + u->vex_b1 = inp_next(u); + if (u->vex_op == 0xc4) { + uint8_t pp, m; + /* 3-byte vex */ + u->vex_b2 = inp_next(u); + UD_RETURN_ON_ERROR(u); + m = u->vex_b1 & 0x1f; + if (m == 0 || m > 3) { + UD_RETURN_WITH_ERROR(u, "reserved vex.m-mmmm value"); + } + pp = u->vex_b2 & 0x3; + index = (pp << 2) | m; + } else { + /* 2-byte vex */ + UD_ASSERT(u->vex_op == 0xc5); + index = 0x1 | ((u->vex_b1 & 0x3) << 2); + } + } + return decode_ext(u, u->le->table[index]); +} + + +/* + * decode_ext() + * + * Decode opcode extensions (if any) + */ +static int +decode_ext(struct ud *u, uint16_t ptr) +{ + uint8_t idx = 0; + if ((ptr & 0x8000) == 0) { + return decode_insn(u, ptr); + } + u->le = &ud_lookup_table_list[(~0x8000 & ptr)]; + if (u->le->type == UD_TAB__OPC_3DNOW) { + return decode_3dnow(u); + } + + switch (u->le->type) { + case UD_TAB__OPC_MOD: + /* !11 = 0, 11 = 1 */ + idx = (MODRM_MOD(modrm(u)) + 1) / 4; + break; + /* disassembly mode/operand size/address size based tables. + * 16 = 0,, 32 = 1, 64 = 2 + */ + case UD_TAB__OPC_MODE: + idx = u->dis_mode != 64 ? 0 : 1; + break; + case UD_TAB__OPC_OSIZE: + idx = eff_opr_mode(u->dis_mode, REX_W(u->pfx_rex), u->pfx_opr) / 32; + break; + case UD_TAB__OPC_ASIZE: + idx = eff_adr_mode(u->dis_mode, u->pfx_adr) / 32; + break; + case UD_TAB__OPC_X87: + idx = modrm(u) - 0xC0; + break; + case UD_TAB__OPC_VENDOR: + if (u->vendor == UD_VENDOR_ANY) { + /* choose a valid entry */ + idx = (u->le->table[idx] != 0) ? 0 : 1; + } else if (u->vendor == UD_VENDOR_AMD) { + idx = 0; + } else { + idx = 1; + } + break; + case UD_TAB__OPC_RM: + idx = MODRM_RM(modrm(u)); + break; + case UD_TAB__OPC_REG: + idx = MODRM_REG(modrm(u)); + break; + case UD_TAB__OPC_SSE: + return decode_ssepfx(u); + case UD_TAB__OPC_VEX: + return decode_vex(u); + case UD_TAB__OPC_VEX_W: + idx = vex_w(u); + break; + case UD_TAB__OPC_VEX_L: + idx = vex_l(u); + break; + case UD_TAB__OPC_TABLE: + inp_next(u); + return decode_opcode(u); + default: + UD_ASSERT(!"not reached"); + break; + } + + return decode_ext(u, u->le->table[idx]); +} + + +static int +decode_opcode(struct ud *u) +{ + uint16_t ptr; + UD_ASSERT(u->le->type == UD_TAB__OPC_TABLE); + UD_RETURN_ON_ERROR(u); + ptr = u->le->table[inp_curr(u)]; + return decode_ext(u, ptr); +} + + +/* ============================================================================= + * ud_decode() - Instruction decoder. Returns the number of bytes decoded. + * ============================================================================= + */ +unsigned int +ud_decode(struct ud *u) +{ + inp_start(u); + clear_insn(u); + u->le = &ud_lookup_table_list[0]; + u->error = decode_prefixes(u) == -1 || + decode_opcode(u) == -1 || + u->error; + /* Handle decode error. */ + if (u->error) { + /* clear out the decode data. */ + clear_insn(u); + /* mark the sequence of bytes as invalid. */ + u->itab_entry = &ud_itab[0]; /* entry 0 is invalid */ + u->mnemonic = u->itab_entry->mnemonic; + } + + /* maybe this stray segment override byte + * should be spewed out? + */ + if ( !P_SEG( u->itab_entry->prefix ) && + u->operand[0].type != UD_OP_MEM && + u->operand[1].type != UD_OP_MEM ) + u->pfx_seg = 0; + + u->insn_offset = u->pc; /* set offset of instruction */ + u->asm_buf_fill = 0; /* set translation buffer index to 0 */ + u->pc += u->inp_ctr; /* move program counter by bytes decoded */ + + /* return number of bytes disassembled. */ + return u->inp_ctr; +} + +/* +vim: set ts=2 sw=2 expandtab +*/ diff --git a/ext/udis86/decode.h b/ext/udis86/decode.h new file mode 100644 index 0000000000..3949c4e269 --- /dev/null +++ b/ext/udis86/decode.h @@ -0,0 +1,197 @@ +/* udis86 - libudis86/decode.h + * + * Copyright (c) 2002-2009 Vivek Thampi + * All rights reserved. + * + * Redistribution and use in source and binary forms, with or without modification, + * are permitted provided that the following conditions are met: + * + * * Redistributions of source code must retain the above copyright notice, + * this list of conditions and the following disclaimer. + * * Redistributions in binary form must reproduce the above copyright notice, + * this list of conditions and the following disclaimer in the documentation + * and/or other materials provided with the distribution. + * + * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND + * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED + * WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE + * DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE FOR + * ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES + * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; + * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON + * ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT + * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + */ +#ifndef UD_DECODE_H +#define UD_DECODE_H + +#include "types.h" +#include "udint.h" +#include "itab.h" + +#define MAX_INSN_LENGTH 15 + +/* itab prefix bits */ +#define P_none ( 0 ) + +#define P_inv64 ( 1 << 0 ) +#define P_INV64(n) ( ( n >> 0 ) & 1 ) +#define P_def64 ( 1 << 1 ) +#define P_DEF64(n) ( ( n >> 1 ) & 1 ) + +#define P_oso ( 1 << 2 ) +#define P_OSO(n) ( ( n >> 2 ) & 1 ) +#define P_aso ( 1 << 3 ) +#define P_ASO(n) ( ( n >> 3 ) & 1 ) + +#define P_rexb ( 1 << 4 ) +#define P_REXB(n) ( ( n >> 4 ) & 1 ) +#define P_rexw ( 1 << 5 ) +#define P_REXW(n) ( ( n >> 5 ) & 1 ) +#define P_rexr ( 1 << 6 ) +#define P_REXR(n) ( ( n >> 6 ) & 1 ) +#define P_rexx ( 1 << 7 ) +#define P_REXX(n) ( ( n >> 7 ) & 1 ) + +#define P_seg ( 1 << 8 ) +#define P_SEG(n) ( ( n >> 8 ) & 1 ) + +#define P_vexl ( 1 << 9 ) +#define P_VEXL(n) ( ( n >> 9 ) & 1 ) +#define P_vexw ( 1 << 10 ) +#define P_VEXW(n) ( ( n >> 10 ) & 1 ) + +#define P_str ( 1 << 11 ) +#define P_STR(n) ( ( n >> 11 ) & 1 ) +#define P_strz ( 1 << 12 ) +#define P_STR_ZF(n) ( ( n >> 12 ) & 1 ) + +/* operand type constants -- order is important! */ + +enum ud_operand_code { + OP_NONE, + + OP_A, OP_E, OP_M, OP_G, + OP_I, OP_F, + + OP_R0, OP_R1, OP_R2, OP_R3, + OP_R4, OP_R5, OP_R6, OP_R7, + + OP_AL, OP_CL, OP_DL, + OP_AX, OP_CX, OP_DX, + OP_eAX, OP_eCX, OP_eDX, + OP_rAX, OP_rCX, OP_rDX, + + OP_ES, OP_CS, OP_SS, OP_DS, + OP_FS, OP_GS, + + OP_ST0, OP_ST1, OP_ST2, OP_ST3, + OP_ST4, OP_ST5, OP_ST6, OP_ST7, + + OP_J, OP_S, OP_O, + OP_I1, OP_I3, OP_sI, + + OP_V, OP_W, OP_Q, OP_P, + OP_U, OP_N, OP_MU, OP_H, + OP_L, + + OP_R, OP_C, OP_D, + + OP_MR +} UD_ATTR_PACKED; + + +/* + * Operand size constants + * + * Symbolic constants for various operand sizes. Some of these constants + * are given a value equal to the width of the data (SZ_B == 8), such + * that they maybe used interchangeably in the internals. Modifying them + * will most certainly break things! + */ +typedef uint16_t ud_operand_size_t; + +#define SZ_NA 0 +#define SZ_Z 1 +#define SZ_V 2 +#define SZ_Y 3 +#define SZ_X 4 +#define SZ_RDQ 7 +#define SZ_B 8 +#define SZ_W 16 +#define SZ_D 32 +#define SZ_Q 64 +#define SZ_T 80 +#define SZ_O 12 +#define SZ_DQ 128 /* double quad */ +#define SZ_QQ 256 /* quad quad */ + +/* + * Complex size types; that encode sizes for operands of type MR (memory or + * register); for internal use only. Id space above 256. + */ +#define SZ_BD ((SZ_B << 8) | SZ_D) +#define SZ_BV ((SZ_B << 8) | SZ_V) +#define SZ_WD ((SZ_W << 8) | SZ_D) +#define SZ_WV ((SZ_W << 8) | SZ_V) +#define SZ_WY ((SZ_W << 8) | SZ_Y) +#define SZ_DY ((SZ_D << 8) | SZ_Y) +#define SZ_WO ((SZ_W << 8) | SZ_O) +#define SZ_DO ((SZ_D << 8) | SZ_O) +#define SZ_QO ((SZ_Q << 8) | SZ_O) + + +/* resolve complex size type. + */ +static UD_INLINE ud_operand_size_t +Mx_mem_size(ud_operand_size_t size) +{ + return (size >> 8) & 0xff; +} + +static UD_INLINE ud_operand_size_t +Mx_reg_size(ud_operand_size_t size) +{ + return size & 0xff; +} + +/* A single operand of an entry in the instruction table. + * (internal use only) + */ +struct ud_itab_entry_operand +{ + enum ud_operand_code type; + ud_operand_size_t size; +}; + + +/* A single entry in an instruction table. + *(internal use only) + */ +struct ud_itab_entry +{ + enum ud_mnemonic_code mnemonic; + struct ud_itab_entry_operand operand1; + struct ud_itab_entry_operand operand2; + struct ud_itab_entry_operand operand3; + struct ud_itab_entry_operand operand4; + uint32_t prefix; +}; + +struct ud_lookup_table_list_entry { + const uint16_t *table; + enum ud_table_type type; + const char *meta; +}; + +extern struct ud_itab_entry ud_itab[]; +extern struct ud_lookup_table_list_entry ud_lookup_table_list[]; + +#endif /* UD_DECODE_H */ + +/* vim:cindent + * vim:expandtab + * vim:ts=4 + * vim:sw=4 + */ diff --git a/ext/udis86/extern.h b/ext/udis86/extern.h new file mode 100644 index 0000000000..71a01fd9b4 --- /dev/null +++ b/ext/udis86/extern.h @@ -0,0 +1,113 @@ +/* udis86 - libudis86/extern.h + * + * Copyright (c) 2002-2009, 2013 Vivek Thampi + * All rights reserved. + * + * Redistribution and use in source and binary forms, with or without modification, + * are permitted provided that the following conditions are met: + * + * * Redistributions of source code must retain the above copyright notice, + * this list of conditions and the following disclaimer. + * * Redistributions in binary form must reproduce the above copyright notice, + * this list of conditions and the following disclaimer in the documentation + * and/or other materials provided with the distribution. + * + * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND + * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED + * WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE + * DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE FOR + * ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES + * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; + * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON + * ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT + * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + */ +#ifndef UD_EXTERN_H +#define UD_EXTERN_H + +#ifdef __cplusplus +extern "C" { +#endif + +#include "types.h" + +#if defined(_MSC_VER) && defined(_USRDLL) +# ifdef LIBUDIS86_EXPORTS +# define LIBUDIS86_DLLEXTERN __declspec(dllexport) +# else +# define LIBUDIS86_DLLEXTERN __declspec(dllimport) +# endif +#else +# define LIBUDIS86_DLLEXTERN +#endif + +/* ============================= PUBLIC API ================================= */ + +extern LIBUDIS86_DLLEXTERN void ud_init(struct ud*); + +extern LIBUDIS86_DLLEXTERN void ud_set_mode(struct ud*, uint8_t); + +extern LIBUDIS86_DLLEXTERN void ud_set_pc(struct ud*, uint64_t); + +extern LIBUDIS86_DLLEXTERN void ud_set_input_hook(struct ud*, int (*)(struct ud*)); + +extern LIBUDIS86_DLLEXTERN void ud_set_input_buffer(struct ud*, const uint8_t*, size_t); + +#ifndef __UD_STANDALONE__ +extern LIBUDIS86_DLLEXTERN void ud_set_input_file(struct ud*, FILE*); +#endif /* __UD_STANDALONE__ */ + +extern LIBUDIS86_DLLEXTERN void ud_set_vendor(struct ud*, unsigned); + +extern LIBUDIS86_DLLEXTERN void ud_set_syntax(struct ud*, void (*)(struct ud*)); + +extern LIBUDIS86_DLLEXTERN void ud_input_skip(struct ud*, size_t); + +extern LIBUDIS86_DLLEXTERN int ud_input_end(const struct ud*); + +extern LIBUDIS86_DLLEXTERN unsigned int ud_decode(struct ud*); + +extern LIBUDIS86_DLLEXTERN unsigned int ud_disassemble(struct ud*); + +extern LIBUDIS86_DLLEXTERN void ud_translate_intel(struct ud*); + +extern LIBUDIS86_DLLEXTERN void ud_translate_att(struct ud*); + +extern LIBUDIS86_DLLEXTERN const char* ud_insn_asm(const struct ud* u); + +extern LIBUDIS86_DLLEXTERN const uint8_t* ud_insn_ptr(const struct ud* u); + +extern LIBUDIS86_DLLEXTERN uint64_t ud_insn_off(const struct ud*); + +extern LIBUDIS86_DLLEXTERN const char* ud_insn_hex(struct ud*); + +extern LIBUDIS86_DLLEXTERN unsigned int ud_insn_len(const struct ud* u); + +extern LIBUDIS86_DLLEXTERN const struct ud_operand* ud_insn_opr(const struct ud *u, unsigned int n); + +extern LIBUDIS86_DLLEXTERN int ud_opr_is_sreg(const struct ud_operand *opr); + +extern LIBUDIS86_DLLEXTERN int ud_opr_is_gpr(const struct ud_operand *opr); + +extern LIBUDIS86_DLLEXTERN enum ud_mnemonic_code ud_insn_mnemonic(const struct ud *u); + +extern LIBUDIS86_DLLEXTERN const char* ud_lookup_mnemonic(enum ud_mnemonic_code c); + +extern LIBUDIS86_DLLEXTERN void ud_set_user_opaque_data(struct ud*, void*); + +extern LIBUDIS86_DLLEXTERN void* ud_get_user_opaque_data(const struct ud*); + +extern LIBUDIS86_DLLEXTERN void ud_set_asm_buffer(struct ud *u, char *buf, size_t size); + +extern LIBUDIS86_DLLEXTERN void ud_set_sym_resolver(struct ud *u, + const char* (*resolver)(struct ud*, + uint64_t addr, + int64_t *offset)); + +/* ========================================================================== */ + +#ifdef __cplusplus +} +#endif +#endif /* UD_EXTERN_H */ diff --git a/ext/udis86/itab.c b/ext/udis86/itab.c new file mode 100644 index 0000000000..7ea0569ebf --- /dev/null +++ b/ext/udis86/itab.c @@ -0,0 +1,5937 @@ +/* itab.c -- generated by udis86:scripts/ud_itab.py, do no edit */ +#include "decode.h" + +#define GROUP(n) (0x8000 | (n)) +#define INVALID 0 + + +const uint16_t ud_itab__0[] = { + /* 0 */ 15, 16, 17, 18, + /* 4 */ 19, 20, GROUP(1), GROUP(2), + /* 8 */ 960, 961, 962, 963, + /* c */ 964, 965, GROUP(3), GROUP(4), + /* 10 */ 5, 6, 7, 8, + /* 14 */ 9, 10, GROUP(284), GROUP(285), + /* 18 */ 1332, 1333, 1334, 1335, + /* 1c */ 1336, 1337, GROUP(286), GROUP(287), + /* 20 */ 49, 50, 51, 52, + /* 24 */ 53, 54, INVALID, GROUP(288), + /* 28 */ 1403, 1404, 1405, 1406, + /* 2c */ 1407, 1408, INVALID, GROUP(289), + /* 30 */ 1483, 1484, 1485, 1486, + /* 34 */ 1487, 1488, INVALID, GROUP(290), + /* 38 */ 100, 101, 102, 103, + /* 3c */ 104, 105, INVALID, GROUP(291), + /* 40 */ 695, 696, 697, 698, + /* 44 */ 699, 700, 701, 702, + /* 48 */ 175, 176, 177, 178, + /* 4c */ 179, 180, 181, 182, + /* 50 */ 1242, 1243, 1244, 1245, + /* 54 */ 1246, 1247, 1248, 1249, + /* 58 */ 1097, 1098, 1099, 1100, + /* 5c */ 1101, 1102, 1103, 1104, + /* 60 */ GROUP(292), GROUP(295), GROUP(298), GROUP(299), + /* 64 */ INVALID, INVALID, INVALID, INVALID, + /* 68 */ 1250, 693, 1252, 694, + /* 6c */ 705, GROUP(300), 978, GROUP(301), + /* 70 */ 722, 724, 726, 728, + /* 74 */ 730, 732, 734, 736, + /* 78 */ 738, 740, 742, 744, + /* 7c */ 746, 748, 750, 752, + /* 80 */ GROUP(302), GROUP(303), GROUP(304), GROUP(313), + /* 84 */ 1429, 1430, 1471, 1472, + /* 88 */ 824, 825, 826, 827, + /* 8c */ 828, 766, 829, GROUP(314), + /* 90 */ 1473, 1474, 1475, 1476, + /* 94 */ 1477, 1478, 1479, 1480, + /* 98 */ GROUP(315), GROUP(316), GROUP(317), 1466, + /* 9c */ GROUP(318), GROUP(322), 1306, 762, + /* a0 */ 830, 831, 832, 833, + /* a4 */ 918, GROUP(326), 114, GROUP(327), + /* a8 */ 1431, 1432, 1398, GROUP(328), + /* ac */ 786, GROUP(329), 1342, GROUP(330), + /* b0 */ 834, 835, 836, 837, + /* b4 */ 838, 839, 840, 841, + /* b8 */ 842, 843, 844, 845, + /* bc */ 846, 847, 848, 849, + /* c0 */ GROUP(331), GROUP(332), 1297, 1298, + /* c4 */ GROUP(333), GROUP(403), GROUP(405), GROUP(406), + /* c8 */ 200, 772, 1299, 1300, + /* cc */ 709, 710, GROUP(407), GROUP(408), + /* d0 */ GROUP(409), GROUP(410), GROUP(411), GROUP(412), + /* d4 */ GROUP(413), GROUP(414), GROUP(415), 1482, + /* d8 */ GROUP(416), GROUP(419), GROUP(422), GROUP(425), + /* dc */ GROUP(428), GROUP(431), GROUP(434), GROUP(437), + /* e0 */ 790, 791, 792, GROUP(440), + /* e4 */ 686, 687, 974, 975, + /* e8 */ 72, 759, GROUP(441), 761, + /* ec */ 688, 689, 976, 977, + /* f0 */ 785, 708, 1295, 1296, + /* f4 */ 683, 83, GROUP(442), GROUP(443), + /* f8 */ 77, 1391, 81, 1394, + /* fc */ 78, 1392, GROUP(444), GROUP(445), +}; + +static const uint16_t ud_itab__1[] = { + /* 0 */ 1236, INVALID, +}; + +static const uint16_t ud_itab__2[] = { + /* 0 */ 1092, INVALID, +}; + +static const uint16_t ud_itab__3[] = { + /* 0 */ 1237, INVALID, +}; + +static const uint16_t ud_itab__4[] = { + /* 0 */ GROUP(5), GROUP(6), 763, 793, + /* 4 */ INVALID, 1422, 82, 1427, + /* 8 */ 712, 1467, INVALID, 1440, + /* c */ INVALID, GROUP(27), 430, GROUP(28), + /* 10 */ GROUP(29), GROUP(30), GROUP(31), GROUP(34), + /* 14 */ GROUP(35), GROUP(36), GROUP(37), GROUP(40), + /* 18 */ GROUP(41), 951, 952, 953, + /* 1c */ 954, 955, 956, 957, + /* 20 */ 850, 851, 852, 853, + /* 24 */ INVALID, INVALID, INVALID, INVALID, + /* 28 */ GROUP(42), GROUP(43), GROUP(44), GROUP(45), + /* 2c */ GROUP(46), GROUP(47), GROUP(48), GROUP(49), + /* 30 */ 1468, 1293, 1291, 1292, + /* 34 */ GROUP(50), GROUP(52), INVALID, 1510, + /* 38 */ GROUP(54), INVALID, GROUP(116), INVALID, + /* 3c */ INVALID, INVALID, INVALID, INVALID, + /* 40 */ 84, 85, 86, 87, + /* 44 */ 88, 89, 90, 91, + /* 48 */ 92, 93, 94, 95, + /* 4c */ 96, 97, 98, 99, + /* 50 */ GROUP(143), GROUP(144), GROUP(145), GROUP(146), + /* 54 */ GROUP(147), GROUP(148), GROUP(149), GROUP(150), + /* 58 */ GROUP(151), GROUP(152), GROUP(153), GROUP(154), + /* 5c */ GROUP(155), GROUP(156), GROUP(157), GROUP(158), + /* 60 */ GROUP(159), GROUP(160), GROUP(161), GROUP(162), + /* 64 */ GROUP(163), GROUP(164), GROUP(165), GROUP(166), + /* 68 */ GROUP(167), GROUP(168), GROUP(169), GROUP(170), + /* 6c */ GROUP(171), GROUP(172), GROUP(173), GROUP(176), + /* 70 */ GROUP(177), GROUP(178), GROUP(182), GROUP(186), + /* 74 */ GROUP(191), GROUP(192), GROUP(193), 199, + /* 78 */ GROUP(194), GROUP(195), INVALID, INVALID, + /* 7c */ GROUP(196), GROUP(197), GROUP(198), GROUP(201), + /* 80 */ 723, 725, 727, 729, + /* 84 */ 731, 733, 735, 737, + /* 88 */ 739, 741, 743, 745, + /* 8c */ 747, 749, 751, 753, + /* 90 */ 1346, 1347, 1348, 1349, + /* 94 */ 1350, 1351, 1352, 1353, + /* 98 */ 1354, 1355, 1356, 1357, + /* 9c */ 1358, 1359, 1360, 1361, + /* a0 */ 1241, 1096, 131, 1665, + /* a4 */ 1371, 1372, GROUP(202), GROUP(207), + /* a8 */ 1240, 1095, 1301, 1670, + /* ac */ 1373, 1374, GROUP(215), 690, + /* b0 */ 122, 123, 771, 1668, + /* b4 */ 768, 769, 936, 937, + /* b8 */ GROUP(221), INVALID, GROUP(222), 1666, + /* bc */ 1654, 1655, 926, 927, + /* c0 */ 1469, 1470, GROUP(223), 900, + /* c4 */ GROUP(224), GROUP(225), GROUP(226), GROUP(227), + /* c8 */ 1656, 1657, 1658, 1659, + /* cc */ 1660, 1661, 1662, 1663, + /* d0 */ GROUP(236), GROUP(237), GROUP(238), GROUP(239), + /* d4 */ GROUP(240), GROUP(241), GROUP(242), GROUP(243), + /* d8 */ GROUP(244), GROUP(245), GROUP(246), GROUP(247), + /* dc */ GROUP(248), GROUP(249), GROUP(250), GROUP(251), + /* e0 */ GROUP(252), GROUP(253), GROUP(254), GROUP(255), + /* e4 */ GROUP(256), GROUP(257), GROUP(258), GROUP(259), + /* e8 */ GROUP(260), GROUP(261), GROUP(262), GROUP(263), + /* ec */ GROUP(264), GROUP(265), GROUP(266), GROUP(267), + /* f0 */ GROUP(268), GROUP(269), GROUP(270), GROUP(271), + /* f4 */ GROUP(272), GROUP(273), GROUP(274), GROUP(275), + /* f8 */ GROUP(277), GROUP(278), GROUP(279), GROUP(280), + /* fc */ GROUP(281), GROUP(282), GROUP(283), INVALID, +}; + +static const uint16_t ud_itab__5[] = { + /* 0 */ 1380, 1402, 782, 794, + /* 4 */ 1449, 1450, INVALID, INVALID, +}; + +static const uint16_t ud_itab__6[] = { + /* 0 */ GROUP(7), GROUP(8), +}; + +static const uint16_t ud_itab__7[] = { + /* 0 */ 1370, 1379, 781, 770, + /* 4 */ 1381, INVALID, 783, 715, +}; + +static const uint16_t ud_itab__8[] = { + /* 0 */ GROUP(9), GROUP(14), GROUP(15), GROUP(16), + /* 4 */ 1382, INVALID, 784, GROUP(25), +}; + +static const uint16_t ud_itab__9[] = { + /* 0 */ INVALID, GROUP(10), GROUP(11), GROUP(12), + /* 4 */ GROUP(13), INVALID, INVALID, INVALID, +}; + +static const uint16_t ud_itab__10[] = { + /* 0 */ INVALID, 1451, INVALID, +}; + +static const uint16_t ud_itab__11[] = { + /* 0 */ INVALID, 1457, INVALID, +}; + +static const uint16_t ud_itab__12[] = { + /* 0 */ INVALID, 1458, INVALID, +}; + +static const uint16_t ud_itab__13[] = { + /* 0 */ INVALID, 1459, INVALID, +}; + +static const uint16_t ud_itab__14[] = { + /* 0 */ 820, 948, INVALID, INVALID, + /* 4 */ INVALID, INVALID, INVALID, INVALID, +}; + +static const uint16_t ud_itab__15[] = { + /* 0 */ 1481, 1504, INVALID, INVALID, + /* 4 */ INVALID, INVALID, INVALID, INVALID, +}; + +static const uint16_t ud_itab__16[] = { + /* 0 */ GROUP(17), GROUP(18), GROUP(19), GROUP(20), + /* 4 */ GROUP(21), GROUP(22), GROUP(23), GROUP(24), +}; + +static const uint16_t ud_itab__17[] = { + /* 0 */ 1462, INVALID, INVALID, +}; + +static const uint16_t ud_itab__18[] = { + /* 0 */ 1463, INVALID, INVALID, +}; + +static const uint16_t ud_itab__19[] = { + /* 0 */ 1464, INVALID, INVALID, +}; + +static const uint16_t ud_itab__20[] = { + /* 0 */ 1465, INVALID, INVALID, +}; + +static const uint16_t ud_itab__21[] = { + /* 0 */ 1393, INVALID, INVALID, +}; + +static const uint16_t ud_itab__22[] = { + /* 0 */ 80, INVALID, INVALID, +}; + +static const uint16_t ud_itab__23[] = { + /* 0 */ 1395, INVALID, INVALID, +}; + +static const uint16_t ud_itab__24[] = { + /* 0 */ 716, INVALID, INVALID, +}; + +static const uint16_t ud_itab__25[] = { + /* 0 */ 1421, GROUP(26), INVALID, INVALID, + /* 4 */ INVALID, INVALID, INVALID, INVALID, +}; + +static const uint16_t ud_itab__26[] = { + /* 0 */ 1294, INVALID, INVALID, +}; + +static const uint16_t ud_itab__27[] = { + /* 0 */ 1115, 1116, 1117, 1118, + /* 4 */ 1119, 1120, 1121, 1122, +}; + +static const uint16_t ud_itab__28[] = { + /* 0 */ INVALID, INVALID, INVALID, INVALID, + /* 4 */ INVALID, INVALID, INVALID, INVALID, + /* 8 */ INVALID, INVALID, INVALID, INVALID, + /* c */ 1212, 1213, INVALID, INVALID, + /* 10 */ INVALID, INVALID, INVALID, INVALID, + /* 14 */ INVALID, INVALID, INVALID, INVALID, + /* 18 */ INVALID, INVALID, INVALID, INVALID, + /* 1c */ 1214, 1215, INVALID, INVALID, + /* 20 */ INVALID, INVALID, INVALID, INVALID, + /* 24 */ INVALID, INVALID, INVALID, INVALID, + /* 28 */ INVALID, INVALID, INVALID, INVALID, + /* 2c */ INVALID, INVALID, INVALID, INVALID, + /* 30 */ INVALID, INVALID, INVALID, INVALID, + /* 34 */ INVALID, INVALID, INVALID, INVALID, + /* 38 */ INVALID, INVALID, INVALID, INVALID, + /* 3c */ INVALID, INVALID, INVALID, INVALID, + /* 40 */ INVALID, INVALID, INVALID, INVALID, + /* 44 */ INVALID, INVALID, INVALID, INVALID, + /* 48 */ INVALID, INVALID, INVALID, INVALID, + /* 4c */ INVALID, INVALID, INVALID, INVALID, + /* 50 */ INVALID, INVALID, INVALID, INVALID, + /* 54 */ INVALID, INVALID, INVALID, INVALID, + /* 58 */ INVALID, INVALID, INVALID, INVALID, + /* 5c */ INVALID, INVALID, INVALID, INVALID, + /* 60 */ INVALID, INVALID, INVALID, INVALID, + /* 64 */ INVALID, INVALID, INVALID, INVALID, + /* 68 */ INVALID, INVALID, INVALID, INVALID, + /* 6c */ INVALID, INVALID, INVALID, INVALID, + /* 70 */ INVALID, INVALID, INVALID, INVALID, + /* 74 */ INVALID, INVALID, INVALID, INVALID, + /* 78 */ INVALID, INVALID, INVALID, INVALID, + /* 7c */ INVALID, INVALID, INVALID, INVALID, + /* 80 */ INVALID, INVALID, INVALID, INVALID, + /* 84 */ INVALID, INVALID, INVALID, INVALID, + /* 88 */ INVALID, INVALID, 1216, INVALID, + /* 8c */ INVALID, INVALID, 1217, INVALID, + /* 90 */ 1218, INVALID, INVALID, INVALID, + /* 94 */ 1219, INVALID, 1220, 1221, + /* 98 */ INVALID, INVALID, 1222, INVALID, + /* 9c */ INVALID, INVALID, 1223, INVALID, + /* a0 */ 1224, INVALID, INVALID, INVALID, + /* a4 */ 1225, INVALID, 1226, 1227, + /* a8 */ INVALID, INVALID, 1228, INVALID, + /* ac */ INVALID, INVALID, 1229, INVALID, + /* b0 */ 1230, INVALID, INVALID, INVALID, + /* b4 */ 1231, INVALID, 1232, 1233, + /* b8 */ INVALID, INVALID, INVALID, 1234, + /* bc */ INVALID, INVALID, INVALID, 1235, + /* c0 */ INVALID, INVALID, INVALID, INVALID, + /* c4 */ INVALID, INVALID, INVALID, INVALID, + /* c8 */ INVALID, INVALID, INVALID, INVALID, + /* cc */ INVALID, INVALID, INVALID, INVALID, + /* d0 */ INVALID, INVALID, INVALID, INVALID, + /* d4 */ INVALID, INVALID, INVALID, INVALID, + /* d8 */ INVALID, INVALID, INVALID, INVALID, + /* dc */ INVALID, INVALID, INVALID, INVALID, + /* e0 */ INVALID, INVALID, INVALID, INVALID, + /* e4 */ INVALID, INVALID, INVALID, INVALID, + /* e8 */ INVALID, INVALID, INVALID, INVALID, + /* ec */ INVALID, INVALID, INVALID, INVALID, + /* f0 */ INVALID, INVALID, INVALID, INVALID, + /* f4 */ INVALID, INVALID, INVALID, INVALID, + /* f8 */ INVALID, INVALID, INVALID, INVALID, + /* fc */ INVALID, INVALID, INVALID, INVALID, +}; + +static const uint16_t ud_itab__29[] = { + /* 0 */ 932, 921, 924, 928, +}; + +static const uint16_t ud_itab__30[] = { + /* 0 */ 934, 922, 925, 930, +}; + +static const uint16_t ud_itab__31[] = { + /* 0 */ GROUP(32), GROUP(33), +}; + +static const uint16_t ud_itab__32[] = { + /* 0 */ 888, 1558, 1566, 884, +}; + +static const uint16_t ud_itab__33[] = { + /* 0 */ 892, 1556, 1564, INVALID, +}; + +static const uint16_t ud_itab__34[] = { + /* 0 */ 890, INVALID, INVALID, 886, +}; + +static const uint16_t ud_itab__35[] = { + /* 0 */ 1445, INVALID, INVALID, 1447, +}; + +static const uint16_t ud_itab__36[] = { + /* 0 */ 1443, INVALID, INVALID, 1441, +}; + +static const uint16_t ud_itab__37[] = { + /* 0 */ GROUP(38), GROUP(39), +}; + +static const uint16_t ud_itab__38[] = { + /* 0 */ 878, INVALID, 1562, 874, +}; + +static const uint16_t ud_itab__39[] = { + /* 0 */ 882, INVALID, 1560, INVALID, +}; + +static const uint16_t ud_itab__40[] = { + /* 0 */ 880, INVALID, INVALID, 876, +}; + +static const uint16_t ud_itab__41[] = { + /* 0 */ 1123, 1124, 1125, 1126, + /* 4 */ INVALID, INVALID, INVALID, INVALID, +}; + +static const uint16_t ud_itab__42[] = { + /* 0 */ 858, INVALID, INVALID, 854, +}; + +static const uint16_t ud_itab__43[] = { + /* 0 */ 860, INVALID, INVALID, 856, +}; + +static const uint16_t ud_itab__44[] = { + /* 0 */ 141, 152, 154, 142, +}; + +static const uint16_t ud_itab__45[] = { + /* 0 */ 903, INVALID, INVALID, 901, +}; + +static const uint16_t ud_itab__46[] = { + /* 0 */ 165, 166, 168, 162, +}; + +static const uint16_t ud_itab__47[] = { + /* 0 */ 147, 148, 158, 138, +}; + +static const uint16_t ud_itab__48[] = { + /* 0 */ 1438, INVALID, INVALID, 1436, +}; + +static const uint16_t ud_itab__49[] = { + /* 0 */ 129, INVALID, INVALID, 127, +}; + +static const uint16_t ud_itab__50[] = { + /* 0 */ 1423, GROUP(51), +}; + +static const uint16_t ud_itab__51[] = { + /* 0 */ INVALID, 1424, INVALID, +}; + +static const uint16_t ud_itab__52[] = { + /* 0 */ 1425, GROUP(53), +}; + +static const uint16_t ud_itab__53[] = { + /* 0 */ INVALID, 1426, INVALID, +}; + +static const uint16_t ud_itab__54[] = { + /* 0 */ GROUP(67), GROUP(68), GROUP(63), GROUP(64), + /* 4 */ GROUP(65), GROUP(66), GROUP(86), GROUP(90), + /* 8 */ GROUP(69), GROUP(70), GROUP(71), GROUP(72), + /* c */ INVALID, INVALID, INVALID, INVALID, + /* 10 */ GROUP(73), INVALID, INVALID, INVALID, + /* 14 */ GROUP(75), GROUP(76), INVALID, GROUP(77), + /* 18 */ INVALID, INVALID, INVALID, INVALID, + /* 1c */ GROUP(78), GROUP(79), GROUP(80), INVALID, + /* 20 */ GROUP(81), GROUP(82), GROUP(83), GROUP(84), + /* 24 */ GROUP(85), GROUP(108), INVALID, INVALID, + /* 28 */ GROUP(87), GROUP(88), GROUP(89), GROUP(74), + /* 2c */ INVALID, INVALID, INVALID, INVALID, + /* 30 */ GROUP(91), GROUP(92), GROUP(93), GROUP(94), + /* 34 */ GROUP(95), GROUP(96), INVALID, GROUP(97), + /* 38 */ GROUP(98), GROUP(99), GROUP(100), GROUP(101), + /* 3c */ GROUP(102), GROUP(103), GROUP(104), GROUP(105), + /* 40 */ GROUP(106), GROUP(107), INVALID, INVALID, + /* 44 */ INVALID, INVALID, INVALID, INVALID, + /* 48 */ INVALID, INVALID, INVALID, INVALID, + /* 4c */ INVALID, INVALID, INVALID, INVALID, + /* 50 */ INVALID, INVALID, INVALID, INVALID, + /* 54 */ INVALID, INVALID, INVALID, INVALID, + /* 58 */ INVALID, INVALID, INVALID, INVALID, + /* 5c */ INVALID, INVALID, INVALID, INVALID, + /* 60 */ INVALID, INVALID, INVALID, INVALID, + /* 64 */ INVALID, INVALID, INVALID, INVALID, + /* 68 */ INVALID, INVALID, INVALID, INVALID, + /* 6c */ INVALID, INVALID, INVALID, INVALID, + /* 70 */ INVALID, INVALID, INVALID, INVALID, + /* 74 */ INVALID, INVALID, INVALID, INVALID, + /* 78 */ INVALID, INVALID, INVALID, INVALID, + /* 7c */ INVALID, INVALID, INVALID, INVALID, + /* 80 */ GROUP(55), GROUP(59), INVALID, INVALID, + /* 84 */ INVALID, INVALID, INVALID, INVALID, + /* 88 */ INVALID, INVALID, INVALID, INVALID, + /* 8c */ INVALID, INVALID, INVALID, INVALID, + /* 90 */ INVALID, INVALID, INVALID, INVALID, + /* 94 */ INVALID, INVALID, INVALID, INVALID, + /* 98 */ INVALID, INVALID, INVALID, INVALID, + /* 9c */ INVALID, INVALID, INVALID, INVALID, + /* a0 */ INVALID, INVALID, INVALID, INVALID, + /* a4 */ INVALID, INVALID, INVALID, INVALID, + /* a8 */ INVALID, INVALID, INVALID, INVALID, + /* ac */ INVALID, INVALID, INVALID, INVALID, + /* b0 */ INVALID, INVALID, INVALID, INVALID, + /* b4 */ INVALID, INVALID, INVALID, INVALID, + /* b8 */ INVALID, INVALID, INVALID, INVALID, + /* bc */ INVALID, INVALID, INVALID, INVALID, + /* c0 */ INVALID, INVALID, INVALID, INVALID, + /* c4 */ INVALID, INVALID, INVALID, INVALID, + /* c8 */ INVALID, INVALID, INVALID, INVALID, + /* cc */ INVALID, INVALID, INVALID, INVALID, + /* d0 */ INVALID, INVALID, INVALID, INVALID, + /* d4 */ INVALID, INVALID, INVALID, INVALID, + /* d8 */ INVALID, INVALID, INVALID, GROUP(109), + /* dc */ GROUP(110), GROUP(111), GROUP(112), GROUP(113), + /* e0 */ INVALID, INVALID, INVALID, INVALID, + /* e4 */ INVALID, INVALID, INVALID, INVALID, + /* e8 */ INVALID, INVALID, INVALID, INVALID, + /* ec */ INVALID, INVALID, INVALID, INVALID, + /* f0 */ GROUP(114), GROUP(115), INVALID, INVALID, + /* f4 */ INVALID, INVALID, INVALID, INVALID, + /* f8 */ INVALID, INVALID, INVALID, INVALID, + /* fc */ INVALID, INVALID, INVALID, INVALID, +}; + +static const uint16_t ud_itab__55[] = { + /* 0 */ INVALID, INVALID, INVALID, GROUP(56), +}; + +static const uint16_t ud_itab__56[] = { + /* 0 */ GROUP(57), GROUP(58), +}; + +static const uint16_t ud_itab__57[] = { + /* 0 */ INVALID, 713, INVALID, +}; + +static const uint16_t ud_itab__58[] = { + /* 0 */ INVALID, 714, INVALID, +}; + +static const uint16_t ud_itab__59[] = { + /* 0 */ INVALID, INVALID, INVALID, GROUP(60), +}; + +static const uint16_t ud_itab__60[] = { + /* 0 */ GROUP(61), GROUP(62), +}; + +static const uint16_t ud_itab__61[] = { + /* 0 */ INVALID, 717, INVALID, +}; + +static const uint16_t ud_itab__62[] = { + /* 0 */ INVALID, 718, INVALID, +}; + +static const uint16_t ud_itab__63[] = { + /* 0 */ 1583, INVALID, INVALID, 1584, +}; + +static const uint16_t ud_itab__64[] = { + /* 0 */ 1586, INVALID, INVALID, 1587, +}; + +static const uint16_t ud_itab__65[] = { + /* 0 */ 1589, INVALID, INVALID, 1590, +}; + +static const uint16_t ud_itab__66[] = { + /* 0 */ 1592, INVALID, INVALID, 1593, +}; + +static const uint16_t ud_itab__67[] = { + /* 0 */ 1577, INVALID, INVALID, 1578, +}; + +static const uint16_t ud_itab__68[] = { + /* 0 */ 1580, INVALID, INVALID, 1581, +}; + +static const uint16_t ud_itab__69[] = { + /* 0 */ 1601, INVALID, INVALID, 1602, +}; + +static const uint16_t ud_itab__70[] = { + /* 0 */ 1607, INVALID, INVALID, 1608, +}; + +static const uint16_t ud_itab__71[] = { + /* 0 */ 1604, INVALID, INVALID, 1605, +}; + +static const uint16_t ud_itab__72[] = { + /* 0 */ 1610, INVALID, INVALID, 1611, +}; + +static const uint16_t ud_itab__73[] = { + /* 0 */ INVALID, INVALID, INVALID, 1616, +}; + +static const uint16_t ud_itab__74[] = { + /* 0 */ INVALID, INVALID, INVALID, 1678, +}; + +static const uint16_t ud_itab__75[] = { + /* 0 */ INVALID, INVALID, INVALID, 1652, +}; + +static const uint16_t ud_itab__76[] = { + /* 0 */ INVALID, INVALID, INVALID, 1651, +}; + +static const uint16_t ud_itab__77[] = { + /* 0 */ INVALID, INVALID, INVALID, 1706, +}; + +static const uint16_t ud_itab__78[] = { + /* 0 */ 1568, INVALID, INVALID, 1569, +}; + +static const uint16_t ud_itab__79[] = { + /* 0 */ 1571, INVALID, INVALID, 1572, +}; + +static const uint16_t ud_itab__80[] = { + /* 0 */ 1574, INVALID, INVALID, 1575, +}; + +static const uint16_t ud_itab__81[] = { + /* 0 */ INVALID, INVALID, INVALID, 1680, +}; + +static const uint16_t ud_itab__82[] = { + /* 0 */ INVALID, INVALID, INVALID, 1682, +}; + +static const uint16_t ud_itab__83[] = { + /* 0 */ INVALID, INVALID, INVALID, 1684, +}; + +static const uint16_t ud_itab__84[] = { + /* 0 */ INVALID, INVALID, INVALID, 1686, +}; + +static const uint16_t ud_itab__85[] = { + /* 0 */ INVALID, INVALID, INVALID, 1688, +}; + +static const uint16_t ud_itab__86[] = { + /* 0 */ 1595, INVALID, INVALID, 1596, +}; + +static const uint16_t ud_itab__87[] = { + /* 0 */ INVALID, INVALID, INVALID, 1617, +}; + +static const uint16_t ud_itab__88[] = { + /* 0 */ INVALID, INVALID, INVALID, 1703, +}; + +static const uint16_t ud_itab__89[] = { + /* 0 */ INVALID, INVALID, INVALID, 1676, +}; + +static const uint16_t ud_itab__90[] = { + /* 0 */ 1598, INVALID, INVALID, 1599, +}; + +static const uint16_t ud_itab__91[] = { + /* 0 */ INVALID, INVALID, INVALID, 1691, +}; + +static const uint16_t ud_itab__92[] = { + /* 0 */ INVALID, INVALID, INVALID, 1693, +}; + +static const uint16_t ud_itab__93[] = { + /* 0 */ INVALID, INVALID, INVALID, 1695, +}; + +static const uint16_t ud_itab__94[] = { + /* 0 */ INVALID, INVALID, INVALID, 1697, +}; + +static const uint16_t ud_itab__95[] = { + /* 0 */ INVALID, INVALID, INVALID, 1699, +}; + +static const uint16_t ud_itab__96[] = { + /* 0 */ INVALID, INVALID, INVALID, 1701, +}; + +static const uint16_t ud_itab__97[] = { + /* 0 */ INVALID, INVALID, INVALID, 1712, +}; + +static const uint16_t ud_itab__98[] = { + /* 0 */ INVALID, INVALID, INVALID, 1619, +}; + +static const uint16_t ud_itab__99[] = { + /* 0 */ INVALID, INVALID, INVALID, 1621, +}; + +static const uint16_t ud_itab__100[] = { + /* 0 */ INVALID, INVALID, INVALID, 1623, +}; + +static const uint16_t ud_itab__101[] = { + /* 0 */ INVALID, INVALID, INVALID, 1625, +}; + +static const uint16_t ud_itab__102[] = { + /* 0 */ INVALID, INVALID, INVALID, 1627, +}; + +static const uint16_t ud_itab__103[] = { + /* 0 */ INVALID, INVALID, INVALID, 1629, +}; + +static const uint16_t ud_itab__104[] = { + /* 0 */ INVALID, INVALID, INVALID, 1633, +}; + +static const uint16_t ud_itab__105[] = { + /* 0 */ INVALID, INVALID, INVALID, 1631, +}; + +static const uint16_t ud_itab__106[] = { + /* 0 */ INVALID, INVALID, INVALID, 1635, +}; + +static const uint16_t ud_itab__107[] = { + /* 0 */ INVALID, INVALID, INVALID, 1637, +}; + +static const uint16_t ud_itab__108[] = { + /* 0 */ INVALID, INVALID, INVALID, 1690, +}; + +static const uint16_t ud_itab__109[] = { + /* 0 */ INVALID, INVALID, INVALID, 45, +}; + +static const uint16_t ud_itab__110[] = { + /* 0 */ INVALID, INVALID, INVALID, 41, +}; + +static const uint16_t ud_itab__111[] = { + /* 0 */ INVALID, INVALID, INVALID, 43, +}; + +static const uint16_t ud_itab__112[] = { + /* 0 */ INVALID, INVALID, INVALID, 37, +}; + +static const uint16_t ud_itab__113[] = { + /* 0 */ INVALID, INVALID, INVALID, 39, +}; + +static const uint16_t ud_itab__114[] = { + /* 0 */ 1718, 1720, INVALID, INVALID, +}; + +static const uint16_t ud_itab__115[] = { + /* 0 */ 1719, 1721, INVALID, INVALID, +}; + +static const uint16_t ud_itab__116[] = { + /* 0 */ INVALID, INVALID, INVALID, INVALID, + /* 4 */ INVALID, INVALID, INVALID, INVALID, + /* 8 */ GROUP(117), GROUP(118), GROUP(119), GROUP(120), + /* c */ GROUP(121), GROUP(122), GROUP(123), GROUP(124), + /* 10 */ INVALID, INVALID, INVALID, INVALID, + /* 14 */ GROUP(125), GROUP(126), GROUP(127), GROUP(129), + /* 18 */ INVALID, INVALID, INVALID, INVALID, + /* 1c */ INVALID, INVALID, INVALID, INVALID, + /* 20 */ GROUP(130), GROUP(131), GROUP(132), INVALID, + /* 24 */ INVALID, INVALID, INVALID, INVALID, + /* 28 */ INVALID, INVALID, INVALID, INVALID, + /* 2c */ INVALID, INVALID, INVALID, INVALID, + /* 30 */ INVALID, INVALID, INVALID, INVALID, + /* 34 */ INVALID, INVALID, INVALID, INVALID, + /* 38 */ INVALID, INVALID, INVALID, INVALID, + /* 3c */ INVALID, INVALID, INVALID, INVALID, + /* 40 */ GROUP(134), GROUP(135), GROUP(136), INVALID, + /* 44 */ GROUP(137), INVALID, INVALID, INVALID, + /* 48 */ INVALID, INVALID, INVALID, INVALID, + /* 4c */ INVALID, INVALID, INVALID, INVALID, + /* 50 */ INVALID, INVALID, INVALID, INVALID, + /* 54 */ INVALID, INVALID, INVALID, INVALID, + /* 58 */ INVALID, INVALID, INVALID, INVALID, + /* 5c */ INVALID, INVALID, INVALID, INVALID, + /* 60 */ GROUP(139), GROUP(140), GROUP(141), GROUP(142), + /* 64 */ INVALID, INVALID, INVALID, INVALID, + /* 68 */ INVALID, INVALID, INVALID, INVALID, + /* 6c */ INVALID, INVALID, INVALID, INVALID, + /* 70 */ INVALID, INVALID, INVALID, INVALID, + /* 74 */ INVALID, INVALID, INVALID, INVALID, + /* 78 */ INVALID, INVALID, INVALID, INVALID, + /* 7c */ INVALID, INVALID, INVALID, INVALID, + /* 80 */ INVALID, INVALID, INVALID, INVALID, + /* 84 */ INVALID, INVALID, INVALID, INVALID, + /* 88 */ INVALID, INVALID, INVALID, INVALID, + /* 8c */ INVALID, INVALID, INVALID, INVALID, + /* 90 */ INVALID, INVALID, INVALID, INVALID, + /* 94 */ INVALID, INVALID, INVALID, INVALID, + /* 98 */ INVALID, INVALID, INVALID, INVALID, + /* 9c */ INVALID, INVALID, INVALID, INVALID, + /* a0 */ INVALID, INVALID, INVALID, INVALID, + /* a4 */ INVALID, INVALID, INVALID, INVALID, + /* a8 */ INVALID, INVALID, INVALID, INVALID, + /* ac */ INVALID, INVALID, INVALID, INVALID, + /* b0 */ INVALID, INVALID, INVALID, INVALID, + /* b4 */ INVALID, INVALID, INVALID, INVALID, + /* b8 */ INVALID, INVALID, INVALID, INVALID, + /* bc */ INVALID, INVALID, INVALID, INVALID, + /* c0 */ INVALID, INVALID, INVALID, INVALID, + /* c4 */ INVALID, INVALID, INVALID, INVALID, + /* c8 */ INVALID, INVALID, INVALID, INVALID, + /* cc */ INVALID, INVALID, INVALID, INVALID, + /* d0 */ INVALID, INVALID, INVALID, INVALID, + /* d4 */ INVALID, INVALID, INVALID, INVALID, + /* d8 */ INVALID, INVALID, INVALID, INVALID, + /* dc */ INVALID, INVALID, INVALID, GROUP(138), + /* e0 */ INVALID, INVALID, INVALID, INVALID, + /* e4 */ INVALID, INVALID, INVALID, INVALID, + /* e8 */ INVALID, INVALID, INVALID, INVALID, + /* ec */ INVALID, INVALID, INVALID, INVALID, + /* f0 */ INVALID, INVALID, INVALID, INVALID, + /* f4 */ INVALID, INVALID, INVALID, INVALID, + /* f8 */ INVALID, INVALID, INVALID, INVALID, + /* fc */ INVALID, INVALID, INVALID, INVALID, +}; + +static const uint16_t ud_itab__117[] = { + /* 0 */ INVALID, INVALID, INVALID, 1639, +}; + +static const uint16_t ud_itab__118[] = { + /* 0 */ INVALID, INVALID, INVALID, 1641, +}; + +static const uint16_t ud_itab__119[] = { + /* 0 */ INVALID, INVALID, INVALID, 1643, +}; + +static const uint16_t ud_itab__120[] = { + /* 0 */ INVALID, INVALID, INVALID, 1645, +}; + +static const uint16_t ud_itab__121[] = { + /* 0 */ INVALID, INVALID, INVALID, 1649, +}; + +static const uint16_t ud_itab__122[] = { + /* 0 */ INVALID, INVALID, INVALID, 1647, +}; + +static const uint16_t ud_itab__123[] = { + /* 0 */ INVALID, INVALID, INVALID, 1672, +}; + +static const uint16_t ud_itab__124[] = { + /* 0 */ 1613, INVALID, INVALID, 1614, +}; + +static const uint16_t ud_itab__125[] = { + /* 0 */ INVALID, INVALID, INVALID, 1041, +}; + +static const uint16_t ud_itab__126[] = { + /* 0 */ INVALID, INVALID, INVALID, 1052, +}; + +static const uint16_t ud_itab__127[] = { + /* 0 */ INVALID, INVALID, INVALID, GROUP(128), +}; + +static const uint16_t ud_itab__128[] = { + /* 0 */ 1043, 1045, 1047, +}; + +static const uint16_t ud_itab__129[] = { + /* 0 */ INVALID, INVALID, INVALID, 201, +}; + +static const uint16_t ud_itab__130[] = { + /* 0 */ INVALID, INVALID, INVALID, 1054, +}; + +static const uint16_t ud_itab__131[] = { + /* 0 */ INVALID, INVALID, INVALID, 1552, +}; + +static const uint16_t ud_itab__132[] = { + /* 0 */ INVALID, INVALID, INVALID, GROUP(133), +}; + +static const uint16_t ud_itab__133[] = { + /* 0 */ 1058, 1059, 1060, +}; + +static const uint16_t ud_itab__134[] = { + /* 0 */ INVALID, INVALID, INVALID, 197, +}; + +static const uint16_t ud_itab__135[] = { + /* 0 */ INVALID, INVALID, INVALID, 195, +}; + +static const uint16_t ud_itab__136[] = { + /* 0 */ INVALID, INVALID, INVALID, 1674, +}; + +static const uint16_t ud_itab__137[] = { + /* 0 */ INVALID, INVALID, INVALID, 1508, +}; + +static const uint16_t ud_itab__138[] = { + /* 0 */ INVALID, INVALID, INVALID, 47, +}; + +static const uint16_t ud_itab__139[] = { + /* 0 */ INVALID, INVALID, INVALID, 1710, +}; + +static const uint16_t ud_itab__140[] = { + /* 0 */ INVALID, INVALID, INVALID, 1708, +}; + +static const uint16_t ud_itab__141[] = { + /* 0 */ INVALID, INVALID, INVALID, 1716, +}; + +static const uint16_t ud_itab__142[] = { + /* 0 */ INVALID, INVALID, INVALID, 1714, +}; + +static const uint16_t ud_itab__143[] = { + /* 0 */ 896, INVALID, INVALID, 894, +}; + +static const uint16_t ud_itab__144[] = { + /* 0 */ 1383, 1387, 1389, 1385, +}; + +static const uint16_t ud_itab__145[] = { + /* 0 */ 1302, INVALID, 1304, INVALID, +}; + +static const uint16_t ud_itab__146[] = { + /* 0 */ 1287, INVALID, 1289, INVALID, +}; + +static const uint16_t ud_itab__147[] = { + /* 0 */ 61, INVALID, INVALID, 59, +}; + +static const uint16_t ud_itab__148[] = { + /* 0 */ 65, INVALID, INVALID, 63, +}; + +static const uint16_t ud_itab__149[] = { + /* 0 */ 972, INVALID, INVALID, 970, +}; + +static const uint16_t ud_itab__150[] = { + /* 0 */ 1495, INVALID, INVALID, 1493, +}; + +static const uint16_t ud_itab__151[] = { + /* 0 */ 27, 29, 31, 25, +}; + +static const uint16_t ud_itab__152[] = { + /* 0 */ 942, 944, 946, 940, +}; + +static const uint16_t ud_itab__153[] = { + /* 0 */ 145, 150, 156, 139, +}; + +static const uint16_t ud_itab__154[] = { + /* 0 */ 134, INVALID, 163, 143, +}; + +static const uint16_t ud_itab__155[] = { + /* 0 */ 1415, 1417, 1419, 1413, +}; + +static const uint16_t ud_itab__156[] = { + /* 0 */ 814, 816, 818, 812, +}; + +static const uint16_t ud_itab__157[] = { + /* 0 */ 189, 191, 193, 187, +}; + +static const uint16_t ud_itab__158[] = { + /* 0 */ 798, 800, 802, 796, +}; + +static const uint16_t ud_itab__159[] = { + /* 0 */ 1205, INVALID, INVALID, 1203, +}; + +static const uint16_t ud_itab__160[] = { + /* 0 */ 1208, INVALID, INVALID, 1206, +}; + +static const uint16_t ud_itab__161[] = { + /* 0 */ 1211, INVALID, INVALID, 1209, +}; + +static const uint16_t ud_itab__162[] = { + /* 0 */ 983, INVALID, INVALID, 981, +}; + +static const uint16_t ud_itab__163[] = { + /* 0 */ 1034, INVALID, INVALID, 1032, +}; + +static const uint16_t ud_itab__164[] = { + /* 0 */ 1037, INVALID, INVALID, 1035, +}; + +static const uint16_t ud_itab__165[] = { + /* 0 */ 1040, INVALID, INVALID, 1038, +}; + +static const uint16_t ud_itab__166[] = { + /* 0 */ 989, INVALID, INVALID, 987, +}; + +static const uint16_t ud_itab__167[] = { + /* 0 */ 1196, INVALID, INVALID, 1194, +}; + +static const uint16_t ud_itab__168[] = { + /* 0 */ 1199, INVALID, INVALID, 1197, +}; + +static const uint16_t ud_itab__169[] = { + /* 0 */ 1202, INVALID, INVALID, 1200, +}; + +static const uint16_t ud_itab__170[] = { + /* 0 */ 986, INVALID, INVALID, 984, +}; + +static const uint16_t ud_itab__171[] = { + /* 0 */ INVALID, INVALID, INVALID, 1542, +}; + +static const uint16_t ud_itab__172[] = { + /* 0 */ INVALID, INVALID, INVALID, 1540, +}; + +static const uint16_t ud_itab__173[] = { + /* 0 */ GROUP(174), INVALID, INVALID, GROUP(175), +}; + +static const uint16_t ud_itab__174[] = { + /* 0 */ 862, 863, 906, +}; + +static const uint16_t ud_itab__175[] = { + /* 0 */ 864, 866, 907, +}; + +static const uint16_t ud_itab__176[] = { + /* 0 */ 916, INVALID, 1518, 1513, +}; + +static const uint16_t ud_itab__177[] = { + /* 0 */ 1130, 1532, 1530, 1534, +}; + +static const uint16_t ud_itab__178[] = { + /* 0 */ INVALID, INVALID, GROUP(179), INVALID, + /* 4 */ GROUP(180), INVALID, GROUP(181), INVALID, +}; + +static const uint16_t ud_itab__179[] = { + /* 0 */ 1155, INVALID, INVALID, 1159, +}; + +static const uint16_t ud_itab__180[] = { + /* 0 */ 1148, INVALID, INVALID, 1146, +}; + +static const uint16_t ud_itab__181[] = { + /* 0 */ 1134, INVALID, INVALID, 1133, +}; + +static const uint16_t ud_itab__182[] = { + /* 0 */ INVALID, INVALID, GROUP(183), INVALID, + /* 4 */ GROUP(184), INVALID, GROUP(185), INVALID, +}; + +static const uint16_t ud_itab__183[] = { + /* 0 */ 1161, INVALID, INVALID, 1165, +}; + +static const uint16_t ud_itab__184[] = { + /* 0 */ 1149, INVALID, INVALID, 1153, +}; + +static const uint16_t ud_itab__185[] = { + /* 0 */ 1138, INVALID, INVALID, 1137, +}; + +static const uint16_t ud_itab__186[] = { + /* 0 */ INVALID, INVALID, GROUP(187), GROUP(188), + /* 4 */ INVALID, INVALID, GROUP(189), GROUP(190), +}; + +static const uint16_t ud_itab__187[] = { + /* 0 */ 1167, INVALID, INVALID, 1171, +}; + +static const uint16_t ud_itab__188[] = { + /* 0 */ INVALID, INVALID, INVALID, 1538, +}; + +static const uint16_t ud_itab__189[] = { + /* 0 */ 1142, INVALID, INVALID, 1141, +}; + +static const uint16_t ud_itab__190[] = { + /* 0 */ INVALID, INVALID, INVALID, 1536, +}; + +static const uint16_t ud_itab__191[] = { + /* 0 */ 1023, INVALID, INVALID, 1024, +}; + +static const uint16_t ud_itab__192[] = { + /* 0 */ 1026, INVALID, INVALID, 1027, +}; + +static const uint16_t ud_itab__193[] = { + /* 0 */ 1029, INVALID, INVALID, 1030, +}; + +static const uint16_t ud_itab__194[] = { + /* 0 */ INVALID, 1460, INVALID, +}; + +static const uint16_t ud_itab__195[] = { + /* 0 */ INVALID, 1461, INVALID, +}; + +static const uint16_t ud_itab__196[] = { + /* 0 */ INVALID, 1546, INVALID, 1544, +}; + +static const uint16_t ud_itab__197[] = { + /* 0 */ INVALID, 1550, INVALID, 1548, +}; + +static const uint16_t ud_itab__198[] = { + /* 0 */ GROUP(199), INVALID, 912, GROUP(200), +}; + +static const uint16_t ud_itab__199[] = { + /* 0 */ 868, 869, 909, +}; + +static const uint16_t ud_itab__200[] = { + /* 0 */ 870, 872, 910, +}; + +static const uint16_t ud_itab__201[] = { + /* 0 */ 917, INVALID, 1520, 1511, +}; + +static const uint16_t ud_itab__202[] = { + /* 0 */ INVALID, GROUP(203), +}; + +static const uint16_t ud_itab__203[] = { + /* 0 */ GROUP(204), GROUP(205), GROUP(206), INVALID, + /* 4 */ INVALID, INVALID, INVALID, INVALID, +}; + +static const uint16_t ud_itab__204[] = { + /* 0 */ 821, INVALID, INVALID, INVALID, + /* 4 */ INVALID, INVALID, INVALID, INVALID, +}; + +static const uint16_t ud_itab__205[] = { + /* 0 */ 1505, INVALID, INVALID, INVALID, + /* 4 */ INVALID, INVALID, INVALID, INVALID, +}; + +static const uint16_t ud_itab__206[] = { + /* 0 */ 1506, INVALID, INVALID, INVALID, + /* 4 */ INVALID, INVALID, INVALID, INVALID, +}; + +static const uint16_t ud_itab__207[] = { + /* 0 */ INVALID, GROUP(208), +}; + +static const uint16_t ud_itab__208[] = { + /* 0 */ GROUP(209), GROUP(210), GROUP(211), GROUP(212), + /* 4 */ GROUP(213), GROUP(214), INVALID, INVALID, +}; + +static const uint16_t ud_itab__209[] = { + /* 0 */ 1507, INVALID, INVALID, INVALID, + /* 4 */ INVALID, INVALID, INVALID, INVALID, +}; + +static const uint16_t ud_itab__210[] = { + /* 0 */ 1497, INVALID, INVALID, INVALID, + /* 4 */ INVALID, INVALID, INVALID, INVALID, +}; + +static const uint16_t ud_itab__211[] = { + /* 0 */ 1498, INVALID, INVALID, INVALID, + /* 4 */ INVALID, INVALID, INVALID, INVALID, +}; + +static const uint16_t ud_itab__212[] = { + /* 0 */ 1499, INVALID, INVALID, INVALID, + /* 4 */ INVALID, INVALID, INVALID, INVALID, +}; + +static const uint16_t ud_itab__213[] = { + /* 0 */ 1500, INVALID, INVALID, INVALID, + /* 4 */ INVALID, INVALID, INVALID, INVALID, +}; + +static const uint16_t ud_itab__214[] = { + /* 0 */ 1501, INVALID, INVALID, INVALID, + /* 4 */ INVALID, INVALID, INVALID, INVALID, +}; + +static const uint16_t ud_itab__215[] = { + /* 0 */ GROUP(216), GROUP(217), +}; + +static const uint16_t ud_itab__216[] = { + /* 0 */ 679, 678, 764, 1396, + /* 4 */ 1503, 1502, INVALID, 79, +}; + +static const uint16_t ud_itab__217[] = { + /* 0 */ INVALID, INVALID, INVALID, INVALID, + /* 4 */ INVALID, GROUP(218), GROUP(219), GROUP(220), +}; + +static const uint16_t ud_itab__218[] = { + /* 0 */ 773, 774, 775, 776, + /* 4 */ 777, 778, 779, 780, +}; + +static const uint16_t ud_itab__219[] = { + /* 0 */ 804, 805, 806, 807, + /* 4 */ 808, 809, 810, 811, +}; + +static const uint16_t ud_itab__220[] = { + /* 0 */ 1362, 1363, 1364, 1365, + /* 4 */ 1366, 1367, 1368, 1369, +}; + +static const uint16_t ud_itab__221[] = { + /* 0 */ INVALID, INVALID, 1705, INVALID, +}; + +static const uint16_t ud_itab__222[] = { + /* 0 */ INVALID, INVALID, INVALID, INVALID, + /* 4 */ 1664, 1671, 1669, 1667, +}; + +static const uint16_t ud_itab__223[] = { + /* 0 */ 112, 117, 120, 110, +}; + +static const uint16_t ud_itab__224[] = { + /* 0 */ 1055, INVALID, INVALID, 1056, +}; + +static const uint16_t ud_itab__225[] = { + /* 0 */ 1051, INVALID, INVALID, 1049, +}; + +static const uint16_t ud_itab__226[] = { + /* 0 */ 1377, INVALID, INVALID, 1375, +}; + +static const uint16_t ud_itab__227[] = { + /* 0 */ GROUP(228), GROUP(235), +}; + +static const uint16_t ud_itab__228[] = { + /* 0 */ INVALID, GROUP(229), INVALID, INVALID, + /* 4 */ INVALID, INVALID, GROUP(230), GROUP(234), +}; + +static const uint16_t ud_itab__229[] = { + /* 0 */ 124, 125, 126, +}; + +static const uint16_t ud_itab__230[] = { + /* 0 */ GROUP(231), INVALID, GROUP(232), GROUP(233), +}; + +static const uint16_t ud_itab__231[] = { + /* 0 */ INVALID, 1455, INVALID, +}; + +static const uint16_t ud_itab__232[] = { + /* 0 */ INVALID, 1454, INVALID, +}; + +static const uint16_t ud_itab__233[] = { + /* 0 */ INVALID, 1453, INVALID, +}; + +static const uint16_t ud_itab__234[] = { + /* 0 */ INVALID, 1456, INVALID, +}; + +static const uint16_t ud_itab__235[] = { + /* 0 */ INVALID, INVALID, INVALID, INVALID, + /* 4 */ INVALID, INVALID, 1452, INVALID, +}; + +static const uint16_t ud_itab__236[] = { + /* 0 */ INVALID, 35, INVALID, 33, +}; + +static const uint16_t ud_itab__237[] = { + /* 0 */ 1156, INVALID, INVALID, 1157, +}; + +static const uint16_t ud_itab__238[] = { + /* 0 */ 1162, INVALID, INVALID, 1163, +}; + +static const uint16_t ud_itab__239[] = { + /* 0 */ 1168, INVALID, INVALID, 1169, +}; + +static const uint16_t ud_itab__240[] = { + /* 0 */ 1522, INVALID, INVALID, 1523, +}; + +static const uint16_t ud_itab__241[] = { + /* 0 */ 1089, INVALID, INVALID, 1090, +}; + +static const uint16_t ud_itab__242[] = { + /* 0 */ INVALID, 1517, 1521, 914, +}; + +static const uint16_t ud_itab__243[] = { + /* 0 */ 1082, INVALID, INVALID, 1080, +}; + +static const uint16_t ud_itab__244[] = { + /* 0 */ 1188, INVALID, INVALID, 1189, +}; + +static const uint16_t ud_itab__245[] = { + /* 0 */ 1191, INVALID, INVALID, 1192, +}; + +static const uint16_t ud_itab__246[] = { + /* 0 */ 1079, INVALID, INVALID, 1077, +}; + +static const uint16_t ud_itab__247[] = { + /* 0 */ 1013, INVALID, INVALID, 1011, +}; + +static const uint16_t ud_itab__248[] = { + /* 0 */ 1005, INVALID, INVALID, 1006, +}; + +static const uint16_t ud_itab__249[] = { + /* 0 */ 1008, INVALID, INVALID, 1009, +}; + +static const uint16_t ud_itab__250[] = { + /* 0 */ 1071, INVALID, INVALID, 1072, +}; + +static const uint16_t ud_itab__251[] = { + /* 0 */ 1016, INVALID, INVALID, 1014, +}; + +static const uint16_t ud_itab__252[] = { + /* 0 */ 1019, INVALID, INVALID, 1017, +}; + +static const uint16_t ud_itab__253[] = { + /* 0 */ 1143, INVALID, INVALID, 1144, +}; + +static const uint16_t ud_itab__254[] = { + /* 0 */ 1152, INVALID, INVALID, 1150, +}; + +static const uint16_t ud_itab__255[] = { + /* 0 */ 1022, INVALID, INVALID, 1020, +}; + +static const uint16_t ud_itab__256[] = { + /* 0 */ 1083, INVALID, INVALID, 1084, +}; + +static const uint16_t ud_itab__257[] = { + /* 0 */ 1088, INVALID, INVALID, 1086, +}; + +static const uint16_t ud_itab__258[] = { + /* 0 */ INVALID, 136, 132, 160, +}; + +static const uint16_t ud_itab__259[] = { + /* 0 */ 905, INVALID, INVALID, 898, +}; + +static const uint16_t ud_itab__260[] = { + /* 0 */ 1182, INVALID, INVALID, 1183, +}; + +static const uint16_t ud_itab__261[] = { + /* 0 */ 1185, INVALID, INVALID, 1186, +}; + +static const uint16_t ud_itab__262[] = { + /* 0 */ 1076, INVALID, INVALID, 1074, +}; + +static const uint16_t ud_itab__263[] = { + /* 0 */ 1114, INVALID, INVALID, 1112, +}; + +static const uint16_t ud_itab__264[] = { + /* 0 */ 999, INVALID, INVALID, 1000, +}; + +static const uint16_t ud_itab__265[] = { + /* 0 */ 1002, INVALID, INVALID, 1003, +}; + +static const uint16_t ud_itab__266[] = { + /* 0 */ 1070, INVALID, INVALID, 1068, +}; + +static const uint16_t ud_itab__267[] = { + /* 0 */ 1262, INVALID, INVALID, 1260, +}; + +static const uint16_t ud_itab__268[] = { + /* 0 */ INVALID, 1554, INVALID, INVALID, +}; + +static const uint16_t ud_itab__269[] = { + /* 0 */ 1132, INVALID, INVALID, 1131, +}; + +static const uint16_t ud_itab__270[] = { + /* 0 */ 1136, INVALID, INVALID, 1135, +}; + +static const uint16_t ud_itab__271[] = { + /* 0 */ 1140, INVALID, INVALID, 1139, +}; + +static const uint16_t ud_itab__272[] = { + /* 0 */ 1528, INVALID, INVALID, 1529, +}; + +static const uint16_t ud_itab__273[] = { + /* 0 */ 1065, INVALID, INVALID, 1066, +}; + +static const uint16_t ud_itab__274[] = { + /* 0 */ 1129, INVALID, INVALID, 1127, +}; + +static const uint16_t ud_itab__275[] = { + /* 0 */ INVALID, GROUP(276), +}; + +static const uint16_t ud_itab__276[] = { + /* 0 */ 795, INVALID, INVALID, 1515, +}; + +static const uint16_t ud_itab__277[] = { + /* 0 */ 1175, INVALID, INVALID, 1173, +}; + +static const uint16_t ud_itab__278[] = { + /* 0 */ 1178, INVALID, INVALID, 1176, +}; + +static const uint16_t ud_itab__279[] = { + /* 0 */ 1179, INVALID, INVALID, 1180, +}; + +static const uint16_t ud_itab__280[] = { + /* 0 */ 1527, INVALID, INVALID, 1525, +}; + +static const uint16_t ud_itab__281[] = { + /* 0 */ 992, INVALID, INVALID, 990, +}; + +static const uint16_t ud_itab__282[] = { + /* 0 */ 993, INVALID, INVALID, 994, +}; + +static const uint16_t ud_itab__283[] = { + /* 0 */ 996, INVALID, INVALID, 997, +}; + +static const uint16_t ud_itab__284[] = { + /* 0 */ 1238, INVALID, +}; + +static const uint16_t ud_itab__285[] = { + /* 0 */ 1093, INVALID, +}; + +static const uint16_t ud_itab__286[] = { + /* 0 */ 1239, INVALID, +}; + +static const uint16_t ud_itab__287[] = { + /* 0 */ 1094, INVALID, +}; + +static const uint16_t ud_itab__288[] = { + /* 0 */ 173, INVALID, +}; + +static const uint16_t ud_itab__289[] = { + /* 0 */ 174, INVALID, +}; + +static const uint16_t ud_itab__290[] = { + /* 0 */ 1, INVALID, +}; + +static const uint16_t ud_itab__291[] = { + /* 0 */ 4, INVALID, +}; + +static const uint16_t ud_itab__292[] = { + /* 0 */ GROUP(293), GROUP(294), INVALID, +}; + +static const uint16_t ud_itab__293[] = { + /* 0 */ 1253, INVALID, +}; + +static const uint16_t ud_itab__294[] = { + /* 0 */ 1254, INVALID, +}; + +static const uint16_t ud_itab__295[] = { + /* 0 */ GROUP(296), GROUP(297), INVALID, +}; + +static const uint16_t ud_itab__296[] = { + /* 0 */ 1106, INVALID, +}; + +static const uint16_t ud_itab__297[] = { + /* 0 */ 1107, INVALID, +}; + +static const uint16_t ud_itab__298[] = { + /* 0 */ 1653, INVALID, +}; + +static const uint16_t ud_itab__299[] = { + /* 0 */ 67, 68, +}; + +static const uint16_t ud_itab__300[] = { + /* 0 */ 706, 707, INVALID, +}; + +static const uint16_t ud_itab__301[] = { + /* 0 */ 979, 980, INVALID, +}; + +static const uint16_t ud_itab__302[] = { + /* 0 */ 21, 966, 11, 1338, + /* 4 */ 55, 1409, 1489, 106, +}; + +static const uint16_t ud_itab__303[] = { + /* 0 */ 23, 967, 13, 1339, + /* 4 */ 57, 1410, 1490, 108, +}; + +static const uint16_t ud_itab__304[] = { + /* 0 */ GROUP(305), GROUP(306), GROUP(307), GROUP(308), + /* 4 */ GROUP(309), GROUP(310), GROUP(311), GROUP(312), +}; + +static const uint16_t ud_itab__305[] = { + /* 0 */ 22, INVALID, +}; + +static const uint16_t ud_itab__306[] = { + /* 0 */ 968, INVALID, +}; + +static const uint16_t ud_itab__307[] = { + /* 0 */ 12, INVALID, +}; + +static const uint16_t ud_itab__308[] = { + /* 0 */ 1340, INVALID, +}; + +static const uint16_t ud_itab__309[] = { + /* 0 */ 56, INVALID, +}; + +static const uint16_t ud_itab__310[] = { + /* 0 */ 1411, INVALID, +}; + +static const uint16_t ud_itab__311[] = { + /* 0 */ 1491, INVALID, +}; + +static const uint16_t ud_itab__312[] = { + /* 0 */ 107, INVALID, +}; + +static const uint16_t ud_itab__313[] = { + /* 0 */ 24, 969, 14, 1341, + /* 4 */ 58, 1412, 1492, 109, +}; + +static const uint16_t ud_itab__314[] = { + /* 0 */ 1105, INVALID, INVALID, INVALID, + /* 4 */ INVALID, INVALID, INVALID, INVALID, +}; + +static const uint16_t ud_itab__315[] = { + /* 0 */ 74, 75, 76, +}; + +static const uint16_t ud_itab__316[] = { + /* 0 */ 170, 171, 172, +}; + +static const uint16_t ud_itab__317[] = { + /* 0 */ 73, INVALID, +}; + +static const uint16_t ud_itab__318[] = { + /* 0 */ GROUP(319), GROUP(320), GROUP(321), +}; + +static const uint16_t ud_itab__319[] = { + /* 0 */ 1255, 1256, +}; + +static const uint16_t ud_itab__320[] = { + /* 0 */ 1257, 1258, +}; + +static const uint16_t ud_itab__321[] = { + /* 0 */ INVALID, 1259, +}; + +static const uint16_t ud_itab__322[] = { + /* 0 */ GROUP(323), GROUP(324), GROUP(325), +}; + +static const uint16_t ud_itab__323[] = { + /* 0 */ 1108, INVALID, +}; + +static const uint16_t ud_itab__324[] = { + /* 0 */ 1109, 1110, +}; + +static const uint16_t ud_itab__325[] = { + /* 0 */ INVALID, 1111, +}; + +static const uint16_t ud_itab__326[] = { + /* 0 */ 919, 920, 923, +}; + +static const uint16_t ud_itab__327[] = { + /* 0 */ 115, 116, 119, +}; + +static const uint16_t ud_itab__328[] = { + /* 0 */ 1399, 1400, 1401, +}; + +static const uint16_t ud_itab__329[] = { + /* 0 */ 787, 788, 789, +}; + +static const uint16_t ud_itab__330[] = { + /* 0 */ 1343, 1344, 1345, +}; + +static const uint16_t ud_itab__331[] = { + /* 0 */ 1275, 1282, 1263, 1271, + /* 4 */ 1323, 1330, 1314, 1309, +}; + +static const uint16_t ud_itab__332[] = { + /* 0 */ 1280, 1283, 1264, 1270, + /* 4 */ 1319, 1326, 1315, 1311, +}; + +static const uint16_t ud_itab__333[] = { + /* 0 */ GROUP(334), GROUP(335), INVALID, INVALID, + /* 4 */ INVALID, GROUP(341), GROUP(357), GROUP(369), + /* 8 */ INVALID, GROUP(394), INVALID, INVALID, + /* c */ INVALID, GROUP(399), INVALID, INVALID, +}; + +static const uint16_t ud_itab__334[] = { + /* 0 */ 767, INVALID, +}; + +static const uint16_t ud_itab__335[] = { + /* 0 */ INVALID, INVALID, INVALID, INVALID, + /* 4 */ INVALID, INVALID, INVALID, INVALID, + /* 8 */ INVALID, INVALID, INVALID, INVALID, + /* c */ INVALID, INVALID, INVALID, INVALID, + /* 10 */ 933, 935, GROUP(336), 891, + /* 14 */ 1446, 1444, GROUP(337), 881, + /* 18 */ INVALID, INVALID, INVALID, INVALID, + /* 1c */ INVALID, INVALID, INVALID, INVALID, + /* 20 */ INVALID, INVALID, INVALID, INVALID, + /* 24 */ INVALID, INVALID, INVALID, INVALID, + /* 28 */ 859, 861, INVALID, 904, + /* 2c */ INVALID, INVALID, 1439, 130, + /* 30 */ INVALID, INVALID, INVALID, INVALID, + /* 34 */ INVALID, INVALID, INVALID, INVALID, + /* 38 */ INVALID, INVALID, INVALID, INVALID, + /* 3c */ INVALID, INVALID, INVALID, INVALID, + /* 40 */ INVALID, INVALID, INVALID, INVALID, + /* 44 */ INVALID, INVALID, INVALID, INVALID, + /* 48 */ INVALID, INVALID, INVALID, INVALID, + /* 4c */ INVALID, INVALID, INVALID, INVALID, + /* 50 */ 897, 1384, 1303, 1288, + /* 54 */ 62, 66, 973, 1496, + /* 58 */ 28, 943, 146, 135, + /* 5c */ 1416, 815, 190, 799, + /* 60 */ INVALID, INVALID, INVALID, INVALID, + /* 64 */ INVALID, INVALID, INVALID, INVALID, + /* 68 */ INVALID, INVALID, INVALID, INVALID, + /* 6c */ INVALID, INVALID, INVALID, INVALID, + /* 70 */ INVALID, INVALID, INVALID, INVALID, + /* 74 */ INVALID, INVALID, INVALID, GROUP(340), + /* 78 */ INVALID, INVALID, INVALID, INVALID, + /* 7c */ INVALID, INVALID, INVALID, INVALID, + /* 80 */ INVALID, INVALID, INVALID, INVALID, + /* 84 */ INVALID, INVALID, INVALID, INVALID, + /* 88 */ INVALID, INVALID, INVALID, INVALID, + /* 8c */ INVALID, INVALID, INVALID, INVALID, + /* 90 */ INVALID, INVALID, INVALID, INVALID, + /* 94 */ INVALID, INVALID, INVALID, INVALID, + /* 98 */ INVALID, INVALID, INVALID, INVALID, + /* 9c */ INVALID, INVALID, INVALID, INVALID, + /* a0 */ INVALID, INVALID, INVALID, INVALID, + /* a4 */ INVALID, INVALID, INVALID, INVALID, + /* a8 */ INVALID, INVALID, INVALID, INVALID, + /* ac */ INVALID, INVALID, GROUP(338), INVALID, + /* b0 */ INVALID, INVALID, INVALID, INVALID, + /* b4 */ INVALID, INVALID, INVALID, INVALID, + /* b8 */ INVALID, INVALID, INVALID, INVALID, + /* bc */ INVALID, INVALID, INVALID, INVALID, + /* c0 */ INVALID, INVALID, 113, INVALID, + /* c4 */ INVALID, INVALID, 1378, INVALID, + /* c8 */ INVALID, INVALID, INVALID, INVALID, + /* cc */ INVALID, INVALID, INVALID, INVALID, + /* d0 */ INVALID, INVALID, INVALID, INVALID, + /* d4 */ INVALID, INVALID, INVALID, INVALID, + /* d8 */ INVALID, INVALID, INVALID, INVALID, + /* dc */ INVALID, INVALID, INVALID, INVALID, + /* e0 */ INVALID, INVALID, INVALID, INVALID, + /* e4 */ INVALID, INVALID, INVALID, INVALID, + /* e8 */ INVALID, INVALID, INVALID, INVALID, + /* ec */ INVALID, INVALID, INVALID, INVALID, + /* f0 */ INVALID, INVALID, INVALID, INVALID, + /* f4 */ INVALID, INVALID, INVALID, INVALID, + /* f8 */ INVALID, INVALID, INVALID, INVALID, + /* fc */ INVALID, INVALID, INVALID, INVALID, +}; + +static const uint16_t ud_itab__336[] = { + /* 0 */ 889, 893, +}; + +static const uint16_t ud_itab__337[] = { + /* 0 */ 879, 883, +}; + +static const uint16_t ud_itab__338[] = { + /* 0 */ GROUP(339), INVALID, +}; + +static const uint16_t ud_itab__339[] = { + /* 0 */ INVALID, INVALID, INVALID, 1397, + /* 4 */ INVALID, INVALID, INVALID, INVALID, +}; + +static const uint16_t ud_itab__340[] = { + /* 0 */ 1737, 1738, +}; + +static const uint16_t ud_itab__341[] = { + /* 0 */ INVALID, INVALID, INVALID, INVALID, + /* 4 */ INVALID, INVALID, INVALID, INVALID, + /* 8 */ INVALID, INVALID, INVALID, INVALID, + /* c */ INVALID, INVALID, INVALID, INVALID, + /* 10 */ 929, 931, GROUP(342), 887, + /* 14 */ 1448, 1442, GROUP(343), 877, + /* 18 */ INVALID, INVALID, INVALID, INVALID, + /* 1c */ INVALID, INVALID, INVALID, INVALID, + /* 20 */ INVALID, INVALID, INVALID, INVALID, + /* 24 */ INVALID, INVALID, INVALID, INVALID, + /* 28 */ 855, 857, INVALID, 902, + /* 2c */ INVALID, INVALID, 1437, 128, + /* 30 */ INVALID, INVALID, INVALID, INVALID, + /* 34 */ INVALID, INVALID, INVALID, INVALID, + /* 38 */ INVALID, INVALID, INVALID, INVALID, + /* 3c */ INVALID, INVALID, INVALID, INVALID, + /* 40 */ INVALID, INVALID, INVALID, INVALID, + /* 44 */ INVALID, INVALID, INVALID, INVALID, + /* 48 */ INVALID, INVALID, INVALID, INVALID, + /* 4c */ INVALID, INVALID, INVALID, INVALID, + /* 50 */ 895, 1386, INVALID, INVALID, + /* 54 */ 60, 64, 971, 1494, + /* 58 */ 26, 941, 140, 144, + /* 5c */ 1414, 813, 188, 797, + /* 60 */ 1204, 1207, 1210, 982, + /* 64 */ 1033, 1036, 1039, 988, + /* 68 */ 1195, 1198, 1201, 985, + /* 6c */ 1543, 1541, GROUP(344), 1514, + /* 70 */ 1535, GROUP(345), GROUP(347), GROUP(349), + /* 74 */ 1025, 1028, 1031, INVALID, + /* 78 */ INVALID, INVALID, INVALID, INVALID, + /* 7c */ 1545, 1549, GROUP(351), 1512, + /* 80 */ INVALID, INVALID, INVALID, INVALID, + /* 84 */ INVALID, INVALID, INVALID, INVALID, + /* 88 */ INVALID, INVALID, INVALID, INVALID, + /* 8c */ INVALID, INVALID, INVALID, INVALID, + /* 90 */ INVALID, INVALID, INVALID, INVALID, + /* 94 */ INVALID, INVALID, INVALID, INVALID, + /* 98 */ INVALID, INVALID, INVALID, INVALID, + /* 9c */ INVALID, INVALID, INVALID, INVALID, + /* a0 */ INVALID, INVALID, INVALID, INVALID, + /* a4 */ INVALID, INVALID, INVALID, INVALID, + /* a8 */ INVALID, INVALID, INVALID, INVALID, + /* ac */ INVALID, INVALID, INVALID, INVALID, + /* b0 */ INVALID, INVALID, INVALID, INVALID, + /* b4 */ INVALID, INVALID, INVALID, INVALID, + /* b8 */ INVALID, INVALID, INVALID, INVALID, + /* bc */ INVALID, INVALID, INVALID, INVALID, + /* c0 */ INVALID, INVALID, 111, INVALID, + /* c4 */ 1057, 1050, 1376, INVALID, + /* c8 */ INVALID, INVALID, INVALID, INVALID, + /* cc */ INVALID, INVALID, INVALID, INVALID, + /* d0 */ 34, 1158, 1164, 1170, + /* d4 */ 1524, 1091, 915, GROUP(352), + /* d8 */ 1190, 1193, 1078, 1012, + /* dc */ 1007, 1010, 1073, 1015, + /* e0 */ 1018, 1145, 1151, 1021, + /* e4 */ 1085, 1087, 161, 899, + /* e8 */ 1184, 1187, 1075, 1113, + /* ec */ 1001, 1004, 1069, 1261, + /* f0 */ INVALID, GROUP(353), GROUP(354), GROUP(355), + /* f4 */ INVALID, 1067, 1128, GROUP(356), + /* f8 */ 1174, 1177, 1181, 1526, + /* fc */ 991, 995, 998, INVALID, +}; + +static const uint16_t ud_itab__342[] = { + /* 0 */ 885, INVALID, +}; + +static const uint16_t ud_itab__343[] = { + /* 0 */ 875, INVALID, +}; + +static const uint16_t ud_itab__344[] = { + /* 0 */ 865, 867, 908, +}; + +static const uint16_t ud_itab__345[] = { + /* 0 */ INVALID, INVALID, 1160, INVALID, + /* 4 */ 1147, INVALID, GROUP(346), INVALID, +}; + +static const uint16_t ud_itab__346[] = { + /* 0 */ 1751, INVALID, +}; + +static const uint16_t ud_itab__347[] = { + /* 0 */ INVALID, INVALID, 1166, INVALID, + /* 4 */ 1154, INVALID, GROUP(348), INVALID, +}; + +static const uint16_t ud_itab__348[] = { + /* 0 */ 1753, INVALID, +}; + +static const uint16_t ud_itab__349[] = { + /* 0 */ INVALID, INVALID, 1172, 1539, + /* 4 */ INVALID, INVALID, GROUP(350), 1537, +}; + +static const uint16_t ud_itab__350[] = { + /* 0 */ 1755, INVALID, +}; + +static const uint16_t ud_itab__351[] = { + /* 0 */ 871, 873, 911, +}; + +static const uint16_t ud_itab__352[] = { + /* 0 */ 1081, INVALID, +}; + +static const uint16_t ud_itab__353[] = { + /* 0 */ 1750, INVALID, +}; + +static const uint16_t ud_itab__354[] = { + /* 0 */ 1752, INVALID, +}; + +static const uint16_t ud_itab__355[] = { + /* 0 */ 1754, INVALID, +}; + +static const uint16_t ud_itab__356[] = { + /* 0 */ INVALID, 1516, +}; + +static const uint16_t ud_itab__357[] = { + /* 0 */ 1579, 1582, 1585, 1588, + /* 4 */ 1591, 1594, 1597, 1600, + /* 8 */ 1603, 1609, 1606, 1612, + /* c */ GROUP(358), GROUP(359), GROUP(360), GROUP(361), + /* 10 */ INVALID, INVALID, INVALID, INVALID, + /* 14 */ INVALID, INVALID, INVALID, 1707, + /* 18 */ GROUP(362), GROUP(363), INVALID, INVALID, + /* 1c */ 1570, 1573, 1576, INVALID, + /* 20 */ 1681, 1683, 1685, 1687, + /* 24 */ 1689, INVALID, INVALID, INVALID, + /* 28 */ 1618, 1704, 1677, 1679, + /* 2c */ GROUP(365), GROUP(366), GROUP(367), GROUP(368), + /* 30 */ 1692, 1694, 1696, 1698, + /* 34 */ 1700, 1702, INVALID, 1713, + /* 38 */ 1620, 1622, 1624, 1626, + /* 3c */ 1628, 1630, 1634, 1632, + /* 40 */ 1636, 1638, INVALID, INVALID, + /* 44 */ INVALID, INVALID, INVALID, INVALID, + /* 48 */ INVALID, INVALID, INVALID, INVALID, + /* 4c */ INVALID, INVALID, INVALID, INVALID, + /* 50 */ INVALID, INVALID, INVALID, INVALID, + /* 54 */ INVALID, INVALID, INVALID, INVALID, + /* 58 */ INVALID, INVALID, INVALID, INVALID, + /* 5c */ INVALID, INVALID, INVALID, INVALID, + /* 60 */ INVALID, INVALID, INVALID, INVALID, + /* 64 */ INVALID, INVALID, INVALID, INVALID, + /* 68 */ INVALID, INVALID, INVALID, INVALID, + /* 6c */ INVALID, INVALID, INVALID, INVALID, + /* 70 */ INVALID, INVALID, INVALID, INVALID, + /* 74 */ INVALID, INVALID, INVALID, INVALID, + /* 78 */ INVALID, INVALID, INVALID, INVALID, + /* 7c */ INVALID, INVALID, INVALID, INVALID, + /* 80 */ INVALID, INVALID, INVALID, INVALID, + /* 84 */ INVALID, INVALID, INVALID, INVALID, + /* 88 */ INVALID, INVALID, INVALID, INVALID, + /* 8c */ INVALID, INVALID, INVALID, INVALID, + /* 90 */ INVALID, INVALID, INVALID, INVALID, + /* 94 */ INVALID, INVALID, INVALID, INVALID, + /* 98 */ INVALID, INVALID, INVALID, INVALID, + /* 9c */ INVALID, INVALID, INVALID, INVALID, + /* a0 */ INVALID, INVALID, INVALID, INVALID, + /* a4 */ INVALID, INVALID, INVALID, INVALID, + /* a8 */ INVALID, INVALID, INVALID, INVALID, + /* ac */ INVALID, INVALID, INVALID, INVALID, + /* b0 */ INVALID, INVALID, INVALID, INVALID, + /* b4 */ INVALID, INVALID, INVALID, INVALID, + /* b8 */ INVALID, INVALID, INVALID, INVALID, + /* bc */ INVALID, INVALID, INVALID, INVALID, + /* c0 */ INVALID, INVALID, INVALID, INVALID, + /* c4 */ INVALID, INVALID, INVALID, INVALID, + /* c8 */ INVALID, INVALID, INVALID, INVALID, + /* cc */ INVALID, INVALID, INVALID, INVALID, + /* d0 */ INVALID, INVALID, INVALID, INVALID, + /* d4 */ INVALID, INVALID, INVALID, INVALID, + /* d8 */ INVALID, INVALID, INVALID, 46, + /* dc */ 42, 44, 38, 40, + /* e0 */ INVALID, INVALID, INVALID, INVALID, + /* e4 */ INVALID, INVALID, INVALID, INVALID, + /* e8 */ INVALID, INVALID, INVALID, INVALID, + /* ec */ INVALID, INVALID, INVALID, INVALID, + /* f0 */ INVALID, INVALID, INVALID, INVALID, + /* f4 */ INVALID, INVALID, INVALID, INVALID, + /* f8 */ INVALID, INVALID, INVALID, INVALID, + /* fc */ INVALID, INVALID, INVALID, INVALID, +}; + +static const uint16_t ud_itab__358[] = { + /* 0 */ 1732, INVALID, +}; + +static const uint16_t ud_itab__359[] = { + /* 0 */ 1730, INVALID, +}; + +static const uint16_t ud_itab__360[] = { + /* 0 */ 1735, INVALID, +}; + +static const uint16_t ud_itab__361[] = { + /* 0 */ 1736, INVALID, +}; + +static const uint16_t ud_itab__362[] = { + /* 0 */ 1722, INVALID, +}; + +static const uint16_t ud_itab__363[] = { + /* 0 */ GROUP(364), INVALID, +}; + +static const uint16_t ud_itab__364[] = { + /* 0 */ INVALID, 1723, +}; + +static const uint16_t ud_itab__365[] = { + /* 0 */ 1726, INVALID, +}; + +static const uint16_t ud_itab__366[] = { + /* 0 */ 1728, INVALID, +}; + +static const uint16_t ud_itab__367[] = { + /* 0 */ 1727, INVALID, +}; + +static const uint16_t ud_itab__368[] = { + /* 0 */ 1729, INVALID, +}; + +static const uint16_t ud_itab__369[] = { + /* 0 */ INVALID, INVALID, INVALID, INVALID, + /* 4 */ GROUP(370), GROUP(371), GROUP(372), INVALID, + /* 8 */ 1640, 1642, 1644, 1646, + /* c */ 1650, 1648, 1673, 1615, + /* 10 */ INVALID, INVALID, INVALID, INVALID, + /* 14 */ GROUP(374), 1053, GROUP(375), 202, + /* 18 */ GROUP(379), GROUP(381), INVALID, INVALID, + /* 1c */ INVALID, INVALID, INVALID, INVALID, + /* 20 */ GROUP(383), 1553, GROUP(385), INVALID, + /* 24 */ INVALID, INVALID, INVALID, INVALID, + /* 28 */ INVALID, INVALID, INVALID, INVALID, + /* 2c */ INVALID, INVALID, INVALID, INVALID, + /* 30 */ INVALID, INVALID, INVALID, INVALID, + /* 34 */ INVALID, INVALID, INVALID, INVALID, + /* 38 */ INVALID, INVALID, INVALID, INVALID, + /* 3c */ INVALID, INVALID, INVALID, INVALID, + /* 40 */ 198, 196, 1675, INVALID, + /* 44 */ 1509, INVALID, INVALID, INVALID, + /* 48 */ INVALID, INVALID, GROUP(391), GROUP(392), + /* 4c */ GROUP(393), INVALID, INVALID, INVALID, + /* 50 */ INVALID, INVALID, INVALID, INVALID, + /* 54 */ INVALID, INVALID, INVALID, INVALID, + /* 58 */ INVALID, INVALID, INVALID, INVALID, + /* 5c */ INVALID, INVALID, INVALID, INVALID, + /* 60 */ 1711, 1709, 1717, 1715, + /* 64 */ INVALID, INVALID, INVALID, INVALID, + /* 68 */ INVALID, INVALID, INVALID, INVALID, + /* 6c */ INVALID, INVALID, INVALID, INVALID, + /* 70 */ INVALID, INVALID, INVALID, INVALID, + /* 74 */ INVALID, INVALID, INVALID, INVALID, + /* 78 */ INVALID, INVALID, INVALID, INVALID, + /* 7c */ INVALID, INVALID, INVALID, INVALID, + /* 80 */ INVALID, INVALID, INVALID, INVALID, + /* 84 */ INVALID, INVALID, INVALID, INVALID, + /* 88 */ INVALID, INVALID, INVALID, INVALID, + /* 8c */ INVALID, INVALID, INVALID, INVALID, + /* 90 */ INVALID, INVALID, INVALID, INVALID, + /* 94 */ INVALID, INVALID, INVALID, INVALID, + /* 98 */ INVALID, INVALID, INVALID, INVALID, + /* 9c */ INVALID, INVALID, INVALID, INVALID, + /* a0 */ INVALID, INVALID, INVALID, INVALID, + /* a4 */ INVALID, INVALID, INVALID, INVALID, + /* a8 */ INVALID, INVALID, INVALID, INVALID, + /* ac */ INVALID, INVALID, INVALID, INVALID, + /* b0 */ INVALID, INVALID, INVALID, INVALID, + /* b4 */ INVALID, INVALID, INVALID, INVALID, + /* b8 */ INVALID, INVALID, INVALID, INVALID, + /* bc */ INVALID, INVALID, INVALID, INVALID, + /* c0 */ INVALID, INVALID, INVALID, INVALID, + /* c4 */ INVALID, INVALID, INVALID, INVALID, + /* c8 */ INVALID, INVALID, INVALID, INVALID, + /* cc */ INVALID, INVALID, INVALID, INVALID, + /* d0 */ INVALID, INVALID, INVALID, INVALID, + /* d4 */ INVALID, INVALID, INVALID, INVALID, + /* d8 */ INVALID, INVALID, INVALID, INVALID, + /* dc */ INVALID, INVALID, INVALID, 48, + /* e0 */ INVALID, INVALID, INVALID, INVALID, + /* e4 */ INVALID, INVALID, INVALID, INVALID, + /* e8 */ INVALID, INVALID, INVALID, INVALID, + /* ec */ INVALID, INVALID, INVALID, INVALID, + /* f0 */ INVALID, INVALID, INVALID, INVALID, + /* f4 */ INVALID, INVALID, INVALID, INVALID, + /* f8 */ INVALID, INVALID, INVALID, INVALID, + /* fc */ INVALID, INVALID, INVALID, INVALID, +}; + +static const uint16_t ud_itab__370[] = { + /* 0 */ 1733, INVALID, +}; + +static const uint16_t ud_itab__371[] = { + /* 0 */ 1731, INVALID, +}; + +static const uint16_t ud_itab__372[] = { + /* 0 */ GROUP(373), INVALID, +}; + +static const uint16_t ud_itab__373[] = { + /* 0 */ INVALID, 1734, +}; + +static const uint16_t ud_itab__374[] = { + /* 0 */ 1042, INVALID, +}; + +static const uint16_t ud_itab__375[] = { + /* 0 */ GROUP(376), GROUP(377), GROUP(378), +}; + +static const uint16_t ud_itab__376[] = { + /* 0 */ 1044, INVALID, +}; + +static const uint16_t ud_itab__377[] = { + /* 0 */ 1046, INVALID, +}; + +static const uint16_t ud_itab__378[] = { + /* 0 */ INVALID, 1048, +}; + +static const uint16_t ud_itab__379[] = { + /* 0 */ GROUP(380), INVALID, +}; + +static const uint16_t ud_itab__380[] = { + /* 0 */ INVALID, 1725, +}; + +static const uint16_t ud_itab__381[] = { + /* 0 */ GROUP(382), INVALID, +}; + +static const uint16_t ud_itab__382[] = { + /* 0 */ INVALID, 1724, +}; + +static const uint16_t ud_itab__383[] = { + /* 0 */ GROUP(384), INVALID, +}; + +static const uint16_t ud_itab__384[] = { + /* 0 */ 1061, INVALID, +}; + +static const uint16_t ud_itab__385[] = { + /* 0 */ GROUP(386), GROUP(388), +}; + +static const uint16_t ud_itab__386[] = { + /* 0 */ GROUP(387), INVALID, +}; + +static const uint16_t ud_itab__387[] = { + /* 0 */ 1062, INVALID, +}; + +static const uint16_t ud_itab__388[] = { + /* 0 */ GROUP(389), GROUP(390), +}; + +static const uint16_t ud_itab__389[] = { + /* 0 */ 1063, INVALID, +}; + +static const uint16_t ud_itab__390[] = { + /* 0 */ 1064, INVALID, +}; + +static const uint16_t ud_itab__391[] = { + /* 0 */ 1740, INVALID, +}; + +static const uint16_t ud_itab__392[] = { + /* 0 */ 1739, INVALID, +}; + +static const uint16_t ud_itab__393[] = { + /* 0 */ 1749, INVALID, +}; + +static const uint16_t ud_itab__394[] = { + /* 0 */ INVALID, INVALID, INVALID, INVALID, + /* 4 */ INVALID, INVALID, INVALID, INVALID, + /* 8 */ INVALID, INVALID, INVALID, INVALID, + /* c */ INVALID, INVALID, INVALID, INVALID, + /* 10 */ GROUP(395), GROUP(396), GROUP(397), INVALID, + /* 14 */ INVALID, INVALID, GROUP(398), INVALID, + /* 18 */ INVALID, INVALID, INVALID, INVALID, + /* 1c */ INVALID, INVALID, INVALID, INVALID, + /* 20 */ INVALID, INVALID, INVALID, INVALID, + /* 24 */ INVALID, INVALID, INVALID, INVALID, + /* 28 */ INVALID, INVALID, 155, INVALID, + /* 2c */ 169, 159, INVALID, INVALID, + /* 30 */ INVALID, INVALID, INVALID, INVALID, + /* 34 */ INVALID, INVALID, INVALID, INVALID, + /* 38 */ INVALID, INVALID, INVALID, INVALID, + /* 3c */ INVALID, INVALID, INVALID, INVALID, + /* 40 */ INVALID, INVALID, INVALID, INVALID, + /* 44 */ INVALID, INVALID, INVALID, INVALID, + /* 48 */ INVALID, INVALID, INVALID, INVALID, + /* 4c */ INVALID, INVALID, INVALID, INVALID, + /* 50 */ INVALID, 1390, 1305, 1290, + /* 54 */ INVALID, INVALID, INVALID, INVALID, + /* 58 */ 32, 947, 157, 164, + /* 5c */ 1420, 819, 194, 803, + /* 60 */ INVALID, INVALID, INVALID, INVALID, + /* 64 */ INVALID, INVALID, INVALID, INVALID, + /* 68 */ INVALID, INVALID, INVALID, INVALID, + /* 6c */ INVALID, INVALID, INVALID, 1519, + /* 70 */ 1531, INVALID, INVALID, INVALID, + /* 74 */ INVALID, INVALID, INVALID, INVALID, + /* 78 */ INVALID, INVALID, INVALID, INVALID, + /* 7c */ INVALID, INVALID, 913, INVALID, + /* 80 */ INVALID, INVALID, INVALID, INVALID, + /* 84 */ INVALID, INVALID, INVALID, INVALID, + /* 88 */ INVALID, INVALID, INVALID, INVALID, + /* 8c */ INVALID, INVALID, INVALID, INVALID, + /* 90 */ INVALID, INVALID, INVALID, INVALID, + /* 94 */ INVALID, INVALID, INVALID, INVALID, + /* 98 */ INVALID, INVALID, INVALID, INVALID, + /* 9c */ INVALID, INVALID, INVALID, INVALID, + /* a0 */ INVALID, INVALID, INVALID, INVALID, + /* a4 */ INVALID, INVALID, INVALID, INVALID, + /* a8 */ INVALID, INVALID, INVALID, INVALID, + /* ac */ INVALID, INVALID, INVALID, INVALID, + /* b0 */ INVALID, INVALID, INVALID, INVALID, + /* b4 */ INVALID, INVALID, INVALID, INVALID, + /* b8 */ INVALID, INVALID, INVALID, INVALID, + /* bc */ INVALID, INVALID, INVALID, INVALID, + /* c0 */ INVALID, INVALID, 121, INVALID, + /* c4 */ INVALID, INVALID, INVALID, INVALID, + /* c8 */ INVALID, INVALID, INVALID, INVALID, + /* cc */ INVALID, INVALID, INVALID, INVALID, + /* d0 */ INVALID, INVALID, INVALID, INVALID, + /* d4 */ INVALID, INVALID, INVALID, INVALID, + /* d8 */ INVALID, INVALID, INVALID, INVALID, + /* dc */ INVALID, INVALID, INVALID, INVALID, + /* e0 */ INVALID, INVALID, INVALID, INVALID, + /* e4 */ INVALID, INVALID, 133, INVALID, + /* e8 */ INVALID, INVALID, INVALID, INVALID, + /* ec */ INVALID, INVALID, INVALID, INVALID, + /* f0 */ INVALID, INVALID, INVALID, INVALID, + /* f4 */ INVALID, INVALID, INVALID, INVALID, + /* f8 */ INVALID, INVALID, INVALID, INVALID, + /* fc */ INVALID, INVALID, INVALID, INVALID, +}; + +static const uint16_t ud_itab__395[] = { + /* 0 */ 1746, 1745, +}; + +static const uint16_t ud_itab__396[] = { + /* 0 */ 1748, 1747, +}; + +static const uint16_t ud_itab__397[] = { + /* 0 */ 1567, 1565, +}; + +static const uint16_t ud_itab__398[] = { + /* 0 */ 1563, 1561, +}; + +static const uint16_t ud_itab__399[] = { + /* 0 */ INVALID, INVALID, INVALID, INVALID, + /* 4 */ INVALID, INVALID, INVALID, INVALID, + /* 8 */ INVALID, INVALID, INVALID, INVALID, + /* c */ INVALID, INVALID, INVALID, INVALID, + /* 10 */ GROUP(402), GROUP(400), GROUP(401), INVALID, + /* 14 */ INVALID, INVALID, INVALID, INVALID, + /* 18 */ INVALID, INVALID, INVALID, INVALID, + /* 1c */ INVALID, INVALID, INVALID, INVALID, + /* 20 */ INVALID, INVALID, INVALID, INVALID, + /* 24 */ INVALID, INVALID, INVALID, INVALID, + /* 28 */ INVALID, INVALID, 153, INVALID, + /* 2c */ 167, 149, INVALID, INVALID, + /* 30 */ INVALID, INVALID, INVALID, INVALID, + /* 34 */ INVALID, INVALID, INVALID, INVALID, + /* 38 */ INVALID, INVALID, INVALID, INVALID, + /* 3c */ INVALID, INVALID, INVALID, INVALID, + /* 40 */ INVALID, INVALID, INVALID, INVALID, + /* 44 */ INVALID, INVALID, INVALID, INVALID, + /* 48 */ INVALID, INVALID, INVALID, INVALID, + /* 4c */ INVALID, INVALID, INVALID, INVALID, + /* 50 */ INVALID, 1388, INVALID, INVALID, + /* 54 */ INVALID, INVALID, INVALID, INVALID, + /* 58 */ 30, 945, 151, INVALID, + /* 5c */ 1418, 817, 192, 801, + /* 60 */ INVALID, INVALID, INVALID, INVALID, + /* 64 */ INVALID, INVALID, INVALID, INVALID, + /* 68 */ INVALID, INVALID, INVALID, INVALID, + /* 6c */ INVALID, INVALID, INVALID, INVALID, + /* 70 */ 1533, INVALID, INVALID, INVALID, + /* 74 */ INVALID, INVALID, INVALID, INVALID, + /* 78 */ INVALID, INVALID, INVALID, INVALID, + /* 7c */ 1547, 1551, INVALID, INVALID, + /* 80 */ INVALID, INVALID, INVALID, INVALID, + /* 84 */ INVALID, INVALID, INVALID, INVALID, + /* 88 */ INVALID, INVALID, INVALID, INVALID, + /* 8c */ INVALID, INVALID, INVALID, INVALID, + /* 90 */ INVALID, INVALID, INVALID, INVALID, + /* 94 */ INVALID, INVALID, INVALID, INVALID, + /* 98 */ INVALID, INVALID, INVALID, INVALID, + /* 9c */ INVALID, INVALID, INVALID, INVALID, + /* a0 */ INVALID, INVALID, INVALID, INVALID, + /* a4 */ INVALID, INVALID, INVALID, INVALID, + /* a8 */ INVALID, INVALID, INVALID, INVALID, + /* ac */ INVALID, INVALID, INVALID, INVALID, + /* b0 */ INVALID, INVALID, INVALID, INVALID, + /* b4 */ INVALID, INVALID, INVALID, INVALID, + /* b8 */ INVALID, INVALID, INVALID, INVALID, + /* bc */ INVALID, INVALID, INVALID, INVALID, + /* c0 */ INVALID, INVALID, 118, INVALID, + /* c4 */ INVALID, INVALID, INVALID, INVALID, + /* c8 */ INVALID, INVALID, INVALID, INVALID, + /* cc */ INVALID, INVALID, INVALID, INVALID, + /* d0 */ 36, INVALID, INVALID, INVALID, + /* d4 */ INVALID, INVALID, INVALID, INVALID, + /* d8 */ INVALID, INVALID, INVALID, INVALID, + /* dc */ INVALID, INVALID, INVALID, INVALID, + /* e0 */ INVALID, INVALID, INVALID, INVALID, + /* e4 */ INVALID, INVALID, 137, INVALID, + /* e8 */ INVALID, INVALID, INVALID, INVALID, + /* ec */ INVALID, INVALID, INVALID, INVALID, + /* f0 */ 1555, INVALID, INVALID, INVALID, + /* f4 */ INVALID, INVALID, INVALID, INVALID, + /* f8 */ INVALID, INVALID, INVALID, INVALID, + /* fc */ INVALID, INVALID, INVALID, INVALID, +}; + +static const uint16_t ud_itab__400[] = { + /* 0 */ 1744, 1743, +}; + +static const uint16_t ud_itab__401[] = { + /* 0 */ 1559, 1557, +}; + +static const uint16_t ud_itab__402[] = { + /* 0 */ 1742, 1741, +}; + +static const uint16_t ud_itab__403[] = { + /* 0 */ GROUP(404), GROUP(335), INVALID, INVALID, + /* 4 */ INVALID, GROUP(341), GROUP(357), GROUP(369), + /* 8 */ INVALID, GROUP(394), INVALID, INVALID, + /* c */ INVALID, GROUP(399), INVALID, INVALID, +}; + +static const uint16_t ud_itab__404[] = { + /* 0 */ 765, INVALID, +}; + +static const uint16_t ud_itab__405[] = { + /* 0 */ 822, INVALID, INVALID, INVALID, + /* 4 */ INVALID, INVALID, INVALID, INVALID, +}; + +static const uint16_t ud_itab__406[] = { + /* 0 */ 823, INVALID, INVALID, INVALID, + /* 4 */ INVALID, INVALID, INVALID, INVALID, +}; + +static const uint16_t ud_itab__407[] = { + /* 0 */ 711, INVALID, +}; + +static const uint16_t ud_itab__408[] = { + /* 0 */ 719, 720, 721, +}; + +static const uint16_t ud_itab__409[] = { + /* 0 */ 1276, 1281, 1265, 1269, + /* 4 */ 1322, 1329, 1316, 1310, +}; + +static const uint16_t ud_itab__410[] = { + /* 0 */ 1277, 1284, 1268, 1272, + /* 4 */ 1321, 1328, 1325, 1308, +}; + +static const uint16_t ud_itab__411[] = { + /* 0 */ 1278, 1285, 1266, 1273, + /* 4 */ 1320, 1327, 1317, 1312, +}; + +static const uint16_t ud_itab__412[] = { + /* 0 */ 1279, 1286, 1267, 1274, + /* 4 */ 1324, 1331, 1318, 1313, +}; + +static const uint16_t ud_itab__413[] = { + /* 0 */ 3, INVALID, +}; + +static const uint16_t ud_itab__414[] = { + /* 0 */ 2, INVALID, +}; + +static const uint16_t ud_itab__415[] = { + /* 0 */ 1307, INVALID, +}; + +static const uint16_t ud_itab__416[] = { + /* 0 */ GROUP(417), GROUP(418), +}; + +static const uint16_t ud_itab__417[] = { + /* 0 */ 206, 503, 307, 357, + /* 4 */ 583, 626, 387, 413, +}; + +static const uint16_t ud_itab__418[] = { + /* 0 */ 215, 216, 217, 218, + /* 4 */ 219, 220, 221, 222, + /* 8 */ 504, 505, 506, 507, + /* c */ 508, 509, 510, 511, + /* 10 */ 309, 310, 311, 312, + /* 14 */ 313, 314, 315, 316, + /* 18 */ 359, 360, 361, 362, + /* 1c */ 363, 364, 365, 366, + /* 20 */ 585, 586, 587, 588, + /* 24 */ 589, 590, 591, 592, + /* 28 */ 610, 611, 612, 613, + /* 2c */ 614, 615, 616, 617, + /* 30 */ 388, 389, 390, 391, + /* 34 */ 392, 393, 394, 395, + /* 38 */ 414, 415, 416, 417, + /* 3c */ 418, 419, 420, 421, +}; + +static const uint16_t ud_itab__419[] = { + /* 0 */ GROUP(420), GROUP(421), +}; + +static const uint16_t ud_itab__420[] = { + /* 0 */ 476, INVALID, 569, 536, + /* 4 */ 493, 492, 580, 579, +}; + +static const uint16_t ud_itab__421[] = { + /* 0 */ 477, 478, 479, 480, + /* 4 */ 481, 482, 483, 484, + /* 8 */ 654, 655, 656, 657, + /* c */ 658, 659, 660, 661, + /* 10 */ 522, INVALID, INVALID, INVALID, + /* 14 */ INVALID, INVALID, INVALID, INVALID, + /* 18 */ 545, 546, 547, 548, + /* 1c */ 549, 550, 551, 552, + /* 20 */ 233, 204, INVALID, INVALID, + /* 24 */ 635, 653, INVALID, INVALID, + /* 28 */ 485, 486, 487, 488, + /* 2c */ 489, 490, 491, INVALID, + /* 30 */ 203, 681, 526, 523, + /* 34 */ 680, 525, 377, 454, + /* 38 */ 524, 682, 533, 532, + /* 3c */ 527, 530, 531, 376, +}; + +static const uint16_t ud_itab__422[] = { + /* 0 */ GROUP(423), GROUP(424), +}; + +static const uint16_t ud_itab__423[] = { + /* 0 */ 456, 520, 448, 450, + /* 4 */ 462, 464, 460, 458, +}; + +static const uint16_t ud_itab__424[] = { + /* 0 */ 235, 236, 237, 238, + /* 4 */ 239, 240, 241, 242, + /* 8 */ 243, 244, 245, 246, + /* c */ 247, 248, 249, 250, + /* 10 */ 251, 252, 253, 254, + /* 14 */ 255, 256, 257, 258, + /* 18 */ 259, 260, 261, 262, + /* 1c */ 263, 264, 265, 266, + /* 20 */ INVALID, INVALID, INVALID, INVALID, + /* 24 */ INVALID, INVALID, INVALID, INVALID, + /* 28 */ INVALID, 652, INVALID, INVALID, + /* 2c */ INVALID, INVALID, INVALID, INVALID, + /* 30 */ INVALID, INVALID, INVALID, INVALID, + /* 34 */ INVALID, INVALID, INVALID, INVALID, + /* 38 */ INVALID, INVALID, INVALID, INVALID, + /* 3c */ INVALID, INVALID, INVALID, INVALID, +}; + +static const uint16_t ud_itab__425[] = { + /* 0 */ GROUP(426), GROUP(427), +}; + +static const uint16_t ud_itab__426[] = { + /* 0 */ 453, 471, 467, 470, + /* 4 */ INVALID, 474, INVALID, 534, +}; + +static const uint16_t ud_itab__427[] = { + /* 0 */ 267, 268, 269, 270, + /* 4 */ 271, 272, 273, 274, + /* 8 */ 275, 276, 277, 278, + /* c */ 279, 280, 281, 282, + /* 10 */ 283, 284, 285, 286, + /* 14 */ 287, 288, 289, 290, + /* 18 */ 291, 292, 293, 294, + /* 1c */ 295, 296, 297, 298, + /* 20 */ INVALID, INVALID, 234, 455, + /* 24 */ INVALID, INVALID, INVALID, INVALID, + /* 28 */ 299, 300, 301, 302, + /* 2c */ 303, 304, 305, 306, + /* 30 */ 333, 334, 335, 336, + /* 34 */ 337, 338, 339, 340, + /* 38 */ INVALID, INVALID, INVALID, INVALID, + /* 3c */ INVALID, INVALID, INVALID, INVALID, +}; + +static const uint16_t ud_itab__428[] = { + /* 0 */ GROUP(429), GROUP(430), +}; + +static const uint16_t ud_itab__429[] = { + /* 0 */ 205, 494, 308, 358, + /* 4 */ 584, 609, 378, 404, +}; + +static const uint16_t ud_itab__430[] = { + /* 0 */ 207, 208, 209, 210, + /* 4 */ 211, 212, 213, 214, + /* 8 */ 495, 496, 497, 498, + /* c */ 499, 500, 501, 502, + /* 10 */ 317, 318, 319, 320, + /* 14 */ 321, 322, 323, 324, + /* 18 */ 325, 326, 327, 328, + /* 1c */ 329, 330, 331, 332, + /* 20 */ 618, 619, 620, 621, + /* 24 */ 622, 623, 624, 625, + /* 28 */ 593, 594, 595, 596, + /* 2c */ 597, 598, 599, 600, + /* 30 */ 405, 406, 407, 408, + /* 34 */ 409, 410, 411, 412, + /* 38 */ 379, 380, 381, 382, + /* 3c */ 383, 384, 385, 386, +}; + +static const uint16_t ud_itab__431[] = { + /* 0 */ GROUP(432), GROUP(433), +}; + +static const uint16_t ud_itab__432[] = { + /* 0 */ 475, 472, 570, 535, + /* 4 */ 528, INVALID, 529, 581, +}; + +static const uint16_t ud_itab__433[] = { + /* 0 */ 431, 432, 433, 434, + /* 4 */ 435, 436, 437, 438, + /* 8 */ 662, 663, 664, 665, + /* c */ 666, 667, 668, 669, + /* 10 */ 571, 572, 573, 574, + /* 14 */ 575, 576, 577, 578, + /* 18 */ 537, 538, 539, 540, + /* 1c */ 541, 542, 543, 544, + /* 20 */ 636, 637, 638, 639, + /* 24 */ 640, 641, 642, 643, + /* 28 */ 644, 645, 646, 647, + /* 2c */ 648, 649, 650, 651, + /* 30 */ INVALID, INVALID, INVALID, INVALID, + /* 34 */ INVALID, INVALID, INVALID, INVALID, + /* 38 */ INVALID, INVALID, INVALID, INVALID, + /* 3c */ INVALID, INVALID, INVALID, INVALID, +}; + +static const uint16_t ud_itab__434[] = { + /* 0 */ GROUP(435), GROUP(436), +}; + +static const uint16_t ud_itab__435[] = { + /* 0 */ 457, 521, 447, 449, + /* 4 */ 463, 465, 461, 459, +}; + +static const uint16_t ud_itab__436[] = { + /* 0 */ 223, 224, 225, 226, + /* 4 */ 227, 228, 229, 230, + /* 8 */ 512, 513, 514, 515, + /* c */ 516, 517, 518, 519, + /* 10 */ 367, 368, 369, 370, + /* 14 */ 371, 372, 373, 374, + /* 18 */ INVALID, 375, INVALID, INVALID, + /* 1c */ INVALID, INVALID, INVALID, INVALID, + /* 20 */ 627, 628, 629, 630, + /* 24 */ 631, 632, 633, 634, + /* 28 */ 601, 602, 603, 604, + /* 2c */ 605, 606, 607, 608, + /* 30 */ 422, 423, 424, 425, + /* 34 */ 426, 427, 428, 429, + /* 38 */ 396, 397, 398, 399, + /* 3c */ 400, 401, 402, 403, +}; + +static const uint16_t ud_itab__437[] = { + /* 0 */ GROUP(438), GROUP(439), +}; + +static const uint16_t ud_itab__438[] = { + /* 0 */ 451, 473, 466, 468, + /* 4 */ 231, 452, 232, 469, +}; + +static const uint16_t ud_itab__439[] = { + /* 0 */ 439, 440, 441, 442, + /* 4 */ 443, 444, 445, 446, + /* 8 */ 670, 671, 672, 673, + /* c */ 674, 675, 676, 677, + /* 10 */ 553, 554, 555, 556, + /* 14 */ 557, 558, 559, 560, + /* 18 */ 561, 562, 563, 564, + /* 1c */ 565, 566, 567, 568, + /* 20 */ 582, INVALID, INVALID, INVALID, + /* 24 */ INVALID, INVALID, INVALID, INVALID, + /* 28 */ 341, 342, 343, 344, + /* 2c */ 345, 346, 347, 348, + /* 30 */ 349, 350, 351, 352, + /* 34 */ 353, 354, 355, 356, + /* 38 */ INVALID, INVALID, INVALID, INVALID, + /* 3c */ INVALID, INVALID, INVALID, INVALID, +}; + +static const uint16_t ud_itab__440[] = { + /* 0 */ 754, 755, 756, +}; + +static const uint16_t ud_itab__441[] = { + /* 0 */ 760, INVALID, +}; + +static const uint16_t ud_itab__442[] = { + /* 0 */ 1428, 1433, 958, 949, + /* 4 */ 938, 691, 186, 685, +}; + +static const uint16_t ud_itab__443[] = { + /* 0 */ 1434, 1435, 959, 950, + /* 4 */ 939, 692, 185, 684, +}; + +static const uint16_t ud_itab__444[] = { + /* 0 */ 704, 183, INVALID, INVALID, + /* 4 */ INVALID, INVALID, INVALID, INVALID, +}; + +static const uint16_t ud_itab__445[] = { + /* 0 */ 703, 184, GROUP(446), 71, + /* 4 */ 757, 758, 1251, INVALID, +}; + +static const uint16_t ud_itab__446[] = { + /* 0 */ 69, 70, +}; + + +struct ud_lookup_table_list_entry ud_lookup_table_list[] = { + /* 000 */ { ud_itab__0, UD_TAB__OPC_TABLE, "opctbl" }, + /* 001 */ { ud_itab__1, UD_TAB__OPC_MODE, "/m" }, + /* 002 */ { ud_itab__2, UD_TAB__OPC_MODE, "/m" }, + /* 003 */ { ud_itab__3, UD_TAB__OPC_MODE, "/m" }, + /* 004 */ { ud_itab__4, UD_TAB__OPC_TABLE, "opctbl" }, + /* 005 */ { ud_itab__5, UD_TAB__OPC_REG, "/reg" }, + /* 006 */ { ud_itab__6, UD_TAB__OPC_MOD, "/mod" }, + /* 007 */ { ud_itab__7, UD_TAB__OPC_REG, "/reg" }, + /* 008 */ { ud_itab__8, UD_TAB__OPC_REG, "/reg" }, + /* 009 */ { ud_itab__9, UD_TAB__OPC_RM, "/rm" }, + /* 010 */ { ud_itab__10, UD_TAB__OPC_VENDOR, "/vendor" }, + /* 011 */ { ud_itab__11, UD_TAB__OPC_VENDOR, "/vendor" }, + /* 012 */ { ud_itab__12, UD_TAB__OPC_VENDOR, "/vendor" }, + /* 013 */ { ud_itab__13, UD_TAB__OPC_VENDOR, "/vendor" }, + /* 014 */ { ud_itab__14, UD_TAB__OPC_RM, "/rm" }, + /* 015 */ { ud_itab__15, UD_TAB__OPC_RM, "/rm" }, + /* 016 */ { ud_itab__16, UD_TAB__OPC_RM, "/rm" }, + /* 017 */ { ud_itab__17, UD_TAB__OPC_VENDOR, "/vendor" }, + /* 018 */ { ud_itab__18, UD_TAB__OPC_VENDOR, "/vendor" }, + /* 019 */ { ud_itab__19, UD_TAB__OPC_VENDOR, "/vendor" }, + /* 020 */ { ud_itab__20, UD_TAB__OPC_VENDOR, "/vendor" }, + /* 021 */ { ud_itab__21, UD_TAB__OPC_VENDOR, "/vendor" }, + /* 022 */ { ud_itab__22, UD_TAB__OPC_VENDOR, "/vendor" }, + /* 023 */ { ud_itab__23, UD_TAB__OPC_VENDOR, "/vendor" }, + /* 024 */ { ud_itab__24, UD_TAB__OPC_VENDOR, "/vendor" }, + /* 025 */ { ud_itab__25, UD_TAB__OPC_RM, "/rm" }, + /* 026 */ { ud_itab__26, UD_TAB__OPC_VENDOR, "/vendor" }, + /* 027 */ { ud_itab__27, UD_TAB__OPC_REG, "/reg" }, + /* 028 */ { ud_itab__28, UD_TAB__OPC_3DNOW, "/3dnow" }, + /* 029 */ { ud_itab__29, UD_TAB__OPC_SSE, "/sse" }, + /* 030 */ { ud_itab__30, UD_TAB__OPC_SSE, "/sse" }, + /* 031 */ { ud_itab__31, UD_TAB__OPC_MOD, "/mod" }, + /* 032 */ { ud_itab__32, UD_TAB__OPC_SSE, "/sse" }, + /* 033 */ { ud_itab__33, UD_TAB__OPC_SSE, "/sse" }, + /* 034 */ { ud_itab__34, UD_TAB__OPC_SSE, "/sse" }, + /* 035 */ { ud_itab__35, UD_TAB__OPC_SSE, "/sse" }, + /* 036 */ { ud_itab__36, UD_TAB__OPC_SSE, "/sse" }, + /* 037 */ { ud_itab__37, UD_TAB__OPC_MOD, "/mod" }, + /* 038 */ { ud_itab__38, UD_TAB__OPC_SSE, "/sse" }, + /* 039 */ { ud_itab__39, UD_TAB__OPC_SSE, "/sse" }, + /* 040 */ { ud_itab__40, UD_TAB__OPC_SSE, "/sse" }, + /* 041 */ { ud_itab__41, UD_TAB__OPC_REG, "/reg" }, + /* 042 */ { ud_itab__42, UD_TAB__OPC_SSE, "/sse" }, + /* 043 */ { ud_itab__43, UD_TAB__OPC_SSE, "/sse" }, + /* 044 */ { ud_itab__44, UD_TAB__OPC_SSE, "/sse" }, + /* 045 */ { ud_itab__45, UD_TAB__OPC_SSE, "/sse" }, + /* 046 */ { ud_itab__46, UD_TAB__OPC_SSE, "/sse" }, + /* 047 */ { ud_itab__47, UD_TAB__OPC_SSE, "/sse" }, + /* 048 */ { ud_itab__48, UD_TAB__OPC_SSE, "/sse" }, + /* 049 */ { ud_itab__49, UD_TAB__OPC_SSE, "/sse" }, + /* 050 */ { ud_itab__50, UD_TAB__OPC_MODE, "/m" }, + /* 051 */ { ud_itab__51, UD_TAB__OPC_VENDOR, "/vendor" }, + /* 052 */ { ud_itab__52, UD_TAB__OPC_MODE, "/m" }, + /* 053 */ { ud_itab__53, UD_TAB__OPC_VENDOR, "/vendor" }, + /* 054 */ { ud_itab__54, UD_TAB__OPC_TABLE, "opctbl" }, + /* 055 */ { ud_itab__55, UD_TAB__OPC_SSE, "/sse" }, + /* 056 */ { ud_itab__56, UD_TAB__OPC_MODE, "/m" }, + /* 057 */ { ud_itab__57, UD_TAB__OPC_VENDOR, "/vendor" }, + /* 058 */ { ud_itab__58, UD_TAB__OPC_VENDOR, "/vendor" }, + /* 059 */ { ud_itab__59, UD_TAB__OPC_SSE, "/sse" }, + /* 060 */ { ud_itab__60, UD_TAB__OPC_MODE, "/m" }, + /* 061 */ { ud_itab__61, UD_TAB__OPC_VENDOR, "/vendor" }, + /* 062 */ { ud_itab__62, UD_TAB__OPC_VENDOR, "/vendor" }, + /* 063 */ { ud_itab__63, UD_TAB__OPC_SSE, "/sse" }, + /* 064 */ { ud_itab__64, UD_TAB__OPC_SSE, "/sse" }, + /* 065 */ { ud_itab__65, UD_TAB__OPC_SSE, "/sse" }, + /* 066 */ { ud_itab__66, UD_TAB__OPC_SSE, "/sse" }, + /* 067 */ { ud_itab__67, UD_TAB__OPC_SSE, "/sse" }, + /* 068 */ { ud_itab__68, UD_TAB__OPC_SSE, "/sse" }, + /* 069 */ { ud_itab__69, UD_TAB__OPC_SSE, "/sse" }, + /* 070 */ { ud_itab__70, UD_TAB__OPC_SSE, "/sse" }, + /* 071 */ { ud_itab__71, UD_TAB__OPC_SSE, "/sse" }, + /* 072 */ { ud_itab__72, UD_TAB__OPC_SSE, "/sse" }, + /* 073 */ { ud_itab__73, UD_TAB__OPC_SSE, "/sse" }, + /* 074 */ { ud_itab__74, UD_TAB__OPC_SSE, "/sse" }, + /* 075 */ { ud_itab__75, UD_TAB__OPC_SSE, "/sse" }, + /* 076 */ { ud_itab__76, UD_TAB__OPC_SSE, "/sse" }, + /* 077 */ { ud_itab__77, UD_TAB__OPC_SSE, "/sse" }, + /* 078 */ { ud_itab__78, UD_TAB__OPC_SSE, "/sse" }, + /* 079 */ { ud_itab__79, UD_TAB__OPC_SSE, "/sse" }, + /* 080 */ { ud_itab__80, UD_TAB__OPC_SSE, "/sse" }, + /* 081 */ { ud_itab__81, UD_TAB__OPC_SSE, "/sse" }, + /* 082 */ { ud_itab__82, UD_TAB__OPC_SSE, "/sse" }, + /* 083 */ { ud_itab__83, UD_TAB__OPC_SSE, "/sse" }, + /* 084 */ { ud_itab__84, UD_TAB__OPC_SSE, "/sse" }, + /* 085 */ { ud_itab__85, UD_TAB__OPC_SSE, "/sse" }, + /* 086 */ { ud_itab__86, UD_TAB__OPC_SSE, "/sse" }, + /* 087 */ { ud_itab__87, UD_TAB__OPC_SSE, "/sse" }, + /* 088 */ { ud_itab__88, UD_TAB__OPC_SSE, "/sse" }, + /* 089 */ { ud_itab__89, UD_TAB__OPC_SSE, "/sse" }, + /* 090 */ { ud_itab__90, UD_TAB__OPC_SSE, "/sse" }, + /* 091 */ { ud_itab__91, UD_TAB__OPC_SSE, "/sse" }, + /* 092 */ { ud_itab__92, UD_TAB__OPC_SSE, "/sse" }, + /* 093 */ { ud_itab__93, UD_TAB__OPC_SSE, "/sse" }, + /* 094 */ { ud_itab__94, UD_TAB__OPC_SSE, "/sse" }, + /* 095 */ { ud_itab__95, UD_TAB__OPC_SSE, "/sse" }, + /* 096 */ { ud_itab__96, UD_TAB__OPC_SSE, "/sse" }, + /* 097 */ { ud_itab__97, UD_TAB__OPC_SSE, "/sse" }, + /* 098 */ { ud_itab__98, UD_TAB__OPC_SSE, "/sse" }, + /* 099 */ { ud_itab__99, UD_TAB__OPC_SSE, "/sse" }, + /* 100 */ { ud_itab__100, UD_TAB__OPC_SSE, "/sse" }, + /* 101 */ { ud_itab__101, UD_TAB__OPC_SSE, "/sse" }, + /* 102 */ { ud_itab__102, UD_TAB__OPC_SSE, "/sse" }, + /* 103 */ { ud_itab__103, UD_TAB__OPC_SSE, "/sse" }, + /* 104 */ { ud_itab__104, UD_TAB__OPC_SSE, "/sse" }, + /* 105 */ { ud_itab__105, UD_TAB__OPC_SSE, "/sse" }, + /* 106 */ { ud_itab__106, UD_TAB__OPC_SSE, "/sse" }, + /* 107 */ { ud_itab__107, UD_TAB__OPC_SSE, "/sse" }, + /* 108 */ { ud_itab__108, UD_TAB__OPC_SSE, "/sse" }, + /* 109 */ { ud_itab__109, UD_TAB__OPC_SSE, "/sse" }, + /* 110 */ { ud_itab__110, UD_TAB__OPC_SSE, "/sse" }, + /* 111 */ { ud_itab__111, UD_TAB__OPC_SSE, "/sse" }, + /* 112 */ { ud_itab__112, UD_TAB__OPC_SSE, "/sse" }, + /* 113 */ { ud_itab__113, UD_TAB__OPC_SSE, "/sse" }, + /* 114 */ { ud_itab__114, UD_TAB__OPC_SSE, "/sse" }, + /* 115 */ { ud_itab__115, UD_TAB__OPC_SSE, "/sse" }, + /* 116 */ { ud_itab__116, UD_TAB__OPC_TABLE, "opctbl" }, + /* 117 */ { ud_itab__117, UD_TAB__OPC_SSE, "/sse" }, + /* 118 */ { ud_itab__118, UD_TAB__OPC_SSE, "/sse" }, + /* 119 */ { ud_itab__119, UD_TAB__OPC_SSE, "/sse" }, + /* 120 */ { ud_itab__120, UD_TAB__OPC_SSE, "/sse" }, + /* 121 */ { ud_itab__121, UD_TAB__OPC_SSE, "/sse" }, + /* 122 */ { ud_itab__122, UD_TAB__OPC_SSE, "/sse" }, + /* 123 */ { ud_itab__123, UD_TAB__OPC_SSE, "/sse" }, + /* 124 */ { ud_itab__124, UD_TAB__OPC_SSE, "/sse" }, + /* 125 */ { ud_itab__125, UD_TAB__OPC_SSE, "/sse" }, + /* 126 */ { ud_itab__126, UD_TAB__OPC_SSE, "/sse" }, + /* 127 */ { ud_itab__127, UD_TAB__OPC_SSE, "/sse" }, + /* 128 */ { ud_itab__128, UD_TAB__OPC_OSIZE, "/o" }, + /* 129 */ { ud_itab__129, UD_TAB__OPC_SSE, "/sse" }, + /* 130 */ { ud_itab__130, UD_TAB__OPC_SSE, "/sse" }, + /* 131 */ { ud_itab__131, UD_TAB__OPC_SSE, "/sse" }, + /* 132 */ { ud_itab__132, UD_TAB__OPC_SSE, "/sse" }, + /* 133 */ { ud_itab__133, UD_TAB__OPC_OSIZE, "/o" }, + /* 134 */ { ud_itab__134, UD_TAB__OPC_SSE, "/sse" }, + /* 135 */ { ud_itab__135, UD_TAB__OPC_SSE, "/sse" }, + /* 136 */ { ud_itab__136, UD_TAB__OPC_SSE, "/sse" }, + /* 137 */ { ud_itab__137, UD_TAB__OPC_SSE, "/sse" }, + /* 138 */ { ud_itab__138, UD_TAB__OPC_SSE, "/sse" }, + /* 139 */ { ud_itab__139, UD_TAB__OPC_SSE, "/sse" }, + /* 140 */ { ud_itab__140, UD_TAB__OPC_SSE, "/sse" }, + /* 141 */ { ud_itab__141, UD_TAB__OPC_SSE, "/sse" }, + /* 142 */ { ud_itab__142, UD_TAB__OPC_SSE, "/sse" }, + /* 143 */ { ud_itab__143, UD_TAB__OPC_SSE, "/sse" }, + /* 144 */ { ud_itab__144, UD_TAB__OPC_SSE, "/sse" }, + /* 145 */ { ud_itab__145, UD_TAB__OPC_SSE, "/sse" }, + /* 146 */ { ud_itab__146, UD_TAB__OPC_SSE, "/sse" }, + /* 147 */ { ud_itab__147, UD_TAB__OPC_SSE, "/sse" }, + /* 148 */ { ud_itab__148, UD_TAB__OPC_SSE, "/sse" }, + /* 149 */ { ud_itab__149, UD_TAB__OPC_SSE, "/sse" }, + /* 150 */ { ud_itab__150, UD_TAB__OPC_SSE, "/sse" }, + /* 151 */ { ud_itab__151, UD_TAB__OPC_SSE, "/sse" }, + /* 152 */ { ud_itab__152, UD_TAB__OPC_SSE, "/sse" }, + /* 153 */ { ud_itab__153, UD_TAB__OPC_SSE, "/sse" }, + /* 154 */ { ud_itab__154, UD_TAB__OPC_SSE, "/sse" }, + /* 155 */ { ud_itab__155, UD_TAB__OPC_SSE, "/sse" }, + /* 156 */ { ud_itab__156, UD_TAB__OPC_SSE, "/sse" }, + /* 157 */ { ud_itab__157, UD_TAB__OPC_SSE, "/sse" }, + /* 158 */ { ud_itab__158, UD_TAB__OPC_SSE, "/sse" }, + /* 159 */ { ud_itab__159, UD_TAB__OPC_SSE, "/sse" }, + /* 160 */ { ud_itab__160, UD_TAB__OPC_SSE, "/sse" }, + /* 161 */ { ud_itab__161, UD_TAB__OPC_SSE, "/sse" }, + /* 162 */ { ud_itab__162, UD_TAB__OPC_SSE, "/sse" }, + /* 163 */ { ud_itab__163, UD_TAB__OPC_SSE, "/sse" }, + /* 164 */ { ud_itab__164, UD_TAB__OPC_SSE, "/sse" }, + /* 165 */ { ud_itab__165, UD_TAB__OPC_SSE, "/sse" }, + /* 166 */ { ud_itab__166, UD_TAB__OPC_SSE, "/sse" }, + /* 167 */ { ud_itab__167, UD_TAB__OPC_SSE, "/sse" }, + /* 168 */ { ud_itab__168, UD_TAB__OPC_SSE, "/sse" }, + /* 169 */ { ud_itab__169, UD_TAB__OPC_SSE, "/sse" }, + /* 170 */ { ud_itab__170, UD_TAB__OPC_SSE, "/sse" }, + /* 171 */ { ud_itab__171, UD_TAB__OPC_SSE, "/sse" }, + /* 172 */ { ud_itab__172, UD_TAB__OPC_SSE, "/sse" }, + /* 173 */ { ud_itab__173, UD_TAB__OPC_SSE, "/sse" }, + /* 174 */ { ud_itab__174, UD_TAB__OPC_OSIZE, "/o" }, + /* 175 */ { ud_itab__175, UD_TAB__OPC_OSIZE, "/o" }, + /* 176 */ { ud_itab__176, UD_TAB__OPC_SSE, "/sse" }, + /* 177 */ { ud_itab__177, UD_TAB__OPC_SSE, "/sse" }, + /* 178 */ { ud_itab__178, UD_TAB__OPC_REG, "/reg" }, + /* 179 */ { ud_itab__179, UD_TAB__OPC_SSE, "/sse" }, + /* 180 */ { ud_itab__180, UD_TAB__OPC_SSE, "/sse" }, + /* 181 */ { ud_itab__181, UD_TAB__OPC_SSE, "/sse" }, + /* 182 */ { ud_itab__182, UD_TAB__OPC_REG, "/reg" }, + /* 183 */ { ud_itab__183, UD_TAB__OPC_SSE, "/sse" }, + /* 184 */ { ud_itab__184, UD_TAB__OPC_SSE, "/sse" }, + /* 185 */ { ud_itab__185, UD_TAB__OPC_SSE, "/sse" }, + /* 186 */ { ud_itab__186, UD_TAB__OPC_REG, "/reg" }, + /* 187 */ { ud_itab__187, UD_TAB__OPC_SSE, "/sse" }, + /* 188 */ { ud_itab__188, UD_TAB__OPC_SSE, "/sse" }, + /* 189 */ { ud_itab__189, UD_TAB__OPC_SSE, "/sse" }, + /* 190 */ { ud_itab__190, UD_TAB__OPC_SSE, "/sse" }, + /* 191 */ { ud_itab__191, UD_TAB__OPC_SSE, "/sse" }, + /* 192 */ { ud_itab__192, UD_TAB__OPC_SSE, "/sse" }, + /* 193 */ { ud_itab__193, UD_TAB__OPC_SSE, "/sse" }, + /* 194 */ { ud_itab__194, UD_TAB__OPC_VENDOR, "/vendor" }, + /* 195 */ { ud_itab__195, UD_TAB__OPC_VENDOR, "/vendor" }, + /* 196 */ { ud_itab__196, UD_TAB__OPC_SSE, "/sse" }, + /* 197 */ { ud_itab__197, UD_TAB__OPC_SSE, "/sse" }, + /* 198 */ { ud_itab__198, UD_TAB__OPC_SSE, "/sse" }, + /* 199 */ { ud_itab__199, UD_TAB__OPC_OSIZE, "/o" }, + /* 200 */ { ud_itab__200, UD_TAB__OPC_OSIZE, "/o" }, + /* 201 */ { ud_itab__201, UD_TAB__OPC_SSE, "/sse" }, + /* 202 */ { ud_itab__202, UD_TAB__OPC_MOD, "/mod" }, + /* 203 */ { ud_itab__203, UD_TAB__OPC_REG, "/reg" }, + /* 204 */ { ud_itab__204, UD_TAB__OPC_RM, "/rm" }, + /* 205 */ { ud_itab__205, UD_TAB__OPC_RM, "/rm" }, + /* 206 */ { ud_itab__206, UD_TAB__OPC_RM, "/rm" }, + /* 207 */ { ud_itab__207, UD_TAB__OPC_MOD, "/mod" }, + /* 208 */ { ud_itab__208, UD_TAB__OPC_REG, "/reg" }, + /* 209 */ { ud_itab__209, UD_TAB__OPC_RM, "/rm" }, + /* 210 */ { ud_itab__210, UD_TAB__OPC_RM, "/rm" }, + /* 211 */ { ud_itab__211, UD_TAB__OPC_RM, "/rm" }, + /* 212 */ { ud_itab__212, UD_TAB__OPC_RM, "/rm" }, + /* 213 */ { ud_itab__213, UD_TAB__OPC_RM, "/rm" }, + /* 214 */ { ud_itab__214, UD_TAB__OPC_RM, "/rm" }, + /* 215 */ { ud_itab__215, UD_TAB__OPC_MOD, "/mod" }, + /* 216 */ { ud_itab__216, UD_TAB__OPC_REG, "/reg" }, + /* 217 */ { ud_itab__217, UD_TAB__OPC_REG, "/reg" }, + /* 218 */ { ud_itab__218, UD_TAB__OPC_RM, "/rm" }, + /* 219 */ { ud_itab__219, UD_TAB__OPC_RM, "/rm" }, + /* 220 */ { ud_itab__220, UD_TAB__OPC_RM, "/rm" }, + /* 221 */ { ud_itab__221, UD_TAB__OPC_SSE, "/sse" }, + /* 222 */ { ud_itab__222, UD_TAB__OPC_REG, "/reg" }, + /* 223 */ { ud_itab__223, UD_TAB__OPC_SSE, "/sse" }, + /* 224 */ { ud_itab__224, UD_TAB__OPC_SSE, "/sse" }, + /* 225 */ { ud_itab__225, UD_TAB__OPC_SSE, "/sse" }, + /* 226 */ { ud_itab__226, UD_TAB__OPC_SSE, "/sse" }, + /* 227 */ { ud_itab__227, UD_TAB__OPC_MOD, "/mod" }, + /* 228 */ { ud_itab__228, UD_TAB__OPC_REG, "/reg" }, + /* 229 */ { ud_itab__229, UD_TAB__OPC_OSIZE, "/o" }, + /* 230 */ { ud_itab__230, UD_TAB__OPC_SSE, "/sse" }, + /* 231 */ { ud_itab__231, UD_TAB__OPC_VENDOR, "/vendor" }, + /* 232 */ { ud_itab__232, UD_TAB__OPC_VENDOR, "/vendor" }, + /* 233 */ { ud_itab__233, UD_TAB__OPC_VENDOR, "/vendor" }, + /* 234 */ { ud_itab__234, UD_TAB__OPC_VENDOR, "/vendor" }, + /* 235 */ { ud_itab__235, UD_TAB__OPC_REG, "/reg" }, + /* 236 */ { ud_itab__236, UD_TAB__OPC_SSE, "/sse" }, + /* 237 */ { ud_itab__237, UD_TAB__OPC_SSE, "/sse" }, + /* 238 */ { ud_itab__238, UD_TAB__OPC_SSE, "/sse" }, + /* 239 */ { ud_itab__239, UD_TAB__OPC_SSE, "/sse" }, + /* 240 */ { ud_itab__240, UD_TAB__OPC_SSE, "/sse" }, + /* 241 */ { ud_itab__241, UD_TAB__OPC_SSE, "/sse" }, + /* 242 */ { ud_itab__242, UD_TAB__OPC_SSE, "/sse" }, + /* 243 */ { ud_itab__243, UD_TAB__OPC_SSE, "/sse" }, + /* 244 */ { ud_itab__244, UD_TAB__OPC_SSE, "/sse" }, + /* 245 */ { ud_itab__245, UD_TAB__OPC_SSE, "/sse" }, + /* 246 */ { ud_itab__246, UD_TAB__OPC_SSE, "/sse" }, + /* 247 */ { ud_itab__247, UD_TAB__OPC_SSE, "/sse" }, + /* 248 */ { ud_itab__248, UD_TAB__OPC_SSE, "/sse" }, + /* 249 */ { ud_itab__249, UD_TAB__OPC_SSE, "/sse" }, + /* 250 */ { ud_itab__250, UD_TAB__OPC_SSE, "/sse" }, + /* 251 */ { ud_itab__251, UD_TAB__OPC_SSE, "/sse" }, + /* 252 */ { ud_itab__252, UD_TAB__OPC_SSE, "/sse" }, + /* 253 */ { ud_itab__253, UD_TAB__OPC_SSE, "/sse" }, + /* 254 */ { ud_itab__254, UD_TAB__OPC_SSE, "/sse" }, + /* 255 */ { ud_itab__255, UD_TAB__OPC_SSE, "/sse" }, + /* 256 */ { ud_itab__256, UD_TAB__OPC_SSE, "/sse" }, + /* 257 */ { ud_itab__257, UD_TAB__OPC_SSE, "/sse" }, + /* 258 */ { ud_itab__258, UD_TAB__OPC_SSE, "/sse" }, + /* 259 */ { ud_itab__259, UD_TAB__OPC_SSE, "/sse" }, + /* 260 */ { ud_itab__260, UD_TAB__OPC_SSE, "/sse" }, + /* 261 */ { ud_itab__261, UD_TAB__OPC_SSE, "/sse" }, + /* 262 */ { ud_itab__262, UD_TAB__OPC_SSE, "/sse" }, + /* 263 */ { ud_itab__263, UD_TAB__OPC_SSE, "/sse" }, + /* 264 */ { ud_itab__264, UD_TAB__OPC_SSE, "/sse" }, + /* 265 */ { ud_itab__265, UD_TAB__OPC_SSE, "/sse" }, + /* 266 */ { ud_itab__266, UD_TAB__OPC_SSE, "/sse" }, + /* 267 */ { ud_itab__267, UD_TAB__OPC_SSE, "/sse" }, + /* 268 */ { ud_itab__268, UD_TAB__OPC_SSE, "/sse" }, + /* 269 */ { ud_itab__269, UD_TAB__OPC_SSE, "/sse" }, + /* 270 */ { ud_itab__270, UD_TAB__OPC_SSE, "/sse" }, + /* 271 */ { ud_itab__271, UD_TAB__OPC_SSE, "/sse" }, + /* 272 */ { ud_itab__272, UD_TAB__OPC_SSE, "/sse" }, + /* 273 */ { ud_itab__273, UD_TAB__OPC_SSE, "/sse" }, + /* 274 */ { ud_itab__274, UD_TAB__OPC_SSE, "/sse" }, + /* 275 */ { ud_itab__275, UD_TAB__OPC_MOD, "/mod" }, + /* 276 */ { ud_itab__276, UD_TAB__OPC_SSE, "/sse" }, + /* 277 */ { ud_itab__277, UD_TAB__OPC_SSE, "/sse" }, + /* 278 */ { ud_itab__278, UD_TAB__OPC_SSE, "/sse" }, + /* 279 */ { ud_itab__279, UD_TAB__OPC_SSE, "/sse" }, + /* 280 */ { ud_itab__280, UD_TAB__OPC_SSE, "/sse" }, + /* 281 */ { ud_itab__281, UD_TAB__OPC_SSE, "/sse" }, + /* 282 */ { ud_itab__282, UD_TAB__OPC_SSE, "/sse" }, + /* 283 */ { ud_itab__283, UD_TAB__OPC_SSE, "/sse" }, + /* 284 */ { ud_itab__284, UD_TAB__OPC_MODE, "/m" }, + /* 285 */ { ud_itab__285, UD_TAB__OPC_MODE, "/m" }, + /* 286 */ { ud_itab__286, UD_TAB__OPC_MODE, "/m" }, + /* 287 */ { ud_itab__287, UD_TAB__OPC_MODE, "/m" }, + /* 288 */ { ud_itab__288, UD_TAB__OPC_MODE, "/m" }, + /* 289 */ { ud_itab__289, UD_TAB__OPC_MODE, "/m" }, + /* 290 */ { ud_itab__290, UD_TAB__OPC_MODE, "/m" }, + /* 291 */ { ud_itab__291, UD_TAB__OPC_MODE, "/m" }, + /* 292 */ { ud_itab__292, UD_TAB__OPC_OSIZE, "/o" }, + /* 293 */ { ud_itab__293, UD_TAB__OPC_MODE, "/m" }, + /* 294 */ { ud_itab__294, UD_TAB__OPC_MODE, "/m" }, + /* 295 */ { ud_itab__295, UD_TAB__OPC_OSIZE, "/o" }, + /* 296 */ { ud_itab__296, UD_TAB__OPC_MODE, "/m" }, + /* 297 */ { ud_itab__297, UD_TAB__OPC_MODE, "/m" }, + /* 298 */ { ud_itab__298, UD_TAB__OPC_MODE, "/m" }, + /* 299 */ { ud_itab__299, UD_TAB__OPC_MODE, "/m" }, + /* 300 */ { ud_itab__300, UD_TAB__OPC_OSIZE, "/o" }, + /* 301 */ { ud_itab__301, UD_TAB__OPC_OSIZE, "/o" }, + /* 302 */ { ud_itab__302, UD_TAB__OPC_REG, "/reg" }, + /* 303 */ { ud_itab__303, UD_TAB__OPC_REG, "/reg" }, + /* 304 */ { ud_itab__304, UD_TAB__OPC_REG, "/reg" }, + /* 305 */ { ud_itab__305, UD_TAB__OPC_MODE, "/m" }, + /* 306 */ { ud_itab__306, UD_TAB__OPC_MODE, "/m" }, + /* 307 */ { ud_itab__307, UD_TAB__OPC_MODE, "/m" }, + /* 308 */ { ud_itab__308, UD_TAB__OPC_MODE, "/m" }, + /* 309 */ { ud_itab__309, UD_TAB__OPC_MODE, "/m" }, + /* 310 */ { ud_itab__310, UD_TAB__OPC_MODE, "/m" }, + /* 311 */ { ud_itab__311, UD_TAB__OPC_MODE, "/m" }, + /* 312 */ { ud_itab__312, UD_TAB__OPC_MODE, "/m" }, + /* 313 */ { ud_itab__313, UD_TAB__OPC_REG, "/reg" }, + /* 314 */ { ud_itab__314, UD_TAB__OPC_REG, "/reg" }, + /* 315 */ { ud_itab__315, UD_TAB__OPC_OSIZE, "/o" }, + /* 316 */ { ud_itab__316, UD_TAB__OPC_OSIZE, "/o" }, + /* 317 */ { ud_itab__317, UD_TAB__OPC_MODE, "/m" }, + /* 318 */ { ud_itab__318, UD_TAB__OPC_OSIZE, "/o" }, + /* 319 */ { ud_itab__319, UD_TAB__OPC_MODE, "/m" }, + /* 320 */ { ud_itab__320, UD_TAB__OPC_MODE, "/m" }, + /* 321 */ { ud_itab__321, UD_TAB__OPC_MODE, "/m" }, + /* 322 */ { ud_itab__322, UD_TAB__OPC_OSIZE, "/o" }, + /* 323 */ { ud_itab__323, UD_TAB__OPC_MODE, "/m" }, + /* 324 */ { ud_itab__324, UD_TAB__OPC_MODE, "/m" }, + /* 325 */ { ud_itab__325, UD_TAB__OPC_MODE, "/m" }, + /* 326 */ { ud_itab__326, UD_TAB__OPC_OSIZE, "/o" }, + /* 327 */ { ud_itab__327, UD_TAB__OPC_OSIZE, "/o" }, + /* 328 */ { ud_itab__328, UD_TAB__OPC_OSIZE, "/o" }, + /* 329 */ { ud_itab__329, UD_TAB__OPC_OSIZE, "/o" }, + /* 330 */ { ud_itab__330, UD_TAB__OPC_OSIZE, "/o" }, + /* 331 */ { ud_itab__331, UD_TAB__OPC_REG, "/reg" }, + /* 332 */ { ud_itab__332, UD_TAB__OPC_REG, "/reg" }, + /* 333 */ { ud_itab__333, UD_TAB__OPC_VEX, "/vex" }, + /* 334 */ { ud_itab__334, UD_TAB__OPC_MODE, "/m" }, + /* 335 */ { ud_itab__335, UD_TAB__OPC_TABLE, "opctbl" }, + /* 336 */ { ud_itab__336, UD_TAB__OPC_MOD, "/mod" }, + /* 337 */ { ud_itab__337, UD_TAB__OPC_MOD, "/mod" }, + /* 338 */ { ud_itab__338, UD_TAB__OPC_MOD, "/mod" }, + /* 339 */ { ud_itab__339, UD_TAB__OPC_REG, "/reg" }, + /* 340 */ { ud_itab__340, UD_TAB__OPC_VEX_L, "/vexl" }, + /* 341 */ { ud_itab__341, UD_TAB__OPC_TABLE, "opctbl" }, + /* 342 */ { ud_itab__342, UD_TAB__OPC_MOD, "/mod" }, + /* 343 */ { ud_itab__343, UD_TAB__OPC_MOD, "/mod" }, + /* 344 */ { ud_itab__344, UD_TAB__OPC_OSIZE, "/o" }, + /* 345 */ { ud_itab__345, UD_TAB__OPC_REG, "/reg" }, + /* 346 */ { ud_itab__346, UD_TAB__OPC_VEX_L, "/vexl" }, + /* 347 */ { ud_itab__347, UD_TAB__OPC_REG, "/reg" }, + /* 348 */ { ud_itab__348, UD_TAB__OPC_VEX_L, "/vexl" }, + /* 349 */ { ud_itab__349, UD_TAB__OPC_REG, "/reg" }, + /* 350 */ { ud_itab__350, UD_TAB__OPC_VEX_L, "/vexl" }, + /* 351 */ { ud_itab__351, UD_TAB__OPC_OSIZE, "/o" }, + /* 352 */ { ud_itab__352, UD_TAB__OPC_VEX_L, "/vexl" }, + /* 353 */ { ud_itab__353, UD_TAB__OPC_VEX_L, "/vexl" }, + /* 354 */ { ud_itab__354, UD_TAB__OPC_VEX_L, "/vexl" }, + /* 355 */ { ud_itab__355, UD_TAB__OPC_VEX_L, "/vexl" }, + /* 356 */ { ud_itab__356, UD_TAB__OPC_MOD, "/mod" }, + /* 357 */ { ud_itab__357, UD_TAB__OPC_TABLE, "opctbl" }, + /* 358 */ { ud_itab__358, UD_TAB__OPC_VEX_W, "/vexw" }, + /* 359 */ { ud_itab__359, UD_TAB__OPC_VEX_W, "/vexw" }, + /* 360 */ { ud_itab__360, UD_TAB__OPC_VEX_W, "/vexw" }, + /* 361 */ { ud_itab__361, UD_TAB__OPC_VEX_W, "/vexw" }, + /* 362 */ { ud_itab__362, UD_TAB__OPC_VEX_W, "/vexw" }, + /* 363 */ { ud_itab__363, UD_TAB__OPC_VEX_W, "/vexw" }, + /* 364 */ { ud_itab__364, UD_TAB__OPC_VEX_L, "/vexl" }, + /* 365 */ { ud_itab__365, UD_TAB__OPC_VEX_W, "/vexw" }, + /* 366 */ { ud_itab__366, UD_TAB__OPC_VEX_W, "/vexw" }, + /* 367 */ { ud_itab__367, UD_TAB__OPC_VEX_W, "/vexw" }, + /* 368 */ { ud_itab__368, UD_TAB__OPC_VEX_W, "/vexw" }, + /* 369 */ { ud_itab__369, UD_TAB__OPC_TABLE, "opctbl" }, + /* 370 */ { ud_itab__370, UD_TAB__OPC_VEX_W, "/vexw" }, + /* 371 */ { ud_itab__371, UD_TAB__OPC_VEX_W, "/vexw" }, + /* 372 */ { ud_itab__372, UD_TAB__OPC_VEX_W, "/vexw" }, + /* 373 */ { ud_itab__373, UD_TAB__OPC_VEX_L, "/vexl" }, + /* 374 */ { ud_itab__374, UD_TAB__OPC_VEX_W, "/vexw" }, + /* 375 */ { ud_itab__375, UD_TAB__OPC_OSIZE, "/o" }, + /* 376 */ { ud_itab__376, UD_TAB__OPC_VEX_W, "/vexw" }, + /* 377 */ { ud_itab__377, UD_TAB__OPC_VEX_W, "/vexw" }, + /* 378 */ { ud_itab__378, UD_TAB__OPC_VEX_W, "/vexw" }, + /* 379 */ { ud_itab__379, UD_TAB__OPC_VEX_W, "/vexw" }, + /* 380 */ { ud_itab__380, UD_TAB__OPC_VEX_L, "/vexl" }, + /* 381 */ { ud_itab__381, UD_TAB__OPC_VEX_W, "/vexw" }, + /* 382 */ { ud_itab__382, UD_TAB__OPC_VEX_L, "/vexl" }, + /* 383 */ { ud_itab__383, UD_TAB__OPC_VEX_W, "/vexw" }, + /* 384 */ { ud_itab__384, UD_TAB__OPC_VEX_L, "/vexl" }, + /* 385 */ { ud_itab__385, UD_TAB__OPC_MODE, "/m" }, + /* 386 */ { ud_itab__386, UD_TAB__OPC_VEX_W, "/vexw" }, + /* 387 */ { ud_itab__387, UD_TAB__OPC_VEX_L, "/vexl" }, + /* 388 */ { ud_itab__388, UD_TAB__OPC_VEX_W, "/vexw" }, + /* 389 */ { ud_itab__389, UD_TAB__OPC_VEX_L, "/vexl" }, + /* 390 */ { ud_itab__390, UD_TAB__OPC_VEX_L, "/vexl" }, + /* 391 */ { ud_itab__391, UD_TAB__OPC_VEX_W, "/vexw" }, + /* 392 */ { ud_itab__392, UD_TAB__OPC_VEX_W, "/vexw" }, + /* 393 */ { ud_itab__393, UD_TAB__OPC_VEX_W, "/vexw" }, + /* 394 */ { ud_itab__394, UD_TAB__OPC_TABLE, "opctbl" }, + /* 395 */ { ud_itab__395, UD_TAB__OPC_MOD, "/mod" }, + /* 396 */ { ud_itab__396, UD_TAB__OPC_MOD, "/mod" }, + /* 397 */ { ud_itab__397, UD_TAB__OPC_MOD, "/mod" }, + /* 398 */ { ud_itab__398, UD_TAB__OPC_MOD, "/mod" }, + /* 399 */ { ud_itab__399, UD_TAB__OPC_TABLE, "opctbl" }, + /* 400 */ { ud_itab__400, UD_TAB__OPC_MOD, "/mod" }, + /* 401 */ { ud_itab__401, UD_TAB__OPC_MOD, "/mod" }, + /* 402 */ { ud_itab__402, UD_TAB__OPC_MOD, "/mod" }, + /* 403 */ { ud_itab__403, UD_TAB__OPC_VEX, "/vex" }, + /* 404 */ { ud_itab__404, UD_TAB__OPC_MODE, "/m" }, + /* 405 */ { ud_itab__405, UD_TAB__OPC_REG, "/reg" }, + /* 406 */ { ud_itab__406, UD_TAB__OPC_REG, "/reg" }, + /* 407 */ { ud_itab__407, UD_TAB__OPC_MODE, "/m" }, + /* 408 */ { ud_itab__408, UD_TAB__OPC_OSIZE, "/o" }, + /* 409 */ { ud_itab__409, UD_TAB__OPC_REG, "/reg" }, + /* 410 */ { ud_itab__410, UD_TAB__OPC_REG, "/reg" }, + /* 411 */ { ud_itab__411, UD_TAB__OPC_REG, "/reg" }, + /* 412 */ { ud_itab__412, UD_TAB__OPC_REG, "/reg" }, + /* 413 */ { ud_itab__413, UD_TAB__OPC_MODE, "/m" }, + /* 414 */ { ud_itab__414, UD_TAB__OPC_MODE, "/m" }, + /* 415 */ { ud_itab__415, UD_TAB__OPC_MODE, "/m" }, + /* 416 */ { ud_itab__416, UD_TAB__OPC_MOD, "/mod" }, + /* 417 */ { ud_itab__417, UD_TAB__OPC_REG, "/reg" }, + /* 418 */ { ud_itab__418, UD_TAB__OPC_X87, "/x87" }, + /* 419 */ { ud_itab__419, UD_TAB__OPC_MOD, "/mod" }, + /* 420 */ { ud_itab__420, UD_TAB__OPC_REG, "/reg" }, + /* 421 */ { ud_itab__421, UD_TAB__OPC_X87, "/x87" }, + /* 422 */ { ud_itab__422, UD_TAB__OPC_MOD, "/mod" }, + /* 423 */ { ud_itab__423, UD_TAB__OPC_REG, "/reg" }, + /* 424 */ { ud_itab__424, UD_TAB__OPC_X87, "/x87" }, + /* 425 */ { ud_itab__425, UD_TAB__OPC_MOD, "/mod" }, + /* 426 */ { ud_itab__426, UD_TAB__OPC_REG, "/reg" }, + /* 427 */ { ud_itab__427, UD_TAB__OPC_X87, "/x87" }, + /* 428 */ { ud_itab__428, UD_TAB__OPC_MOD, "/mod" }, + /* 429 */ { ud_itab__429, UD_TAB__OPC_REG, "/reg" }, + /* 430 */ { ud_itab__430, UD_TAB__OPC_X87, "/x87" }, + /* 431 */ { ud_itab__431, UD_TAB__OPC_MOD, "/mod" }, + /* 432 */ { ud_itab__432, UD_TAB__OPC_REG, "/reg" }, + /* 433 */ { ud_itab__433, UD_TAB__OPC_X87, "/x87" }, + /* 434 */ { ud_itab__434, UD_TAB__OPC_MOD, "/mod" }, + /* 435 */ { ud_itab__435, UD_TAB__OPC_REG, "/reg" }, + /* 436 */ { ud_itab__436, UD_TAB__OPC_X87, "/x87" }, + /* 437 */ { ud_itab__437, UD_TAB__OPC_MOD, "/mod" }, + /* 438 */ { ud_itab__438, UD_TAB__OPC_REG, "/reg" }, + /* 439 */ { ud_itab__439, UD_TAB__OPC_X87, "/x87" }, + /* 440 */ { ud_itab__440, UD_TAB__OPC_ASIZE, "/a" }, + /* 441 */ { ud_itab__441, UD_TAB__OPC_MODE, "/m" }, + /* 442 */ { ud_itab__442, UD_TAB__OPC_REG, "/reg" }, + /* 443 */ { ud_itab__443, UD_TAB__OPC_REG, "/reg" }, + /* 444 */ { ud_itab__444, UD_TAB__OPC_REG, "/reg" }, + /* 445 */ { ud_itab__445, UD_TAB__OPC_REG, "/reg" }, + /* 446 */ { ud_itab__446, UD_TAB__OPC_MODE, "/m" }, +}; + +/* itab entry operand definitions (for readability) */ +#define O_AL { OP_AL, SZ_B } +#define O_AX { OP_AX, SZ_W } +#define O_Av { OP_A, SZ_V } +#define O_C { OP_C, SZ_NA } +#define O_CL { OP_CL, SZ_B } +#define O_CS { OP_CS, SZ_NA } +#define O_CX { OP_CX, SZ_W } +#define O_D { OP_D, SZ_NA } +#define O_DL { OP_DL, SZ_B } +#define O_DS { OP_DS, SZ_NA } +#define O_DX { OP_DX, SZ_W } +#define O_E { OP_E, SZ_NA } +#define O_ES { OP_ES, SZ_NA } +#define O_Eb { OP_E, SZ_B } +#define O_Ed { OP_E, SZ_D } +#define O_Eq { OP_E, SZ_Q } +#define O_Ev { OP_E, SZ_V } +#define O_Ew { OP_E, SZ_W } +#define O_Ey { OP_E, SZ_Y } +#define O_Ez { OP_E, SZ_Z } +#define O_FS { OP_FS, SZ_NA } +#define O_Fv { OP_F, SZ_V } +#define O_G { OP_G, SZ_NA } +#define O_GS { OP_GS, SZ_NA } +#define O_Gb { OP_G, SZ_B } +#define O_Gd { OP_G, SZ_D } +#define O_Gq { OP_G, SZ_Q } +#define O_Gv { OP_G, SZ_V } +#define O_Gw { OP_G, SZ_W } +#define O_Gy { OP_G, SZ_Y } +#define O_Gz { OP_G, SZ_Z } +#define O_H { OP_H, SZ_X } +#define O_Hqq { OP_H, SZ_QQ } +#define O_Hx { OP_H, SZ_X } +#define O_I1 { OP_I1, SZ_NA } +#define O_I3 { OP_I3, SZ_NA } +#define O_Ib { OP_I, SZ_B } +#define O_Iv { OP_I, SZ_V } +#define O_Iw { OP_I, SZ_W } +#define O_Iz { OP_I, SZ_Z } +#define O_Jb { OP_J, SZ_B } +#define O_Jv { OP_J, SZ_V } +#define O_Jz { OP_J, SZ_Z } +#define O_L { OP_L, SZ_O } +#define O_Lx { OP_L, SZ_X } +#define O_M { OP_M, SZ_NA } +#define O_Mb { OP_M, SZ_B } +#define O_MbRd { OP_MR, SZ_BD } +#define O_MbRv { OP_MR, SZ_BV } +#define O_Md { OP_M, SZ_D } +#define O_MdRy { OP_MR, SZ_DY } +#define O_MdU { OP_MU, SZ_DO } +#define O_Mdq { OP_M, SZ_DQ } +#define O_Mo { OP_M, SZ_O } +#define O_Mq { OP_M, SZ_Q } +#define O_MqU { OP_MU, SZ_QO } +#define O_Ms { OP_M, SZ_W } +#define O_Mt { OP_M, SZ_T } +#define O_Mv { OP_M, SZ_V } +#define O_Mw { OP_M, SZ_W } +#define O_MwRd { OP_MR, SZ_WD } +#define O_MwRv { OP_MR, SZ_WV } +#define O_MwRy { OP_MR, SZ_WY } +#define O_MwU { OP_MU, SZ_WO } +#define O_N { OP_N, SZ_Q } +#define O_NONE { OP_NONE, SZ_NA } +#define O_Ob { OP_O, SZ_B } +#define O_Ov { OP_O, SZ_V } +#define O_Ow { OP_O, SZ_W } +#define O_P { OP_P, SZ_Q } +#define O_Q { OP_Q, SZ_Q } +#define O_R { OP_R, SZ_RDQ } +#define O_R0b { OP_R0, SZ_B } +#define O_R0v { OP_R0, SZ_V } +#define O_R0w { OP_R0, SZ_W } +#define O_R0y { OP_R0, SZ_Y } +#define O_R0z { OP_R0, SZ_Z } +#define O_R1b { OP_R1, SZ_B } +#define O_R1v { OP_R1, SZ_V } +#define O_R1w { OP_R1, SZ_W } +#define O_R1y { OP_R1, SZ_Y } +#define O_R1z { OP_R1, SZ_Z } +#define O_R2b { OP_R2, SZ_B } +#define O_R2v { OP_R2, SZ_V } +#define O_R2w { OP_R2, SZ_W } +#define O_R2y { OP_R2, SZ_Y } +#define O_R2z { OP_R2, SZ_Z } +#define O_R3b { OP_R3, SZ_B } +#define O_R3v { OP_R3, SZ_V } +#define O_R3w { OP_R3, SZ_W } +#define O_R3y { OP_R3, SZ_Y } +#define O_R3z { OP_R3, SZ_Z } +#define O_R4b { OP_R4, SZ_B } +#define O_R4v { OP_R4, SZ_V } +#define O_R4w { OP_R4, SZ_W } +#define O_R4y { OP_R4, SZ_Y } +#define O_R4z { OP_R4, SZ_Z } +#define O_R5b { OP_R5, SZ_B } +#define O_R5v { OP_R5, SZ_V } +#define O_R5w { OP_R5, SZ_W } +#define O_R5y { OP_R5, SZ_Y } +#define O_R5z { OP_R5, SZ_Z } +#define O_R6b { OP_R6, SZ_B } +#define O_R6v { OP_R6, SZ_V } +#define O_R6w { OP_R6, SZ_W } +#define O_R6y { OP_R6, SZ_Y } +#define O_R6z { OP_R6, SZ_Z } +#define O_R7b { OP_R7, SZ_B } +#define O_R7v { OP_R7, SZ_V } +#define O_R7w { OP_R7, SZ_W } +#define O_R7y { OP_R7, SZ_Y } +#define O_R7z { OP_R7, SZ_Z } +#define O_S { OP_S, SZ_W } +#define O_SS { OP_SS, SZ_NA } +#define O_ST0 { OP_ST0, SZ_NA } +#define O_ST1 { OP_ST1, SZ_NA } +#define O_ST2 { OP_ST2, SZ_NA } +#define O_ST3 { OP_ST3, SZ_NA } +#define O_ST4 { OP_ST4, SZ_NA } +#define O_ST5 { OP_ST5, SZ_NA } +#define O_ST6 { OP_ST6, SZ_NA } +#define O_ST7 { OP_ST7, SZ_NA } +#define O_U { OP_U, SZ_O } +#define O_Ux { OP_U, SZ_X } +#define O_V { OP_V, SZ_DQ } +#define O_Vdq { OP_V, SZ_DQ } +#define O_Vqq { OP_V, SZ_QQ } +#define O_Vsd { OP_V, SZ_Q } +#define O_Vx { OP_V, SZ_X } +#define O_W { OP_W, SZ_DQ } +#define O_Wdq { OP_W, SZ_DQ } +#define O_Wqq { OP_W, SZ_QQ } +#define O_Wsd { OP_W, SZ_Q } +#define O_Wx { OP_W, SZ_X } +#define O_eAX { OP_eAX, SZ_Z } +#define O_eCX { OP_eCX, SZ_Z } +#define O_eDX { OP_eDX, SZ_Z } +#define O_rAX { OP_rAX, SZ_V } +#define O_rCX { OP_rCX, SZ_V } +#define O_rDX { OP_rDX, SZ_V } +#define O_sIb { OP_sI, SZ_B } +#define O_sIv { OP_sI, SZ_V } +#define O_sIz { OP_sI, SZ_Z } + +struct ud_itab_entry ud_itab[] = { + /* 0000 */ { UD_Iinvalid, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0001 */ { UD_Iaaa, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0002 */ { UD_Iaad, O_Ib, O_NONE, O_NONE, O_NONE, P_none }, + /* 0003 */ { UD_Iaam, O_Ib, O_NONE, O_NONE, O_NONE, P_none }, + /* 0004 */ { UD_Iaas, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0005 */ { UD_Iadc, O_Eb, O_Gb, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0006 */ { UD_Iadc, O_Ev, O_Gv, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0007 */ { UD_Iadc, O_Gb, O_Eb, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0008 */ { UD_Iadc, O_Gv, O_Ev, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0009 */ { UD_Iadc, O_AL, O_Ib, O_NONE, O_NONE, P_none }, + /* 0010 */ { UD_Iadc, O_rAX, O_sIz, O_NONE, O_NONE, P_oso|P_rexw }, + /* 0011 */ { UD_Iadc, O_Eb, O_Ib, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0012 */ { UD_Iadc, O_Eb, O_Ib, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_inv64 }, + /* 0013 */ { UD_Iadc, O_Ev, O_sIz, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0014 */ { UD_Iadc, O_Ev, O_sIb, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0015 */ { UD_Iadd, O_Eb, O_Gb, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0016 */ { UD_Iadd, O_Ev, O_Gv, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0017 */ { UD_Iadd, O_Gb, O_Eb, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0018 */ { UD_Iadd, O_Gv, O_Ev, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0019 */ { UD_Iadd, O_AL, O_Ib, O_NONE, O_NONE, P_none }, + /* 0020 */ { UD_Iadd, O_rAX, O_sIz, O_NONE, O_NONE, P_oso|P_rexw }, + /* 0021 */ { UD_Iadd, O_Eb, O_Ib, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0022 */ { UD_Iadd, O_Eb, O_Ib, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_inv64 }, + /* 0023 */ { UD_Iadd, O_Ev, O_sIz, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0024 */ { UD_Iadd, O_Ev, O_sIb, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0025 */ { UD_Iaddpd, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0026 */ { UD_Ivaddpd, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 0027 */ { UD_Iaddps, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0028 */ { UD_Ivaddps, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 0029 */ { UD_Iaddsd, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0030 */ { UD_Ivaddsd, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0031 */ { UD_Iaddss, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0032 */ { UD_Ivaddss, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0033 */ { UD_Iaddsubpd, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0034 */ { UD_Ivaddsubpd, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0035 */ { UD_Iaddsubps, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0036 */ { UD_Ivaddsubps, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0037 */ { UD_Iaesdec, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0038 */ { UD_Ivaesdec, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0039 */ { UD_Iaesdeclast, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0040 */ { UD_Ivaesdeclast, O_Vx, O_Wx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0041 */ { UD_Iaesenc, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0042 */ { UD_Ivaesenc, O_Vx, O_Wx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0043 */ { UD_Iaesenclast, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0044 */ { UD_Ivaesenclast, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0045 */ { UD_Iaesimc, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0046 */ { UD_Ivaesimc, O_Vx, O_Wx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0047 */ { UD_Iaeskeygenassist, O_V, O_W, O_Ib, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0048 */ { UD_Ivaeskeygenassist, O_Vx, O_Wx, O_Ib, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0049 */ { UD_Iand, O_Eb, O_Gb, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0050 */ { UD_Iand, O_Ev, O_Gv, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0051 */ { UD_Iand, O_Gb, O_Eb, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0052 */ { UD_Iand, O_Gv, O_Ev, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0053 */ { UD_Iand, O_AL, O_Ib, O_NONE, O_NONE, P_none }, + /* 0054 */ { UD_Iand, O_rAX, O_sIz, O_NONE, O_NONE, P_oso|P_rexw }, + /* 0055 */ { UD_Iand, O_Eb, O_Ib, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0056 */ { UD_Iand, O_Eb, O_Ib, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_inv64 }, + /* 0057 */ { UD_Iand, O_Ev, O_sIz, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0058 */ { UD_Iand, O_Ev, O_sIb, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0059 */ { UD_Iandpd, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0060 */ { UD_Ivandpd, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 0061 */ { UD_Iandps, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0062 */ { UD_Ivandps, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0063 */ { UD_Iandnpd, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0064 */ { UD_Ivandnpd, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0065 */ { UD_Iandnps, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0066 */ { UD_Ivandnps, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0067 */ { UD_Iarpl, O_Ew, O_Gw, O_NONE, O_NONE, P_aso }, + /* 0068 */ { UD_Imovsxd, O_Gq, O_Ed, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexx|P_rexr|P_rexb }, + /* 0069 */ { UD_Icall, O_Ev, O_NONE, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0070 */ { UD_Icall, O_Eq, O_NONE, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb|P_def64 }, + /* 0071 */ { UD_Icall, O_Fv, O_NONE, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0072 */ { UD_Icall, O_Jz, O_NONE, O_NONE, O_NONE, P_oso|P_def64 }, + /* 0073 */ { UD_Icall, O_Av, O_NONE, O_NONE, O_NONE, P_oso }, + /* 0074 */ { UD_Icbw, O_NONE, O_NONE, O_NONE, O_NONE, P_oso|P_rexw }, + /* 0075 */ { UD_Icwde, O_NONE, O_NONE, O_NONE, O_NONE, P_oso|P_rexw }, + /* 0076 */ { UD_Icdqe, O_NONE, O_NONE, O_NONE, O_NONE, P_oso|P_rexw }, + /* 0077 */ { UD_Iclc, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0078 */ { UD_Icld, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0079 */ { UD_Iclflush, O_M, O_NONE, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0080 */ { UD_Iclgi, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0081 */ { UD_Icli, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0082 */ { UD_Iclts, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0083 */ { UD_Icmc, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0084 */ { UD_Icmovo, O_Gv, O_Ev, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0085 */ { UD_Icmovno, O_Gv, O_Ev, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0086 */ { UD_Icmovb, O_Gv, O_Ev, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0087 */ { UD_Icmovae, O_Gv, O_Ev, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0088 */ { UD_Icmovz, O_Gv, O_Ev, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0089 */ { UD_Icmovnz, O_Gv, O_Ev, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0090 */ { UD_Icmovbe, O_Gv, O_Ev, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0091 */ { UD_Icmova, O_Gv, O_Ev, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0092 */ { UD_Icmovs, O_Gv, O_Ev, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0093 */ { UD_Icmovns, O_Gv, O_Ev, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0094 */ { UD_Icmovp, O_Gv, O_Ev, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0095 */ { UD_Icmovnp, O_Gv, O_Ev, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0096 */ { UD_Icmovl, O_Gv, O_Ev, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0097 */ { UD_Icmovge, O_Gv, O_Ev, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0098 */ { UD_Icmovle, O_Gv, O_Ev, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0099 */ { UD_Icmovg, O_Gv, O_Ev, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0100 */ { UD_Icmp, O_Eb, O_Gb, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0101 */ { UD_Icmp, O_Ev, O_Gv, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0102 */ { UD_Icmp, O_Gb, O_Eb, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0103 */ { UD_Icmp, O_Gv, O_Ev, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0104 */ { UD_Icmp, O_AL, O_Ib, O_NONE, O_NONE, P_none }, + /* 0105 */ { UD_Icmp, O_rAX, O_sIz, O_NONE, O_NONE, P_oso|P_rexw }, + /* 0106 */ { UD_Icmp, O_Eb, O_Ib, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0107 */ { UD_Icmp, O_Eb, O_Ib, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_inv64 }, + /* 0108 */ { UD_Icmp, O_Ev, O_sIz, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0109 */ { UD_Icmp, O_Ev, O_sIb, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0110 */ { UD_Icmppd, O_V, O_W, O_Ib, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0111 */ { UD_Ivcmppd, O_Vx, O_Hx, O_Wx, O_Ib, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 0112 */ { UD_Icmpps, O_V, O_W, O_Ib, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0113 */ { UD_Ivcmpps, O_Vx, O_Hx, O_Wx, O_Ib, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 0114 */ { UD_Icmpsb, O_NONE, O_NONE, O_NONE, O_NONE, P_strz|P_seg }, + /* 0115 */ { UD_Icmpsw, O_NONE, O_NONE, O_NONE, O_NONE, P_strz|P_oso|P_rexw|P_seg }, + /* 0116 */ { UD_Icmpsd, O_NONE, O_NONE, O_NONE, O_NONE, P_strz|P_oso|P_rexw|P_seg }, + /* 0117 */ { UD_Icmpsd, O_V, O_W, O_Ib, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0118 */ { UD_Ivcmpsd, O_Vx, O_Hx, O_Wx, O_Ib, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0119 */ { UD_Icmpsq, O_NONE, O_NONE, O_NONE, O_NONE, P_strz|P_oso|P_rexw|P_seg }, + /* 0120 */ { UD_Icmpss, O_V, O_W, O_Ib, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0121 */ { UD_Ivcmpss, O_Vx, O_Hx, O_Wx, O_Ib, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0122 */ { UD_Icmpxchg, O_Eb, O_Gb, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0123 */ { UD_Icmpxchg, O_Ev, O_Gv, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0124 */ { UD_Icmpxchg8b, O_M, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0125 */ { UD_Icmpxchg8b, O_M, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0126 */ { UD_Icmpxchg16b, O_M, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0127 */ { UD_Icomisd, O_Vsd, O_Wsd, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0128 */ { UD_Ivcomisd, O_Vsd, O_Wsd, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0129 */ { UD_Icomiss, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0130 */ { UD_Ivcomiss, O_Vx, O_Wx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0131 */ { UD_Icpuid, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0132 */ { UD_Icvtdq2pd, O_V, O_Wdq, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0133 */ { UD_Ivcvtdq2pd, O_Vx, O_Wdq, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 0134 */ { UD_Icvtdq2ps, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0135 */ { UD_Ivcvtdq2ps, O_Vx, O_Wx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0136 */ { UD_Icvtpd2dq, O_Vdq, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0137 */ { UD_Ivcvtpd2dq, O_Vdq, O_Wx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 0138 */ { UD_Icvtpd2pi, O_P, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0139 */ { UD_Icvtpd2ps, O_Vdq, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0140 */ { UD_Ivcvtpd2ps, O_Vdq, O_Wx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 0141 */ { UD_Icvtpi2ps, O_V, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0142 */ { UD_Icvtpi2pd, O_V, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0143 */ { UD_Icvtps2dq, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0144 */ { UD_Ivcvtps2dq, O_Vx, O_Wx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 0145 */ { UD_Icvtps2pd, O_V, O_Wdq, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0146 */ { UD_Ivcvtps2pd, O_Vx, O_Wdq, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 0147 */ { UD_Icvtps2pi, O_P, O_MqU, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0148 */ { UD_Icvtsd2si, O_Gy, O_MqU, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0149 */ { UD_Ivcvtsd2si, O_Gy, O_MqU, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0150 */ { UD_Icvtsd2ss, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0151 */ { UD_Ivcvtsd2ss, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0152 */ { UD_Icvtsi2sd, O_V, O_Ey, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0153 */ { UD_Ivcvtsi2sd, O_Vx, O_Hx, O_Ey, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0154 */ { UD_Icvtsi2ss, O_V, O_Ey, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0155 */ { UD_Ivcvtsi2ss, O_Vx, O_Hx, O_Ey, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0156 */ { UD_Icvtss2sd, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0157 */ { UD_Ivcvtss2sd, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0158 */ { UD_Icvtss2si, O_Gy, O_MdU, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0159 */ { UD_Ivcvtss2si, O_Gy, O_MdU, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0160 */ { UD_Icvttpd2dq, O_Vdq, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0161 */ { UD_Ivcvttpd2dq, O_Vdq, O_Wx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 0162 */ { UD_Icvttpd2pi, O_P, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0163 */ { UD_Icvttps2dq, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0164 */ { UD_Ivcvttps2dq, O_Vx, O_Wx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 0165 */ { UD_Icvttps2pi, O_P, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0166 */ { UD_Icvttsd2si, O_Gy, O_MqU, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0167 */ { UD_Ivcvttsd2si, O_Gy, O_MqU, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0168 */ { UD_Icvttss2si, O_Gy, O_MdU, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0169 */ { UD_Ivcvttss2si, O_Gy, O_MdU, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0170 */ { UD_Icwd, O_NONE, O_NONE, O_NONE, O_NONE, P_oso|P_rexw }, + /* 0171 */ { UD_Icdq, O_NONE, O_NONE, O_NONE, O_NONE, P_oso|P_rexw }, + /* 0172 */ { UD_Icqo, O_NONE, O_NONE, O_NONE, O_NONE, P_oso|P_rexw }, + /* 0173 */ { UD_Idaa, O_NONE, O_NONE, O_NONE, O_NONE, P_inv64 }, + /* 0174 */ { UD_Idas, O_NONE, O_NONE, O_NONE, O_NONE, P_inv64 }, + /* 0175 */ { UD_Idec, O_R0z, O_NONE, O_NONE, O_NONE, P_oso }, + /* 0176 */ { UD_Idec, O_R1z, O_NONE, O_NONE, O_NONE, P_oso }, + /* 0177 */ { UD_Idec, O_R2z, O_NONE, O_NONE, O_NONE, P_oso }, + /* 0178 */ { UD_Idec, O_R3z, O_NONE, O_NONE, O_NONE, P_oso }, + /* 0179 */ { UD_Idec, O_R4z, O_NONE, O_NONE, O_NONE, P_oso }, + /* 0180 */ { UD_Idec, O_R5z, O_NONE, O_NONE, O_NONE, P_oso }, + /* 0181 */ { UD_Idec, O_R6z, O_NONE, O_NONE, O_NONE, P_oso }, + /* 0182 */ { UD_Idec, O_R7z, O_NONE, O_NONE, O_NONE, P_oso }, + /* 0183 */ { UD_Idec, O_Eb, O_NONE, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0184 */ { UD_Idec, O_Ev, O_NONE, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0185 */ { UD_Idiv, O_Ev, O_NONE, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0186 */ { UD_Idiv, O_Eb, O_NONE, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0187 */ { UD_Idivpd, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0188 */ { UD_Ivdivpd, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 0189 */ { UD_Idivps, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0190 */ { UD_Ivdivps, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0191 */ { UD_Idivsd, O_V, O_MqU, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0192 */ { UD_Ivdivsd, O_Vx, O_Hx, O_MqU, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0193 */ { UD_Idivss, O_V, O_MdU, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0194 */ { UD_Ivdivss, O_Vx, O_Hx, O_MdU, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0195 */ { UD_Idppd, O_V, O_W, O_Ib, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0196 */ { UD_Ivdppd, O_Vx, O_Hx, O_Wx, O_Ib, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0197 */ { UD_Idpps, O_V, O_W, O_Ib, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0198 */ { UD_Ivdpps, O_Vx, O_Hx, O_Wx, O_Ib, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 0199 */ { UD_Iemms, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0200 */ { UD_Ienter, O_Iw, O_Ib, O_NONE, O_NONE, P_def64 }, + /* 0201 */ { UD_Iextractps, O_MdRy, O_V, O_Ib, O_NONE, P_aso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 0202 */ { UD_Ivextractps, O_MdRy, O_Vx, O_Ib, O_NONE, P_aso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 0203 */ { UD_If2xm1, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0204 */ { UD_Ifabs, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0205 */ { UD_Ifadd, O_Mq, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0206 */ { UD_Ifadd, O_Md, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0207 */ { UD_Ifadd, O_ST0, O_ST0, O_NONE, O_NONE, P_none }, + /* 0208 */ { UD_Ifadd, O_ST1, O_ST0, O_NONE, O_NONE, P_none }, + /* 0209 */ { UD_Ifadd, O_ST2, O_ST0, O_NONE, O_NONE, P_none }, + /* 0210 */ { UD_Ifadd, O_ST3, O_ST0, O_NONE, O_NONE, P_none }, + /* 0211 */ { UD_Ifadd, O_ST4, O_ST0, O_NONE, O_NONE, P_none }, + /* 0212 */ { UD_Ifadd, O_ST5, O_ST0, O_NONE, O_NONE, P_none }, + /* 0213 */ { UD_Ifadd, O_ST6, O_ST0, O_NONE, O_NONE, P_none }, + /* 0214 */ { UD_Ifadd, O_ST7, O_ST0, O_NONE, O_NONE, P_none }, + /* 0215 */ { UD_Ifadd, O_ST0, O_ST0, O_NONE, O_NONE, P_none }, + /* 0216 */ { UD_Ifadd, O_ST0, O_ST1, O_NONE, O_NONE, P_none }, + /* 0217 */ { UD_Ifadd, O_ST0, O_ST2, O_NONE, O_NONE, P_none }, + /* 0218 */ { UD_Ifadd, O_ST0, O_ST3, O_NONE, O_NONE, P_none }, + /* 0219 */ { UD_Ifadd, O_ST0, O_ST4, O_NONE, O_NONE, P_none }, + /* 0220 */ { UD_Ifadd, O_ST0, O_ST5, O_NONE, O_NONE, P_none }, + /* 0221 */ { UD_Ifadd, O_ST0, O_ST6, O_NONE, O_NONE, P_none }, + /* 0222 */ { UD_Ifadd, O_ST0, O_ST7, O_NONE, O_NONE, P_none }, + /* 0223 */ { UD_Ifaddp, O_ST0, O_ST0, O_NONE, O_NONE, P_none }, + /* 0224 */ { UD_Ifaddp, O_ST1, O_ST0, O_NONE, O_NONE, P_none }, + /* 0225 */ { UD_Ifaddp, O_ST2, O_ST0, O_NONE, O_NONE, P_none }, + /* 0226 */ { UD_Ifaddp, O_ST3, O_ST0, O_NONE, O_NONE, P_none }, + /* 0227 */ { UD_Ifaddp, O_ST4, O_ST0, O_NONE, O_NONE, P_none }, + /* 0228 */ { UD_Ifaddp, O_ST5, O_ST0, O_NONE, O_NONE, P_none }, + /* 0229 */ { UD_Ifaddp, O_ST6, O_ST0, O_NONE, O_NONE, P_none }, + /* 0230 */ { UD_Ifaddp, O_ST7, O_ST0, O_NONE, O_NONE, P_none }, + /* 0231 */ { UD_Ifbld, O_Mt, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0232 */ { UD_Ifbstp, O_Mt, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0233 */ { UD_Ifchs, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0234 */ { UD_Ifclex, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0235 */ { UD_Ifcmovb, O_ST0, O_ST0, O_NONE, O_NONE, P_none }, + /* 0236 */ { UD_Ifcmovb, O_ST0, O_ST1, O_NONE, O_NONE, P_none }, + /* 0237 */ { UD_Ifcmovb, O_ST0, O_ST2, O_NONE, O_NONE, P_none }, + /* 0238 */ { UD_Ifcmovb, O_ST0, O_ST3, O_NONE, O_NONE, P_none }, + /* 0239 */ { UD_Ifcmovb, O_ST0, O_ST4, O_NONE, O_NONE, P_none }, + /* 0240 */ { UD_Ifcmovb, O_ST0, O_ST5, O_NONE, O_NONE, P_none }, + /* 0241 */ { UD_Ifcmovb, O_ST0, O_ST6, O_NONE, O_NONE, P_none }, + /* 0242 */ { UD_Ifcmovb, O_ST0, O_ST7, O_NONE, O_NONE, P_none }, + /* 0243 */ { UD_Ifcmove, O_ST0, O_ST0, O_NONE, O_NONE, P_none }, + /* 0244 */ { UD_Ifcmove, O_ST0, O_ST1, O_NONE, O_NONE, P_none }, + /* 0245 */ { UD_Ifcmove, O_ST0, O_ST2, O_NONE, O_NONE, P_none }, + /* 0246 */ { UD_Ifcmove, O_ST0, O_ST3, O_NONE, O_NONE, P_none }, + /* 0247 */ { UD_Ifcmove, O_ST0, O_ST4, O_NONE, O_NONE, P_none }, + /* 0248 */ { UD_Ifcmove, O_ST0, O_ST5, O_NONE, O_NONE, P_none }, + /* 0249 */ { UD_Ifcmove, O_ST0, O_ST6, O_NONE, O_NONE, P_none }, + /* 0250 */ { UD_Ifcmove, O_ST0, O_ST7, O_NONE, O_NONE, P_none }, + /* 0251 */ { UD_Ifcmovbe, O_ST0, O_ST0, O_NONE, O_NONE, P_none }, + /* 0252 */ { UD_Ifcmovbe, O_ST0, O_ST1, O_NONE, O_NONE, P_none }, + /* 0253 */ { UD_Ifcmovbe, O_ST0, O_ST2, O_NONE, O_NONE, P_none }, + /* 0254 */ { UD_Ifcmovbe, O_ST0, O_ST3, O_NONE, O_NONE, P_none }, + /* 0255 */ { UD_Ifcmovbe, O_ST0, O_ST4, O_NONE, O_NONE, P_none }, + /* 0256 */ { UD_Ifcmovbe, O_ST0, O_ST5, O_NONE, O_NONE, P_none }, + /* 0257 */ { UD_Ifcmovbe, O_ST0, O_ST6, O_NONE, O_NONE, P_none }, + /* 0258 */ { UD_Ifcmovbe, O_ST0, O_ST7, O_NONE, O_NONE, P_none }, + /* 0259 */ { UD_Ifcmovu, O_ST0, O_ST0, O_NONE, O_NONE, P_none }, + /* 0260 */ { UD_Ifcmovu, O_ST0, O_ST1, O_NONE, O_NONE, P_none }, + /* 0261 */ { UD_Ifcmovu, O_ST0, O_ST2, O_NONE, O_NONE, P_none }, + /* 0262 */ { UD_Ifcmovu, O_ST0, O_ST3, O_NONE, O_NONE, P_none }, + /* 0263 */ { UD_Ifcmovu, O_ST0, O_ST4, O_NONE, O_NONE, P_none }, + /* 0264 */ { UD_Ifcmovu, O_ST0, O_ST5, O_NONE, O_NONE, P_none }, + /* 0265 */ { UD_Ifcmovu, O_ST0, O_ST6, O_NONE, O_NONE, P_none }, + /* 0266 */ { UD_Ifcmovu, O_ST0, O_ST7, O_NONE, O_NONE, P_none }, + /* 0267 */ { UD_Ifcmovnb, O_ST0, O_ST0, O_NONE, O_NONE, P_none }, + /* 0268 */ { UD_Ifcmovnb, O_ST0, O_ST1, O_NONE, O_NONE, P_none }, + /* 0269 */ { UD_Ifcmovnb, O_ST0, O_ST2, O_NONE, O_NONE, P_none }, + /* 0270 */ { UD_Ifcmovnb, O_ST0, O_ST3, O_NONE, O_NONE, P_none }, + /* 0271 */ { UD_Ifcmovnb, O_ST0, O_ST4, O_NONE, O_NONE, P_none }, + /* 0272 */ { UD_Ifcmovnb, O_ST0, O_ST5, O_NONE, O_NONE, P_none }, + /* 0273 */ { UD_Ifcmovnb, O_ST0, O_ST6, O_NONE, O_NONE, P_none }, + /* 0274 */ { UD_Ifcmovnb, O_ST0, O_ST7, O_NONE, O_NONE, P_none }, + /* 0275 */ { UD_Ifcmovne, O_ST0, O_ST0, O_NONE, O_NONE, P_none }, + /* 0276 */ { UD_Ifcmovne, O_ST0, O_ST1, O_NONE, O_NONE, P_none }, + /* 0277 */ { UD_Ifcmovne, O_ST0, O_ST2, O_NONE, O_NONE, P_none }, + /* 0278 */ { UD_Ifcmovne, O_ST0, O_ST3, O_NONE, O_NONE, P_none }, + /* 0279 */ { UD_Ifcmovne, O_ST0, O_ST4, O_NONE, O_NONE, P_none }, + /* 0280 */ { UD_Ifcmovne, O_ST0, O_ST5, O_NONE, O_NONE, P_none }, + /* 0281 */ { UD_Ifcmovne, O_ST0, O_ST6, O_NONE, O_NONE, P_none }, + /* 0282 */ { UD_Ifcmovne, O_ST0, O_ST7, O_NONE, O_NONE, P_none }, + /* 0283 */ { UD_Ifcmovnbe, O_ST0, O_ST0, O_NONE, O_NONE, P_none }, + /* 0284 */ { UD_Ifcmovnbe, O_ST0, O_ST1, O_NONE, O_NONE, P_none }, + /* 0285 */ { UD_Ifcmovnbe, O_ST0, O_ST2, O_NONE, O_NONE, P_none }, + /* 0286 */ { UD_Ifcmovnbe, O_ST0, O_ST3, O_NONE, O_NONE, P_none }, + /* 0287 */ { UD_Ifcmovnbe, O_ST0, O_ST4, O_NONE, O_NONE, P_none }, + /* 0288 */ { UD_Ifcmovnbe, O_ST0, O_ST5, O_NONE, O_NONE, P_none }, + /* 0289 */ { UD_Ifcmovnbe, O_ST0, O_ST6, O_NONE, O_NONE, P_none }, + /* 0290 */ { UD_Ifcmovnbe, O_ST0, O_ST7, O_NONE, O_NONE, P_none }, + /* 0291 */ { UD_Ifcmovnu, O_ST0, O_ST0, O_NONE, O_NONE, P_none }, + /* 0292 */ { UD_Ifcmovnu, O_ST0, O_ST1, O_NONE, O_NONE, P_none }, + /* 0293 */ { UD_Ifcmovnu, O_ST0, O_ST2, O_NONE, O_NONE, P_none }, + /* 0294 */ { UD_Ifcmovnu, O_ST0, O_ST3, O_NONE, O_NONE, P_none }, + /* 0295 */ { UD_Ifcmovnu, O_ST0, O_ST4, O_NONE, O_NONE, P_none }, + /* 0296 */ { UD_Ifcmovnu, O_ST0, O_ST5, O_NONE, O_NONE, P_none }, + /* 0297 */ { UD_Ifcmovnu, O_ST0, O_ST6, O_NONE, O_NONE, P_none }, + /* 0298 */ { UD_Ifcmovnu, O_ST0, O_ST7, O_NONE, O_NONE, P_none }, + /* 0299 */ { UD_Ifucomi, O_ST0, O_ST0, O_NONE, O_NONE, P_none }, + /* 0300 */ { UD_Ifucomi, O_ST0, O_ST1, O_NONE, O_NONE, P_none }, + /* 0301 */ { UD_Ifucomi, O_ST0, O_ST2, O_NONE, O_NONE, P_none }, + /* 0302 */ { UD_Ifucomi, O_ST0, O_ST3, O_NONE, O_NONE, P_none }, + /* 0303 */ { UD_Ifucomi, O_ST0, O_ST4, O_NONE, O_NONE, P_none }, + /* 0304 */ { UD_Ifucomi, O_ST0, O_ST5, O_NONE, O_NONE, P_none }, + /* 0305 */ { UD_Ifucomi, O_ST0, O_ST6, O_NONE, O_NONE, P_none }, + /* 0306 */ { UD_Ifucomi, O_ST0, O_ST7, O_NONE, O_NONE, P_none }, + /* 0307 */ { UD_Ifcom, O_Md, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0308 */ { UD_Ifcom, O_Mq, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0309 */ { UD_Ifcom, O_ST0, O_ST0, O_NONE, O_NONE, P_none }, + /* 0310 */ { UD_Ifcom, O_ST0, O_ST1, O_NONE, O_NONE, P_none }, + /* 0311 */ { UD_Ifcom, O_ST0, O_ST2, O_NONE, O_NONE, P_none }, + /* 0312 */ { UD_Ifcom, O_ST0, O_ST3, O_NONE, O_NONE, P_none }, + /* 0313 */ { UD_Ifcom, O_ST0, O_ST4, O_NONE, O_NONE, P_none }, + /* 0314 */ { UD_Ifcom, O_ST0, O_ST5, O_NONE, O_NONE, P_none }, + /* 0315 */ { UD_Ifcom, O_ST0, O_ST6, O_NONE, O_NONE, P_none }, + /* 0316 */ { UD_Ifcom, O_ST0, O_ST7, O_NONE, O_NONE, P_none }, + /* 0317 */ { UD_Ifcom2, O_ST0, O_NONE, O_NONE, O_NONE, P_none }, + /* 0318 */ { UD_Ifcom2, O_ST1, O_NONE, O_NONE, O_NONE, P_none }, + /* 0319 */ { UD_Ifcom2, O_ST2, O_NONE, O_NONE, O_NONE, P_none }, + /* 0320 */ { UD_Ifcom2, O_ST3, O_NONE, O_NONE, O_NONE, P_none }, + /* 0321 */ { UD_Ifcom2, O_ST4, O_NONE, O_NONE, O_NONE, P_none }, + /* 0322 */ { UD_Ifcom2, O_ST5, O_NONE, O_NONE, O_NONE, P_none }, + /* 0323 */ { UD_Ifcom2, O_ST6, O_NONE, O_NONE, O_NONE, P_none }, + /* 0324 */ { UD_Ifcom2, O_ST7, O_NONE, O_NONE, O_NONE, P_none }, + /* 0325 */ { UD_Ifcomp3, O_ST0, O_NONE, O_NONE, O_NONE, P_none }, + /* 0326 */ { UD_Ifcomp3, O_ST1, O_NONE, O_NONE, O_NONE, P_none }, + /* 0327 */ { UD_Ifcomp3, O_ST2, O_NONE, O_NONE, O_NONE, P_none }, + /* 0328 */ { UD_Ifcomp3, O_ST3, O_NONE, O_NONE, O_NONE, P_none }, + /* 0329 */ { UD_Ifcomp3, O_ST4, O_NONE, O_NONE, O_NONE, P_none }, + /* 0330 */ { UD_Ifcomp3, O_ST5, O_NONE, O_NONE, O_NONE, P_none }, + /* 0331 */ { UD_Ifcomp3, O_ST6, O_NONE, O_NONE, O_NONE, P_none }, + /* 0332 */ { UD_Ifcomp3, O_ST7, O_NONE, O_NONE, O_NONE, P_none }, + /* 0333 */ { UD_Ifcomi, O_ST0, O_ST0, O_NONE, O_NONE, P_none }, + /* 0334 */ { UD_Ifcomi, O_ST0, O_ST1, O_NONE, O_NONE, P_none }, + /* 0335 */ { UD_Ifcomi, O_ST0, O_ST2, O_NONE, O_NONE, P_none }, + /* 0336 */ { UD_Ifcomi, O_ST0, O_ST3, O_NONE, O_NONE, P_none }, + /* 0337 */ { UD_Ifcomi, O_ST0, O_ST4, O_NONE, O_NONE, P_none }, + /* 0338 */ { UD_Ifcomi, O_ST0, O_ST5, O_NONE, O_NONE, P_none }, + /* 0339 */ { UD_Ifcomi, O_ST0, O_ST6, O_NONE, O_NONE, P_none }, + /* 0340 */ { UD_Ifcomi, O_ST0, O_ST7, O_NONE, O_NONE, P_none }, + /* 0341 */ { UD_Ifucomip, O_ST0, O_ST0, O_NONE, O_NONE, P_none }, + /* 0342 */ { UD_Ifucomip, O_ST0, O_ST1, O_NONE, O_NONE, P_none }, + /* 0343 */ { UD_Ifucomip, O_ST0, O_ST2, O_NONE, O_NONE, P_none }, + /* 0344 */ { UD_Ifucomip, O_ST0, O_ST3, O_NONE, O_NONE, P_none }, + /* 0345 */ { UD_Ifucomip, O_ST0, O_ST4, O_NONE, O_NONE, P_none }, + /* 0346 */ { UD_Ifucomip, O_ST0, O_ST5, O_NONE, O_NONE, P_none }, + /* 0347 */ { UD_Ifucomip, O_ST0, O_ST6, O_NONE, O_NONE, P_none }, + /* 0348 */ { UD_Ifucomip, O_ST0, O_ST7, O_NONE, O_NONE, P_none }, + /* 0349 */ { UD_Ifcomip, O_ST0, O_ST0, O_NONE, O_NONE, P_none }, + /* 0350 */ { UD_Ifcomip, O_ST0, O_ST1, O_NONE, O_NONE, P_none }, + /* 0351 */ { UD_Ifcomip, O_ST0, O_ST2, O_NONE, O_NONE, P_none }, + /* 0352 */ { UD_Ifcomip, O_ST0, O_ST3, O_NONE, O_NONE, P_none }, + /* 0353 */ { UD_Ifcomip, O_ST0, O_ST4, O_NONE, O_NONE, P_none }, + /* 0354 */ { UD_Ifcomip, O_ST0, O_ST5, O_NONE, O_NONE, P_none }, + /* 0355 */ { UD_Ifcomip, O_ST0, O_ST6, O_NONE, O_NONE, P_none }, + /* 0356 */ { UD_Ifcomip, O_ST0, O_ST7, O_NONE, O_NONE, P_none }, + /* 0357 */ { UD_Ifcomp, O_Md, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0358 */ { UD_Ifcomp, O_Mq, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0359 */ { UD_Ifcomp, O_ST0, O_ST0, O_NONE, O_NONE, P_none }, + /* 0360 */ { UD_Ifcomp, O_ST0, O_ST1, O_NONE, O_NONE, P_none }, + /* 0361 */ { UD_Ifcomp, O_ST0, O_ST2, O_NONE, O_NONE, P_none }, + /* 0362 */ { UD_Ifcomp, O_ST0, O_ST3, O_NONE, O_NONE, P_none }, + /* 0363 */ { UD_Ifcomp, O_ST0, O_ST4, O_NONE, O_NONE, P_none }, + /* 0364 */ { UD_Ifcomp, O_ST0, O_ST5, O_NONE, O_NONE, P_none }, + /* 0365 */ { UD_Ifcomp, O_ST0, O_ST6, O_NONE, O_NONE, P_none }, + /* 0366 */ { UD_Ifcomp, O_ST0, O_ST7, O_NONE, O_NONE, P_none }, + /* 0367 */ { UD_Ifcomp5, O_ST0, O_NONE, O_NONE, O_NONE, P_none }, + /* 0368 */ { UD_Ifcomp5, O_ST1, O_NONE, O_NONE, O_NONE, P_none }, + /* 0369 */ { UD_Ifcomp5, O_ST2, O_NONE, O_NONE, O_NONE, P_none }, + /* 0370 */ { UD_Ifcomp5, O_ST3, O_NONE, O_NONE, O_NONE, P_none }, + /* 0371 */ { UD_Ifcomp5, O_ST4, O_NONE, O_NONE, O_NONE, P_none }, + /* 0372 */ { UD_Ifcomp5, O_ST5, O_NONE, O_NONE, O_NONE, P_none }, + /* 0373 */ { UD_Ifcomp5, O_ST6, O_NONE, O_NONE, O_NONE, P_none }, + /* 0374 */ { UD_Ifcomp5, O_ST7, O_NONE, O_NONE, O_NONE, P_none }, + /* 0375 */ { UD_Ifcompp, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0376 */ { UD_Ifcos, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0377 */ { UD_Ifdecstp, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0378 */ { UD_Ifdiv, O_Mq, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0379 */ { UD_Ifdiv, O_ST0, O_ST0, O_NONE, O_NONE, P_none }, + /* 0380 */ { UD_Ifdiv, O_ST1, O_ST0, O_NONE, O_NONE, P_none }, + /* 0381 */ { UD_Ifdiv, O_ST2, O_ST0, O_NONE, O_NONE, P_none }, + /* 0382 */ { UD_Ifdiv, O_ST3, O_ST0, O_NONE, O_NONE, P_none }, + /* 0383 */ { UD_Ifdiv, O_ST4, O_ST0, O_NONE, O_NONE, P_none }, + /* 0384 */ { UD_Ifdiv, O_ST5, O_ST0, O_NONE, O_NONE, P_none }, + /* 0385 */ { UD_Ifdiv, O_ST6, O_ST0, O_NONE, O_NONE, P_none }, + /* 0386 */ { UD_Ifdiv, O_ST7, O_ST0, O_NONE, O_NONE, P_none }, + /* 0387 */ { UD_Ifdiv, O_Md, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0388 */ { UD_Ifdiv, O_ST0, O_ST0, O_NONE, O_NONE, P_none }, + /* 0389 */ { UD_Ifdiv, O_ST0, O_ST1, O_NONE, O_NONE, P_none }, + /* 0390 */ { UD_Ifdiv, O_ST0, O_ST2, O_NONE, O_NONE, P_none }, + /* 0391 */ { UD_Ifdiv, O_ST0, O_ST3, O_NONE, O_NONE, P_none }, + /* 0392 */ { UD_Ifdiv, O_ST0, O_ST4, O_NONE, O_NONE, P_none }, + /* 0393 */ { UD_Ifdiv, O_ST0, O_ST5, O_NONE, O_NONE, P_none }, + /* 0394 */ { UD_Ifdiv, O_ST0, O_ST6, O_NONE, O_NONE, P_none }, + /* 0395 */ { UD_Ifdiv, O_ST0, O_ST7, O_NONE, O_NONE, P_none }, + /* 0396 */ { UD_Ifdivp, O_ST0, O_ST0, O_NONE, O_NONE, P_none }, + /* 0397 */ { UD_Ifdivp, O_ST1, O_ST0, O_NONE, O_NONE, P_none }, + /* 0398 */ { UD_Ifdivp, O_ST2, O_ST0, O_NONE, O_NONE, P_none }, + /* 0399 */ { UD_Ifdivp, O_ST3, O_ST0, O_NONE, O_NONE, P_none }, + /* 0400 */ { UD_Ifdivp, O_ST4, O_ST0, O_NONE, O_NONE, P_none }, + /* 0401 */ { UD_Ifdivp, O_ST5, O_ST0, O_NONE, O_NONE, P_none }, + /* 0402 */ { UD_Ifdivp, O_ST6, O_ST0, O_NONE, O_NONE, P_none }, + /* 0403 */ { UD_Ifdivp, O_ST7, O_ST0, O_NONE, O_NONE, P_none }, + /* 0404 */ { UD_Ifdivr, O_Mq, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0405 */ { UD_Ifdivr, O_ST0, O_ST0, O_NONE, O_NONE, P_none }, + /* 0406 */ { UD_Ifdivr, O_ST1, O_ST0, O_NONE, O_NONE, P_none }, + /* 0407 */ { UD_Ifdivr, O_ST2, O_ST0, O_NONE, O_NONE, P_none }, + /* 0408 */ { UD_Ifdivr, O_ST3, O_ST0, O_NONE, O_NONE, P_none }, + /* 0409 */ { UD_Ifdivr, O_ST4, O_ST0, O_NONE, O_NONE, P_none }, + /* 0410 */ { UD_Ifdivr, O_ST5, O_ST0, O_NONE, O_NONE, P_none }, + /* 0411 */ { UD_Ifdivr, O_ST6, O_ST0, O_NONE, O_NONE, P_none }, + /* 0412 */ { UD_Ifdivr, O_ST7, O_ST0, O_NONE, O_NONE, P_none }, + /* 0413 */ { UD_Ifdivr, O_Md, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0414 */ { UD_Ifdivr, O_ST0, O_ST0, O_NONE, O_NONE, P_none }, + /* 0415 */ { UD_Ifdivr, O_ST0, O_ST1, O_NONE, O_NONE, P_none }, + /* 0416 */ { UD_Ifdivr, O_ST0, O_ST2, O_NONE, O_NONE, P_none }, + /* 0417 */ { UD_Ifdivr, O_ST0, O_ST3, O_NONE, O_NONE, P_none }, + /* 0418 */ { UD_Ifdivr, O_ST0, O_ST4, O_NONE, O_NONE, P_none }, + /* 0419 */ { UD_Ifdivr, O_ST0, O_ST5, O_NONE, O_NONE, P_none }, + /* 0420 */ { UD_Ifdivr, O_ST0, O_ST6, O_NONE, O_NONE, P_none }, + /* 0421 */ { UD_Ifdivr, O_ST0, O_ST7, O_NONE, O_NONE, P_none }, + /* 0422 */ { UD_Ifdivrp, O_ST0, O_ST0, O_NONE, O_NONE, P_none }, + /* 0423 */ { UD_Ifdivrp, O_ST1, O_ST0, O_NONE, O_NONE, P_none }, + /* 0424 */ { UD_Ifdivrp, O_ST2, O_ST0, O_NONE, O_NONE, P_none }, + /* 0425 */ { UD_Ifdivrp, O_ST3, O_ST0, O_NONE, O_NONE, P_none }, + /* 0426 */ { UD_Ifdivrp, O_ST4, O_ST0, O_NONE, O_NONE, P_none }, + /* 0427 */ { UD_Ifdivrp, O_ST5, O_ST0, O_NONE, O_NONE, P_none }, + /* 0428 */ { UD_Ifdivrp, O_ST6, O_ST0, O_NONE, O_NONE, P_none }, + /* 0429 */ { UD_Ifdivrp, O_ST7, O_ST0, O_NONE, O_NONE, P_none }, + /* 0430 */ { UD_Ifemms, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0431 */ { UD_Iffree, O_ST0, O_NONE, O_NONE, O_NONE, P_none }, + /* 0432 */ { UD_Iffree, O_ST1, O_NONE, O_NONE, O_NONE, P_none }, + /* 0433 */ { UD_Iffree, O_ST2, O_NONE, O_NONE, O_NONE, P_none }, + /* 0434 */ { UD_Iffree, O_ST3, O_NONE, O_NONE, O_NONE, P_none }, + /* 0435 */ { UD_Iffree, O_ST4, O_NONE, O_NONE, O_NONE, P_none }, + /* 0436 */ { UD_Iffree, O_ST5, O_NONE, O_NONE, O_NONE, P_none }, + /* 0437 */ { UD_Iffree, O_ST6, O_NONE, O_NONE, O_NONE, P_none }, + /* 0438 */ { UD_Iffree, O_ST7, O_NONE, O_NONE, O_NONE, P_none }, + /* 0439 */ { UD_Iffreep, O_ST0, O_NONE, O_NONE, O_NONE, P_none }, + /* 0440 */ { UD_Iffreep, O_ST1, O_NONE, O_NONE, O_NONE, P_none }, + /* 0441 */ { UD_Iffreep, O_ST2, O_NONE, O_NONE, O_NONE, P_none }, + /* 0442 */ { UD_Iffreep, O_ST3, O_NONE, O_NONE, O_NONE, P_none }, + /* 0443 */ { UD_Iffreep, O_ST4, O_NONE, O_NONE, O_NONE, P_none }, + /* 0444 */ { UD_Iffreep, O_ST5, O_NONE, O_NONE, O_NONE, P_none }, + /* 0445 */ { UD_Iffreep, O_ST6, O_NONE, O_NONE, O_NONE, P_none }, + /* 0446 */ { UD_Iffreep, O_ST7, O_NONE, O_NONE, O_NONE, P_none }, + /* 0447 */ { UD_Ificom, O_Mw, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0448 */ { UD_Ificom, O_Md, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0449 */ { UD_Ificomp, O_Mw, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0450 */ { UD_Ificomp, O_Md, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0451 */ { UD_Ifild, O_Mw, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0452 */ { UD_Ifild, O_Mq, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0453 */ { UD_Ifild, O_Md, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0454 */ { UD_Ifincstp, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0455 */ { UD_Ifninit, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0456 */ { UD_Ifiadd, O_Md, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0457 */ { UD_Ifiadd, O_Mw, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0458 */ { UD_Ifidivr, O_Md, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0459 */ { UD_Ifidivr, O_Mw, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0460 */ { UD_Ifidiv, O_Md, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0461 */ { UD_Ifidiv, O_Mw, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0462 */ { UD_Ifisub, O_Md, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0463 */ { UD_Ifisub, O_Mw, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0464 */ { UD_Ifisubr, O_Md, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0465 */ { UD_Ifisubr, O_Mw, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0466 */ { UD_Ifist, O_Mw, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0467 */ { UD_Ifist, O_Md, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0468 */ { UD_Ifistp, O_Mw, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0469 */ { UD_Ifistp, O_Mq, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0470 */ { UD_Ifistp, O_Md, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0471 */ { UD_Ifisttp, O_Md, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0472 */ { UD_Ifisttp, O_Mq, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0473 */ { UD_Ifisttp, O_Mw, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0474 */ { UD_Ifld, O_Mt, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0475 */ { UD_Ifld, O_Mq, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0476 */ { UD_Ifld, O_Md, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0477 */ { UD_Ifld, O_ST0, O_NONE, O_NONE, O_NONE, P_none }, + /* 0478 */ { UD_Ifld, O_ST1, O_NONE, O_NONE, O_NONE, P_none }, + /* 0479 */ { UD_Ifld, O_ST2, O_NONE, O_NONE, O_NONE, P_none }, + /* 0480 */ { UD_Ifld, O_ST3, O_NONE, O_NONE, O_NONE, P_none }, + /* 0481 */ { UD_Ifld, O_ST4, O_NONE, O_NONE, O_NONE, P_none }, + /* 0482 */ { UD_Ifld, O_ST5, O_NONE, O_NONE, O_NONE, P_none }, + /* 0483 */ { UD_Ifld, O_ST6, O_NONE, O_NONE, O_NONE, P_none }, + /* 0484 */ { UD_Ifld, O_ST7, O_NONE, O_NONE, O_NONE, P_none }, + /* 0485 */ { UD_Ifld1, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0486 */ { UD_Ifldl2t, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0487 */ { UD_Ifldl2e, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0488 */ { UD_Ifldpi, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0489 */ { UD_Ifldlg2, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0490 */ { UD_Ifldln2, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0491 */ { UD_Ifldz, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0492 */ { UD_Ifldcw, O_Mw, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0493 */ { UD_Ifldenv, O_M, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0494 */ { UD_Ifmul, O_Mq, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0495 */ { UD_Ifmul, O_ST0, O_ST0, O_NONE, O_NONE, P_none }, + /* 0496 */ { UD_Ifmul, O_ST1, O_ST0, O_NONE, O_NONE, P_none }, + /* 0497 */ { UD_Ifmul, O_ST2, O_ST0, O_NONE, O_NONE, P_none }, + /* 0498 */ { UD_Ifmul, O_ST3, O_ST0, O_NONE, O_NONE, P_none }, + /* 0499 */ { UD_Ifmul, O_ST4, O_ST0, O_NONE, O_NONE, P_none }, + /* 0500 */ { UD_Ifmul, O_ST5, O_ST0, O_NONE, O_NONE, P_none }, + /* 0501 */ { UD_Ifmul, O_ST6, O_ST0, O_NONE, O_NONE, P_none }, + /* 0502 */ { UD_Ifmul, O_ST7, O_ST0, O_NONE, O_NONE, P_none }, + /* 0503 */ { UD_Ifmul, O_Md, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0504 */ { UD_Ifmul, O_ST0, O_ST0, O_NONE, O_NONE, P_none }, + /* 0505 */ { UD_Ifmul, O_ST0, O_ST1, O_NONE, O_NONE, P_none }, + /* 0506 */ { UD_Ifmul, O_ST0, O_ST2, O_NONE, O_NONE, P_none }, + /* 0507 */ { UD_Ifmul, O_ST0, O_ST3, O_NONE, O_NONE, P_none }, + /* 0508 */ { UD_Ifmul, O_ST0, O_ST4, O_NONE, O_NONE, P_none }, + /* 0509 */ { UD_Ifmul, O_ST0, O_ST5, O_NONE, O_NONE, P_none }, + /* 0510 */ { UD_Ifmul, O_ST0, O_ST6, O_NONE, O_NONE, P_none }, + /* 0511 */ { UD_Ifmul, O_ST0, O_ST7, O_NONE, O_NONE, P_none }, + /* 0512 */ { UD_Ifmulp, O_ST0, O_ST0, O_NONE, O_NONE, P_none }, + /* 0513 */ { UD_Ifmulp, O_ST1, O_ST0, O_NONE, O_NONE, P_none }, + /* 0514 */ { UD_Ifmulp, O_ST2, O_ST0, O_NONE, O_NONE, P_none }, + /* 0515 */ { UD_Ifmulp, O_ST3, O_ST0, O_NONE, O_NONE, P_none }, + /* 0516 */ { UD_Ifmulp, O_ST4, O_ST0, O_NONE, O_NONE, P_none }, + /* 0517 */ { UD_Ifmulp, O_ST5, O_ST0, O_NONE, O_NONE, P_none }, + /* 0518 */ { UD_Ifmulp, O_ST6, O_ST0, O_NONE, O_NONE, P_none }, + /* 0519 */ { UD_Ifmulp, O_ST7, O_ST0, O_NONE, O_NONE, P_none }, + /* 0520 */ { UD_Ifimul, O_Md, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0521 */ { UD_Ifimul, O_Mw, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0522 */ { UD_Ifnop, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0523 */ { UD_Ifpatan, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0524 */ { UD_Ifprem, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0525 */ { UD_Ifprem1, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0526 */ { UD_Ifptan, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0527 */ { UD_Ifrndint, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0528 */ { UD_Ifrstor, O_M, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0529 */ { UD_Ifnsave, O_M, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0530 */ { UD_Ifscale, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0531 */ { UD_Ifsin, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0532 */ { UD_Ifsincos, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0533 */ { UD_Ifsqrt, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0534 */ { UD_Ifstp, O_Mt, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0535 */ { UD_Ifstp, O_Mq, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0536 */ { UD_Ifstp, O_Md, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0537 */ { UD_Ifstp, O_ST0, O_NONE, O_NONE, O_NONE, P_none }, + /* 0538 */ { UD_Ifstp, O_ST1, O_NONE, O_NONE, O_NONE, P_none }, + /* 0539 */ { UD_Ifstp, O_ST2, O_NONE, O_NONE, O_NONE, P_none }, + /* 0540 */ { UD_Ifstp, O_ST3, O_NONE, O_NONE, O_NONE, P_none }, + /* 0541 */ { UD_Ifstp, O_ST4, O_NONE, O_NONE, O_NONE, P_none }, + /* 0542 */ { UD_Ifstp, O_ST5, O_NONE, O_NONE, O_NONE, P_none }, + /* 0543 */ { UD_Ifstp, O_ST6, O_NONE, O_NONE, O_NONE, P_none }, + /* 0544 */ { UD_Ifstp, O_ST7, O_NONE, O_NONE, O_NONE, P_none }, + /* 0545 */ { UD_Ifstp1, O_ST0, O_NONE, O_NONE, O_NONE, P_none }, + /* 0546 */ { UD_Ifstp1, O_ST1, O_NONE, O_NONE, O_NONE, P_none }, + /* 0547 */ { UD_Ifstp1, O_ST2, O_NONE, O_NONE, O_NONE, P_none }, + /* 0548 */ { UD_Ifstp1, O_ST3, O_NONE, O_NONE, O_NONE, P_none }, + /* 0549 */ { UD_Ifstp1, O_ST4, O_NONE, O_NONE, O_NONE, P_none }, + /* 0550 */ { UD_Ifstp1, O_ST5, O_NONE, O_NONE, O_NONE, P_none }, + /* 0551 */ { UD_Ifstp1, O_ST6, O_NONE, O_NONE, O_NONE, P_none }, + /* 0552 */ { UD_Ifstp1, O_ST7, O_NONE, O_NONE, O_NONE, P_none }, + /* 0553 */ { UD_Ifstp8, O_ST0, O_NONE, O_NONE, O_NONE, P_none }, + /* 0554 */ { UD_Ifstp8, O_ST1, O_NONE, O_NONE, O_NONE, P_none }, + /* 0555 */ { UD_Ifstp8, O_ST2, O_NONE, O_NONE, O_NONE, P_none }, + /* 0556 */ { UD_Ifstp8, O_ST3, O_NONE, O_NONE, O_NONE, P_none }, + /* 0557 */ { UD_Ifstp8, O_ST4, O_NONE, O_NONE, O_NONE, P_none }, + /* 0558 */ { UD_Ifstp8, O_ST5, O_NONE, O_NONE, O_NONE, P_none }, + /* 0559 */ { UD_Ifstp8, O_ST6, O_NONE, O_NONE, O_NONE, P_none }, + /* 0560 */ { UD_Ifstp8, O_ST7, O_NONE, O_NONE, O_NONE, P_none }, + /* 0561 */ { UD_Ifstp9, O_ST0, O_NONE, O_NONE, O_NONE, P_none }, + /* 0562 */ { UD_Ifstp9, O_ST1, O_NONE, O_NONE, O_NONE, P_none }, + /* 0563 */ { UD_Ifstp9, O_ST2, O_NONE, O_NONE, O_NONE, P_none }, + /* 0564 */ { UD_Ifstp9, O_ST3, O_NONE, O_NONE, O_NONE, P_none }, + /* 0565 */ { UD_Ifstp9, O_ST4, O_NONE, O_NONE, O_NONE, P_none }, + /* 0566 */ { UD_Ifstp9, O_ST5, O_NONE, O_NONE, O_NONE, P_none }, + /* 0567 */ { UD_Ifstp9, O_ST6, O_NONE, O_NONE, O_NONE, P_none }, + /* 0568 */ { UD_Ifstp9, O_ST7, O_NONE, O_NONE, O_NONE, P_none }, + /* 0569 */ { UD_Ifst, O_Md, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0570 */ { UD_Ifst, O_Mq, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0571 */ { UD_Ifst, O_ST0, O_NONE, O_NONE, O_NONE, P_none }, + /* 0572 */ { UD_Ifst, O_ST1, O_NONE, O_NONE, O_NONE, P_none }, + /* 0573 */ { UD_Ifst, O_ST2, O_NONE, O_NONE, O_NONE, P_none }, + /* 0574 */ { UD_Ifst, O_ST3, O_NONE, O_NONE, O_NONE, P_none }, + /* 0575 */ { UD_Ifst, O_ST4, O_NONE, O_NONE, O_NONE, P_none }, + /* 0576 */ { UD_Ifst, O_ST5, O_NONE, O_NONE, O_NONE, P_none }, + /* 0577 */ { UD_Ifst, O_ST6, O_NONE, O_NONE, O_NONE, P_none }, + /* 0578 */ { UD_Ifst, O_ST7, O_NONE, O_NONE, O_NONE, P_none }, + /* 0579 */ { UD_Ifnstcw, O_Mw, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0580 */ { UD_Ifnstenv, O_M, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0581 */ { UD_Ifnstsw, O_Mw, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0582 */ { UD_Ifnstsw, O_AX, O_NONE, O_NONE, O_NONE, P_none }, + /* 0583 */ { UD_Ifsub, O_Md, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0584 */ { UD_Ifsub, O_Mq, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0585 */ { UD_Ifsub, O_ST0, O_ST0, O_NONE, O_NONE, P_none }, + /* 0586 */ { UD_Ifsub, O_ST0, O_ST1, O_NONE, O_NONE, P_none }, + /* 0587 */ { UD_Ifsub, O_ST0, O_ST2, O_NONE, O_NONE, P_none }, + /* 0588 */ { UD_Ifsub, O_ST0, O_ST3, O_NONE, O_NONE, P_none }, + /* 0589 */ { UD_Ifsub, O_ST0, O_ST4, O_NONE, O_NONE, P_none }, + /* 0590 */ { UD_Ifsub, O_ST0, O_ST5, O_NONE, O_NONE, P_none }, + /* 0591 */ { UD_Ifsub, O_ST0, O_ST6, O_NONE, O_NONE, P_none }, + /* 0592 */ { UD_Ifsub, O_ST0, O_ST7, O_NONE, O_NONE, P_none }, + /* 0593 */ { UD_Ifsub, O_ST0, O_ST0, O_NONE, O_NONE, P_none }, + /* 0594 */ { UD_Ifsub, O_ST1, O_ST0, O_NONE, O_NONE, P_none }, + /* 0595 */ { UD_Ifsub, O_ST2, O_ST0, O_NONE, O_NONE, P_none }, + /* 0596 */ { UD_Ifsub, O_ST3, O_ST0, O_NONE, O_NONE, P_none }, + /* 0597 */ { UD_Ifsub, O_ST4, O_ST0, O_NONE, O_NONE, P_none }, + /* 0598 */ { UD_Ifsub, O_ST5, O_ST0, O_NONE, O_NONE, P_none }, + /* 0599 */ { UD_Ifsub, O_ST6, O_ST0, O_NONE, O_NONE, P_none }, + /* 0600 */ { UD_Ifsub, O_ST7, O_ST0, O_NONE, O_NONE, P_none }, + /* 0601 */ { UD_Ifsubp, O_ST0, O_ST0, O_NONE, O_NONE, P_none }, + /* 0602 */ { UD_Ifsubp, O_ST1, O_ST0, O_NONE, O_NONE, P_none }, + /* 0603 */ { UD_Ifsubp, O_ST2, O_ST0, O_NONE, O_NONE, P_none }, + /* 0604 */ { UD_Ifsubp, O_ST3, O_ST0, O_NONE, O_NONE, P_none }, + /* 0605 */ { UD_Ifsubp, O_ST4, O_ST0, O_NONE, O_NONE, P_none }, + /* 0606 */ { UD_Ifsubp, O_ST5, O_ST0, O_NONE, O_NONE, P_none }, + /* 0607 */ { UD_Ifsubp, O_ST6, O_ST0, O_NONE, O_NONE, P_none }, + /* 0608 */ { UD_Ifsubp, O_ST7, O_ST0, O_NONE, O_NONE, P_none }, + /* 0609 */ { UD_Ifsubr, O_Mq, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0610 */ { UD_Ifsubr, O_ST0, O_ST0, O_NONE, O_NONE, P_none }, + /* 0611 */ { UD_Ifsubr, O_ST0, O_ST1, O_NONE, O_NONE, P_none }, + /* 0612 */ { UD_Ifsubr, O_ST0, O_ST2, O_NONE, O_NONE, P_none }, + /* 0613 */ { UD_Ifsubr, O_ST0, O_ST3, O_NONE, O_NONE, P_none }, + /* 0614 */ { UD_Ifsubr, O_ST0, O_ST4, O_NONE, O_NONE, P_none }, + /* 0615 */ { UD_Ifsubr, O_ST0, O_ST5, O_NONE, O_NONE, P_none }, + /* 0616 */ { UD_Ifsubr, O_ST0, O_ST6, O_NONE, O_NONE, P_none }, + /* 0617 */ { UD_Ifsubr, O_ST0, O_ST7, O_NONE, O_NONE, P_none }, + /* 0618 */ { UD_Ifsubr, O_ST0, O_ST0, O_NONE, O_NONE, P_none }, + /* 0619 */ { UD_Ifsubr, O_ST1, O_ST0, O_NONE, O_NONE, P_none }, + /* 0620 */ { UD_Ifsubr, O_ST2, O_ST0, O_NONE, O_NONE, P_none }, + /* 0621 */ { UD_Ifsubr, O_ST3, O_ST0, O_NONE, O_NONE, P_none }, + /* 0622 */ { UD_Ifsubr, O_ST4, O_ST0, O_NONE, O_NONE, P_none }, + /* 0623 */ { UD_Ifsubr, O_ST5, O_ST0, O_NONE, O_NONE, P_none }, + /* 0624 */ { UD_Ifsubr, O_ST6, O_ST0, O_NONE, O_NONE, P_none }, + /* 0625 */ { UD_Ifsubr, O_ST7, O_ST0, O_NONE, O_NONE, P_none }, + /* 0626 */ { UD_Ifsubr, O_Md, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0627 */ { UD_Ifsubrp, O_ST0, O_ST0, O_NONE, O_NONE, P_none }, + /* 0628 */ { UD_Ifsubrp, O_ST1, O_ST0, O_NONE, O_NONE, P_none }, + /* 0629 */ { UD_Ifsubrp, O_ST2, O_ST0, O_NONE, O_NONE, P_none }, + /* 0630 */ { UD_Ifsubrp, O_ST3, O_ST0, O_NONE, O_NONE, P_none }, + /* 0631 */ { UD_Ifsubrp, O_ST4, O_ST0, O_NONE, O_NONE, P_none }, + /* 0632 */ { UD_Ifsubrp, O_ST5, O_ST0, O_NONE, O_NONE, P_none }, + /* 0633 */ { UD_Ifsubrp, O_ST6, O_ST0, O_NONE, O_NONE, P_none }, + /* 0634 */ { UD_Ifsubrp, O_ST7, O_ST0, O_NONE, O_NONE, P_none }, + /* 0635 */ { UD_Iftst, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0636 */ { UD_Ifucom, O_ST0, O_NONE, O_NONE, O_NONE, P_none }, + /* 0637 */ { UD_Ifucom, O_ST1, O_NONE, O_NONE, O_NONE, P_none }, + /* 0638 */ { UD_Ifucom, O_ST2, O_NONE, O_NONE, O_NONE, P_none }, + /* 0639 */ { UD_Ifucom, O_ST3, O_NONE, O_NONE, O_NONE, P_none }, + /* 0640 */ { UD_Ifucom, O_ST4, O_NONE, O_NONE, O_NONE, P_none }, + /* 0641 */ { UD_Ifucom, O_ST5, O_NONE, O_NONE, O_NONE, P_none }, + /* 0642 */ { UD_Ifucom, O_ST6, O_NONE, O_NONE, O_NONE, P_none }, + /* 0643 */ { UD_Ifucom, O_ST7, O_NONE, O_NONE, O_NONE, P_none }, + /* 0644 */ { UD_Ifucomp, O_ST0, O_NONE, O_NONE, O_NONE, P_none }, + /* 0645 */ { UD_Ifucomp, O_ST1, O_NONE, O_NONE, O_NONE, P_none }, + /* 0646 */ { UD_Ifucomp, O_ST2, O_NONE, O_NONE, O_NONE, P_none }, + /* 0647 */ { UD_Ifucomp, O_ST3, O_NONE, O_NONE, O_NONE, P_none }, + /* 0648 */ { UD_Ifucomp, O_ST4, O_NONE, O_NONE, O_NONE, P_none }, + /* 0649 */ { UD_Ifucomp, O_ST5, O_NONE, O_NONE, O_NONE, P_none }, + /* 0650 */ { UD_Ifucomp, O_ST6, O_NONE, O_NONE, O_NONE, P_none }, + /* 0651 */ { UD_Ifucomp, O_ST7, O_NONE, O_NONE, O_NONE, P_none }, + /* 0652 */ { UD_Ifucompp, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0653 */ { UD_Ifxam, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0654 */ { UD_Ifxch, O_ST0, O_ST0, O_NONE, O_NONE, P_none }, + /* 0655 */ { UD_Ifxch, O_ST0, O_ST1, O_NONE, O_NONE, P_none }, + /* 0656 */ { UD_Ifxch, O_ST0, O_ST2, O_NONE, O_NONE, P_none }, + /* 0657 */ { UD_Ifxch, O_ST0, O_ST3, O_NONE, O_NONE, P_none }, + /* 0658 */ { UD_Ifxch, O_ST0, O_ST4, O_NONE, O_NONE, P_none }, + /* 0659 */ { UD_Ifxch, O_ST0, O_ST5, O_NONE, O_NONE, P_none }, + /* 0660 */ { UD_Ifxch, O_ST0, O_ST6, O_NONE, O_NONE, P_none }, + /* 0661 */ { UD_Ifxch, O_ST0, O_ST7, O_NONE, O_NONE, P_none }, + /* 0662 */ { UD_Ifxch4, O_ST0, O_NONE, O_NONE, O_NONE, P_none }, + /* 0663 */ { UD_Ifxch4, O_ST1, O_NONE, O_NONE, O_NONE, P_none }, + /* 0664 */ { UD_Ifxch4, O_ST2, O_NONE, O_NONE, O_NONE, P_none }, + /* 0665 */ { UD_Ifxch4, O_ST3, O_NONE, O_NONE, O_NONE, P_none }, + /* 0666 */ { UD_Ifxch4, O_ST4, O_NONE, O_NONE, O_NONE, P_none }, + /* 0667 */ { UD_Ifxch4, O_ST5, O_NONE, O_NONE, O_NONE, P_none }, + /* 0668 */ { UD_Ifxch4, O_ST6, O_NONE, O_NONE, O_NONE, P_none }, + /* 0669 */ { UD_Ifxch4, O_ST7, O_NONE, O_NONE, O_NONE, P_none }, + /* 0670 */ { UD_Ifxch7, O_ST0, O_NONE, O_NONE, O_NONE, P_none }, + /* 0671 */ { UD_Ifxch7, O_ST1, O_NONE, O_NONE, O_NONE, P_none }, + /* 0672 */ { UD_Ifxch7, O_ST2, O_NONE, O_NONE, O_NONE, P_none }, + /* 0673 */ { UD_Ifxch7, O_ST3, O_NONE, O_NONE, O_NONE, P_none }, + /* 0674 */ { UD_Ifxch7, O_ST4, O_NONE, O_NONE, O_NONE, P_none }, + /* 0675 */ { UD_Ifxch7, O_ST5, O_NONE, O_NONE, O_NONE, P_none }, + /* 0676 */ { UD_Ifxch7, O_ST6, O_NONE, O_NONE, O_NONE, P_none }, + /* 0677 */ { UD_Ifxch7, O_ST7, O_NONE, O_NONE, O_NONE, P_none }, + /* 0678 */ { UD_Ifxrstor, O_M, O_NONE, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0679 */ { UD_Ifxsave, O_M, O_NONE, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0680 */ { UD_Ifxtract, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0681 */ { UD_Ifyl2x, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0682 */ { UD_Ifyl2xp1, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0683 */ { UD_Ihlt, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0684 */ { UD_Iidiv, O_Ev, O_NONE, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0685 */ { UD_Iidiv, O_Eb, O_NONE, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0686 */ { UD_Iin, O_AL, O_Ib, O_NONE, O_NONE, P_none }, + /* 0687 */ { UD_Iin, O_eAX, O_Ib, O_NONE, O_NONE, P_oso }, + /* 0688 */ { UD_Iin, O_AL, O_DX, O_NONE, O_NONE, P_none }, + /* 0689 */ { UD_Iin, O_eAX, O_DX, O_NONE, O_NONE, P_oso }, + /* 0690 */ { UD_Iimul, O_Gv, O_Ev, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0691 */ { UD_Iimul, O_Eb, O_NONE, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0692 */ { UD_Iimul, O_Ev, O_NONE, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0693 */ { UD_Iimul, O_Gv, O_Ev, O_Iz, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0694 */ { UD_Iimul, O_Gv, O_Ev, O_sIb, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0695 */ { UD_Iinc, O_R0z, O_NONE, O_NONE, O_NONE, P_oso }, + /* 0696 */ { UD_Iinc, O_R1z, O_NONE, O_NONE, O_NONE, P_oso }, + /* 0697 */ { UD_Iinc, O_R2z, O_NONE, O_NONE, O_NONE, P_oso }, + /* 0698 */ { UD_Iinc, O_R3z, O_NONE, O_NONE, O_NONE, P_oso }, + /* 0699 */ { UD_Iinc, O_R4z, O_NONE, O_NONE, O_NONE, P_oso }, + /* 0700 */ { UD_Iinc, O_R5z, O_NONE, O_NONE, O_NONE, P_oso }, + /* 0701 */ { UD_Iinc, O_R6z, O_NONE, O_NONE, O_NONE, P_oso }, + /* 0702 */ { UD_Iinc, O_R7z, O_NONE, O_NONE, O_NONE, P_oso }, + /* 0703 */ { UD_Iinc, O_Ev, O_NONE, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0704 */ { UD_Iinc, O_Eb, O_NONE, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0705 */ { UD_Iinsb, O_NONE, O_NONE, O_NONE, O_NONE, P_str|P_seg }, + /* 0706 */ { UD_Iinsw, O_NONE, O_NONE, O_NONE, O_NONE, P_str|P_oso|P_seg }, + /* 0707 */ { UD_Iinsd, O_NONE, O_NONE, O_NONE, O_NONE, P_str|P_oso|P_seg }, + /* 0708 */ { UD_Iint1, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0709 */ { UD_Iint3, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0710 */ { UD_Iint, O_Ib, O_NONE, O_NONE, O_NONE, P_none }, + /* 0711 */ { UD_Iinto, O_NONE, O_NONE, O_NONE, O_NONE, P_inv64 }, + /* 0712 */ { UD_Iinvd, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0713 */ { UD_Iinvept, O_Gd, O_Mo, O_NONE, O_NONE, P_none }, + /* 0714 */ { UD_Iinvept, O_Gq, O_Mo, O_NONE, O_NONE, P_none }, + /* 0715 */ { UD_Iinvlpg, O_M, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0716 */ { UD_Iinvlpga, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0717 */ { UD_Iinvvpid, O_Gd, O_Mo, O_NONE, O_NONE, P_none }, + /* 0718 */ { UD_Iinvvpid, O_Gq, O_Mo, O_NONE, O_NONE, P_none }, + /* 0719 */ { UD_Iiretw, O_NONE, O_NONE, O_NONE, O_NONE, P_oso|P_rexw }, + /* 0720 */ { UD_Iiretd, O_NONE, O_NONE, O_NONE, O_NONE, P_oso|P_rexw }, + /* 0721 */ { UD_Iiretq, O_NONE, O_NONE, O_NONE, O_NONE, P_oso|P_rexw }, + /* 0722 */ { UD_Ijo, O_Jb, O_NONE, O_NONE, O_NONE, P_none }, + /* 0723 */ { UD_Ijo, O_Jz, O_NONE, O_NONE, O_NONE, P_oso|P_def64 }, + /* 0724 */ { UD_Ijno, O_Jb, O_NONE, O_NONE, O_NONE, P_none }, + /* 0725 */ { UD_Ijno, O_Jz, O_NONE, O_NONE, O_NONE, P_oso|P_def64 }, + /* 0726 */ { UD_Ijb, O_Jb, O_NONE, O_NONE, O_NONE, P_none }, + /* 0727 */ { UD_Ijb, O_Jz, O_NONE, O_NONE, O_NONE, P_oso|P_def64 }, + /* 0728 */ { UD_Ijae, O_Jb, O_NONE, O_NONE, O_NONE, P_none }, + /* 0729 */ { UD_Ijae, O_Jz, O_NONE, O_NONE, O_NONE, P_oso|P_def64 }, + /* 0730 */ { UD_Ijz, O_Jb, O_NONE, O_NONE, O_NONE, P_none }, + /* 0731 */ { UD_Ijz, O_Jz, O_NONE, O_NONE, O_NONE, P_oso|P_def64 }, + /* 0732 */ { UD_Ijnz, O_Jb, O_NONE, O_NONE, O_NONE, P_none }, + /* 0733 */ { UD_Ijnz, O_Jz, O_NONE, O_NONE, O_NONE, P_oso|P_def64 }, + /* 0734 */ { UD_Ijbe, O_Jb, O_NONE, O_NONE, O_NONE, P_none }, + /* 0735 */ { UD_Ijbe, O_Jz, O_NONE, O_NONE, O_NONE, P_oso|P_def64 }, + /* 0736 */ { UD_Ija, O_Jb, O_NONE, O_NONE, O_NONE, P_none }, + /* 0737 */ { UD_Ija, O_Jz, O_NONE, O_NONE, O_NONE, P_oso|P_def64 }, + /* 0738 */ { UD_Ijs, O_Jb, O_NONE, O_NONE, O_NONE, P_none }, + /* 0739 */ { UD_Ijs, O_Jz, O_NONE, O_NONE, O_NONE, P_oso|P_def64 }, + /* 0740 */ { UD_Ijns, O_Jb, O_NONE, O_NONE, O_NONE, P_none }, + /* 0741 */ { UD_Ijns, O_Jz, O_NONE, O_NONE, O_NONE, P_oso|P_def64 }, + /* 0742 */ { UD_Ijp, O_Jb, O_NONE, O_NONE, O_NONE, P_none }, + /* 0743 */ { UD_Ijp, O_Jz, O_NONE, O_NONE, O_NONE, P_oso|P_def64 }, + /* 0744 */ { UD_Ijnp, O_Jb, O_NONE, O_NONE, O_NONE, P_none }, + /* 0745 */ { UD_Ijnp, O_Jz, O_NONE, O_NONE, O_NONE, P_oso|P_def64 }, + /* 0746 */ { UD_Ijl, O_Jb, O_NONE, O_NONE, O_NONE, P_none }, + /* 0747 */ { UD_Ijl, O_Jz, O_NONE, O_NONE, O_NONE, P_oso|P_def64 }, + /* 0748 */ { UD_Ijge, O_Jb, O_NONE, O_NONE, O_NONE, P_none }, + /* 0749 */ { UD_Ijge, O_Jz, O_NONE, O_NONE, O_NONE, P_oso|P_def64 }, + /* 0750 */ { UD_Ijle, O_Jb, O_NONE, O_NONE, O_NONE, P_none }, + /* 0751 */ { UD_Ijle, O_Jz, O_NONE, O_NONE, O_NONE, P_oso|P_def64 }, + /* 0752 */ { UD_Ijg, O_Jb, O_NONE, O_NONE, O_NONE, P_none }, + /* 0753 */ { UD_Ijg, O_Jz, O_NONE, O_NONE, O_NONE, P_oso|P_def64 }, + /* 0754 */ { UD_Ijcxz, O_Jb, O_NONE, O_NONE, O_NONE, P_aso }, + /* 0755 */ { UD_Ijecxz, O_Jb, O_NONE, O_NONE, O_NONE, P_aso }, + /* 0756 */ { UD_Ijrcxz, O_Jb, O_NONE, O_NONE, O_NONE, P_aso }, + /* 0757 */ { UD_Ijmp, O_Ev, O_NONE, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb|P_def64 }, + /* 0758 */ { UD_Ijmp, O_Fv, O_NONE, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0759 */ { UD_Ijmp, O_Jz, O_NONE, O_NONE, O_NONE, P_oso|P_def64 }, + /* 0760 */ { UD_Ijmp, O_Av, O_NONE, O_NONE, O_NONE, P_oso }, + /* 0761 */ { UD_Ijmp, O_Jb, O_NONE, O_NONE, O_NONE, P_def64 }, + /* 0762 */ { UD_Ilahf, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0763 */ { UD_Ilar, O_Gv, O_Ew, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0764 */ { UD_Ildmxcsr, O_Md, O_NONE, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0765 */ { UD_Ilds, O_Gv, O_M, O_NONE, O_NONE, P_aso|P_oso }, + /* 0766 */ { UD_Ilea, O_Gv, O_M, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0767 */ { UD_Iles, O_Gv, O_M, O_NONE, O_NONE, P_aso|P_oso }, + /* 0768 */ { UD_Ilfs, O_Gz, O_M, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0769 */ { UD_Ilgs, O_Gz, O_M, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0770 */ { UD_Ilidt, O_M, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0771 */ { UD_Ilss, O_Gv, O_M, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0772 */ { UD_Ileave, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0773 */ { UD_Ilfence, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0774 */ { UD_Ilfence, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0775 */ { UD_Ilfence, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0776 */ { UD_Ilfence, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0777 */ { UD_Ilfence, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0778 */ { UD_Ilfence, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0779 */ { UD_Ilfence, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0780 */ { UD_Ilfence, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0781 */ { UD_Ilgdt, O_M, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0782 */ { UD_Illdt, O_Ew, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0783 */ { UD_Ilmsw, O_Ew, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0784 */ { UD_Ilmsw, O_Ew, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0785 */ { UD_Ilock, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0786 */ { UD_Ilodsb, O_NONE, O_NONE, O_NONE, O_NONE, P_str|P_seg }, + /* 0787 */ { UD_Ilodsw, O_NONE, O_NONE, O_NONE, O_NONE, P_str|P_seg|P_oso|P_rexw }, + /* 0788 */ { UD_Ilodsd, O_NONE, O_NONE, O_NONE, O_NONE, P_str|P_seg|P_oso|P_rexw }, + /* 0789 */ { UD_Ilodsq, O_NONE, O_NONE, O_NONE, O_NONE, P_str|P_seg|P_oso|P_rexw }, + /* 0790 */ { UD_Iloopne, O_Jb, O_NONE, O_NONE, O_NONE, P_none }, + /* 0791 */ { UD_Iloope, O_Jb, O_NONE, O_NONE, O_NONE, P_none }, + /* 0792 */ { UD_Iloop, O_Jb, O_NONE, O_NONE, O_NONE, P_none }, + /* 0793 */ { UD_Ilsl, O_Gv, O_Ew, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0794 */ { UD_Iltr, O_Ew, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0795 */ { UD_Imaskmovq, O_P, O_N, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0796 */ { UD_Imaxpd, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0797 */ { UD_Ivmaxpd, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 0798 */ { UD_Imaxps, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0799 */ { UD_Ivmaxps, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 0800 */ { UD_Imaxsd, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0801 */ { UD_Ivmaxsd, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0802 */ { UD_Imaxss, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0803 */ { UD_Ivmaxss, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0804 */ { UD_Imfence, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0805 */ { UD_Imfence, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0806 */ { UD_Imfence, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0807 */ { UD_Imfence, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0808 */ { UD_Imfence, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0809 */ { UD_Imfence, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0810 */ { UD_Imfence, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0811 */ { UD_Imfence, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0812 */ { UD_Iminpd, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0813 */ { UD_Ivminpd, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 0814 */ { UD_Iminps, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0815 */ { UD_Ivminps, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0816 */ { UD_Iminsd, O_V, O_MqU, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0817 */ { UD_Ivminsd, O_Vx, O_Hx, O_MqU, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0818 */ { UD_Iminss, O_V, O_MdU, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0819 */ { UD_Ivminss, O_Vx, O_Hx, O_MdU, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0820 */ { UD_Imonitor, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0821 */ { UD_Imontmul, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0822 */ { UD_Imov, O_Eb, O_Ib, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0823 */ { UD_Imov, O_Ev, O_sIz, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0824 */ { UD_Imov, O_Eb, O_Gb, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0825 */ { UD_Imov, O_Ev, O_Gv, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0826 */ { UD_Imov, O_Gb, O_Eb, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0827 */ { UD_Imov, O_Gv, O_Ev, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0828 */ { UD_Imov, O_MwRv, O_S, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0829 */ { UD_Imov, O_S, O_MwRv, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0830 */ { UD_Imov, O_AL, O_Ob, O_NONE, O_NONE, P_none }, + /* 0831 */ { UD_Imov, O_rAX, O_Ov, O_NONE, O_NONE, P_aso|P_oso|P_rexw }, + /* 0832 */ { UD_Imov, O_Ob, O_AL, O_NONE, O_NONE, P_none }, + /* 0833 */ { UD_Imov, O_Ov, O_rAX, O_NONE, O_NONE, P_aso|P_oso|P_rexw }, + /* 0834 */ { UD_Imov, O_R0b, O_Ib, O_NONE, O_NONE, P_rexb }, + /* 0835 */ { UD_Imov, O_R1b, O_Ib, O_NONE, O_NONE, P_rexb }, + /* 0836 */ { UD_Imov, O_R2b, O_Ib, O_NONE, O_NONE, P_rexb }, + /* 0837 */ { UD_Imov, O_R3b, O_Ib, O_NONE, O_NONE, P_rexb }, + /* 0838 */ { UD_Imov, O_R4b, O_Ib, O_NONE, O_NONE, P_rexb }, + /* 0839 */ { UD_Imov, O_R5b, O_Ib, O_NONE, O_NONE, P_rexb }, + /* 0840 */ { UD_Imov, O_R6b, O_Ib, O_NONE, O_NONE, P_rexb }, + /* 0841 */ { UD_Imov, O_R7b, O_Ib, O_NONE, O_NONE, P_rexb }, + /* 0842 */ { UD_Imov, O_R0v, O_Iv, O_NONE, O_NONE, P_oso|P_rexw|P_rexb }, + /* 0843 */ { UD_Imov, O_R1v, O_Iv, O_NONE, O_NONE, P_oso|P_rexw|P_rexb }, + /* 0844 */ { UD_Imov, O_R2v, O_Iv, O_NONE, O_NONE, P_oso|P_rexw|P_rexb }, + /* 0845 */ { UD_Imov, O_R3v, O_Iv, O_NONE, O_NONE, P_oso|P_rexw|P_rexb }, + /* 0846 */ { UD_Imov, O_R4v, O_Iv, O_NONE, O_NONE, P_oso|P_rexw|P_rexb }, + /* 0847 */ { UD_Imov, O_R5v, O_Iv, O_NONE, O_NONE, P_oso|P_rexw|P_rexb }, + /* 0848 */ { UD_Imov, O_R6v, O_Iv, O_NONE, O_NONE, P_oso|P_rexw|P_rexb }, + /* 0849 */ { UD_Imov, O_R7v, O_Iv, O_NONE, O_NONE, P_oso|P_rexw|P_rexb }, + /* 0850 */ { UD_Imov, O_R, O_C, O_NONE, O_NONE, P_rexr|P_rexw|P_rexb }, + /* 0851 */ { UD_Imov, O_R, O_D, O_NONE, O_NONE, P_rexr|P_rexw|P_rexb }, + /* 0852 */ { UD_Imov, O_C, O_R, O_NONE, O_NONE, P_rexr|P_rexw|P_rexb }, + /* 0853 */ { UD_Imov, O_D, O_R, O_NONE, O_NONE, P_rexr|P_rexw|P_rexb }, + /* 0854 */ { UD_Imovapd, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0855 */ { UD_Ivmovapd, O_Vx, O_Wx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 0856 */ { UD_Imovapd, O_W, O_V, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0857 */ { UD_Ivmovapd, O_Wx, O_Vx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 0858 */ { UD_Imovaps, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0859 */ { UD_Ivmovaps, O_Vx, O_Wx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 0860 */ { UD_Imovaps, O_W, O_V, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0861 */ { UD_Ivmovaps, O_Wx, O_Vx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 0862 */ { UD_Imovd, O_P, O_Ey, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0863 */ { UD_Imovd, O_P, O_Ey, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0864 */ { UD_Imovd, O_V, O_Ey, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0865 */ { UD_Ivmovd, O_Vx, O_Ey, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0866 */ { UD_Imovd, O_V, O_Ey, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0867 */ { UD_Ivmovd, O_Vx, O_Ey, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0868 */ { UD_Imovd, O_Ey, O_P, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0869 */ { UD_Imovd, O_Ey, O_P, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0870 */ { UD_Imovd, O_Ey, O_V, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0871 */ { UD_Ivmovd, O_Ey, O_Vx, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0872 */ { UD_Imovd, O_Ey, O_V, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0873 */ { UD_Ivmovd, O_Ey, O_Vx, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0874 */ { UD_Imovhpd, O_V, O_M, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0875 */ { UD_Ivmovhpd, O_Vx, O_Hx, O_M, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0876 */ { UD_Imovhpd, O_M, O_V, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0877 */ { UD_Ivmovhpd, O_M, O_Vx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0878 */ { UD_Imovhps, O_V, O_M, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0879 */ { UD_Ivmovhps, O_Vx, O_Hx, O_M, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0880 */ { UD_Imovhps, O_M, O_V, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0881 */ { UD_Ivmovhps, O_M, O_Vx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0882 */ { UD_Imovlhps, O_V, O_U, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0883 */ { UD_Ivmovlhps, O_Vx, O_Hx, O_Ux, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0884 */ { UD_Imovlpd, O_V, O_M, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0885 */ { UD_Ivmovlpd, O_Vx, O_M, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0886 */ { UD_Imovlpd, O_M, O_V, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0887 */ { UD_Ivmovlpd, O_M, O_Vx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0888 */ { UD_Imovlps, O_V, O_M, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0889 */ { UD_Ivmovlps, O_Vx, O_M, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0890 */ { UD_Imovlps, O_M, O_V, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0891 */ { UD_Ivmovlps, O_M, O_Vx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0892 */ { UD_Imovhlps, O_V, O_U, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0893 */ { UD_Ivmovhlps, O_Vx, O_Ux, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0894 */ { UD_Imovmskpd, O_Gd, O_U, O_NONE, O_NONE, P_oso|P_rexr|P_rexb }, + /* 0895 */ { UD_Ivmovmskpd, O_Gd, O_Ux, O_NONE, O_NONE, P_oso|P_rexr|P_rexb|P_vexl }, + /* 0896 */ { UD_Imovmskps, O_Gd, O_U, O_NONE, O_NONE, P_oso|P_rexr|P_rexb }, + /* 0897 */ { UD_Ivmovmskps, O_Gd, O_Ux, O_NONE, O_NONE, P_oso|P_rexr|P_rexb }, + /* 0898 */ { UD_Imovntdq, O_M, O_V, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0899 */ { UD_Ivmovntdq, O_M, O_Vx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 0900 */ { UD_Imovnti, O_M, O_Gy, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0901 */ { UD_Imovntpd, O_M, O_V, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0902 */ { UD_Ivmovntpd, O_M, O_Vx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 0903 */ { UD_Imovntps, O_M, O_V, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0904 */ { UD_Ivmovntps, O_M, O_Vx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 0905 */ { UD_Imovntq, O_M, O_P, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0906 */ { UD_Imovq, O_P, O_Eq, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0907 */ { UD_Imovq, O_V, O_Eq, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0908 */ { UD_Ivmovq, O_Vx, O_Eq, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0909 */ { UD_Imovq, O_Eq, O_P, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0910 */ { UD_Imovq, O_Eq, O_V, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0911 */ { UD_Ivmovq, O_Eq, O_Vx, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0912 */ { UD_Imovq, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0913 */ { UD_Ivmovq, O_Vx, O_Wx, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0914 */ { UD_Imovq, O_W, O_V, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0915 */ { UD_Ivmovq, O_Wx, O_Vx, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0916 */ { UD_Imovq, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0917 */ { UD_Imovq, O_Q, O_P, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0918 */ { UD_Imovsb, O_NONE, O_NONE, O_NONE, O_NONE, P_str|P_seg }, + /* 0919 */ { UD_Imovsw, O_NONE, O_NONE, O_NONE, O_NONE, P_str|P_seg|P_oso|P_rexw }, + /* 0920 */ { UD_Imovsd, O_NONE, O_NONE, O_NONE, O_NONE, P_str|P_seg|P_oso|P_rexw }, + /* 0921 */ { UD_Imovsd, O_V, O_MqU, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0922 */ { UD_Imovsd, O_W, O_V, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0923 */ { UD_Imovsq, O_NONE, O_NONE, O_NONE, O_NONE, P_str|P_seg|P_oso|P_rexw }, + /* 0924 */ { UD_Imovss, O_V, O_MdU, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0925 */ { UD_Imovss, O_W, O_V, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0926 */ { UD_Imovsx, O_Gv, O_Eb, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0927 */ { UD_Imovsx, O_Gy, O_Ew, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0928 */ { UD_Imovupd, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0929 */ { UD_Ivmovupd, O_Vx, O_Wx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 0930 */ { UD_Imovupd, O_W, O_V, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0931 */ { UD_Ivmovupd, O_Wx, O_Vx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 0932 */ { UD_Imovups, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0933 */ { UD_Ivmovups, O_Vx, O_Wx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 0934 */ { UD_Imovups, O_W, O_V, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0935 */ { UD_Ivmovups, O_Wx, O_Vx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 0936 */ { UD_Imovzx, O_Gv, O_Eb, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0937 */ { UD_Imovzx, O_Gy, O_Ew, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0938 */ { UD_Imul, O_Eb, O_NONE, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0939 */ { UD_Imul, O_Ev, O_NONE, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0940 */ { UD_Imulpd, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0941 */ { UD_Ivmulpd, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 0942 */ { UD_Imulps, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0943 */ { UD_Ivmulps, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 0944 */ { UD_Imulsd, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0945 */ { UD_Ivmulsd, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0946 */ { UD_Imulss, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0947 */ { UD_Ivmulss, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0948 */ { UD_Imwait, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 0949 */ { UD_Ineg, O_Eb, O_NONE, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0950 */ { UD_Ineg, O_Ev, O_NONE, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0951 */ { UD_Inop, O_M, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0952 */ { UD_Inop, O_M, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0953 */ { UD_Inop, O_M, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0954 */ { UD_Inop, O_M, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0955 */ { UD_Inop, O_M, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0956 */ { UD_Inop, O_M, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0957 */ { UD_Inop, O_M, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0958 */ { UD_Inot, O_Eb, O_NONE, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0959 */ { UD_Inot, O_Ev, O_NONE, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0960 */ { UD_Ior, O_Eb, O_Gb, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0961 */ { UD_Ior, O_Ev, O_Gv, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0962 */ { UD_Ior, O_Gb, O_Eb, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0963 */ { UD_Ior, O_Gv, O_Ev, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0964 */ { UD_Ior, O_AL, O_Ib, O_NONE, O_NONE, P_none }, + /* 0965 */ { UD_Ior, O_rAX, O_sIz, O_NONE, O_NONE, P_oso|P_rexw }, + /* 0966 */ { UD_Ior, O_Eb, O_Ib, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0967 */ { UD_Ior, O_Ev, O_sIz, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0968 */ { UD_Ior, O_Eb, O_Ib, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0969 */ { UD_Ior, O_Ev, O_sIb, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 0970 */ { UD_Iorpd, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0971 */ { UD_Ivorpd, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 0972 */ { UD_Iorps, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0973 */ { UD_Ivorps, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 0974 */ { UD_Iout, O_Ib, O_AL, O_NONE, O_NONE, P_none }, + /* 0975 */ { UD_Iout, O_Ib, O_eAX, O_NONE, O_NONE, P_oso }, + /* 0976 */ { UD_Iout, O_DX, O_AL, O_NONE, O_NONE, P_none }, + /* 0977 */ { UD_Iout, O_DX, O_eAX, O_NONE, O_NONE, P_oso }, + /* 0978 */ { UD_Ioutsb, O_NONE, O_NONE, O_NONE, O_NONE, P_str|P_seg }, + /* 0979 */ { UD_Ioutsw, O_NONE, O_NONE, O_NONE, O_NONE, P_str|P_oso|P_seg }, + /* 0980 */ { UD_Ioutsd, O_NONE, O_NONE, O_NONE, O_NONE, P_str|P_oso|P_seg }, + /* 0981 */ { UD_Ipacksswb, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0982 */ { UD_Ivpacksswb, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 0983 */ { UD_Ipacksswb, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0984 */ { UD_Ipackssdw, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0985 */ { UD_Ivpackssdw, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 0986 */ { UD_Ipackssdw, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0987 */ { UD_Ipackuswb, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0988 */ { UD_Ivpackuswb, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 0989 */ { UD_Ipackuswb, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0990 */ { UD_Ipaddb, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0991 */ { UD_Ivpaddb, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 0992 */ { UD_Ipaddb, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0993 */ { UD_Ipaddw, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0994 */ { UD_Ipaddw, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0995 */ { UD_Ivpaddw, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 0996 */ { UD_Ipaddd, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0997 */ { UD_Ipaddd, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 0998 */ { UD_Ivpaddd, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 0999 */ { UD_Ipaddsb, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1000 */ { UD_Ipaddsb, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1001 */ { UD_Ivpaddsb, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1002 */ { UD_Ipaddsw, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1003 */ { UD_Ipaddsw, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1004 */ { UD_Ivpaddsw, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1005 */ { UD_Ipaddusb, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1006 */ { UD_Ipaddusb, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1007 */ { UD_Ivpaddusb, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1008 */ { UD_Ipaddusw, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1009 */ { UD_Ipaddusw, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1010 */ { UD_Ivpaddusw, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1011 */ { UD_Ipand, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1012 */ { UD_Ivpand, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1013 */ { UD_Ipand, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1014 */ { UD_Ipandn, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1015 */ { UD_Ivpandn, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1016 */ { UD_Ipandn, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1017 */ { UD_Ipavgb, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1018 */ { UD_Ivpavgb, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1019 */ { UD_Ipavgb, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1020 */ { UD_Ipavgw, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1021 */ { UD_Ivpavgw, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1022 */ { UD_Ipavgw, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1023 */ { UD_Ipcmpeqb, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1024 */ { UD_Ipcmpeqb, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1025 */ { UD_Ivpcmpeqb, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1026 */ { UD_Ipcmpeqw, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1027 */ { UD_Ipcmpeqw, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1028 */ { UD_Ivpcmpeqw, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1029 */ { UD_Ipcmpeqd, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1030 */ { UD_Ipcmpeqd, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1031 */ { UD_Ivpcmpeqd, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1032 */ { UD_Ipcmpgtb, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1033 */ { UD_Ivpcmpgtb, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1034 */ { UD_Ipcmpgtb, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1035 */ { UD_Ipcmpgtw, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1036 */ { UD_Ivpcmpgtw, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1037 */ { UD_Ipcmpgtw, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1038 */ { UD_Ipcmpgtd, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1039 */ { UD_Ivpcmpgtd, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1040 */ { UD_Ipcmpgtd, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1041 */ { UD_Ipextrb, O_MbRv, O_V, O_Ib, O_NONE, P_aso|P_rexx|P_rexr|P_rexb|P_def64 }, + /* 1042 */ { UD_Ivpextrb, O_MbRv, O_Vx, O_Ib, O_NONE, P_aso|P_rexx|P_rexr|P_rexb|P_def64 }, + /* 1043 */ { UD_Ipextrd, O_Ed, O_V, O_Ib, O_NONE, P_aso|P_rexr|P_rexx|P_rexw|P_rexb }, + /* 1044 */ { UD_Ivpextrd, O_Ed, O_Vx, O_Ib, O_NONE, P_aso|P_rexr|P_rexx|P_rexw|P_rexb }, + /* 1045 */ { UD_Ipextrd, O_Ed, O_V, O_Ib, O_NONE, P_aso|P_rexr|P_rexx|P_rexw|P_rexb }, + /* 1046 */ { UD_Ivpextrd, O_Ed, O_Vx, O_Ib, O_NONE, P_aso|P_rexr|P_rexx|P_rexw|P_rexb }, + /* 1047 */ { UD_Ipextrq, O_Eq, O_V, O_Ib, O_NONE, P_aso|P_rexr|P_rexw|P_rexb|P_def64 }, + /* 1048 */ { UD_Ivpextrq, O_Eq, O_Vx, O_Ib, O_NONE, P_aso|P_rexr|P_rexw|P_rexb|P_def64 }, + /* 1049 */ { UD_Ipextrw, O_Gd, O_U, O_Ib, O_NONE, P_aso|P_rexw|P_rexr|P_rexb }, + /* 1050 */ { UD_Ivpextrw, O_Gd, O_Ux, O_Ib, O_NONE, P_aso|P_rexw|P_rexr|P_rexb }, + /* 1051 */ { UD_Ipextrw, O_Gd, O_N, O_Ib, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1052 */ { UD_Ipextrw, O_MwRd, O_V, O_Ib, O_NONE, P_aso|P_rexw|P_rexx|P_rexr|P_rexb }, + /* 1053 */ { UD_Ivpextrw, O_MwRd, O_Vx, O_Ib, O_NONE, P_aso|P_rexw|P_rexx|P_rexr|P_rexb }, + /* 1054 */ { UD_Ipinsrb, O_V, O_MbRd, O_Ib, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1055 */ { UD_Ipinsrw, O_P, O_MwRy, O_Ib, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb|P_def64 }, + /* 1056 */ { UD_Ipinsrw, O_V, O_MwRy, O_Ib, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb|P_def64 }, + /* 1057 */ { UD_Ivpinsrw, O_Vx, O_MwRy, O_Ib, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb|P_def64 }, + /* 1058 */ { UD_Ipinsrd, O_V, O_Ed, O_Ib, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1059 */ { UD_Ipinsrd, O_V, O_Ed, O_Ib, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1060 */ { UD_Ipinsrq, O_V, O_Eq, O_Ib, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1061 */ { UD_Ivpinsrb, O_V, O_H, O_MbRd, O_Ib, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1062 */ { UD_Ivpinsrd, O_V, O_H, O_Ed, O_Ib, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1063 */ { UD_Ivpinsrd, O_V, O_H, O_Ed, O_Ib, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1064 */ { UD_Ivpinsrq, O_V, O_H, O_Eq, O_Ib, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1065 */ { UD_Ipmaddwd, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1066 */ { UD_Ipmaddwd, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1067 */ { UD_Ivpmaddwd, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1068 */ { UD_Ipmaxsw, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1069 */ { UD_Ivpmaxsw, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1070 */ { UD_Ipmaxsw, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1071 */ { UD_Ipmaxub, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1072 */ { UD_Ipmaxub, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1073 */ { UD_Ivpmaxub, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1074 */ { UD_Ipminsw, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1075 */ { UD_Ivpminsw, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1076 */ { UD_Ipminsw, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1077 */ { UD_Ipminub, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1078 */ { UD_Ivpminub, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1079 */ { UD_Ipminub, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1080 */ { UD_Ipmovmskb, O_Gd, O_U, O_NONE, O_NONE, P_oso|P_rexr|P_rexw|P_rexb }, + /* 1081 */ { UD_Ivpmovmskb, O_Gd, O_Ux, O_NONE, O_NONE, P_oso|P_rexr|P_rexw|P_rexb }, + /* 1082 */ { UD_Ipmovmskb, O_Gd, O_N, O_NONE, O_NONE, P_oso|P_rexr|P_rexw|P_rexb }, + /* 1083 */ { UD_Ipmulhuw, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1084 */ { UD_Ipmulhuw, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1085 */ { UD_Ivpmulhuw, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1086 */ { UD_Ipmulhw, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1087 */ { UD_Ivpmulhw, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1088 */ { UD_Ipmulhw, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1089 */ { UD_Ipmullw, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1090 */ { UD_Ipmullw, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1091 */ { UD_Ivpmullw, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1092 */ { UD_Ipop, O_ES, O_NONE, O_NONE, O_NONE, P_inv64 }, + /* 1093 */ { UD_Ipop, O_SS, O_NONE, O_NONE, O_NONE, P_inv64 }, + /* 1094 */ { UD_Ipop, O_DS, O_NONE, O_NONE, O_NONE, P_inv64 }, + /* 1095 */ { UD_Ipop, O_GS, O_NONE, O_NONE, O_NONE, P_none }, + /* 1096 */ { UD_Ipop, O_FS, O_NONE, O_NONE, O_NONE, P_none }, + /* 1097 */ { UD_Ipop, O_R0v, O_NONE, O_NONE, O_NONE, P_oso|P_rexb|P_def64 }, + /* 1098 */ { UD_Ipop, O_R1v, O_NONE, O_NONE, O_NONE, P_oso|P_rexb|P_def64 }, + /* 1099 */ { UD_Ipop, O_R2v, O_NONE, O_NONE, O_NONE, P_oso|P_rexb|P_def64 }, + /* 1100 */ { UD_Ipop, O_R3v, O_NONE, O_NONE, O_NONE, P_oso|P_rexb|P_def64 }, + /* 1101 */ { UD_Ipop, O_R4v, O_NONE, O_NONE, O_NONE, P_oso|P_rexb|P_def64 }, + /* 1102 */ { UD_Ipop, O_R5v, O_NONE, O_NONE, O_NONE, P_oso|P_rexb|P_def64 }, + /* 1103 */ { UD_Ipop, O_R6v, O_NONE, O_NONE, O_NONE, P_oso|P_rexb|P_def64 }, + /* 1104 */ { UD_Ipop, O_R7v, O_NONE, O_NONE, O_NONE, P_oso|P_rexb|P_def64 }, + /* 1105 */ { UD_Ipop, O_Ev, O_NONE, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb|P_def64 }, + /* 1106 */ { UD_Ipopa, O_NONE, O_NONE, O_NONE, O_NONE, P_oso|P_inv64 }, + /* 1107 */ { UD_Ipopad, O_NONE, O_NONE, O_NONE, O_NONE, P_oso|P_inv64 }, + /* 1108 */ { UD_Ipopfw, O_NONE, O_NONE, O_NONE, O_NONE, P_oso }, + /* 1109 */ { UD_Ipopfd, O_NONE, O_NONE, O_NONE, O_NONE, P_oso }, + /* 1110 */ { UD_Ipopfq, O_NONE, O_NONE, O_NONE, O_NONE, P_oso|P_def64 }, + /* 1111 */ { UD_Ipopfq, O_NONE, O_NONE, O_NONE, O_NONE, P_oso|P_def64 }, + /* 1112 */ { UD_Ipor, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1113 */ { UD_Ivpor, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1114 */ { UD_Ipor, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1115 */ { UD_Iprefetch, O_M, O_NONE, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1116 */ { UD_Iprefetch, O_M, O_NONE, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1117 */ { UD_Iprefetch, O_M, O_NONE, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1118 */ { UD_Iprefetch, O_M, O_NONE, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1119 */ { UD_Iprefetch, O_M, O_NONE, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1120 */ { UD_Iprefetch, O_M, O_NONE, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1121 */ { UD_Iprefetch, O_M, O_NONE, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1122 */ { UD_Iprefetch, O_M, O_NONE, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1123 */ { UD_Iprefetchnta, O_M, O_NONE, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1124 */ { UD_Iprefetcht0, O_M, O_NONE, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1125 */ { UD_Iprefetcht1, O_M, O_NONE, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1126 */ { UD_Iprefetcht2, O_M, O_NONE, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1127 */ { UD_Ipsadbw, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1128 */ { UD_Ivpsadbw, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1129 */ { UD_Ipsadbw, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1130 */ { UD_Ipshufw, O_P, O_Q, O_Ib, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1131 */ { UD_Ipsllw, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1132 */ { UD_Ipsllw, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1133 */ { UD_Ipsllw, O_U, O_Ib, O_NONE, O_NONE, P_rexb }, + /* 1134 */ { UD_Ipsllw, O_N, O_Ib, O_NONE, O_NONE, P_none }, + /* 1135 */ { UD_Ipslld, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1136 */ { UD_Ipslld, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1137 */ { UD_Ipslld, O_U, O_Ib, O_NONE, O_NONE, P_rexb }, + /* 1138 */ { UD_Ipslld, O_N, O_Ib, O_NONE, O_NONE, P_none }, + /* 1139 */ { UD_Ipsllq, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1140 */ { UD_Ipsllq, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1141 */ { UD_Ipsllq, O_U, O_Ib, O_NONE, O_NONE, P_rexb }, + /* 1142 */ { UD_Ipsllq, O_N, O_Ib, O_NONE, O_NONE, P_none }, + /* 1143 */ { UD_Ipsraw, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1144 */ { UD_Ipsraw, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1145 */ { UD_Ivpsraw, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1146 */ { UD_Ipsraw, O_U, O_Ib, O_NONE, O_NONE, P_rexb }, + /* 1147 */ { UD_Ivpsraw, O_Hx, O_Ux, O_Ib, O_NONE, P_rexb }, + /* 1148 */ { UD_Ipsraw, O_N, O_Ib, O_NONE, O_NONE, P_none }, + /* 1149 */ { UD_Ipsrad, O_N, O_Ib, O_NONE, O_NONE, P_none }, + /* 1150 */ { UD_Ipsrad, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1151 */ { UD_Ivpsrad, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1152 */ { UD_Ipsrad, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1153 */ { UD_Ipsrad, O_U, O_Ib, O_NONE, O_NONE, P_rexb }, + /* 1154 */ { UD_Ivpsrad, O_Hx, O_Ux, O_Ib, O_NONE, P_rexb }, + /* 1155 */ { UD_Ipsrlw, O_N, O_Ib, O_NONE, O_NONE, P_none }, + /* 1156 */ { UD_Ipsrlw, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1157 */ { UD_Ipsrlw, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1158 */ { UD_Ivpsrlw, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1159 */ { UD_Ipsrlw, O_U, O_Ib, O_NONE, O_NONE, P_rexb }, + /* 1160 */ { UD_Ivpsrlw, O_Hx, O_Ux, O_Ib, O_NONE, P_rexb }, + /* 1161 */ { UD_Ipsrld, O_N, O_Ib, O_NONE, O_NONE, P_none }, + /* 1162 */ { UD_Ipsrld, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1163 */ { UD_Ipsrld, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1164 */ { UD_Ivpsrld, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1165 */ { UD_Ipsrld, O_U, O_Ib, O_NONE, O_NONE, P_rexb }, + /* 1166 */ { UD_Ivpsrld, O_Hx, O_Ux, O_Ib, O_NONE, P_rexb }, + /* 1167 */ { UD_Ipsrlq, O_N, O_Ib, O_NONE, O_NONE, P_none }, + /* 1168 */ { UD_Ipsrlq, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1169 */ { UD_Ipsrlq, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1170 */ { UD_Ivpsrlq, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1171 */ { UD_Ipsrlq, O_U, O_Ib, O_NONE, O_NONE, P_rexb }, + /* 1172 */ { UD_Ivpsrlq, O_Hx, O_Ux, O_Ib, O_NONE, P_rexb }, + /* 1173 */ { UD_Ipsubb, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1174 */ { UD_Ivpsubb, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1175 */ { UD_Ipsubb, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1176 */ { UD_Ipsubw, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1177 */ { UD_Ivpsubw, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1178 */ { UD_Ipsubw, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1179 */ { UD_Ipsubd, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1180 */ { UD_Ipsubd, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1181 */ { UD_Ivpsubd, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1182 */ { UD_Ipsubsb, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1183 */ { UD_Ipsubsb, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1184 */ { UD_Ivpsubsb, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1185 */ { UD_Ipsubsw, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1186 */ { UD_Ipsubsw, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1187 */ { UD_Ivpsubsw, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1188 */ { UD_Ipsubusb, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1189 */ { UD_Ipsubusb, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1190 */ { UD_Ivpsubusb, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1191 */ { UD_Ipsubusw, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1192 */ { UD_Ipsubusw, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1193 */ { UD_Ivpsubusw, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1194 */ { UD_Ipunpckhbw, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1195 */ { UD_Ivpunpckhbw, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1196 */ { UD_Ipunpckhbw, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1197 */ { UD_Ipunpckhwd, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1198 */ { UD_Ivpunpckhwd, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1199 */ { UD_Ipunpckhwd, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1200 */ { UD_Ipunpckhdq, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1201 */ { UD_Ivpunpckhdq, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1202 */ { UD_Ipunpckhdq, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1203 */ { UD_Ipunpcklbw, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1204 */ { UD_Ivpunpcklbw, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1205 */ { UD_Ipunpcklbw, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1206 */ { UD_Ipunpcklwd, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1207 */ { UD_Ivpunpcklwd, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1208 */ { UD_Ipunpcklwd, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1209 */ { UD_Ipunpckldq, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1210 */ { UD_Ivpunpckldq, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1211 */ { UD_Ipunpckldq, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1212 */ { UD_Ipi2fw, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1213 */ { UD_Ipi2fd, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1214 */ { UD_Ipf2iw, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1215 */ { UD_Ipf2id, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1216 */ { UD_Ipfnacc, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1217 */ { UD_Ipfpnacc, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1218 */ { UD_Ipfcmpge, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1219 */ { UD_Ipfmin, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1220 */ { UD_Ipfrcp, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1221 */ { UD_Ipfrsqrt, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1222 */ { UD_Ipfsub, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1223 */ { UD_Ipfadd, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1224 */ { UD_Ipfcmpgt, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1225 */ { UD_Ipfmax, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1226 */ { UD_Ipfrcpit1, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1227 */ { UD_Ipfrsqit1, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1228 */ { UD_Ipfsubr, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1229 */ { UD_Ipfacc, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1230 */ { UD_Ipfcmpeq, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1231 */ { UD_Ipfmul, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1232 */ { UD_Ipfrcpit2, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1233 */ { UD_Ipmulhrw, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1234 */ { UD_Ipswapd, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1235 */ { UD_Ipavgusb, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1236 */ { UD_Ipush, O_ES, O_NONE, O_NONE, O_NONE, P_inv64 }, + /* 1237 */ { UD_Ipush, O_CS, O_NONE, O_NONE, O_NONE, P_inv64 }, + /* 1238 */ { UD_Ipush, O_SS, O_NONE, O_NONE, O_NONE, P_inv64 }, + /* 1239 */ { UD_Ipush, O_DS, O_NONE, O_NONE, O_NONE, P_inv64 }, + /* 1240 */ { UD_Ipush, O_GS, O_NONE, O_NONE, O_NONE, P_none }, + /* 1241 */ { UD_Ipush, O_FS, O_NONE, O_NONE, O_NONE, P_none }, + /* 1242 */ { UD_Ipush, O_R0v, O_NONE, O_NONE, O_NONE, P_oso|P_rexb|P_def64 }, + /* 1243 */ { UD_Ipush, O_R1v, O_NONE, O_NONE, O_NONE, P_oso|P_rexb|P_def64 }, + /* 1244 */ { UD_Ipush, O_R2v, O_NONE, O_NONE, O_NONE, P_oso|P_rexb|P_def64 }, + /* 1245 */ { UD_Ipush, O_R3v, O_NONE, O_NONE, O_NONE, P_oso|P_rexb|P_def64 }, + /* 1246 */ { UD_Ipush, O_R4v, O_NONE, O_NONE, O_NONE, P_oso|P_rexb|P_def64 }, + /* 1247 */ { UD_Ipush, O_R5v, O_NONE, O_NONE, O_NONE, P_oso|P_rexb|P_def64 }, + /* 1248 */ { UD_Ipush, O_R6v, O_NONE, O_NONE, O_NONE, P_oso|P_rexb|P_def64 }, + /* 1249 */ { UD_Ipush, O_R7v, O_NONE, O_NONE, O_NONE, P_oso|P_rexb|P_def64 }, + /* 1250 */ { UD_Ipush, O_sIz, O_NONE, O_NONE, O_NONE, P_oso|P_def64 }, + /* 1251 */ { UD_Ipush, O_Ev, O_NONE, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb|P_def64 }, + /* 1252 */ { UD_Ipush, O_sIb, O_NONE, O_NONE, O_NONE, P_oso|P_def64 }, + /* 1253 */ { UD_Ipusha, O_NONE, O_NONE, O_NONE, O_NONE, P_oso|P_inv64 }, + /* 1254 */ { UD_Ipushad, O_NONE, O_NONE, O_NONE, O_NONE, P_oso|P_inv64 }, + /* 1255 */ { UD_Ipushfw, O_NONE, O_NONE, O_NONE, O_NONE, P_oso }, + /* 1256 */ { UD_Ipushfw, O_NONE, O_NONE, O_NONE, O_NONE, P_oso|P_rexw|P_def64 }, + /* 1257 */ { UD_Ipushfd, O_NONE, O_NONE, O_NONE, O_NONE, P_oso }, + /* 1258 */ { UD_Ipushfq, O_NONE, O_NONE, O_NONE, O_NONE, P_oso|P_rexw|P_def64 }, + /* 1259 */ { UD_Ipushfq, O_NONE, O_NONE, O_NONE, O_NONE, P_oso|P_rexw|P_def64 }, + /* 1260 */ { UD_Ipxor, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1261 */ { UD_Ivpxor, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1262 */ { UD_Ipxor, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1263 */ { UD_Ircl, O_Eb, O_Ib, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1264 */ { UD_Ircl, O_Ev, O_Ib, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1265 */ { UD_Ircl, O_Eb, O_I1, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1266 */ { UD_Ircl, O_Eb, O_CL, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1267 */ { UD_Ircl, O_Ev, O_CL, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1268 */ { UD_Ircl, O_Ev, O_I1, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1269 */ { UD_Ircr, O_Eb, O_I1, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1270 */ { UD_Ircr, O_Ev, O_Ib, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1271 */ { UD_Ircr, O_Eb, O_Ib, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1272 */ { UD_Ircr, O_Ev, O_I1, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1273 */ { UD_Ircr, O_Eb, O_CL, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1274 */ { UD_Ircr, O_Ev, O_CL, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1275 */ { UD_Irol, O_Eb, O_Ib, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1276 */ { UD_Irol, O_Eb, O_I1, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1277 */ { UD_Irol, O_Ev, O_I1, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1278 */ { UD_Irol, O_Eb, O_CL, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1279 */ { UD_Irol, O_Ev, O_CL, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1280 */ { UD_Irol, O_Ev, O_Ib, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1281 */ { UD_Iror, O_Eb, O_I1, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1282 */ { UD_Iror, O_Eb, O_Ib, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1283 */ { UD_Iror, O_Ev, O_Ib, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1284 */ { UD_Iror, O_Ev, O_I1, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1285 */ { UD_Iror, O_Eb, O_CL, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1286 */ { UD_Iror, O_Ev, O_CL, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1287 */ { UD_Ircpps, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1288 */ { UD_Ivrcpps, O_Vx, O_Wx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 1289 */ { UD_Ircpss, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1290 */ { UD_Ivrcpss, O_Vx, O_Wx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1291 */ { UD_Irdmsr, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1292 */ { UD_Irdpmc, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1293 */ { UD_Irdtsc, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1294 */ { UD_Irdtscp, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1295 */ { UD_Irepne, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1296 */ { UD_Irep, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1297 */ { UD_Iret, O_Iw, O_NONE, O_NONE, O_NONE, P_none }, + /* 1298 */ { UD_Iret, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1299 */ { UD_Iretf, O_Iw, O_NONE, O_NONE, O_NONE, P_none }, + /* 1300 */ { UD_Iretf, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1301 */ { UD_Irsm, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1302 */ { UD_Irsqrtps, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1303 */ { UD_Ivrsqrtps, O_Vx, O_Wx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 1304 */ { UD_Irsqrtss, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1305 */ { UD_Ivrsqrtss, O_Vx, O_Wx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1306 */ { UD_Isahf, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1307 */ { UD_Isalc, O_NONE, O_NONE, O_NONE, O_NONE, P_inv64 }, + /* 1308 */ { UD_Isar, O_Ev, O_I1, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1309 */ { UD_Isar, O_Eb, O_Ib, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1310 */ { UD_Isar, O_Eb, O_I1, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1311 */ { UD_Isar, O_Ev, O_Ib, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1312 */ { UD_Isar, O_Eb, O_CL, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1313 */ { UD_Isar, O_Ev, O_CL, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1314 */ { UD_Ishl, O_Eb, O_Ib, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1315 */ { UD_Ishl, O_Ev, O_Ib, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1316 */ { UD_Ishl, O_Eb, O_I1, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1317 */ { UD_Ishl, O_Eb, O_CL, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1318 */ { UD_Ishl, O_Ev, O_CL, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1319 */ { UD_Ishl, O_Ev, O_Ib, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1320 */ { UD_Ishl, O_Eb, O_CL, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1321 */ { UD_Ishl, O_Ev, O_I1, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1322 */ { UD_Ishl, O_Eb, O_I1, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1323 */ { UD_Ishl, O_Eb, O_Ib, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1324 */ { UD_Ishl, O_Ev, O_CL, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1325 */ { UD_Ishl, O_Ev, O_I1, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1326 */ { UD_Ishr, O_Ev, O_Ib, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1327 */ { UD_Ishr, O_Eb, O_CL, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1328 */ { UD_Ishr, O_Ev, O_I1, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1329 */ { UD_Ishr, O_Eb, O_I1, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1330 */ { UD_Ishr, O_Eb, O_Ib, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1331 */ { UD_Ishr, O_Ev, O_CL, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1332 */ { UD_Isbb, O_Eb, O_Gb, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1333 */ { UD_Isbb, O_Ev, O_Gv, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1334 */ { UD_Isbb, O_Gb, O_Eb, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1335 */ { UD_Isbb, O_Gv, O_Ev, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1336 */ { UD_Isbb, O_AL, O_Ib, O_NONE, O_NONE, P_none }, + /* 1337 */ { UD_Isbb, O_rAX, O_sIz, O_NONE, O_NONE, P_oso|P_rexw }, + /* 1338 */ { UD_Isbb, O_Eb, O_Ib, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1339 */ { UD_Isbb, O_Ev, O_sIz, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1340 */ { UD_Isbb, O_Eb, O_Ib, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_inv64 }, + /* 1341 */ { UD_Isbb, O_Ev, O_sIb, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1342 */ { UD_Iscasb, O_NONE, O_NONE, O_NONE, O_NONE, P_strz }, + /* 1343 */ { UD_Iscasw, O_NONE, O_NONE, O_NONE, O_NONE, P_strz|P_oso|P_rexw }, + /* 1344 */ { UD_Iscasd, O_NONE, O_NONE, O_NONE, O_NONE, P_strz|P_oso|P_rexw }, + /* 1345 */ { UD_Iscasq, O_NONE, O_NONE, O_NONE, O_NONE, P_strz|P_oso|P_rexw }, + /* 1346 */ { UD_Iseto, O_Eb, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1347 */ { UD_Isetno, O_Eb, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1348 */ { UD_Isetb, O_Eb, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1349 */ { UD_Isetae, O_Eb, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1350 */ { UD_Isetz, O_Eb, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1351 */ { UD_Isetnz, O_Eb, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1352 */ { UD_Isetbe, O_Eb, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1353 */ { UD_Iseta, O_Eb, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1354 */ { UD_Isets, O_Eb, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1355 */ { UD_Isetns, O_Eb, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1356 */ { UD_Isetp, O_Eb, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1357 */ { UD_Isetnp, O_Eb, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1358 */ { UD_Isetl, O_Eb, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1359 */ { UD_Isetge, O_Eb, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1360 */ { UD_Isetle, O_Eb, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1361 */ { UD_Isetg, O_Eb, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1362 */ { UD_Isfence, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1363 */ { UD_Isfence, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1364 */ { UD_Isfence, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1365 */ { UD_Isfence, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1366 */ { UD_Isfence, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1367 */ { UD_Isfence, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1368 */ { UD_Isfence, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1369 */ { UD_Isfence, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1370 */ { UD_Isgdt, O_M, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1371 */ { UD_Ishld, O_Ev, O_Gv, O_Ib, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1372 */ { UD_Ishld, O_Ev, O_Gv, O_CL, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1373 */ { UD_Ishrd, O_Ev, O_Gv, O_Ib, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1374 */ { UD_Ishrd, O_Ev, O_Gv, O_CL, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1375 */ { UD_Ishufpd, O_V, O_W, O_Ib, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1376 */ { UD_Ivshufpd, O_Vx, O_Hx, O_Wx, O_Ib, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 1377 */ { UD_Ishufps, O_V, O_W, O_Ib, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1378 */ { UD_Ivshufps, O_Vx, O_Hx, O_Wx, O_Ib, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1379 */ { UD_Isidt, O_M, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1380 */ { UD_Isldt, O_MwRv, O_NONE, O_NONE, O_NONE, P_aso|P_oso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1381 */ { UD_Ismsw, O_MwRv, O_NONE, O_NONE, O_NONE, P_aso|P_oso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1382 */ { UD_Ismsw, O_MwRv, O_NONE, O_NONE, O_NONE, P_aso|P_oso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1383 */ { UD_Isqrtps, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1384 */ { UD_Ivsqrtps, O_Vx, O_Wx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 1385 */ { UD_Isqrtpd, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1386 */ { UD_Ivsqrtpd, O_Vx, O_Wx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1387 */ { UD_Isqrtsd, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1388 */ { UD_Ivsqrtsd, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1389 */ { UD_Isqrtss, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1390 */ { UD_Ivsqrtss, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1391 */ { UD_Istc, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1392 */ { UD_Istd, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1393 */ { UD_Istgi, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1394 */ { UD_Isti, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1395 */ { UD_Iskinit, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1396 */ { UD_Istmxcsr, O_Md, O_NONE, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1397 */ { UD_Ivstmxcsr, O_Md, O_NONE, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1398 */ { UD_Istosb, O_NONE, O_NONE, O_NONE, O_NONE, P_str|P_seg }, + /* 1399 */ { UD_Istosw, O_NONE, O_NONE, O_NONE, O_NONE, P_str|P_seg|P_oso|P_rexw }, + /* 1400 */ { UD_Istosd, O_NONE, O_NONE, O_NONE, O_NONE, P_str|P_seg|P_oso|P_rexw }, + /* 1401 */ { UD_Istosq, O_NONE, O_NONE, O_NONE, O_NONE, P_str|P_seg|P_oso|P_rexw }, + /* 1402 */ { UD_Istr, O_MwRv, O_NONE, O_NONE, O_NONE, P_aso|P_oso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1403 */ { UD_Isub, O_Eb, O_Gb, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1404 */ { UD_Isub, O_Ev, O_Gv, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1405 */ { UD_Isub, O_Gb, O_Eb, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1406 */ { UD_Isub, O_Gv, O_Ev, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1407 */ { UD_Isub, O_AL, O_Ib, O_NONE, O_NONE, P_none }, + /* 1408 */ { UD_Isub, O_rAX, O_sIz, O_NONE, O_NONE, P_oso|P_rexw }, + /* 1409 */ { UD_Isub, O_Eb, O_Ib, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1410 */ { UD_Isub, O_Ev, O_sIz, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1411 */ { UD_Isub, O_Eb, O_Ib, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_inv64 }, + /* 1412 */ { UD_Isub, O_Ev, O_sIb, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1413 */ { UD_Isubpd, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1414 */ { UD_Ivsubpd, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 1415 */ { UD_Isubps, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1416 */ { UD_Ivsubps, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 1417 */ { UD_Isubsd, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1418 */ { UD_Ivsubsd, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1419 */ { UD_Isubss, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1420 */ { UD_Ivsubss, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1421 */ { UD_Iswapgs, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1422 */ { UD_Isyscall, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1423 */ { UD_Isysenter, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1424 */ { UD_Isysenter, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1425 */ { UD_Isysexit, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1426 */ { UD_Isysexit, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1427 */ { UD_Isysret, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1428 */ { UD_Itest, O_Eb, O_Ib, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1429 */ { UD_Itest, O_Eb, O_Gb, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1430 */ { UD_Itest, O_Ev, O_Gv, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1431 */ { UD_Itest, O_AL, O_Ib, O_NONE, O_NONE, P_none }, + /* 1432 */ { UD_Itest, O_rAX, O_sIz, O_NONE, O_NONE, P_oso|P_rexw }, + /* 1433 */ { UD_Itest, O_Eb, O_Ib, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1434 */ { UD_Itest, O_Ev, O_sIz, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1435 */ { UD_Itest, O_Ev, O_Iz, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1436 */ { UD_Iucomisd, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1437 */ { UD_Ivucomisd, O_Vx, O_Wx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1438 */ { UD_Iucomiss, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1439 */ { UD_Ivucomiss, O_Vx, O_Wx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1440 */ { UD_Iud2, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1441 */ { UD_Iunpckhpd, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1442 */ { UD_Ivunpckhpd, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 1443 */ { UD_Iunpckhps, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1444 */ { UD_Ivunpckhps, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1445 */ { UD_Iunpcklps, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1446 */ { UD_Ivunpcklps, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1447 */ { UD_Iunpcklpd, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1448 */ { UD_Ivunpcklpd, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 1449 */ { UD_Iverr, O_Ew, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1450 */ { UD_Iverw, O_Ew, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1451 */ { UD_Ivmcall, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1452 */ { UD_Irdrand, O_R, O_NONE, O_NONE, O_NONE, P_oso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1453 */ { UD_Ivmclear, O_Mq, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1454 */ { UD_Ivmxon, O_Mq, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1455 */ { UD_Ivmptrld, O_Mq, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1456 */ { UD_Ivmptrst, O_Mq, O_NONE, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1457 */ { UD_Ivmlaunch, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1458 */ { UD_Ivmresume, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1459 */ { UD_Ivmxoff, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1460 */ { UD_Ivmread, O_Ey, O_Gy, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_def64 }, + /* 1461 */ { UD_Ivmwrite, O_Gy, O_Ey, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_def64 }, + /* 1462 */ { UD_Ivmrun, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1463 */ { UD_Ivmmcall, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1464 */ { UD_Ivmload, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1465 */ { UD_Ivmsave, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1466 */ { UD_Iwait, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1467 */ { UD_Iwbinvd, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1468 */ { UD_Iwrmsr, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1469 */ { UD_Ixadd, O_Eb, O_Gb, O_NONE, O_NONE, P_aso|P_oso|P_rexr|P_rexx|P_rexb }, + /* 1470 */ { UD_Ixadd, O_Ev, O_Gv, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1471 */ { UD_Ixchg, O_Eb, O_Gb, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1472 */ { UD_Ixchg, O_Ev, O_Gv, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1473 */ { UD_Ixchg, O_R0v, O_rAX, O_NONE, O_NONE, P_oso|P_rexw|P_rexb }, + /* 1474 */ { UD_Ixchg, O_R1v, O_rAX, O_NONE, O_NONE, P_oso|P_rexw|P_rexb }, + /* 1475 */ { UD_Ixchg, O_R2v, O_rAX, O_NONE, O_NONE, P_oso|P_rexw|P_rexb }, + /* 1476 */ { UD_Ixchg, O_R3v, O_rAX, O_NONE, O_NONE, P_oso|P_rexw|P_rexb }, + /* 1477 */ { UD_Ixchg, O_R4v, O_rAX, O_NONE, O_NONE, P_oso|P_rexw|P_rexb }, + /* 1478 */ { UD_Ixchg, O_R5v, O_rAX, O_NONE, O_NONE, P_oso|P_rexw|P_rexb }, + /* 1479 */ { UD_Ixchg, O_R6v, O_rAX, O_NONE, O_NONE, P_oso|P_rexw|P_rexb }, + /* 1480 */ { UD_Ixchg, O_R7v, O_rAX, O_NONE, O_NONE, P_oso|P_rexw|P_rexb }, + /* 1481 */ { UD_Ixgetbv, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1482 */ { UD_Ixlatb, O_NONE, O_NONE, O_NONE, O_NONE, P_rexw|P_seg }, + /* 1483 */ { UD_Ixor, O_Eb, O_Gb, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1484 */ { UD_Ixor, O_Ev, O_Gv, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1485 */ { UD_Ixor, O_Gb, O_Eb, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1486 */ { UD_Ixor, O_Gv, O_Ev, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1487 */ { UD_Ixor, O_AL, O_Ib, O_NONE, O_NONE, P_none }, + /* 1488 */ { UD_Ixor, O_rAX, O_sIz, O_NONE, O_NONE, P_oso|P_rexw }, + /* 1489 */ { UD_Ixor, O_Eb, O_Ib, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1490 */ { UD_Ixor, O_Ev, O_sIz, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1491 */ { UD_Ixor, O_Eb, O_Ib, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_inv64 }, + /* 1492 */ { UD_Ixor, O_Ev, O_sIb, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1493 */ { UD_Ixorpd, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1494 */ { UD_Ivxorpd, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 1495 */ { UD_Ixorps, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1496 */ { UD_Ivxorps, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1497 */ { UD_Ixcryptecb, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1498 */ { UD_Ixcryptcbc, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1499 */ { UD_Ixcryptctr, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1500 */ { UD_Ixcryptcfb, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1501 */ { UD_Ixcryptofb, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1502 */ { UD_Ixrstor, O_M, O_NONE, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1503 */ { UD_Ixsave, O_M, O_NONE, O_NONE, O_NONE, P_aso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1504 */ { UD_Ixsetbv, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1505 */ { UD_Ixsha1, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1506 */ { UD_Ixsha256, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1507 */ { UD_Ixstore, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1508 */ { UD_Ipclmulqdq, O_V, O_W, O_Ib, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1509 */ { UD_Ivpclmulqdq, O_Vx, O_Hx, O_Wx, O_Ib, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1510 */ { UD_Igetsec, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1511 */ { UD_Imovdqa, O_W, O_V, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1512 */ { UD_Ivmovdqa, O_Wx, O_Vx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 1513 */ { UD_Imovdqa, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1514 */ { UD_Ivmovdqa, O_Vx, O_Wx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 1515 */ { UD_Imaskmovdqu, O_V, O_U, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1516 */ { UD_Ivmaskmovdqu, O_Vx, O_Ux, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1517 */ { UD_Imovdq2q, O_P, O_U, O_NONE, O_NONE, P_aso|P_rexb }, + /* 1518 */ { UD_Imovdqu, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1519 */ { UD_Ivmovdqu, O_Vx, O_Wx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 1520 */ { UD_Imovdqu, O_W, O_V, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1521 */ { UD_Imovq2dq, O_V, O_N, O_NONE, O_NONE, P_aso|P_rexr }, + /* 1522 */ { UD_Ipaddq, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1523 */ { UD_Ipaddq, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1524 */ { UD_Ivpaddq, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1525 */ { UD_Ipsubq, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1526 */ { UD_Ivpsubq, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1527 */ { UD_Ipsubq, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1528 */ { UD_Ipmuludq, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1529 */ { UD_Ipmuludq, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1530 */ { UD_Ipshufhw, O_V, O_W, O_Ib, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1531 */ { UD_Ivpshufhw, O_Vx, O_Wx, O_Ib, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1532 */ { UD_Ipshuflw, O_V, O_W, O_Ib, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1533 */ { UD_Ivpshuflw, O_Vx, O_Wx, O_Ib, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1534 */ { UD_Ipshufd, O_V, O_W, O_Ib, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1535 */ { UD_Ivpshufd, O_Vx, O_Wx, O_Ib, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1536 */ { UD_Ipslldq, O_U, O_Ib, O_NONE, O_NONE, P_rexb }, + /* 1537 */ { UD_Ivpslldq, O_Hx, O_Ux, O_Ib, O_NONE, P_rexb }, + /* 1538 */ { UD_Ipsrldq, O_U, O_Ib, O_NONE, O_NONE, P_rexb }, + /* 1539 */ { UD_Ivpsrldq, O_Hx, O_Ux, O_Ib, O_NONE, P_rexb }, + /* 1540 */ { UD_Ipunpckhqdq, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1541 */ { UD_Ivpunpckhqdq, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1542 */ { UD_Ipunpcklqdq, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1543 */ { UD_Ivpunpcklqdq, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1544 */ { UD_Ihaddpd, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1545 */ { UD_Ivhaddpd, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 1546 */ { UD_Ihaddps, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1547 */ { UD_Ivhaddps, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 1548 */ { UD_Ihsubpd, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1549 */ { UD_Ivhsubpd, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 1550 */ { UD_Ihsubps, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1551 */ { UD_Ivhsubps, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 1552 */ { UD_Iinsertps, O_V, O_Md, O_Ib, O_NONE, P_aso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1553 */ { UD_Ivinsertps, O_Vx, O_Hx, O_Md, O_Ib, P_aso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1554 */ { UD_Ilddqu, O_V, O_M, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1555 */ { UD_Ivlddqu, O_Vx, O_M, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 1556 */ { UD_Imovddup, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1557 */ { UD_Ivmovddup, O_Vx, O_Wx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1558 */ { UD_Imovddup, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1559 */ { UD_Ivmovddup, O_Vx, O_Wx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1560 */ { UD_Imovshdup, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1561 */ { UD_Ivmovshdup, O_Vx, O_Wx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 1562 */ { UD_Imovshdup, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1563 */ { UD_Ivmovshdup, O_Vx, O_Wx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 1564 */ { UD_Imovsldup, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1565 */ { UD_Ivmovsldup, O_Vx, O_Wx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 1566 */ { UD_Imovsldup, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1567 */ { UD_Ivmovsldup, O_Vx, O_Wx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 1568 */ { UD_Ipabsb, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1569 */ { UD_Ipabsb, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1570 */ { UD_Ivpabsb, O_Vx, O_Wx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 1571 */ { UD_Ipabsw, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1572 */ { UD_Ipabsw, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1573 */ { UD_Ivpabsw, O_Vx, O_Wx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 1574 */ { UD_Ipabsd, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1575 */ { UD_Ipabsd, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1576 */ { UD_Ivpabsd, O_Vx, O_Wx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 1577 */ { UD_Ipshufb, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1578 */ { UD_Ipshufb, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1579 */ { UD_Ivpshufb, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1580 */ { UD_Iphaddw, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1581 */ { UD_Iphaddw, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1582 */ { UD_Ivphaddw, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1583 */ { UD_Iphaddd, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1584 */ { UD_Iphaddd, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1585 */ { UD_Ivphaddd, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1586 */ { UD_Iphaddsw, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1587 */ { UD_Iphaddsw, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1588 */ { UD_Ivphaddsw, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1589 */ { UD_Ipmaddubsw, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1590 */ { UD_Ipmaddubsw, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1591 */ { UD_Ivpmaddubsw, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1592 */ { UD_Iphsubw, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1593 */ { UD_Iphsubw, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1594 */ { UD_Ivphsubw, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1595 */ { UD_Iphsubd, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1596 */ { UD_Iphsubd, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1597 */ { UD_Ivphsubd, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1598 */ { UD_Iphsubsw, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1599 */ { UD_Iphsubsw, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1600 */ { UD_Ivphsubsw, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1601 */ { UD_Ipsignb, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1602 */ { UD_Ipsignb, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1603 */ { UD_Ivpsignb, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1604 */ { UD_Ipsignd, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1605 */ { UD_Ipsignd, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1606 */ { UD_Ivpsignd, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1607 */ { UD_Ipsignw, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1608 */ { UD_Ipsignw, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1609 */ { UD_Ivpsignw, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1610 */ { UD_Ipmulhrsw, O_P, O_Q, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1611 */ { UD_Ipmulhrsw, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1612 */ { UD_Ivpmulhrsw, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1613 */ { UD_Ipalignr, O_P, O_Q, O_Ib, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1614 */ { UD_Ipalignr, O_V, O_W, O_Ib, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1615 */ { UD_Ivpalignr, O_Vx, O_Hx, O_Wx, O_Ib, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1616 */ { UD_Ipblendvb, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1617 */ { UD_Ipmuldq, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1618 */ { UD_Ivpmuldq, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1619 */ { UD_Ipminsb, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1620 */ { UD_Ivpminsb, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1621 */ { UD_Ipminsd, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1622 */ { UD_Ivpminsd, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1623 */ { UD_Ipminuw, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1624 */ { UD_Ivpminuw, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1625 */ { UD_Ipminud, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1626 */ { UD_Ivpminud, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1627 */ { UD_Ipmaxsb, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1628 */ { UD_Ivpmaxsb, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1629 */ { UD_Ipmaxsd, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1630 */ { UD_Ivpmaxsd, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1631 */ { UD_Ipmaxud, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1632 */ { UD_Ivpmaxud, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1633 */ { UD_Ipmaxuw, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1634 */ { UD_Ivpmaxuw, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1635 */ { UD_Ipmulld, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1636 */ { UD_Ivpmulld, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1637 */ { UD_Iphminposuw, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1638 */ { UD_Ivphminposuw, O_Vx, O_Wx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1639 */ { UD_Iroundps, O_V, O_W, O_Ib, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1640 */ { UD_Ivroundps, O_Vx, O_Wx, O_Ib, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 1641 */ { UD_Iroundpd, O_V, O_W, O_Ib, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1642 */ { UD_Ivroundpd, O_Vx, O_Wx, O_Ib, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 1643 */ { UD_Iroundss, O_V, O_W, O_Ib, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1644 */ { UD_Ivroundss, O_Vx, O_Hx, O_Wx, O_Ib, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1645 */ { UD_Iroundsd, O_V, O_W, O_Ib, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1646 */ { UD_Ivroundsd, O_Vx, O_Hx, O_Wx, O_Ib, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1647 */ { UD_Iblendpd, O_V, O_W, O_Ib, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1648 */ { UD_Ivblendpd, O_Vx, O_Hx, O_Wx, O_Ib, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 1649 */ { UD_Iblendps, O_V, O_W, O_Ib, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1650 */ { UD_Ivblendps, O_Vx, O_Hx, O_Wx, O_Ib, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1651 */ { UD_Iblendvpd, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1652 */ { UD_Iblendvps, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1653 */ { UD_Ibound, O_Gv, O_M, O_NONE, O_NONE, P_aso|P_oso }, + /* 1654 */ { UD_Ibsf, O_Gv, O_Ev, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1655 */ { UD_Ibsr, O_Gv, O_Ev, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1656 */ { UD_Ibswap, O_R0y, O_NONE, O_NONE, O_NONE, P_oso|P_rexw|P_rexb }, + /* 1657 */ { UD_Ibswap, O_R1y, O_NONE, O_NONE, O_NONE, P_oso|P_rexw|P_rexb }, + /* 1658 */ { UD_Ibswap, O_R2y, O_NONE, O_NONE, O_NONE, P_oso|P_rexw|P_rexb }, + /* 1659 */ { UD_Ibswap, O_R3y, O_NONE, O_NONE, O_NONE, P_oso|P_rexw|P_rexb }, + /* 1660 */ { UD_Ibswap, O_R4y, O_NONE, O_NONE, O_NONE, P_oso|P_rexw|P_rexb }, + /* 1661 */ { UD_Ibswap, O_R5y, O_NONE, O_NONE, O_NONE, P_oso|P_rexw|P_rexb }, + /* 1662 */ { UD_Ibswap, O_R6y, O_NONE, O_NONE, O_NONE, P_oso|P_rexw|P_rexb }, + /* 1663 */ { UD_Ibswap, O_R7y, O_NONE, O_NONE, O_NONE, P_oso|P_rexw|P_rexb }, + /* 1664 */ { UD_Ibt, O_Ev, O_Ib, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1665 */ { UD_Ibt, O_Ev, O_Gv, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1666 */ { UD_Ibtc, O_Ev, O_Gv, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1667 */ { UD_Ibtc, O_Ev, O_Ib, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1668 */ { UD_Ibtr, O_Ev, O_Gv, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1669 */ { UD_Ibtr, O_Ev, O_Ib, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1670 */ { UD_Ibts, O_Ev, O_Gv, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1671 */ { UD_Ibts, O_Ev, O_Ib, O_NONE, O_NONE, P_aso|P_oso|P_rexw|P_rexr|P_rexx|P_rexb }, + /* 1672 */ { UD_Ipblendw, O_V, O_W, O_Ib, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1673 */ { UD_Ivpblendw, O_Vx, O_Hx, O_Wx, O_Ib, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1674 */ { UD_Impsadbw, O_V, O_W, O_Ib, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1675 */ { UD_Ivmpsadbw, O_Vx, O_Hx, O_Wx, O_Ib, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 1676 */ { UD_Imovntdqa, O_V, O_M, O_NONE, O_NONE, P_aso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1677 */ { UD_Ivmovntdqa, O_Vx, O_M, O_NONE, O_NONE, P_aso|P_rexr|P_rexw|P_rexx|P_rexb|P_vexl }, + /* 1678 */ { UD_Ipackusdw, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1679 */ { UD_Ivpackusdw, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexw|P_rexx|P_rexb|P_vexl }, + /* 1680 */ { UD_Ipmovsxbw, O_V, O_MqU, O_NONE, O_NONE, P_aso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1681 */ { UD_Ivpmovsxbw, O_Vx, O_MqU, O_NONE, O_NONE, P_aso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1682 */ { UD_Ipmovsxbd, O_V, O_MdU, O_NONE, O_NONE, P_aso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1683 */ { UD_Ivpmovsxbd, O_Vx, O_MdU, O_NONE, O_NONE, P_aso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1684 */ { UD_Ipmovsxbq, O_V, O_MwU, O_NONE, O_NONE, P_aso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1685 */ { UD_Ivpmovsxbq, O_Vx, O_MwU, O_NONE, O_NONE, P_aso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1686 */ { UD_Ipmovsxwd, O_V, O_MqU, O_NONE, O_NONE, P_aso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1687 */ { UD_Ivpmovsxwd, O_Vx, O_MqU, O_NONE, O_NONE, P_aso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1688 */ { UD_Ipmovsxwq, O_V, O_MdU, O_NONE, O_NONE, P_aso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1689 */ { UD_Ivpmovsxwq, O_Vx, O_MdU, O_NONE, O_NONE, P_aso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1690 */ { UD_Ipmovsxdq, O_V, O_MqU, O_NONE, O_NONE, P_aso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1691 */ { UD_Ipmovzxbw, O_V, O_MqU, O_NONE, O_NONE, P_aso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1692 */ { UD_Ivpmovzxbw, O_Vx, O_MqU, O_NONE, O_NONE, P_aso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1693 */ { UD_Ipmovzxbd, O_V, O_MdU, O_NONE, O_NONE, P_aso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1694 */ { UD_Ivpmovzxbd, O_Vx, O_MdU, O_NONE, O_NONE, P_aso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1695 */ { UD_Ipmovzxbq, O_V, O_MwU, O_NONE, O_NONE, P_aso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1696 */ { UD_Ivpmovzxbq, O_Vx, O_MwU, O_NONE, O_NONE, P_aso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1697 */ { UD_Ipmovzxwd, O_V, O_MqU, O_NONE, O_NONE, P_aso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1698 */ { UD_Ivpmovzxwd, O_Vx, O_MqU, O_NONE, O_NONE, P_aso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1699 */ { UD_Ipmovzxwq, O_V, O_MdU, O_NONE, O_NONE, P_aso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1700 */ { UD_Ivpmovzxwq, O_Vx, O_MdU, O_NONE, O_NONE, P_aso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1701 */ { UD_Ipmovzxdq, O_V, O_MqU, O_NONE, O_NONE, P_aso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1702 */ { UD_Ivpmovzxdq, O_Vx, O_MqU, O_NONE, O_NONE, P_aso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1703 */ { UD_Ipcmpeqq, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1704 */ { UD_Ivpcmpeqq, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1705 */ { UD_Ipopcnt, O_Gv, O_Ev, O_NONE, O_NONE, P_aso|P_oso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1706 */ { UD_Iptest, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1707 */ { UD_Ivptest, O_Vx, O_Wx, O_NONE, O_NONE, P_aso|P_rexr|P_rexw|P_rexx|P_rexb|P_vexl }, + /* 1708 */ { UD_Ipcmpestri, O_V, O_W, O_Ib, O_NONE, P_aso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1709 */ { UD_Ivpcmpestri, O_Vx, O_Wx, O_Ib, O_NONE, P_aso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1710 */ { UD_Ipcmpestrm, O_V, O_W, O_Ib, O_NONE, P_aso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1711 */ { UD_Ivpcmpestrm, O_Vx, O_Wx, O_Ib, O_NONE, P_aso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1712 */ { UD_Ipcmpgtq, O_V, O_W, O_NONE, O_NONE, P_aso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1713 */ { UD_Ivpcmpgtq, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1714 */ { UD_Ipcmpistri, O_V, O_W, O_Ib, O_NONE, P_aso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1715 */ { UD_Ivpcmpistri, O_Vx, O_Wx, O_Ib, O_NONE, P_aso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1716 */ { UD_Ipcmpistrm, O_V, O_W, O_Ib, O_NONE, P_aso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1717 */ { UD_Ivpcmpistrm, O_Vx, O_Wx, O_Ib, O_NONE, P_aso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1718 */ { UD_Imovbe, O_Gv, O_Mv, O_NONE, O_NONE, P_aso|P_oso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1719 */ { UD_Imovbe, O_Mv, O_Gv, O_NONE, O_NONE, P_aso|P_oso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1720 */ { UD_Icrc32, O_Gy, O_Eb, O_NONE, O_NONE, P_aso|P_oso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1721 */ { UD_Icrc32, O_Gy, O_Ev, O_NONE, O_NONE, P_aso|P_oso|P_rexr|P_rexw|P_rexx|P_rexb }, + /* 1722 */ { UD_Ivbroadcastss, O_V, O_Md, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 1723 */ { UD_Ivbroadcastsd, O_Vqq, O_Mq, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 1724 */ { UD_Ivextractf128, O_Wdq, O_Vqq, O_Ib, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 1725 */ { UD_Ivinsertf128, O_Vqq, O_Hqq, O_Wdq, O_Ib, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 1726 */ { UD_Ivmaskmovps, O_V, O_H, O_M, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 1727 */ { UD_Ivmaskmovps, O_M, O_H, O_V, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 1728 */ { UD_Ivmaskmovpd, O_V, O_H, O_M, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 1729 */ { UD_Ivmaskmovpd, O_M, O_H, O_V, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 1730 */ { UD_Ivpermilpd, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 1731 */ { UD_Ivpermilpd, O_V, O_W, O_Ib, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 1732 */ { UD_Ivpermilps, O_Vx, O_Hx, O_Wx, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 1733 */ { UD_Ivpermilps, O_Vx, O_Wx, O_Ib, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 1734 */ { UD_Ivperm2f128, O_Vqq, O_Hqq, O_Wqq, O_Ib, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 1735 */ { UD_Ivtestps, O_Vx, O_Wx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 1736 */ { UD_Ivtestpd, O_Vx, O_Wx, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 1737 */ { UD_Ivzeroupper, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1738 */ { UD_Ivzeroall, O_NONE, O_NONE, O_NONE, O_NONE, P_none }, + /* 1739 */ { UD_Ivblendvpd, O_Vx, O_Hx, O_Wx, O_Lx, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 1740 */ { UD_Ivblendvps, O_Vx, O_Hx, O_Wx, O_Lx, P_aso|P_rexr|P_rexx|P_rexb|P_vexl }, + /* 1741 */ { UD_Ivmovsd, O_V, O_H, O_U, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1742 */ { UD_Ivmovsd, O_V, O_Mq, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1743 */ { UD_Ivmovsd, O_U, O_H, O_V, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1744 */ { UD_Ivmovsd, O_Mq, O_V, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1745 */ { UD_Ivmovss, O_V, O_H, O_U, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1746 */ { UD_Ivmovss, O_V, O_Md, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1747 */ { UD_Ivmovss, O_U, O_H, O_V, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1748 */ { UD_Ivmovss, O_Md, O_V, O_NONE, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1749 */ { UD_Ivpblendvb, O_V, O_H, O_W, O_L, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1750 */ { UD_Ivpsllw, O_V, O_H, O_W, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1751 */ { UD_Ivpsllw, O_H, O_V, O_W, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1752 */ { UD_Ivpslld, O_V, O_H, O_W, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1753 */ { UD_Ivpslld, O_H, O_V, O_W, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1754 */ { UD_Ivpsllq, O_V, O_H, O_W, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, + /* 1755 */ { UD_Ivpsllq, O_H, O_V, O_W, O_NONE, P_aso|P_rexr|P_rexx|P_rexb }, +}; + + +const char* ud_mnemonics_str[] = { + "aaa", + "aad", + "aam", + "aas", + "adc", + "add", + "addpd", + "addps", + "addsd", + "addss", + "addsubpd", + "addsubps", + "aesdec", + "aesdeclast", + "aesenc", + "aesenclast", + "aesimc", + "aeskeygenassist", + "and", + "andnpd", + "andnps", + "andpd", + "andps", + "arpl", + "blendpd", + "blendps", + "blendvpd", + "blendvps", + "bound", + "bsf", + "bsr", + "bswap", + "bt", + "btc", + "btr", + "bts", + "call", + "cbw", + "cdq", + "cdqe", + "clc", + "cld", + "clflush", + "clgi", + "cli", + "clts", + "cmc", + "cmova", + "cmovae", + "cmovb", + "cmovbe", + "cmovg", + "cmovge", + "cmovl", + "cmovle", + "cmovno", + "cmovnp", + "cmovns", + "cmovnz", + "cmovo", + "cmovp", + "cmovs", + "cmovz", + "cmp", + "cmppd", + "cmpps", + "cmpsb", + "cmpsd", + "cmpsq", + "cmpss", + "cmpsw", + "cmpxchg", + "cmpxchg16b", + "cmpxchg8b", + "comisd", + "comiss", + "cpuid", + "cqo", + "crc32", + "cvtdq2pd", + "cvtdq2ps", + "cvtpd2dq", + "cvtpd2pi", + "cvtpd2ps", + "cvtpi2pd", + "cvtpi2ps", + "cvtps2dq", + "cvtps2pd", + "cvtps2pi", + "cvtsd2si", + "cvtsd2ss", + "cvtsi2sd", + "cvtsi2ss", + "cvtss2sd", + "cvtss2si", + "cvttpd2dq", + "cvttpd2pi", + "cvttps2dq", + "cvttps2pi", + "cvttsd2si", + "cvttss2si", + "cwd", + "cwde", + "daa", + "das", + "dec", + "div", + "divpd", + "divps", + "divsd", + "divss", + "dppd", + "dpps", + "emms", + "enter", + "extractps", + "f2xm1", + "fabs", + "fadd", + "faddp", + "fbld", + "fbstp", + "fchs", + "fclex", + "fcmovb", + "fcmovbe", + "fcmove", + "fcmovnb", + "fcmovnbe", + "fcmovne", + "fcmovnu", + "fcmovu", + "fcom", + "fcom2", + "fcomi", + "fcomip", + "fcomp", + "fcomp3", + "fcomp5", + "fcompp", + "fcos", + "fdecstp", + "fdiv", + "fdivp", + "fdivr", + "fdivrp", + "femms", + "ffree", + "ffreep", + "fiadd", + "ficom", + "ficomp", + "fidiv", + "fidivr", + "fild", + "fimul", + "fincstp", + "fist", + "fistp", + "fisttp", + "fisub", + "fisubr", + "fld", + "fld1", + "fldcw", + "fldenv", + "fldl2e", + "fldl2t", + "fldlg2", + "fldln2", + "fldpi", + "fldz", + "fmul", + "fmulp", + "fninit", + "fnop", + "fnsave", + "fnstcw", + "fnstenv", + "fnstsw", + "fpatan", + "fprem", + "fprem1", + "fptan", + "frndint", + "frstor", + "fscale", + "fsin", + "fsincos", + "fsqrt", + "fst", + "fstp", + "fstp1", + "fstp8", + "fstp9", + "fsub", + "fsubp", + "fsubr", + "fsubrp", + "ftst", + "fucom", + "fucomi", + "fucomip", + "fucomp", + "fucompp", + "fxam", + "fxch", + "fxch4", + "fxch7", + "fxrstor", + "fxsave", + "fxtract", + "fyl2x", + "fyl2xp1", + "getsec", + "haddpd", + "haddps", + "hlt", + "hsubpd", + "hsubps", + "idiv", + "imul", + "in", + "inc", + "insb", + "insd", + "insertps", + "insw", + "int", + "int1", + "int3", + "into", + "invd", + "invept", + "invlpg", + "invlpga", + "invvpid", + "iretd", + "iretq", + "iretw", + "ja", + "jae", + "jb", + "jbe", + "jcxz", + "jecxz", + "jg", + "jge", + "jl", + "jle", + "jmp", + "jno", + "jnp", + "jns", + "jnz", + "jo", + "jp", + "jrcxz", + "js", + "jz", + "lahf", + "lar", + "lddqu", + "ldmxcsr", + "lds", + "lea", + "leave", + "les", + "lfence", + "lfs", + "lgdt", + "lgs", + "lidt", + "lldt", + "lmsw", + "lock", + "lodsb", + "lodsd", + "lodsq", + "lodsw", + "loop", + "loope", + "loopne", + "lsl", + "lss", + "ltr", + "maskmovdqu", + "maskmovq", + "maxpd", + "maxps", + "maxsd", + "maxss", + "mfence", + "minpd", + "minps", + "minsd", + "minss", + "monitor", + "montmul", + "mov", + "movapd", + "movaps", + "movbe", + "movd", + "movddup", + "movdq2q", + "movdqa", + "movdqu", + "movhlps", + "movhpd", + "movhps", + "movlhps", + "movlpd", + "movlps", + "movmskpd", + "movmskps", + "movntdq", + "movntdqa", + "movnti", + "movntpd", + "movntps", + "movntq", + "movq", + "movq2dq", + "movsb", + "movsd", + "movshdup", + "movsldup", + "movsq", + "movss", + "movsw", + "movsx", + "movsxd", + "movupd", + "movups", + "movzx", + "mpsadbw", + "mul", + "mulpd", + "mulps", + "mulsd", + "mulss", + "mwait", + "neg", + "nop", + "not", + "or", + "orpd", + "orps", + "out", + "outsb", + "outsd", + "outsw", + "pabsb", + "pabsd", + "pabsw", + "packssdw", + "packsswb", + "packusdw", + "packuswb", + "paddb", + "paddd", + "paddq", + "paddsb", + "paddsw", + "paddusb", + "paddusw", + "paddw", + "palignr", + "pand", + "pandn", + "pavgb", + "pavgusb", + "pavgw", + "pblendvb", + "pblendw", + "pclmulqdq", + "pcmpeqb", + "pcmpeqd", + "pcmpeqq", + "pcmpeqw", + "pcmpestri", + "pcmpestrm", + "pcmpgtb", + "pcmpgtd", + "pcmpgtq", + "pcmpgtw", + "pcmpistri", + "pcmpistrm", + "pextrb", + "pextrd", + "pextrq", + "pextrw", + "pf2id", + "pf2iw", + "pfacc", + "pfadd", + "pfcmpeq", + "pfcmpge", + "pfcmpgt", + "pfmax", + "pfmin", + "pfmul", + "pfnacc", + "pfpnacc", + "pfrcp", + "pfrcpit1", + "pfrcpit2", + "pfrsqit1", + "pfrsqrt", + "pfsub", + "pfsubr", + "phaddd", + "phaddsw", + "phaddw", + "phminposuw", + "phsubd", + "phsubsw", + "phsubw", + "pi2fd", + "pi2fw", + "pinsrb", + "pinsrd", + "pinsrq", + "pinsrw", + "pmaddubsw", + "pmaddwd", + "pmaxsb", + "pmaxsd", + "pmaxsw", + "pmaxub", + "pmaxud", + "pmaxuw", + "pminsb", + "pminsd", + "pminsw", + "pminub", + "pminud", + "pminuw", + "pmovmskb", + "pmovsxbd", + "pmovsxbq", + "pmovsxbw", + "pmovsxdq", + "pmovsxwd", + "pmovsxwq", + "pmovzxbd", + "pmovzxbq", + "pmovzxbw", + "pmovzxdq", + "pmovzxwd", + "pmovzxwq", + "pmuldq", + "pmulhrsw", + "pmulhrw", + "pmulhuw", + "pmulhw", + "pmulld", + "pmullw", + "pmuludq", + "pop", + "popa", + "popad", + "popcnt", + "popfd", + "popfq", + "popfw", + "por", + "prefetch", + "prefetchnta", + "prefetcht0", + "prefetcht1", + "prefetcht2", + "psadbw", + "pshufb", + "pshufd", + "pshufhw", + "pshuflw", + "pshufw", + "psignb", + "psignd", + "psignw", + "pslld", + "pslldq", + "psllq", + "psllw", + "psrad", + "psraw", + "psrld", + "psrldq", + "psrlq", + "psrlw", + "psubb", + "psubd", + "psubq", + "psubsb", + "psubsw", + "psubusb", + "psubusw", + "psubw", + "pswapd", + "ptest", + "punpckhbw", + "punpckhdq", + "punpckhqdq", + "punpckhwd", + "punpcklbw", + "punpckldq", + "punpcklqdq", + "punpcklwd", + "push", + "pusha", + "pushad", + "pushfd", + "pushfq", + "pushfw", + "pxor", + "rcl", + "rcpps", + "rcpss", + "rcr", + "rdmsr", + "rdpmc", + "rdrand", + "rdtsc", + "rdtscp", + "rep", + "repne", + "ret", + "retf", + "rol", + "ror", + "roundpd", + "roundps", + "roundsd", + "roundss", + "rsm", + "rsqrtps", + "rsqrtss", + "sahf", + "salc", + "sar", + "sbb", + "scasb", + "scasd", + "scasq", + "scasw", + "seta", + "setae", + "setb", + "setbe", + "setg", + "setge", + "setl", + "setle", + "setno", + "setnp", + "setns", + "setnz", + "seto", + "setp", + "sets", + "setz", + "sfence", + "sgdt", + "shl", + "shld", + "shr", + "shrd", + "shufpd", + "shufps", + "sidt", + "skinit", + "sldt", + "smsw", + "sqrtpd", + "sqrtps", + "sqrtsd", + "sqrtss", + "stc", + "std", + "stgi", + "sti", + "stmxcsr", + "stosb", + "stosd", + "stosq", + "stosw", + "str", + "sub", + "subpd", + "subps", + "subsd", + "subss", + "swapgs", + "syscall", + "sysenter", + "sysexit", + "sysret", + "test", + "ucomisd", + "ucomiss", + "ud2", + "unpckhpd", + "unpckhps", + "unpcklpd", + "unpcklps", + "vaddpd", + "vaddps", + "vaddsd", + "vaddss", + "vaddsubpd", + "vaddsubps", + "vaesdec", + "vaesdeclast", + "vaesenc", + "vaesenclast", + "vaesimc", + "vaeskeygenassist", + "vandnpd", + "vandnps", + "vandpd", + "vandps", + "vblendpd", + "vblendps", + "vblendvpd", + "vblendvps", + "vbroadcastsd", + "vbroadcastss", + "vcmppd", + "vcmpps", + "vcmpsd", + "vcmpss", + "vcomisd", + "vcomiss", + "vcvtdq2pd", + "vcvtdq2ps", + "vcvtpd2dq", + "vcvtpd2ps", + "vcvtps2dq", + "vcvtps2pd", + "vcvtsd2si", + "vcvtsd2ss", + "vcvtsi2sd", + "vcvtsi2ss", + "vcvtss2sd", + "vcvtss2si", + "vcvttpd2dq", + "vcvttps2dq", + "vcvttsd2si", + "vcvttss2si", + "vdivpd", + "vdivps", + "vdivsd", + "vdivss", + "vdppd", + "vdpps", + "verr", + "verw", + "vextractf128", + "vextractps", + "vhaddpd", + "vhaddps", + "vhsubpd", + "vhsubps", + "vinsertf128", + "vinsertps", + "vlddqu", + "vmaskmovdqu", + "vmaskmovpd", + "vmaskmovps", + "vmaxpd", + "vmaxps", + "vmaxsd", + "vmaxss", + "vmcall", + "vmclear", + "vminpd", + "vminps", + "vminsd", + "vminss", + "vmlaunch", + "vmload", + "vmmcall", + "vmovapd", + "vmovaps", + "vmovd", + "vmovddup", + "vmovdqa", + "vmovdqu", + "vmovhlps", + "vmovhpd", + "vmovhps", + "vmovlhps", + "vmovlpd", + "vmovlps", + "vmovmskpd", + "vmovmskps", + "vmovntdq", + "vmovntdqa", + "vmovntpd", + "vmovntps", + "vmovq", + "vmovsd", + "vmovshdup", + "vmovsldup", + "vmovss", + "vmovupd", + "vmovups", + "vmpsadbw", + "vmptrld", + "vmptrst", + "vmread", + "vmresume", + "vmrun", + "vmsave", + "vmulpd", + "vmulps", + "vmulsd", + "vmulss", + "vmwrite", + "vmxoff", + "vmxon", + "vorpd", + "vorps", + "vpabsb", + "vpabsd", + "vpabsw", + "vpackssdw", + "vpacksswb", + "vpackusdw", + "vpackuswb", + "vpaddb", + "vpaddd", + "vpaddq", + "vpaddsb", + "vpaddsw", + "vpaddusb", + "vpaddusw", + "vpaddw", + "vpalignr", + "vpand", + "vpandn", + "vpavgb", + "vpavgw", + "vpblendvb", + "vpblendw", + "vpclmulqdq", + "vpcmpeqb", + "vpcmpeqd", + "vpcmpeqq", + "vpcmpeqw", + "vpcmpestri", + "vpcmpestrm", + "vpcmpgtb", + "vpcmpgtd", + "vpcmpgtq", + "vpcmpgtw", + "vpcmpistri", + "vpcmpistrm", + "vperm2f128", + "vpermilpd", + "vpermilps", + "vpextrb", + "vpextrd", + "vpextrq", + "vpextrw", + "vphaddd", + "vphaddsw", + "vphaddw", + "vphminposuw", + "vphsubd", + "vphsubsw", + "vphsubw", + "vpinsrb", + "vpinsrd", + "vpinsrq", + "vpinsrw", + "vpmaddubsw", + "vpmaddwd", + "vpmaxsb", + "vpmaxsd", + "vpmaxsw", + "vpmaxub", + "vpmaxud", + "vpmaxuw", + "vpminsb", + "vpminsd", + "vpminsw", + "vpminub", + "vpminud", + "vpminuw", + "vpmovmskb", + "vpmovsxbd", + "vpmovsxbq", + "vpmovsxbw", + "vpmovsxwd", + "vpmovsxwq", + "vpmovzxbd", + "vpmovzxbq", + "vpmovzxbw", + "vpmovzxdq", + "vpmovzxwd", + "vpmovzxwq", + "vpmuldq", + "vpmulhrsw", + "vpmulhuw", + "vpmulhw", + "vpmulld", + "vpmullw", + "vpor", + "vpsadbw", + "vpshufb", + "vpshufd", + "vpshufhw", + "vpshuflw", + "vpsignb", + "vpsignd", + "vpsignw", + "vpslld", + "vpslldq", + "vpsllq", + "vpsllw", + "vpsrad", + "vpsraw", + "vpsrld", + "vpsrldq", + "vpsrlq", + "vpsrlw", + "vpsubb", + "vpsubd", + "vpsubq", + "vpsubsb", + "vpsubsw", + "vpsubusb", + "vpsubusw", + "vpsubw", + "vptest", + "vpunpckhbw", + "vpunpckhdq", + "vpunpckhqdq", + "vpunpckhwd", + "vpunpcklbw", + "vpunpckldq", + "vpunpcklqdq", + "vpunpcklwd", + "vpxor", + "vrcpps", + "vrcpss", + "vroundpd", + "vroundps", + "vroundsd", + "vroundss", + "vrsqrtps", + "vrsqrtss", + "vshufpd", + "vshufps", + "vsqrtpd", + "vsqrtps", + "vsqrtsd", + "vsqrtss", + "vstmxcsr", + "vsubpd", + "vsubps", + "vsubsd", + "vsubss", + "vtestpd", + "vtestps", + "vucomisd", + "vucomiss", + "vunpckhpd", + "vunpckhps", + "vunpcklpd", + "vunpcklps", + "vxorpd", + "vxorps", + "vzeroall", + "vzeroupper", + "wait", + "wbinvd", + "wrmsr", + "xadd", + "xchg", + "xcryptcbc", + "xcryptcfb", + "xcryptctr", + "xcryptecb", + "xcryptofb", + "xgetbv", + "xlatb", + "xor", + "xorpd", + "xorps", + "xrstor", + "xsave", + "xsetbv", + "xsha1", + "xsha256", + "xstore", + "invalid", + "3dnow", + "none", + "db", + "pause" +}; diff --git a/ext/udis86/itab.h b/ext/udis86/itab.h new file mode 100644 index 0000000000..329fa07e8c --- /dev/null +++ b/ext/udis86/itab.h @@ -0,0 +1,935 @@ +#ifndef UD_ITAB_H +#define UD_ITAB_H + +/* itab.h -- generated by udis86:scripts/ud_itab.py, do no edit */ + +/* ud_table_type -- lookup table types (see decode.c) */ +enum ud_table_type { + UD_TAB__OPC_VEX, + UD_TAB__OPC_TABLE, + UD_TAB__OPC_X87, + UD_TAB__OPC_MOD, + UD_TAB__OPC_RM, + UD_TAB__OPC_OSIZE, + UD_TAB__OPC_MODE, + UD_TAB__OPC_VEX_L, + UD_TAB__OPC_3DNOW, + UD_TAB__OPC_REG, + UD_TAB__OPC_ASIZE, + UD_TAB__OPC_VEX_W, + UD_TAB__OPC_SSE, + UD_TAB__OPC_VENDOR +}; + +/* ud_mnemonic -- mnemonic constants */ +enum ud_mnemonic_code { + UD_Iaaa, + UD_Iaad, + UD_Iaam, + UD_Iaas, + UD_Iadc, + UD_Iadd, + UD_Iaddpd, + UD_Iaddps, + UD_Iaddsd, + UD_Iaddss, + UD_Iaddsubpd, + UD_Iaddsubps, + UD_Iaesdec, + UD_Iaesdeclast, + UD_Iaesenc, + UD_Iaesenclast, + UD_Iaesimc, + UD_Iaeskeygenassist, + UD_Iand, + UD_Iandnpd, + UD_Iandnps, + UD_Iandpd, + UD_Iandps, + UD_Iarpl, + UD_Iblendpd, + UD_Iblendps, + UD_Iblendvpd, + UD_Iblendvps, + UD_Ibound, + UD_Ibsf, + UD_Ibsr, + UD_Ibswap, + UD_Ibt, + UD_Ibtc, + UD_Ibtr, + UD_Ibts, + UD_Icall, + UD_Icbw, + UD_Icdq, + UD_Icdqe, + UD_Iclc, + UD_Icld, + UD_Iclflush, + UD_Iclgi, + UD_Icli, + UD_Iclts, + UD_Icmc, + UD_Icmova, + UD_Icmovae, + UD_Icmovb, + UD_Icmovbe, + UD_Icmovg, + UD_Icmovge, + UD_Icmovl, + UD_Icmovle, + UD_Icmovno, + UD_Icmovnp, + UD_Icmovns, + UD_Icmovnz, + UD_Icmovo, + UD_Icmovp, + UD_Icmovs, + UD_Icmovz, + UD_Icmp, + UD_Icmppd, + UD_Icmpps, + UD_Icmpsb, + UD_Icmpsd, + UD_Icmpsq, + UD_Icmpss, + UD_Icmpsw, + UD_Icmpxchg, + UD_Icmpxchg16b, + UD_Icmpxchg8b, + UD_Icomisd, + UD_Icomiss, + UD_Icpuid, + UD_Icqo, + UD_Icrc32, + UD_Icvtdq2pd, + UD_Icvtdq2ps, + UD_Icvtpd2dq, + UD_Icvtpd2pi, + UD_Icvtpd2ps, + UD_Icvtpi2pd, + UD_Icvtpi2ps, + UD_Icvtps2dq, + UD_Icvtps2pd, + UD_Icvtps2pi, + UD_Icvtsd2si, + UD_Icvtsd2ss, + UD_Icvtsi2sd, + UD_Icvtsi2ss, + UD_Icvtss2sd, + UD_Icvtss2si, + UD_Icvttpd2dq, + UD_Icvttpd2pi, + UD_Icvttps2dq, + UD_Icvttps2pi, + UD_Icvttsd2si, + UD_Icvttss2si, + UD_Icwd, + UD_Icwde, + UD_Idaa, + UD_Idas, + UD_Idec, + UD_Idiv, + UD_Idivpd, + UD_Idivps, + UD_Idivsd, + UD_Idivss, + UD_Idppd, + UD_Idpps, + UD_Iemms, + UD_Ienter, + UD_Iextractps, + UD_If2xm1, + UD_Ifabs, + UD_Ifadd, + UD_Ifaddp, + UD_Ifbld, + UD_Ifbstp, + UD_Ifchs, + UD_Ifclex, + UD_Ifcmovb, + UD_Ifcmovbe, + UD_Ifcmove, + UD_Ifcmovnb, + UD_Ifcmovnbe, + UD_Ifcmovne, + UD_Ifcmovnu, + UD_Ifcmovu, + UD_Ifcom, + UD_Ifcom2, + UD_Ifcomi, + UD_Ifcomip, + UD_Ifcomp, + UD_Ifcomp3, + UD_Ifcomp5, + UD_Ifcompp, + UD_Ifcos, + UD_Ifdecstp, + UD_Ifdiv, + UD_Ifdivp, + UD_Ifdivr, + UD_Ifdivrp, + UD_Ifemms, + UD_Iffree, + UD_Iffreep, + UD_Ifiadd, + UD_Ificom, + UD_Ificomp, + UD_Ifidiv, + UD_Ifidivr, + UD_Ifild, + UD_Ifimul, + UD_Ifincstp, + UD_Ifist, + UD_Ifistp, + UD_Ifisttp, + UD_Ifisub, + UD_Ifisubr, + UD_Ifld, + UD_Ifld1, + UD_Ifldcw, + UD_Ifldenv, + UD_Ifldl2e, + UD_Ifldl2t, + UD_Ifldlg2, + UD_Ifldln2, + UD_Ifldpi, + UD_Ifldz, + UD_Ifmul, + UD_Ifmulp, + UD_Ifninit, + UD_Ifnop, + UD_Ifnsave, + UD_Ifnstcw, + UD_Ifnstenv, + UD_Ifnstsw, + UD_Ifpatan, + UD_Ifprem, + UD_Ifprem1, + UD_Ifptan, + UD_Ifrndint, + UD_Ifrstor, + UD_Ifscale, + UD_Ifsin, + UD_Ifsincos, + UD_Ifsqrt, + UD_Ifst, + UD_Ifstp, + UD_Ifstp1, + UD_Ifstp8, + UD_Ifstp9, + UD_Ifsub, + UD_Ifsubp, + UD_Ifsubr, + UD_Ifsubrp, + UD_Iftst, + UD_Ifucom, + UD_Ifucomi, + UD_Ifucomip, + UD_Ifucomp, + UD_Ifucompp, + UD_Ifxam, + UD_Ifxch, + UD_Ifxch4, + UD_Ifxch7, + UD_Ifxrstor, + UD_Ifxsave, + UD_Ifxtract, + UD_Ifyl2x, + UD_Ifyl2xp1, + UD_Igetsec, + UD_Ihaddpd, + UD_Ihaddps, + UD_Ihlt, + UD_Ihsubpd, + UD_Ihsubps, + UD_Iidiv, + UD_Iimul, + UD_Iin, + UD_Iinc, + UD_Iinsb, + UD_Iinsd, + UD_Iinsertps, + UD_Iinsw, + UD_Iint, + UD_Iint1, + UD_Iint3, + UD_Iinto, + UD_Iinvd, + UD_Iinvept, + UD_Iinvlpg, + UD_Iinvlpga, + UD_Iinvvpid, + UD_Iiretd, + UD_Iiretq, + UD_Iiretw, + UD_Ija, + UD_Ijae, + UD_Ijb, + UD_Ijbe, + UD_Ijcxz, + UD_Ijecxz, + UD_Ijg, + UD_Ijge, + UD_Ijl, + UD_Ijle, + UD_Ijmp, + UD_Ijno, + UD_Ijnp, + UD_Ijns, + UD_Ijnz, + UD_Ijo, + UD_Ijp, + UD_Ijrcxz, + UD_Ijs, + UD_Ijz, + UD_Ilahf, + UD_Ilar, + UD_Ilddqu, + UD_Ildmxcsr, + UD_Ilds, + UD_Ilea, + UD_Ileave, + UD_Iles, + UD_Ilfence, + UD_Ilfs, + UD_Ilgdt, + UD_Ilgs, + UD_Ilidt, + UD_Illdt, + UD_Ilmsw, + UD_Ilock, + UD_Ilodsb, + UD_Ilodsd, + UD_Ilodsq, + UD_Ilodsw, + UD_Iloop, + UD_Iloope, + UD_Iloopne, + UD_Ilsl, + UD_Ilss, + UD_Iltr, + UD_Imaskmovdqu, + UD_Imaskmovq, + UD_Imaxpd, + UD_Imaxps, + UD_Imaxsd, + UD_Imaxss, + UD_Imfence, + UD_Iminpd, + UD_Iminps, + UD_Iminsd, + UD_Iminss, + UD_Imonitor, + UD_Imontmul, + UD_Imov, + UD_Imovapd, + UD_Imovaps, + UD_Imovbe, + UD_Imovd, + UD_Imovddup, + UD_Imovdq2q, + UD_Imovdqa, + UD_Imovdqu, + UD_Imovhlps, + UD_Imovhpd, + UD_Imovhps, + UD_Imovlhps, + UD_Imovlpd, + UD_Imovlps, + UD_Imovmskpd, + UD_Imovmskps, + UD_Imovntdq, + UD_Imovntdqa, + UD_Imovnti, + UD_Imovntpd, + UD_Imovntps, + UD_Imovntq, + UD_Imovq, + UD_Imovq2dq, + UD_Imovsb, + UD_Imovsd, + UD_Imovshdup, + UD_Imovsldup, + UD_Imovsq, + UD_Imovss, + UD_Imovsw, + UD_Imovsx, + UD_Imovsxd, + UD_Imovupd, + UD_Imovups, + UD_Imovzx, + UD_Impsadbw, + UD_Imul, + UD_Imulpd, + UD_Imulps, + UD_Imulsd, + UD_Imulss, + UD_Imwait, + UD_Ineg, + UD_Inop, + UD_Inot, + UD_Ior, + UD_Iorpd, + UD_Iorps, + UD_Iout, + UD_Ioutsb, + UD_Ioutsd, + UD_Ioutsw, + UD_Ipabsb, + UD_Ipabsd, + UD_Ipabsw, + UD_Ipackssdw, + UD_Ipacksswb, + UD_Ipackusdw, + UD_Ipackuswb, + UD_Ipaddb, + UD_Ipaddd, + UD_Ipaddq, + UD_Ipaddsb, + UD_Ipaddsw, + UD_Ipaddusb, + UD_Ipaddusw, + UD_Ipaddw, + UD_Ipalignr, + UD_Ipand, + UD_Ipandn, + UD_Ipavgb, + UD_Ipavgusb, + UD_Ipavgw, + UD_Ipblendvb, + UD_Ipblendw, + UD_Ipclmulqdq, + UD_Ipcmpeqb, + UD_Ipcmpeqd, + UD_Ipcmpeqq, + UD_Ipcmpeqw, + UD_Ipcmpestri, + UD_Ipcmpestrm, + UD_Ipcmpgtb, + UD_Ipcmpgtd, + UD_Ipcmpgtq, + UD_Ipcmpgtw, + UD_Ipcmpistri, + UD_Ipcmpistrm, + UD_Ipextrb, + UD_Ipextrd, + UD_Ipextrq, + UD_Ipextrw, + UD_Ipf2id, + UD_Ipf2iw, + UD_Ipfacc, + UD_Ipfadd, + UD_Ipfcmpeq, + UD_Ipfcmpge, + UD_Ipfcmpgt, + UD_Ipfmax, + UD_Ipfmin, + UD_Ipfmul, + UD_Ipfnacc, + UD_Ipfpnacc, + UD_Ipfrcp, + UD_Ipfrcpit1, + UD_Ipfrcpit2, + UD_Ipfrsqit1, + UD_Ipfrsqrt, + UD_Ipfsub, + UD_Ipfsubr, + UD_Iphaddd, + UD_Iphaddsw, + UD_Iphaddw, + UD_Iphminposuw, + UD_Iphsubd, + UD_Iphsubsw, + UD_Iphsubw, + UD_Ipi2fd, + UD_Ipi2fw, + UD_Ipinsrb, + UD_Ipinsrd, + UD_Ipinsrq, + UD_Ipinsrw, + UD_Ipmaddubsw, + UD_Ipmaddwd, + UD_Ipmaxsb, + UD_Ipmaxsd, + UD_Ipmaxsw, + UD_Ipmaxub, + UD_Ipmaxud, + UD_Ipmaxuw, + UD_Ipminsb, + UD_Ipminsd, + UD_Ipminsw, + UD_Ipminub, + UD_Ipminud, + UD_Ipminuw, + UD_Ipmovmskb, + UD_Ipmovsxbd, + UD_Ipmovsxbq, + UD_Ipmovsxbw, + UD_Ipmovsxdq, + UD_Ipmovsxwd, + UD_Ipmovsxwq, + UD_Ipmovzxbd, + UD_Ipmovzxbq, + UD_Ipmovzxbw, + UD_Ipmovzxdq, + UD_Ipmovzxwd, + UD_Ipmovzxwq, + UD_Ipmuldq, + UD_Ipmulhrsw, + UD_Ipmulhrw, + UD_Ipmulhuw, + UD_Ipmulhw, + UD_Ipmulld, + UD_Ipmullw, + UD_Ipmuludq, + UD_Ipop, + UD_Ipopa, + UD_Ipopad, + UD_Ipopcnt, + UD_Ipopfd, + UD_Ipopfq, + UD_Ipopfw, + UD_Ipor, + UD_Iprefetch, + UD_Iprefetchnta, + UD_Iprefetcht0, + UD_Iprefetcht1, + UD_Iprefetcht2, + UD_Ipsadbw, + UD_Ipshufb, + UD_Ipshufd, + UD_Ipshufhw, + UD_Ipshuflw, + UD_Ipshufw, + UD_Ipsignb, + UD_Ipsignd, + UD_Ipsignw, + UD_Ipslld, + UD_Ipslldq, + UD_Ipsllq, + UD_Ipsllw, + UD_Ipsrad, + UD_Ipsraw, + UD_Ipsrld, + UD_Ipsrldq, + UD_Ipsrlq, + UD_Ipsrlw, + UD_Ipsubb, + UD_Ipsubd, + UD_Ipsubq, + UD_Ipsubsb, + UD_Ipsubsw, + UD_Ipsubusb, + UD_Ipsubusw, + UD_Ipsubw, + UD_Ipswapd, + UD_Iptest, + UD_Ipunpckhbw, + UD_Ipunpckhdq, + UD_Ipunpckhqdq, + UD_Ipunpckhwd, + UD_Ipunpcklbw, + UD_Ipunpckldq, + UD_Ipunpcklqdq, + UD_Ipunpcklwd, + UD_Ipush, + UD_Ipusha, + UD_Ipushad, + UD_Ipushfd, + UD_Ipushfq, + UD_Ipushfw, + UD_Ipxor, + UD_Ircl, + UD_Ircpps, + UD_Ircpss, + UD_Ircr, + UD_Irdmsr, + UD_Irdpmc, + UD_Irdrand, + UD_Irdtsc, + UD_Irdtscp, + UD_Irep, + UD_Irepne, + UD_Iret, + UD_Iretf, + UD_Irol, + UD_Iror, + UD_Iroundpd, + UD_Iroundps, + UD_Iroundsd, + UD_Iroundss, + UD_Irsm, + UD_Irsqrtps, + UD_Irsqrtss, + UD_Isahf, + UD_Isalc, + UD_Isar, + UD_Isbb, + UD_Iscasb, + UD_Iscasd, + UD_Iscasq, + UD_Iscasw, + UD_Iseta, + UD_Isetae, + UD_Isetb, + UD_Isetbe, + UD_Isetg, + UD_Isetge, + UD_Isetl, + UD_Isetle, + UD_Isetno, + UD_Isetnp, + UD_Isetns, + UD_Isetnz, + UD_Iseto, + UD_Isetp, + UD_Isets, + UD_Isetz, + UD_Isfence, + UD_Isgdt, + UD_Ishl, + UD_Ishld, + UD_Ishr, + UD_Ishrd, + UD_Ishufpd, + UD_Ishufps, + UD_Isidt, + UD_Iskinit, + UD_Isldt, + UD_Ismsw, + UD_Isqrtpd, + UD_Isqrtps, + UD_Isqrtsd, + UD_Isqrtss, + UD_Istc, + UD_Istd, + UD_Istgi, + UD_Isti, + UD_Istmxcsr, + UD_Istosb, + UD_Istosd, + UD_Istosq, + UD_Istosw, + UD_Istr, + UD_Isub, + UD_Isubpd, + UD_Isubps, + UD_Isubsd, + UD_Isubss, + UD_Iswapgs, + UD_Isyscall, + UD_Isysenter, + UD_Isysexit, + UD_Isysret, + UD_Itest, + UD_Iucomisd, + UD_Iucomiss, + UD_Iud2, + UD_Iunpckhpd, + UD_Iunpckhps, + UD_Iunpcklpd, + UD_Iunpcklps, + UD_Ivaddpd, + UD_Ivaddps, + UD_Ivaddsd, + UD_Ivaddss, + UD_Ivaddsubpd, + UD_Ivaddsubps, + UD_Ivaesdec, + UD_Ivaesdeclast, + UD_Ivaesenc, + UD_Ivaesenclast, + UD_Ivaesimc, + UD_Ivaeskeygenassist, + UD_Ivandnpd, + UD_Ivandnps, + UD_Ivandpd, + UD_Ivandps, + UD_Ivblendpd, + UD_Ivblendps, + UD_Ivblendvpd, + UD_Ivblendvps, + UD_Ivbroadcastsd, + UD_Ivbroadcastss, + UD_Ivcmppd, + UD_Ivcmpps, + UD_Ivcmpsd, + UD_Ivcmpss, + UD_Ivcomisd, + UD_Ivcomiss, + UD_Ivcvtdq2pd, + UD_Ivcvtdq2ps, + UD_Ivcvtpd2dq, + UD_Ivcvtpd2ps, + UD_Ivcvtps2dq, + UD_Ivcvtps2pd, + UD_Ivcvtsd2si, + UD_Ivcvtsd2ss, + UD_Ivcvtsi2sd, + UD_Ivcvtsi2ss, + UD_Ivcvtss2sd, + UD_Ivcvtss2si, + UD_Ivcvttpd2dq, + UD_Ivcvttps2dq, + UD_Ivcvttsd2si, + UD_Ivcvttss2si, + UD_Ivdivpd, + UD_Ivdivps, + UD_Ivdivsd, + UD_Ivdivss, + UD_Ivdppd, + UD_Ivdpps, + UD_Iverr, + UD_Iverw, + UD_Ivextractf128, + UD_Ivextractps, + UD_Ivhaddpd, + UD_Ivhaddps, + UD_Ivhsubpd, + UD_Ivhsubps, + UD_Ivinsertf128, + UD_Ivinsertps, + UD_Ivlddqu, + UD_Ivmaskmovdqu, + UD_Ivmaskmovpd, + UD_Ivmaskmovps, + UD_Ivmaxpd, + UD_Ivmaxps, + UD_Ivmaxsd, + UD_Ivmaxss, + UD_Ivmcall, + UD_Ivmclear, + UD_Ivminpd, + UD_Ivminps, + UD_Ivminsd, + UD_Ivminss, + UD_Ivmlaunch, + UD_Ivmload, + UD_Ivmmcall, + UD_Ivmovapd, + UD_Ivmovaps, + UD_Ivmovd, + UD_Ivmovddup, + UD_Ivmovdqa, + UD_Ivmovdqu, + UD_Ivmovhlps, + UD_Ivmovhpd, + UD_Ivmovhps, + UD_Ivmovlhps, + UD_Ivmovlpd, + UD_Ivmovlps, + UD_Ivmovmskpd, + UD_Ivmovmskps, + UD_Ivmovntdq, + UD_Ivmovntdqa, + UD_Ivmovntpd, + UD_Ivmovntps, + UD_Ivmovq, + UD_Ivmovsd, + UD_Ivmovshdup, + UD_Ivmovsldup, + UD_Ivmovss, + UD_Ivmovupd, + UD_Ivmovups, + UD_Ivmpsadbw, + UD_Ivmptrld, + UD_Ivmptrst, + UD_Ivmread, + UD_Ivmresume, + UD_Ivmrun, + UD_Ivmsave, + UD_Ivmulpd, + UD_Ivmulps, + UD_Ivmulsd, + UD_Ivmulss, + UD_Ivmwrite, + UD_Ivmxoff, + UD_Ivmxon, + UD_Ivorpd, + UD_Ivorps, + UD_Ivpabsb, + UD_Ivpabsd, + UD_Ivpabsw, + UD_Ivpackssdw, + UD_Ivpacksswb, + UD_Ivpackusdw, + UD_Ivpackuswb, + UD_Ivpaddb, + UD_Ivpaddd, + UD_Ivpaddq, + UD_Ivpaddsb, + UD_Ivpaddsw, + UD_Ivpaddusb, + UD_Ivpaddusw, + UD_Ivpaddw, + UD_Ivpalignr, + UD_Ivpand, + UD_Ivpandn, + UD_Ivpavgb, + UD_Ivpavgw, + UD_Ivpblendvb, + UD_Ivpblendw, + UD_Ivpclmulqdq, + UD_Ivpcmpeqb, + UD_Ivpcmpeqd, + UD_Ivpcmpeqq, + UD_Ivpcmpeqw, + UD_Ivpcmpestri, + UD_Ivpcmpestrm, + UD_Ivpcmpgtb, + UD_Ivpcmpgtd, + UD_Ivpcmpgtq, + UD_Ivpcmpgtw, + UD_Ivpcmpistri, + UD_Ivpcmpistrm, + UD_Ivperm2f128, + UD_Ivpermilpd, + UD_Ivpermilps, + UD_Ivpextrb, + UD_Ivpextrd, + UD_Ivpextrq, + UD_Ivpextrw, + UD_Ivphaddd, + UD_Ivphaddsw, + UD_Ivphaddw, + UD_Ivphminposuw, + UD_Ivphsubd, + UD_Ivphsubsw, + UD_Ivphsubw, + UD_Ivpinsrb, + UD_Ivpinsrd, + UD_Ivpinsrq, + UD_Ivpinsrw, + UD_Ivpmaddubsw, + UD_Ivpmaddwd, + UD_Ivpmaxsb, + UD_Ivpmaxsd, + UD_Ivpmaxsw, + UD_Ivpmaxub, + UD_Ivpmaxud, + UD_Ivpmaxuw, + UD_Ivpminsb, + UD_Ivpminsd, + UD_Ivpminsw, + UD_Ivpminub, + UD_Ivpminud, + UD_Ivpminuw, + UD_Ivpmovmskb, + UD_Ivpmovsxbd, + UD_Ivpmovsxbq, + UD_Ivpmovsxbw, + UD_Ivpmovsxwd, + UD_Ivpmovsxwq, + UD_Ivpmovzxbd, + UD_Ivpmovzxbq, + UD_Ivpmovzxbw, + UD_Ivpmovzxdq, + UD_Ivpmovzxwd, + UD_Ivpmovzxwq, + UD_Ivpmuldq, + UD_Ivpmulhrsw, + UD_Ivpmulhuw, + UD_Ivpmulhw, + UD_Ivpmulld, + UD_Ivpmullw, + UD_Ivpor, + UD_Ivpsadbw, + UD_Ivpshufb, + UD_Ivpshufd, + UD_Ivpshufhw, + UD_Ivpshuflw, + UD_Ivpsignb, + UD_Ivpsignd, + UD_Ivpsignw, + UD_Ivpslld, + UD_Ivpslldq, + UD_Ivpsllq, + UD_Ivpsllw, + UD_Ivpsrad, + UD_Ivpsraw, + UD_Ivpsrld, + UD_Ivpsrldq, + UD_Ivpsrlq, + UD_Ivpsrlw, + UD_Ivpsubb, + UD_Ivpsubd, + UD_Ivpsubq, + UD_Ivpsubsb, + UD_Ivpsubsw, + UD_Ivpsubusb, + UD_Ivpsubusw, + UD_Ivpsubw, + UD_Ivptest, + UD_Ivpunpckhbw, + UD_Ivpunpckhdq, + UD_Ivpunpckhqdq, + UD_Ivpunpckhwd, + UD_Ivpunpcklbw, + UD_Ivpunpckldq, + UD_Ivpunpcklqdq, + UD_Ivpunpcklwd, + UD_Ivpxor, + UD_Ivrcpps, + UD_Ivrcpss, + UD_Ivroundpd, + UD_Ivroundps, + UD_Ivroundsd, + UD_Ivroundss, + UD_Ivrsqrtps, + UD_Ivrsqrtss, + UD_Ivshufpd, + UD_Ivshufps, + UD_Ivsqrtpd, + UD_Ivsqrtps, + UD_Ivsqrtsd, + UD_Ivsqrtss, + UD_Ivstmxcsr, + UD_Ivsubpd, + UD_Ivsubps, + UD_Ivsubsd, + UD_Ivsubss, + UD_Ivtestpd, + UD_Ivtestps, + UD_Ivucomisd, + UD_Ivucomiss, + UD_Ivunpckhpd, + UD_Ivunpckhps, + UD_Ivunpcklpd, + UD_Ivunpcklps, + UD_Ivxorpd, + UD_Ivxorps, + UD_Ivzeroall, + UD_Ivzeroupper, + UD_Iwait, + UD_Iwbinvd, + UD_Iwrmsr, + UD_Ixadd, + UD_Ixchg, + UD_Ixcryptcbc, + UD_Ixcryptcfb, + UD_Ixcryptctr, + UD_Ixcryptecb, + UD_Ixcryptofb, + UD_Ixgetbv, + UD_Ixlatb, + UD_Ixor, + UD_Ixorpd, + UD_Ixorps, + UD_Ixrstor, + UD_Ixsave, + UD_Ixsetbv, + UD_Ixsha1, + UD_Ixsha256, + UD_Ixstore, + UD_Iinvalid, + UD_I3dnow, + UD_Inone, + UD_Idb, + UD_Ipause, + UD_MAX_MNEMONIC_CODE +} UD_ATTR_PACKED; + +extern const char * ud_mnemonics_str[]; + +#endif /* UD_ITAB_H */ diff --git a/ext/udis86/syn-att.c b/ext/udis86/syn-att.c new file mode 100644 index 0000000000..cc598389e2 --- /dev/null +++ b/ext/udis86/syn-att.c @@ -0,0 +1,228 @@ +/* udis86 - libudis86/syn-att.c + * + * Copyright (c) 2002-2009 Vivek Thampi + * All rights reserved. + * + * Redistribution and use in source and binary forms, with or without modification, + * are permitted provided that the following conditions are met: + * + * * Redistributions of source code must retain the above copyright notice, + * this list of conditions and the following disclaimer. + * * Redistributions in binary form must reproduce the above copyright notice, + * this list of conditions and the following disclaimer in the documentation + * and/or other materials provided with the distribution. + * + * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND + * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED + * WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE + * DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE FOR + * ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES + * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; + * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON + * ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT + * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + */ +#include "types.h" +#include "extern.h" +#include "decode.h" +#include "itab.h" +#include "syn.h" +#include "udint.h" + +/* ----------------------------------------------------------------------------- + * opr_cast() - Prints an operand cast. + * ----------------------------------------------------------------------------- + */ +static void +opr_cast(struct ud* u, struct ud_operand* op) +{ + switch(op->size) { + case 16 : case 32 : + ud_asmprintf(u, "*"); break; + default: break; + } +} + +/* ----------------------------------------------------------------------------- + * gen_operand() - Generates assembly output for each operand. + * ----------------------------------------------------------------------------- + */ +static void +gen_operand(struct ud* u, struct ud_operand* op) +{ + switch(op->type) { + case UD_OP_CONST: + ud_asmprintf(u, "$0x%x", op->lval.udword); + break; + + case UD_OP_REG: + ud_asmprintf(u, "%%%s", ud_reg_tab[op->base - UD_R_AL]); + break; + + case UD_OP_MEM: + if (u->br_far) { + opr_cast(u, op); + } + if (u->pfx_seg) { + ud_asmprintf(u, "%%%s:", ud_reg_tab[u->pfx_seg - UD_R_AL]); + } + if (op->offset != 0) { + ud_syn_print_mem_disp(u, op, 0); + } + if (op->base) { + ud_asmprintf(u, "(%%%s", ud_reg_tab[op->base - UD_R_AL]); + } + if (op->index) { + if (op->base) { + ud_asmprintf(u, ","); + } else { + ud_asmprintf(u, "("); + } + ud_asmprintf(u, "%%%s", ud_reg_tab[op->index - UD_R_AL]); + } + if (op->scale) { + ud_asmprintf(u, ",%d", op->scale); + } + if (op->base || op->index) { + ud_asmprintf(u, ")"); + } + break; + + case UD_OP_IMM: + ud_asmprintf(u, "$"); + ud_syn_print_imm(u, op); + break; + + case UD_OP_JIMM: + ud_syn_print_addr(u, ud_syn_rel_target(u, op)); + break; + + case UD_OP_PTR: + switch (op->size) { + case 32: + ud_asmprintf(u, "$0x%x, $0x%x", op->lval.ptr.seg, + op->lval.ptr.off & 0xFFFF); + break; + case 48: + ud_asmprintf(u, "$0x%x, $0x%x", op->lval.ptr.seg, + op->lval.ptr.off); + break; + } + break; + + default: return; + } +} + +/* ============================================================================= + * translates to AT&T syntax + * ============================================================================= + */ +extern void +ud_translate_att(struct ud *u) +{ + int size = 0; + int star = 0; + + /* check if P_OSO prefix is used */ + if (! P_OSO(u->itab_entry->prefix) && u->pfx_opr) { + switch (u->dis_mode) { + case 16: + ud_asmprintf(u, "o32 "); + break; + case 32: + case 64: + ud_asmprintf(u, "o16 "); + break; + } + } + + /* check if P_ASO prefix was used */ + if (! P_ASO(u->itab_entry->prefix) && u->pfx_adr) { + switch (u->dis_mode) { + case 16: + ud_asmprintf(u, "a32 "); + break; + case 32: + ud_asmprintf(u, "a16 "); + break; + case 64: + ud_asmprintf(u, "a32 "); + break; + } + } + + if (u->pfx_lock) + ud_asmprintf(u, "lock "); + if (u->pfx_rep) { + ud_asmprintf(u, "rep "); + } else if (u->pfx_rep) { + ud_asmprintf(u, "repe "); + } else if (u->pfx_repne) { + ud_asmprintf(u, "repne "); + } + + /* special instructions */ + switch (u->mnemonic) { + case UD_Iretf: + ud_asmprintf(u, "lret "); + break; + case UD_Idb: + ud_asmprintf(u, ".byte 0x%x", u->operand[0].lval.ubyte); + return; + case UD_Ijmp: + case UD_Icall: + if (u->br_far) ud_asmprintf(u, "l"); + if (u->operand[0].type == UD_OP_REG) { + star = 1; + } + ud_asmprintf(u, "%s", ud_lookup_mnemonic(u->mnemonic)); + break; + case UD_Ibound: + case UD_Ienter: + if (u->operand[0].type != UD_NONE) + gen_operand(u, &u->operand[0]); + if (u->operand[1].type != UD_NONE) { + ud_asmprintf(u, ","); + gen_operand(u, &u->operand[1]); + } + return; + default: + ud_asmprintf(u, "%s", ud_lookup_mnemonic(u->mnemonic)); + } + + if (size == 8) { + ud_asmprintf(u, "b"); + } else if (size == 16) { + ud_asmprintf(u, "w"); + } else if (size == 64) { + ud_asmprintf(u, "q"); + } + + if (star) { + ud_asmprintf(u, " *"); + } else { + ud_asmprintf(u, " "); + } + + if (u->operand[3].type != UD_NONE) { + gen_operand(u, &u->operand[3]); + ud_asmprintf(u, ", "); + } + if (u->operand[2].type != UD_NONE) { + gen_operand(u, &u->operand[2]); + ud_asmprintf(u, ", "); + } + if (u->operand[1].type != UD_NONE) { + gen_operand(u, &u->operand[1]); + ud_asmprintf(u, ", "); + } + if (u->operand[0].type != UD_NONE) { + gen_operand(u, &u->operand[0]); + } +} + +/* +vim: set ts=2 sw=2 expandtab +*/ diff --git a/ext/udis86/syn-intel.c b/ext/udis86/syn-intel.c new file mode 100644 index 0000000000..0664fea092 --- /dev/null +++ b/ext/udis86/syn-intel.c @@ -0,0 +1,224 @@ +/* udis86 - libudis86/syn-intel.c + * + * Copyright (c) 2002-2013 Vivek Thampi + * All rights reserved. + * + * Redistribution and use in source and binary forms, with or without modification, + * are permitted provided that the following conditions are met: + * + * * Redistributions of source code must retain the above copyright notice, + * this list of conditions and the following disclaimer. + * * Redistributions in binary form must reproduce the above copyright notice, + * this list of conditions and the following disclaimer in the documentation + * and/or other materials provided with the distribution. + * + * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND + * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED + * WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE + * DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE FOR + * ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES + * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; + * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON + * ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT + * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + */ +#include "types.h" +#include "extern.h" +#include "decode.h" +#include "itab.h" +#include "syn.h" +#include "udint.h" + +/* ----------------------------------------------------------------------------- + * opr_cast() - Prints an operand cast. + * ----------------------------------------------------------------------------- + */ +static void +opr_cast(struct ud* u, struct ud_operand* op) +{ + if (u->br_far) { + ud_asmprintf(u, "far "); + } + switch(op->size) { + case 8: ud_asmprintf(u, "byte " ); break; + case 16: ud_asmprintf(u, "word " ); break; + case 32: ud_asmprintf(u, "dword "); break; + case 64: ud_asmprintf(u, "qword "); break; + case 80: ud_asmprintf(u, "tword "); break; + case 128: ud_asmprintf(u, "oword "); break; + case 256: ud_asmprintf(u, "yword "); break; + default: break; + } +} + +/* ----------------------------------------------------------------------------- + * gen_operand() - Generates assembly output for each operand. + * ----------------------------------------------------------------------------- + */ +static void gen_operand(struct ud* u, struct ud_operand* op, int syn_cast) +{ + switch(op->type) { + case UD_OP_REG: + ud_asmprintf(u, "%s", ud_reg_tab[op->base - UD_R_AL]); + break; + + case UD_OP_MEM: + if (syn_cast) { + opr_cast(u, op); + } + ud_asmprintf(u, "["); + if (u->pfx_seg) { + ud_asmprintf(u, "%s:", ud_reg_tab[u->pfx_seg - UD_R_AL]); + } + if (op->base) { + ud_asmprintf(u, "%s", ud_reg_tab[op->base - UD_R_AL]); + } + if (op->index) { + ud_asmprintf(u, "%s%s", op->base != UD_NONE? "+" : "", + ud_reg_tab[op->index - UD_R_AL]); + if (op->scale) { + ud_asmprintf(u, "*%d", op->scale); + } + } + if (op->offset != 0) { + ud_syn_print_mem_disp(u, op, (op->base != UD_NONE || + op->index != UD_NONE) ? 1 : 0); + } + ud_asmprintf(u, "]"); + break; + + case UD_OP_IMM: + ud_syn_print_imm(u, op); + break; + + + case UD_OP_JIMM: + ud_syn_print_addr(u, ud_syn_rel_target(u, op)); + break; + + case UD_OP_PTR: + switch (op->size) { + case 32: + ud_asmprintf(u, "word 0x%x:0x%x", op->lval.ptr.seg, + op->lval.ptr.off & 0xFFFF); + break; + case 48: + ud_asmprintf(u, "dword 0x%x:0x%x", op->lval.ptr.seg, + op->lval.ptr.off); + break; + } + break; + + case UD_OP_CONST: + if (syn_cast) opr_cast(u, op); + ud_asmprintf(u, "%d", op->lval.udword); + break; + + default: return; + } +} + +/* ============================================================================= + * translates to intel syntax + * ============================================================================= + */ +extern void +ud_translate_intel(struct ud* u) +{ + /* check if P_OSO prefix is used */ + if (!P_OSO(u->itab_entry->prefix) && u->pfx_opr) { + switch (u->dis_mode) { + case 16: ud_asmprintf(u, "o32 "); break; + case 32: + case 64: ud_asmprintf(u, "o16 "); break; + } + } + + /* check if P_ASO prefix was used */ + if (!P_ASO(u->itab_entry->prefix) && u->pfx_adr) { + switch (u->dis_mode) { + case 16: ud_asmprintf(u, "a32 "); break; + case 32: ud_asmprintf(u, "a16 "); break; + case 64: ud_asmprintf(u, "a32 "); break; + } + } + + if (u->pfx_seg && + u->operand[0].type != UD_OP_MEM && + u->operand[1].type != UD_OP_MEM ) { + ud_asmprintf(u, "%s ", ud_reg_tab[u->pfx_seg - UD_R_AL]); + } + + if (u->pfx_lock) { + ud_asmprintf(u, "lock "); + } + if (u->pfx_rep) { + ud_asmprintf(u, "rep "); + } else if (u->pfx_repe) { + ud_asmprintf(u, "repe "); + } else if (u->pfx_repne) { + ud_asmprintf(u, "repne "); + } + + /* print the instruction mnemonic */ + ud_asmprintf(u, "%s", ud_lookup_mnemonic(u->mnemonic)); + + if (u->operand[0].type != UD_NONE) { + int cast = 0; + ud_asmprintf(u, " "); + if (u->operand[0].type == UD_OP_MEM) { + if (u->operand[1].type == UD_OP_IMM || + u->operand[1].type == UD_OP_CONST || + u->operand[1].type == UD_NONE || + (u->operand[0].size != u->operand[1].size)) { + cast = 1; + } else if (u->operand[1].type == UD_OP_REG && + u->operand[1].base == UD_R_CL) { + switch (u->mnemonic) { + case UD_Ircl: + case UD_Irol: + case UD_Iror: + case UD_Ircr: + case UD_Ishl: + case UD_Ishr: + case UD_Isar: + cast = 1; + break; + default: break; + } + } + } + gen_operand(u, &u->operand[0], cast); + } + + if (u->operand[1].type != UD_NONE) { + int cast = 0; + ud_asmprintf(u, ", "); + if (u->operand[1].type == UD_OP_MEM && + u->operand[0].size != u->operand[1].size && + !ud_opr_is_sreg(&u->operand[0])) { + cast = 1; + } + gen_operand(u, &u->operand[1], cast); + } + + if (u->operand[2].type != UD_NONE) { + int cast = 0; + ud_asmprintf(u, ", "); + if (u->operand[2].type == UD_OP_MEM && + u->operand[2].size != u->operand[1].size) { + cast = 1; + } + gen_operand(u, &u->operand[2], cast); + } + + if (u->operand[3].type != UD_NONE) { + ud_asmprintf(u, ", "); + gen_operand(u, &u->operand[3], 0); + } +} + +/* +vim: set ts=2 sw=2 expandtab +*/ diff --git a/ext/udis86/syn.c b/ext/udis86/syn.c new file mode 100644 index 0000000000..1b9e1d42a5 --- /dev/null +++ b/ext/udis86/syn.c @@ -0,0 +1,212 @@ +/* udis86 - libudis86/syn.c + * + * Copyright (c) 2002-2013 Vivek Thampi + * All rights reserved. + * + * Redistribution and use in source and binary forms, with or without modification, + * are permitted provided that the following conditions are met: + * + * * Redistributions of source code must retain the above copyright notice, + * this list of conditions and the following disclaimer. + * * Redistributions in binary form must reproduce the above copyright notice, + * this list of conditions and the following disclaimer in the documentation + * and/or other materials provided with the distribution. + * + * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND + * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED + * WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE + * DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE FOR + * ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES + * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; + * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON + * ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT + * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + */ +#include "types.h" +#include "decode.h" +#include "syn.h" +#include "udint.h" + +/* + * Register Table - Order Matters (types.h)! + * + */ +const char* ud_reg_tab[] = +{ + "al", "cl", "dl", "bl", + "ah", "ch", "dh", "bh", + "spl", "bpl", "sil", "dil", + "r8b", "r9b", "r10b", "r11b", + "r12b", "r13b", "r14b", "r15b", + + "ax", "cx", "dx", "bx", + "sp", "bp", "si", "di", + "r8w", "r9w", "r10w", "r11w", + "r12w", "r13w", "r14w", "r15w", + + "eax", "ecx", "edx", "ebx", + "esp", "ebp", "esi", "edi", + "r8d", "r9d", "r10d", "r11d", + "r12d", "r13d", "r14d", "r15d", + + "rax", "rcx", "rdx", "rbx", + "rsp", "rbp", "rsi", "rdi", + "r8", "r9", "r10", "r11", + "r12", "r13", "r14", "r15", + + "es", "cs", "ss", "ds", + "fs", "gs", + + "cr0", "cr1", "cr2", "cr3", + "cr4", "cr5", "cr6", "cr7", + "cr8", "cr9", "cr10", "cr11", + "cr12", "cr13", "cr14", "cr15", + + "dr0", "dr1", "dr2", "dr3", + "dr4", "dr5", "dr6", "dr7", + "dr8", "dr9", "dr10", "dr11", + "dr12", "dr13", "dr14", "dr15", + + "mm0", "mm1", "mm2", "mm3", + "mm4", "mm5", "mm6", "mm7", + + "st0", "st1", "st2", "st3", + "st4", "st5", "st6", "st7", + + "xmm0", "xmm1", "xmm2", "xmm3", + "xmm4", "xmm5", "xmm6", "xmm7", + "xmm8", "xmm9", "xmm10", "xmm11", + "xmm12", "xmm13", "xmm14", "xmm15", + + "ymm0", "ymm1", "ymm2", "ymm3", + "ymm4", "ymm5", "ymm6", "ymm7", + "ymm8", "ymm9", "ymm10", "ymm11", + "ymm12", "ymm13", "ymm14", "ymm15", + + "rip" +}; + + +uint64_t +ud_syn_rel_target(struct ud *u, struct ud_operand *opr) +{ + const uint64_t trunc_mask = 0xffffffffffffffffull >> (64 - u->opr_mode); + switch (opr->size) { + case 8 : return (u->pc + opr->lval.sbyte) & trunc_mask; + case 16: return (u->pc + opr->lval.sword) & trunc_mask; + case 32: return (u->pc + opr->lval.sdword) & trunc_mask; + default: UD_ASSERT(!"invalid relative offset size."); + return 0ull; + } +} + + +/* + * asmprintf + * Printf style function for printing translated assembly + * output. Returns the number of characters written and + * moves the buffer pointer forward. On an overflow, + * returns a negative number and truncates the output. + */ +int +ud_asmprintf(struct ud *u, const char *fmt, ...) +{ + int ret; + int avail; + va_list ap; + va_start(ap, fmt); + avail = u->asm_buf_size - u->asm_buf_fill - 1 /* nullchar */; + ret = vsnprintf((char*) u->asm_buf + u->asm_buf_fill, avail, fmt, ap); + if (ret < 0 || ret > avail) { + u->asm_buf_fill = u->asm_buf_size - 1; + } else { + u->asm_buf_fill += ret; + } + va_end(ap); + return ret; +} + + +void +ud_syn_print_addr(struct ud *u, uint64_t addr) +{ + const char *name = NULL; + if (u->sym_resolver) { + int64_t offset = 0; + name = u->sym_resolver(u, addr, &offset); + if (name) { + if (offset) { + ud_asmprintf(u, "%s%+" FMT64 "d", name, offset); + } else { + ud_asmprintf(u, "%s", name); + } + return; + } + } + ud_asmprintf(u, "0x%" FMT64 "x", addr); +} + + +void +ud_syn_print_imm(struct ud* u, const struct ud_operand *op) +{ + uint64_t v; + if (op->_oprcode == OP_sI && op->size != u->opr_mode) { + if (op->size == 8) { + v = (int64_t)op->lval.sbyte; + } else { + UD_ASSERT(op->size == 32); + v = (int64_t)op->lval.sdword; + } + if (u->opr_mode < 64) { + v = v & ((1ull << u->opr_mode) - 1ull); + } + } else { + switch (op->size) { + case 8 : v = op->lval.ubyte; break; + case 16: v = op->lval.uword; break; + case 32: v = op->lval.udword; break; + case 64: v = op->lval.uqword; break; + default: UD_ASSERT(!"invalid offset"); v = 0; /* keep cc happy */ + } + } + ud_asmprintf(u, "0x%" FMT64 "x", v); +} + + +void +ud_syn_print_mem_disp(struct ud* u, const struct ud_operand *op, int sign) +{ + UD_ASSERT(op->offset != 0); + if (op->base == UD_NONE && op->index == UD_NONE) { + uint64_t v; + UD_ASSERT(op->scale == UD_NONE && op->offset != 8); + /* unsigned mem-offset */ + switch (op->offset) { + case 16: v = op->lval.uword; break; + case 32: v = op->lval.udword; break; + case 64: v = op->lval.uqword; break; + default: UD_ASSERT(!"invalid offset"); v = 0; /* keep cc happy */ + } + ud_asmprintf(u, "0x%" FMT64 "x", v); + } else { + int64_t v; + UD_ASSERT(op->offset != 64); + switch (op->offset) { + case 8 : v = op->lval.sbyte; break; + case 16: v = op->lval.sword; break; + case 32: v = op->lval.sdword; break; + default: UD_ASSERT(!"invalid offset"); v = 0; /* keep cc happy */ + } + if (v < 0) { + ud_asmprintf(u, "-0x%" FMT64 "x", -v); + } else if (v > 0) { + ud_asmprintf(u, "%s0x%" FMT64 "x", sign? "+" : "", v); + } + } +} + +/* +vim: set ts=2 sw=2 expandtab +*/ diff --git a/ext/udis86/syn.h b/ext/udis86/syn.h new file mode 100644 index 0000000000..d3b1e3fe04 --- /dev/null +++ b/ext/udis86/syn.h @@ -0,0 +1,53 @@ +/* udis86 - libudis86/syn.h + * + * Copyright (c) 2002-2009 + * All rights reserved. + * + * Redistribution and use in source and binary forms, with or without modification, + * are permitted provided that the following conditions are met: + * + * * Redistributions of source code must retain the above copyright notice, + * this list of conditions and the following disclaimer. + * * Redistributions in binary form must reproduce the above copyright notice, + * this list of conditions and the following disclaimer in the documentation + * and/or other materials provided with the distribution. + * + * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND + * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED + * WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE + * DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE FOR + * ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES + * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; + * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON + * ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT + * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + */ +#ifndef UD_SYN_H +#define UD_SYN_H + +#include "types.h" +#ifndef __UD_STANDALONE__ +# include +#endif /* __UD_STANDALONE__ */ + +extern const char* ud_reg_tab[]; + +uint64_t ud_syn_rel_target(struct ud*, struct ud_operand*); + +#ifdef __GNUC__ +int ud_asmprintf(struct ud *u, const char *fmt, ...) + __attribute__ ((format (printf, 2, 3))); +#else +int ud_asmprintf(struct ud *u, const char *fmt, ...); +#endif + +void ud_syn_print_addr(struct ud *u, uint64_t addr); +void ud_syn_print_imm(struct ud* u, const struct ud_operand *op); +void ud_syn_print_mem_disp(struct ud* u, const struct ud_operand *, int sign); + +#endif /* UD_SYN_H */ + +/* +vim: set ts=2 sw=2 expandtab +*/ diff --git a/ext/udis86/types.h b/ext/udis86/types.h new file mode 100644 index 0000000000..d79dae937b --- /dev/null +++ b/ext/udis86/types.h @@ -0,0 +1,259 @@ +/* udis86 - libudis86/types.h + * + * Copyright (c) 2002-2013 Vivek Thampi + * All rights reserved. + * + * Redistribution and use in source and binary forms, with or without modification, + * are permitted provided that the following conditions are met: + * + * * Redistributions of source code must retain the above copyright notice, + * this list of conditions and the following disclaimer. + * * Redistributions in binary form must reproduce the above copyright notice, + * this list of conditions and the following disclaimer in the documentation + * and/or other materials provided with the distribution. + * + * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND + * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED + * WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE + * DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE FOR + * ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES + * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; + * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON + * ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT + * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + */ +#ifndef UD_TYPES_H +#define UD_TYPES_H + +#ifdef __KERNEL__ + /* + * -D__KERNEL__ is automatically passed on the command line when + * building something as part of the Linux kernel. Assume standalone + * mode. + */ +# include +# include +# ifndef __UD_STANDALONE__ +# define __UD_STANDALONE__ 1 +# endif +#endif /* __KERNEL__ */ + +#if !defined(__UD_STANDALONE__) +# include +# include +#endif + +/* gcc specific extensions */ +#ifdef __GNUC__ +# define UD_ATTR_PACKED __attribute__((packed)) +#else +# define UD_ATTR_PACKED +#endif /* UD_ATTR_PACKED */ + + +/* ----------------------------------------------------------------------------- + * All possible "types" of objects in udis86. Order is Important! + * ----------------------------------------------------------------------------- + */ +enum ud_type +{ + UD_NONE, + + /* 8 bit GPRs */ + UD_R_AL, UD_R_CL, UD_R_DL, UD_R_BL, + UD_R_AH, UD_R_CH, UD_R_DH, UD_R_BH, + UD_R_SPL, UD_R_BPL, UD_R_SIL, UD_R_DIL, + UD_R_R8B, UD_R_R9B, UD_R_R10B, UD_R_R11B, + UD_R_R12B, UD_R_R13B, UD_R_R14B, UD_R_R15B, + + /* 16 bit GPRs */ + UD_R_AX, UD_R_CX, UD_R_DX, UD_R_BX, + UD_R_SP, UD_R_BP, UD_R_SI, UD_R_DI, + UD_R_R8W, UD_R_R9W, UD_R_R10W, UD_R_R11W, + UD_R_R12W, UD_R_R13W, UD_R_R14W, UD_R_R15W, + + /* 32 bit GPRs */ + UD_R_EAX, UD_R_ECX, UD_R_EDX, UD_R_EBX, + UD_R_ESP, UD_R_EBP, UD_R_ESI, UD_R_EDI, + UD_R_R8D, UD_R_R9D, UD_R_R10D, UD_R_R11D, + UD_R_R12D, UD_R_R13D, UD_R_R14D, UD_R_R15D, + + /* 64 bit GPRs */ + UD_R_RAX, UD_R_RCX, UD_R_RDX, UD_R_RBX, + UD_R_RSP, UD_R_RBP, UD_R_RSI, UD_R_RDI, + UD_R_R8, UD_R_R9, UD_R_R10, UD_R_R11, + UD_R_R12, UD_R_R13, UD_R_R14, UD_R_R15, + + /* segment registers */ + UD_R_ES, UD_R_CS, UD_R_SS, UD_R_DS, + UD_R_FS, UD_R_GS, + + /* control registers*/ + UD_R_CR0, UD_R_CR1, UD_R_CR2, UD_R_CR3, + UD_R_CR4, UD_R_CR5, UD_R_CR6, UD_R_CR7, + UD_R_CR8, UD_R_CR9, UD_R_CR10, UD_R_CR11, + UD_R_CR12, UD_R_CR13, UD_R_CR14, UD_R_CR15, + + /* debug registers */ + UD_R_DR0, UD_R_DR1, UD_R_DR2, UD_R_DR3, + UD_R_DR4, UD_R_DR5, UD_R_DR6, UD_R_DR7, + UD_R_DR8, UD_R_DR9, UD_R_DR10, UD_R_DR11, + UD_R_DR12, UD_R_DR13, UD_R_DR14, UD_R_DR15, + + /* mmx registers */ + UD_R_MM0, UD_R_MM1, UD_R_MM2, UD_R_MM3, + UD_R_MM4, UD_R_MM5, UD_R_MM6, UD_R_MM7, + + /* x87 registers */ + UD_R_ST0, UD_R_ST1, UD_R_ST2, UD_R_ST3, + UD_R_ST4, UD_R_ST5, UD_R_ST6, UD_R_ST7, + + /* extended multimedia registers */ + UD_R_XMM0, UD_R_XMM1, UD_R_XMM2, UD_R_XMM3, + UD_R_XMM4, UD_R_XMM5, UD_R_XMM6, UD_R_XMM7, + UD_R_XMM8, UD_R_XMM9, UD_R_XMM10, UD_R_XMM11, + UD_R_XMM12, UD_R_XMM13, UD_R_XMM14, UD_R_XMM15, + + /* 256B multimedia registers */ + UD_R_YMM0, UD_R_YMM1, UD_R_YMM2, UD_R_YMM3, + UD_R_YMM4, UD_R_YMM5, UD_R_YMM6, UD_R_YMM7, + UD_R_YMM8, UD_R_YMM9, UD_R_YMM10, UD_R_YMM11, + UD_R_YMM12, UD_R_YMM13, UD_R_YMM14, UD_R_YMM15, + + UD_R_RIP, + + /* Operand Types */ + UD_OP_REG, UD_OP_MEM, UD_OP_PTR, UD_OP_IMM, + UD_OP_JIMM, UD_OP_CONST +}; + +#include "itab.h" + +union ud_lval { + int8_t sbyte; + uint8_t ubyte; + int16_t sword; + uint16_t uword; + int32_t sdword; + uint32_t udword; + int64_t sqword; + uint64_t uqword; + struct { + uint16_t seg; + uint32_t off; + } ptr; +}; + +/* ----------------------------------------------------------------------------- + * struct ud_operand - Disassembled instruction Operand. + * ----------------------------------------------------------------------------- + */ +struct ud_operand { + enum ud_type type; + uint16_t size; + enum ud_type base; + enum ud_type index; + uint8_t scale; + uint8_t offset; + union ud_lval lval; + /* + * internal use only + */ + uint64_t _legacy; /* this will be removed in 1.8 */ + uint8_t _oprcode; +}; + +/* ----------------------------------------------------------------------------- + * struct ud - The udis86 object. + * ----------------------------------------------------------------------------- + */ +struct ud +{ + /* + * input buffering + */ + int (*inp_hook) (struct ud*); +#ifndef __UD_STANDALONE__ + FILE* inp_file; +#endif + const uint8_t* inp_buf; + size_t inp_buf_size; + size_t inp_buf_index; + uint8_t inp_curr; + size_t inp_ctr; + uint8_t inp_sess[64]; + int inp_end; + int inp_peek; + + void (*translator)(struct ud*); + uint64_t insn_offset; + char insn_hexcode[64]; + + /* + * Assembly output buffer + */ + char *asm_buf; + size_t asm_buf_size; + size_t asm_buf_fill; + char asm_buf_int[128]; + + /* + * Symbol resolver for use in the translation phase. + */ + const char* (*sym_resolver)(struct ud*, uint64_t addr, int64_t *offset); + + uint8_t dis_mode; + uint64_t pc; + uint8_t vendor; + enum ud_mnemonic_code mnemonic; + struct ud_operand operand[4]; + uint8_t error; + uint8_t _rex; + uint8_t pfx_rex; + uint8_t pfx_seg; + uint8_t pfx_opr; + uint8_t pfx_adr; + uint8_t pfx_lock; + uint8_t pfx_str; + uint8_t pfx_rep; + uint8_t pfx_repe; + uint8_t pfx_repne; + uint8_t opr_mode; + uint8_t adr_mode; + uint8_t br_far; + uint8_t br_near; + uint8_t have_modrm; + uint8_t modrm; + uint8_t vex_op; + uint8_t vex_b1; + uint8_t vex_b2; + uint8_t primary_opcode; + void * user_opaque_data; + struct ud_itab_entry * itab_entry; + struct ud_lookup_table_list_entry *le; +}; + +/* ----------------------------------------------------------------------------- + * Type-definitions + * ----------------------------------------------------------------------------- + */ +typedef enum ud_type ud_type_t; +typedef enum ud_mnemonic_code ud_mnemonic_code_t; + +typedef struct ud ud_t; +typedef struct ud_operand ud_operand_t; + +#define UD_SYN_INTEL ud_translate_intel +#define UD_SYN_ATT ud_translate_att +#define UD_EOI (-1) +#define UD_INP_CACHE_SZ 32 +#define UD_VENDOR_AMD 0 +#define UD_VENDOR_INTEL 1 +#define UD_VENDOR_ANY 2 + +#endif + +/* +vim: set ts=2 sw=2 expandtab +*/ diff --git a/ext/udis86/udint.h b/ext/udis86/udint.h new file mode 100644 index 0000000000..38103bbc95 --- /dev/null +++ b/ext/udis86/udint.h @@ -0,0 +1,93 @@ +/* udis86 - libudis86/udint.h -- definitions for internal use only + * + * Copyright (c) 2002-2009 Vivek Thampi + * All rights reserved. + * + * Redistribution and use in source and binary forms, with or without modification, + * are permitted provided that the following conditions are met: + * + * * Redistributions of source code must retain the above copyright notice, + * this list of conditions and the following disclaimer. + * * Redistributions in binary form must reproduce the above copyright notice, + * this list of conditions and the following disclaimer in the documentation + * and/or other materials provided with the distribution. + * + * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND + * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED + * WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE + * DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE FOR + * ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES + * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; + * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON + * ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT + * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + */ +#ifndef _UDINT_H_ +#define _UDINT_H_ + +#ifdef HAVE_CONFIG_H +# include +#endif /* HAVE_CONFIG_H */ + +#include +#define UD_ASSERT(_x) assert(_x) + +#if defined(UD_DEBUG) + #define UDERR(u, msg) \ + do { \ + (u)->error = 1; \ + fprintf(stderr, "decode-error: %s:%d: %s", \ + __FILE__, __LINE__, (msg)); \ + } while (0) +#else + #define UDERR(u, m) \ + do { \ + (u)->error = 1; \ + } while (0) +#endif /* !LOGERR */ + +#define UD_RETURN_ON_ERROR(u) \ + do { \ + if ((u)->error != 0) { \ + return (u)->error; \ + } \ + } while (0) + +#define UD_RETURN_WITH_ERROR(u, m) \ + do { \ + UDERR(u, m); \ + return (u)->error; \ + } while (0) + +#ifndef __UD_STANDALONE__ +# define UD_NON_STANDALONE(x) x +#else +# define UD_NON_STANDALONE(x) +#endif + +/* printf formatting int64 specifier */ +#ifdef FMT64 +# undef FMT64 +#endif +#if defined(_MSC_VER) || defined(__BORLANDC__) +# define FMT64 "I64" +#else +# if defined(__APPLE__) +# define FMT64 "ll" +# elif defined(__amd64__) || defined(__x86_64__) +# define FMT64 "l" +# else +# define FMT64 "ll" +# endif /* !x64 */ +#endif + +/* define an inline macro */ +#if defined(_MSC_VER) || defined(__BORLANDC__) +# define UD_INLINE __inline /* MS Visual Studio requires __inline + instead of inline for C code */ +#else +# define UD_INLINE inline +#endif + +#endif /* _UDINT_H_ */ diff --git a/ext/udis86/udis86.c b/ext/udis86/udis86.c new file mode 100644 index 0000000000..ede9106d2d --- /dev/null +++ b/ext/udis86/udis86.c @@ -0,0 +1,456 @@ +/* udis86 - libudis86/udis86.c + * + * Copyright (c) 2002-2013 Vivek Thampi + * All rights reserved. + * + * Redistribution and use in source and binary forms, with or without modification, + * are permitted provided that the following conditions are met: + * + * * Redistributions of source code must retain the above copyright notice, + * this list of conditions and the following disclaimer. + * * Redistributions in binary form must reproduce the above copyright notice, + * this list of conditions and the following disclaimer in the documentation + * and/or other materials provided with the distribution. + * + * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND + * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED + * WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE + * DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE FOR + * ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES + * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; + * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON + * ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT + * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + */ + +#include "udint.h" +#include "extern.h" +#include "decode.h" + +#if !defined(__UD_STANDALONE__) +#include +#endif /* !__UD_STANDALONE__ */ + +static void ud_inp_init(struct ud *u); + +/* ============================================================================= + * ud_init + * Initializes ud_t object. + * ============================================================================= + */ +extern void +ud_init(struct ud* u) +{ + memset((void*)u, 0, sizeof(struct ud)); + ud_set_mode(u, 16); + u->mnemonic = UD_Iinvalid; + ud_set_pc(u, 0); +#ifndef __UD_STANDALONE__ + ud_set_input_file(u, stdin); +#endif /* __UD_STANDALONE__ */ + + ud_set_asm_buffer(u, u->asm_buf_int, sizeof(u->asm_buf_int)); +} + + +/* ============================================================================= + * ud_disassemble + * Disassembles one instruction and returns the number of + * bytes disassembled. A zero means end of disassembly. + * ============================================================================= + */ +extern unsigned int +ud_disassemble(struct ud* u) +{ + int len; + if (u->inp_end) { + return 0; + } + if ((len = ud_decode(u)) > 0) { + if (u->translator != NULL) { + u->asm_buf[0] = '\0'; + u->translator(u); + } + } + return len; +} + + +/* ============================================================================= + * ud_set_mode() - Set Disassemly Mode. + * ============================================================================= + */ +extern void +ud_set_mode(struct ud* u, uint8_t m) +{ + switch(m) { + case 16: + case 32: + case 64: u->dis_mode = m ; return; + default: u->dis_mode = 16; return; + } +} + +/* ============================================================================= + * ud_set_vendor() - Set vendor. + * ============================================================================= + */ +extern void +ud_set_vendor(struct ud* u, unsigned v) +{ + switch(v) { + case UD_VENDOR_INTEL: + u->vendor = v; + break; + case UD_VENDOR_ANY: + u->vendor = v; + break; + default: + u->vendor = UD_VENDOR_AMD; + } +} + +/* ============================================================================= + * ud_set_pc() - Sets code origin. + * ============================================================================= + */ +extern void +ud_set_pc(struct ud* u, uint64_t o) +{ + u->pc = o; +} + +/* ============================================================================= + * ud_set_syntax() - Sets the output syntax. + * ============================================================================= + */ +extern void +ud_set_syntax(struct ud* u, void (*t)(struct ud*)) +{ + u->translator = t; +} + +/* ============================================================================= + * ud_insn() - returns the disassembled instruction + * ============================================================================= + */ +const char* +ud_insn_asm(const struct ud* u) +{ + return u->asm_buf; +} + +/* ============================================================================= + * ud_insn_offset() - Returns the offset. + * ============================================================================= + */ +uint64_t +ud_insn_off(const struct ud* u) +{ + return u->insn_offset; +} + + +/* ============================================================================= + * ud_insn_hex() - Returns hex form of disassembled instruction. + * ============================================================================= + */ +const char* +ud_insn_hex(struct ud* u) +{ + u->insn_hexcode[0] = 0; + if (!u->error) { + unsigned int i; + const unsigned char *src_ptr = ud_insn_ptr(u); + char* src_hex; + src_hex = (char*) u->insn_hexcode; + /* for each byte used to decode instruction */ + for (i = 0; i < ud_insn_len(u) && i < sizeof(u->insn_hexcode) / 2; + ++i, ++src_ptr) { + sprintf(src_hex, "%02x", *src_ptr & 0xFF); + src_hex += 2; + } + } + return u->insn_hexcode; +} + + +/* ============================================================================= + * ud_insn_ptr + * Returns a pointer to buffer containing the bytes that were + * disassembled. + * ============================================================================= + */ +extern const uint8_t* +ud_insn_ptr(const struct ud* u) +{ + return (u->inp_buf == NULL) ? + u->inp_sess : u->inp_buf + (u->inp_buf_index - u->inp_ctr); +} + + +/* ============================================================================= + * ud_insn_len + * Returns the count of bytes disassembled. + * ============================================================================= + */ +extern unsigned int +ud_insn_len(const struct ud* u) +{ + return u->inp_ctr; +} + + +/* ============================================================================= + * ud_insn_get_opr + * Return the operand struct representing the nth operand of + * the currently disassembled instruction. Returns NULL if + * there's no such operand. + * ============================================================================= + */ +const struct ud_operand* +ud_insn_opr(const struct ud *u, unsigned int n) +{ + if (n > 3 || u->operand[n].type == UD_NONE) { + return NULL; + } else { + return &u->operand[n]; + } +} + + +/* ============================================================================= + * ud_opr_is_sreg + * Returns non-zero if the given operand is of a segment register type. + * ============================================================================= + */ +int +ud_opr_is_sreg(const struct ud_operand *opr) +{ + return opr->type == UD_OP_REG && + opr->base >= UD_R_ES && + opr->base <= UD_R_GS; +} + + +/* ============================================================================= + * ud_opr_is_sreg + * Returns non-zero if the given operand is of a general purpose + * register type. + * ============================================================================= + */ +int +ud_opr_is_gpr(const struct ud_operand *opr) +{ + return opr->type == UD_OP_REG && + opr->base >= UD_R_AL && + opr->base <= UD_R_R15; +} + + +/* ============================================================================= + * ud_set_user_opaque_data + * ud_get_user_opaque_data + * Get/set user opaqute data pointer + * ============================================================================= + */ +void +ud_set_user_opaque_data(struct ud * u, void* opaque) +{ + u->user_opaque_data = opaque; +} + +void* +ud_get_user_opaque_data(const struct ud *u) +{ + return u->user_opaque_data; +} + + +/* ============================================================================= + * ud_set_asm_buffer + * Allow the user to set an assembler output buffer. If `buf` is NULL, + * we switch back to the internal buffer. + * ============================================================================= + */ +void +ud_set_asm_buffer(struct ud *u, char *buf, size_t size) +{ + if (buf == NULL) { + ud_set_asm_buffer(u, u->asm_buf_int, sizeof(u->asm_buf_int)); + } else { + u->asm_buf = buf; + u->asm_buf_size = size; + } +} + + +/* ============================================================================= + * ud_set_sym_resolver + * Set symbol resolver for relative targets used in the translation + * phase. + * + * The resolver is a function that takes a uint64_t address and returns a + * symbolic name for the that address. The function also takes a second + * argument pointing to an integer that the client can optionally set to a + * non-zero value for offsetted targets. (symbol+offset) The function may + * also return NULL, in which case the translator only prints the target + * address. + * + * The function pointer maybe NULL which resets symbol resolution. + * ============================================================================= + */ +void +ud_set_sym_resolver(struct ud *u, const char* (*resolver)(struct ud*, + uint64_t addr, + int64_t *offset)) +{ + u->sym_resolver = resolver; +} + + +/* ============================================================================= + * ud_insn_mnemonic + * Return the current instruction mnemonic. + * ============================================================================= + */ +enum ud_mnemonic_code +ud_insn_mnemonic(const struct ud *u) +{ + return u->mnemonic; +} + + +/* ============================================================================= + * ud_lookup_mnemonic + * Looks up mnemonic code in the mnemonic string table. + * Returns NULL if the mnemonic code is invalid. + * ============================================================================= + */ +const char* +ud_lookup_mnemonic(enum ud_mnemonic_code c) +{ + if (c < UD_MAX_MNEMONIC_CODE) { + return ud_mnemonics_str[c]; + } else { + return NULL; + } +} + + +/* + * ud_inp_init + * Initializes the input system. + */ +static void +ud_inp_init(struct ud *u) +{ + u->inp_hook = NULL; + u->inp_buf = NULL; + u->inp_buf_size = 0; + u->inp_buf_index = 0; + u->inp_curr = 0; + u->inp_ctr = 0; + u->inp_end = 0; + u->inp_peek = UD_EOI; + UD_NON_STANDALONE(u->inp_file = NULL); +} + + +/* ============================================================================= + * ud_inp_set_hook + * Sets input hook. + * ============================================================================= + */ +void +ud_set_input_hook(register struct ud* u, int (*hook)(struct ud*)) +{ + ud_inp_init(u); + u->inp_hook = hook; +} + +/* ============================================================================= + * ud_inp_set_buffer + * Set buffer as input. + * ============================================================================= + */ +void +ud_set_input_buffer(register struct ud* u, const uint8_t* buf, size_t len) +{ + ud_inp_init(u); + u->inp_buf = buf; + u->inp_buf_size = len; + u->inp_buf_index = 0; +} + + +#ifndef __UD_STANDALONE__ +/* ============================================================================= + * ud_input_set_file + * Set FILE as input. + * ============================================================================= + */ +static int +inp_file_hook(struct ud* u) +{ + return fgetc(u->inp_file); +} + +void +ud_set_input_file(register struct ud* u, FILE* f) +{ + ud_inp_init(u); + u->inp_hook = inp_file_hook; + u->inp_file = f; +} +#endif /* __UD_STANDALONE__ */ + + +/* ============================================================================= + * ud_input_skip + * Skip n input bytes. + * ============================================================================ + */ +void +ud_input_skip(struct ud* u, size_t n) +{ + if (u->inp_end) { + return; + } + if (u->inp_buf == NULL) { + while (n--) { + int c = u->inp_hook(u); + if (c == UD_EOI) { + goto eoi; + } + } + return; + } else { + if (n > u->inp_buf_size || + u->inp_buf_index > u->inp_buf_size - n) { + u->inp_buf_index = u->inp_buf_size; + goto eoi; + } + u->inp_buf_index += n; + return; + } +eoi: + u->inp_end = 1; + UDERR(u, "cannot skip, eoi received\b"); + return; +} + + +/* ============================================================================= + * ud_input_end + * Returns non-zero on end-of-input. + * ============================================================================= + */ +int +ud_input_end(const struct ud *u) +{ + return u->inp_end; +} + +/* vim:set ts=2 sw=2 expandtab */ diff --git a/ext/udis86/udis86.h b/ext/udis86/udis86.h new file mode 100644 index 0000000000..18e71ccba9 --- /dev/null +++ b/ext/udis86/udis86.h @@ -0,0 +1,33 @@ +/* udis86 - udis86.h + * + * Copyright (c) 2002-2009 Vivek Thampi + * All rights reserved. + * + * Redistribution and use in source and binary forms, with or without modification, + * are permitted provided that the following conditions are met: + * + * * Redistributions of source code must retain the above copyright notice, + * this list of conditions and the following disclaimer. + * * Redistributions in binary form must reproduce the above copyright notice, + * this list of conditions and the following disclaimer in the documentation + * and/or other materials provided with the distribution. + * + * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND + * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED + * WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE + * DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE FOR + * ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES + * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; + * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON + * ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT + * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + */ +#ifndef UDIS86_H +#define UDIS86_H + +#include "types.h" +#include "extern.h" +#include "itab.h" + +#endif From 6f9ea6ef4b803ffabda9b96475eefe7d8db9ab67 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Sat, 11 Oct 2014 08:43:35 -0700 Subject: [PATCH 071/105] Avoid some warnings in udis86. These should all be safe. --- ext/udis86/decode.c | 4 ++-- ext/udis86/syn.c | 2 +- ext/udis86/udis86.c | 2 +- 3 files changed, 4 insertions(+), 4 deletions(-) diff --git a/ext/udis86/decode.c b/ext/udis86/decode.c index 83e8e185bc..44c3bdb452 100644 --- a/ext/udis86/decode.c +++ b/ext/udis86/decode.c @@ -544,7 +544,7 @@ decode_modrm_rm(struct ud *u, unsigned int size) /* operand size */ { - size_t offset = 0; + unsigned int offset = 0; unsigned char mod, rm; /* get mod, r/m and reg fields */ @@ -1257,7 +1257,7 @@ ud_decode(struct ud *u) u->pc += u->inp_ctr; /* move program counter by bytes decoded */ /* return number of bytes disassembled. */ - return u->inp_ctr; + return (unsigned int)u->inp_ctr; } /* diff --git a/ext/udis86/syn.c b/ext/udis86/syn.c index 1b9e1d42a5..1af48e5bf9 100644 --- a/ext/udis86/syn.c +++ b/ext/udis86/syn.c @@ -116,7 +116,7 @@ ud_asmprintf(struct ud *u, const char *fmt, ...) int avail; va_list ap; va_start(ap, fmt); - avail = u->asm_buf_size - u->asm_buf_fill - 1 /* nullchar */; + avail = (int)(u->asm_buf_size - u->asm_buf_fill - 1 /* nullchar */); ret = vsnprintf((char*) u->asm_buf + u->asm_buf_fill, avail, fmt, ap); if (ret < 0 || ret > avail) { u->asm_buf_fill = u->asm_buf_size - 1; diff --git a/ext/udis86/udis86.c b/ext/udis86/udis86.c index ede9106d2d..f957de45ec 100644 --- a/ext/udis86/udis86.c +++ b/ext/udis86/udis86.c @@ -198,7 +198,7 @@ ud_insn_ptr(const struct ud* u) extern unsigned int ud_insn_len(const struct ud* u) { - return u->inp_ctr; + return (unsigned int)u->inp_ctr; } From 498d0b9adcbf39dbd5e20044eb9137354c688276 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Sat, 11 Oct 2014 09:06:52 -0700 Subject: [PATCH 072/105] Enable x86 disassembly in jit compare. --- UI/DevScreens.cpp | 15 ++++++++++++++- 1 file changed, 14 insertions(+), 1 deletion(-) diff --git a/UI/DevScreens.cpp b/UI/DevScreens.cpp index 8313663159..e17de3de3d 100644 --- a/UI/DevScreens.cpp +++ b/UI/DevScreens.cpp @@ -25,6 +25,7 @@ #include "ui/viewgroup.h" #include "ui/ui.h" #include "ext/disarm.h" +#include "ext/udis86/udis86.h" #include "Common/LogManager.h" #include "Common/CPUDetect.h" @@ -491,7 +492,19 @@ void JitCompareScreen::UpdateDisasm() { rightDisasm_->Add(new TextView(targetDis[i])); } #else - rightDisasm_->Add(new TextView("No x86 disassembler available")); + ud_t ud_obj; + ud_init(&ud_obj); +#ifdef _M_X64 + ud_set_mode(&ud_obj, 64); +#endif + ud_set_pc(&ud_obj, (intptr_t)block->normalEntry); + ud_set_vendor(&ud_obj, UD_VENDOR_ANY); + ud_set_syntax(&ud_obj, UD_SYN_INTEL); + + ud_set_input_buffer(&ud_obj, block->normalEntry, block->codeSize); + while (ud_disassemble(&ud_obj) != 0) { + rightDisasm_->Add(new TextView(ud_insn_asm(&ud_obj))); + } #endif } From 4027731a71b62bc5b7acd288194ce90b8eeaeaa2 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Sat, 11 Oct 2014 09:28:52 -0700 Subject: [PATCH 073/105] Avoid spinning the CPU in dev menu screens. --- UI/DevScreens.cpp | 3 +++ 1 file changed, 3 insertions(+) diff --git a/UI/DevScreens.cpp b/UI/DevScreens.cpp index e17de3de3d..8967831a65 100644 --- a/UI/DevScreens.cpp +++ b/UI/DevScreens.cpp @@ -62,16 +62,19 @@ void DevMenu::CreatePopupContents(UI::ViewGroup *parent) { } UI::EventReturn DevMenu::OnLogConfig(UI::EventParams &e) { + UpdateUIState(UISTATE_PAUSEMENU); screenManager()->push(new LogConfigScreen()); return UI::EVENT_DONE; } UI::EventReturn DevMenu::OnDeveloperTools(UI::EventParams &e) { + UpdateUIState(UISTATE_PAUSEMENU); screenManager()->push(new DeveloperToolsScreen()); return UI::EVENT_DONE; } UI::EventReturn DevMenu::OnJitCompare(UI::EventParams &e) { + UpdateUIState(UISTATE_PAUSEMENU); screenManager()->push(new JitCompareScreen()); return UI::EVENT_DONE; } From 25a26eeaf9a4c2bcf7e5670c366884072adf0be6 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Sat, 11 Oct 2014 12:29:31 -0700 Subject: [PATCH 074/105] Fix paths. Hmm, this didn't happen when I added other things in ext... I blame 2013. --- Core/Core.vcxproj | 26 +++++++++++++------------- Core/Core.vcxproj.filters | 26 +++++++++++++------------- 2 files changed, 26 insertions(+), 26 deletions(-) diff --git a/Core/Core.vcxproj b/Core/Core.vcxproj index 3036bdb00a..b2c9d6ae0e 100644 --- a/Core/Core.vcxproj +++ b/Core/Core.vcxproj @@ -169,12 +169,12 @@ - - - - - - + + + + + + @@ -447,13 +447,13 @@ - - - - - - - + + + + + + + diff --git a/Core/Core.vcxproj.filters b/Core/Core.vcxproj.filters index 5bcc838c6a..ad2b007744 100644 --- a/Core/Core.vcxproj.filters +++ b/Core/Core.vcxproj.filters @@ -535,22 +535,22 @@ HLE\Libraries - + Ext\udis86 - + Ext\udis86 - + Ext\udis86 - + Ext\udis86 - + Ext\udis86 - + Ext\udis86 @@ -1005,25 +1005,25 @@ HLE\Libraries - + Ext\udis86 - + Ext\udis86 - + Ext\udis86 - + Ext\udis86 - + Ext\udis86 - + Ext\udis86 - + Ext\udis86 From f99c2cd01075d5d393d419bdaceec56220b0fff5 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Henrik=20Rydg=C3=A5rd?= Date: Sun, 12 Oct 2014 15:51:16 +0200 Subject: [PATCH 075/105] x86 Jit: Generate nicer code for some cases of addiu --- Common/x64Emitter.h | 7 +++++++ Core/MIPS/x86/Asm.cpp | 1 - Core/MIPS/x86/CompALU.cpp | 22 ++++++++++++++-------- 3 files changed, 21 insertions(+), 9 deletions(-) diff --git a/Common/x64Emitter.h b/Common/x64Emitter.h index ec055c6f39..cf739c7a30 100644 --- a/Common/x64Emitter.h +++ b/Common/x64Emitter.h @@ -216,6 +216,13 @@ inline OpArg Imm8 (u8 imm) {return OpArg(imm, SCALE_IMM8);} inline OpArg Imm16(u16 imm) {return OpArg(imm, SCALE_IMM16);} //rarely used inline OpArg Imm32(u32 imm) {return OpArg(imm, SCALE_IMM32);} inline OpArg Imm64(u64 imm) {return OpArg(imm, SCALE_IMM64);} +inline OpArg UImmAuto(u32 imm) { + return OpArg(imm, imm >= 128 ? SCALE_IMM32 : SCALE_IMM8); +} +inline OpArg SImmAuto(s32 imm) { + return OpArg(imm, (imm >= 128 || imm < -128) ? SCALE_IMM32 : SCALE_IMM8); +} + #ifdef _M_X64 inline OpArg ImmPtr(const void *imm) {return Imm64((u64)imm);} #else diff --git a/Core/MIPS/x86/Asm.cpp b/Core/MIPS/x86/Asm.cpp index b94c48b286..bdcfa0f712 100644 --- a/Core/MIPS/x86/Asm.cpp +++ b/Core/MIPS/x86/Asm.cpp @@ -115,7 +115,6 @@ void AsmRoutineManager::Generate(MIPSState *mips, MIPSComp::Jit *jit) AND(32, R(EDX), Imm32(MIPS_JITBLOCK_MASK)); CMP(32, R(EDX), Imm32(MIPS_EMUHACK_OPCODE)); FixupBranch notfound = J_CC(CC_NZ); - // IDEA - we have 24 bits, why not just use offsets from base of code? if (enableDebug) { ADD(32, M(&mips->debugCount), Imm8(1)); diff --git a/Core/MIPS/x86/CompALU.cpp b/Core/MIPS/x86/CompALU.cpp index ae27029f24..bf28b4689c 100644 --- a/Core/MIPS/x86/CompALU.cpp +++ b/Core/MIPS/x86/CompALU.cpp @@ -75,28 +75,34 @@ namespace MIPSComp case 8: // same as addiu? case 9: // R(rt) = R(rs) + simm; break; //addiu { - if (gpr.IsImm(rs)) - { + if (gpr.IsImm(rs)) { gpr.SetImm(rt, gpr.GetImm(rs) + simm); break; } gpr.Lock(rt, rs); gpr.MapReg(rt, rt == rs, true); - if (rt == rs || gpr.R(rs).IsSimpleReg()) + if (rt == rs) { + if (simm > 0) { + ADD(32, gpr.R(rt), UImmAuto(simm)); + } else if (simm < 0) { + SUB(32, gpr.R(rt), UImmAuto(-simm)); + } + } else if (gpr.R(rs).IsSimpleReg()) { LEA(32, gpr.RX(rt), MDisp(gpr.RX(rs), simm)); - else - { + } else { MOV(32, gpr.R(rt), gpr.R(rs)); - if (suimm != 0) - ADD(32, gpr.R(rt), Imm32(suimm)); + if (simm > 0) + ADD(32, gpr.R(rt), UImmAuto(simm)); + else if (simm < 0) { + SUB(32, gpr.R(rt), UImmAuto(-simm)); + } } gpr.UnlockAll(); } break; case 10: // R(rt) = (s32)R(rs) < simm; break; //slti - // There's a mips compiler out there asking questions it already knows the answer to... if (gpr.IsImm(rs)) { gpr.SetImm(rt, (s32)gpr.GetImm(rs) < simm); From e389fcefcf7708fa5b318055bd68ff89251c5e1e Mon Sep 17 00:00:00 2001 From: Henrik Rydgard Date: Sun, 12 Oct 2014 18:54:21 +0200 Subject: [PATCH 076/105] Fix x86 disassembly in 32-bit mode --- UI/DevScreens.cpp | 2 ++ 1 file changed, 2 insertions(+) diff --git a/UI/DevScreens.cpp b/UI/DevScreens.cpp index 8967831a65..18263cd2e0 100644 --- a/UI/DevScreens.cpp +++ b/UI/DevScreens.cpp @@ -499,6 +499,8 @@ void JitCompareScreen::UpdateDisasm() { ud_init(&ud_obj); #ifdef _M_X64 ud_set_mode(&ud_obj, 64); +#else + ud_set_mode(&ud_obj, 32); #endif ud_set_pc(&ud_obj, (intptr_t)block->normalEntry); ud_set_vendor(&ud_obj, UD_VENDOR_ANY); From 0b3fe61a35adca2a5aa0755ff5c48d73fd1d35b5 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Sun, 12 Oct 2014 09:44:16 -0700 Subject: [PATCH 077/105] Start the jit compare off on the current block. --- UI/DevScreens.cpp | 3 +++ 1 file changed, 3 insertions(+) diff --git a/UI/DevScreens.cpp b/UI/DevScreens.cpp index 18263cd2e0..959dd78511 100644 --- a/UI/DevScreens.cpp +++ b/UI/DevScreens.cpp @@ -427,6 +427,9 @@ void JitCompareScreen::CreateViews() { leftColumn->Add(new Choice("Random VFPU"))->OnClick.Handle(this, &JitCompareScreen::OnRandomVFPUBlock); leftColumn->Add(new Choice(d->T("Back")))->OnClick.Handle(this, &UIScreen::OnBack); blockName_ = leftColumn->Add(new TextView("no block")); + + EventParams ignore = {0}; + OnCurrentBlock(ignore); } #ifdef ARM From 1284b109d0e66f6572638aa55abaa4b6210d4da4 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Sun, 12 Oct 2014 09:48:47 -0700 Subject: [PATCH 078/105] Small warning fix. --- UI/ControlMappingScreen.cpp | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/UI/ControlMappingScreen.cpp b/UI/ControlMappingScreen.cpp index 4113e0dad1..1a4e137d87 100644 --- a/UI/ControlMappingScreen.cpp +++ b/UI/ControlMappingScreen.cpp @@ -416,7 +416,7 @@ void JoystickHistoryView::Draw(UIContext &dc) { void JoystickHistoryView::Update(const InputState &input_state) { locations_.push_back(Location(curX_, curY_)); - if (locations_.size() > maxCount_) { + if ((int)locations_.size() > maxCount_) { locations_.pop_front(); } } From 62fdb6be92e53cdda59ef962359722a04b8982dd Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Sun, 12 Oct 2014 09:51:47 -0700 Subject: [PATCH 079/105] Switch Travis to SDL2. --- .travis.yml | 9 ++++++--- 1 file changed, 6 insertions(+), 3 deletions(-) diff --git a/.travis.yml b/.travis.yml index c7f40fb672..28d2a7ab35 100644 --- a/.travis.yml +++ b/.travis.yml @@ -35,10 +35,13 @@ before_install: install: # Ubuntu Linux + GCC 4.8 - if [ "$PPSSPP_BUILD_TYPE" == "Linux" ]; then - sudo apt-get install libsdl1.2-dev -qq && + sudo add-apt-repository ppa:zoogie/sdl2-snapshots -y && + if [[ "$CXX" == g++ ]]; then + sudo add-apt-repository ppa:ubuntu-toolchain-r/test -y; + fi; + sudo apt-get update && + sudo apt-get install libsdl2-dev -qq && if [[ "$CXX" == g++ ]]; then - sudo add-apt-repository ppa:ubuntu-toolchain-r/test -y && - sudo apt-get update && sudo apt-get install g++-4.8 -qq && export CXX="g++-4.8" CC="gcc-4.8"; fi; From eab010a0c010ccda6e2d2cb57ca148fb9be1ebb6 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Henrik=20Rydg=C3=A5rd?= Date: Sun, 12 Oct 2014 14:09:35 +0200 Subject: [PATCH 080/105] x86 JIT: Sacrifice a register for a pointer to the MIPS context. Shrinks emitted x86 code considerably. Nice in 64-bit, but might be a bit too much in 32-bit though... Needs testing. --- Core/MIPS/x86/Asm.cpp | 8 ++++++++ Core/MIPS/x86/RegCache.cpp | 11 ++++++----- Core/MIPS/x86/RegCache.h | 6 ++++++ Core/MIPS/x86/RegCacheFPU.cpp | 3 ++- 4 files changed, 22 insertions(+), 6 deletions(-) diff --git a/Core/MIPS/x86/Asm.cpp b/Core/MIPS/x86/Asm.cpp index bdcfa0f712..d642f9e205 100644 --- a/Core/MIPS/x86/Asm.cpp +++ b/Core/MIPS/x86/Asm.cpp @@ -103,6 +103,14 @@ void AsmRoutineManager::Generate(MIPSState *mips, MIPSComp::Jit *jit) dispatcherNoCheck = GetCodePtr(); + // TODO: Find a less costly place to put this (or multiple..) +#ifdef _M_X64 + // From the start of the FP reg, a single byte offset can reach all GPR + all FPR (but no VFPUR) + MOV(64, R(CTXREG), ImmPtr(&mips->f[0])); +#else + MOV(32, R(CTXREG), ImmPtr(&mips->f[0])); +#endif + MOV(32, R(EAX), M(&mips->pc)); #ifdef _M_IX86 AND(32, R(EAX), Imm32(Memory::MEMVIEW32_MASK)); diff --git a/Core/MIPS/x86/RegCache.cpp b/Core/MIPS/x86/RegCache.cpp index 14674d35d7..f0765b9cee 100644 --- a/Core/MIPS/x86/RegCache.cpp +++ b/Core/MIPS/x86/RegCache.cpp @@ -33,12 +33,12 @@ static const int allocationOrder[] = // On x64, RCX and RDX are the first args. CallProtectedFunction() assumes they're not regcached. #ifdef _M_X64 #ifdef _WIN32 - RSI, RDI, R13, R14, R8, R9, R10, R11, R12, + RSI, RDI, R13, R8, R9, R10, R11, R12, #else - RBP, R13, R14, R8, R9, R10, R11, R12, + RBP, R13, R8, R9, R10, R11, R12, #endif #elif _M_IX86 - ESI, EDI, EBP, EDX, ECX, // Let's try to free up EBX as well. + ESI, EDI, EDX, ECX, // Let's try to free up EBX as well. #endif }; @@ -218,8 +218,9 @@ const int *GPRRegCache::GetAllocationOrder(int &count) { OpArg GPRRegCache::GetDefaultLocation(MIPSGPReg reg) const { - if (reg < 32) - return M(&mips->r[reg]); + if (reg < 32) { + return MDisp(CTXREG, -128 + reg * 4); + } switch (reg) { case MIPS_REG_HI: return M(&mips->hi); diff --git a/Core/MIPS/x86/RegCache.h b/Core/MIPS/x86/RegCache.h index 297d59635d..930761ac11 100644 --- a/Core/MIPS/x86/RegCache.h +++ b/Core/MIPS/x86/RegCache.h @@ -31,6 +31,12 @@ using namespace Gen; #define NUM_MIPS_GPRS 36 +#ifdef _M_X64 +#define CTXREG R14 +#else +#define CTXREG EBP +#endif + struct MIPSCachedReg { OpArg location; bool away; // value not in source register diff --git a/Core/MIPS/x86/RegCacheFPU.cpp b/Core/MIPS/x86/RegCacheFPU.cpp index 865231d590..71cd63da87 100644 --- a/Core/MIPS/x86/RegCacheFPU.cpp +++ b/Core/MIPS/x86/RegCacheFPU.cpp @@ -20,6 +20,7 @@ #include "Common/Log.h" #include "Common/x64Emitter.h" #include "Core/MIPS/MIPSAnalyst.h" +#include "Core/MIPS/x86/RegCache.h" #include "Core/MIPS/x86/RegCacheFPU.h" u32 FPURegCache::tempValues[NUM_TEMPS]; @@ -215,7 +216,7 @@ void FPURegCache::Flush() { OpArg FPURegCache::GetDefaultLocation(int reg) const { if (reg < 32) { - return M(&mips->f[reg]); + return MDisp(CTXREG, reg * 4); } else if (reg < 32 + 128) { return M(&mips->v[voffset[reg - 32]]); } else { From 8177b4c43bf4238bc68dbcc04dc6fb9f314acd2a Mon Sep 17 00:00:00 2001 From: Henrik Rydgard Date: Sun, 12 Oct 2014 18:53:56 +0200 Subject: [PATCH 081/105] Avoid an ifdef using PTRBITS --- Common/x64Emitter.h | 6 ++++++ Core/MIPS/x86/Asm.cpp | 8 ++------ GPU/Common/VertexDecoderX86.cpp | 7 ------- 3 files changed, 8 insertions(+), 13 deletions(-) diff --git a/Common/x64Emitter.h b/Common/x64Emitter.h index cf739c7a30..1c7e26a001 100644 --- a/Common/x64Emitter.h +++ b/Common/x64Emitter.h @@ -26,6 +26,12 @@ #error "Don't build this on arm." #endif +#ifdef _M_X64 +#define PTRBITS 64 +#else +#define PTRBITS 32 +#endif + namespace Gen { diff --git a/Core/MIPS/x86/Asm.cpp b/Core/MIPS/x86/Asm.cpp index d642f9e205..b4fbbe7161 100644 --- a/Core/MIPS/x86/Asm.cpp +++ b/Core/MIPS/x86/Asm.cpp @@ -103,13 +103,9 @@ void AsmRoutineManager::Generate(MIPSState *mips, MIPSComp::Jit *jit) dispatcherNoCheck = GetCodePtr(); - // TODO: Find a less costly place to put this (or multiple..) -#ifdef _M_X64 + // TODO: Find a less costly place to put this (or multiple..)? // From the start of the FP reg, a single byte offset can reach all GPR + all FPR (but no VFPUR) - MOV(64, R(CTXREG), ImmPtr(&mips->f[0])); -#else - MOV(32, R(CTXREG), ImmPtr(&mips->f[0])); -#endif + MOV(PTRBITS, R(CTXREG), ImmPtr(&mips->f[0])); MOV(32, R(EAX), M(&mips->pc)); #ifdef _M_IX86 diff --git a/GPU/Common/VertexDecoderX86.cpp b/GPU/Common/VertexDecoderX86.cpp index 9feb3be6da..7386fa329b 100644 --- a/GPU/Common/VertexDecoderX86.cpp +++ b/GPU/Common/VertexDecoderX86.cpp @@ -150,13 +150,6 @@ static const JitLookup jitLookup[] = { {&VertexDecoder::Step_Color5551Morph, &VertexDecoderJitCache::Jit_Color5551Morph}, }; -// TODO: This should probably be global... -#ifdef _M_X64 -#define PTRBITS 64 -#else -#define PTRBITS 32 -#endif - JittedVertexDecoder VertexDecoderJitCache::Compile(const VertexDecoder &dec) { dec_ = &dec; const u8 *start = this->GetCodePtr(); From 281ab5f9cb9553375d5a020b13d48e3dcf38aa32 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Henrik=20Rydg=C3=A5rd?= Date: Thu, 9 Oct 2014 20:01:47 +0200 Subject: [PATCH 082/105] Sync x64 emitter to Dolphin's. --- Common/x64Emitter.cpp | 296 +++++++++++++++++++++++++----------------- Common/x64Emitter.h | 237 +++++++++++++++++---------------- 2 files changed, 301 insertions(+), 232 deletions(-) diff --git a/Common/x64Emitter.cpp b/Common/x64Emitter.cpp index 4151ef1518..f454296470 100644 --- a/Common/x64Emitter.cpp +++ b/Common/x64Emitter.cpp @@ -23,6 +23,11 @@ #include "MemoryUtil.h" #include "MsgHandler.h" +#define PRIx64 "llx" + +// Minimize the diff against Dolphin +#define DYNA_REC JIT + namespace Gen { @@ -32,7 +37,7 @@ struct NormalOpDef u8 toRm8, toRm32, fromRm8, fromRm32, imm8, imm32, simm8, ext; }; -static const NormalOpDef nops[11] = +static const NormalOpDef nops[11] = { {0x00, 0x01, 0x02, 0x03, 0x80, 0x81, 0x83, 0}, //ADD {0x10, 0x11, 0x12, 0x13, 0x80, 0x81, 0x83, 2}, //ADC @@ -54,30 +59,30 @@ static const NormalOpDef nops[11] = enum NormalSSEOps { - sseCMP = 0xC2, - sseADD = 0x58, //ADD - sseSUB = 0x5C, //SUB - sseAND = 0x54, //AND - sseANDN = 0x55, //ANDN - sseOR = 0x56, - sseXOR = 0x57, - sseMUL = 0x59, //MUL, - sseDIV = 0x5E, //DIV - sseMIN = 0x5D, //MIN - sseMAX = 0x5F, //MAX - sseCOMIS = 0x2F, //COMIS - sseUCOMIS = 0x2E, //UCOMIS - sseSQRT = 0x51, //SQRT - sseRSQRT = 0x52, //RSQRT (NO DOUBLE PRECISION!!!) + sseCMP = 0xC2, + sseADD = 0x58, //ADD + sseSUB = 0x5C, //SUB + sseAND = 0x54, //AND + sseANDN = 0x55, //ANDN + sseOR = 0x56, + sseXOR = 0x57, + sseMUL = 0x59, //MUL + sseDIV = 0x5E, //DIV + sseMIN = 0x5D, //MIN + sseMAX = 0x5F, //MAX + sseCOMIS = 0x2F, //COMIS + sseUCOMIS = 0x2E, //UCOMIS + sseSQRT = 0x51, //SQRT + sseRSQRT = 0x52, //RSQRT (NO DOUBLE PRECISION!!!) sseMOVAPfromRM = 0x28, //MOVAP from RM - sseMOVAPtoRM = 0x29, //MOVAP to RM - sseMOVUPfromRM = 0x10, //MOVUP from RM - sseMOVUPtoRM = 0x11, //MOVUP to RM + sseMOVAPtoRM = 0x29, //MOVAP to RM + sseMOVUPfromRM = 0x10, //MOVUP from RM sseMOVDQfromRM = 0x6F, sseMOVDQtoRM = 0x7F, - sseMASKMOVDQU = 0xF7, - sseLDDQU = 0xF0, - sseSHUF = 0xC6, + sseMOVUPtoRM = 0x11, //MOVUP to RM + sseMASKMOVDQU = 0xF7, + sseLDDQU = 0xF0, + sseSHUF = 0xC6, sseMOVNTDQ = 0xE7, sseMOVNTP = 0x2B, }; @@ -128,9 +133,9 @@ const u8 *XEmitter::AlignCodePage() return code; } -void XEmitter::WriteModRM(int mod, int rm, int reg) +void XEmitter::WriteModRM(int mod, int reg, int rm) { - Write8((u8)((mod << 6) | ((rm & 7) << 3) | (reg & 7))); + Write8((u8)((mod << 6) | ((reg & 7) << 3) | (rm & 7))); } void XEmitter::WriteSIB(int scale, int index, int base) @@ -148,32 +153,66 @@ void OpArg::WriteRex(XEmitter *emit, int opBits, int bits, int customOp) const if (indexReg & 8) op |= 2; if (offsetOrBaseReg & 8) op |= 1; //TODO investigate if this is dangerous if (op != 0x40 || - (scale == SCALE_NONE && bits == 8 && (offsetOrBaseReg & 0x10c) == 4) || - (opBits == 8 && (customOp & 0x10c) == 4)) { + (bits == 8 && (offsetOrBaseReg & 0x10c) == 4) || + (opBits == 8 && (customOp & 0x10c) == 4)) { emit->Write8(op); - _dbg_assert_(JIT, (offsetOrBaseReg & 0x100) == 0 || bits != 8); - _dbg_assert_(JIT, (customOp & 0x100) == 0 || opBits != 8); + _dbg_assert_(DYNA_REC, (offsetOrBaseReg & 0x100) == 0 || bits != 8); + _dbg_assert_(DYNA_REC, (customOp & 0x100) == 0 || opBits != 8); } else { - _dbg_assert_(JIT, (offsetOrBaseReg & 0x10c) == 0 || - (offsetOrBaseReg & 0x10c) == 0x104 || - bits != 8); - _dbg_assert_(JIT, (customOp & 0x10c) == 0 || - (customOp & 0x10c) == 0x104 || - opBits != 8); + _dbg_assert_(DYNA_REC, (offsetOrBaseReg & 0x10c) == 0 || + (offsetOrBaseReg & 0x10c) == 0x104 || + bits != 8); + _dbg_assert_(DYNA_REC, (customOp & 0x10c) == 0 || + (customOp & 0x10c) == 0x104 || + opBits != 8); } #else - _dbg_assert_(JIT, opBits != 64); - _dbg_assert_(JIT, (customOp & 8) == 0 || customOp == -1); - _dbg_assert_(JIT, (indexReg & 8) == 0); - _dbg_assert_(JIT, (offsetOrBaseReg & 8) == 0); - _dbg_assert_(JIT, opBits != 8 || (customOp & 0x10c) != 4 || customOp == -1); - _dbg_assert_(JIT, scale == SCALE_ATREG || bits != 8 || (offsetOrBaseReg & 0x10c) != 4); + _dbg_assert_(DYNA_REC, opBits != 64); + _dbg_assert_(DYNA_REC, (customOp & 8) == 0 || customOp == -1); + _dbg_assert_(DYNA_REC, (indexReg & 8) == 0); + _dbg_assert_(DYNA_REC, (offsetOrBaseReg & 8) == 0); + _dbg_assert_(DYNA_REC, opBits != 8 || (customOp & 0x10c) != 4 || customOp == -1); + _dbg_assert_(DYNA_REC, bits != 8 || (offsetOrBaseReg & 0x10c) != 4); #endif } +void OpArg::WriteVex(XEmitter* emit, int size, int packed, Gen::X64Reg regOp1, Gen::X64Reg regOp2) const +{ + int R = !(regOp1 & 8); + int X = !(indexReg & 8); + int B = !(offsetOrBaseReg & 8); + + // not so sure about this one... + int W = 0; + + // aka map_select in AMD manuals + // only support VEX opcode map 1 for now (analog to secondary opcode map) + int mmmmm = 1; + + int vvvv = (regOp2 == X64Reg::INVALID_REG) ? 0xf : (regOp2 ^ 0xf); + int L = size == 256; + int pp = (packed << 1) | (size == 64); + + // do we need any VEX fields that only appear in the three-byte form? + if (X == 1 && B == 1 && W == 0 && mmmmm == 1) + { + u8 RvvvvLpp = (R << 7) | (vvvv << 3) | (L << 1) | pp; + emit->Write8(0xC5); + emit->Write8(RvvvvLpp); + } + else + { + u8 RXBmmmmm = (R << 7) | (X << 6) | (B << 5) | mmmmm; + u8 WvvvvLpp = (W << 7) | (vvvv << 3) | (L << 1) | pp; + emit->Write8(0xC4); + emit->Write8(RXBmmmmm); + emit->Write8(WvvvvLpp); + } +} + void OpArg::WriteRest(XEmitter *emit, int extraBytes, X64Reg _operandReg, - bool warn_64bit_offset) const + bool warn_64bit_offset) const { if (_operandReg == 0xff) _operandReg = (X64Reg)this->operandReg; @@ -191,10 +230,10 @@ void OpArg::WriteRest(XEmitter *emit, int extraBytes, X64Reg _operandReg, #ifdef _M_X64 u64 ripAddr = (u64)emit->GetCodePtr() + 4 + extraBytes; s64 distance = (s64)offset - (s64)ripAddr; - _assert_msg_(JIT, (distance < 0x80000000LL + _assert_msg_(DYNA_REC, (distance < 0x80000000LL && distance >= -0x80000000LL) || !warn_64bit_offset, - "WriteRest: op out of range (0x%llx uses 0x%llx)", + "WriteRest: op out of range (0x%" PRIx64 " uses 0x%" PRIx64 ")", ripAddr, offset); s32 offs = (s32)distance; emit->Write32((u32)offs); @@ -248,7 +287,7 @@ void OpArg::WriteRest(XEmitter *emit, int extraBytes, X64Reg _operandReg, SIB = true; } - if (scale == SCALE_ATREG && ((_offsetOrBaseReg & 7) == 4)) + if (scale == SCALE_ATREG && ((_offsetOrBaseReg & 7) == 4)) { SIB = true; ireg = _offsetOrBaseReg; @@ -273,7 +312,7 @@ void OpArg::WriteRest(XEmitter *emit, int extraBytes, X64Reg _operandReg, int oreg = _offsetOrBaseReg; if (SIB) oreg = 4; - + // TODO(ector): WTF is this if about? I don't remember writing it :-) //if (RIP) // oreg = 5; @@ -286,7 +325,7 @@ void OpArg::WriteRest(XEmitter *emit, int extraBytes, X64Reg _operandReg, int ss; switch (scale) { - case SCALE_NONE: _offsetOrBaseReg = 4; ss = 0; break; //RSP + case SCALE_NONE: _offsetOrBaseReg = 4; ss = 0; break; //RSP case SCALE_1: ss = 0; break; case SCALE_2: ss = 1; break; case SCALE_4: ss = 2; break; @@ -295,7 +334,7 @@ void OpArg::WriteRest(XEmitter *emit, int extraBytes, X64Reg _operandReg, case SCALE_NOBASE_4: ss = 2; break; case SCALE_NOBASE_8: ss = 3; break; case SCALE_ATREG: ss = 0; break; - default: _assert_msg_(JIT, 0, "Invalid scale for SIB byte"); ss = 0; break; + default: _assert_msg_(DYNA_REC, 0, "Invalid scale for SIB byte"); ss = 0; break; } emit->Write8((u8)((ss << 6) | ((ireg&7)<<3) | (_offsetOrBaseReg&7))); } @@ -317,7 +356,7 @@ void OpArg::WriteRest(XEmitter *emit, int extraBytes, X64Reg _operandReg, // B = base register# upper bit void XEmitter::Rex(int w, int r, int x, int b) { - w = w ? 1 : 0; + w = w ? 1 : 0; r = r ? 1 : 0; x = x ? 1 : 0; b = b ? 1 : 0; @@ -332,7 +371,7 @@ void XEmitter::JMP(const u8 *addr, bool force5Bytes) if (!force5Bytes) { s64 distance = (s64)(fn - ((u64)code + 2)); - _assert_msg_(JIT, distance >= -0x80 && distance < 0x80, + _assert_msg_(DYNA_REC, distance >= -0x80 && distance < 0x80, "Jump target too far away, needs force5Bytes = true"); //8 bits will do Write8(0xEB); @@ -342,7 +381,7 @@ void XEmitter::JMP(const u8 *addr, bool force5Bytes) { s64 distance = (s64)(fn - ((u64)code + 5)); - _assert_msg_(JIT, distance >= -0x80000000LL + _assert_msg_(DYNA_REC, distance >= -0x80000000LL && distance < 0x80000000LL, "Jump target too far away, needs indirect register"); Write8(0xE9); @@ -353,7 +392,7 @@ void XEmitter::JMP(const u8 *addr, bool force5Bytes) void XEmitter::JMPptr(const OpArg &arg2) { OpArg arg = arg2; - if (arg.IsImm()) _assert_msg_(JIT, 0, "JMPptr - Imm argument"); + if (arg.IsImm()) _assert_msg_(DYNA_REC, 0, "JMPptr - Imm argument"); arg.operandReg = 4; arg.WriteRex(this, 0, 0); Write8(0xFF); @@ -370,7 +409,7 @@ void XEmitter::JMPself() void XEmitter::CALLptr(OpArg arg) { - if (arg.IsImm()) _assert_msg_(JIT, 0, "CALLptr - Imm argument"); + if (arg.IsImm()) _assert_msg_(DYNA_REC, 0, "CALLptr - Imm argument"); arg.operandReg = 2; arg.WriteRex(this, 0, 0); Write8(0xFF); @@ -380,7 +419,7 @@ void XEmitter::CALLptr(OpArg arg) void XEmitter::CALL(const void *fnptr) { u64 distance = u64(fnptr) - (u64(code) + 5); - _assert_msg_(JIT, distance < 0x0000000080000000ULL + _assert_msg_(DYNA_REC, distance < 0x0000000080000000ULL || distance >= 0xFFFFFFFF80000000ULL, "CALL out of range (%p calls %p)", code, fnptr); Write8(0xE8); @@ -432,7 +471,7 @@ void XEmitter::J_CC(CCFlags conditionCode, const u8 * addr, bool force5Bytes) if (!force5Bytes) { s64 distance = (s64)(fn - ((u64)code + 2)); - _assert_msg_(JIT, distance >= -0x80 && distance < 0x80, "Jump target too far away, needs force5Bytes = true"); + _assert_msg_(DYNA_REC, distance >= -0x80 && distance < 0x80, "Jump target too far away, needs force5Bytes = true"); //8 bits will do Write8(0x70 + conditionCode); Write8((u8)(s8)distance); @@ -440,7 +479,7 @@ void XEmitter::J_CC(CCFlags conditionCode, const u8 * addr, bool force5Bytes) else { s64 distance = (s64)(fn - ((u64)code + 6)); - _assert_msg_(JIT, distance >= -0x80000000LL + _assert_msg_(DYNA_REC, distance >= -0x80000000LL && distance < 0x80000000LL, "Jump target too far away, needs indirect register"); Write8(0x0F); @@ -454,13 +493,13 @@ void XEmitter::SetJumpTarget(const FixupBranch &branch) if (branch.type == 0) { s64 distance = (s64)(code - branch.ptr); - _assert_msg_(JIT, distance >= -0x80 && distance < 0x80, "Jump target too far away, needs force5Bytes = true"); + _assert_msg_(DYNA_REC, distance >= -0x80 && distance < 0x80, "Jump target too far away, needs force5Bytes = true"); branch.ptr[-1] = (u8)(s8)distance; } else if (branch.type == 1) { s64 distance = (s64)(code - branch.ptr); - _assert_msg_(JIT, distance >= -0x80000000LL && distance < 0x80000000LL, "Jump target too far away, needs indirect register"); + _assert_msg_(DYNA_REC, distance >= -0x80000000LL && distance < 0x80000000LL, "Jump target too far away, needs indirect register"); ((s32*)branch.ptr)[-1] = (s32)distance; } } @@ -491,9 +530,7 @@ void XEmitter::DEC(int bits, OpArg arg) //Single byte opcodes //There is no PUSHAD/POPAD in 64-bit mode. -void XEmitter::INT3() { - Write8(0xCC); -} +void XEmitter::INT3() {Write8(0xCC);} void XEmitter::RET() {Write8(0xC3);} void XEmitter::RET_FAST() {Write8(0xF3); Write8(0xC3);} //two-byte return (rep ret) - recommended by AMD optimization manual for the case of jumping to a ret @@ -515,7 +552,7 @@ void XEmitter::NOP(int count) } break; } -} +} void XEmitter::PAUSE() {Write8(0xF3); NOP();} //use in tight spinloops for energy saving on some cpu void XEmitter::CLC() {Write8(0xF8);} //clear carry @@ -577,8 +614,8 @@ void XEmitter::CBW(int bits) void XEmitter::PUSH(X64Reg reg) {WriteSimple1Byte(32, 0x50, reg);} void XEmitter::POP(X64Reg reg) {WriteSimple1Byte(32, 0x58, reg);} -void XEmitter::PUSH(int bits, const OpArg ®) -{ +void XEmitter::PUSH(int bits, const OpArg ®) +{ if (reg.IsSimpleReg()) PUSH(reg.GetSimpleReg()); else if (reg.IsImm()) @@ -599,7 +636,7 @@ void XEmitter::PUSH(int bits, const OpArg ®) Write32((u32)reg.offset); break; default: - _assert_msg_(JIT, 0, "PUSH - Bad imm bits"); + _assert_msg_(DYNA_REC, 0, "PUSH - Bad imm bits"); break; } } @@ -614,7 +651,7 @@ void XEmitter::PUSH(int bits, const OpArg ®) } void XEmitter::POP(int /*bits*/, const OpArg ®) -{ +{ if (reg.IsSimpleReg()) POP(reg.GetSimpleReg()); else @@ -637,7 +674,7 @@ void XEmitter::BSWAP(int bits, X64Reg reg) } else { - _assert_msg_(JIT, 0, "BSWAP - Wrong number of bits"); + _assert_msg_(DYNA_REC, 0, "BSWAP - Wrong number of bits"); } } @@ -651,7 +688,7 @@ void XEmitter::UD2() void XEmitter::PREFETCH(PrefetchLevel level, OpArg arg) { - if (arg.IsImm()) _assert_msg_(JIT, 0, "PREFETCH - Imm argument"); + if (arg.IsImm()) _assert_msg_(DYNA_REC, 0, "PREFETCH - Imm argument");; arg.operandReg = (u8)level; arg.WriteRex(this, 0, 0); Write8(0x0F); @@ -661,7 +698,7 @@ void XEmitter::PREFETCH(PrefetchLevel level, OpArg arg) void XEmitter::SETcc(CCFlags flag, OpArg dest) { - if (dest.IsImm()) _assert_msg_(JIT, 0, "SETcc - Imm argument"); + if (dest.IsImm()) _assert_msg_(DYNA_REC, 0, "SETcc - Imm argument"); dest.operandReg = 0; dest.WriteRex(this, 0, 0); Write8(0x0F); @@ -671,7 +708,7 @@ void XEmitter::SETcc(CCFlags flag, OpArg dest) void XEmitter::CMOVcc(int bits, X64Reg dest, OpArg src, CCFlags flag) { - if (src.IsImm()) _assert_msg_(JIT, 0, "CMOVcc - Imm argument"); + if (src.IsImm()) _assert_msg_(DYNA_REC, 0, "CMOVcc - Imm argument"); src.operandReg = dest; src.WriteRex(this, bits, bits); Write8(0x0F); @@ -681,7 +718,7 @@ void XEmitter::CMOVcc(int bits, X64Reg dest, OpArg src, CCFlags flag) void XEmitter::WriteMulDivType(int bits, OpArg src, int ext) { - if (src.IsImm()) _assert_msg_(JIT, 0, "WriteMulDivType - Imm argument"); + if (src.IsImm()) _assert_msg_(DYNA_REC, 0, "WriteMulDivType - Imm argument"); src.operandReg = ext; if (bits == 16) Write8(0x66); src.WriteRex(this, bits, bits); @@ -705,7 +742,7 @@ void XEmitter::NOT(int bits, OpArg src) {WriteMulDivType(bits, src, 2);} void XEmitter::WriteBitSearchType(int bits, X64Reg dest, OpArg src, u8 byte2) { - if (src.IsImm()) _assert_msg_(JIT, 0, "WriteBitSearchType - Imm argument"); + if (src.IsImm()) _assert_msg_(DYNA_REC, 0, "WriteBitSearchType - Imm argument"); src.operandReg = (u8)dest; if (bits == 16) Write8(0x66); src.WriteRex(this, bits, bits); @@ -716,7 +753,7 @@ void XEmitter::WriteBitSearchType(int bits, X64Reg dest, OpArg src, u8 byte2) void XEmitter::MOVNTI(int bits, OpArg dest, X64Reg src) { - if (bits <= 16) _assert_msg_(JIT, 0, "MOVNTI - bits<=16"); + if (bits <= 16) _assert_msg_(DYNA_REC, 0, "MOVNTI - bits<=16"); WriteBitSearchType(bits, src, dest, 0xC3); } @@ -725,7 +762,7 @@ void XEmitter::BSR(int bits, X64Reg dest, OpArg src) {WriteBitSearchType(bits,de void XEmitter::MOVSX(int dbits, int sbits, X64Reg dest, OpArg src) { - if (src.IsImm()) _assert_msg_(JIT, 0, "MOVSX - Imm argument"); + if (src.IsImm()) _assert_msg_(DYNA_REC, 0, "MOVSX - Imm argument"); if (dbits == sbits) { MOV(dbits, R(dest), src); return; @@ -756,7 +793,7 @@ void XEmitter::MOVSX(int dbits, int sbits, X64Reg dest, OpArg src) void XEmitter::MOVZX(int dbits, int sbits, X64Reg dest, OpArg src) { - if (src.IsImm()) _assert_msg_(JIT, 0, "MOVZX - Imm argument"); + if (src.IsImm()) _assert_msg_(DYNA_REC, 0, "MOVZX - Imm argument"); if (dbits == sbits) { MOV(dbits, R(dest), src); return; @@ -775,6 +812,10 @@ void XEmitter::MOVZX(int dbits, int sbits, X64Reg dest, OpArg src) Write8(0x0F); Write8(0xB7); } + else if (sbits == 32 && dbits == 64) + { + Write8(0x8B); + } else { Crash(); @@ -785,7 +826,7 @@ void XEmitter::MOVZX(int dbits, int sbits, X64Reg dest, OpArg src) void XEmitter::LEA(int bits, X64Reg dest, OpArg src) { - if (src.IsImm()) _assert_msg_(JIT, 0, "LEA - Imm argument"); + if (src.IsImm()) _assert_msg_(DYNA_REC, 0, "LEA - Imm argument"); src.operandReg = (u8)dest; if (bits == 16) Write8(0x66); //TODO: performance warning src.WriteRex(this, bits, bits); @@ -799,11 +840,11 @@ void XEmitter::WriteShift(int bits, OpArg dest, OpArg &shift, int ext) bool writeImm = false; if (dest.IsImm()) { - _assert_msg_(JIT, 0, "WriteShift - can't shift imms"); + _assert_msg_(DYNA_REC, 0, "WriteShift - can't shift imms"); } if ((shift.IsSimpleReg() && shift.GetSimpleReg() != ECX) || (shift.IsImm() && shift.GetImmBits() != 8)) { - _assert_msg_(JIT, 0, "WriteShift - illegal argument"); + _assert_msg_(DYNA_REC, 0, "WriteShift - illegal argument"); } dest.operandReg = ext; if (bits == 16) Write8(0x66); @@ -846,11 +887,11 @@ void XEmitter::WriteBitTest(int bits, OpArg &dest, OpArg &index, int ext) { if (dest.IsImm()) { - _assert_msg_(JIT, 0, "WriteBitTest - can't test imms"); + _assert_msg_(DYNA_REC, 0, "WriteBitTest - can't test imms"); } if ((index.IsImm() && index.GetImmBits() != 8)) { - _assert_msg_(JIT, 0, "WriteBitTest - illegal argument"); + _assert_msg_(DYNA_REC, 0, "WriteBitTest - illegal argument"); } if (bits == 16) Write8(0x66); if (index.IsImm()) @@ -879,15 +920,15 @@ void XEmitter::SHRD(int bits, OpArg dest, OpArg src, OpArg shift) { if (dest.IsImm()) { - _assert_msg_(JIT, 0, "SHRD - can't use imms as destination"); + _assert_msg_(DYNA_REC, 0, "SHRD - can't use imms as destination"); } if (!src.IsSimpleReg()) { - _assert_msg_(JIT, 0, "SHRD - must use simple register as source"); + _assert_msg_(DYNA_REC, 0, "SHRD - must use simple register as source"); } if ((shift.IsSimpleReg() && shift.GetSimpleReg() != ECX) || (shift.IsImm() && shift.GetImmBits() != 8)) { - _assert_msg_(JIT, 0, "SHRD - illegal shift"); + _assert_msg_(DYNA_REC, 0, "SHRD - illegal shift"); } if (bits == 16) Write8(0x66); X64Reg operand = src.GetSimpleReg(); @@ -909,15 +950,15 @@ void XEmitter::SHLD(int bits, OpArg dest, OpArg src, OpArg shift) { if (dest.IsImm()) { - _assert_msg_(JIT, 0, "SHLD - can't use imms as destination"); + _assert_msg_(DYNA_REC, 0, "SHLD - can't use imms as destination"); } if (!src.IsSimpleReg()) { - _assert_msg_(JIT, 0, "SHLD - must use simple register as source"); + _assert_msg_(DYNA_REC, 0, "SHLD - must use simple register as source"); } if ((shift.IsSimpleReg() && shift.GetSimpleReg() != ECX) || (shift.IsImm() && shift.GetImmBits() != 8)) { - _assert_msg_(JIT, 0, "SHLD - illegal shift"); + _assert_msg_(DYNA_REC, 0, "SHLD - illegal shift"); } if (bits == 16) Write8(0x66); X64Reg operand = src.GetSimpleReg(); @@ -952,7 +993,7 @@ void OpArg::WriteNormalOp(XEmitter *emit, bool toRM, NormalOp op, const OpArg &o X64Reg _operandReg = (X64Reg)this->operandReg; if (IsImm()) { - _assert_msg_(JIT, 0, "WriteNormalOp - Imm argument, wrong order"); + _assert_msg_(DYNA_REC, 0, "WriteNormalOp - Imm argument, wrong order"); } if (bits == 16) @@ -967,24 +1008,24 @@ void OpArg::WriteNormalOp(XEmitter *emit, bool toRM, NormalOp op, const OpArg &o if (!toRM) { - _assert_msg_(JIT, 0, "WriteNormalOp - Writing to Imm (!toRM)"); + _assert_msg_(DYNA_REC, 0, "WriteNormalOp - Writing to Imm (!toRM)"); } - if (operand.scale == SCALE_IMM8 && bits == 8) + if (operand.scale == SCALE_IMM8 && bits == 8) { emit->Write8(nops[op].imm8); immToWrite = 8; } else if ((operand.scale == SCALE_IMM16 && bits == 16) || - (operand.scale == SCALE_IMM32 && bits == 32) || - (operand.scale == SCALE_IMM32 && bits == 64)) + (operand.scale == SCALE_IMM32 && bits == 32) || + (operand.scale == SCALE_IMM32 && bits == 64)) { emit->Write8(nops[op].imm32); immToWrite = bits == 16 ? 16 : 32; } else if ((operand.scale == SCALE_IMM8 && bits == 16) || - (operand.scale == SCALE_IMM8 && bits == 32) || - (operand.scale == SCALE_IMM8 && bits == 64)) + (operand.scale == SCALE_IMM8 && bits == 32) || + (operand.scale == SCALE_IMM8 && bits == 64)) { emit->Write8(nops[op].simm8); immToWrite = 8; @@ -997,11 +1038,11 @@ void OpArg::WriteNormalOp(XEmitter *emit, bool toRM, NormalOp op, const OpArg &o emit->Write64((u64)operand.offset); return; } - _assert_msg_(JIT, 0, "WriteNormalOp - Only MOV can take 64-bit imm"); + _assert_msg_(DYNA_REC, 0, "WriteNormalOp - Only MOV can take 64-bit imm"); } else { - _assert_msg_(JIT, 0, "WriteNormalOp - Unhandled case"); + _assert_msg_(DYNA_REC, 0, "WriteNormalOp - Unhandled case"); } _operandReg = (X64Reg)nops[op].ext; //pass extension in REG of ModRM } @@ -1036,7 +1077,7 @@ void OpArg::WriteNormalOp(XEmitter *emit, bool toRM, NormalOp op, const OpArg &o emit->Write32((u32)operand.offset); break; default: - _assert_msg_(JIT, 0, "WriteNormalOp - Unhandled case"); + _assert_msg_(DYNA_REC, 0, "WriteNormalOp - Unhandled case"); } } @@ -1045,7 +1086,7 @@ void XEmitter::WriteNormalOp(XEmitter *emit, int bits, NormalOp op, const OpArg if (a1.IsImm()) { //Booh! Can't write to an imm - _assert_msg_(JIT, 0, "WriteNormalOp - a1 cannot be imm"); + _assert_msg_(DYNA_REC, 0, "WriteNormalOp - a1 cannot be imm"); return; } if (a2.IsImm()) @@ -1072,11 +1113,11 @@ void XEmitter::SBB (int bits, const OpArg &a1, const OpArg &a2) {WriteNormalOp(t void XEmitter::AND (int bits, const OpArg &a1, const OpArg &a2) {WriteNormalOp(this, bits, nrmAND, a1, a2);} void XEmitter::OR (int bits, const OpArg &a1, const OpArg &a2) {WriteNormalOp(this, bits, nrmOR , a1, a2);} void XEmitter::XOR (int bits, const OpArg &a1, const OpArg &a2) {WriteNormalOp(this, bits, nrmXOR, a1, a2);} -void XEmitter::MOV (int bits, const OpArg &a1, const OpArg &a2) +void XEmitter::MOV (int bits, const OpArg &a1, const OpArg &a2) { #ifdef _DEBUG - _assert_msg_(JIT, !a1.IsSimpleReg() || !a2.IsSimpleReg() || a1.GetSimpleReg() != a2.GetSimpleReg(), "Redundant MOV @ %p - bug in JIT?", - code); + _assert_msg_(DYNA_REC, !a1.IsSimpleReg() || !a2.IsSimpleReg() || a1.GetSimpleReg() != a2.GetSimpleReg(), "Redundant MOV @ %p - bug in DYNA_REC?", + code); #endif WriteNormalOp(this, bits, nrmMOV, a1, a2); } @@ -1087,16 +1128,16 @@ void XEmitter::XCHG(int bits, const OpArg &a1, const OpArg &a2) {WriteNormalOp(t void XEmitter::IMUL(int bits, X64Reg regOp, OpArg a1, OpArg a2) { if (bits == 8) { - _assert_msg_(JIT, 0, "IMUL - illegal bit size!"); + _assert_msg_(DYNA_REC, 0, "IMUL - illegal bit size!"); return; } if (a1.IsImm()) { - _assert_msg_(JIT, 0, "IMUL - second arg cannot be imm!"); + _assert_msg_(DYNA_REC, 0, "IMUL - second arg cannot be imm!"); return; } if (!a2.IsImm()) { - _assert_msg_(JIT, 0, "IMUL - third arg must be imm!"); + _assert_msg_(DYNA_REC, 0, "IMUL - third arg must be imm!"); return; } @@ -1118,7 +1159,7 @@ void XEmitter::IMUL(int bits, X64Reg regOp, OpArg a1, OpArg a2) a1.WriteRest(this, 4, regOp); Write32((u32)a2.offset); } else { - _assert_msg_(JIT, 0, "IMUL - unhandled case!"); + _assert_msg_(DYNA_REC, 0, "IMUL - unhandled case!"); } } } @@ -1126,7 +1167,7 @@ void XEmitter::IMUL(int bits, X64Reg regOp, OpArg a1, OpArg a2) void XEmitter::IMUL(int bits, X64Reg regOp, OpArg a) { if (bits == 8) { - _assert_msg_(JIT, 0, "IMUL - illegal bit size!"); + _assert_msg_(DYNA_REC, 0, "IMUL - illegal bit size!"); return; } if (a.IsImm()) @@ -1160,7 +1201,7 @@ void XEmitter::WriteSSEOp(int size, u8 sseOp, bool packed, X64Reg regOp, OpArg a void XEmitter::WriteSSEOp2(int size, u8 sseOp, bool packed, X64Reg regOp, OpArg arg, int extrabytes) { if (size == 64 && packed) - Write8(0x66); //this time, override goes upwards + Write8(0x66); //this time, override goes upwards if (!packed) Write8(size == 64 ? 0xF2 : 0xF3); arg.operandReg = regOp; @@ -1171,6 +1212,18 @@ void XEmitter::WriteSSEOp2(int size, u8 sseOp, bool packed, X64Reg regOp, OpArg arg.WriteRest(this, extrabytes); } +void XEmitter::WriteAVXOp(int size, u8 sseOp, bool packed, X64Reg regOp, OpArg arg, int extrabytes) +{ + WriteAVXOp(size, sseOp, packed, regOp, X64Reg::INVALID_REG, arg, extrabytes); +} + +void XEmitter::WriteAVXOp(int size, u8 sseOp, bool packed, X64Reg regOp1, X64Reg regOp2, OpArg arg, int extrabytes) +{ + arg.WriteVex(this, size, packed, regOp1, regOp2); + Write8(sseOp); + arg.WriteRest(this, extrabytes, regOp1); +} + void XEmitter::MOVD_xmm(X64Reg dest, const OpArg &arg) {WriteSSEOp(64, 0x6E, true, dest, arg, 0);} void XEmitter::MOVD_xmm(const OpArg &arg, X64Reg src) {WriteSSEOp(64, 0x7E, true, src, arg, 0);} @@ -1218,8 +1271,8 @@ void XEmitter::MOVQ_xmm(OpArg arg, X64Reg src) { void XEmitter::WriteMXCSR(OpArg arg, int ext) { - if (arg.IsImm() || arg.IsSimpleReg()) - _assert_msg_(JIT, 0, "MXCSR - invalid operand"); + if (arg.IsImm() || arg.IsSimpleReg()) + _assert_msg_(DYNA_REC, 0, "MXCSR - invalid operand"); arg.operandReg = ext; arg.WriteRex(this, 0, 0); @@ -1278,8 +1331,8 @@ void XEmitter::MAXPD(X64Reg regOp, OpArg arg) {WriteSSEOp(64, sseMAX, true, re void XEmitter::SQRTPS(X64Reg regOp, OpArg arg) {WriteSSEOp(32, sseSQRT, true, regOp, arg);} void XEmitter::SQRTPD(X64Reg regOp, OpArg arg) {WriteSSEOp(64, sseSQRT, true, regOp, arg);} void XEmitter::RSQRTPS(X64Reg regOp, OpArg arg) {WriteSSEOp(32, sseRSQRT, true, regOp, arg);} -void XEmitter::SHUFPS(X64Reg regOp, OpArg arg, u8 shuffle) {WriteSSEOp(32, sseSHUF, true, regOp, arg,1); Write8(shuffle);} -void XEmitter::SHUFPD(X64Reg regOp, OpArg arg, u8 shuffle) {WriteSSEOp(64, sseSHUF, true, regOp, arg,1); Write8(shuffle);} +void XEmitter::SHUFPS(X64Reg regOp, OpArg arg, u8 shuffle) {WriteSSEOp(32, sseSHUF, true, regOp, arg,1); Write8(shuffle);} +void XEmitter::SHUFPD(X64Reg regOp, OpArg arg, u8 shuffle) {WriteSSEOp(64, sseSHUF, true, regOp, arg,1); Write8(shuffle);} void XEmitter::COMISS(X64Reg regOp, OpArg arg) {WriteSSEOp(32, sseCOMIS, true, regOp, arg);} //weird that these should be packed void XEmitter::COMISD(X64Reg regOp, OpArg arg) {WriteSSEOp(64, sseCOMIS, true, regOp, arg);} //ordered @@ -1287,13 +1340,13 @@ void XEmitter::UCOMISS(X64Reg regOp, OpArg arg) {WriteSSEOp(32, sseUCOMIS, true, void XEmitter::UCOMISD(X64Reg regOp, OpArg arg) {WriteSSEOp(64, sseUCOMIS, true, regOp, arg);} void XEmitter::MOVAPS(X64Reg regOp, OpArg arg) {WriteSSEOp(32, sseMOVAPfromRM, true, regOp, arg);} -void XEmitter::MOVAPS(OpArg arg, X64Reg regOp) {WriteSSEOp(32, sseMOVAPtoRM, true, regOp, arg);} -void XEmitter::MOVUPS(X64Reg regOp, OpArg arg) {WriteSSEOp(32, sseMOVUPfromRM, true, regOp, arg);} -void XEmitter::MOVUPS(OpArg arg, X64Reg regOp) {WriteSSEOp(32, sseMOVUPtoRM, true, regOp, arg);} - void XEmitter::MOVAPD(X64Reg regOp, OpArg arg) {WriteSSEOp(64, sseMOVAPfromRM, true, regOp, arg);} +void XEmitter::MOVAPS(OpArg arg, X64Reg regOp) {WriteSSEOp(32, sseMOVAPtoRM, true, regOp, arg);} void XEmitter::MOVAPD(OpArg arg, X64Reg regOp) {WriteSSEOp(64, sseMOVAPtoRM, true, regOp, arg);} + +void XEmitter::MOVUPS(X64Reg regOp, OpArg arg) {WriteSSEOp(32, sseMOVUPfromRM, true, regOp, arg);} void XEmitter::MOVUPD(X64Reg regOp, OpArg arg) {WriteSSEOp(64, sseMOVUPfromRM, true, regOp, arg);} +void XEmitter::MOVUPS(OpArg arg, X64Reg regOp) {WriteSSEOp(32, sseMOVUPtoRM, true, regOp, arg);} void XEmitter::MOVUPD(OpArg arg, X64Reg regOp) {WriteSSEOp(64, sseMOVUPtoRM, true, regOp, arg);} void XEmitter::MOVDQA(X64Reg regOp, OpArg arg) {WriteSSEOp(64, sseMOVDQfromRM, true, regOp, arg);} @@ -1311,7 +1364,7 @@ void XEmitter::CVTPD2PS(X64Reg regOp, OpArg arg) {WriteSSEOp(64, 0x5A, true, reg void XEmitter::CVTSD2SS(X64Reg regOp, OpArg arg) {WriteSSEOp(64, 0x5A, false, regOp, arg);} void XEmitter::CVTSS2SD(X64Reg regOp, OpArg arg) {WriteSSEOp(32, 0x5A, false, regOp, arg);} -void XEmitter::CVTSD2SI(X64Reg regOp, OpArg arg) {WriteSSEOp(32, 0xF2, false, regOp, arg);} +void XEmitter::CVTSD2SI(X64Reg regOp, OpArg arg) {WriteSSEOp(64, 0x2D, false, regOp, arg);} void XEmitter::CVTDQ2PD(X64Reg regOp, OpArg arg) {WriteSSEOp(32, 0xE6, false, regOp, arg);} void XEmitter::CVTDQ2PS(X64Reg regOp, OpArg arg) {WriteSSEOp(32, 0x5B, true, regOp, arg);} @@ -1339,7 +1392,7 @@ void XEmitter::UNPCKHPS(X64Reg dest, OpArg arg) {WriteSSEOp(32, 0x15, true, dest void XEmitter::UNPCKLPD(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0x14, true, dest, arg);} void XEmitter::UNPCKHPD(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0x15, true, dest, arg);} -void XEmitter::MOVDDUP(X64Reg regOp, OpArg arg) +void XEmitter::MOVDDUP(X64Reg regOp, OpArg arg) { if (cpu_info.bSSE3) { @@ -1356,7 +1409,7 @@ void XEmitter::MOVDDUP(X64Reg regOp, OpArg arg) //There are a few more left -// Also some integer instrucitons are missing +// Also some integer instructions are missing void XEmitter::PACKSSDW(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0x6B, true, dest, arg);} void XEmitter::PACKSSWB(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0x63, true, dest, arg);} //void PACKUSDW(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0x66, true, dest, arg);} // WRONG @@ -1515,8 +1568,8 @@ void XEmitter::PCMPGTB(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0x64, true, dest void XEmitter::PCMPGTW(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0x65, true, dest, arg);} void XEmitter::PCMPGTD(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0x66, true, dest, arg);} -void XEmitter::PEXTRW(X64Reg dest, OpArg arg, u8 subreg) { WriteSSEOp(64, 0xC5, true, dest, arg); Write8(subreg); } -void XEmitter::PINSRW(X64Reg dest, OpArg arg, u8 subreg) { WriteSSEOp(64, 0xC4, true, dest, arg); Write8(subreg); } +void XEmitter::PEXTRW(X64Reg dest, OpArg arg, u8 subreg) {WriteSSEOp(64, 0xC5, true, dest, arg); Write8(subreg);} +void XEmitter::PINSRW(X64Reg dest, OpArg arg, u8 subreg) {WriteSSEOp(64, 0xC4, true, dest, arg); Write8(subreg);} void XEmitter::PMADDWD(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0xF5, true, dest, arg); } void XEmitter::PSADBW(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0xF6, true, dest, arg);} @@ -1531,6 +1584,13 @@ void XEmitter::PMOVMSKB(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0xD7, true, d void XEmitter::PSHUFD(X64Reg regOp, OpArg arg, u8 shuffle) {WriteSSEOp(64, 0x70, true, regOp, arg, 1); Write8(shuffle);} void XEmitter::PSHUFLW(X64Reg regOp, OpArg arg, u8 shuffle) {WriteSSEOp(64, 0x70, false, regOp, arg, 1); Write8(shuffle);} +// VEX +void XEmitter::VADDSD(X64Reg regOp1, X64Reg regOp2, OpArg arg) {WriteAVXOp(64, sseADD, false, regOp1, regOp2, arg);} +void XEmitter::VSUBSD(X64Reg regOp1, X64Reg regOp2, OpArg arg) {WriteAVXOp(64, sseSUB, false, regOp1, regOp2, arg);} +void XEmitter::VMULSD(X64Reg regOp1, X64Reg regOp2, OpArg arg) {WriteAVXOp(64, sseMUL, false, regOp1, regOp2, arg);} +void XEmitter::VDIVSD(X64Reg regOp1, X64Reg regOp2, OpArg arg) {WriteAVXOp(64, sseDIV, false, regOp1, regOp2, arg);} +void XEmitter::VSQRTSD(X64Reg regOp1, X64Reg regOp2, OpArg arg) {WriteAVXOp(64, sseSQRT, false, regOp1, regOp2, arg);} + // Prefixes void XEmitter::LOCK() { Write8(0xF0); } diff --git a/Common/x64Emitter.h b/Common/x64Emitter.h index cf739c7a30..2b163ff52a 100644 --- a/Common/x64Emitter.h +++ b/Common/x64Emitter.h @@ -22,10 +22,6 @@ #include "Common.h" -#if !defined(_M_IX86) && !defined(_M_X64) -#error "Don't build this on arm." -#endif - namespace Gen { @@ -33,7 +29,7 @@ enum X64Reg { EAX = 0, EBX = 3, ECX = 1, EDX = 2, ESI = 6, EDI = 7, EBP = 5, ESP = 4, - + RAX = 0, RBX = 3, RCX = 1, RDX = 2, RSI = 6, RDI = 7, RBP = 5, RSP = 4, R8 = 8, R9 = 9, R10 = 10,R11 = 11, @@ -46,9 +42,12 @@ enum X64Reg AX = 0, BX = 3, CX = 1, DX = 2, SI = 6, DI = 7, BP = 5, SP = 4, - XMM0=0, XMM1, XMM2, XMM3, XMM4, XMM5, XMM6, XMM7, + XMM0=0, XMM1, XMM2, XMM3, XMM4, XMM5, XMM6, XMM7, XMM8, XMM9, XMM10, XMM11, XMM12, XMM13, XMM14, XMM15, + YMM0=0, YMM1, YMM2, YMM3, YMM4, YMM5, YMM6, YMM7, + YMM8, YMM9, YMM10, YMM11, YMM12, YMM13, YMM14, YMM15, + INVALID_REG = 0xFFFFFFFF }; @@ -59,7 +58,7 @@ enum CCFlags CC_B = 2, CC_C = 2, CC_NAE = 2, CC_NB = 3, CC_NC = 3, CC_AE = 3, CC_Z = 4, CC_E = 4, - CC_NZ = 5, CC_NE = 5, + CC_NZ = 5, CC_NE = 5, CC_BE = 6, CC_NA = 6, CC_NBE = 7, CC_A = 7, CC_S = 8, @@ -111,8 +110,7 @@ enum NormalOp { nrmXCHG, }; -enum -{ +enum { CMP_EQ = 0, CMP_LT = 1, CMP_LE = 2, @@ -125,6 +123,7 @@ enum class XEmitter; +// RIP addressing does not benefit from micro op fusion on Core arch struct OpArg { OpArg() {} // dummy op arg, used for storage @@ -134,10 +133,11 @@ struct OpArg scale = (u8)_scale; offsetOrBaseReg = (u16)rmReg; indexReg = (u16)scaledReg; - //if scale == 0 never mind offseting + //if scale == 0 never mind offsetting offset = _offset; } void WriteRex(XEmitter *emit, int opBits, int bits, int customOp = -1) const; + void WriteVex(XEmitter* emit, int size, int packed, Gen::X64Reg regOp1, X64Reg regOp2) const; void WriteRest(XEmitter *emit, int extraBytes=0, X64Reg operandReg=(X64Reg)0xFF, bool warn_64bit_offset = true) const; void WriteSingleByteOp(XEmitter *emit, u8 op, X64Reg operandReg, int bits); // This one is public - must be written to @@ -148,6 +148,8 @@ struct OpArg bool IsImm() const {return scale == SCALE_IMM8 || scale == SCALE_IMM16 || scale == SCALE_IMM32 || scale == SCALE_IMM64;} bool IsSimpleReg() const {return scale == SCALE_NONE;} bool IsSimpleReg(X64Reg reg) const { + if (!IsSimpleReg()) + return false; return GetSimpleReg() == reg; } @@ -186,16 +188,17 @@ struct OpArg void IncreaseOffset(int sz) { offset += sz; } + private: u8 scale; u16 offsetOrBaseReg; u16 indexReg; }; -inline OpArg M(const void *ptr) {return OpArg((u64)ptr, (int)SCALE_RIP);} +inline OpArg M(void *ptr) {return OpArg((u64)ptr, (int)SCALE_RIP);} template inline OpArg M(const T *ptr) {return OpArg((u64)(const void *)ptr, (int)SCALE_RIP);} -inline OpArg R(X64Reg value) {return OpArg(0, SCALE_NONE, value);} +inline OpArg R(X64Reg value) {return OpArg(0, SCALE_NONE, value);} inline OpArg MatR(X64Reg value) {return OpArg(0, SCALE_ATREG, value);} inline OpArg MDisp(X64Reg value, int offset) { return OpArg((u32)offset, SCALE_ATREG, value); @@ -224,11 +227,11 @@ inline OpArg SImmAuto(s32 imm) { } #ifdef _M_X64 -inline OpArg ImmPtr(const void *imm) {return Imm64((u64)imm);} +inline OpArg ImmPtr(const void* imm) {return Imm64((u64)imm);} #else -inline OpArg ImmPtr(const void *imm) {return Imm32((u32)imm);} +inline OpArg ImmPtr(const void* imm) {return Imm32((u32)imm);} #endif -inline u32 PtrOffset(const void *ptr, const void *base) { +inline u32 PtrOffset(const void* ptr, const void* base) { #ifdef _M_X64 s64 distance = (s64)ptr-(s64)base; if (distance >= 0x80000000LL || @@ -253,6 +256,18 @@ struct FixupBranch int type; //0 = 8bit 1 = 32bit }; +enum SSECompare +{ + EQ = 0, + LT, + LE, + UNORD, + NEQ, + NLT, + NLE, + ORD, +}; + typedef const u8* JumpTarget; class XEmitter @@ -271,15 +286,12 @@ private: void WriteMXCSR(OpArg arg, int ext); void WriteSSEOp(int size, u8 sseOp, bool packed, X64Reg regOp, OpArg arg, int extrabytes = 0); void WriteSSEOp2(int size, u8 sseOp, bool packed, X64Reg regOp, OpArg arg, int extrabytes = 0); + void WriteAVXOp(int size, u8 sseOp, bool packed, X64Reg regOp, OpArg arg, int extrabytes = 0); + void WriteAVXOp(int size, u8 sseOp, bool packed, X64Reg regOp1, X64Reg regOp2, OpArg arg, int extrabytes = 0); void WriteNormalOp(XEmitter *emit, int bits, NormalOp op, const OpArg &a1, const OpArg &a2); protected: - inline void Write8(u8 value) { - //if (value == 0xcc) { - // value = 0xcc; // set breakpoint here to find where mysterious 0xcc are written - //} - *code++ = value; - } + inline void Write8(u8 value) {*code++ = value;} inline void Write16(u16 value) {*(u16*)code = (value); code += 2;} inline void Write32(u32 value) {*(u32*)code = (value); code += 4;} inline void Write64(u64 value) {*(u64*)code = (value); code += 8;} @@ -301,7 +313,7 @@ public: u8 *GetWritableCodePtr(); // Looking for one of these? It's BANNED!! Some instructions are slow on modern CPU - // INC, DEC, LOOP, LOOPNE, LOOPE, ENTER, LEAVE, XCHG, XLAT, REP MOVSB/MOVSD, REP SCASD + other string instr., + // INC, DEC, LOOP, LOOPNE, LOOPE, ENTER, LEAVE, XCHG, XLAT, REP MOVSB/MOVSD, REP SCASD + other string instr., // INC and DEC are slow on Intel Core, but not on AMD. They create a // false flag dependency because they only update a subset of the flags. // XCHG is SLOW and should be avoided. @@ -390,7 +402,7 @@ public: void DIV(int bits, OpArg src); void IDIV(int bits, OpArg src); - // Shift + // Shift void ROL(int bits, OpArg dest, OpArg shift); void ROR(int bits, OpArg dest, OpArg shift); void RCL(int bits, OpArg dest, OpArg shift); @@ -445,7 +457,7 @@ public: // Sign/zero extension void MOVSX(int dbits, int sbits, X64Reg dest, OpArg src); //automatically uses MOVSXD if necessary - void MOVZX(int dbits, int sbits, X64Reg dest, OpArg src); + void MOVZX(int dbits, int sbits, X64Reg dest, OpArg src); // WARNING - These two take 11-13 cycles and are VectorPath! (AMD64) void STMXCSR(OpArg memloc); @@ -459,25 +471,33 @@ public: void FWAIT(); // SSE/SSE2: Floating point arithmetic - void ADDSS(X64Reg regOp, OpArg arg); - void ADDSD(X64Reg regOp, OpArg arg); - void SUBSS(X64Reg regOp, OpArg arg); - void SUBSD(X64Reg regOp, OpArg arg); - void MULSS(X64Reg regOp, OpArg arg); - void MULSD(X64Reg regOp, OpArg arg); - void DIVSS(X64Reg regOp, OpArg arg); - void DIVSD(X64Reg regOp, OpArg arg); - void MINSS(X64Reg regOp, OpArg arg); - void MINSD(X64Reg regOp, OpArg arg); - void MAXSS(X64Reg regOp, OpArg arg); - void MAXSD(X64Reg regOp, OpArg arg); - void SQRTSS(X64Reg regOp, OpArg arg); - void SQRTSD(X64Reg regOp, OpArg arg); + void ADDSS(X64Reg regOp, OpArg arg); + void ADDSD(X64Reg regOp, OpArg arg); + void SUBSS(X64Reg regOp, OpArg arg); + void SUBSD(X64Reg regOp, OpArg arg); + void MULSS(X64Reg regOp, OpArg arg); + void MULSD(X64Reg regOp, OpArg arg); + void DIVSS(X64Reg regOp, OpArg arg); + void DIVSD(X64Reg regOp, OpArg arg); + void MINSS(X64Reg regOp, OpArg arg); + void MINSD(X64Reg regOp, OpArg arg); + void MAXSS(X64Reg regOp, OpArg arg); + void MAXSD(X64Reg regOp, OpArg arg); + void SQRTSS(X64Reg regOp, OpArg arg); + void SQRTSD(X64Reg regOp, OpArg arg); void RSQRTSS(X64Reg regOp, OpArg arg); // SSE/SSE2: Floating point bitwise (yes) - void CMPSS(X64Reg regOp, OpArg arg, u8 compare); - void CMPSD(X64Reg regOp, OpArg arg, u8 compare); + void CMPSS(X64Reg regOp, OpArg arg, u8 compare); + void CMPSD(X64Reg regOp, OpArg arg, u8 compare); + void ANDSS(X64Reg regOp, OpArg arg); + void ANDSD(X64Reg regOp, OpArg arg); + void ANDNSS(X64Reg regOp, OpArg arg); + void ANDNSD(X64Reg regOp, OpArg arg); + void ORSS(X64Reg regOp, OpArg arg); + void ORSD(X64Reg regOp, OpArg arg); + void XORSS(X64Reg regOp, OpArg arg); + void XORSD(X64Reg regOp, OpArg arg); inline void CMPEQSS(X64Reg regOp, OpArg arg) { CMPSS(regOp, arg, CMP_EQ); } inline void CMPLTSS(X64Reg regOp, OpArg arg) { CMPSS(regOp, arg, CMP_LT); } @@ -487,24 +507,12 @@ public: inline void CMPNLTSS(X64Reg regOp, OpArg arg) { CMPSS(regOp, arg, CMP_NLT); } inline void CMPORDSS(X64Reg regOp, OpArg arg) { CMPSS(regOp, arg, CMP_ORD); } - - // I don't think these exist - /* - void ANDSD(X64Reg regOp, OpArg arg); - void ANDNSS(X64Reg regOp, OpArg arg); - void ANDNSD(X64Reg regOp, OpArg arg); - void ORSS(X64Reg regOp, OpArg arg); - void ORSD(X64Reg regOp, OpArg arg); - void XORSS(X64Reg regOp, OpArg arg); - void XORSD(X64Reg regOp, OpArg arg); - */ - // SSE/SSE2: Floating point packed arithmetic (x4 for float, x2 for double) - void ADDPS(X64Reg regOp, OpArg arg); - void ADDPD(X64Reg regOp, OpArg arg); - void SUBPS(X64Reg regOp, OpArg arg); - void SUBPD(X64Reg regOp, OpArg arg); - void CMPPS(X64Reg regOp, OpArg arg, u8 compare); + void ADDPS(X64Reg regOp, OpArg arg); + void ADDPD(X64Reg regOp, OpArg arg); + void SUBPS(X64Reg regOp, OpArg arg); + void SUBPD(X64Reg regOp, OpArg arg); + void CMPPS(X64Reg regOp, OpArg arg, u8 compare); void CMPPD(X64Reg regOp, OpArg arg, u8 compare); void MULPS(X64Reg regOp, OpArg arg); void MULPD(X64Reg regOp, OpArg arg); @@ -519,8 +527,8 @@ public: void RSQRTPS(X64Reg regOp, OpArg arg); // SSE/SSE2: Floating point packed bitwise (x4 for float, x2 for double) - void ANDPS(X64Reg regOp, OpArg arg); - void ANDPD(X64Reg regOp, OpArg arg); + void ANDPS(X64Reg regOp, OpArg arg); + void ANDPD(X64Reg regOp, OpArg arg); void ANDNPS(X64Reg regOp, OpArg arg); void ANDNPD(X64Reg regOp, OpArg arg); void ORPS(X64Reg regOp, OpArg arg); @@ -529,9 +537,9 @@ public: void XORPD(X64Reg regOp, OpArg arg); // SSE/SSE2: Shuffle components. These are tricky - see Intel documentation. - void SHUFPS(X64Reg regOp, OpArg arg, u8 shuffle); - void SHUFPD(X64Reg regOp, OpArg arg, u8 shuffle); - + void SHUFPS(X64Reg regOp, OpArg arg, u8 shuffle); + void SHUFPD(X64Reg regOp, OpArg arg, u8 shuffle); + // SSE/SSE2: Useful alternative to shuffle in some cases. void MOVDDUP(X64Reg regOp, OpArg arg); @@ -549,18 +557,17 @@ public: void UCOMISS(X64Reg regOp, OpArg arg); void UCOMISD(X64Reg regOp, OpArg arg); - // SSE/SSE2: Moves. Use the right data type for your data to avoid slight penalties on some CPUs. - - // Singles + // SSE/SSE2: Moves. Use the right data type for your data, in most cases. void MOVAPS(X64Reg regOp, OpArg arg); - void MOVAPS(OpArg arg, X64Reg regOp); - void MOVUPS(X64Reg regOp, OpArg arg); - void MOVUPS(OpArg arg, X64Reg regOp); - // Doubles void MOVAPD(X64Reg regOp, OpArg arg); + void MOVAPS(OpArg arg, X64Reg regOp); void MOVAPD(OpArg arg, X64Reg regOp); + + void MOVUPS(X64Reg regOp, OpArg arg); void MOVUPD(X64Reg regOp, OpArg arg); + void MOVUPS(OpArg arg, X64Reg regOp); void MOVUPD(OpArg arg, X64Reg regOp); + // Integers (NOTE: untested - I added these then it turned out I didn't have a use for them after all). void MOVDQA(X64Reg regOp, OpArg arg); void MOVDQA(OpArg arg, X64Reg regOp); @@ -596,11 +603,11 @@ public: void CVTDQ2PS(X64Reg regOp, OpArg arg); void CVTPS2DQ(X64Reg regOp, OpArg arg); + void CVTTSS2SI(X64Reg xregdest, OpArg arg); // Yeah, destination really is a GPR like EAX! + void CVTTPS2DQ(X64Reg regOp, OpArg arg); void CVTSI2SS(X64Reg xregdest, OpArg arg); // Yeah, destination really is a GPR like EAX! void CVTSS2SI(X64Reg xregdest, OpArg arg); // Yeah, destination really is a GPR like EAX! - void CVTTSS2SI(X64Reg xregdest, OpArg arg); // Yeah, destination really is a GPR like EAX! void CVTTSD2SI(X64Reg xregdest, OpArg arg); // Yeah, destination really is a GPR like EAX! - void CVTTPS2DQ(X64Reg regOp, OpArg arg); void CVTTPD2DQ(X64Reg xregdest, OpArg arg); // SSE2: Packed integer instructions @@ -621,57 +628,57 @@ public: void PMOVZXWD(X64Reg dest, const OpArg &arg); void PAND(X64Reg dest, OpArg arg); - void PANDN(X64Reg dest, OpArg arg); - void PXOR(X64Reg dest, OpArg arg); - void POR(X64Reg dest, OpArg arg); + void PANDN(X64Reg dest, OpArg arg); + void PXOR(X64Reg dest, OpArg arg); + void POR(X64Reg dest, OpArg arg); void PADDB(X64Reg dest, OpArg arg); - void PADDW(X64Reg dest, OpArg arg); - void PADDD(X64Reg dest, OpArg arg); - void PADDQ(X64Reg dest, OpArg arg); + void PADDW(X64Reg dest, OpArg arg); + void PADDD(X64Reg dest, OpArg arg); + void PADDQ(X64Reg dest, OpArg arg); - void PADDSB(X64Reg dest, OpArg arg); - void PADDSW(X64Reg dest, OpArg arg); - void PADDUSB(X64Reg dest, OpArg arg); - void PADDUSW(X64Reg dest, OpArg arg); + void PADDSB(X64Reg dest, OpArg arg); + void PADDSW(X64Reg dest, OpArg arg); + void PADDUSB(X64Reg dest, OpArg arg); + void PADDUSW(X64Reg dest, OpArg arg); - void PSUBB(X64Reg dest, OpArg arg); - void PSUBW(X64Reg dest, OpArg arg); - void PSUBD(X64Reg dest, OpArg arg); - void PSUBQ(X64Reg dest, OpArg arg); + void PSUBB(X64Reg dest, OpArg arg); + void PSUBW(X64Reg dest, OpArg arg); + void PSUBD(X64Reg dest, OpArg arg); + void PSUBQ(X64Reg dest, OpArg arg); - void PSUBSB(X64Reg dest, OpArg arg); - void PSUBSW(X64Reg dest, OpArg arg); - void PSUBUSB(X64Reg dest, OpArg arg); - void PSUBUSW(X64Reg dest, OpArg arg); + void PSUBSB(X64Reg dest, OpArg arg); + void PSUBSW(X64Reg dest, OpArg arg); + void PSUBUSB(X64Reg dest, OpArg arg); + void PSUBUSW(X64Reg dest, OpArg arg); - void PAVGB(X64Reg dest, OpArg arg); - void PAVGW(X64Reg dest, OpArg arg); + void PAVGB(X64Reg dest, OpArg arg); + void PAVGW(X64Reg dest, OpArg arg); - void PCMPEQB(X64Reg dest, OpArg arg); - void PCMPEQW(X64Reg dest, OpArg arg); - void PCMPEQD(X64Reg dest, OpArg arg); + void PCMPEQB(X64Reg dest, OpArg arg); + void PCMPEQW(X64Reg dest, OpArg arg); + void PCMPEQD(X64Reg dest, OpArg arg); - void PCMPGTB(X64Reg dest, OpArg arg); - void PCMPGTW(X64Reg dest, OpArg arg); - void PCMPGTD(X64Reg dest, OpArg arg); + void PCMPGTB(X64Reg dest, OpArg arg); + void PCMPGTW(X64Reg dest, OpArg arg); + void PCMPGTD(X64Reg dest, OpArg arg); void PEXTRW(X64Reg dest, OpArg arg, u8 subreg); void PINSRW(X64Reg dest, OpArg arg, u8 subreg); - void PMADDWD(X64Reg dest, OpArg arg); - void PSADBW(X64Reg dest, OpArg arg); + void PMADDWD(X64Reg dest, OpArg arg); + void PSADBW(X64Reg dest, OpArg arg); - void PMAXSW(X64Reg dest, OpArg arg); - void PMAXUB(X64Reg dest, OpArg arg); - void PMINSW(X64Reg dest, OpArg arg); - void PMINUB(X64Reg dest, OpArg arg); + void PMAXSW(X64Reg dest, OpArg arg); + void PMAXUB(X64Reg dest, OpArg arg); + void PMINSW(X64Reg dest, OpArg arg); + void PMINUB(X64Reg dest, OpArg arg); // SSE4 has PMAXSB and PMINSB and PMAXUW and PMINUW too if we need them. - + void PMOVMSKB(X64Reg dest, OpArg arg); + void PSHUFD(X64Reg dest, OpArg arg, u8 shuffle); void PSHUFB(X64Reg dest, OpArg arg); - void PSHUFD(X64Reg dest, OpArg arg, u8 shuffle); void PSHUFLW(X64Reg dest, OpArg arg, u8 shuffle); void PSRLW(X64Reg reg, int shift); @@ -688,13 +695,19 @@ public: void PSRAW(X64Reg reg, int shift); void PSRAD(X64Reg reg, int shift); + // AVX + void VADDSD(X64Reg regOp1, X64Reg regOp2, OpArg arg); + void VSUBSD(X64Reg regOp1, X64Reg regOp2, OpArg arg); + void VMULSD(X64Reg regOp1, X64Reg regOp2, OpArg arg); + void VDIVSD(X64Reg regOp1, X64Reg regOp2, OpArg arg); + void VSQRTSD(X64Reg regOp1, X64Reg regOp2, OpArg arg); + void RTDSC(); // Utility functions // The difference between this and CALL is that this aligns the stack // where appropriate. void ABI_CallFunction(const void *func); - template void ABI_CallFunction(T (*func)()) { ABI_CallFunction((const void *)func); @@ -703,10 +716,9 @@ public: void ABI_CallFunction(const u8 *func) { ABI_CallFunction((const void *)func); } - void ABI_CallFunctionC16(const void *func, u16 param1); void ABI_CallFunctionCC16(const void *func, u32 param1, u16 param2); - + // These only support u32 parameters, but that's enough for a lot of uses. // These will destroy the 1 or 2 first "parameter regs". void ABI_CallFunctionC(const void *func, u32 param1); @@ -783,8 +795,7 @@ public: // Call this when shutting down. Don't rely on the destructor, even though it'll do the job. void FreeCodeSpace(); - bool IsInSpace(const u8 *ptr) const - { + bool IsInSpace(const u8 *ptr) const { return ptr >= region && ptr < region + region_size; } @@ -792,13 +803,11 @@ public: // Start over if you need to change the code (call FreeCodeSpace(), AllocCodeSpace()). void WriteProtect(); - void ResetCodePtr() - { + void ResetCodePtr() { SetCodePtr(region); } - size_t GetSpaceLeft() const - { + size_t GetSpaceLeft() const { return region_size - (GetCodePtr() - region); } From 3b1476c8ecb59982b0349cd85e75a0a2777cd777 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Henrik=20Rydg=C3=A5rd?= Date: Thu, 9 Oct 2014 21:38:25 +0200 Subject: [PATCH 083/105] MIPSTables: Annotate fp and hi/lo in/out more accurately than just "other" Some typo fixes --- Core/Debugger/DisassemblyManager.cpp | 4 +- Core/MIPS/MIPS.h | 10 +-- Core/MIPS/MIPSAnalyst.cpp | 8 +- Core/MIPS/MIPSAnalyst.h | 2 +- Core/MIPS/MIPSInt.cpp | 36 +++------ Core/MIPS/MIPSTables.cpp | 106 +++++++++++++-------------- Core/MIPS/MIPSTables.h | 89 ++++++++++++---------- Windows/Debugger/CtrlDisAsmView.cpp | 2 +- 8 files changed, 124 insertions(+), 133 deletions(-) diff --git a/Core/Debugger/DisassemblyManager.cpp b/Core/Debugger/DisassemblyManager.cpp index 874207b2f6..fa0203878d 100644 --- a/Core/Debugger/DisassemblyManager.cpp +++ b/Core/Debugger/DisassemblyManager.cpp @@ -772,7 +772,7 @@ bool DisassemblyMacro::disassemble(u32 address, DisassemblyLineInfo& dest, bool dest.params = buffer; dest.info.hasRelevantAddress = true; - dest.info.releventAddress = immediate; + dest.info.relevantAddress = immediate; break; case MACRO_MEMORYIMM: dest.name = name; @@ -792,7 +792,7 @@ bool DisassemblyMacro::disassemble(u32 address, DisassemblyLineInfo& dest, bool dest.info.dataSize = dataSize; dest.info.hasRelevantAddress = true; - dest.info.releventAddress = immediate; + dest.info.relevantAddress = immediate; break; default: return false; diff --git a/Core/MIPS/MIPS.h b/Core/MIPS/MIPS.h index 4a69dcd345..61ccc639ff 100644 --- a/Core/MIPS/MIPS.h +++ b/Core/MIPS/MIPS.h @@ -26,8 +26,7 @@ class PointerWrap; typedef Memory::Opcode MIPSOpcode; -enum MIPSGPReg -{ +enum MIPSGPReg { MIPS_REG_ZERO=0, MIPS_REG_COMPILER_SCRATCH=1, @@ -65,17 +64,16 @@ enum MIPSGPReg MIPS_REG_FP=30, MIPS_REG_RA=31, - MIPS_REG_INVALID=-1, - // Not real regs, just for convenience/jit mapping. MIPS_REG_HI = 32, MIPS_REG_LO = 33, MIPS_REG_FPCOND = 34, MIPS_REG_VFPUCC = 35, + + MIPS_REG_INVALID=-1, }; -enum -{ +enum { VFPU_CTRL_SPREFIX, VFPU_CTRL_TPREFIX, VFPU_CTRL_DPREFIX, diff --git a/Core/MIPS/MIPSAnalyst.cpp b/Core/MIPS/MIPSAnalyst.cpp index 626931a650..9dc3ea4cee 100644 --- a/Core/MIPS/MIPSAnalyst.cpp +++ b/Core/MIPS/MIPSAnalyst.cpp @@ -1204,19 +1204,19 @@ skip: case 0x20: // add case 0x21: // addu info.hasRelevantAddress = true; - info.releventAddress = cpu->GetRegValue(0,MIPS_GET_RS(op))+cpu->GetRegValue(0,MIPS_GET_RT(op)); + info.relevantAddress = cpu->GetRegValue(0,MIPS_GET_RS(op))+cpu->GetRegValue(0,MIPS_GET_RT(op)); break; case 0x22: // sub case 0x23: // subu info.hasRelevantAddress = true; - info.releventAddress = cpu->GetRegValue(0,MIPS_GET_RS(op))-cpu->GetRegValue(0,MIPS_GET_RT(op)); + info.relevantAddress = cpu->GetRegValue(0,MIPS_GET_RS(op))-cpu->GetRegValue(0,MIPS_GET_RT(op)); break; } break; case 0x08: // addi case 0x09: // adiu info.hasRelevantAddress = true; - info.releventAddress = cpu->GetRegValue(0,MIPS_GET_RS(op))+((s16)(op & 0xFFFF)); + info.relevantAddress = cpu->GetRegValue(0,MIPS_GET_RS(op))+((s16)(op & 0xFFFF)); break; } @@ -1323,7 +1323,7 @@ skip: info.dataAddress = rs + imm16; info.hasRelevantAddress = true; - info.releventAddress = info.dataAddress; + info.relevantAddress = info.dataAddress; } return info; diff --git a/Core/MIPS/MIPSAnalyst.h b/Core/MIPS/MIPSAnalyst.h index f296e88a66..4eaefbb91b 100644 --- a/Core/MIPS/MIPSAnalyst.h +++ b/Core/MIPS/MIPSAnalyst.h @@ -154,7 +154,7 @@ namespace MIPSAnalyst u32 dataAddress; bool hasRelevantAddress; - u32 releventAddress; + u32 relevantAddress; } MipsOpcodeInfo; MipsOpcodeInfo GetOpcodeInfo(DebugInterface* cpu, u32 address); diff --git a/Core/MIPS/MIPSInt.cpp b/Core/MIPS/MIPSInt.cpp index 330f29a054..792f85a0ae 100644 --- a/Core/MIPS/MIPSInt.cpp +++ b/Core/MIPS/MIPSInt.cpp @@ -74,29 +74,13 @@ int MIPS_SingleStep() #else MIPSOpcode op = Memory::Read_Opcode_JIT(mipsr4k.pc); #endif - /* - // Choke on VFPU - MIPSInfo info = MIPSGetInfo(op); - if (info & IS_VFPU) - { - if (!Core_IsStepping() && !GetAsyncKeyState(VK_LSHIFT)) - { - Core_EnableStepping(true); - return; - } - }*/ - - if (mipsr4k.inDelaySlot) - { + if (mipsr4k.inDelaySlot) { MIPSInterpret(op); - if (mipsr4k.inDelaySlot) - { + if (mipsr4k.inDelaySlot) { mipsr4k.pc = mipsr4k.nextPC; mipsr4k.inDelaySlot = false; } - } - else - { + } else { MIPSInterpret(op); } return 1; @@ -872,14 +856,12 @@ namespace MIPSInt int pos = _POS; // Don't change $zr. - if (rt == 0) - { + if (rt == 0) { PC += 4; return; } - switch (op & 0x3f) - { + switch (op & 0x3f) { case 0x0: //ext { int size = _SIZE + 1; @@ -1025,10 +1007,10 @@ namespace MIPSInt switch (op & 0x3f) { - case 0: F(fd) = F(fs) + F(ft); break; //add - case 1: F(fd) = F(fs) - F(ft); break; //sub - case 2: F(fd) = F(fs) * F(ft); break; //mul - case 3: F(fd) = F(fs) / F(ft); break; //div + case 0: F(fd) = F(fs) + F(ft); break; // add.s + case 1: F(fd) = F(fs) - F(ft); break; // sub.s + case 2: F(fd) = F(fs) * F(ft); break; // mul.s + case 3: F(fd) = F(fs) / F(ft); break; // div.s default: _dbg_assert_msg_(CPU,0,"Trying to interpret FPU3Op instruction that can't be interpreted"); break; diff --git a/Core/MIPS/MIPSTables.cpp b/Core/MIPS/MIPSTables.cpp index fe5e269904..ff03a09fbc 100644 --- a/Core/MIPS/MIPSTables.cpp +++ b/Core/MIPS/MIPSTables.cpp @@ -31,8 +31,7 @@ #include "JitCommon/JitCommon.h" -enum MipsEncoding -{ +enum MipsEncoding { Imme, Spec, Spe2, @@ -66,8 +65,7 @@ enum MipsEncoding Inval = -2, }; -struct MIPSInstruction -{ +struct MIPSInstruction { MipsEncoding altEncoding; const char *name; MIPSComp::MIPSCompileFunc compile; @@ -152,7 +150,7 @@ const MIPSInstruction tableImmediate[64] = // xxxxxx ..... ..... ............... INVALID, INVALID, INSTR("swr", &Jit::Comp_ITypeMem, Dis_ITypeMem, Int_ITypeMem, IN_IMM16|IN_RS_ADDR|IN_RT|OUT_MEM|MEMTYPE_WORD), - INSTR("cache", &Jit::Comp_Cache, Dis_Cache, Int_Cache, IN_MEM|IN_IMM16|IN_RS_ADDR|IN_OTHER|OUT_OTHER), + INSTR("cache", &Jit::Comp_Cache, Dis_Cache, Int_Cache, IN_MEM|IN_IMM16|IN_RS_ADDR), //48 INSTR("ll", &Jit::Comp_Generic, Dis_Generic, Int_StoreSync, IN_MEM|IN_IMM16|IN_RS_ADDR|OUT_RT|OUT_OTHER|MEMTYPE_WORD), INSTR("lwc1", &Jit::Comp_FPULS, Dis_FPULS, Int_FPULS, IN_MEM|IN_IMM16|IN_RS_ADDR|OUT_OTHER|MEMTYPE_FLOAT), @@ -198,22 +196,22 @@ const MIPSInstruction tableSpecial[64] = // 000000 ..... ..... ..... ..... xxxxx INSTR("sync", &Jit::Comp_DoNothing, Dis_Generic, Int_Sync, 0), //16 - INSTR("mfhi", &Jit::Comp_MulDivType, Dis_FromHiloTransfer, Int_MulDivType, OUT_RD|IN_OTHER), - INSTR("mthi", &Jit::Comp_MulDivType, Dis_ToHiloTransfer, Int_MulDivType, IN_RS|OUT_OTHER), - INSTR("mflo", &Jit::Comp_MulDivType, Dis_FromHiloTransfer, Int_MulDivType, OUT_RD|IN_OTHER), - INSTR("mtlo", &Jit::Comp_MulDivType, Dis_ToHiloTransfer, Int_MulDivType, IN_RS|OUT_OTHER), + INSTR("mfhi", &Jit::Comp_MulDivType, Dis_FromHiloTransfer, Int_MulDivType, OUT_RD|IN_HI), + INSTR("mthi", &Jit::Comp_MulDivType, Dis_ToHiloTransfer, Int_MulDivType, IN_RS|OUT_HI), + INSTR("mflo", &Jit::Comp_MulDivType, Dis_FromHiloTransfer, Int_MulDivType, OUT_RD|IN_LO), + INSTR("mtlo", &Jit::Comp_MulDivType, Dis_ToHiloTransfer, Int_MulDivType, IN_RS|OUT_LO), INVALID, INVALID, INSTR("clz", &Jit::Comp_RType2, Dis_RType2, Int_RType2, OUT_RD|IN_RS), INSTR("clo", &Jit::Comp_RType2, Dis_RType2, Int_RType2, OUT_RD|IN_RS), //24 - INSTR("mult", &Jit::Comp_MulDivType, Dis_MulDivType, Int_MulDivType, IN_RS|IN_RT|OUT_OTHER), - INSTR("multu", &Jit::Comp_MulDivType, Dis_MulDivType, Int_MulDivType, IN_RS|IN_RT|OUT_OTHER), - INSTR("div", &Jit::Comp_MulDivType, Dis_MulDivType, Int_MulDivType, IN_RS|IN_RT|OUT_OTHER), - INSTR("divu", &Jit::Comp_MulDivType, Dis_MulDivType, Int_MulDivType, IN_RS|IN_RT|OUT_OTHER), - INSTR("madd", &Jit::Comp_MulDivType, Dis_MulDivType, Int_MulDivType, IN_RS|IN_RT|IN_OTHER|OUT_OTHER), - INSTR("maddu", &Jit::Comp_MulDivType, Dis_MulDivType, Int_MulDivType, IN_RS|IN_RT|IN_OTHER|OUT_OTHER), + INSTR("mult", &Jit::Comp_MulDivType, Dis_MulDivType, Int_MulDivType, IN_RS|IN_RT|OUT_HI|OUT_LO), + INSTR("multu", &Jit::Comp_MulDivType, Dis_MulDivType, Int_MulDivType, IN_RS|IN_RT|OUT_HI|OUT_LO), + INSTR("div", &Jit::Comp_MulDivType, Dis_MulDivType, Int_MulDivType, IN_RS|IN_RT|OUT_HI|OUT_LO), + INSTR("divu", &Jit::Comp_MulDivType, Dis_MulDivType, Int_MulDivType, IN_RS|IN_RT|OUT_HI|OUT_LO), + INSTR("madd", &Jit::Comp_MulDivType, Dis_MulDivType, Int_MulDivType, IN_RS|IN_RT|IN_HI|IN_LO|OUT_HI|OUT_LO), + INSTR("maddu", &Jit::Comp_MulDivType, Dis_MulDivType, Int_MulDivType, IN_RS|IN_RT|IN_HI|IN_LO|OUT_HI|OUT_LO), INVALID, INVALID, @@ -234,8 +232,8 @@ const MIPSInstruction tableSpecial[64] = // 000000 ..... ..... ..... ..... xxxxx INSTR("sltu", &Jit::Comp_RType3, Dis_RType3, Int_RType3, IN_RS|IN_RT|OUT_RD), INSTR("max", &Jit::Comp_RType3, Dis_RType3, Int_RType3, IN_RS|IN_RT|OUT_RD), INSTR("min", &Jit::Comp_RType3, Dis_RType3, Int_RType3, IN_RS|IN_RT|OUT_RD), - INSTR("msub", &Jit::Comp_MulDivType, Dis_MulDivType, Int_MulDivType, IN_RS|IN_RT|IN_OTHER|OUT_OTHER), - INSTR("msubu", &Jit::Comp_MulDivType, Dis_MulDivType, Int_MulDivType, IN_RS|IN_RT|IN_OTHER|OUT_OTHER), + INSTR("msub", &Jit::Comp_MulDivType, Dis_MulDivType, Int_MulDivType, IN_RS|IN_RT|IN_HI|IN_LO|OUT_HI|OUT_LO), + INSTR("msubu", &Jit::Comp_MulDivType, Dis_MulDivType, Int_MulDivType, IN_RS|IN_RT|IN_HI|IN_LO|OUT_HI|OUT_LO), //48 INSTR("tge", &Jit::Comp_Generic, Dis_RType3, 0, 0), @@ -262,9 +260,9 @@ const MIPSInstruction tableSpecial2[64] = // 011100 ..... ..... ..... ..... xxxx INVALID_X_8, //32 INVALID, INVALID, INVALID, INVALID, - INSTR("mfic", &Jit::Comp_Generic, Dis_Generic, Int_Special2, 0), + INSTR("mfic", &Jit::Comp_Generic, Dis_Generic, Int_Special2, OUT_OTHER), INVALID, - INSTR("mtic", &Jit::Comp_Generic, Dis_Generic, Int_Special2, 0), + INSTR("mtic", &Jit::Comp_Generic, Dis_Generic, Int_Special2, OUT_OTHER), INVALID, //40 INVALID_X_8, @@ -369,11 +367,11 @@ const MIPSInstruction tableCop2BC2[4] = // 010010 01000 ...xx ................ const MIPSInstruction tableCop0[32] = // 010000 xxxxx ..... ................ { - INSTR("mfc0", &Jit::Comp_Generic, Dis_Generic, 0, OUT_RT), + INSTR("mfc0", &Jit::Comp_Generic, Dis_Generic, 0, OUT_RT), // unused INVALID, INVALID, INVALID, - INSTR("mtc0", &Jit::Comp_Generic, Dis_Generic, 0, IN_RT), + INSTR("mtc0", &Jit::Comp_Generic, Dis_Generic, 0, IN_RT), // unused INVALID, INVALID, INVALID, @@ -423,11 +421,11 @@ const MIPSInstruction tableCop0CO[64] = // 010000 1.... ..... ..... ..... xxxxxx const MIPSInstruction tableCop1[32] = // 010001 xxxxx ..... ..... ........... { - INSTR("mfc1", &Jit::Comp_mxc1, Dis_mxc1, Int_mxc1, IN_OTHER|OUT_RT), + INSTR("mfc1", &Jit::Comp_mxc1, Dis_mxc1, Int_mxc1, IN_FS|OUT_RT), INVALID, INSTR("cfc1", &Jit::Comp_mxc1, Dis_mxc1, Int_mxc1, IN_OTHER|IN_FPUFLAG|OUT_RT), INVALID, - INSTR("mtc1", &Jit::Comp_mxc1, Dis_mxc1, Int_mxc1, IN_RT|OUT_OTHER), + INSTR("mtc1", &Jit::Comp_mxc1, Dis_mxc1, Int_mxc1, IN_RT|OUT_FS), INVALID, INSTR("ctc1", &Jit::Comp_mxc1, Dis_mxc1, Int_mxc1, IN_RT|OUT_FPUFLAG|OUT_OTHER), INVALID, @@ -455,20 +453,20 @@ const MIPSInstruction tableCop1BC[32] = // 010001 01000 xxxxx ................ const MIPSInstruction tableCop1S[64] = // 010001 10000 ..... ..... ..... xxxxxx { - INSTR("add.s", &Jit::Comp_FPU3op, Dis_FPU3op, Int_FPU3op, IN_OTHER|OUT_OTHER), - INSTR("sub.s", &Jit::Comp_FPU3op, Dis_FPU3op, Int_FPU3op, IN_OTHER|OUT_OTHER), - INSTR("mul.s", &Jit::Comp_FPU3op, Dis_FPU3op, Int_FPU3op, IN_OTHER|OUT_OTHER), - INSTR("div.s", &Jit::Comp_FPU3op, Dis_FPU3op, Int_FPU3op, IN_OTHER|OUT_OTHER), - INSTR("sqrt.s", &Jit::Comp_FPU2op, Dis_FPU2op, Int_FPU2op, IN_OTHER|OUT_OTHER), - INSTR("abs.s", &Jit::Comp_FPU2op, Dis_FPU2op, Int_FPU2op, IN_OTHER|OUT_OTHER), - INSTR("mov.s", &Jit::Comp_FPU2op, Dis_FPU2op, Int_FPU2op, IN_OTHER|OUT_OTHER), - INSTR("neg.s", &Jit::Comp_FPU2op, Dis_FPU2op, Int_FPU2op, IN_OTHER|OUT_OTHER), + INSTR("add.s", &Jit::Comp_FPU3op, Dis_FPU3op, Int_FPU3op, OUT_FD|IN_FS|IN_FT), + INSTR("sub.s", &Jit::Comp_FPU3op, Dis_FPU3op, Int_FPU3op, OUT_FD|IN_FS|IN_FT), + INSTR("mul.s", &Jit::Comp_FPU3op, Dis_FPU3op, Int_FPU3op, OUT_FD|IN_FS|IN_FT), + INSTR("div.s", &Jit::Comp_FPU3op, Dis_FPU3op, Int_FPU3op, OUT_FD|IN_FS|IN_FT), + INSTR("sqrt.s", &Jit::Comp_FPU2op, Dis_FPU2op, Int_FPU2op, OUT_FD|IN_FS), + INSTR("abs.s", &Jit::Comp_FPU2op, Dis_FPU2op, Int_FPU2op, OUT_FD|IN_FS), + INSTR("mov.s", &Jit::Comp_FPU2op, Dis_FPU2op, Int_FPU2op, OUT_FD|IN_FS), + INSTR("neg.s", &Jit::Comp_FPU2op, Dis_FPU2op, Int_FPU2op, OUT_FD|IN_FS), //8 INVALID, INVALID, INVALID, INVALID, - INSTR("round.w.s", &Jit::Comp_FPU2op, Dis_FPU2op, Int_FPU2op, IN_OTHER|OUT_OTHER), - INSTR("trunc.w.s", &Jit::Comp_FPU2op, Dis_FPU2op, Int_FPU2op, IN_OTHER|OUT_OTHER), - INSTR("ceil.w.s", &Jit::Comp_FPU2op, Dis_FPU2op, Int_FPU2op, IN_OTHER|OUT_OTHER), - INSTR("floor.w.s", &Jit::Comp_FPU2op, Dis_FPU2op, Int_FPU2op, IN_OTHER|OUT_OTHER), + INSTR("round.w.s", &Jit::Comp_FPU2op, Dis_FPU2op, Int_FPU2op, OUT_FD|IN_FS), + INSTR("trunc.w.s", &Jit::Comp_FPU2op, Dis_FPU2op, Int_FPU2op, OUT_FD|IN_FS), + INSTR("ceil.w.s", &Jit::Comp_FPU2op, Dis_FPU2op, Int_FPU2op, OUT_FD|IN_FS), + INSTR("floor.w.s", &Jit::Comp_FPU2op, Dis_FPU2op, Int_FPU2op, OUT_FD|IN_FS), //16 INVALID_X_8, //24 @@ -476,29 +474,29 @@ const MIPSInstruction tableCop1S[64] = // 010001 10000 ..... ..... ..... xxxxxx //32 INVALID, INVALID, INVALID, INVALID, //36 - INSTR("cvt.w.s", &Jit::Comp_FPU2op, Dis_FPU2op, Int_FPU2op, IN_OTHER|OUT_OTHER), + INSTR("cvt.w.s", &Jit::Comp_FPU2op, Dis_FPU2op, Int_FPU2op, OUT_FD|IN_FS), INVALID, INSTR("dis.int", &Jit::Comp_Generic, Dis_Generic, Int_Interrupt, 0), INVALID, //40 INVALID_X_8, //48 - 010001 10000 ..... ..... ..... 11xxxx - INSTR("c.f", &Jit::Comp_FPUComp, Dis_FPUComp, Int_FPUComp, IN_OTHER|OUT_FPUFLAG), - INSTR("c.un", &Jit::Comp_FPUComp, Dis_FPUComp, Int_FPUComp, IN_OTHER|OUT_FPUFLAG), - INSTR("c.eq", &Jit::Comp_FPUComp, Dis_FPUComp, Int_FPUComp, IN_OTHER|OUT_FPUFLAG), - INSTR("c.ueq", &Jit::Comp_FPUComp, Dis_FPUComp, Int_FPUComp, IN_OTHER|OUT_FPUFLAG), - INSTR("c.olt", &Jit::Comp_FPUComp, Dis_FPUComp, Int_FPUComp, IN_OTHER|OUT_FPUFLAG), - INSTR("c.ult", &Jit::Comp_FPUComp, Dis_FPUComp, Int_FPUComp, IN_OTHER|OUT_FPUFLAG), - INSTR("c.ole", &Jit::Comp_FPUComp, Dis_FPUComp, Int_FPUComp, IN_OTHER|OUT_FPUFLAG), - INSTR("c.ule", &Jit::Comp_FPUComp, Dis_FPUComp, Int_FPUComp, IN_OTHER|OUT_FPUFLAG), - INSTR("c.sf", &Jit::Comp_FPUComp, Dis_FPUComp, Int_FPUComp, IN_OTHER|OUT_FPUFLAG), - INSTR("c.ngle",&Jit::Comp_FPUComp, Dis_FPUComp, Int_FPUComp, IN_OTHER|OUT_FPUFLAG), - INSTR("c.seq", &Jit::Comp_FPUComp, Dis_FPUComp, Int_FPUComp, IN_OTHER|OUT_FPUFLAG), - INSTR("c.ngl", &Jit::Comp_FPUComp, Dis_FPUComp, Int_FPUComp, IN_OTHER|OUT_FPUFLAG), - INSTR("c.lt", &Jit::Comp_FPUComp, Dis_FPUComp, Int_FPUComp, IN_OTHER|OUT_FPUFLAG), - INSTR("c.nge", &Jit::Comp_FPUComp, Dis_FPUComp, Int_FPUComp, IN_OTHER|OUT_FPUFLAG), - INSTR("c.le", &Jit::Comp_FPUComp, Dis_FPUComp, Int_FPUComp, IN_OTHER|OUT_FPUFLAG), - INSTR("c.ngt", &Jit::Comp_FPUComp, Dis_FPUComp, Int_FPUComp, IN_OTHER|OUT_FPUFLAG), + INSTR("c.f", &Jit::Comp_FPUComp, Dis_FPUComp, Int_FPUComp, OUT_FPUFLAG), + INSTR("c.un", &Jit::Comp_FPUComp, Dis_FPUComp, Int_FPUComp, IN_FS|IN_FT|OUT_FPUFLAG), + INSTR("c.eq", &Jit::Comp_FPUComp, Dis_FPUComp, Int_FPUComp, IN_FS|IN_FT|OUT_FPUFLAG), + INSTR("c.ueq", &Jit::Comp_FPUComp, Dis_FPUComp, Int_FPUComp, IN_FS|IN_FT|OUT_FPUFLAG), + INSTR("c.olt", &Jit::Comp_FPUComp, Dis_FPUComp, Int_FPUComp, IN_FS|IN_FT|OUT_FPUFLAG), + INSTR("c.ult", &Jit::Comp_FPUComp, Dis_FPUComp, Int_FPUComp, IN_FS|IN_FT|OUT_FPUFLAG), + INSTR("c.ole", &Jit::Comp_FPUComp, Dis_FPUComp, Int_FPUComp, IN_FS|IN_FT|OUT_FPUFLAG), + INSTR("c.ule", &Jit::Comp_FPUComp, Dis_FPUComp, Int_FPUComp, IN_FS|IN_FT|OUT_FPUFLAG), + INSTR("c.sf", &Jit::Comp_FPUComp, Dis_FPUComp, Int_FPUComp, OUT_FPUFLAG), + INSTR("c.ngle",&Jit::Comp_FPUComp, Dis_FPUComp, Int_FPUComp, IN_FS|IN_FT|OUT_FPUFLAG), + INSTR("c.seq", &Jit::Comp_FPUComp, Dis_FPUComp, Int_FPUComp, IN_FS|IN_FT|OUT_FPUFLAG), + INSTR("c.ngl", &Jit::Comp_FPUComp, Dis_FPUComp, Int_FPUComp, IN_FS|IN_FT|OUT_FPUFLAG), + INSTR("c.lt", &Jit::Comp_FPUComp, Dis_FPUComp, Int_FPUComp, IN_FS|IN_FT|OUT_FPUFLAG), + INSTR("c.nge", &Jit::Comp_FPUComp, Dis_FPUComp, Int_FPUComp, IN_FS|IN_FT|OUT_FPUFLAG), + INSTR("c.le", &Jit::Comp_FPUComp, Dis_FPUComp, Int_FPUComp, IN_FS|IN_FT|OUT_FPUFLAG), + INSTR("c.ngt", &Jit::Comp_FPUComp, Dis_FPUComp, Int_FPUComp, IN_FS|IN_FT|OUT_FPUFLAG), }; const MIPSInstruction tableCop1W[64] = // 010001 10100 ..... ..... ..... xxxxxx @@ -511,7 +509,7 @@ const MIPSInstruction tableCop1W[64] = // 010001 10100 ..... ..... ..... xxxxxx //24 INVALID_X_8, //32 - INSTR("cvt.s.w", &Jit::Comp_FPU2op, Dis_FPU2op, Int_FPU2op, IN_OTHER|OUT_OTHER), + INSTR("cvt.s.w", &Jit::Comp_FPU2op, Dis_FPU2op, Int_FPU2op, OUT_FD|IN_FS), INVALID, INVALID, INVALID, //36 INVALID, @@ -890,8 +888,6 @@ const MIPSInstruction *mipsTables[NumEncodings] = 0, }; - - //arm encoding table //const MIPSInstruction mipsinstructions[] = //{ diff --git a/Core/MIPS/MIPSTables.h b/Core/MIPS/MIPSTables.h index b05e04e199..1f70366eb2 100644 --- a/Core/MIPS/MIPSTables.h +++ b/Core/MIPS/MIPSTables.h @@ -25,14 +25,14 @@ struct MIPSInfo { value = 0; } - explicit MIPSInfo(u32 v) : value(v) { + explicit MIPSInfo(u64 v) : value(v) { } - u32 operator & (const u32 &arg) const { + u64 operator & (const u32 &arg) const { return value & arg; } - u32 value; + u64 value; }; #define CONDTYPE_MASK 0x00000007 @@ -49,44 +49,59 @@ struct MIPSInfo { // as long as the other flags are checked, // there is no way to misinterpret these // as CONDTYPE_X -#define MEMTYPE_MASK 0x00000007 -#define MEMTYPE_BYTE 0x00000001 -#define MEMTYPE_HWORD 0x00000002 -#define MEMTYPE_WORD 0x00000003 -#define MEMTYPE_FLOAT 0x00000004 -#define MEMTYPE_VQUAD 0x00000005 +#define MEMTYPE_MASK 0x00000007ULL +#define MEMTYPE_BYTE 0x00000001ULL +#define MEMTYPE_HWORD 0x00000002ULL +#define MEMTYPE_WORD 0x00000003ULL +#define MEMTYPE_FLOAT 0x00000004ULL +#define MEMTYPE_VQUAD 0x00000005ULL -#define IS_CONDMOVE 0x00000008 -#define DELAYSLOT 0x00000010 -#define BAD_INSTRUCTION 0x00000020 -#define LIKELY 0x00000040 -#define IS_CONDBRANCH 0x00000080 -#define IS_JUMP 0x00000100 +#define IS_CONDMOVE 0x00000008ULL +#define DELAYSLOT 0x00000010ULL +#define BAD_INSTRUCTION 0x00000020ULL +#define LIKELY 0x00000040ULL +#define IS_CONDBRANCH 0x00000080ULL +#define IS_JUMP 0x00000100ULL -#define IN_RS 0x00000200 -#define IN_RS_ADDR (0x00000400 | IN_RS) -#define IN_RS_SHIFT (0x00000800 | IN_RS) -#define IN_RT 0x00001000 -#define IN_SA 0x00002000 -#define IN_IMM16 0x00004000 -#define IN_IMM26 0x00008000 -#define IN_MEM 0x00010000 -#define IN_OTHER 0x00020000 -#define IN_FPUFLAG 0x00040000 -#define IN_VFPU_CC 0x00080000 +#define IN_RS 0x00000200ULL +#define IN_RS_ADDR (0x00000400ULL | IN_RS) +#define IN_RS_SHIFT (0x00000800ULL | IN_RS) +#define IN_RT 0x00001000ULL +#define IN_SA 0x00002000ULL +#define IN_IMM16 0x00004000ULL +#define IN_IMM26 0x00008000ULL +#define IN_MEM 0x00010000ULL +#define IN_OTHER 0x00020000ULL +#define IN_FPUFLAG 0x00040000ULL +#define IN_VFPU_CC 0x00080000ULL -#define OUT_RT 0x00100000 -#define OUT_RD 0x00200000 -#define OUT_RA 0x00400000 -#define OUT_MEM 0x00800000 -#define OUT_OTHER 0x01000000 -#define OUT_FPUFLAG 0x02000000 -#define OUT_VFPU_CC 0x04000000 -#define OUT_EAT_PREFIX 0x08000000 +#define OUT_RT 0x00100000ULL +#define OUT_RD 0x00200000ULL +#define OUT_RA 0x00400000ULL +#define OUT_MEM 0x00800000ULL +#define OUT_OTHER 0x01000000ULL +#define OUT_FPUFLAG 0x02000000ULL +#define OUT_VFPU_CC 0x04000000ULL +#define OUT_EAT_PREFIX 0x08000000ULL -#define VFPU_NO_PREFIX 0x10000000 -#define IS_VFPU 0x20000000 -#define IS_FPU 0x40000000 +#define VFPU_NO_PREFIX 0x10000000ULL +#define IS_VFPU 0x20000000ULL +#define IS_FPU 0x40000000ULL + +#define IN_FS 0x000100000000ULL +#define IN_FT 0x000200000000ULL +#define IN_LO 0x000400000000ULL +#define IN_HI 0x000800000000ULL + +#define OUT_FD 0x001000000000ULL +#define OUT_FS 0x002000000000ULL +#define OUT_LO 0x004000000000ULL +#define OUT_HI 0x008000000000ULL + +#define IN_VS 0x010000000000ULL +#define IN_VT 0x020000000000ULL + +#define OUT_VD 0x100000000000ULL #ifndef CDECL #define CDECL diff --git a/Windows/Debugger/CtrlDisAsmView.cpp b/Windows/Debugger/CtrlDisAsmView.cpp index ec90a340e4..3fe114c135 100644 --- a/Windows/Debugger/CtrlDisAsmView.cpp +++ b/Windows/Debugger/CtrlDisAsmView.cpp @@ -643,7 +643,7 @@ void CtrlDisAsmView::followBranch() } else if (line.info.hasRelevantAddress) { // well, not exactly a branch, but we can do something anyway - SendMessage(GetParent(wnd),WM_DEB_GOTOHEXEDIT,line.info.releventAddress,0); + SendMessage(GetParent(wnd),WM_DEB_GOTOHEXEDIT,line.info.relevantAddress,0); SetFocus(wnd); } } else if (line.type == DISTYPE_DATA) From 7bde97606919f8a012b5991645903e21d906daf1 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Henrik=20Rydg=C3=A5rd?= Date: Fri, 10 Oct 2014 20:41:00 +0200 Subject: [PATCH 084/105] Merge x64 emitter from a newer Dolphin version. This one can generate slightly smaller code by exploiting some EAX-only encoding and various other short forms, and adds support for many newer CPU instructions. --- Common/CPUDetect.cpp | 43 ++ Common/CPUDetect.h | 5 + Common/x64Emitter.cpp | 1086 ++++++++++++++++++++++++++--------------- Common/x64Emitter.h | 233 ++++++--- 4 files changed, 912 insertions(+), 455 deletions(-) diff --git a/Common/CPUDetect.cpp b/Common/CPUDetect.cpp index cb2e72eb23..188fbdaf28 100644 --- a/Common/CPUDetect.cpp +++ b/Common/CPUDetect.cpp @@ -49,6 +49,17 @@ void do_cpuid(u32 regs[4], u32 cpuid_leaf) { #ifdef _M_SSE #include + +#define _XCR_XFEATURE_ENABLED_MASK 0 +static unsigned long long _xgetbv(unsigned int index) +{ + unsigned int eax, edx; + __asm__ __volatile__("xgetbv" : "=a"(eax), "=d"(edx) : "c"(index)); + return ((unsigned long long)edx << 32) | eax; +} + +#else +#define _XCR_XFEATURE_ENABLED_MASK 0 #endif #if defined __FreeBSD__ @@ -172,6 +183,38 @@ void CPUInfo::Detect() { bFMA = true; } if ((cpu_id[2] >> 25) & 1) bAES = true; + + if ((cpu_id[3] >> 24) & 1) + { + // We can use FXSAVE. + bFXSR = true; + } + + // AVX support requires 3 separate checks: + // - Is the AVX bit set in CPUID? + // - Is the XSAVE bit set in CPUID? + // - XGETBV result has the XCR bit set. + if (((cpu_id[2] >> 28) & 1) && ((cpu_id[2] >> 27) & 1)) + { + if ((_xgetbv(_XCR_XFEATURE_ENABLED_MASK) & 0x6) == 0x6) + { + bAVX = true; + if ((cpu_id[2] >> 12) & 1) + bFMA = true; + } + } + + if (max_std_fn >= 7) + { + do_cpuid(cpu_id, 0x00000007); + // careful; we can't enable AVX2 unless the XSAVE/XGETBV checks above passed + if ((cpu_id[1] >> 5) & 1) + bAVX2 = bAVX; + if ((cpu_id[1] >> 3) & 1) + bBMI1 = true; + if ((cpu_id[1] >> 8) & 1) + bBMI2 = true; + } } if (max_ex_fn >= 0x80000004) { // Extract brand string diff --git a/Common/CPUDetect.h b/Common/CPUDetect.h index 04c615b412..091e8f9713 100644 --- a/Common/CPUDetect.h +++ b/Common/CPUDetect.h @@ -56,10 +56,15 @@ struct CPUInfo { bool bLZCNT; bool bSSE4A; bool bAVX; + bool bAVX2; bool bFMA; bool bAES; bool bLAHFSAHF64; bool bLongMode; + bool bBMI1; + bool bBMI2; + bool bMOVBE; + bool bFXSR; // ARM specific CPUInfo bool bSwp; diff --git a/Common/x64Emitter.cpp b/Common/x64Emitter.cpp index f454296470..c4455067e4 100644 --- a/Common/x64Emitter.cpp +++ b/Common/x64Emitter.cpp @@ -34,27 +34,28 @@ namespace Gen // TODO(ector): Add EAX special casing, for ever so slightly smaller code. struct NormalOpDef { - u8 toRm8, toRm32, fromRm8, fromRm32, imm8, imm32, simm8, ext; + u8 toRm8, toRm32, fromRm8, fromRm32, imm8, imm32, simm8, eaximm8, eaximm32, ext; }; -static const NormalOpDef nops[11] = +// 0xCC is code for invalid combination of immediates +static const NormalOpDef normalops[11] = { - {0x00, 0x01, 0x02, 0x03, 0x80, 0x81, 0x83, 0}, //ADD - {0x10, 0x11, 0x12, 0x13, 0x80, 0x81, 0x83, 2}, //ADC + {0x00, 0x01, 0x02, 0x03, 0x80, 0x81, 0x83, 0x04, 0x05, 0}, //ADD + {0x10, 0x11, 0x12, 0x13, 0x80, 0x81, 0x83, 0x14, 0x15, 2}, //ADC - {0x28, 0x29, 0x2A, 0x2B, 0x80, 0x81, 0x83, 5}, //SUB - {0x18, 0x19, 0x1A, 0x1B, 0x80, 0x81, 0x83, 3}, //SBB + {0x28, 0x29, 0x2A, 0x2B, 0x80, 0x81, 0x83, 0x2C, 0x2D, 5}, //SUB + {0x18, 0x19, 0x1A, 0x1B, 0x80, 0x81, 0x83, 0x1C, 0x1D, 3}, //SBB - {0x20, 0x21, 0x22, 0x23, 0x80, 0x81, 0x83, 4}, //AND - {0x08, 0x09, 0x0A, 0x0B, 0x80, 0x81, 0x83, 1}, //OR + {0x20, 0x21, 0x22, 0x23, 0x80, 0x81, 0x83, 0x24, 0x25, 4}, //AND + {0x08, 0x09, 0x0A, 0x0B, 0x80, 0x81, 0x83, 0x0C, 0x0D, 1}, //OR - {0x30, 0x31, 0x32, 0x33, 0x80, 0x81, 0x83, 6}, //XOR - {0x88, 0x89, 0x8A, 0x8B, 0xC6, 0xC7, 0xCC, 0}, //MOV + {0x30, 0x31, 0x32, 0x33, 0x80, 0x81, 0x83, 0x34, 0x35, 6}, //XOR + {0x88, 0x89, 0x8A, 0x8B, 0xC6, 0xC7, 0xCC, 0xCC, 0xCC, 0}, //MOV - {0x84, 0x85, 0x84, 0x85, 0xF6, 0xF7, 0xCC, 0}, //TEST (to == from) - {0x38, 0x39, 0x3A, 0x3B, 0x80, 0x81, 0x83, 7}, //CMP + {0x84, 0x85, 0x84, 0x85, 0xF6, 0xF7, 0xCC, 0xA8, 0xA9, 0}, //TEST (to == from) + {0x38, 0x39, 0x3A, 0x3B, 0x80, 0x81, 0x83, 0x3C, 0x3D, 7}, //CMP - {0x86, 0x87, 0x86, 0x87, 0xCC, 0xCC, 0xCC, 7}, //XCHG + {0x86, 0x87, 0x86, 0x87, 0xCC, 0xCC, 0xCC, 0xCC, 0xCC, 7}, //XCHG }; enum NormalSSEOps @@ -76,10 +77,16 @@ enum NormalSSEOps sseRSQRT = 0x52, //RSQRT (NO DOUBLE PRECISION!!!) sseMOVAPfromRM = 0x28, //MOVAP from RM sseMOVAPtoRM = 0x29, //MOVAP to RM - sseMOVUPfromRM = 0x10, //MOVUP from RM + sseMOVUPfromRM = 0x10, //MOVUP from RM + sseMOVUPtoRM = 0x11, //MOVUP to RM + sseMOVLPDfromRM= 0x12, + sseMOVLPDtoRM = 0x13, + sseMOVHPDfromRM= 0x16, + sseMOVHPDtoRM = 0x17, + sseMOVHLPS = 0x12, + sseMOVLHPS = 0x16, sseMOVDQfromRM = 0x6F, sseMOVDQtoRM = 0x7F, - sseMOVUPtoRM = 0x11, //MOVUP to RM sseMASKMOVDQU = 0xF7, sseLDDQU = 0xF0, sseSHUF = 0xC6, @@ -133,6 +140,14 @@ const u8 *XEmitter::AlignCodePage() return code; } +// This operation modifies flags; check to see the flags are locked. +// If the flags are locked, we should immediately and loudly fail before +// causing a subtle JIT bug. +void XEmitter::CheckFlags() +{ + _assert_msg_(DYNA_REC, !flags_locked, "Attempt to modify flags while flags locked!"); +} + void XEmitter::WriteModRM(int mod, int reg, int rm) { Write8((u8)((mod << 6) | ((reg & 7) << 3) | (rm & 7))); @@ -148,51 +163,42 @@ void OpArg::WriteRex(XEmitter *emit, int opBits, int bits, int customOp) const if (customOp == -1) customOp = operandReg; #ifdef _M_X64 u8 op = 0x40; + // REX.W (whether operation is a 64-bit operation) if (opBits == 64) op |= 8; + // REX.R (whether ModR/M reg field refers to R8-R15. if (customOp & 8) op |= 4; + // REX.X (whether ModR/M SIB index field refers to R8-R15) if (indexReg & 8) op |= 2; - if (offsetOrBaseReg & 8) op |= 1; //TODO investigate if this is dangerous + // REX.B (whether ModR/M rm or SIB base or opcode reg field refers to R8-R15) + if (offsetOrBaseReg & 8) op |= 1; + // Write REX if wr have REX bits to write, or if the operation accesses + // SIL, DIL, BPL, or SPL. if (op != 0x40 || - (bits == 8 && (offsetOrBaseReg & 0x10c) == 4) || - (opBits == 8 && (customOp & 0x10c) == 4)) { + (scale == SCALE_NONE && bits == 8 && (offsetOrBaseReg & 0x10c) == 4) || + (opBits == 8 && (customOp & 0x10c) == 4)) + { emit->Write8(op); - _dbg_assert_(DYNA_REC, (offsetOrBaseReg & 0x100) == 0 || bits != 8); - _dbg_assert_(DYNA_REC, (customOp & 0x100) == 0 || opBits != 8); - } else { - _dbg_assert_(DYNA_REC, (offsetOrBaseReg & 0x10c) == 0 || - (offsetOrBaseReg & 0x10c) == 0x104 || - bits != 8); - _dbg_assert_(DYNA_REC, (customOp & 0x10c) == 0 || - (customOp & 0x10c) == 0x104 || - opBits != 8); + // Check the operation doesn't access AH, BH, CH, or DH. + _dbg_assert_(DYNA_REC, (offsetOrBaseReg & 0x100) == 0); + _dbg_assert_(DYNA_REC, (customOp & 0x100) == 0); } - #else _dbg_assert_(DYNA_REC, opBits != 64); _dbg_assert_(DYNA_REC, (customOp & 8) == 0 || customOp == -1); _dbg_assert_(DYNA_REC, (indexReg & 8) == 0); _dbg_assert_(DYNA_REC, (offsetOrBaseReg & 8) == 0); _dbg_assert_(DYNA_REC, opBits != 8 || (customOp & 0x10c) != 4 || customOp == -1); - _dbg_assert_(DYNA_REC, bits != 8 || (offsetOrBaseReg & 0x10c) != 4); + _dbg_assert_(DYNA_REC, scale == SCALE_ATREG || bits != 8 || (offsetOrBaseReg & 0x10c) != 4); #endif } -void OpArg::WriteVex(XEmitter* emit, int size, int packed, Gen::X64Reg regOp1, Gen::X64Reg regOp2) const +void OpArg::WriteVex(XEmitter* emit, X64Reg regOp1, X64Reg regOp2, int L, int pp, int mmmmm, int W) const { int R = !(regOp1 & 8); int X = !(indexReg & 8); int B = !(offsetOrBaseReg & 8); - // not so sure about this one... - int W = 0; - - // aka map_select in AMD manuals - // only support VEX opcode map 1 for now (analog to secondary opcode map) - int mmmmm = 1; - int vvvv = (regOp2 == X64Reg::INVALID_REG) ? 0xf : (regOp2 ^ 0xf); - int L = size == 256; - int pp = (packed << 1) | (size == 64); // do we need any VEX fields that only appear in the three-byte form? if (X == 1 && B == 1 && W == 0 && mmmmm == 1) @@ -214,7 +220,7 @@ void OpArg::WriteVex(XEmitter* emit, int size, int packed, Gen::X64Reg regOp1, G void OpArg::WriteRest(XEmitter *emit, int extraBytes, X64Reg _operandReg, bool warn_64bit_offset) const { - if (_operandReg == 0xff) + if (_operandReg == INVALID_REG) _operandReg = (X64Reg)this->operandReg; int mod = 0; int ireg = indexReg; @@ -225,16 +231,17 @@ void OpArg::WriteRest(XEmitter *emit, int extraBytes, X64Reg _operandReg, { // Oh, RIP addressing. _offsetOrBaseReg = 5; - emit->WriteModRM(0, _operandReg&7, 5); + emit->WriteModRM(0, _operandReg, _offsetOrBaseReg); //TODO : add some checks #ifdef _M_X64 u64 ripAddr = (u64)emit->GetCodePtr() + 4 + extraBytes; s64 distance = (s64)offset - (s64)ripAddr; - _assert_msg_(DYNA_REC, (distance < 0x80000000LL - && distance >= -0x80000000LL) || - !warn_64bit_offset, - "WriteRest: op out of range (0x%" PRIx64 " uses 0x%" PRIx64 ")", - ripAddr, offset); + _assert_msg_(DYNA_REC, + (distance < 0x80000000LL && + distance >= -0x80000000LL) || + !warn_64bit_offset, + "WriteRest: op out of range (0x%" PRIx64 " uses 0x%" PRIx64 ")", + ripAddr, offset); s32 offs = (s32)distance; emit->Write32((u32)offs); #else @@ -349,7 +356,6 @@ void OpArg::WriteRest(XEmitter *emit, int extraBytes, X64Reg _operandReg, } } - // W = operand extended width (1 if 64-bit) // R = register# upper bit // X = scale amnt upper bit @@ -381,9 +387,9 @@ void XEmitter::JMP(const u8 *addr, bool force5Bytes) { s64 distance = (s64)(fn - ((u64)code + 5)); - _assert_msg_(DYNA_REC, distance >= -0x80000000LL - && distance < 0x80000000LL, - "Jump target too far away, needs indirect register"); + _assert_msg_(DYNA_REC, + distance >= -0x80000000LL && distance < 0x80000000LL, + "Jump target too far away, needs indirect register"); Write8(0xE9); Write32((u32)(s32)distance); } @@ -419,9 +425,10 @@ void XEmitter::CALLptr(OpArg arg) void XEmitter::CALL(const void *fnptr) { u64 distance = u64(fnptr) - (u64(code) + 5); - _assert_msg_(DYNA_REC, distance < 0x0000000080000000ULL - || distance >= 0xFFFFFFFF80000000ULL, - "CALL out of range (%p calls %p)", code, fnptr); + _assert_msg_(DYNA_REC, + distance < 0x0000000080000000ULL || + distance >= 0xFFFFFFFF80000000ULL, + "CALL out of range (%p calls %p)", code, fnptr); Write8(0xE8); Write32(u32(distance)); } @@ -465,27 +472,25 @@ FixupBranch XEmitter::J_CC(CCFlags conditionCode, bool force5bytes) return branch; } -void XEmitter::J_CC(CCFlags conditionCode, const u8 * addr, bool force5Bytes) +void XEmitter::J_CC(CCFlags conditionCode, const u8* addr, bool force5bytes) { u64 fn = (u64)addr; - if (!force5Bytes) + s64 distance = (s64)(fn - ((u64)code + 2)); + if (distance < -0x80 || distance >= 0x80 || force5bytes) { - s64 distance = (s64)(fn - ((u64)code + 2)); - _assert_msg_(DYNA_REC, distance >= -0x80 && distance < 0x80, "Jump target too far away, needs force5Bytes = true"); - //8 bits will do - Write8(0x70 + conditionCode); - Write8((u8)(s8)distance); - } - else - { - s64 distance = (s64)(fn - ((u64)code + 6)); - _assert_msg_(DYNA_REC, distance >= -0x80000000LL - && distance < 0x80000000LL, - "Jump target too far away, needs indirect register"); + distance = (s64)(fn - ((u64)code + 6)); + _assert_msg_(DYNA_REC, + distance >= -0x80000000LL && distance < 0x80000000LL, + "Jump target too far away, needs indirect register"); Write8(0x0F); Write8(0x80 + conditionCode); Write32((u32)(s32)distance); } + else + { + Write8(0x70 + conditionCode); + Write8((u8)(s8)distance); + } } void XEmitter::SetJumpTarget(const FixupBranch &branch) @@ -534,30 +539,71 @@ void XEmitter::INT3() {Write8(0xCC);} void XEmitter::RET() {Write8(0xC3);} void XEmitter::RET_FAST() {Write8(0xF3); Write8(0xC3);} //two-byte return (rep ret) - recommended by AMD optimization manual for the case of jumping to a ret -void XEmitter::NOP(int count) +// The first sign of decadence: optimized NOPs. +void XEmitter::NOP(size_t size) { - // TODO: look up the fastest nop sleds for various sizes - int i; - switch (count) { - case 1: - Write8(0x90); - break; - case 2: - Write8(0x66); - Write8(0x90); - break; - default: - for (i = 0; i < count; i++) { + _dbg_assert_(DYNA_REC, (int)size > 0); + while (true) + { + switch (size) + { + case 0: + return; + case 1: Write8(0x90); + return; + case 2: + Write8(0x66); Write8(0x90); + return; + case 3: + Write8(0x0F); Write8(0x1F); Write8(0x00); + return; + case 4: + Write8(0x0F); Write8(0x1F); Write8(0x40); Write8(0x00); + return; + case 5: + Write8(0x0F); Write8(0x1F); Write8(0x44); Write8(0x00); + Write8(0x00); + return; + case 6: + Write8(0x66); Write8(0x0F); Write8(0x1F); Write8(0x44); + Write8(0x00); Write8(0x00); + return; + case 7: + Write8(0x0F); Write8(0x1F); Write8(0x80); Write8(0x00); + Write8(0x00); Write8(0x00); Write8(0x00); + return; + case 8: + Write8(0x0F); Write8(0x1F); Write8(0x84); Write8(0x00); + Write8(0x00); Write8(0x00); Write8(0x00); Write8(0x00); + return; + case 9: + Write8(0x66); Write8(0x0F); Write8(0x1F); Write8(0x84); + Write8(0x00); Write8(0x00); Write8(0x00); Write8(0x00); + Write8(0x00); + return; + case 10: + Write8(0x66); Write8(0x66); Write8(0x0F); Write8(0x1F); + Write8(0x84); Write8(0x00); Write8(0x00); Write8(0x00); + Write8(0x00); Write8(0x00); + return; + default: + // Even though x86 instructions are allowed to be up to 15 bytes long, + // AMD advises against using NOPs longer than 11 bytes because they + // carry a performance penalty on CPUs older than AMD family 16h. + Write8(0x66); Write8(0x66); Write8(0x66); Write8(0x0F); + Write8(0x1F); Write8(0x84); Write8(0x00); Write8(0x00); + Write8(0x00); Write8(0x00); Write8(0x00); + size -= 11; + continue; } - break; } } void XEmitter::PAUSE() {Write8(0xF3); NOP();} //use in tight spinloops for energy saving on some cpu -void XEmitter::CLC() {Write8(0xF8);} //clear carry -void XEmitter::CMC() {Write8(0xF5);} //flip carry -void XEmitter::STC() {Write8(0xF9);} //set carry +void XEmitter::CLC() {CheckFlags(); Write8(0xF8);} //clear carry +void XEmitter::CMC() {CheckFlags(); Write8(0xF5);} //flip carry +void XEmitter::STC() {CheckFlags(); Write8(0xF9);} //set carry //TODO: xchg ah, al ??? void XEmitter::XCHG_AHAL() @@ -569,10 +615,10 @@ void XEmitter::XCHG_AHAL() //These two can not be executed on early Intel 64-bit CPU:s, only on AMD! void XEmitter::LAHF() {Write8(0x9F);} -void XEmitter::SAHF() {Write8(0x9E);} +void XEmitter::SAHF() {CheckFlags(); Write8(0x9E);} void XEmitter::PUSHF() {Write8(0x9C);} -void XEmitter::POPF() {Write8(0x9D);} +void XEmitter::POPF() {CheckFlags(); Write8(0x9D);} void XEmitter::LFENCE() {Write8(0x0F); Write8(0xAE); Write8(0xE8);} void XEmitter::MFENCE() {Write8(0x0F); Write8(0xAE); Write8(0xF0);} @@ -580,14 +626,16 @@ void XEmitter::SFENCE() {Write8(0x0F); Write8(0xAE); Write8(0xF8);} void XEmitter::WriteSimple1Byte(int bits, u8 byte, X64Reg reg) { - if (bits == 16) {Write8(0x66);} + if (bits == 16) + Write8(0x66); Rex(bits == 64, 0, 0, (int)reg >> 3); Write8(byte + ((int)reg & 7)); } void XEmitter::WriteSimple2Byte(int bits, u8 byte1, u8 byte2, X64Reg reg) { - if (bits == 16) {Write8(0x66);} + if (bits == 16) + Write8(0x66); Rex(bits==64, 0, 0, (int)reg >> 3); Write8(byte1); Write8(byte2 + ((int)reg & 7)); @@ -595,14 +643,16 @@ void XEmitter::WriteSimple2Byte(int bits, u8 byte1, u8 byte2, X64Reg reg) void XEmitter::CWD(int bits) { - if (bits == 16) {Write8(0x66);} + if (bits == 16) + Write8(0x66); Rex(bits == 64, 0, 0, 0); Write8(0x99); } void XEmitter::CBW(int bits) { - if (bits == 8) {Write8(0x66);} + if (bits == 8) + Write8(0x66); Rex(bits == 32, 0, 0, 0); Write8(0x98); } @@ -655,7 +705,7 @@ void XEmitter::POP(int /*bits*/, const OpArg ®) if (reg.IsSimpleReg()) POP(reg.GetSimpleReg()); else - INT3(); + _assert_msg_(DYNA_REC, 0, "POP - Unsupported encoding"); } void XEmitter::BSWAP(int bits, X64Reg reg) @@ -688,7 +738,7 @@ void XEmitter::UD2() void XEmitter::PREFETCH(PrefetchLevel level, OpArg arg) { - if (arg.IsImm()) _assert_msg_(DYNA_REC, 0, "PREFETCH - Imm argument");; + _assert_msg_(DYNA_REC, !arg.IsImm(), "PREFETCH - Imm argument"); arg.operandReg = (u8)level; arg.WriteRex(this, 0, 0); Write8(0x0F); @@ -698,9 +748,9 @@ void XEmitter::PREFETCH(PrefetchLevel level, OpArg arg) void XEmitter::SETcc(CCFlags flag, OpArg dest) { - if (dest.IsImm()) _assert_msg_(DYNA_REC, 0, "SETcc - Imm argument"); + _assert_msg_(DYNA_REC, !dest.IsImm(), "SETcc - Imm argument"); dest.operandReg = 0; - dest.WriteRex(this, 0, 0); + dest.WriteRex(this, 0, 8); Write8(0x0F); Write8(0x90 + (u8)flag); dest.WriteRest(this); @@ -708,7 +758,10 @@ void XEmitter::SETcc(CCFlags flag, OpArg dest) void XEmitter::CMOVcc(int bits, X64Reg dest, OpArg src, CCFlags flag) { - if (src.IsImm()) _assert_msg_(DYNA_REC, 0, "CMOVcc - Imm argument"); + _assert_msg_(DYNA_REC, !src.IsImm(), "CMOVcc - Imm argument"); + _assert_msg_(DYNA_REC, bits != 8, "CMOVcc - 8 bits unsupported"); + if (bits == 16) + Write8(0x66); src.operandReg = dest; src.WriteRex(this, bits, bits); Write8(0x0F); @@ -718,10 +771,12 @@ void XEmitter::CMOVcc(int bits, X64Reg dest, OpArg src, CCFlags flag) void XEmitter::WriteMulDivType(int bits, OpArg src, int ext) { - if (src.IsImm()) _assert_msg_(DYNA_REC, 0, "WriteMulDivType - Imm argument"); + _assert_msg_(DYNA_REC, !src.IsImm(), "WriteMulDivType - Imm argument"); + CheckFlags(); src.operandReg = ext; - if (bits == 16) Write8(0x66); - src.WriteRex(this, bits, bits); + if (bits == 16) + Write8(0x66); + src.WriteRex(this, bits, bits, 0); if (bits == 8) { Write8(0xF6); @@ -740,11 +795,15 @@ void XEmitter::IDIV(int bits, OpArg src) {WriteMulDivType(bits, src, 7);} void XEmitter::NEG(int bits, OpArg src) {WriteMulDivType(bits, src, 3);} void XEmitter::NOT(int bits, OpArg src) {WriteMulDivType(bits, src, 2);} -void XEmitter::WriteBitSearchType(int bits, X64Reg dest, OpArg src, u8 byte2) +void XEmitter::WriteBitSearchType(int bits, X64Reg dest, OpArg src, u8 byte2, bool rep) { - if (src.IsImm()) _assert_msg_(DYNA_REC, 0, "WriteBitSearchType - Imm argument"); + _assert_msg_(DYNA_REC, !src.IsImm(), "WriteBitSearchType - Imm argument"); + CheckFlags(); src.operandReg = (u8)dest; - if (bits == 16) Write8(0x66); + if (bits == 16) + Write8(0x66); + if (rep) + Write8(0xF3); src.WriteRex(this, bits, bits); Write8(0x0F); Write8(byte2); @@ -753,22 +812,40 @@ void XEmitter::WriteBitSearchType(int bits, X64Reg dest, OpArg src, u8 byte2) void XEmitter::MOVNTI(int bits, OpArg dest, X64Reg src) { - if (bits <= 16) _assert_msg_(DYNA_REC, 0, "MOVNTI - bits<=16"); + if (bits <= 16) + _assert_msg_(DYNA_REC, 0, "MOVNTI - bits<=16"); WriteBitSearchType(bits, src, dest, 0xC3); } void XEmitter::BSF(int bits, X64Reg dest, OpArg src) {WriteBitSearchType(bits,dest,src,0xBC);} //bottom bit to top bit void XEmitter::BSR(int bits, X64Reg dest, OpArg src) {WriteBitSearchType(bits,dest,src,0xBD);} //top bit to bottom bit +void XEmitter::TZCNT(int bits, X64Reg dest, OpArg src) +{ + CheckFlags(); + if (!cpu_info.bBMI1) + PanicAlert("Trying to use BMI1 on a system that doesn't support it. Bad programmer."); + WriteBitSearchType(bits, dest, src, 0xBC, true); +} +void XEmitter::LZCNT(int bits, X64Reg dest, OpArg src) +{ + CheckFlags(); + if (!cpu_info.bLZCNT) + PanicAlert("Trying to use LZCNT on a system that doesn't support it. Bad programmer."); + WriteBitSearchType(bits, dest, src, 0xBD, true); +} + void XEmitter::MOVSX(int dbits, int sbits, X64Reg dest, OpArg src) { - if (src.IsImm()) _assert_msg_(DYNA_REC, 0, "MOVSX - Imm argument"); - if (dbits == sbits) { + _assert_msg_(DYNA_REC, !src.IsImm(), "MOVSX - Imm argument"); + if (dbits == sbits) + { MOV(dbits, R(dest), src); return; } src.operandReg = (u8)dest; - if (dbits == 16) Write8(0x66); + if (dbits == 16) + Write8(0x66); src.WriteRex(this, dbits, sbits); if (sbits == 8) { @@ -793,13 +870,15 @@ void XEmitter::MOVSX(int dbits, int sbits, X64Reg dest, OpArg src) void XEmitter::MOVZX(int dbits, int sbits, X64Reg dest, OpArg src) { - if (src.IsImm()) _assert_msg_(DYNA_REC, 0, "MOVZX - Imm argument"); - if (dbits == sbits) { + _assert_msg_(DYNA_REC, !src.IsImm(), "MOVZX - Imm argument"); + if (dbits == sbits) + { MOV(dbits, R(dest), src); return; } src.operandReg = (u8)dest; - if (dbits == 16) Write8(0x66); + if (dbits == 16) + Write8(0x66); //the 32bit result is automatically zero extended to 64bit src.WriteRex(this, dbits == 64 ? 32 : dbits, sbits); if (sbits == 8) @@ -818,25 +897,59 @@ void XEmitter::MOVZX(int dbits, int sbits, X64Reg dest, OpArg src) } else { - Crash(); + _assert_msg_(DYNA_REC, 0, "MOVZX - Invalid size"); } src.WriteRest(this); } +void XEmitter::MOVBE(int bits, const OpArg& dest, const OpArg& src) +{ + _assert_msg_(DYNA_REC, cpu_info.bMOVBE, "Generating MOVBE on a system that does not support it."); + if (bits == 8) + { + MOV(bits, dest, src); + return; + } + + if (bits == 16) + Write8(0x66); + + if (dest.IsSimpleReg()) + { + _assert_msg_(DYNA_REC, !src.IsSimpleReg() && !src.IsImm(), "MOVBE: Loading from !mem"); + src.WriteRex(this, bits, bits, dest.GetSimpleReg()); + Write8(0x0F); Write8(0x38); Write8(0xF0); + src.WriteRest(this, 0, dest.GetSimpleReg()); + } + else if (src.IsSimpleReg()) + { + _assert_msg_(DYNA_REC, !dest.IsSimpleReg() && !dest.IsImm(), "MOVBE: Storing to !mem"); + dest.WriteRex(this, bits, bits, src.GetSimpleReg()); + Write8(0x0F); Write8(0x38); Write8(0xF1); + dest.WriteRest(this, 0, src.GetSimpleReg()); + } + else + { + _assert_msg_(DYNA_REC, 0, "MOVBE: Not loading or storing to mem"); + } +} + void XEmitter::LEA(int bits, X64Reg dest, OpArg src) { - if (src.IsImm()) _assert_msg_(DYNA_REC, 0, "LEA - Imm argument"); + _assert_msg_(DYNA_REC, !src.IsImm(), "LEA - Imm argument"); src.operandReg = (u8)dest; - if (bits == 16) Write8(0x66); //TODO: performance warning + if (bits == 16) + Write8(0x66); //TODO: performance warning src.WriteRex(this, bits, bits); Write8(0x8D); - src.WriteRest(this, 0, (X64Reg)0xFF, bits == 64); + src.WriteRest(this, 0, INVALID_REG, bits == 64); } //shift can be either imm8 or cl void XEmitter::WriteShift(int bits, OpArg dest, OpArg &shift, int ext) { + CheckFlags(); bool writeImm = false; if (dest.IsImm()) { @@ -847,7 +960,8 @@ void XEmitter::WriteShift(int bits, OpArg dest, OpArg &shift, int ext) _assert_msg_(DYNA_REC, 0, "WriteShift - illegal argument"); } dest.operandReg = ext; - if (bits == 16) Write8(0x66); + if (bits == 16) + Write8(0x66); dest.WriteRex(this, bits, bits, 0); if (shift.GetImmBits() == 8) { @@ -885,6 +999,7 @@ void XEmitter::SAR(int bits, OpArg dest, OpArg shift) {WriteShift(bits, dest, sh // index can be either imm8 or register, don't use memory destination because it's slow void XEmitter::WriteBitTest(int bits, OpArg &dest, OpArg &index, int ext) { + CheckFlags(); if (dest.IsImm()) { _assert_msg_(DYNA_REC, 0, "WriteBitTest - can't test imms"); @@ -893,7 +1008,8 @@ void XEmitter::WriteBitTest(int bits, OpArg &dest, OpArg &index, int ext) { _assert_msg_(DYNA_REC, 0, "WriteBitTest - illegal argument"); } - if (bits == 16) Write8(0x66); + if (bits == 16) + Write8(0x66); if (index.IsImm()) { dest.WriteRex(this, bits, bits); @@ -918,6 +1034,7 @@ void XEmitter::BTC(int bits, OpArg dest, OpArg index) {WriteBitTest(bits, dest, //shift can be either imm8 or cl void XEmitter::SHRD(int bits, OpArg dest, OpArg src, OpArg shift) { + CheckFlags(); if (dest.IsImm()) { _assert_msg_(DYNA_REC, 0, "SHRD - can't use imms as destination"); @@ -930,7 +1047,8 @@ void XEmitter::SHRD(int bits, OpArg dest, OpArg src, OpArg shift) { _assert_msg_(DYNA_REC, 0, "SHRD - illegal shift"); } - if (bits == 16) Write8(0x66); + if (bits == 16) + Write8(0x66); X64Reg operand = src.GetSimpleReg(); dest.WriteRex(this, bits, bits, operand); if (shift.GetImmBits() == 8) @@ -948,6 +1066,7 @@ void XEmitter::SHRD(int bits, OpArg dest, OpArg src, OpArg shift) void XEmitter::SHLD(int bits, OpArg dest, OpArg src, OpArg shift) { + CheckFlags(); if (dest.IsImm()) { _assert_msg_(DYNA_REC, 0, "SHLD - can't use imms as destination"); @@ -960,7 +1079,8 @@ void XEmitter::SHLD(int bits, OpArg dest, OpArg src, OpArg shift) { _assert_msg_(DYNA_REC, 0, "SHLD - illegal shift"); } - if (bits == 16) Write8(0x66); + if (bits == 16) + Write8(0x66); X64Reg operand = src.GetSimpleReg(); dest.WriteRex(this, bits, bits, operand); if (shift.GetImmBits() == 8) @@ -990,7 +1110,7 @@ void OpArg::WriteSingleByteOp(XEmitter *emit, u8 op, X64Reg _operandReg, int bit //operand can either be immediate or register void OpArg::WriteNormalOp(XEmitter *emit, bool toRM, NormalOp op, const OpArg &operand, int bits) const { - X64Reg _operandReg = (X64Reg)this->operandReg; + X64Reg _operandReg; if (IsImm()) { _assert_msg_(DYNA_REC, 0, "WriteNormalOp - Imm argument, wrong order"); @@ -1003,7 +1123,6 @@ void OpArg::WriteNormalOp(XEmitter *emit, bool toRM, NormalOp op, const OpArg &o if (operand.IsImm()) { - _operandReg = (X64Reg)0; WriteRex(emit, bits, bits); if (!toRM) @@ -1013,26 +1132,81 @@ void OpArg::WriteNormalOp(XEmitter *emit, bool toRM, NormalOp op, const OpArg &o if (operand.scale == SCALE_IMM8 && bits == 8) { - emit->Write8(nops[op].imm8); + // op al, imm8 + if (!scale && offsetOrBaseReg == AL && normalops[op].eaximm8 != 0xCC) + { + emit->Write8(normalops[op].eaximm8); + emit->Write8((u8)operand.offset); + return; + } + // mov reg, imm8 + if (!scale && op == nrmMOV) + { + emit->Write8(0xB0 + (offsetOrBaseReg & 7)); + emit->Write8((u8)operand.offset); + return; + } + // op r/m8, imm8 + emit->Write8(normalops[op].imm8); immToWrite = 8; } else if ((operand.scale == SCALE_IMM16 && bits == 16) || (operand.scale == SCALE_IMM32 && bits == 32) || (operand.scale == SCALE_IMM32 && bits == 64)) { - emit->Write8(nops[op].imm32); - immToWrite = bits == 16 ? 16 : 32; + // Try to save immediate size if we can, but first check to see + // if the instruction supports simm8. + // op r/m, imm8 + if (normalops[op].simm8 != 0xCC && + ((operand.scale == SCALE_IMM16 && (s16)operand.offset == (s8)operand.offset) || + (operand.scale == SCALE_IMM32 && (s32)operand.offset == (s8)operand.offset))) + { + emit->Write8(normalops[op].simm8); + immToWrite = 8; + } + else + { + // mov reg, imm + if (!scale && op == nrmMOV && bits != 64) + { + emit->Write8(0xB8 + (offsetOrBaseReg & 7)); + if (bits == 16) + emit->Write16((u16)operand.offset); + else + emit->Write32((u32)operand.offset); + return; + } + // op eax, imm + if (!scale && offsetOrBaseReg == EAX && normalops[op].eaximm32 != 0xCC) + { + emit->Write8(normalops[op].eaximm32); + if (bits == 16) + emit->Write16((u16)operand.offset); + else + emit->Write32((u32)operand.offset); + return; + } + // op r/m, imm + emit->Write8(normalops[op].imm32); + immToWrite = bits == 16 ? 16 : 32; + } } else if ((operand.scale == SCALE_IMM8 && bits == 16) || (operand.scale == SCALE_IMM8 && bits == 32) || (operand.scale == SCALE_IMM8 && bits == 64)) { - emit->Write8(nops[op].simm8); + // op r/m, imm8 + emit->Write8(normalops[op].simm8); immToWrite = 8; } else if (operand.scale == SCALE_IMM64 && bits == 64) { - if (op == nrmMOV) + if (scale) + { + _assert_msg_(DYNA_REC, 0, "WriteNormalOp - MOV with 64-bit imm requres register destination"); + } + // mov reg64, imm64 + else if (op == nrmMOV) { emit->Write8(0xB8 + (offsetOrBaseReg & 7)); emit->Write64((u64)operand.offset); @@ -1044,25 +1218,24 @@ void OpArg::WriteNormalOp(XEmitter *emit, bool toRM, NormalOp op, const OpArg &o { _assert_msg_(DYNA_REC, 0, "WriteNormalOp - Unhandled case"); } - _operandReg = (X64Reg)nops[op].ext; //pass extension in REG of ModRM + _operandReg = (X64Reg)normalops[op].ext; //pass extension in REG of ModRM } else { _operandReg = (X64Reg)operand.offsetOrBaseReg; WriteRex(emit, bits, bits, _operandReg); - // mem/reg or reg/reg op + // op r/m, reg if (toRM) { - emit->Write8(bits == 8 ? nops[op].toRm8 : nops[op].toRm32); - // _assert_msg_(DYNA_REC, code[-1] != 0xCC, "ARGH4"); + emit->Write8(bits == 8 ? normalops[op].toRm8 : normalops[op].toRm32); } + // op reg, r/m else { - emit->Write8(bits == 8 ? nops[op].fromRm8 : nops[op].fromRm32); - // _assert_msg_(DYNA_REC, code[-1] != 0xCC, "ARGH5"); + emit->Write8(bits == 8 ? normalops[op].fromRm8 : normalops[op].fromRm32); } } - WriteRest(emit, immToWrite>>3, _operandReg); + WriteRest(emit, immToWrite >> 3, _operandReg); switch (immToWrite) { case 0: @@ -1101,40 +1274,44 @@ void XEmitter::WriteNormalOp(XEmitter *emit, int bits, NormalOp op, const OpArg } else { + _assert_msg_(DYNA_REC, a2.IsSimpleReg() || a2.IsImm(), "WriteNormalOp - a1 and a2 cannot both be memory"); a1.WriteNormalOp(emit, true, op, a2, bits); } } } -void XEmitter::ADD (int bits, const OpArg &a1, const OpArg &a2) {WriteNormalOp(this, bits, nrmADD, a1, a2);} -void XEmitter::ADC (int bits, const OpArg &a1, const OpArg &a2) {WriteNormalOp(this, bits, nrmADC, a1, a2);} -void XEmitter::SUB (int bits, const OpArg &a1, const OpArg &a2) {WriteNormalOp(this, bits, nrmSUB, a1, a2);} -void XEmitter::SBB (int bits, const OpArg &a1, const OpArg &a2) {WriteNormalOp(this, bits, nrmSBB, a1, a2);} -void XEmitter::AND (int bits, const OpArg &a1, const OpArg &a2) {WriteNormalOp(this, bits, nrmAND, a1, a2);} -void XEmitter::OR (int bits, const OpArg &a1, const OpArg &a2) {WriteNormalOp(this, bits, nrmOR , a1, a2);} -void XEmitter::XOR (int bits, const OpArg &a1, const OpArg &a2) {WriteNormalOp(this, bits, nrmXOR, a1, a2);} +void XEmitter::ADD (int bits, const OpArg &a1, const OpArg &a2) {CheckFlags(); WriteNormalOp(this, bits, nrmADD, a1, a2);} +void XEmitter::ADC (int bits, const OpArg &a1, const OpArg &a2) {CheckFlags(); WriteNormalOp(this, bits, nrmADC, a1, a2);} +void XEmitter::SUB (int bits, const OpArg &a1, const OpArg &a2) {CheckFlags(); WriteNormalOp(this, bits, nrmSUB, a1, a2);} +void XEmitter::SBB (int bits, const OpArg &a1, const OpArg &a2) {CheckFlags(); WriteNormalOp(this, bits, nrmSBB, a1, a2);} +void XEmitter::AND (int bits, const OpArg &a1, const OpArg &a2) {CheckFlags(); WriteNormalOp(this, bits, nrmAND, a1, a2);} +void XEmitter::OR (int bits, const OpArg &a1, const OpArg &a2) {CheckFlags(); WriteNormalOp(this, bits, nrmOR , a1, a2);} +void XEmitter::XOR (int bits, const OpArg &a1, const OpArg &a2) {CheckFlags(); WriteNormalOp(this, bits, nrmXOR, a1, a2);} void XEmitter::MOV (int bits, const OpArg &a1, const OpArg &a2) { -#ifdef _DEBUG - _assert_msg_(DYNA_REC, !a1.IsSimpleReg() || !a2.IsSimpleReg() || a1.GetSimpleReg() != a2.GetSimpleReg(), "Redundant MOV @ %p - bug in DYNA_REC?", - code); -#endif + if (a1.IsSimpleReg() && a2.IsSimpleReg() && a1.GetSimpleReg() == a2.GetSimpleReg()) + ERROR_LOG(DYNA_REC, "Redundant MOV @ %p - bug in JIT?", code); WriteNormalOp(this, bits, nrmMOV, a1, a2); } -void XEmitter::TEST(int bits, const OpArg &a1, const OpArg &a2) {WriteNormalOp(this, bits, nrmTEST, a1, a2);} -void XEmitter::CMP (int bits, const OpArg &a1, const OpArg &a2) {WriteNormalOp(this, bits, nrmCMP, a1, a2);} +void XEmitter::TEST(int bits, const OpArg &a1, const OpArg &a2) {CheckFlags(); WriteNormalOp(this, bits, nrmTEST, a1, a2);} +void XEmitter::CMP (int bits, const OpArg &a1, const OpArg &a2) {CheckFlags(); WriteNormalOp(this, bits, nrmCMP, a1, a2);} void XEmitter::XCHG(int bits, const OpArg &a1, const OpArg &a2) {WriteNormalOp(this, bits, nrmXCHG, a1, a2);} void XEmitter::IMUL(int bits, X64Reg regOp, OpArg a1, OpArg a2) { - if (bits == 8) { + CheckFlags(); + if (bits == 8) + { _assert_msg_(DYNA_REC, 0, "IMUL - illegal bit size!"); return; } - if (a1.IsImm()) { + + if (a1.IsImm()) + { _assert_msg_(DYNA_REC, 0, "IMUL - second arg cannot be imm!"); return; } + if (!a2.IsImm()) { _assert_msg_(DYNA_REC, 0, "IMUL - third arg must be imm!"); @@ -1145,20 +1322,29 @@ void XEmitter::IMUL(int bits, X64Reg regOp, OpArg a1, OpArg a2) Write8(0x66); a1.WriteRex(this, bits, bits, regOp); - if (a2.GetImmBits() == 8) { + if (a2.GetImmBits() == 8 || + (a2.GetImmBits() == 16 && (s8)a2.offset == (s16)a2.offset) || + (a2.GetImmBits() == 32 && (s8)a2.offset == (s32)a2.offset)) + { Write8(0x6B); a1.WriteRest(this, 1, regOp); Write8((u8)a2.offset); - } else { + } + else + { Write8(0x69); - if (a2.GetImmBits() == 16 && bits == 16) { + if (a2.GetImmBits() == 16 && bits == 16) + { a1.WriteRest(this, 2, regOp); Write16((u16)a2.offset); - } else if (a2.GetImmBits() == 32 && - (bits == 32 || bits == 64)) { - a1.WriteRest(this, 4, regOp); - Write32((u32)a2.offset); - } else { + } + else if (a2.GetImmBits() == 32 && (bits == 32 || bits == 64)) + { + a1.WriteRest(this, 4, regOp); + Write32((u32)a2.offset); + } + else + { _assert_msg_(DYNA_REC, 0, "IMUL - unhandled case!"); } } @@ -1166,10 +1352,13 @@ void XEmitter::IMUL(int bits, X64Reg regOp, OpArg a1, OpArg a2) void XEmitter::IMUL(int bits, X64Reg regOp, OpArg a) { - if (bits == 8) { + CheckFlags(); + if (bits == 8) + { _assert_msg_(DYNA_REC, 0, "IMUL - illegal bit size!"); return; } + if (a.IsImm()) { IMUL(bits, regOp, R(regOp), a) ; @@ -1185,49 +1374,92 @@ void XEmitter::IMUL(int bits, X64Reg regOp, OpArg a) } -void XEmitter::WriteSSEOp(int size, u8 sseOp, bool packed, X64Reg regOp, OpArg arg, int extrabytes) +void XEmitter::WriteSSEOp(u8 opPrefix, u16 op, X64Reg regOp, OpArg arg, int extrabytes) { - if (size == 64 && packed) - Write8(0x66); //this time, override goes upwards - if (!packed) - Write8(size == 64 ? 0xF2 : 0xF3); + if (opPrefix) + Write8(opPrefix); arg.operandReg = regOp; arg.WriteRex(this, 0, 0); Write8(0x0F); - Write8(sseOp); + if (op > 0xFF) + Write8((op >> 8) & 0xFF); + Write8(op & 0xFF); arg.WriteRest(this, extrabytes); } -void XEmitter::WriteSSEOp2(int size, u8 sseOp, bool packed, X64Reg regOp, OpArg arg, int extrabytes) +void XEmitter::WriteAVXOp(u8 opPrefix, u16 op, X64Reg regOp, OpArg arg, int extrabytes) { - if (size == 64 && packed) - Write8(0x66); //this time, override goes upwards - if (!packed) - Write8(size == 64 ? 0xF2 : 0xF3); - arg.operandReg = regOp; - arg.WriteRex(this, 0, 0); - Write8(0x0F); - Write8(0x38); - Write8(sseOp); - arg.WriteRest(this, extrabytes); + WriteAVXOp(opPrefix, op, regOp, INVALID_REG, arg, extrabytes); } -void XEmitter::WriteAVXOp(int size, u8 sseOp, bool packed, X64Reg regOp, OpArg arg, int extrabytes) +static int GetVEXmmmmm(u16 op) { - WriteAVXOp(size, sseOp, packed, regOp, X64Reg::INVALID_REG, arg, extrabytes); + // Currently, only 0x38 and 0x3A are used as secondary escape byte. + if ((op >> 8) == 0x3A) + return 3; + else if ((op >> 8) == 0x38) + return 2; + else + return 1; } -void XEmitter::WriteAVXOp(int size, u8 sseOp, bool packed, X64Reg regOp1, X64Reg regOp2, OpArg arg, int extrabytes) +static int GetVEXpp(u8 opPrefix) { - arg.WriteVex(this, size, packed, regOp1, regOp2); - Write8(sseOp); + if (opPrefix == 0x66) + return 1; + else if (opPrefix == 0xF3) + return 2; + else if (opPrefix == 0xF2) + return 3; + else + return 0; +} + +void XEmitter::WriteAVXOp(u8 opPrefix, u16 op, X64Reg regOp1, X64Reg regOp2, OpArg arg, int extrabytes) +{ + if (!cpu_info.bAVX) + PanicAlert("Trying to use AVX on a system that doesn't support it. Bad programmer."); + int mmmmm = GetVEXmmmmm(op); + int pp = GetVEXpp(opPrefix); + // FIXME: we currently don't support 256-bit instructions, and "size" is not the vector size here + arg.WriteVex(this, regOp1, regOp2, 0, pp, mmmmm); + Write8(op & 0xFF); arg.WriteRest(this, extrabytes, regOp1); } -void XEmitter::MOVD_xmm(X64Reg dest, const OpArg &arg) {WriteSSEOp(64, 0x6E, true, dest, arg, 0);} -void XEmitter::MOVD_xmm(const OpArg &arg, X64Reg src) {WriteSSEOp(64, 0x7E, true, src, arg, 0);} +// Like the above, but more general; covers GPR-based VEX operations, like BMI1/2 +void XEmitter::WriteVEXOp(int size, u8 opPrefix, u16 op, X64Reg regOp1, X64Reg regOp2, OpArg arg, int extrabytes) +{ + if (size != 32 && size != 64) + PanicAlert("VEX GPR instructions only support 32-bit and 64-bit modes!"); + int mmmmm = GetVEXmmmmm(op); + int pp = GetVEXpp(opPrefix); + arg.WriteVex(this, regOp1, regOp2, 0, pp, mmmmm, size == 64); + Write8(op & 0xFF); + arg.WriteRest(this, extrabytes, regOp1); +} -void XEmitter::MOVQ_xmm(X64Reg dest, OpArg arg) { +void XEmitter::WriteBMI1Op(int size, u8 opPrefix, u16 op, X64Reg regOp1, X64Reg regOp2, OpArg arg, int extrabytes) +{ + CheckFlags(); + if (!cpu_info.bBMI1) + PanicAlert("Trying to use BMI1 on a system that doesn't support it. Bad programmer."); + WriteVEXOp(size, opPrefix, op, regOp1, regOp2, arg, extrabytes); +} + +void XEmitter::WriteBMI2Op(int size, u8 opPrefix, u16 op, X64Reg regOp1, X64Reg regOp2, OpArg arg, int extrabytes) +{ + CheckFlags(); + if (!cpu_info.bBMI2) + PanicAlert("Trying to use BMI2 on a system that doesn't support it. Bad programmer."); + WriteVEXOp(size, opPrefix, op, regOp1, regOp2, arg, extrabytes); +} + +void XEmitter::MOVD_xmm(X64Reg dest, const OpArg &arg) {WriteSSEOp(0x66, 0x6E, dest, arg, 0);} +void XEmitter::MOVD_xmm(const OpArg &arg, X64Reg src) {WriteSSEOp(0x66, 0x7E, src, arg, 0);} + +void XEmitter::MOVQ_xmm(X64Reg dest, OpArg arg) +{ #ifdef _M_X64 // Alternate encoding // This does not display correctly in MSVC's debugger, it thinks it's a MOVD @@ -1246,10 +1478,9 @@ void XEmitter::MOVQ_xmm(X64Reg dest, OpArg arg) { #endif } -void XEmitter::MOVQ_xmm(OpArg arg, X64Reg src) { - if (arg.IsSimpleReg()) - PanicAlert("Emitter: MOVQ_xmm doesn't support single registers as destination"); - if (src > 7) +void XEmitter::MOVQ_xmm(OpArg arg, X64Reg src) +{ + if (src > 7 || arg.IsSimpleReg()) { // Alternate encoding // This does not display correctly in MSVC's debugger, it thinks it's a MOVD @@ -1259,7 +1490,9 @@ void XEmitter::MOVQ_xmm(OpArg arg, X64Reg src) { Write8(0x0f); Write8(0x7E); arg.WriteRest(this, 0); - } else { + } + else + { arg.operandReg = src; arg.WriteRex(this, 0, 0); Write8(0x66); @@ -1284,119 +1517,128 @@ void XEmitter::WriteMXCSR(OpArg arg, int ext) void XEmitter::STMXCSR(OpArg memloc) {WriteMXCSR(memloc, 3);} void XEmitter::LDMXCSR(OpArg memloc) {WriteMXCSR(memloc, 2);} -void XEmitter::MOVNTDQ(OpArg arg, X64Reg regOp) {WriteSSEOp(64, sseMOVNTDQ, true, regOp, arg);} -void XEmitter::MOVNTPS(OpArg arg, X64Reg regOp) {WriteSSEOp(32, sseMOVNTP, true, regOp, arg);} -void XEmitter::MOVNTPD(OpArg arg, X64Reg regOp) {WriteSSEOp(64, sseMOVNTP, true, regOp, arg);} +void XEmitter::MOVNTDQ(OpArg arg, X64Reg regOp) {WriteSSEOp(0x66, sseMOVNTDQ, regOp, arg);} +void XEmitter::MOVNTPS(OpArg arg, X64Reg regOp) {WriteSSEOp(0x00, sseMOVNTP, regOp, arg);} +void XEmitter::MOVNTPD(OpArg arg, X64Reg regOp) {WriteSSEOp(0x66, sseMOVNTP, regOp, arg);} -void XEmitter::ADDSS(X64Reg regOp, OpArg arg) {WriteSSEOp(32, sseADD, false, regOp, arg);} -void XEmitter::ADDSD(X64Reg regOp, OpArg arg) {WriteSSEOp(64, sseADD, false, regOp, arg);} -void XEmitter::SUBSS(X64Reg regOp, OpArg arg) {WriteSSEOp(32, sseSUB, false, regOp, arg);} -void XEmitter::SUBSD(X64Reg regOp, OpArg arg) {WriteSSEOp(64, sseSUB, false, regOp, arg);} -void XEmitter::CMPSS(X64Reg regOp, OpArg arg, u8 compare) {WriteSSEOp(32, sseCMP, false, regOp, arg,1); Write8(compare);} -void XEmitter::CMPSD(X64Reg regOp, OpArg arg, u8 compare) {WriteSSEOp(64, sseCMP, false, regOp, arg,1); Write8(compare);} -void XEmitter::MULSS(X64Reg regOp, OpArg arg) {WriteSSEOp(32, sseMUL, false, regOp, arg);} -void XEmitter::MULSD(X64Reg regOp, OpArg arg) {WriteSSEOp(64, sseMUL, false, regOp, arg);} -void XEmitter::DIVSS(X64Reg regOp, OpArg arg) {WriteSSEOp(32, sseDIV, false, regOp, arg);} -void XEmitter::DIVSD(X64Reg regOp, OpArg arg) {WriteSSEOp(64, sseDIV, false, regOp, arg);} -void XEmitter::MINSS(X64Reg regOp, OpArg arg) {WriteSSEOp(32, sseMIN, false, regOp, arg);} -void XEmitter::MINSD(X64Reg regOp, OpArg arg) {WriteSSEOp(64, sseMIN, false, regOp, arg);} -void XEmitter::MAXSS(X64Reg regOp, OpArg arg) {WriteSSEOp(32, sseMAX, false, regOp, arg);} -void XEmitter::MAXSD(X64Reg regOp, OpArg arg) {WriteSSEOp(64, sseMAX, false, regOp, arg);} -void XEmitter::SQRTSS(X64Reg regOp, OpArg arg) {WriteSSEOp(32, sseSQRT, false, regOp, arg);} -void XEmitter::SQRTSD(X64Reg regOp, OpArg arg) {WriteSSEOp(64, sseSQRT, false, regOp, arg);} -void XEmitter::RSQRTSS(X64Reg regOp, OpArg arg) {WriteSSEOp(32, sseRSQRT, false, regOp, arg);} +void XEmitter::ADDSS(X64Reg regOp, OpArg arg) {WriteSSEOp(0xF3, sseADD, regOp, arg);} +void XEmitter::ADDSD(X64Reg regOp, OpArg arg) {WriteSSEOp(0xF2, sseADD, regOp, arg);} +void XEmitter::SUBSS(X64Reg regOp, OpArg arg) {WriteSSEOp(0xF3, sseSUB, regOp, arg);} +void XEmitter::SUBSD(X64Reg regOp, OpArg arg) {WriteSSEOp(0xF2, sseSUB, regOp, arg);} +void XEmitter::CMPSS(X64Reg regOp, OpArg arg, u8 compare) {WriteSSEOp(0xF3, sseCMP, regOp, arg, 1); Write8(compare);} +void XEmitter::CMPSD(X64Reg regOp, OpArg arg, u8 compare) {WriteSSEOp(0xF2, sseCMP, regOp, arg, 1); Write8(compare);} +void XEmitter::MULSS(X64Reg regOp, OpArg arg) {WriteSSEOp(0xF3, sseMUL, regOp, arg);} +void XEmitter::MULSD(X64Reg regOp, OpArg arg) {WriteSSEOp(0xF2, sseMUL, regOp, arg);} +void XEmitter::DIVSS(X64Reg regOp, OpArg arg) {WriteSSEOp(0xF3, sseDIV, regOp, arg);} +void XEmitter::DIVSD(X64Reg regOp, OpArg arg) {WriteSSEOp(0xF2, sseDIV, regOp, arg);} +void XEmitter::MINSS(X64Reg regOp, OpArg arg) {WriteSSEOp(0xF3, sseMIN, regOp, arg);} +void XEmitter::MINSD(X64Reg regOp, OpArg arg) {WriteSSEOp(0xF2, sseMIN, regOp, arg);} +void XEmitter::MAXSS(X64Reg regOp, OpArg arg) {WriteSSEOp(0xF3, sseMAX, regOp, arg);} +void XEmitter::MAXSD(X64Reg regOp, OpArg arg) {WriteSSEOp(0xF2, sseMAX, regOp, arg);} +void XEmitter::SQRTSS(X64Reg regOp, OpArg arg) {WriteSSEOp(0xF3, sseSQRT, regOp, arg);} +void XEmitter::SQRTSD(X64Reg regOp, OpArg arg) {WriteSSEOp(0xF2, sseSQRT, regOp, arg);} +void XEmitter::RSQRTSS(X64Reg regOp, OpArg arg) {WriteSSEOp(0xF3, sseRSQRT, regOp, arg);} -void XEmitter::ADDPS(X64Reg regOp, OpArg arg) {WriteSSEOp(32, sseADD, true, regOp, arg);} -void XEmitter::ADDPD(X64Reg regOp, OpArg arg) {WriteSSEOp(64, sseADD, true, regOp, arg);} -void XEmitter::SUBPS(X64Reg regOp, OpArg arg) {WriteSSEOp(32, sseSUB, true, regOp, arg);} -void XEmitter::SUBPD(X64Reg regOp, OpArg arg) {WriteSSEOp(64, sseSUB, true, regOp, arg);} -void XEmitter::CMPPS(X64Reg regOp, OpArg arg, u8 compare) {WriteSSEOp(32, sseCMP, true, regOp, arg,1); Write8(compare);} -void XEmitter::CMPPD(X64Reg regOp, OpArg arg, u8 compare) {WriteSSEOp(64, sseCMP, true, regOp, arg,1); Write8(compare);} -void XEmitter::ANDPS(X64Reg regOp, OpArg arg) {WriteSSEOp(32, sseAND, true, regOp, arg);} -void XEmitter::ANDPD(X64Reg regOp, OpArg arg) {WriteSSEOp(64, sseAND, true, regOp, arg);} -void XEmitter::ANDNPS(X64Reg regOp, OpArg arg) {WriteSSEOp(32, sseANDN, true, regOp, arg);} -void XEmitter::ANDNPD(X64Reg regOp, OpArg arg) {WriteSSEOp(64, sseANDN, true, regOp, arg);} -void XEmitter::ORPS(X64Reg regOp, OpArg arg) {WriteSSEOp(32, sseOR, true, regOp, arg);} -void XEmitter::ORPD(X64Reg regOp, OpArg arg) {WriteSSEOp(64, sseOR, true, regOp, arg);} -void XEmitter::XORPS(X64Reg regOp, OpArg arg) {WriteSSEOp(32, sseXOR, true, regOp, arg);} -void XEmitter::XORPD(X64Reg regOp, OpArg arg) {WriteSSEOp(64, sseXOR, true, regOp, arg);} -void XEmitter::MULPS(X64Reg regOp, OpArg arg) {WriteSSEOp(32, sseMUL, true, regOp, arg);} -void XEmitter::MULPD(X64Reg regOp, OpArg arg) {WriteSSEOp(64, sseMUL, true, regOp, arg);} -void XEmitter::DIVPS(X64Reg regOp, OpArg arg) {WriteSSEOp(32, sseDIV, true, regOp, arg);} -void XEmitter::DIVPD(X64Reg regOp, OpArg arg) {WriteSSEOp(64, sseDIV, true, regOp, arg);} -void XEmitter::MINPS(X64Reg regOp, OpArg arg) {WriteSSEOp(32, sseMIN, true, regOp, arg);} -void XEmitter::MINPD(X64Reg regOp, OpArg arg) {WriteSSEOp(64, sseMIN, true, regOp, arg);} -void XEmitter::MAXPS(X64Reg regOp, OpArg arg) {WriteSSEOp(32, sseMAX, true, regOp, arg);} -void XEmitter::MAXPD(X64Reg regOp, OpArg arg) {WriteSSEOp(64, sseMAX, true, regOp, arg);} -void XEmitter::SQRTPS(X64Reg regOp, OpArg arg) {WriteSSEOp(32, sseSQRT, true, regOp, arg);} -void XEmitter::SQRTPD(X64Reg regOp, OpArg arg) {WriteSSEOp(64, sseSQRT, true, regOp, arg);} -void XEmitter::RSQRTPS(X64Reg regOp, OpArg arg) {WriteSSEOp(32, sseRSQRT, true, regOp, arg);} -void XEmitter::SHUFPS(X64Reg regOp, OpArg arg, u8 shuffle) {WriteSSEOp(32, sseSHUF, true, regOp, arg,1); Write8(shuffle);} -void XEmitter::SHUFPD(X64Reg regOp, OpArg arg, u8 shuffle) {WriteSSEOp(64, sseSHUF, true, regOp, arg,1); Write8(shuffle);} +void XEmitter::ADDPS(X64Reg regOp, OpArg arg) {WriteSSEOp(0x00, sseADD, regOp, arg);} +void XEmitter::ADDPD(X64Reg regOp, OpArg arg) {WriteSSEOp(0x66, sseADD, regOp, arg);} +void XEmitter::SUBPS(X64Reg regOp, OpArg arg) {WriteSSEOp(0x00, sseSUB, regOp, arg);} +void XEmitter::SUBPD(X64Reg regOp, OpArg arg) {WriteSSEOp(0x66, sseSUB, regOp, arg);} +void XEmitter::CMPPS(X64Reg regOp, OpArg arg, u8 compare) {WriteSSEOp(0x00, sseCMP, regOp, arg, 1); Write8(compare);} +void XEmitter::CMPPD(X64Reg regOp, OpArg arg, u8 compare) {WriteSSEOp(0x66, sseCMP, regOp, arg, 1); Write8(compare);} +void XEmitter::ANDPS(X64Reg regOp, OpArg arg) {WriteSSEOp(0x00, sseAND, regOp, arg);} +void XEmitter::ANDPD(X64Reg regOp, OpArg arg) {WriteSSEOp(0x66, sseAND, regOp, arg);} +void XEmitter::ANDNPS(X64Reg regOp, OpArg arg) {WriteSSEOp(0x00, sseANDN, regOp, arg);} +void XEmitter::ANDNPD(X64Reg regOp, OpArg arg) {WriteSSEOp(0x66, sseANDN, regOp, arg);} +void XEmitter::ORPS(X64Reg regOp, OpArg arg) {WriteSSEOp(0x00, sseOR, regOp, arg);} +void XEmitter::ORPD(X64Reg regOp, OpArg arg) {WriteSSEOp(0x66, sseOR, regOp, arg);} +void XEmitter::XORPS(X64Reg regOp, OpArg arg) {WriteSSEOp(0x00, sseXOR, regOp, arg);} +void XEmitter::XORPD(X64Reg regOp, OpArg arg) {WriteSSEOp(0x66, sseXOR, regOp, arg);} +void XEmitter::MULPS(X64Reg regOp, OpArg arg) {WriteSSEOp(0x00, sseMUL, regOp, arg);} +void XEmitter::MULPD(X64Reg regOp, OpArg arg) {WriteSSEOp(0x66, sseMUL, regOp, arg);} +void XEmitter::DIVPS(X64Reg regOp, OpArg arg) {WriteSSEOp(0x00, sseDIV, regOp, arg);} +void XEmitter::DIVPD(X64Reg regOp, OpArg arg) {WriteSSEOp(0x66, sseDIV, regOp, arg);} +void XEmitter::MINPS(X64Reg regOp, OpArg arg) {WriteSSEOp(0x00, sseMIN, regOp, arg);} +void XEmitter::MINPD(X64Reg regOp, OpArg arg) {WriteSSEOp(0x66, sseMIN, regOp, arg);} +void XEmitter::MAXPS(X64Reg regOp, OpArg arg) {WriteSSEOp(0x00, sseMAX, regOp, arg);} +void XEmitter::MAXPD(X64Reg regOp, OpArg arg) {WriteSSEOp(0x66, sseMAX, regOp, arg);} +void XEmitter::SQRTPS(X64Reg regOp, OpArg arg) {WriteSSEOp(0x00, sseSQRT, regOp, arg);} +void XEmitter::SQRTPD(X64Reg regOp, OpArg arg) {WriteSSEOp(0x66, sseSQRT, regOp, arg);} +void XEmitter::RSQRTPS(X64Reg regOp, OpArg arg) {WriteSSEOp(0x00, sseRSQRT, regOp, arg);} +void XEmitter::SHUFPS(X64Reg regOp, OpArg arg, u8 shuffle) {WriteSSEOp(0x00, sseSHUF, regOp, arg,1); Write8(shuffle);} +void XEmitter::SHUFPD(X64Reg regOp, OpArg arg, u8 shuffle) {WriteSSEOp(0x66, sseSHUF, regOp, arg,1); Write8(shuffle);} -void XEmitter::COMISS(X64Reg regOp, OpArg arg) {WriteSSEOp(32, sseCOMIS, true, regOp, arg);} //weird that these should be packed -void XEmitter::COMISD(X64Reg regOp, OpArg arg) {WriteSSEOp(64, sseCOMIS, true, regOp, arg);} //ordered -void XEmitter::UCOMISS(X64Reg regOp, OpArg arg) {WriteSSEOp(32, sseUCOMIS, true, regOp, arg);} //unordered -void XEmitter::UCOMISD(X64Reg regOp, OpArg arg) {WriteSSEOp(64, sseUCOMIS, true, regOp, arg);} +void XEmitter::COMISS(X64Reg regOp, OpArg arg) {WriteSSEOp(0x00, sseCOMIS, regOp, arg);} //weird that these should be packed +void XEmitter::COMISD(X64Reg regOp, OpArg arg) {WriteSSEOp(0x66, sseCOMIS, regOp, arg);} //ordered +void XEmitter::UCOMISS(X64Reg regOp, OpArg arg) {WriteSSEOp(0x00, sseUCOMIS, regOp, arg);} //unordered +void XEmitter::UCOMISD(X64Reg regOp, OpArg arg) {WriteSSEOp(0x66, sseUCOMIS, regOp, arg);} -void XEmitter::MOVAPS(X64Reg regOp, OpArg arg) {WriteSSEOp(32, sseMOVAPfromRM, true, regOp, arg);} -void XEmitter::MOVAPD(X64Reg regOp, OpArg arg) {WriteSSEOp(64, sseMOVAPfromRM, true, regOp, arg);} -void XEmitter::MOVAPS(OpArg arg, X64Reg regOp) {WriteSSEOp(32, sseMOVAPtoRM, true, regOp, arg);} -void XEmitter::MOVAPD(OpArg arg, X64Reg regOp) {WriteSSEOp(64, sseMOVAPtoRM, true, regOp, arg);} +void XEmitter::MOVAPS(X64Reg regOp, OpArg arg) {WriteSSEOp(0x00, sseMOVAPfromRM, regOp, arg);} +void XEmitter::MOVAPD(X64Reg regOp, OpArg arg) {WriteSSEOp(0x66, sseMOVAPfromRM, regOp, arg);} +void XEmitter::MOVAPS(OpArg arg, X64Reg regOp) {WriteSSEOp(0x00, sseMOVAPtoRM, regOp, arg);} +void XEmitter::MOVAPD(OpArg arg, X64Reg regOp) {WriteSSEOp(0x66, sseMOVAPtoRM, regOp, arg);} -void XEmitter::MOVUPS(X64Reg regOp, OpArg arg) {WriteSSEOp(32, sseMOVUPfromRM, true, regOp, arg);} -void XEmitter::MOVUPD(X64Reg regOp, OpArg arg) {WriteSSEOp(64, sseMOVUPfromRM, true, regOp, arg);} -void XEmitter::MOVUPS(OpArg arg, X64Reg regOp) {WriteSSEOp(32, sseMOVUPtoRM, true, regOp, arg);} -void XEmitter::MOVUPD(OpArg arg, X64Reg regOp) {WriteSSEOp(64, sseMOVUPtoRM, true, regOp, arg);} +void XEmitter::MOVUPS(X64Reg regOp, OpArg arg) {WriteSSEOp(0x00, sseMOVUPfromRM, regOp, arg);} +void XEmitter::MOVUPD(X64Reg regOp, OpArg arg) {WriteSSEOp(0x66, sseMOVUPfromRM, regOp, arg);} +void XEmitter::MOVUPS(OpArg arg, X64Reg regOp) {WriteSSEOp(0x00, sseMOVUPtoRM, regOp, arg);} +void XEmitter::MOVUPD(OpArg arg, X64Reg regOp) {WriteSSEOp(0x66, sseMOVUPtoRM, regOp, arg);} -void XEmitter::MOVDQA(X64Reg regOp, OpArg arg) {WriteSSEOp(64, sseMOVDQfromRM, true, regOp, arg);} -void XEmitter::MOVDQA(OpArg arg, X64Reg regOp) {WriteSSEOp(64, sseMOVDQtoRM, true, regOp, arg);} -void XEmitter::MOVDQU(X64Reg regOp, OpArg arg) {WriteSSEOp(32, sseMOVDQfromRM, false, regOp, arg);} -void XEmitter::MOVDQU(OpArg arg, X64Reg regOp) {WriteSSEOp(32, sseMOVDQtoRM, false, regOp, arg);} +void XEmitter::MOVDQA(X64Reg regOp, OpArg arg) {WriteSSEOp(0x66, sseMOVDQfromRM, regOp, arg);} +void XEmitter::MOVDQA(OpArg arg, X64Reg regOp) {WriteSSEOp(0x66, sseMOVDQtoRM, regOp, arg);} +void XEmitter::MOVDQU(X64Reg regOp, OpArg arg) {WriteSSEOp(0xF3, sseMOVDQfromRM, regOp, arg);} +void XEmitter::MOVDQU(OpArg arg, X64Reg regOp) {WriteSSEOp(0xF3, sseMOVDQtoRM, regOp, arg);} -void XEmitter::MOVSS(X64Reg regOp, OpArg arg) {WriteSSEOp(32, sseMOVUPfromRM, false, regOp, arg);} -void XEmitter::MOVSD(X64Reg regOp, OpArg arg) {WriteSSEOp(64, sseMOVUPfromRM, false, regOp, arg);} -void XEmitter::MOVSS(OpArg arg, X64Reg regOp) {WriteSSEOp(32, sseMOVUPtoRM, false, regOp, arg);} -void XEmitter::MOVSD(OpArg arg, X64Reg regOp) {WriteSSEOp(64, sseMOVUPtoRM, false, regOp, arg);} +void XEmitter::MOVSS(X64Reg regOp, OpArg arg) {WriteSSEOp(0xF3, sseMOVUPfromRM, regOp, arg);} +void XEmitter::MOVSD(X64Reg regOp, OpArg arg) {WriteSSEOp(0xF2, sseMOVUPfromRM, regOp, arg);} +void XEmitter::MOVSS(OpArg arg, X64Reg regOp) {WriteSSEOp(0xF3, sseMOVUPtoRM, regOp, arg);} +void XEmitter::MOVSD(OpArg arg, X64Reg regOp) {WriteSSEOp(0xF2, sseMOVUPtoRM, regOp, arg);} -void XEmitter::CVTPS2PD(X64Reg regOp, OpArg arg) {WriteSSEOp(32, 0x5A, true, regOp, arg);} -void XEmitter::CVTPD2PS(X64Reg regOp, OpArg arg) {WriteSSEOp(64, 0x5A, true, regOp, arg);} +void XEmitter::MOVLPD(X64Reg regOp, OpArg arg) {WriteSSEOp(0xF2, sseMOVLPDfromRM, regOp, arg);} +void XEmitter::MOVHPD(X64Reg regOp, OpArg arg) {WriteSSEOp(0xF2, sseMOVHPDfromRM, regOp, arg);} +void XEmitter::MOVLPD(OpArg arg, X64Reg regOp) {WriteSSEOp(0xF2, sseMOVLPDtoRM, regOp, arg);} +void XEmitter::MOVHPD(OpArg arg, X64Reg regOp) {WriteSSEOp(0xF2, sseMOVHPDtoRM, regOp, arg);} -void XEmitter::CVTSD2SS(X64Reg regOp, OpArg arg) {WriteSSEOp(64, 0x5A, false, regOp, arg);} -void XEmitter::CVTSS2SD(X64Reg regOp, OpArg arg) {WriteSSEOp(32, 0x5A, false, regOp, arg);} -void XEmitter::CVTSD2SI(X64Reg regOp, OpArg arg) {WriteSSEOp(64, 0x2D, false, regOp, arg);} +void XEmitter::MOVHLPS(X64Reg regOp1, X64Reg regOp2) {WriteSSEOp(0x00, sseMOVHLPS, regOp1, R(regOp2));} +void XEmitter::MOVLHPS(X64Reg regOp1, X64Reg regOp2) {WriteSSEOp(0x00, sseMOVLHPS, regOp1, R(regOp2));} -void XEmitter::CVTDQ2PD(X64Reg regOp, OpArg arg) {WriteSSEOp(32, 0xE6, false, regOp, arg);} -void XEmitter::CVTDQ2PS(X64Reg regOp, OpArg arg) {WriteSSEOp(32, 0x5B, true, regOp, arg);} -void XEmitter::CVTPD2DQ(X64Reg regOp, OpArg arg) {WriteSSEOp(64, 0xE6, false, regOp, arg);} -void XEmitter::CVTPS2DQ(X64Reg regOp, OpArg arg) {WriteSSEOp(64, 0x5B, true, regOp, arg);} +void XEmitter::CVTPS2PD(X64Reg regOp, OpArg arg) {WriteSSEOp(0x00, 0x5A, regOp, arg);} +void XEmitter::CVTPD2PS(X64Reg regOp, OpArg arg) {WriteSSEOp(0x66, 0x5A, regOp, arg);} -void XEmitter::CVTSI2SS(X64Reg xregdest, OpArg arg) {WriteSSEOp(32, 0x2A, false, xregdest, arg);} -void XEmitter::CVTSS2SI(X64Reg xregdest, OpArg arg) {WriteSSEOp(32, 0x2D, false, xregdest, arg);} -void XEmitter::CVTTSS2SI(X64Reg xregdest, OpArg arg) {WriteSSEOp(32, 0x2C, false, xregdest, arg);} -void XEmitter::CVTTPS2DQ(X64Reg xregdest, OpArg arg) {WriteSSEOp(32, 0x5B, false, xregdest, arg);} -void XEmitter::CVTTSD2SI(X64Reg xregdest, OpArg arg) {WriteSSEOp(64, 0x2C, false, xregdest, arg);} -void XEmitter::CVTTPD2DQ(X64Reg xregdest, OpArg arg) {WriteSSEOp(64, 0xE6, true, xregdest, arg); } +void XEmitter::CVTSD2SS(X64Reg regOp, OpArg arg) {WriteSSEOp(0xF2, 0x5A, regOp, arg);} +void XEmitter::CVTSS2SD(X64Reg regOp, OpArg arg) {WriteSSEOp(0xF3, 0x5A, regOp, arg);} +void XEmitter::CVTSD2SI(X64Reg regOp, OpArg arg) {WriteSSEOp(0xF2, 0x2D, regOp, arg);} +void XEmitter::CVTSS2SI(X64Reg regOp, OpArg arg) {WriteSSEOp(0xF3, 0x2D, regOp, arg);} +void XEmitter::CVTSI2SD(X64Reg regOp, OpArg arg) {WriteSSEOp(0xF2, 0x2A, regOp, arg);} +void XEmitter::CVTSI2SS(X64Reg regOp, OpArg arg) {WriteSSEOp(0xF3, 0x2A, regOp, arg);} -void XEmitter::MASKMOVDQU(X64Reg dest, X64Reg src) {WriteSSEOp(64, sseMASKMOVDQU, true, dest, R(src));} +void XEmitter::CVTDQ2PD(X64Reg regOp, OpArg arg) {WriteSSEOp(0xF3, 0xE6, regOp, arg);} +void XEmitter::CVTDQ2PS(X64Reg regOp, OpArg arg) {WriteSSEOp(0x00, 0x5B, regOp, arg);} +void XEmitter::CVTPD2DQ(X64Reg regOp, OpArg arg) {WriteSSEOp(0xF2, 0xE6, regOp, arg);} +void XEmitter::CVTPS2DQ(X64Reg regOp, OpArg arg) {WriteSSEOp(0x66, 0x5B, regOp, arg);} -void XEmitter::MOVMSKPS(X64Reg dest, OpArg arg) {WriteSSEOp(32, 0x50, true, dest, arg);} -void XEmitter::MOVMSKPD(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0x50, true, dest, arg);} +void XEmitter::CVTTSD2SI(X64Reg regOp, OpArg arg) {WriteSSEOp(0xF2, 0x2C, regOp, arg);} +void XEmitter::CVTTSS2SI(X64Reg regOp, OpArg arg) {WriteSSEOp(0xF3, 0x2C, regOp, arg);} +void XEmitter::CVTTPS2DQ(X64Reg regOp, OpArg arg) {WriteSSEOp(0xF3, 0x5B, regOp, arg);} +void XEmitter::CVTTPD2DQ(X64Reg regOp, OpArg arg) {WriteSSEOp(0x66, 0xE6, regOp, arg);} -void XEmitter::LDDQU(X64Reg dest, OpArg arg) {WriteSSEOp(64, sseLDDQU, false, dest, arg);} // For integer data only +void XEmitter::MASKMOVDQU(X64Reg dest, X64Reg src) {WriteSSEOp(0x66, sseMASKMOVDQU, dest, R(src));} + +void XEmitter::MOVMSKPS(X64Reg dest, OpArg arg) {WriteSSEOp(0x00, 0x50, dest, arg);} +void XEmitter::MOVMSKPD(X64Reg dest, OpArg arg) {WriteSSEOp(0x66, 0x50, dest, arg);} + +void XEmitter::LDDQU(X64Reg dest, OpArg arg) {WriteSSEOp(0xF2, sseLDDQU, dest, arg);} // For integer data only // THESE TWO ARE UNTESTED. -void XEmitter::UNPCKLPS(X64Reg dest, OpArg arg) {WriteSSEOp(32, 0x14, true, dest, arg);} -void XEmitter::UNPCKHPS(X64Reg dest, OpArg arg) {WriteSSEOp(32, 0x15, true, dest, arg);} +void XEmitter::UNPCKLPS(X64Reg dest, OpArg arg) {WriteSSEOp(0x00, 0x14, dest, arg);} +void XEmitter::UNPCKHPS(X64Reg dest, OpArg arg) {WriteSSEOp(0x00, 0x15, dest, arg);} -void XEmitter::UNPCKLPD(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0x14, true, dest, arg);} -void XEmitter::UNPCKHPD(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0x15, true, dest, arg);} +void XEmitter::UNPCKLPD(X64Reg dest, OpArg arg) {WriteSSEOp(0x66, 0x14, dest, arg);} +void XEmitter::UNPCKHPD(X64Reg dest, OpArg arg) {WriteSSEOp(0x66, 0x15, dest, arg);} void XEmitter::MOVDDUP(X64Reg regOp, OpArg arg) { if (cpu_info.bSSE3) { - WriteSSEOp(64, 0x12, false, regOp, arg); //SSE3 movddup + WriteSSEOp(0xF2, 0x12, regOp, arg); //SSE3 movddup } else { @@ -1410,101 +1652,69 @@ void XEmitter::MOVDDUP(X64Reg regOp, OpArg arg) //There are a few more left // Also some integer instructions are missing -void XEmitter::PACKSSDW(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0x6B, true, dest, arg);} -void XEmitter::PACKSSWB(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0x63, true, dest, arg);} -//void PACKUSDW(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0x66, true, dest, arg);} // WRONG -void XEmitter::PACKUSWB(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0x67, true, dest, arg);} +void XEmitter::PACKSSDW(X64Reg dest, OpArg arg) {WriteSSEOp(0x66, 0x6B, dest, arg);} +void XEmitter::PACKSSWB(X64Reg dest, OpArg arg) {WriteSSEOp(0x66, 0x63, dest, arg);} +void XEmitter::PACKUSWB(X64Reg dest, OpArg arg) {WriteSSEOp(0x66, 0x67, dest, arg);} -void XEmitter::PUNPCKLBW(X64Reg dest, const OpArg &arg) {WriteSSEOp(64, 0x60, true, dest, arg);} -void XEmitter::PUNPCKLWD(X64Reg dest, const OpArg &arg) {WriteSSEOp(64, 0x61, true, dest, arg);} -void XEmitter::PUNPCKLDQ(X64Reg dest, const OpArg &arg) {WriteSSEOp(64, 0x62, true, dest, arg);} -//void PUNPCKLQDQ(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0x60, true, dest, arg);} +void XEmitter::PUNPCKLBW(X64Reg dest, const OpArg &arg) {WriteSSEOp(0x66, 0x60, dest, arg);} +void XEmitter::PUNPCKLWD(X64Reg dest, const OpArg &arg) {WriteSSEOp(0x66, 0x61, dest, arg);} +void XEmitter::PUNPCKLDQ(X64Reg dest, const OpArg &arg) {WriteSSEOp(0x66, 0x62, dest, arg);} -void XEmitter::PMOVSXBW(X64Reg dest, const OpArg &arg) { - if (!cpu_info.bSSE4_1) { - PanicAlert("Trying to use PMOVSXBW on a system that doesn't support it. Bad programmer."); - } - WriteSSEOp2(64, 0x20, true, dest, arg); -} - -void XEmitter::PMOVSXBD(X64Reg dest, const OpArg &arg) { - if (!cpu_info.bSSE4_1) { - PanicAlert("Trying to use PMOVSXBD on a system that doesn't support it. Bad programmer."); - } - WriteSSEOp2(64, 0x21, true, dest, arg); -} - -void XEmitter::PMOVSXWD(X64Reg dest, const OpArg &arg) { - if (!cpu_info.bSSE4_1) { - PanicAlert("Trying to use PMOVSXWD on a system that doesn't support it. Bad programmer."); - } - WriteSSEOp2(64, 0x23, true, dest, arg); -} - -void XEmitter::PMOVZXBW(X64Reg dest, const OpArg &arg) { - if (!cpu_info.bSSE4_1) { - PanicAlert("Trying to use PMOVSXBW on a system that doesn't support it. Bad programmer."); - } - WriteSSEOp2(64, 0x30, true, dest, arg); -} - -void XEmitter::PMOVZXBD(X64Reg dest, const OpArg &arg) { - if (!cpu_info.bSSE4_1) { - PanicAlert("Trying to use PMOVSXBD on a system that doesn't support it. Bad programmer."); - } - WriteSSEOp2(64, 0x31, true, dest, arg); -} - -void XEmitter::PMOVZXWD(X64Reg dest, const OpArg &arg) { - if (!cpu_info.bSSE4_1) { - PanicAlert("Trying to use PMOVSXWD on a system that doesn't support it. Bad programmer."); - } - WriteSSEOp2(64, 0x33, true, dest, arg); -} - -void XEmitter::PSRLW(X64Reg reg, int shift) { - WriteSSEOp(64, 0x71, true, (X64Reg)2, R(reg)); +void XEmitter::PSRLW(X64Reg reg, int shift) +{ + WriteSSEOp(0x66, 0x71, (X64Reg)2, R(reg)); Write8(shift); } -void XEmitter::PSRLD(X64Reg reg, int shift) { - WriteSSEOp(64, 0x72, true, (X64Reg)2, R(reg)); +void XEmitter::PSRLD(X64Reg reg, int shift) +{ + WriteSSEOp(0x66, 0x72, (X64Reg)2, R(reg)); Write8(shift); } -void XEmitter::PSRLQ(X64Reg reg, int shift) { - WriteSSEOp(64, 0x73, true, (X64Reg)2, R(reg)); +void XEmitter::PSRLQ(X64Reg reg, int shift) +{ + WriteSSEOp(0x66, 0x73, (X64Reg)2, R(reg)); Write8(shift); } -void XEmitter::PSLLW(X64Reg reg, int shift) { - WriteSSEOp(64, 0x71, true, (X64Reg)6, R(reg)); +void XEmitter::PSRLQ(X64Reg reg, OpArg arg) +{ + WriteSSEOp(0x66, 0xd3, reg, arg); +} + +void XEmitter::PSRLDQ(X64Reg reg, int shift) { + WriteSSEOp(0x66, 0x73, (X64Reg)3, R(reg)); Write8(shift); } -void XEmitter::PSLLD(X64Reg reg, int shift) { - WriteSSEOp(64, 0x72, true, (X64Reg)6, R(reg)); +void XEmitter::PSLLW(X64Reg reg, int shift) +{ + WriteSSEOp(0x66, 0x71, (X64Reg)6, R(reg)); Write8(shift); } -void XEmitter::PSLLQ(X64Reg reg, int shift) { - WriteSSEOp(64, 0x73, true, (X64Reg)6, R(reg)); +void XEmitter::PSLLD(X64Reg reg, int shift) +{ + WriteSSEOp(0x66, 0x72, (X64Reg)6, R(reg)); + Write8(shift); +} + +void XEmitter::PSLLQ(X64Reg reg, int shift) +{ + WriteSSEOp(0x66, 0x73, (X64Reg)6, R(reg)); Write8(shift); } void XEmitter::PSLLDQ(X64Reg reg, int shift) { - WriteSSEOp(64, 0x73, true, (X64Reg)7, R(reg)); - Write8(shift); -} - -void XEmitter::PSRLDQ(X64Reg reg, int shift) { - WriteSSEOp(64, 0x73, true, (X64Reg)3, R(reg)); + WriteSSEOp(0x66, 0x73, (X64Reg)7, R(reg)); Write8(shift); } // WARNING not REX compatible -void XEmitter::PSRAW(X64Reg reg, int shift) { +void XEmitter::PSRAW(X64Reg reg, int shift) +{ if (reg > 7) PanicAlert("The PSRAW-emitter does not support regs above 7"); Write8(0x66); @@ -1515,7 +1725,8 @@ void XEmitter::PSRAW(X64Reg reg, int shift) { } // WARNING not REX compatible -void XEmitter::PSRAD(X64Reg reg, int shift) { +void XEmitter::PSRAD(X64Reg reg, int shift) +{ if (reg > 7) PanicAlert("The PSRAD-emitter does not support regs above 7"); Write8(0x66); @@ -1525,83 +1736,163 @@ void XEmitter::PSRAD(X64Reg reg, int shift) { Write8(shift); } -void XEmitter::PSHUFB(X64Reg dest, OpArg arg) { - if (!cpu_info.bSSSE3) { - PanicAlert("Trying to use PSHUFB on a system that doesn't support it. Bad programmer."); - } - WriteSSEOp2(64, 0x00, true, dest, arg); +void XEmitter::WriteSSSE3Op(u8 opPrefix, u16 op, X64Reg regOp, OpArg arg, int extrabytes) +{ + if (!cpu_info.bSSSE3) + PanicAlert("Trying to use SSSE3 on a system that doesn't support it. Bad programmer."); + WriteSSEOp(opPrefix, op, regOp, arg, extrabytes); } -void XEmitter::PAND(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0xDB, true, dest, arg);} -void XEmitter::PANDN(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0xDF, true, dest, arg);} -void XEmitter::PXOR(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0xEF, true, dest, arg);} -void XEmitter::POR(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0xEB, true, dest, arg);} +void XEmitter::WriteSSE41Op(u8 opPrefix, u16 op, X64Reg regOp, OpArg arg, int extrabytes) +{ + if (!cpu_info.bSSE4_1) + PanicAlert("Trying to use SSE4.1 on a system that doesn't support it. Bad programmer."); + WriteSSEOp(opPrefix, op, regOp, arg, extrabytes); +} -void XEmitter::PADDB(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0xFC, true, dest, arg);} -void XEmitter::PADDW(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0xFD, true, dest, arg);} -void XEmitter::PADDD(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0xFE, true, dest, arg);} -void XEmitter::PADDQ(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0xD4, true, dest, arg);} +void XEmitter::PSHUFB(X64Reg dest, OpArg arg) {WriteSSSE3Op(0x66, 0x3800, dest, arg);} +void XEmitter::PTEST(X64Reg dest, OpArg arg) {WriteSSE41Op(0x66, 0x3817, dest, arg);} +void XEmitter::PACKUSDW(X64Reg dest, OpArg arg) {WriteSSE41Op(0x66, 0x382b, dest, arg);} -void XEmitter::PADDSB(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0xEC, true, dest, arg);} -void XEmitter::PADDSW(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0xED, true, dest, arg);} -void XEmitter::PADDUSB(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0xDC, true, dest, arg);} -void XEmitter::PADDUSW(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0xDD, true, dest, arg);} +void XEmitter::PMOVSXBW(X64Reg dest, OpArg arg) {WriteSSE41Op(0x66, 0x3820, dest, arg);} +void XEmitter::PMOVSXBD(X64Reg dest, OpArg arg) {WriteSSE41Op(0x66, 0x3821, dest, arg);} +void XEmitter::PMOVSXBQ(X64Reg dest, OpArg arg) {WriteSSE41Op(0x66, 0x3822, dest, arg);} +void XEmitter::PMOVSXWD(X64Reg dest, OpArg arg) {WriteSSE41Op(0x66, 0x3823, dest, arg);} +void XEmitter::PMOVSXWQ(X64Reg dest, OpArg arg) {WriteSSE41Op(0x66, 0x3824, dest, arg);} +void XEmitter::PMOVSXDQ(X64Reg dest, OpArg arg) {WriteSSE41Op(0x66, 0x3825, dest, arg);} +void XEmitter::PMOVZXBW(X64Reg dest, OpArg arg) {WriteSSE41Op(0x66, 0x3830, dest, arg);} +void XEmitter::PMOVZXBD(X64Reg dest, OpArg arg) {WriteSSE41Op(0x66, 0x3831, dest, arg);} +void XEmitter::PMOVZXBQ(X64Reg dest, OpArg arg) {WriteSSE41Op(0x66, 0x3832, dest, arg);} +void XEmitter::PMOVZXWD(X64Reg dest, OpArg arg) {WriteSSE41Op(0x66, 0x3833, dest, arg);} +void XEmitter::PMOVZXWQ(X64Reg dest, OpArg arg) {WriteSSE41Op(0x66, 0x3834, dest, arg);} +void XEmitter::PMOVZXDQ(X64Reg dest, OpArg arg) {WriteSSE41Op(0x66, 0x3835, dest, arg);} -void XEmitter::PSUBB(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0xF8, true, dest, arg);} -void XEmitter::PSUBW(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0xF9, true, dest, arg);} -void XEmitter::PSUBD(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0xFA, true, dest, arg);} -void XEmitter::PSUBQ(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0xDB, true, dest, arg);} +void XEmitter::PBLENDVB(X64Reg dest, OpArg arg) {WriteSSE41Op(0x66, 0x3810, dest, arg);} +void XEmitter::BLENDVPS(X64Reg dest, OpArg arg) {WriteSSE41Op(0x66, 0x3814, dest, arg);} +void XEmitter::BLENDVPD(X64Reg dest, OpArg arg) {WriteSSE41Op(0x66, 0x3815, dest, arg);} -void XEmitter::PSUBSB(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0xE8, true, dest, arg);} -void XEmitter::PSUBSW(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0xE9, true, dest, arg);} -void XEmitter::PSUBUSB(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0xD8, true, dest, arg);} -void XEmitter::PSUBUSW(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0xD9, true, dest, arg);} +void XEmitter::PAND(X64Reg dest, OpArg arg) {WriteSSEOp(0x66, 0xDB, dest, arg);} +void XEmitter::PANDN(X64Reg dest, OpArg arg) {WriteSSEOp(0x66, 0xDF, dest, arg);} +void XEmitter::PXOR(X64Reg dest, OpArg arg) {WriteSSEOp(0x66, 0xEF, dest, arg);} +void XEmitter::POR(X64Reg dest, OpArg arg) {WriteSSEOp(0x66, 0xEB, dest, arg);} -void XEmitter::PAVGB(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0xE0, true, dest, arg);} -void XEmitter::PAVGW(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0xE3, true, dest, arg);} +void XEmitter::PADDB(X64Reg dest, OpArg arg) {WriteSSEOp(0x66, 0xFC, dest, arg);} +void XEmitter::PADDW(X64Reg dest, OpArg arg) {WriteSSEOp(0x66, 0xFD, dest, arg);} +void XEmitter::PADDD(X64Reg dest, OpArg arg) {WriteSSEOp(0x66, 0xFE, dest, arg);} +void XEmitter::PADDQ(X64Reg dest, OpArg arg) {WriteSSEOp(0x66, 0xD4, dest, arg);} -void XEmitter::PCMPEQB(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0x74, true, dest, arg);} -void XEmitter::PCMPEQW(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0x75, true, dest, arg);} -void XEmitter::PCMPEQD(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0x76, true, dest, arg);} +void XEmitter::PADDSB(X64Reg dest, OpArg arg) {WriteSSEOp(0x66, 0xEC, dest, arg);} +void XEmitter::PADDSW(X64Reg dest, OpArg arg) {WriteSSEOp(0x66, 0xED, dest, arg);} +void XEmitter::PADDUSB(X64Reg dest, OpArg arg) {WriteSSEOp(0x66, 0xDC, dest, arg);} +void XEmitter::PADDUSW(X64Reg dest, OpArg arg) {WriteSSEOp(0x66, 0xDD, dest, arg);} -void XEmitter::PCMPGTB(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0x64, true, dest, arg);} -void XEmitter::PCMPGTW(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0x65, true, dest, arg);} -void XEmitter::PCMPGTD(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0x66, true, dest, arg);} +void XEmitter::PSUBB(X64Reg dest, OpArg arg) {WriteSSEOp(0x66, 0xF8, dest, arg);} +void XEmitter::PSUBW(X64Reg dest, OpArg arg) {WriteSSEOp(0x66, 0xF9, dest, arg);} +void XEmitter::PSUBD(X64Reg dest, OpArg arg) {WriteSSEOp(0x66, 0xFA, dest, arg);} +void XEmitter::PSUBQ(X64Reg dest, OpArg arg) {WriteSSEOp(0x66, 0xFB, dest, arg);} -void XEmitter::PEXTRW(X64Reg dest, OpArg arg, u8 subreg) {WriteSSEOp(64, 0xC5, true, dest, arg); Write8(subreg);} -void XEmitter::PINSRW(X64Reg dest, OpArg arg, u8 subreg) {WriteSSEOp(64, 0xC4, true, dest, arg); Write8(subreg);} +void XEmitter::PSUBSB(X64Reg dest, OpArg arg) {WriteSSEOp(0x66, 0xE8, dest, arg);} +void XEmitter::PSUBSW(X64Reg dest, OpArg arg) {WriteSSEOp(0x66, 0xE9, dest, arg);} +void XEmitter::PSUBUSB(X64Reg dest, OpArg arg) {WriteSSEOp(0x66, 0xD8, dest, arg);} +void XEmitter::PSUBUSW(X64Reg dest, OpArg arg) {WriteSSEOp(0x66, 0xD9, dest, arg);} -void XEmitter::PMADDWD(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0xF5, true, dest, arg); } -void XEmitter::PSADBW(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0xF6, true, dest, arg);} +void XEmitter::PAVGB(X64Reg dest, OpArg arg) {WriteSSEOp(0x66, 0xE0, dest, arg);} +void XEmitter::PAVGW(X64Reg dest, OpArg arg) {WriteSSEOp(0x66, 0xE3, dest, arg);} -void XEmitter::PMAXSW(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0xEE, true, dest, arg); } -void XEmitter::PMAXUB(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0xDE, true, dest, arg); } -void XEmitter::PMINSW(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0xEA, true, dest, arg); } -void XEmitter::PMINUB(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0xDA, true, dest, arg); } +void XEmitter::PCMPEQB(X64Reg dest, OpArg arg) {WriteSSEOp(0x66, 0x74, dest, arg);} +void XEmitter::PCMPEQW(X64Reg dest, OpArg arg) {WriteSSEOp(0x66, 0x75, dest, arg);} +void XEmitter::PCMPEQD(X64Reg dest, OpArg arg) {WriteSSEOp(0x66, 0x76, dest, arg);} -void XEmitter::PMOVMSKB(X64Reg dest, OpArg arg) {WriteSSEOp(64, 0xD7, true, dest, arg); } +void XEmitter::PCMPGTB(X64Reg dest, OpArg arg) {WriteSSEOp(0x66, 0x64, dest, arg);} +void XEmitter::PCMPGTW(X64Reg dest, OpArg arg) {WriteSSEOp(0x66, 0x65, dest, arg);} +void XEmitter::PCMPGTD(X64Reg dest, OpArg arg) {WriteSSEOp(0x66, 0x66, dest, arg);} -void XEmitter::PSHUFD(X64Reg regOp, OpArg arg, u8 shuffle) {WriteSSEOp(64, 0x70, true, regOp, arg, 1); Write8(shuffle);} -void XEmitter::PSHUFLW(X64Reg regOp, OpArg arg, u8 shuffle) {WriteSSEOp(64, 0x70, false, regOp, arg, 1); Write8(shuffle);} +void XEmitter::PEXTRW(X64Reg dest, OpArg arg, u8 subreg) {WriteSSEOp(0x66, 0xC5, dest, arg); Write8(subreg);} +void XEmitter::PINSRW(X64Reg dest, OpArg arg, u8 subreg) {WriteSSEOp(0x66, 0xC4, dest, arg); Write8(subreg);} + +void XEmitter::PMADDWD(X64Reg dest, OpArg arg) {WriteSSEOp(0x66, 0xF5, dest, arg); } +void XEmitter::PSADBW(X64Reg dest, OpArg arg) {WriteSSEOp(0x66, 0xF6, dest, arg);} + +void XEmitter::PMAXSW(X64Reg dest, OpArg arg) {WriteSSEOp(0x66, 0xEE, dest, arg); } +void XEmitter::PMAXUB(X64Reg dest, OpArg arg) {WriteSSEOp(0x66, 0xDE, dest, arg); } +void XEmitter::PMINSW(X64Reg dest, OpArg arg) {WriteSSEOp(0x66, 0xEA, dest, arg); } +void XEmitter::PMINUB(X64Reg dest, OpArg arg) {WriteSSEOp(0x66, 0xDA, dest, arg); } + +void XEmitter::PMOVMSKB(X64Reg dest, OpArg arg) {WriteSSEOp(0x66, 0xD7, dest, arg); } +void XEmitter::PSHUFD(X64Reg regOp, OpArg arg, u8 shuffle) {WriteSSEOp(0x66, 0x70, regOp, arg, 1); Write8(shuffle);} +void XEmitter::PSHUFLW(X64Reg regOp, OpArg arg, u8 shuffle) {WriteSSEOp(0xF2, 0x70, regOp, arg, 1); Write8(shuffle);} +void XEmitter::PSHUFHW(X64Reg regOp, OpArg arg, u8 shuffle) {WriteSSEOp(0xF3, 0x70, regOp, arg, 1); Write8(shuffle);} // VEX -void XEmitter::VADDSD(X64Reg regOp1, X64Reg regOp2, OpArg arg) {WriteAVXOp(64, sseADD, false, regOp1, regOp2, arg);} -void XEmitter::VSUBSD(X64Reg regOp1, X64Reg regOp2, OpArg arg) {WriteAVXOp(64, sseSUB, false, regOp1, regOp2, arg);} -void XEmitter::VMULSD(X64Reg regOp1, X64Reg regOp2, OpArg arg) {WriteAVXOp(64, sseMUL, false, regOp1, regOp2, arg);} -void XEmitter::VDIVSD(X64Reg regOp1, X64Reg regOp2, OpArg arg) {WriteAVXOp(64, sseDIV, false, regOp1, regOp2, arg);} -void XEmitter::VSQRTSD(X64Reg regOp1, X64Reg regOp2, OpArg arg) {WriteAVXOp(64, sseSQRT, false, regOp1, regOp2, arg);} +void XEmitter::VADDSD(X64Reg regOp1, X64Reg regOp2, OpArg arg) {WriteAVXOp(0xF2, sseADD, regOp1, regOp2, arg);} +void XEmitter::VSUBSD(X64Reg regOp1, X64Reg regOp2, OpArg arg) {WriteAVXOp(0xF2, sseSUB, regOp1, regOp2, arg);} +void XEmitter::VMULSD(X64Reg regOp1, X64Reg regOp2, OpArg arg) {WriteAVXOp(0xF2, sseMUL, regOp1, regOp2, arg);} +void XEmitter::VDIVSD(X64Reg regOp1, X64Reg regOp2, OpArg arg) {WriteAVXOp(0xF2, sseDIV, regOp1, regOp2, arg);} +void XEmitter::VADDPD(X64Reg regOp1, X64Reg regOp2, OpArg arg) {WriteAVXOp(0x66, sseADD, regOp1, regOp2, arg);} +void XEmitter::VSUBPD(X64Reg regOp1, X64Reg regOp2, OpArg arg) {WriteAVXOp(0x66, sseSUB, regOp1, regOp2, arg);} +void XEmitter::VMULPD(X64Reg regOp1, X64Reg regOp2, OpArg arg) {WriteAVXOp(0x66, sseMUL, regOp1, regOp2, arg);} +void XEmitter::VDIVPD(X64Reg regOp1, X64Reg regOp2, OpArg arg) {WriteAVXOp(0x66, sseDIV, regOp1, regOp2, arg);} +void XEmitter::VSQRTSD(X64Reg regOp1, X64Reg regOp2, OpArg arg) {WriteAVXOp(0xF2, sseSQRT, regOp1, regOp2, arg);} +void XEmitter::VPAND(X64Reg regOp1, X64Reg regOp2, OpArg arg) {WriteAVXOp(0x66, sseAND, regOp1, regOp2, arg);} +void XEmitter::VPANDN(X64Reg regOp1, X64Reg regOp2, OpArg arg) {WriteAVXOp(0x66, sseANDN, regOp1, regOp2, arg);} +void XEmitter::VPOR(X64Reg regOp1, X64Reg regOp2, OpArg arg) {WriteAVXOp(0x66, sseOR, regOp1, regOp2, arg);} +void XEmitter::VPXOR(X64Reg regOp1, X64Reg regOp2, OpArg arg) {WriteAVXOp(0x66, sseXOR, regOp1, regOp2, arg);} +void XEmitter::VSHUFPD(X64Reg regOp1, X64Reg regOp2, OpArg arg, u8 shuffle) {WriteAVXOp(0x66, sseSHUF, regOp1, regOp2, arg, 1); Write8(shuffle);} +void XEmitter::VUNPCKLPD(X64Reg regOp1, X64Reg regOp2, OpArg arg){WriteAVXOp(0x66, 0x14, regOp1, regOp2, arg);} +void XEmitter::VUNPCKHPD(X64Reg regOp1, X64Reg regOp2, OpArg arg){WriteAVXOp(0x66, 0x15, regOp1, regOp2, arg);} + +void XEmitter::SARX(int bits, X64Reg regOp1, OpArg arg, X64Reg regOp2) {WriteBMI2Op(bits, 0xF3, 0x38F7, regOp1, regOp2, arg);} +void XEmitter::SHLX(int bits, X64Reg regOp1, OpArg arg, X64Reg regOp2) {WriteBMI2Op(bits, 0x66, 0x38F7, regOp1, regOp2, arg);} +void XEmitter::SHRX(int bits, X64Reg regOp1, OpArg arg, X64Reg regOp2) {WriteBMI2Op(bits, 0xF2, 0x38F7, regOp1, regOp2, arg);} +void XEmitter::RORX(int bits, X64Reg regOp, OpArg arg, u8 rotate) {WriteBMI2Op(bits, 0xF2, 0x3AF0, regOp, INVALID_REG, arg, 1); Write8(rotate);} +void XEmitter::PEXT(int bits, X64Reg regOp1, X64Reg regOp2, OpArg arg) {WriteBMI2Op(bits, 0xF3, 0x38F5, regOp1, regOp2, arg);} +void XEmitter::PDEP(int bits, X64Reg regOp1, X64Reg regOp2, OpArg arg) {WriteBMI2Op(bits, 0xF2, 0x38F5, regOp1, regOp2, arg);} +void XEmitter::MULX(int bits, X64Reg regOp1, X64Reg regOp2, OpArg arg) {WriteBMI2Op(bits, 0xF2, 0x38F6, regOp2, regOp1, arg);} +void XEmitter::BZHI(int bits, X64Reg regOp1, OpArg arg, X64Reg regOp2) {WriteBMI2Op(bits, 0x00, 0x38F5, regOp1, regOp2, arg);} +void XEmitter::BLSR(int bits, X64Reg regOp, OpArg arg) {WriteBMI1Op(bits, 0x00, 0x38F3, (X64Reg)0x1, regOp, arg);} +void XEmitter::BLSMSK(int bits, X64Reg regOp, OpArg arg) {WriteBMI1Op(bits, 0x00, 0x38F3, (X64Reg)0x2, regOp, arg);} +void XEmitter::BLSI(int bits, X64Reg regOp, OpArg arg) {WriteBMI1Op(bits, 0x00, 0x38F3, (X64Reg)0x3, regOp, arg);} +void XEmitter::BEXTR(int bits, X64Reg regOp1, OpArg arg, X64Reg regOp2){WriteBMI1Op(bits, 0x00, 0x38F7, regOp1, regOp2, arg);} +void XEmitter::ANDN(int bits, X64Reg regOp1, X64Reg regOp2, OpArg arg) {WriteBMI1Op(bits, 0x00, 0x38F2, regOp1, regOp2, arg);} // Prefixes void XEmitter::LOCK() { Write8(0xF0); } void XEmitter::REP() { Write8(0xF3); } void XEmitter::REPNE() { Write8(0xF2); } +void XEmitter::FSOverride() { Write8(0x64); } +void XEmitter::GSOverride() { Write8(0x65); } -void XEmitter::FWAIT() { +void XEmitter::FWAIT() +{ Write8(0x9B); } -void XEmitter::RTDSC() { Write8(0x0F); Write8(0x31); } +// TODO: make this more generic +void XEmitter::WriteFloatLoadStore(int bits, FloatOp op, FloatOp op_80b, OpArg arg) +{ + int mf = 0; + _assert_msg_(DYNA_REC, !(bits == 80 && op_80b == floatINVALID), "WriteFloatLoadStore: 80 bits not supported for this instruction"); + switch (bits) + { + case 32: mf = 0; break; + case 64: mf = 4; break; + case 80: mf = 2; break; + default: _assert_msg_(DYNA_REC, 0, "WriteFloatLoadStore: invalid bits (should be 32/64/80)"); + } + Write8(0xd9 | mf); + // x87 instructions use the reg field of the ModR/M byte as opcode: + if (bits == 80) + op = op_80b; + arg.WriteRest(this, 0, (X64Reg) op); +} + +void XEmitter::FLD(int bits, OpArg src) {WriteFloatLoadStore(bits, floatLD, floatLD80, src);} +void XEmitter::FST(int bits, OpArg dest) {WriteFloatLoadStore(bits, floatST, floatINVALID, dest);} +void XEmitter::FSTP(int bits, OpArg dest) {WriteFloatLoadStore(bits, floatSTP, floatSTP80, dest);} +void XEmitter::FNSTSW_AX() { Write8(0xDF); Write8(0xE0); } + +void XEmitter::RDTSC() { Write8(0x0F); Write8(0x31); } void XCodeBlock::AllocCodeSpace(int size) { region_size = size; @@ -1625,5 +1916,4 @@ void XCodeBlock::WriteProtect() { WriteProtectMemory(region, region_size, true); } -} // Gen - +} diff --git a/Common/x64Emitter.h b/Common/x64Emitter.h index 2b163ff52a..3af96eea84 100644 --- a/Common/x64Emitter.h +++ b/Common/x64Emitter.h @@ -22,6 +22,10 @@ #include "Common.h" +#ifdef _M_X64 +#define _ARCH_64 +#endif + namespace Gen { @@ -55,10 +59,10 @@ enum CCFlags { CC_O = 0, CC_NO = 1, - CC_B = 2, CC_C = 2, CC_NAE = 2, - CC_NB = 3, CC_NC = 3, CC_AE = 3, + CC_B = 2, CC_C = 2, CC_NAE = 2, + CC_NB = 3, CC_NC = 3, CC_AE = 3, CC_Z = 4, CC_E = 4, - CC_NZ = 5, CC_NE = 5, + CC_NZ = 5, CC_NE = 5, CC_BE = 6, CC_NA = 6, CC_NBE = 7, CC_A = 7, CC_S = 8, @@ -121,6 +125,16 @@ enum { CMP_ORD = 7, }; +enum FloatOp { + floatLD = 0, + floatST = 2, + floatSTP = 3, + floatLD80 = 5, + floatSTP80 = 7, + + floatINVALID = -1, +}; + class XEmitter; // RIP addressing does not benefit from micro op fusion on Core arch @@ -136,9 +150,15 @@ struct OpArg //if scale == 0 never mind offsetting offset = _offset; } + bool operator==(OpArg b) + { + return operandReg == b.operandReg && scale == b.scale && offsetOrBaseReg == b.offsetOrBaseReg && + indexReg == b.indexReg && offset == b.offset; + } void WriteRex(XEmitter *emit, int opBits, int bits, int customOp = -1) const; - void WriteVex(XEmitter* emit, int size, int packed, Gen::X64Reg regOp1, X64Reg regOp2) const; - void WriteRest(XEmitter *emit, int extraBytes=0, X64Reg operandReg=(X64Reg)0xFF, bool warn_64bit_offset = true) const; + void WriteVex(XEmitter* emit, X64Reg regOp1, X64Reg regOp2, int L, int pp, int mmmmm, int W = 0) const; + void WriteRest(XEmitter *emit, int extraBytes=0, X64Reg operandReg=INVALID_REG, bool warn_64bit_offset = true) const; + void WriteFloatModRM(XEmitter *emit, FloatOp op); void WriteSingleByteOp(XEmitter *emit, u8 op, X64Reg operandReg, int bits); // This one is public - must be written to u64 offset; // use RIP-relative as much as possible - 64-bit immediates are not available. @@ -147,7 +167,8 @@ struct OpArg void WriteNormalOp(XEmitter *emit, bool toRM, NormalOp op, const OpArg &operand, int bits) const; bool IsImm() const {return scale == SCALE_IMM8 || scale == SCALE_IMM16 || scale == SCALE_IMM32 || scale == SCALE_IMM64;} bool IsSimpleReg() const {return scale == SCALE_NONE;} - bool IsSimpleReg(X64Reg reg) const { + bool IsSimpleReg(X64Reg reg) const + { if (!IsSimpleReg()) return false; return GetSimpleReg() == reg; @@ -195,26 +216,35 @@ private: u16 indexReg; }; -inline OpArg M(void *ptr) {return OpArg((u64)ptr, (int)SCALE_RIP);} +inline OpArg M(const void *ptr) {return OpArg((u64)ptr, (int)SCALE_RIP);} template inline OpArg M(const T *ptr) {return OpArg((u64)(const void *)ptr, (int)SCALE_RIP);} -inline OpArg R(X64Reg value) {return OpArg(0, SCALE_NONE, value);} +inline OpArg R(X64Reg value) {return OpArg(0, SCALE_NONE, value);} inline OpArg MatR(X64Reg value) {return OpArg(0, SCALE_ATREG, value);} -inline OpArg MDisp(X64Reg value, int offset) { + +inline OpArg MDisp(X64Reg value, int offset) +{ return OpArg((u32)offset, SCALE_ATREG, value); } -inline OpArg MComplex(X64Reg base, X64Reg scaled, int scale, int offset) { + +inline OpArg MComplex(X64Reg base, X64Reg scaled, int scale, int offset) +{ return OpArg(offset, scale, base, scaled); } -inline OpArg MScaled(X64Reg scaled, int scale, int offset) { + +inline OpArg MScaled(X64Reg scaled, int scale, int offset) +{ if (scale == SCALE_1) return OpArg(offset, SCALE_ATREG, scaled); else return OpArg(offset, scale | 0x20, RAX, scaled); } -inline OpArg MRegSum(X64Reg base, X64Reg offset) { + +inline OpArg MRegSum(X64Reg base, X64Reg offset) +{ return MComplex(base, offset, 1, 0); } + inline OpArg Imm8 (u8 imm) {return OpArg(imm, SCALE_IMM8);} inline OpArg Imm16(u16 imm) {return OpArg(imm, SCALE_IMM16);} //rarely used inline OpArg Imm32(u32 imm) {return OpArg(imm, SCALE_IMM32);} @@ -226,19 +256,23 @@ inline OpArg SImmAuto(s32 imm) { return OpArg(imm, (imm >= 128 || imm < -128) ? SCALE_IMM32 : SCALE_IMM8); } -#ifdef _M_X64 +#ifdef _ARCH_64 inline OpArg ImmPtr(const void* imm) {return Imm64((u64)imm);} #else inline OpArg ImmPtr(const void* imm) {return Imm32((u32)imm);} #endif -inline u32 PtrOffset(const void* ptr, const void* base) { -#ifdef _M_X64 + +inline u32 PtrOffset(const void* ptr, const void* base) +{ +#ifdef _ARCH_64 s64 distance = (s64)ptr-(s64)base; if (distance >= 0x80000000LL || - distance < -0x80000000LL) { - _assert_msg_(JIT, 0, "pointer offset out of range"); + distance < -0x80000000LL) + { + _assert_msg_(DYNA_REC, 0, "pointer offset out of range"); return 0; } + return (u32)distance; #else return (u32)ptr-(u32)base; @@ -275,21 +309,31 @@ class XEmitter friend struct OpArg; // for Write8 etc private: u8 *code; + bool flags_locked; + + void CheckFlags(); void Rex(int w, int r, int x, int b); void WriteSimple1Byte(int bits, u8 byte, X64Reg reg); void WriteSimple2Byte(int bits, u8 byte1, u8 byte2, X64Reg reg); void WriteMulDivType(int bits, OpArg src, int ext); - void WriteBitSearchType(int bits, X64Reg dest, OpArg src, u8 byte2); + void WriteBitSearchType(int bits, X64Reg dest, OpArg src, u8 byte2, bool rep = false); void WriteShift(int bits, OpArg dest, OpArg &shift, int ext); void WriteBitTest(int bits, OpArg &dest, OpArg &index, int ext); void WriteMXCSR(OpArg arg, int ext); - void WriteSSEOp(int size, u8 sseOp, bool packed, X64Reg regOp, OpArg arg, int extrabytes = 0); - void WriteSSEOp2(int size, u8 sseOp, bool packed, X64Reg regOp, OpArg arg, int extrabytes = 0); - void WriteAVXOp(int size, u8 sseOp, bool packed, X64Reg regOp, OpArg arg, int extrabytes = 0); - void WriteAVXOp(int size, u8 sseOp, bool packed, X64Reg regOp1, X64Reg regOp2, OpArg arg, int extrabytes = 0); + void WriteSSEOp(u8 opPrefix, u16 op, X64Reg regOp, OpArg arg, int extrabytes = 0); + void WriteSSSE3Op(u8 opPrefix, u16 op, X64Reg regOp, OpArg arg, int extrabytes = 0); + void WriteSSE41Op(u8 opPrefix, u16 op, X64Reg regOp, OpArg arg, int extrabytes = 0); + void WriteAVXOp(u8 opPrefix, u16 op, X64Reg regOp, OpArg arg, int extrabytes = 0); + void WriteAVXOp(u8 opPrefix, u16 op, X64Reg regOp1, X64Reg regOp2, OpArg arg, int extrabytes = 0); + void WriteVEXOp(int size, u8 opPrefix, u16 op, X64Reg regOp1, X64Reg regOp2, OpArg arg, int extrabytes = 0); + void WriteBMI1Op(int size, u8 opPrefix, u16 op, X64Reg regOp1, X64Reg regOp2, OpArg arg, int extrabytes = 0); + void WriteBMI2Op(int size, u8 opPrefix, u16 op, X64Reg regOp1, X64Reg regOp2, OpArg arg, int extrabytes = 0); + void WriteFloatLoadStore(int bits, FloatOp op, FloatOp op_80b, OpArg arg); void WriteNormalOp(XEmitter *emit, int bits, NormalOp op, const OpArg &a1, const OpArg &a2); + void ABI_CalculateFrameSize(u32 mask, size_t rsp_alignment, size_t needed_frame_size, size_t* shadowp, size_t* subtractionp, size_t* xmm_offsetp); + protected: inline void Write8(u8 value) {*code++ = value;} inline void Write16(u16 value) {*(u16*)code = (value); code += 2;} @@ -297,8 +341,8 @@ protected: inline void Write64(u64 value) {*(u64*)code = (value); code += 8;} public: - XEmitter() { code = NULL; } - XEmitter(u8 *code_ptr) { code = code_ptr; } + XEmitter() { code = nullptr; flags_locked = false; } + XEmitter(u8 *code_ptr) { code = code_ptr; flags_locked = false; } virtual ~XEmitter() {} void WriteModRM(int mod, int rm, int reg); @@ -312,6 +356,9 @@ public: const u8 *GetCodePtr() const; u8 *GetWritableCodePtr(); + void LockFlags() { flags_locked = true; } + void UnlockFlags() { flags_locked = false; } + // Looking for one of these? It's BANNED!! Some instructions are slow on modern CPU // INC, DEC, LOOP, LOOPNE, LOOPE, ENTER, LEAVE, XCHG, XLAT, REP MOVSB/MOVSD, REP SCASD + other string instr., // INC and DEC are slow on Intel Core, but not on AMD. They create a @@ -322,7 +369,7 @@ public: void INT3(); // Do nothing - void NOP(int count = 1); //nop padding - TODO: fast nop slides, for amd and intel (check their manuals) + void NOP(size_t count = 1); // Save energy in wait-loops on P4 only. Probably not too useful. void PAUSE(); @@ -459,6 +506,14 @@ public: void MOVSX(int dbits, int sbits, X64Reg dest, OpArg src); //automatically uses MOVSXD if necessary void MOVZX(int dbits, int sbits, X64Reg dest, OpArg src); + // Available only on Atom or >= Haswell so far. Test with cpu_info.bMOVBE. + void MOVBE(int dbits, const OpArg& dest, const OpArg& src); + + // Available only on AMD >= Phenom or Intel >= Haswell + void LZCNT(int bits, X64Reg dest, OpArg src); + // Note: this one is actually part of BMI1 + void TZCNT(int bits, X64Reg dest, OpArg src); + // WARNING - These two take 11-13 cycles and are VectorPath! (AMD64) void STMXCSR(OpArg memloc); void LDMXCSR(OpArg memloc); @@ -467,7 +522,31 @@ public: void LOCK(); void REP(); void REPNE(); + void FSOverride(); + void GSOverride(); + // x87 + enum x87StatusWordBits { + x87_InvalidOperation = 0x1, + x87_DenormalizedOperand = 0x2, + x87_DivisionByZero = 0x4, + x87_Overflow = 0x8, + x87_Underflow = 0x10, + x87_Precision = 0x20, + x87_StackFault = 0x40, + x87_ErrorSummary = 0x80, + x87_C0 = 0x100, + x87_C1 = 0x200, + x87_C2 = 0x400, + x87_TopOfStack = 0x2000 | 0x1000 | 0x800, + x87_C3 = 0x4000, + x87_FPUBusy = 0x8000, + }; + + void FLD(int bits, OpArg src); + void FST(int bits, OpArg dest); + void FSTP(int bits, OpArg dest); + void FNSTSW_AX(); void FWAIT(); // SSE/SSE2: Floating point arithmetic @@ -490,14 +569,6 @@ public: // SSE/SSE2: Floating point bitwise (yes) void CMPSS(X64Reg regOp, OpArg arg, u8 compare); void CMPSD(X64Reg regOp, OpArg arg, u8 compare); - void ANDSS(X64Reg regOp, OpArg arg); - void ANDSD(X64Reg regOp, OpArg arg); - void ANDNSS(X64Reg regOp, OpArg arg); - void ANDNSD(X64Reg regOp, OpArg arg); - void ORSS(X64Reg regOp, OpArg arg); - void ORSD(X64Reg regOp, OpArg arg); - void XORSS(X64Reg regOp, OpArg arg); - void XORSD(X64Reg regOp, OpArg arg); inline void CMPEQSS(X64Reg regOp, OpArg arg) { CMPSS(regOp, arg, CMP_EQ); } inline void CMPLTSS(X64Reg regOp, OpArg arg) { CMPSS(regOp, arg, CMP_LT); } @@ -543,11 +614,8 @@ public: // SSE/SSE2: Useful alternative to shuffle in some cases. void MOVDDUP(X64Reg regOp, OpArg arg); - // THESE TWO ARE NEW AND UNTESTED void UNPCKLPS(X64Reg dest, OpArg src); void UNPCKHPS(X64Reg dest, OpArg src); - - // These are OK. void UNPCKLPD(X64Reg dest, OpArg src); void UNPCKHPD(X64Reg dest, OpArg src); @@ -568,7 +636,6 @@ public: void MOVUPS(OpArg arg, X64Reg regOp); void MOVUPD(OpArg arg, X64Reg regOp); - // Integers (NOTE: untested - I added these then it turned out I didn't have a use for them after all). void MOVDQA(X64Reg regOp, OpArg arg); void MOVDQA(OpArg arg, X64Reg regOp); void MOVDQU(X64Reg regOp, OpArg arg); @@ -579,6 +646,14 @@ public: void MOVSS(OpArg arg, X64Reg regOp); void MOVSD(OpArg arg, X64Reg regOp); + void MOVLPD(X64Reg regOp, OpArg arg); + void MOVHPD(X64Reg regOp, OpArg arg); + void MOVLPD(OpArg arg, X64Reg regOp); + void MOVHPD(OpArg arg, X64Reg regOp); + + void MOVHLPS(X64Reg regOp1, X64Reg regOp2); + void MOVLHPS(X64Reg regOp1, X64Reg regOp2); + void MOVD_xmm(X64Reg dest, const OpArg &arg); void MOVQ_xmm(X64Reg dest, OpArg arg); void MOVD_xmm(const OpArg &arg, X64Reg src); @@ -596,37 +671,34 @@ public: void CVTPS2PD(X64Reg dest, OpArg src); void CVTPD2PS(X64Reg dest, OpArg src); void CVTSS2SD(X64Reg dest, OpArg src); + void CVTSI2SS(X64Reg dest, OpArg src); void CVTSD2SS(X64Reg dest, OpArg src); - void CVTSD2SI(X64Reg dest, OpArg src); + void CVTSI2SD(X64Reg dest, OpArg src); void CVTDQ2PD(X64Reg regOp, OpArg arg); void CVTPD2DQ(X64Reg regOp, OpArg arg); void CVTDQ2PS(X64Reg regOp, OpArg arg); void CVTPS2DQ(X64Reg regOp, OpArg arg); - void CVTTSS2SI(X64Reg xregdest, OpArg arg); // Yeah, destination really is a GPR like EAX! void CVTTPS2DQ(X64Reg regOp, OpArg arg); - void CVTSI2SS(X64Reg xregdest, OpArg arg); // Yeah, destination really is a GPR like EAX! - void CVTSS2SI(X64Reg xregdest, OpArg arg); // Yeah, destination really is a GPR like EAX! - void CVTTSD2SI(X64Reg xregdest, OpArg arg); // Yeah, destination really is a GPR like EAX! - void CVTTPD2DQ(X64Reg xregdest, OpArg arg); + void CVTTPD2DQ(X64Reg regOp, OpArg arg); + + // Destinations are X64 regs (rax, rbx, ...) for these instructions. + void CVTSS2SI(X64Reg xregdest, OpArg src); + void CVTSD2SI(X64Reg xregdest, OpArg src); + void CVTTSS2SI(X64Reg xregdest, OpArg arg); + void CVTTSD2SI(X64Reg xregdest, OpArg arg); // SSE2: Packed integer instructions void PACKSSDW(X64Reg dest, OpArg arg); void PACKSSWB(X64Reg dest, OpArg arg); - //void PACKUSDW(X64Reg dest, OpArg arg); + void PACKUSDW(X64Reg dest, OpArg arg); void PACKUSWB(X64Reg dest, OpArg arg); void PUNPCKLBW(X64Reg dest, const OpArg &arg); void PUNPCKLWD(X64Reg dest, const OpArg &arg); void PUNPCKLDQ(X64Reg dest, const OpArg &arg); - void PMOVSXBW(X64Reg dest, const OpArg &arg); - void PMOVSXBD(X64Reg dest, const OpArg &arg); - void PMOVSXWD(X64Reg dest, const OpArg &arg); - void PMOVZXBW(X64Reg dest, const OpArg &arg); - void PMOVZXBD(X64Reg dest, const OpArg &arg); - void PMOVZXWD(X64Reg dest, const OpArg &arg); - + void PTEST(X64Reg dest, OpArg arg); void PAND(X64Reg dest, OpArg arg); void PANDN(X64Reg dest, OpArg arg); void PXOR(X64Reg dest, OpArg arg); @@ -680,29 +752,75 @@ public: void PSHUFB(X64Reg dest, OpArg arg); void PSHUFLW(X64Reg dest, OpArg arg, u8 shuffle); + void PSHUFHW(X64Reg dest, OpArg arg, u8 shuffle); void PSRLW(X64Reg reg, int shift); void PSRLD(X64Reg reg, int shift); void PSRLQ(X64Reg reg, int shift); + void PSRLQ(X64Reg reg, OpArg arg); + void PSRLDQ(X64Reg reg, int shift); void PSLLW(X64Reg reg, int shift); void PSLLD(X64Reg reg, int shift); void PSLLQ(X64Reg reg, int shift); - - void PSRLDQ(X64Reg reg, int shift); void PSLLDQ(X64Reg reg, int shift); void PSRAW(X64Reg reg, int shift); void PSRAD(X64Reg reg, int shift); + // SSE4: data type conversions + void PMOVSXBW(X64Reg dest, OpArg arg); + void PMOVSXBD(X64Reg dest, OpArg arg); + void PMOVSXBQ(X64Reg dest, OpArg arg); + void PMOVSXWD(X64Reg dest, OpArg arg); + void PMOVSXWQ(X64Reg dest, OpArg arg); + void PMOVSXDQ(X64Reg dest, OpArg arg); + void PMOVZXBW(X64Reg dest, OpArg arg); + void PMOVZXBD(X64Reg dest, OpArg arg); + void PMOVZXBQ(X64Reg dest, OpArg arg); + void PMOVZXWD(X64Reg dest, OpArg arg); + void PMOVZXWQ(X64Reg dest, OpArg arg); + void PMOVZXDQ(X64Reg dest, OpArg arg); + + // SSE4: variable blend instructions (xmm0 implicit argument) + void PBLENDVB(X64Reg dest, OpArg arg); + void BLENDVPS(X64Reg dest, OpArg arg); + void BLENDVPD(X64Reg dest, OpArg arg); + // AVX void VADDSD(X64Reg regOp1, X64Reg regOp2, OpArg arg); void VSUBSD(X64Reg regOp1, X64Reg regOp2, OpArg arg); void VMULSD(X64Reg regOp1, X64Reg regOp2, OpArg arg); void VDIVSD(X64Reg regOp1, X64Reg regOp2, OpArg arg); + void VADDPD(X64Reg regOp1, X64Reg regOp2, OpArg arg); + void VSUBPD(X64Reg regOp1, X64Reg regOp2, OpArg arg); + void VMULPD(X64Reg regOp1, X64Reg regOp2, OpArg arg); + void VDIVPD(X64Reg regOp1, X64Reg regOp2, OpArg arg); void VSQRTSD(X64Reg regOp1, X64Reg regOp2, OpArg arg); + void VPAND(X64Reg regOp1, X64Reg regOp2, OpArg arg); + void VPANDN(X64Reg regOp1, X64Reg regOp2, OpArg arg); + void VPOR(X64Reg regOp1, X64Reg regOp2, OpArg arg); + void VPXOR(X64Reg regOp1, X64Reg regOp2, OpArg arg); + void VSHUFPD(X64Reg regOp1, X64Reg regOp2, OpArg arg, u8 shuffle); + void VUNPCKLPD(X64Reg regOp1, X64Reg regOp2, OpArg arg); + void VUNPCKHPD(X64Reg regOp1, X64Reg regOp2, OpArg arg); - void RTDSC(); + // VEX GPR instructions + void SARX(int bits, X64Reg regOp1, OpArg arg, X64Reg regOp2); + void SHLX(int bits, X64Reg regOp1, OpArg arg, X64Reg regOp2); + void SHRX(int bits, X64Reg regOp1, OpArg arg, X64Reg regOp2); + void RORX(int bits, X64Reg regOp, OpArg arg, u8 rotate); + void PEXT(int bits, X64Reg regOp1, X64Reg regOp2, OpArg arg); + void PDEP(int bits, X64Reg regOp1, X64Reg regOp2, OpArg arg); + void MULX(int bits, X64Reg regOp1, X64Reg regOp2, OpArg arg); + void BZHI(int bits, X64Reg regOp1, OpArg arg, X64Reg regOp2); + void BLSR(int bits, X64Reg regOp, OpArg arg); + void BLSMSK(int bits, X64Reg regOp, OpArg arg); + void BLSI(int bits, X64Reg regOp, OpArg arg); + void BEXTR(int bits, X64Reg regOp1, OpArg arg, X64Reg regOp2); + void ANDN(int bits, X64Reg regOp1, X64Reg regOp2, OpArg arg); + + void RDTSC(); // Utility functions // The difference between this and CALL is that this aligns the stack @@ -719,6 +837,7 @@ public: void ABI_CallFunctionC16(const void *func, u16 param1); void ABI_CallFunctionCC16(const void *func, u32 param1, u16 param2); + // These only support u32 parameters, but that's enough for a lot of uses. // These will destroy the 1 or 2 first "parameter regs". void ABI_CallFunctionC(const void *func, u32 param1); @@ -736,8 +855,8 @@ public: void ABI_CallFunctionAA(const void *func, const Gen::OpArg &arg1, const Gen::OpArg &arg2); // Pass a register as a parameter. - void ABI_CallFunctionR(const void *func, Gen::X64Reg reg1); - void ABI_CallFunctionRR(const void *func, Gen::X64Reg reg1, Gen::X64Reg reg2); + void ABI_CallFunctionR(const void *func, X64Reg reg1); + void ABI_CallFunctionRR(const void *func, X64Reg reg1, X64Reg reg2); template void ABI_CallFunctionC(Tr (*func)(T1), u32 param1) { @@ -822,4 +941,4 @@ public: } // namespace -#endif // _DOLPHIN_INTEL_CODEGEN_ +#endif From 8d0dca71fe04d84993939de95a31160969032d31 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Sun, 12 Oct 2014 11:03:39 -0700 Subject: [PATCH 085/105] jit: Rename the rounding mode funcs to clarify. They apply/restore the value, set/clear is confusing. --- Core/MIPS/ARM/ArmAsm.cpp | 10 +++++----- Core/MIPS/ARM/ArmCompBranch.cpp | 4 ++-- Core/MIPS/ARM/ArmCompFPU.cpp | 12 ++++++------ Core/MIPS/ARM/ArmJit.cpp | 18 +++++++++--------- Core/MIPS/ARM/ArmJit.h | 4 ++-- Core/MIPS/x86/Asm.cpp | 12 ++++++------ Core/MIPS/x86/CompBranch.cpp | 4 ++-- Core/MIPS/x86/CompFPU.cpp | 8 ++++---- Core/MIPS/x86/Jit.cpp | 28 ++++++++++++++-------------- Core/MIPS/x86/Jit.h | 4 ++-- 10 files changed, 52 insertions(+), 52 deletions(-) diff --git a/Core/MIPS/ARM/ArmAsm.cpp b/Core/MIPS/ARM/ArmAsm.cpp index d3c15afd14..129564e232 100644 --- a/Core/MIPS/ARM/ArmAsm.cpp +++ b/Core/MIPS/ARM/ArmAsm.cpp @@ -114,9 +114,9 @@ void Jit::GenerateFixedCode() MovToPC(R0); outerLoop = GetCodePtr(); SaveDowncount(); - ClearRoundingMode(); + RestoreRoundingMode(); QuickCallFunction(R0, &CoreTiming::Advance); - SetRoundingMode(); + ApplyRoundingMode(); RestoreDowncount(); FixupBranch skipToRealDispatch = B(); //skip the sync and compare first time @@ -175,9 +175,9 @@ void Jit::GenerateFixedCode() // No block found, let's jit SaveDowncount(); - ClearRoundingMode(); + RestoreRoundingMode(); QuickCallFunction(R2, (void *)&JitAt); - SetRoundingMode(); + ApplyRoundingMode(); RestoreDowncount(); B(dispatcherNoCheck); // no point in special casing this @@ -199,7 +199,7 @@ void Jit::GenerateFixedCode() } SaveDowncount(); - ClearRoundingMode(); + RestoreRoundingMode(); ADD(R_SP, R_SP, 4); diff --git a/Core/MIPS/ARM/ArmCompBranch.cpp b/Core/MIPS/ARM/ArmCompBranch.cpp index e58c54e130..24d104d644 100644 --- a/Core/MIPS/ARM/ArmCompBranch.cpp +++ b/Core/MIPS/ARM/ArmCompBranch.cpp @@ -538,7 +538,7 @@ void Jit::Comp_Syscall(MIPSOpcode op) // If we're in a delay slot, this is off by one. const int offset = js.inDelaySlot ? -1 : 0; WriteDownCount(offset); - ClearRoundingMode(); + RestoreRoundingMode(); js.downcountAmount = -offset; // TODO: Maybe discard v0, v1, and some temps? Definitely at? @@ -558,7 +558,7 @@ void Jit::Comp_Syscall(MIPSOpcode op) gpr.SetRegImm(R0, op.encoding); QuickCallFunction(R1, (void *)&CallSyscall); } - SetRoundingMode(); + ApplyRoundingMode(); RestoreDowncount(); WriteSyscallExit(); diff --git a/Core/MIPS/ARM/ArmCompFPU.cpp b/Core/MIPS/ARM/ArmCompFPU.cpp index 2a52d9c8f4..770fe95ea3 100644 --- a/Core/MIPS/ARM/ArmCompFPU.cpp +++ b/Core/MIPS/ARM/ArmCompFPU.cpp @@ -280,7 +280,7 @@ void Jit::Comp_FPU2op(MIPSOpcode op) { VNEG(fpr.R(fd), fpr.R(fs)); break; case 12: //FsI(fd) = (int)floorf(F(fs)+0.5f); break; //round.w.s - ClearRoundingMode(); + RestoreRoundingMode(); fpr.MapDirtyIn(fd, fs); VCVT(fpr.R(fd), fpr.R(fs), TO_INT | IS_SIGNED); break; @@ -295,7 +295,7 @@ void Jit::Comp_FPU2op(MIPSOpcode op) { break; case 14: //FsI(fd) = (int)ceilf (F(fs)); break; //ceil.w.s { - ClearRoundingMode(); + RestoreRoundingMode(); fpr.MapDirtyIn(fd, fs); VMRS(SCRATCHREG2); // Assume we're always in round-to-nearest mode. @@ -313,7 +313,7 @@ void Jit::Comp_FPU2op(MIPSOpcode op) { } case 15: //FsI(fd) = (int)floorf(F(fs)); break; //floor.w.s { - ClearRoundingMode(); + RestoreRoundingMode(); fpr.MapDirtyIn(fd, fs); VMRS(SCRATCHREG2); // Assume we're always in round-to-nearest mode. @@ -399,8 +399,8 @@ void Jit::Comp_mxc1(MIPSOpcode op) case 6: //ctc1 if (fs == 31) { - // Must clear before setting, since SetRoundingMode() assumes it was cleared. - ClearRoundingMode(); + // Must clear before setting, since ApplyRoundingMode() assumes it was cleared. + RestoreRoundingMode(); bool wasImm = gpr.IsImm(rt); if (wasImm) { gpr.SetImm(MIPS_REG_FPCOND, (gpr.GetImm(rt) >> 23) & 1); @@ -420,7 +420,7 @@ void Jit::Comp_mxc1(MIPSOpcode op) AND(gpr.R(MIPS_REG_FPCOND), SCRATCHREG1, Operand2(1)); #endif } - SetRoundingMode(); + ApplyRoundingMode(); } else { Comp_Generic(op); } diff --git a/Core/MIPS/ARM/ArmJit.cpp b/Core/MIPS/ARM/ArmJit.cpp index d6c135fbf4..7a422cf5f6 100644 --- a/Core/MIPS/ARM/ArmJit.cpp +++ b/Core/MIPS/ARM/ArmJit.cpp @@ -381,14 +381,14 @@ bool Jit::ReplaceJalTo(u32 dest) { gpr.SetImm(MIPS_REG_RA, js.compilerPC + 8); CompileDelaySlot(DELAYSLOT_NICE); FlushAll(); - ClearRoundingMode(); + RestoreRoundingMode(); if (BLInRange((const void *)(entry->replaceFunc))) { BL((const void *)(entry->replaceFunc)); } else { MOVI2R(R0, (u32)entry->replaceFunc); BL(R0); } - SetRoundingMode(); + ApplyRoundingMode(); WriteDownCountR(R0); } @@ -435,7 +435,7 @@ void Jit::Comp_ReplacementFunc(MIPSOpcode op) } } else if (entry->replaceFunc) { FlushAll(); - ClearRoundingMode(); + RestoreRoundingMode(); gpr.SetRegImm(SCRATCHREG1, js.compilerPC); MovToPC(SCRATCHREG1); @@ -450,10 +450,10 @@ void Jit::Comp_ReplacementFunc(MIPSOpcode op) if (entry->flags & (REPFLAG_HOOKENTER | REPFLAG_HOOKEXIT)) { // Compile the original instruction at this address. We ignore cycles for hooks. - SetRoundingMode(); + ApplyRoundingMode(); MIPSCompileOp(Memory::Read_Instruction(js.compilerPC, true)); } else { - SetRoundingMode(); + ApplyRoundingMode(); LDR(R1, CTXREG, MIPS_REG_RA * 4); WriteDownCountR(R0); WriteExitDestInR(R1); @@ -472,12 +472,12 @@ void Jit::Comp_Generic(MIPSOpcode op) { SaveDowncount(); // TODO: Perhaps keep the rounding mode for interp? - ClearRoundingMode(); + RestoreRoundingMode(); gpr.SetRegImm(SCRATCHREG1, js.compilerPC); MovToPC(SCRATCHREG1); gpr.SetRegImm(R0, op.encoding); QuickCallFunction(R1, (void *)func); - SetRoundingMode(); + ApplyRoundingMode(); RestoreDowncount(); } @@ -547,7 +547,7 @@ void Jit::WriteDownCountR(ARMReg reg) { } } -void Jit::ClearRoundingMode() { +void Jit::RestoreRoundingMode() { if (g_Config.bSetRoundingMode) { VMRS(SCRATCHREG2); // Assume we're always in round-to-nearest mode beforehand. @@ -560,7 +560,7 @@ void Jit::ClearRoundingMode() { } } -void Jit::SetRoundingMode() { +void Jit::ApplyRoundingMode() { // NOTE: Must not destory R0. if (g_Config.bSetRoundingMode) { LDR(SCRATCHREG2, CTXREG, offsetof(MIPSState, fcr31)); diff --git a/Core/MIPS/ARM/ArmJit.h b/Core/MIPS/ARM/ArmJit.h index b5f1252253..d5cb448e35 100644 --- a/Core/MIPS/ARM/ArmJit.h +++ b/Core/MIPS/ARM/ArmJit.h @@ -192,8 +192,8 @@ private: void WriteDownCount(int offset = 0); void WriteDownCountR(ARMReg reg); - void ClearRoundingMode(); - void SetRoundingMode(); + void RestoreRoundingMode(); + void ApplyRoundingMode(); void MovFromPC(ARMReg r); void MovToPC(ARMReg r); diff --git a/Core/MIPS/x86/Asm.cpp b/Core/MIPS/x86/Asm.cpp index b4fbbe7161..dc25034e34 100644 --- a/Core/MIPS/x86/Asm.cpp +++ b/Core/MIPS/x86/Asm.cpp @@ -77,9 +77,9 @@ void AsmRoutineManager::Generate(MIPSState *mips, MIPSComp::Jit *jit) #endif outerLoop = GetCodePtr(); - jit->ClearRoundingMode(this); + jit->RestoreRoundingMode(this); ABI_CallFunction(reinterpret_cast(&CoreTiming::Advance)); - jit->SetRoundingMode(this); + jit->ApplyRoundingMode(this); FixupBranch skipToRealDispatch = J(); //skip the sync and compare first time dispatcherCheckCoreState = GetCodePtr(); @@ -134,9 +134,9 @@ void AsmRoutineManager::Generate(MIPSState *mips, MIPSComp::Jit *jit) SetJumpTarget(notfound); //Ok, no block, let's jit - jit->ClearRoundingMode(this); + jit->RestoreRoundingMode(this); ABI_CallFunction(&Jit); - jit->SetRoundingMode(this); + jit->ApplyRoundingMode(this); JMP(dispatcherNoCheck, true); // Let's just dispatch again, we'll enter the block since we know it's there. SetJumpTarget(bail); @@ -146,12 +146,12 @@ void AsmRoutineManager::Generate(MIPSState *mips, MIPSComp::Jit *jit) J_CC(CC_Z, outerLoop, true); SetJumpTarget(badCoreState); - jit->ClearRoundingMode(this); + jit->RestoreRoundingMode(this); ABI_PopAllCalleeSavedRegsAndAdjustStack(); RET(); breakpointBailout = GetCodePtr(); - jit->ClearRoundingMode(this); + jit->RestoreRoundingMode(this); ABI_PopAllCalleeSavedRegsAndAdjustStack(); RET(); } diff --git a/Core/MIPS/x86/CompBranch.cpp b/Core/MIPS/x86/CompBranch.cpp index 83ab2b298a..5c04a69c66 100644 --- a/Core/MIPS/x86/CompBranch.cpp +++ b/Core/MIPS/x86/CompBranch.cpp @@ -681,7 +681,7 @@ void Jit::Comp_Syscall(MIPSOpcode op) // If we're in a delay slot, this is off by one. const int offset = js.inDelaySlot ? -1 : 0; WriteDowncount(offset); - ClearRoundingMode(); + RestoreRoundingMode(); js.downcountAmount = -offset; // Skip the CallSyscall where possible. @@ -691,7 +691,7 @@ void Jit::Comp_Syscall(MIPSOpcode op) else ABI_CallFunctionC(&CallSyscall, op.encoding); - SetRoundingMode(); + ApplyRoundingMode(); WriteSyscallExit(); js.compiling = false; } diff --git a/Core/MIPS/x86/CompFPU.cpp b/Core/MIPS/x86/CompFPU.cpp index e0d9bb8046..01561d12af 100644 --- a/Core/MIPS/x86/CompFPU.cpp +++ b/Core/MIPS/x86/CompFPU.cpp @@ -379,15 +379,15 @@ void Jit::Comp_mxc1(MIPSOpcode op) case 6: //currentMIPS->WriteFCR(fs, R(rt)); break; //ctc1 if (fs == 31) { - // Must clear before setting, since SetRoundingMode() assumes it was cleared. - ClearRoundingMode(); + // Must clear before setting, since ApplyRoundingMode() assumes it was cleared. + RestoreRoundingMode(); if (gpr.IsImm(rt)) { gpr.SetImm(MIPS_REG_FPCOND, (gpr.GetImm(rt) >> 23) & 1); MOV(32, M(&mips_->fcr31), Imm32(gpr.GetImm(rt) & 0x0181FFFF)); if ((gpr.GetImm(rt) & 0x1000003) == 0) { // Default nearest / no-flush mode, just leave it cleared. } else { - SetRoundingMode(); + ApplyRoundingMode(); } } else { gpr.Lock(rt, MIPS_REG_FPCOND); @@ -399,7 +399,7 @@ void Jit::Comp_mxc1(MIPSOpcode op) MOV(32, M(&mips_->fcr31), gpr.R(rt)); AND(32, M(&mips_->fcr31), Imm32(0x0181FFFF)); gpr.UnlockAll(); - SetRoundingMode(); + ApplyRoundingMode(); } } else { Comp_Generic(op); diff --git a/Core/MIPS/x86/Jit.cpp b/Core/MIPS/x86/Jit.cpp index dd1699a608..34835962a1 100644 --- a/Core/MIPS/x86/Jit.cpp +++ b/Core/MIPS/x86/Jit.cpp @@ -211,7 +211,7 @@ void Jit::WriteDowncount(int offset) SUB(32, M(¤tMIPS->downcount), downcount > 127 ? Imm32(downcount) : Imm8(downcount)); } -void Jit::ClearRoundingMode(XEmitter *emitter) +void Jit::RestoreRoundingMode(XEmitter *emitter) { if (g_Config.bSetRoundingMode) { @@ -224,7 +224,7 @@ void Jit::ClearRoundingMode(XEmitter *emitter) } } -void Jit::SetRoundingMode(XEmitter *emitter) +void Jit::ApplyRoundingMode(XEmitter *emitter) { if (g_Config.bSetRoundingMode) { @@ -496,10 +496,10 @@ bool Jit::ReplaceJalTo(u32 dest) { CompileDelaySlot(DELAYSLOT_NICE); FlushAll(); MOV(32, M(&mips_->pc), Imm32(js.compilerPC)); - ClearRoundingMode(); + RestoreRoundingMode(); ABI_CallFunction(entry->replaceFunc); SUB(32, M(¤tMIPS->downcount), R(EAX)); - SetRoundingMode(); + ApplyRoundingMode(); } js.compilerPC += 4; @@ -548,17 +548,17 @@ void Jit::Comp_ReplacementFunc(MIPSOpcode op) // Standard function call, nothing fancy. // The function returns the number of cycles it took in EAX. MOV(32, M(&mips_->pc), Imm32(js.compilerPC)); - ClearRoundingMode(); + RestoreRoundingMode(); ABI_CallFunction(entry->replaceFunc); if (entry->flags & (REPFLAG_HOOKENTER | REPFLAG_HOOKEXIT)) { // Compile the original instruction at this address. We ignore cycles for hooks. - SetRoundingMode(); + ApplyRoundingMode(); MIPSCompileOp(Memory::Read_Instruction(js.compilerPC, true)); } else { MOV(32, R(ECX), M(¤tMIPS->r[MIPS_REG_RA])); SUB(32, M(¤tMIPS->downcount), R(EAX)); - SetRoundingMode(); + ApplyRoundingMode(); SUB(32, M(¤tMIPS->downcount), Imm8(0)); WriteExitDestInReg(ECX); js.compiling = false; @@ -577,13 +577,13 @@ void Jit::Comp_Generic(MIPSOpcode op) if (func) { // TODO: Maybe we'd be better off keeping the rounding mode within interp? - ClearRoundingMode(); + RestoreRoundingMode(); MOV(32, M(&mips_->pc), Imm32(js.compilerPC)); if (USE_JIT_MISSMAP) ABI_CallFunctionC(&JitLogMiss, op.encoding); else ABI_CallFunctionC(func, op.encoding); - SetRoundingMode(); + ApplyRoundingMode(); } else ERROR_LOG_REPORT(JIT, "Trying to compile instruction %08x that can't be interpreted", op.encoding); @@ -691,9 +691,9 @@ void Jit::WriteSyscallExit() { WriteDowncount(); if (js.afterOp & JitState::AFTER_MEMCHECK_CLEANUP) { - ClearRoundingMode(); + RestoreRoundingMode(); ABI_CallFunction(&JitMemCheckCleanup); - SetRoundingMode(); + ApplyRoundingMode(); } JMP(asm_.dispatcherCheckCoreState, true); } @@ -705,20 +705,20 @@ bool Jit::CheckJitBreakpoint(u32 addr, int downcountOffset) SAVE_FLAGS; FlushAll(); MOV(32, M(&mips_->pc), Imm32(js.compilerPC)); - ClearRoundingMode(); + RestoreRoundingMode(); ABI_CallFunction(&JitBreakpoint); // If 0, the conditional breakpoint wasn't taken. CMP(32, R(EAX), Imm32(0)); FixupBranch skip = J_CC(CC_Z); WriteDowncount(downcountOffset); + ApplyRoundingMode(); // Just to fix the stack. - SetRoundingMode(); LOAD_FLAGS; JMP(asm_.dispatcherCheckCoreState, true); SetJumpTarget(skip); - SetRoundingMode(); + ApplyRoundingMode(); LOAD_FLAGS; return true; diff --git a/Core/MIPS/x86/Jit.h b/Core/MIPS/x86/Jit.h index 385f5c4ccd..9e58ca6580 100644 --- a/Core/MIPS/x86/Jit.h +++ b/Core/MIPS/x86/Jit.h @@ -165,8 +165,8 @@ public: void GetVectorRegsPrefixD(u8 *regs, VectorSize sz, int vectorReg); void EatPrefix() { js.EatPrefix(); } - void ClearRoundingMode(XEmitter *emitter = NULL); - void SetRoundingMode(XEmitter *emitter = NULL); + void RestoreRoundingMode(XEmitter *emitter = NULL); + void ApplyRoundingMode(XEmitter *emitter = NULL); JitBlockCache *GetBlockCache() { return &blocks; } AsmRoutineManager &Asm() { return asm_; } From 928e2adfc97886f95386eb78471d088ba3cc9634 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Sun, 12 Oct 2014 11:34:26 -0700 Subject: [PATCH 086/105] jit: Avoid applying/restoring the rounding mode. If the game never sets it, we can skip around syscalls, interpreter, replacements, etc. --- Core/MIPS/ARM/ArmAsm.cpp | 10 +++--- Core/MIPS/ARM/ArmCompFPU.cpp | 1 + Core/MIPS/ARM/ArmJit.cpp | 65 +++++++++++++++++++++++++++++----- Core/MIPS/ARM/ArmJit.h | 5 +-- Core/MIPS/JitCommon/JitState.h | 7 +++- Core/MIPS/x86/Asm.cpp | 12 +++---- Core/MIPS/x86/CompFPU.cpp | 2 ++ Core/MIPS/x86/Jit.cpp | 62 ++++++++++++++++++++++++++------ Core/MIPS/x86/Jit.h | 5 +-- 9 files changed, 135 insertions(+), 34 deletions(-) diff --git a/Core/MIPS/ARM/ArmAsm.cpp b/Core/MIPS/ARM/ArmAsm.cpp index 129564e232..773c851ab9 100644 --- a/Core/MIPS/ARM/ArmAsm.cpp +++ b/Core/MIPS/ARM/ArmAsm.cpp @@ -114,9 +114,9 @@ void Jit::GenerateFixedCode() MovToPC(R0); outerLoop = GetCodePtr(); SaveDowncount(); - RestoreRoundingMode(); + RestoreRoundingMode(true); QuickCallFunction(R0, &CoreTiming::Advance); - ApplyRoundingMode(); + ApplyRoundingMode(true); RestoreDowncount(); FixupBranch skipToRealDispatch = B(); //skip the sync and compare first time @@ -175,9 +175,9 @@ void Jit::GenerateFixedCode() // No block found, let's jit SaveDowncount(); - RestoreRoundingMode(); + RestoreRoundingMode(true); QuickCallFunction(R2, (void *)&JitAt); - ApplyRoundingMode(); + ApplyRoundingMode(true); RestoreDowncount(); B(dispatcherNoCheck); // no point in special casing this @@ -199,7 +199,7 @@ void Jit::GenerateFixedCode() } SaveDowncount(); - RestoreRoundingMode(); + RestoreRoundingMode(true); ADD(R_SP, R_SP, 4); diff --git a/Core/MIPS/ARM/ArmCompFPU.cpp b/Core/MIPS/ARM/ArmCompFPU.cpp index 770fe95ea3..279f4ea5b7 100644 --- a/Core/MIPS/ARM/ArmCompFPU.cpp +++ b/Core/MIPS/ARM/ArmCompFPU.cpp @@ -420,6 +420,7 @@ void Jit::Comp_mxc1(MIPSOpcode op) AND(gpr.R(MIPS_REG_FPCOND), SCRATCHREG1, Operand2(1)); #endif } + UpdateRoundingMode(); ApplyRoundingMode(); } else { Comp_Generic(op); diff --git a/Core/MIPS/ARM/ArmJit.cpp b/Core/MIPS/ARM/ArmJit.cpp index 7a422cf5f6..68f347edbd 100644 --- a/Core/MIPS/ARM/ArmJit.cpp +++ b/Core/MIPS/ARM/ArmJit.cpp @@ -94,22 +94,32 @@ Jit::Jit(MIPSState *mips) : blocks(mips, this), gpr(mips, &jo), fpr(mips), mips_ void Jit::DoState(PointerWrap &p) { - auto s = p.Section("Jit", 1); + auto s = p.Section("Jit", 1, 2); if (!s) return; p.Do(js.startDefaultPrefix); + if (s >= 2) { + p.Do(js.hasSetRounding); + js.lastSetRounding = 0; + } else { + js.hasSetRounding = 1; + } } // This is here so the savestate matches between jit and non-jit. void Jit::DoDummyState(PointerWrap &p) { - auto s = p.Section("Jit", 1); + auto s = p.Section("Jit", 1, 2); if (!s) return; bool dummy = false; p.Do(dummy); + if (s >= 2) { + dummy = true; + p.Do(dummy); + } } void Jit::FlushAll() @@ -201,17 +211,28 @@ void Jit::Compile(u32 em_address) { DoJit(em_address, b); blocks.FinalizeBlock(block_num, jo.enableBlocklink); + bool cleanSlate = false; + + if (js.hasSetRounding && !js.lastSetRounding) { + WARN_LOG(JIT, "Detected rounding mode usage, rebuilding jit with checks"); + // Won't loop, since hasSetRounding is only ever set to 1. + js.lastSetRounding = js.hasSetRounding; + cleanSlate = true; + } + // Drat. The VFPU hit an uneaten prefix at the end of a block. if (js.startDefaultPrefix && js.MayHavePrefix()) { WARN_LOG(JIT, "An uneaten prefix at end of block: %08x", js.compilerPC - 4); js.LogPrefix(); + // Let's try that one more time. We won't get back here because we toggled the value. js.startDefaultPrefix = false; + cleanSlate = true; + } + if (cleanSlate) { // Our assumptions are all wrong so it's clean-slate time. ClearCache(); - - // Let's try that one more time. We won't get back here because we toggled the value. Compile(em_address); } } @@ -547,8 +568,9 @@ void Jit::WriteDownCountR(ARMReg reg) { } } -void Jit::RestoreRoundingMode() { - if (g_Config.bSetRoundingMode) { +void Jit::RestoreRoundingMode(bool force) { + // If the game has never set an interesting rounding mode, we can safely skip this. + if (g_Config.bSetRoundingMode && (force || !g_Config.bForceFlushToZero || js.hasSetRounding)) { VMRS(SCRATCHREG2); // Assume we're always in round-to-nearest mode beforehand. // Also on ARM, we're always in flush-to-zero in C++, so stay that way. @@ -560,9 +582,10 @@ void Jit::RestoreRoundingMode() { } } -void Jit::ApplyRoundingMode() { +void Jit::ApplyRoundingMode(bool force) { // NOTE: Must not destory R0. - if (g_Config.bSetRoundingMode) { + // If the game has never set an interesting rounding mode, we can safely skip this. + if (g_Config.bSetRoundingMode && (force || !g_Config.bForceFlushToZero || js.hasSetRounding)) { LDR(SCRATCHREG2, CTXREG, offsetof(MIPSState, fcr31)); if (!g_Config.bForceFlushToZero) { TST(SCRATCHREG2, AssumeMakeOperand2(1 << 24)); @@ -609,6 +632,32 @@ void Jit::ApplyRoundingMode() { } } +void Jit::UpdateRoundingMode() { + // NOTE: Must not destory R0. + if (g_Config.bSetRoundingMode) { + LDR(SCRATCHREG2, CTXREG, offsetof(MIPSState, fcr31)); + if (!g_Config.bForceFlushToZero) { + TST(SCRATCHREG2, AssumeMakeOperand2(1 << 24)); + AND(SCRATCHREG2, SCRATCHREG2, Operand2(3)); + SetCC(CC_NEQ); + ADD(SCRATCHREG2, SCRATCHREG2, Operand2(4)); + SetCC(CC_AL); + // We can only skip if the rounding mode is zero and flush is set. + CMP(SCRATCHREG2, Operand2(4)); + } else { + ANDS(SCRATCHREG2, SCRATCHREG2, Operand2(3)); + } + + FixupBranch skip = B_CC(CC_EQ); + PUSH(1, SCRATCHREG1); + MOVI2R(SCRATCHREG2, 1); + MOVP2R(SCRATCHREG1, &js.hasSetRounding); + STRB(SCRATCHREG2, SCRATCHREG1, 0); + POP(1, SCRATCHREG1); + SetJumpTarget(skip); + } +} + // IDEA - could have a WriteDualExit that takes two destinations and two condition flags, // and just have conditional that set PC "twice". This only works when we fall back to dispatcher // though, as we need to have the SUBS flag set in the end. So with block linking in the mix, diff --git a/Core/MIPS/ARM/ArmJit.h b/Core/MIPS/ARM/ArmJit.h index d5cb448e35..95ba4ec867 100644 --- a/Core/MIPS/ARM/ArmJit.h +++ b/Core/MIPS/ARM/ArmJit.h @@ -192,8 +192,9 @@ private: void WriteDownCount(int offset = 0); void WriteDownCountR(ARMReg reg); - void RestoreRoundingMode(); - void ApplyRoundingMode(); + void RestoreRoundingMode(bool force = false); + void ApplyRoundingMode(bool force = false); + void UpdateRoundingMode(); void MovFromPC(ARMReg r); void MovToPC(ARMReg r); diff --git a/Core/MIPS/JitCommon/JitState.h b/Core/MIPS/JitCommon/JitState.h index ab50648775..c77481499d 100644 --- a/Core/MIPS/JitCommon/JitState.h +++ b/Core/MIPS/JitCommon/JitState.h @@ -55,7 +55,9 @@ namespace MIPSComp { }; JitState() - : startDefaultPrefix(true), + : hasSetRounding(0), + lastSetRounding(0), + startDefaultPrefix(true), prefixSFlag(PREFIX_UNKNOWN), prefixTFlag(PREFIX_UNKNOWN), prefixDFlag(PREFIX_UNKNOWN) {} @@ -72,6 +74,9 @@ namespace MIPSComp { bool compiling; // TODO: get rid of this in favor of using analysis results to determine end of block JitBlock *curBlock; + u8 hasSetRounding; + u8 lastSetRounding; + // VFPU prefix magic bool startDefaultPrefix; u32 prefixS; diff --git a/Core/MIPS/x86/Asm.cpp b/Core/MIPS/x86/Asm.cpp index dc25034e34..12fd1da530 100644 --- a/Core/MIPS/x86/Asm.cpp +++ b/Core/MIPS/x86/Asm.cpp @@ -77,9 +77,9 @@ void AsmRoutineManager::Generate(MIPSState *mips, MIPSComp::Jit *jit) #endif outerLoop = GetCodePtr(); - jit->RestoreRoundingMode(this); + jit->RestoreRoundingMode(true, this); ABI_CallFunction(reinterpret_cast(&CoreTiming::Advance)); - jit->ApplyRoundingMode(this); + jit->ApplyRoundingMode(true, this); FixupBranch skipToRealDispatch = J(); //skip the sync and compare first time dispatcherCheckCoreState = GetCodePtr(); @@ -134,9 +134,9 @@ void AsmRoutineManager::Generate(MIPSState *mips, MIPSComp::Jit *jit) SetJumpTarget(notfound); //Ok, no block, let's jit - jit->RestoreRoundingMode(this); + jit->RestoreRoundingMode(true, this); ABI_CallFunction(&Jit); - jit->ApplyRoundingMode(this); + jit->ApplyRoundingMode(true, this); JMP(dispatcherNoCheck, true); // Let's just dispatch again, we'll enter the block since we know it's there. SetJumpTarget(bail); @@ -146,12 +146,12 @@ void AsmRoutineManager::Generate(MIPSState *mips, MIPSComp::Jit *jit) J_CC(CC_Z, outerLoop, true); SetJumpTarget(badCoreState); - jit->RestoreRoundingMode(this); + jit->RestoreRoundingMode(true, this); ABI_PopAllCalleeSavedRegsAndAdjustStack(); RET(); breakpointBailout = GetCodePtr(); - jit->RestoreRoundingMode(this); + jit->RestoreRoundingMode(true, this); ABI_PopAllCalleeSavedRegsAndAdjustStack(); RET(); } diff --git a/Core/MIPS/x86/CompFPU.cpp b/Core/MIPS/x86/CompFPU.cpp index 01561d12af..4cb788cf31 100644 --- a/Core/MIPS/x86/CompFPU.cpp +++ b/Core/MIPS/x86/CompFPU.cpp @@ -387,6 +387,7 @@ void Jit::Comp_mxc1(MIPSOpcode op) if ((gpr.GetImm(rt) & 0x1000003) == 0) { // Default nearest / no-flush mode, just leave it cleared. } else { + UpdateRoundingMode(); ApplyRoundingMode(); } } else { @@ -399,6 +400,7 @@ void Jit::Comp_mxc1(MIPSOpcode op) MOV(32, M(&mips_->fcr31), gpr.R(rt)); AND(32, M(&mips_->fcr31), Imm32(0x0181FFFF)); gpr.UnlockAll(); + UpdateRoundingMode(); ApplyRoundingMode(); } } else { diff --git a/Core/MIPS/x86/Jit.cpp b/Core/MIPS/x86/Jit.cpp index 34835962a1..4b3ecf8124 100644 --- a/Core/MIPS/x86/Jit.cpp +++ b/Core/MIPS/x86/Jit.cpp @@ -145,22 +145,32 @@ Jit::~Jit() { void Jit::DoState(PointerWrap &p) { - auto s = p.Section("Jit", 1); + auto s = p.Section("Jit", 1, 2); if (!s) return; p.Do(js.startDefaultPrefix); + if (s >= 2) { + p.Do(js.hasSetRounding); + js.lastSetRounding = 0; + } else { + js.hasSetRounding = 1; + } } // This is here so the savestate matches between jit and non-jit. void Jit::DoDummyState(PointerWrap &p) { - auto s = p.Section("Jit", 1); + auto s = p.Section("Jit", 1, 2); if (!s) return; bool dummy = false; p.Do(dummy); + if (s >= 2) { + dummy = true; + p.Do(dummy); + } } @@ -211,9 +221,10 @@ void Jit::WriteDowncount(int offset) SUB(32, M(¤tMIPS->downcount), downcount > 127 ? Imm32(downcount) : Imm8(downcount)); } -void Jit::RestoreRoundingMode(XEmitter *emitter) +void Jit::RestoreRoundingMode(bool force, XEmitter *emitter) { - if (g_Config.bSetRoundingMode) + // If the game has never set an interesting rounding mode, we can safely skip this. + if (g_Config.bSetRoundingMode && (force || g_Config.bForceFlushToZero || js.hasSetRounding)) { if (emitter == NULL) emitter = this; @@ -224,9 +235,10 @@ void Jit::RestoreRoundingMode(XEmitter *emitter) } } -void Jit::ApplyRoundingMode(XEmitter *emitter) +void Jit::ApplyRoundingMode(bool force, XEmitter *emitter) { - if (g_Config.bSetRoundingMode) + // If the game has never set an interesting rounding mode, we can safely skip this. + if (g_Config.bSetRoundingMode && (force || g_Config.bForceFlushToZero || js.hasSetRounding)) { if (emitter == NULL) emitter = this; @@ -265,6 +277,22 @@ void Jit::ApplyRoundingMode(XEmitter *emitter) } } +void Jit::UpdateRoundingMode(XEmitter *emitter) +{ + if (g_Config.bSetRoundingMode) + { + if (emitter == NULL) + emitter = this; + + // If it's only ever 0, we don't actually bother applying or restoring it. + // This is the most common situation. + emitter->TEST(32, M(&mips_->fcr31), Imm32(0x01000003)); + FixupBranch skip = emitter->J_CC(CC_Z); + emitter->MOV(8, M(&js.hasSetRounding), Imm8(1)); + emitter->SetJumpTarget(skip); + } +} + void Jit::ClearCache() { blocks.Clear(); @@ -330,14 +358,28 @@ void Jit::Compile(u32 em_address) DoJit(em_address, b); blocks.FinalizeBlock(block_num, jo.enableBlocklink); + bool cleanSlate = false; + + if (js.hasSetRounding && !js.lastSetRounding) { + WARN_LOG(JIT, "Detected rounding mode usage, rebuilding jit with checks"); + // Won't loop, since hasSetRounding is only ever set to 1. + js.lastSetRounding = js.hasSetRounding; + cleanSlate = true; + } + // Drat. The VFPU hit an uneaten prefix at the end of a block. if (js.startDefaultPrefix && js.MayHavePrefix()) { - WARN_LOG(JIT, "Uneaten prefix at end of block: %08x", js.compilerPC - 4); - js.startDefaultPrefix = false; - // Our assumptions are all wrong so it's clean-slate time. - ClearCache(); + WARN_LOG(JIT, "An uneaten prefix at end of block: %08x", js.compilerPC - 4); + js.LogPrefix(); // Let's try that one more time. We won't get back here because we toggled the value. + js.startDefaultPrefix = false; + cleanSlate = true; + } + + if (cleanSlate) { + // Our assumptions are all wrong so it's clean-slate time. + ClearCache(); Compile(em_address); } } diff --git a/Core/MIPS/x86/Jit.h b/Core/MIPS/x86/Jit.h index 9e58ca6580..c1b088e9b8 100644 --- a/Core/MIPS/x86/Jit.h +++ b/Core/MIPS/x86/Jit.h @@ -165,8 +165,9 @@ public: void GetVectorRegsPrefixD(u8 *regs, VectorSize sz, int vectorReg); void EatPrefix() { js.EatPrefix(); } - void RestoreRoundingMode(XEmitter *emitter = NULL); - void ApplyRoundingMode(XEmitter *emitter = NULL); + void RestoreRoundingMode(bool force = false, XEmitter *emitter = NULL); + void ApplyRoundingMode(bool force = false, XEmitter *emitter = NULL); + void UpdateRoundingMode(XEmitter *emitter = NULL); JitBlockCache *GetBlockCache() { return &blocks; } AsmRoutineManager &Asm() { return asm_; } From 4d30288601f3c9da5589b8c24a441fcadac6cd96 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Sun, 12 Oct 2014 11:37:27 -0700 Subject: [PATCH 087/105] x86jit: Fix force flush to zero. --- Core/MIPS/x86/Jit.cpp | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/Core/MIPS/x86/Jit.cpp b/Core/MIPS/x86/Jit.cpp index 4b3ecf8124..cc96264062 100644 --- a/Core/MIPS/x86/Jit.cpp +++ b/Core/MIPS/x86/Jit.cpp @@ -248,7 +248,9 @@ void Jit::ApplyRoundingMode(bool force, XEmitter *emitter) // If it's 0, we don't actually bother setting. This is the most common. // We always use nearest as the default rounding mode with // flush-to-zero disabled. - FixupBranch skip = emitter->J_CC(CC_Z); + FixupBranch skip; + if (!g_Config.bForceFlushToZero) + skip = emitter->J_CC(CC_Z); emitter->STMXCSR(M(¤tMIPS->temp)); @@ -273,7 +275,8 @@ void Jit::ApplyRoundingMode(bool force, XEmitter *emitter) emitter->LDMXCSR(M(¤tMIPS->temp)); - emitter->SetJumpTarget(skip); + if (!g_Config.bForceFlushToZero) + emitter->SetJumpTarget(skip); } } From 9228ac72da87bcd24b239429dad41f2d11fc2d07 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Sun, 12 Oct 2014 12:00:59 -0700 Subject: [PATCH 088/105] jit: Reorganize imm branch logic a bit. --- Core/MIPS/ARM/ArmCompBranch.cpp | 42 ++++++++++++++++++++------------- Core/MIPS/x86/CompBranch.cpp | 40 ++++++++++++++++++------------- 2 files changed, 50 insertions(+), 32 deletions(-) diff --git a/Core/MIPS/ARM/ArmCompBranch.cpp b/Core/MIPS/ARM/ArmCompBranch.cpp index 24d104d644..2bae11047f 100644 --- a/Core/MIPS/ARM/ArmCompBranch.cpp +++ b/Core/MIPS/ARM/ArmCompBranch.cpp @@ -66,19 +66,24 @@ void Jit::BranchRSRTComp(MIPSOpcode op, ArmGen::CCFlags cc, bool likely) MIPSGPReg rs = _RS; u32 targetAddr = js.compilerPC + offset + 4; - if (jo.immBranches && gpr.IsImm(rs) && gpr.IsImm(rt) && js.numInstructions < jo.continueMaxInstructions) { + bool immBranch = false; + bool immBranchNotTaken = false; + if (gpr.IsImm(rs) && gpr.IsImm(rt)) { // The cc flags are opposites: when NOT to take the branch. - bool skipBranch; s32 rsImm = (s32)gpr.GetImm(rs); s32 rtImm = (s32)gpr.GetImm(rt); - switch (cc) { - case CC_EQ: skipBranch = rsImm == rtImm; break; - case CC_NEQ: skipBranch = rsImm != rtImm; break; - default: skipBranch = false; _dbg_assert_msg_(JIT, false, "Bad cc flag in BranchRSRTComp()."); + switch (cc) + { + case CC_EQ: immBranchNotTaken = rsImm == rtImm; break; + case CC_NEQ: immBranchNotTaken = rsImm != rtImm; break; + default: immBranchNotTaken = false; _dbg_assert_msg_(JIT, false, "Bad cc flag in BranchRSRTComp()."); } + immBranch = true; + } - if (skipBranch) { + if (jo.immBranches && immBranch && js.numInstructions < jo.continueMaxInstructions) { + if (immBranchNotTaken) { // Skip the delay slot if likely, otherwise it'll be the next instruction. if (likely) js.compilerPC += 4; @@ -158,20 +163,25 @@ void Jit::BranchRSZeroComp(MIPSOpcode op, ArmGen::CCFlags cc, bool andLink, bool MIPSGPReg rs = _RS; u32 targetAddr = js.compilerPC + offset + 4; - if (jo.immBranches && gpr.IsImm(rs) && js.numInstructions < jo.continueMaxInstructions) { + bool immBranch = false; + bool immBranchNotTaken = false; + if (gpr.IsImm(rs)) { // The cc flags are opposites: when NOT to take the branch. - bool skipBranch; s32 imm = (s32)gpr.GetImm(rs); - switch (cc) { - case CC_GT: skipBranch = imm > 0; break; - case CC_GE: skipBranch = imm >= 0; break; - case CC_LT: skipBranch = imm < 0; break; - case CC_LE: skipBranch = imm <= 0; break; - default: skipBranch = false; _dbg_assert_msg_(JIT, false, "Bad cc flag in BranchRSZeroComp()."); + switch (cc) + { + case CC_GT: immBranchNotTaken = imm > 0; break; + case CC_GE: immBranchNotTaken = imm >= 0; break; + case CC_LT: immBranchNotTaken = imm < 0; break; + case CC_LE: immBranchNotTaken = imm <= 0; break; + default: immBranchNotTaken = false; _dbg_assert_msg_(JIT, false, "Bad cc flag in BranchRSZeroComp()."); } + immBranch = true; + } - if (skipBranch) { + if (jo.immBranches && immBranch && js.numInstructions < jo.continueMaxInstructions) { + if (immBranchNotTaken) { // Skip the delay slot if likely, otherwise it'll be the next instruction. if (likely) js.compilerPC += 4; diff --git a/Core/MIPS/x86/CompBranch.cpp b/Core/MIPS/x86/CompBranch.cpp index 5c04a69c66..6003bcca64 100644 --- a/Core/MIPS/x86/CompBranch.cpp +++ b/Core/MIPS/x86/CompBranch.cpp @@ -278,21 +278,25 @@ void Jit::BranchRSRTComp(MIPSOpcode op, Gen::CCFlags cc, bool likely) MIPSGPReg rs = _RS; u32 targetAddr = js.compilerPC + offset + 4; - if (jo.immBranches && gpr.IsImm(rs) && gpr.IsImm(rt) && js.numInstructions < jo.continueMaxInstructions) - { + bool immBranch = false; + bool immBranchNotTaken = false; + if (gpr.IsImm(rs) && gpr.IsImm(rt)) { // The cc flags are opposites: when NOT to take the branch. - bool skipBranch; s32 rsImm = (s32)gpr.GetImm(rs); s32 rtImm = (s32)gpr.GetImm(rt); switch (cc) { - case CC_E: skipBranch = rsImm == rtImm; break; - case CC_NE: skipBranch = rsImm != rtImm; break; - default: skipBranch = false; _dbg_assert_msg_(JIT, false, "Bad cc flag in BranchRSRTComp()."); + case CC_E: immBranchNotTaken = rsImm == rtImm; break; + case CC_NE: immBranchNotTaken = rsImm != rtImm; break; + default: immBranchNotTaken = false; _dbg_assert_msg_(JIT, false, "Bad cc flag in BranchRSRTComp()."); } + immBranch = true; + } - if (skipBranch) + if (jo.immBranches && immBranch && js.numInstructions < jo.continueMaxInstructions) + { + if (immBranchNotTaken) { // Skip the delay slot if likely, otherwise it'll be the next instruction. if (likely) @@ -340,22 +344,26 @@ void Jit::BranchRSZeroComp(MIPSOpcode op, Gen::CCFlags cc, bool andLink, bool li MIPSGPReg rs = _RS; u32 targetAddr = js.compilerPC + offset + 4; - if (jo.immBranches && gpr.IsImm(rs) && js.numInstructions < jo.continueMaxInstructions) - { + bool immBranch = false; + bool immBranchNotTaken = false; + if (gpr.IsImm(rs)) { // The cc flags are opposites: when NOT to take the branch. - bool skipBranch; s32 imm = (s32)gpr.GetImm(rs); switch (cc) { - case CC_G: skipBranch = imm > 0; break; - case CC_GE: skipBranch = imm >= 0; break; - case CC_L: skipBranch = imm < 0; break; - case CC_LE: skipBranch = imm <= 0; break; - default: skipBranch = false; _dbg_assert_msg_(JIT, false, "Bad cc flag in BranchRSZeroComp()."); + case CC_G: immBranchNotTaken = imm > 0; break; + case CC_GE: immBranchNotTaken = imm >= 0; break; + case CC_L: immBranchNotTaken = imm < 0; break; + case CC_LE: immBranchNotTaken = imm <= 0; break; + default: immBranchNotTaken = false; _dbg_assert_msg_(JIT, false, "Bad cc flag in BranchRSZeroComp()."); } + immBranch = true; + } - if (skipBranch) + if (jo.immBranches && immBranch && js.numInstructions < jo.continueMaxInstructions) + { + if (immBranchNotTaken) { // Skip the delay slot if likely, otherwise it'll be the next instruction. if (likely) From 2f598e8f3837581eaec4799aa84099c90f8eaa8e Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Sun, 12 Oct 2014 12:37:54 -0700 Subject: [PATCH 089/105] jit: Statically jump for fixed branches. This handles both loops (first step is known) and static branches (some code uses them instead of jumps, and we disassemble that to "b".) Not likely to be a big improvement, but might help if the branch predictor was wrong. This is as opposed to continuing, which would build a larger jit block. --- Core/MIPS/ARM/ArmCompBranch.cpp | 178 +++++++++++++++++++------------- Core/MIPS/x86/CompBranch.cpp | 71 +++++++++---- Core/MIPS/x86/Jit.h | 1 + 3 files changed, 156 insertions(+), 94 deletions(-) diff --git a/Core/MIPS/ARM/ArmCompBranch.cpp b/Core/MIPS/ARM/ArmCompBranch.cpp index 2bae11047f..0f6de65645 100644 --- a/Core/MIPS/ARM/ArmCompBranch.cpp +++ b/Core/MIPS/ARM/ArmCompBranch.cpp @@ -67,9 +67,10 @@ void Jit::BranchRSRTComp(MIPSOpcode op, ArmGen::CCFlags cc, bool likely) u32 targetAddr = js.compilerPC + offset + 4; bool immBranch = false; - bool immBranchNotTaken = false; + bool immBranchTaken = false; if (gpr.IsImm(rs) && gpr.IsImm(rt)) { // The cc flags are opposites: when NOT to take the branch. + bool immBranchNotTaken; s32 rsImm = (s32)gpr.GetImm(rs); s32 rtImm = (s32)gpr.GetImm(rt); @@ -80,10 +81,11 @@ void Jit::BranchRSRTComp(MIPSOpcode op, ArmGen::CCFlags cc, bool likely) default: immBranchNotTaken = false; _dbg_assert_msg_(JIT, false, "Bad cc flag in BranchRSRTComp()."); } immBranch = true; + immBranchTaken = !immBranchNotTaken; } if (jo.immBranches && immBranch && js.numInstructions < jo.continueMaxInstructions) { - if (immBranchNotTaken) { + if (!immBranchTaken) { // Skip the delay slot if likely, otherwise it'll be the next instruction. if (likely) js.compilerPC += 4; @@ -102,53 +104,65 @@ void Jit::BranchRSRTComp(MIPSOpcode op, ArmGen::CCFlags cc, bool likely) MIPSOpcode delaySlotOp = Memory::Read_Instruction(js.compilerPC+4); bool delaySlotIsNice = IsDelaySlotNiceReg(op, delaySlotOp, rt, rs); CONDITIONAL_NICE_DELAYSLOT; - if (!likely && delaySlotIsNice) - CompileDelaySlot(DELAYSLOT_NICE); - // We might be able to flip the condition (EQ/NEQ are easy.) - const bool canFlip = cc == CC_EQ || cc == CC_NEQ; - - Operand2 op2; - bool negated; - if (gpr.IsImm(rt) && TryMakeOperand2_AllowNegation(gpr.GetImm(rt), op2, &negated)) { - gpr.MapReg(rs); - if (!negated) - CMP(gpr.R(rs), op2); - else - CMN(gpr.R(rs), op2); - } else { - if (gpr.IsImm(rs) && TryMakeOperand2_AllowNegation(gpr.GetImm(rs), op2, &negated) && canFlip) { - gpr.MapReg(rt); - if (!negated) - CMP(gpr.R(rt), op2); - else - CMN(gpr.R(rt), op2); - } else { - gpr.MapInIn(rs, rt); - CMP(gpr.R(rs), gpr.R(rt)); - } - } - - ArmGen::FixupBranch ptr; - if (!likely) { - if (!delaySlotIsNice) - CompileDelaySlot(DELAYSLOT_SAFE_FLUSH); + if (immBranch) { + // Continuing is handled above, this is just static jumping. + if (immBranchTaken || !likely) + CompileDelaySlot(DELAYSLOT_FLUSH); else FlushAll(); - ptr = B_CC(cc); + + const u32 destAddr = immBranchTaken ? targetAddr : js.compilerPC + 8; + WriteExit(destAddr, js.nextExit++); } else { - FlushAll(); - ptr = B_CC(cc); - CompileDelaySlot(DELAYSLOT_FLUSH); + if (!likely && delaySlotIsNice) + CompileDelaySlot(DELAYSLOT_NICE); + + // We might be able to flip the condition (EQ/NEQ are easy.) + const bool canFlip = cc == CC_EQ || cc == CC_NEQ; + + Operand2 op2; + bool negated; + if (gpr.IsImm(rt) && TryMakeOperand2_AllowNegation(gpr.GetImm(rt), op2, &negated)) { + gpr.MapReg(rs); + if (!negated) + CMP(gpr.R(rs), op2); + else + CMN(gpr.R(rs), op2); + } else { + if (gpr.IsImm(rs) && TryMakeOperand2_AllowNegation(gpr.GetImm(rs), op2, &negated) && canFlip) { + gpr.MapReg(rt); + if (!negated) + CMP(gpr.R(rt), op2); + else + CMN(gpr.R(rt), op2); + } else { + gpr.MapInIn(rs, rt); + CMP(gpr.R(rs), gpr.R(rt)); + } + } + + ArmGen::FixupBranch ptr; + if (!likely) { + if (!delaySlotIsNice) + CompileDelaySlot(DELAYSLOT_SAFE_FLUSH); + else + FlushAll(); + ptr = B_CC(cc); + } else { + FlushAll(); + ptr = B_CC(cc); + CompileDelaySlot(DELAYSLOT_FLUSH); + } + + // Take the branch + WriteExit(targetAddr, js.nextExit++); + + SetJumpTarget(ptr); + // Not taken + WriteExit(js.compilerPC + 8, js.nextExit++); } - // Take the branch - WriteExit(targetAddr, js.nextExit++); - - SetJumpTarget(ptr); - // Not taken - WriteExit(js.compilerPC+8, js.nextExit++); - js.compiling = false; } @@ -164,9 +178,10 @@ void Jit::BranchRSZeroComp(MIPSOpcode op, ArmGen::CCFlags cc, bool andLink, bool u32 targetAddr = js.compilerPC + offset + 4; bool immBranch = false; - bool immBranchNotTaken = false; + bool immBranchTaken = false; if (gpr.IsImm(rs)) { // The cc flags are opposites: when NOT to take the branch. + bool immBranchNotTaken; s32 imm = (s32)gpr.GetImm(rs); switch (cc) @@ -178,10 +193,11 @@ void Jit::BranchRSZeroComp(MIPSOpcode op, ArmGen::CCFlags cc, bool andLink, bool default: immBranchNotTaken = false; _dbg_assert_msg_(JIT, false, "Bad cc flag in BranchRSZeroComp()."); } immBranch = true; + immBranchTaken = !immBranchNotTaken; } if (jo.immBranches && immBranch && js.numInstructions < jo.continueMaxInstructions) { - if (immBranchNotTaken) { + if (!immBranchTaken) { // Skip the delay slot if likely, otherwise it'll be the next instruction. if (likely) js.compilerPC += 4; @@ -203,40 +219,54 @@ void Jit::BranchRSZeroComp(MIPSOpcode op, ArmGen::CCFlags cc, bool andLink, bool MIPSOpcode delaySlotOp = Memory::Read_Instruction(js.compilerPC + 4); bool delaySlotIsNice = IsDelaySlotNiceReg(op, delaySlotOp, rs); CONDITIONAL_NICE_DELAYSLOT; - if (!likely && delaySlotIsNice) - CompileDelaySlot(DELAYSLOT_NICE); - gpr.MapReg(rs); - CMP(gpr.R(rs), Operand2(0, TYPE_IMM)); - - ArmGen::FixupBranch ptr; - if (!likely) - { - if (!delaySlotIsNice) - CompileDelaySlot(DELAYSLOT_SAFE_FLUSH); + if (immBranch) { + // Continuing is handled above, this is just static jumping. + if (immBranchTaken && andLink) + gpr.SetImm(MIPS_REG_RA, js.compilerPC + 8); + if (immBranchTaken || !likely) + CompileDelaySlot(DELAYSLOT_FLUSH); else FlushAll(); - ptr = B_CC(cc); - } - else - { - FlushAll(); - ptr = B_CC(cc); - CompileDelaySlot(DELAYSLOT_FLUSH); - } - // Take the branch - if (andLink) - { - gpr.SetRegImm(SCRATCHREG1, js.compilerPC + 8); - STR(SCRATCHREG1, CTXREG, MIPS_REG_RA * 4); + const u32 destAddr = immBranchTaken ? targetAddr : js.compilerPC + 8; + WriteExit(destAddr, js.nextExit++); + } else { + if (!likely && delaySlotIsNice) + CompileDelaySlot(DELAYSLOT_NICE); + + gpr.MapReg(rs); + CMP(gpr.R(rs), Operand2(0, TYPE_IMM)); + + ArmGen::FixupBranch ptr; + if (!likely) + { + if (!delaySlotIsNice) + CompileDelaySlot(DELAYSLOT_SAFE_FLUSH); + else + FlushAll(); + ptr = B_CC(cc); + } + else + { + FlushAll(); + ptr = B_CC(cc); + CompileDelaySlot(DELAYSLOT_FLUSH); + } + + // Take the branch + if (andLink) + { + gpr.SetRegImm(SCRATCHREG1, js.compilerPC + 8); + STR(SCRATCHREG1, CTXREG, MIPS_REG_RA * 4); + } + + WriteExit(targetAddr, js.nextExit++); + + SetJumpTarget(ptr); + // Not taken + WriteExit(js.compilerPC + 8, js.nextExit++); } - - WriteExit(targetAddr, js.nextExit++); - - SetJumpTarget(ptr); - // Not taken - WriteExit(js.compilerPC + 8, js.nextExit++); js.compiling = false; } diff --git a/Core/MIPS/x86/CompBranch.cpp b/Core/MIPS/x86/CompBranch.cpp index 6003bcca64..6f7d4444f4 100644 --- a/Core/MIPS/x86/CompBranch.cpp +++ b/Core/MIPS/x86/CompBranch.cpp @@ -266,6 +266,21 @@ void Jit::CompBranchExits(CCFlags cc, u32 targetAddr, u32 notTakenAddr, bool del } } +void Jit::CompBranchExit(bool taken, u32 targetAddr, u32 notTakenAddr, bool delaySlotIsNice, bool likely, bool andLink) { + // Continuing is handled in the imm branch case... TODO: move it here? + if (taken && andLink) + gpr.SetImm(MIPS_REG_RA, js.compilerPC + 8); + if (taken || !likely) + CompileDelaySlot(DELAYSLOT_FLUSH); + else + FlushAll(); + + const u32 destAddr = taken ? targetAddr : notTakenAddr; + CONDITIONAL_LOG_EXIT(destAddr); + WriteExit(destAddr, js.nextExit++); + js.compiling = false; +} + void Jit::BranchRSRTComp(MIPSOpcode op, Gen::CCFlags cc, bool likely) { CONDITIONAL_LOG; @@ -279,9 +294,10 @@ void Jit::BranchRSRTComp(MIPSOpcode op, Gen::CCFlags cc, bool likely) u32 targetAddr = js.compilerPC + offset + 4; bool immBranch = false; - bool immBranchNotTaken = false; + bool immBranchTaken = false; if (gpr.IsImm(rs) && gpr.IsImm(rt)) { // The cc flags are opposites: when NOT to take the branch. + bool immBranchNotTaken; s32 rsImm = (s32)gpr.GetImm(rs); s32 rtImm = (s32)gpr.GetImm(rt); @@ -292,11 +308,12 @@ void Jit::BranchRSRTComp(MIPSOpcode op, Gen::CCFlags cc, bool likely) default: immBranchNotTaken = false; _dbg_assert_msg_(JIT, false, "Bad cc flag in BranchRSRTComp()."); } immBranch = true; + immBranchTaken = !immBranchNotTaken; } if (jo.immBranches && immBranch && js.numInstructions < jo.continueMaxInstructions) { - if (immBranchNotTaken) + if (!immBranchTaken) { // Skip the delay slot if likely, otherwise it'll be the next instruction. if (likely) @@ -316,21 +333,27 @@ void Jit::BranchRSRTComp(MIPSOpcode op, Gen::CCFlags cc, bool likely) MIPSOpcode delaySlotOp = Memory::Read_Instruction(js.compilerPC+4); bool delaySlotIsNice = IsDelaySlotNiceReg(op, delaySlotOp, rt, rs); CONDITIONAL_NICE_DELAYSLOT; - if (!likely && delaySlotIsNice) - CompileDelaySlot(DELAYSLOT_NICE); - if (gpr.IsImm(rt) && gpr.GetImm(rt) == 0) - { - gpr.KillImmediate(rs, true, false); - CMP(32, gpr.R(rs), Imm32(0)); - } + if (immBranch) + CompBranchExit(immBranchTaken, targetAddr, js.compilerPC + 8, delaySlotIsNice, likely, false); else { - gpr.MapReg(rs, true, false); - CMP(32, gpr.R(rs), gpr.R(rt)); - } + if (!likely && delaySlotIsNice) + CompileDelaySlot(DELAYSLOT_NICE); - CompBranchExits(cc, targetAddr, js.compilerPC + 8, delaySlotIsNice, likely, false); + if (gpr.IsImm(rt) && gpr.GetImm(rt) == 0) + { + gpr.KillImmediate(rs, true, false); + CMP(32, gpr.R(rs), Imm32(0)); + } + else + { + gpr.MapReg(rs, true, false); + CMP(32, gpr.R(rs), gpr.R(rt)); + } + + CompBranchExits(cc, targetAddr, js.compilerPC + 8, delaySlotIsNice, likely, false); + } } void Jit::BranchRSZeroComp(MIPSOpcode op, Gen::CCFlags cc, bool andLink, bool likely) @@ -345,9 +368,10 @@ void Jit::BranchRSZeroComp(MIPSOpcode op, Gen::CCFlags cc, bool andLink, bool li u32 targetAddr = js.compilerPC + offset + 4; bool immBranch = false; - bool immBranchNotTaken = false; + bool immBranchTaken = false; if (gpr.IsImm(rs)) { // The cc flags are opposites: when NOT to take the branch. + bool immBranchNotTaken; s32 imm = (s32)gpr.GetImm(rs); switch (cc) @@ -359,11 +383,12 @@ void Jit::BranchRSZeroComp(MIPSOpcode op, Gen::CCFlags cc, bool andLink, bool li default: immBranchNotTaken = false; _dbg_assert_msg_(JIT, false, "Bad cc flag in BranchRSZeroComp()."); } immBranch = true; + immBranchTaken = !immBranchNotTaken; } if (jo.immBranches && immBranch && js.numInstructions < jo.continueMaxInstructions) { - if (immBranchNotTaken) + if (!immBranchTaken) { // Skip the delay slot if likely, otherwise it'll be the next instruction. if (likely) @@ -386,13 +411,19 @@ void Jit::BranchRSZeroComp(MIPSOpcode op, Gen::CCFlags cc, bool andLink, bool li MIPSOpcode delaySlotOp = Memory::Read_Instruction(js.compilerPC + 4); bool delaySlotIsNice = IsDelaySlotNiceReg(op, delaySlotOp, rs); CONDITIONAL_NICE_DELAYSLOT; - if (!likely && delaySlotIsNice) - CompileDelaySlot(DELAYSLOT_NICE); - gpr.MapReg(rs, true, false); - CMP(32, gpr.R(rs), Imm32(0)); + if (immBranch) + CompBranchExit(immBranchTaken, targetAddr, js.compilerPC + 8, delaySlotIsNice, likely, false); + else + { + if (!likely && delaySlotIsNice) + CompileDelaySlot(DELAYSLOT_NICE); - CompBranchExits(cc, targetAddr, js.compilerPC + 8, delaySlotIsNice, likely, andLink); + gpr.MapReg(rs, true, false); + CMP(32, gpr.R(rs), Imm32(0)); + + CompBranchExits(cc, targetAddr, js.compilerPC + 8, delaySlotIsNice, likely, andLink); + } } diff --git a/Core/MIPS/x86/Jit.h b/Core/MIPS/x86/Jit.h index c1b088e9b8..bc3e7f0f0b 100644 --- a/Core/MIPS/x86/Jit.h +++ b/Core/MIPS/x86/Jit.h @@ -228,6 +228,7 @@ private: void CompITypeMemUnpairedLR(MIPSOpcode op, bool isStore); void CompITypeMemUnpairedLRInner(MIPSOpcode op, X64Reg shiftReg); void CompBranchExits(CCFlags cc, u32 targetAddr, u32 notTakenAddr, bool delaySlotIsNice, bool likely, bool andLink); + void CompBranchExit(bool taken, u32 targetAddr, u32 notTakenAddr, bool delaySlotIsNice, bool likely, bool andLink); void CompFPTriArith(MIPSOpcode op, void (XEmitter::*arith)(X64Reg reg, OpArg), bool orderMatters); void CompFPComp(int lhs, int rhs, u8 compare, bool allowNaN = false); From 6fae78cd3f4634f7f4dfa8e91ff69bd513f8b336 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Sun, 12 Oct 2014 12:44:51 -0700 Subject: [PATCH 090/105] x86jit: Fix a bug in branch continuing. When we predict it won't take a likely delay slot, we'd lose our register allocation state. --- Core/MIPS/x86/CompBranch.cpp | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/Core/MIPS/x86/CompBranch.cpp b/Core/MIPS/x86/CompBranch.cpp index 6f7d4444f4..eaaa69f384 100644 --- a/Core/MIPS/x86/CompBranch.cpp +++ b/Core/MIPS/x86/CompBranch.cpp @@ -189,7 +189,12 @@ void Jit::CompBranchExits(CCFlags cc, u32 targetAddr, u32 notTakenAddr, bool del if (predictTakeBranch) GetStateAndFlushAll(state); else + { + // We need to get the state BEFORE the delay slot is compiled. + gpr.GetState(state.gpr); + fpr.GetState(state.fpr); CompileDelaySlot(DELAYSLOT_FLUSH); + } } if (predictTakeBranch) From 0f45c3516d09366180902c325b04b0df2ea7778a Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Sun, 12 Oct 2014 12:47:57 -0700 Subject: [PATCH 091/105] Skip setting a0 in the idle thread. We don't need the param for our fake syscall. This is safe since it's all savestated. --- Core/HLE/sceKernelThread.cpp | 2 -- 1 file changed, 2 deletions(-) diff --git a/Core/HLE/sceKernelThread.cpp b/Core/HLE/sceKernelThread.cpp index fa22c381a2..0001f9dfa8 100644 --- a/Core/HLE/sceKernelThread.cpp +++ b/Core/HLE/sceKernelThread.cpp @@ -1148,10 +1148,8 @@ void __KernelThreadingInit() // Yeah, this is straight out of JPCSP, I should be ashamed. const static u32_le idleThreadCode[] = { - MIPS_MAKE_ADDIU(MIPS_REG_A0, MIPS_REG_ZERO, 0), MIPS_MAKE_LUI(MIPS_REG_RA, 0x0800), MIPS_MAKE_JR_RA(), - //MIPS_MAKE_SYSCALL("ThreadManForUser", "sceKernelDelayThread"), MIPS_MAKE_SYSCALL("FakeSysCalls", "_sceKernelIdle"), MIPS_MAKE_BREAK(0), }; From e3a04aa2d2578bcb59aa5d977bea1efc2a1fcf2f Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Sun, 12 Oct 2014 14:23:59 -0700 Subject: [PATCH 092/105] x86jit: Preload sp and similar regs used often. This can help us avoid using a temporary. Very tiny performance improvement. --- Core/MIPS/MIPSAnalyst.cpp | 34 +++++++++++++++++++-------------- Core/MIPS/MIPSAnalyst.h | 3 ++- Core/MIPS/x86/CompLoadStore.cpp | 2 +- Core/MIPS/x86/CompVFPU.cpp | 12 ++++++++---- Core/MIPS/x86/JitSafeMem.cpp | 6 ++++++ 5 files changed, 37 insertions(+), 20 deletions(-) diff --git a/Core/MIPS/MIPSAnalyst.cpp b/Core/MIPS/MIPSAnalyst.cpp index 9dc3ea4cee..3cf3262f40 100644 --- a/Core/MIPS/MIPSAnalyst.cpp +++ b/Core/MIPS/MIPSAnalyst.cpp @@ -661,29 +661,35 @@ namespace MIPSAnalyst { } } - // Look forwards to find if a register is used again in this block. - // Don't think we use this yet. - bool IsRegisterUsed(MIPSGPReg reg, u32 addr) { - while (true) { - MIPSOpcode op = Memory::Read_Instruction(addr, true); - MIPSInfo info = MIPSGetInfo(op); + bool IsRegisterUsed(MIPSGPReg reg, u32 addr, int instrs) { + u32 end = addr + instrs * sizeof(u32); + while (addr < end) { + const MIPSOpcode op = Memory::Read_Instruction(addr, true); + const MIPSInfo info = MIPSGetInfo(op); + + // Yes, used. if ((info & IN_RS) && (MIPS_GET_RS(op) == reg)) return true; if ((info & IN_RT) && (MIPS_GET_RT(op) == reg)) return true; - if ((info & IS_CONDBRANCH)) - return true; // could also follow both paths - if ((info & IS_JUMP)) - return true; // could also follow the path + + // Clobbered, so not used. if ((info & OUT_RT) && (MIPS_GET_RT(op) == reg)) - return false; //the reg got clobbed! yay! + return false; if ((info & OUT_RD) && (MIPS_GET_RD(op) == reg)) - return false; //the reg got clobbed! yay! + return false; if ((info & OUT_RA) && (reg == MIPS_REG_RA)) - return false; //the reg got clobbed! yay! + return false; + + // Bail early if we hit a branch (could follow each path for continuing?) + if ((info & IS_CONDBRANCH) || (info & IS_JUMP)) { + // Still need to check the delay slot (so end after it.) + // We'll assume likely are taken. + end = addr + 8; + } addr += 4; } - return true; + return false; } void HashFunctions() { diff --git a/Core/MIPS/MIPSAnalyst.h b/Core/MIPS/MIPSAnalyst.h index 4eaefbb91b..fa6188913d 100644 --- a/Core/MIPS/MIPSAnalyst.h +++ b/Core/MIPS/MIPSAnalyst.h @@ -77,7 +77,8 @@ namespace MIPSAnalyst AnalysisResults Analyze(u32 address); - bool IsRegisterUsed(MIPSGPReg reg, u32 addr); + // This tells us if the reg is used within intrs of addr (also includes likely delay slots.) + bool IsRegisterUsed(MIPSGPReg reg, u32 addr, int instrs); struct AnalyzedFunction { u32 start; diff --git a/Core/MIPS/x86/CompLoadStore.cpp b/Core/MIPS/x86/CompLoadStore.cpp index f56afb3cbf..4884c0057c 100644 --- a/Core/MIPS/x86/CompLoadStore.cpp +++ b/Core/MIPS/x86/CompLoadStore.cpp @@ -121,7 +121,7 @@ namespace MIPSComp shiftReg = R9; #endif - gpr.Lock(rt); + gpr.Lock(rt, rs); gpr.MapReg(rt, true, !isStore); // Grab the offset from alignment for shifting (<< 3 for bytes -> bits.) diff --git a/Core/MIPS/x86/CompVFPU.cpp b/Core/MIPS/x86/CompVFPU.cpp index 12359aa4ee..e60a753483 100644 --- a/Core/MIPS/x86/CompVFPU.cpp +++ b/Core/MIPS/x86/CompVFPU.cpp @@ -233,6 +233,7 @@ void Jit::Comp_SV(MIPSOpcode op) { { case 50: //lv.s // VI(vt) = Memory::Read_U32(addr); { + gpr.Lock(rs); gpr.MapReg(rs, true, false); fpr.MapRegV(vt, MAP_NOINIT); @@ -256,7 +257,8 @@ void Jit::Comp_SV(MIPSOpcode op) { case 58: //sv.s // Memory::Write_U32(VI(vt), addr); { - gpr.MapReg(rs, true, true); + gpr.Lock(rs); + gpr.MapReg(rs, true, false); // Even if we don't use real SIMD there's still 8 or 16 scalar float registers. fpr.MapRegV(vt, 0); @@ -302,7 +304,7 @@ void Jit::Comp_SVQ(MIPSOpcode op) } DISABLE; - gpr.MapReg(rs, true, true); + gpr.MapReg(rs, true, false); gpr.FlushLockX(ECX); u8 vregs[4]; GetVectorRegs(vregs, V_Quad, vt); @@ -364,7 +366,8 @@ void Jit::Comp_SVQ(MIPSOpcode op) case 54: //lv.q { - gpr.MapReg(rs, true, true); + gpr.Lock(rs); + gpr.MapReg(rs, true, false); u8 vregs[4]; GetVectorRegs(vregs, V_Quad, vt); @@ -396,7 +399,8 @@ void Jit::Comp_SVQ(MIPSOpcode op) case 62: //sv.q { - gpr.MapReg(rs, true, true); + gpr.Lock(rs); + gpr.MapReg(rs, true, false); u8 vregs[4]; GetVectorRegs(vregs, V_Quad, vt); diff --git a/Core/MIPS/x86/JitSafeMem.cpp b/Core/MIPS/x86/JitSafeMem.cpp index 086ac7a5c9..a79052d1cd 100644 --- a/Core/MIPS/x86/JitSafeMem.cpp +++ b/Core/MIPS/x86/JitSafeMem.cpp @@ -55,6 +55,12 @@ JitSafeMem::JitSafeMem(Jit *jit, MIPSGPReg raddr, s32 offset, u32 alignMask) iaddr_ = (u32) -1; fast_ = g_Config.bFastMemory || raddr == MIPS_REG_SP; + + // If raddr_ is going to get loaded soon, load it now for more optimal code. + // We assume that it was already locked. + const int LOOKAHEAD_OPS = 3; + if (!jit_->gpr.R(raddr_).IsImm() && MIPSAnalyst::IsRegisterUsed(raddr_, jit_->js.compilerPC + 4, LOOKAHEAD_OPS)) + jit_->gpr.MapReg(raddr_, true, false); } void JitSafeMem::SetFar() From 0f3210361594e0ce7bca2351683909333e8ae0e3 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Sun, 12 Oct 2014 15:16:09 -0700 Subject: [PATCH 093/105] x86jit: Consistently use mips_. --- Core/MIPS/ARM/ArmCompVFPU.cpp | 2 -- Core/MIPS/PPC/PpcCompVFPU.cpp | 4 ++-- Core/MIPS/x86/CompBranch.cpp | 2 +- Core/MIPS/x86/CompVFPU.cpp | 8 ++++---- Core/MIPS/x86/Jit.cpp | 34 +++++++++++++++++----------------- 5 files changed, 24 insertions(+), 26 deletions(-) diff --git a/Core/MIPS/ARM/ArmCompVFPU.cpp b/Core/MIPS/ARM/ArmCompVFPU.cpp index eda34c5c5b..b0ba0001b0 100644 --- a/Core/MIPS/ARM/ArmCompVFPU.cpp +++ b/Core/MIPS/ARM/ArmCompVFPU.cpp @@ -1252,8 +1252,6 @@ namespace MIPSComp gpr.MapReg(rt); STR(gpr.R(rt), CTXREG, offsetof(MIPSState, vfpuCtrl) + 4 * (imm - 128)); } - //gpr.BindToRegister(rt, true, false); - //MOV(32, M(¤tMIPS->vfpuCtrl[imm - 128]), gpr.R(rt)); // TODO: Optimization if rt is Imm? // Set these BEFORE disable! diff --git a/Core/MIPS/PPC/PpcCompVFPU.cpp b/Core/MIPS/PPC/PpcCompVFPU.cpp index 65b530e3a7..81dfdc5aa5 100644 --- a/Core/MIPS/PPC/PpcCompVFPU.cpp +++ b/Core/MIPS/PPC/PpcCompVFPU.cpp @@ -671,7 +671,7 @@ namespace MIPSComp // In case we have a saved prefix. //FlushPrefixV(); //gpr.BindToRegister(rt, false, true); - //MOV(32, gpr.R(rt), M(¤tMIPS->vfpuCtrl[imm - 128])); + //MOV(32, gpr.R(rt), M(&mips_->vfpuCtrl[imm - 128])); } else { //ERROR - maybe need to make this value too an "interlock" value? ERROR_LOG(CPU, "mfv - invalid register %i", imm); @@ -688,7 +688,7 @@ namespace MIPSComp gpr.MapReg(rt); STW(gpr.R(rt), CTXREG, offsetof(MIPSState, vfpuCtrl) + 4 * (imm - 128)); //gpr.BindToRegister(rt, true, false); - //MOV(32, M(¤tMIPS->vfpuCtrl[imm - 128]), gpr.R(rt)); + //MOV(32, M(&mips_->vfpuCtrl[imm - 128]), gpr.R(rt)); // TODO: Optimization if rt is Imm? // Set these BEFORE disable! diff --git a/Core/MIPS/x86/CompBranch.cpp b/Core/MIPS/x86/CompBranch.cpp index eaaa69f384..73ec89e7c2 100644 --- a/Core/MIPS/x86/CompBranch.cpp +++ b/Core/MIPS/x86/CompBranch.cpp @@ -650,7 +650,7 @@ void Jit::Comp_JumpReg(MIPSOpcode op) { // If this is a syscall, write the pc (for thread switching and other good reasons.) gpr.MapReg(rs, true, false); - MOV(32, M(¤tMIPS->pc), gpr.R(rs)); + MOV(32, M(&mips_->pc), gpr.R(rs)); if (andLink) gpr.SetImm(rd, js.compilerPC + 8); CompileDelaySlot(DELAYSLOT_FLUSH); diff --git a/Core/MIPS/x86/CompVFPU.cpp b/Core/MIPS/x86/CompVFPU.cpp index e60a753483..5a698a9b3e 100644 --- a/Core/MIPS/x86/CompVFPU.cpp +++ b/Core/MIPS/x86/CompVFPU.cpp @@ -1697,7 +1697,7 @@ void Jit::Comp_Mftv(MIPSOpcode op) { // In case we have a saved prefix. FlushPrefixV(); gpr.MapReg(rt, false, true); - MOV(32, gpr.R(rt), M(¤tMIPS->vfpuCtrl[imm - 128])); + MOV(32, gpr.R(rt), M(&mips_->vfpuCtrl[imm - 128])); } } else { //ERROR - maybe need to make this value too an "interlock" value? @@ -1724,7 +1724,7 @@ void Jit::Comp_Mftv(MIPSOpcode op) { } } else { gpr.MapReg(rt, true, false); - MOV(32, M(¤tMIPS->vfpuCtrl[imm - 128]), gpr.R(rt)); + MOV(32, M(&mips_->vfpuCtrl[imm - 128]), gpr.R(rt)); } // TODO: Optimization if rt is Imm? @@ -1756,7 +1756,7 @@ void Jit::Comp_Vmfvc(MIPSOpcode op) { gpr.MapReg(MIPS_REG_VFPUCC, true, false); MOVD_xmm(fpr.VX(vs), gpr.R(MIPS_REG_VFPUCC)); } else { - MOVSS(fpr.VX(vs), M(¤tMIPS->vfpuCtrl[imm - 128])); + MOVSS(fpr.VX(vs), M(&mips_->vfpuCtrl[imm - 128])); } fpr.ReleaseSpillLocks(); } @@ -1772,7 +1772,7 @@ void Jit::Comp_Vmtvc(MIPSOpcode op) { gpr.MapReg(MIPS_REG_VFPUCC, false, true); MOVD_xmm(gpr.R(MIPS_REG_VFPUCC), fpr.VX(vs)); } else { - MOVSS(M(¤tMIPS->vfpuCtrl[imm - 128]), fpr.VX(vs)); + MOVSS(M(&mips_->vfpuCtrl[imm - 128]), fpr.VX(vs)); } fpr.ReleaseSpillLocks(); diff --git a/Core/MIPS/x86/Jit.cpp b/Core/MIPS/x86/Jit.cpp index cc96264062..3edc7de02f 100644 --- a/Core/MIPS/x86/Jit.cpp +++ b/Core/MIPS/x86/Jit.cpp @@ -218,7 +218,7 @@ void Jit::FlushPrefixV() void Jit::WriteDowncount(int offset) { const int downcount = js.downcountAmount + offset; - SUB(32, M(¤tMIPS->downcount), downcount > 127 ? Imm32(downcount) : Imm8(downcount)); + SUB(32, M(&mips_->downcount), downcount > 127 ? Imm32(downcount) : Imm8(downcount)); } void Jit::RestoreRoundingMode(bool force, XEmitter *emitter) @@ -228,10 +228,10 @@ void Jit::RestoreRoundingMode(bool force, XEmitter *emitter) { if (emitter == NULL) emitter = this; - emitter->STMXCSR(M(¤tMIPS->temp)); + emitter->STMXCSR(M(&mips_->temp)); // Clear the rounding mode and flush-to-zero bits back to 0. - emitter->AND(32, M(¤tMIPS->temp), Imm32(~(7 << 13))); - emitter->LDMXCSR(M(¤tMIPS->temp)); + emitter->AND(32, M(&mips_->temp), Imm32(~(7 << 13))); + emitter->LDMXCSR(M(&mips_->temp)); } } @@ -252,7 +252,7 @@ void Jit::ApplyRoundingMode(bool force, XEmitter *emitter) if (!g_Config.bForceFlushToZero) skip = emitter->J_CC(CC_Z); - emitter->STMXCSR(M(¤tMIPS->temp)); + emitter->STMXCSR(M(&mips_->temp)); // The MIPS bits don't correspond exactly, so we have to adjust. // 0 -> 0 (skip2), 1 -> 3, 2 -> 2 (skip2), 3 -> 1 @@ -262,18 +262,18 @@ void Jit::ApplyRoundingMode(bool force, XEmitter *emitter) emitter->SetJumpTarget(skip2); emitter->SHL(32, R(EAX), Imm8(13)); - emitter->OR(32, M(¤tMIPS->temp), R(EAX)); + emitter->OR(32, M(&mips_->temp), R(EAX)); if (g_Config.bForceFlushToZero) { - emitter->OR(32, M(¤tMIPS->temp), Imm32(1 << 15)); + emitter->OR(32, M(&mips_->temp), Imm32(1 << 15)); } else { emitter->TEST(32, M(&mips_->fcr31), Imm32(1 << 24)); FixupBranch skip3 = emitter->J_CC(CC_Z); - emitter->OR(32, M(¤tMIPS->temp), Imm32(1 << 15)); + emitter->OR(32, M(&mips_->temp), Imm32(1 << 15)); emitter->SetJumpTarget(skip3); } - emitter->LDMXCSR(M(¤tMIPS->temp)); + emitter->LDMXCSR(M(&mips_->temp)); if (!g_Config.bForceFlushToZero) emitter->SetJumpTarget(skip); @@ -543,7 +543,7 @@ bool Jit::ReplaceJalTo(u32 dest) { MOV(32, M(&mips_->pc), Imm32(js.compilerPC)); RestoreRoundingMode(); ABI_CallFunction(entry->replaceFunc); - SUB(32, M(¤tMIPS->downcount), R(EAX)); + SUB(32, M(&mips_->downcount), R(EAX)); ApplyRoundingMode(); } @@ -582,7 +582,7 @@ void Jit::Comp_ReplacementFunc(MIPSOpcode op) MIPSCompileOp(Memory::Read_Instruction(js.compilerPC, true)); } else { FlushAll(); - MOV(32, R(ECX), M(¤tMIPS->r[MIPS_REG_RA])); + MOV(32, R(ECX), M(&mips_->r[MIPS_REG_RA])); js.downcountAmount += cycles; WriteExitDestInReg(ECX); js.compiling = false; @@ -601,10 +601,10 @@ void Jit::Comp_ReplacementFunc(MIPSOpcode op) ApplyRoundingMode(); MIPSCompileOp(Memory::Read_Instruction(js.compilerPC, true)); } else { - MOV(32, R(ECX), M(¤tMIPS->r[MIPS_REG_RA])); - SUB(32, M(¤tMIPS->downcount), R(EAX)); + MOV(32, R(ECX), M(&mips_->r[MIPS_REG_RA])); + SUB(32, M(&mips_->downcount), R(EAX)); ApplyRoundingMode(); - SUB(32, M(¤tMIPS->downcount), Imm8(0)); + SUB(32, M(&mips_->downcount), Imm8(0)); WriteExitDestInReg(ECX); js.compiling = false; } @@ -707,7 +707,7 @@ void Jit::WriteExitDestInReg(X64Reg reg) FixupBranch tooHigh = J_CC(CC_AE); // Need to set neg flag again if necessary. - SUB(32, M(¤tMIPS->downcount), Imm32(0)); + SUB(32, M(&mips_->downcount), Imm32(0)); JMP(asm_.dispatcher, true); SetJumpTarget(tooLow); @@ -721,11 +721,11 @@ void Jit::WriteExitDestInReg(X64Reg reg) if (g_Config.bIgnoreBadMemAccess) CallProtectedFunction(Core_UpdateState, Imm32(CORE_ERROR)); - SUB(32, M(¤tMIPS->downcount), Imm32(0)); + SUB(32, M(&mips_->downcount), Imm32(0)); JMP(asm_.dispatcherCheckCoreState, true); SetJumpTarget(skip); - SUB(32, M(¤tMIPS->downcount), Imm32(0)); + SUB(32, M(&mips_->downcount), Imm32(0)); J_CC(CC_NE, asm_.dispatcher, true); } else From 5fd402222b34a8fa5ad52cf66f9b33b6d6581c61 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Sun, 12 Oct 2014 15:18:22 -0700 Subject: [PATCH 094/105] x86jit: Use the shorter MDisp() offset for andLink. --- Core/MIPS/x86/CompBranch.cpp | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/Core/MIPS/x86/CompBranch.cpp b/Core/MIPS/x86/CompBranch.cpp index 73ec89e7c2..0f391805db 100644 --- a/Core/MIPS/x86/CompBranch.cpp +++ b/Core/MIPS/x86/CompBranch.cpp @@ -224,7 +224,7 @@ void Jit::CompBranchExits(CCFlags cc, u32 targetAddr, u32 notTakenAddr, bool del { // Take the branch if (andLink) - MOV(32, M(&mips_->r[MIPS_REG_RA]), Imm32(js.compilerPC + 8)); + MOV(32, gpr.GetDefaultLocation(MIPS_REG_RA), Imm32(js.compilerPC + 8)); CONDITIONAL_LOG_EXIT(targetAddr); WriteExit(targetAddr, js.nextExit++); @@ -259,7 +259,7 @@ void Jit::CompBranchExits(CCFlags cc, u32 targetAddr, u32 notTakenAddr, bool del // Take the branch if (andLink) - MOV(32, M(&mips_->r[MIPS_REG_RA]), Imm32(js.compilerPC + 8)); + MOV(32, gpr.GetDefaultLocation(MIPS_REG_RA), Imm32(js.compilerPC + 8)); CONDITIONAL_LOG_EXIT(targetAddr); WriteExit(targetAddr, js.nextExit++); From 90821b761de9f2a2f8ece0514f2c2975e8e857b0 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Sun, 12 Oct 2014 16:00:58 -0700 Subject: [PATCH 095/105] x86jit: Pad linked exits with breakpoints. So that we don't get garbage, and so we see if we end up there. --- Core/MIPS/JitCommon/JitBlockCache.cpp | 24 +++++++++++++++++++++--- Core/MIPS/JitCommon/JitBlockCache.h | 2 ++ Core/MIPS/x86/Jit.cpp | 8 ++++++++ 3 files changed, 31 insertions(+), 3 deletions(-) diff --git a/Core/MIPS/JitCommon/JitBlockCache.cpp b/Core/MIPS/JitCommon/JitBlockCache.cpp index 54fa8016ab..e34207c727 100644 --- a/Core/MIPS/JitCommon/JitBlockCache.cpp +++ b/Core/MIPS/JitCommon/JitBlockCache.cpp @@ -386,6 +386,12 @@ void JitBlockCache::LinkBlockExits(int i) { #elif defined(_M_IX86) || defined(_M_X64) XEmitter emit(b.exitPtrs[e]); emit.JMP(blocks_[destinationBlock].checkedEntry, true); + + ptrdiff_t actualSize = emit.GetWritableCodePtr() - b.exitPtrs[e]; + int pad = JitBlockCache::GetBlockExitSize() - (int)actualSize; + for (int i = 0; i < pad; ++i) { + emit.INT3(); + } #elif defined(PPC) PPCXEmitter emit(b.exitPtrs[e]); emit.B(blocks_[destinationBlock].checkedEntry); @@ -397,9 +403,8 @@ void JitBlockCache::LinkBlockExits(int i) { } } -using namespace std; - void JitBlockCache::LinkBlock(int i) { + using namespace std; LinkBlockExits(i); JitBlock &b = blocks_[i]; pair::iterator, multimap::iterator> ppp; @@ -415,6 +420,7 @@ void JitBlockCache::LinkBlock(int i) { } void JitBlockCache::UnlinkBlock(int i) { + using namespace std; JitBlock &b = blocks_[i]; pair::iterator, multimap::iterator> ppp; ppp = links_to_.equal_range(b.originalAddress); @@ -547,7 +553,7 @@ void JitBlockCache::InvalidateICache(u32 address, const u32 length) { // destroy JIT blocks // !! this works correctly under assumption that any two overlapping blocks end at the same address // TODO: This may not be a safe assumption with jit continuing enabled. - std::map, u32>::iterator it1 = block_map_.lower_bound(std::make_pair(pAddr, 0)), it2 = it1; + std::map, u32>::iterator it1 = block_map_.lower_bound(std::make_pair(pAddr, 0)), it2 = it1; while (it2 != block_map_.end() && it2->first.second < pAddr + length) { DestroyBlock(it2->second, true); it2++; @@ -556,3 +562,15 @@ void JitBlockCache::InvalidateICache(u32 address, const u32 length) { if (it1 != it2) block_map_.erase(it1, it2); } + +int JitBlockCache::GetBlockExitSize() { +#if defined(ARM) + // TODO + return 0; +#elif defined(_M_IX86) || defined(_M_X64) + return 15; +#elif defined(PPC) + // TODO + return 0; +#endif +} diff --git a/Core/MIPS/JitCommon/JitBlockCache.h b/Core/MIPS/JitCommon/JitBlockCache.h index 5bb5461b22..9154dbbe3a 100644 --- a/Core/MIPS/JitCommon/JitBlockCache.h +++ b/Core/MIPS/JitCommon/JitBlockCache.h @@ -141,6 +141,8 @@ public: int GetNumBlocks() const { return num_blocks_; } + static int GetBlockExitSize(); + private: void LinkBlockExits(int i); void LinkBlock(int i); diff --git a/Core/MIPS/x86/Jit.cpp b/Core/MIPS/x86/Jit.cpp index 3edc7de02f..49d7108c56 100644 --- a/Core/MIPS/x86/Jit.cpp +++ b/Core/MIPS/x86/Jit.cpp @@ -677,6 +677,14 @@ void Jit::WriteExit(u32 destination, int exit_num) // No blocklinking. MOV(32, M(&mips_->pc), Imm32(destination)); JMP(asm_.dispatcher, true); + + // Normally, exits are 15 bytes (MOV + &pc + dest + JMP + dest) on 64 or 32 bit. + // But just in case we somehow optimized, pad. + ptrdiff_t actualSize = GetWritableCodePtr() - b->exitPtrs[exit_num]; + int pad = JitBlockCache::GetBlockExitSize() - (int)actualSize; + for (int i = 0; i < pad; ++i) { + INT3(); + } } } From 01f9521dc586732f4d98e7f9196a945d897a8d8d Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Sun, 12 Oct 2014 17:13:04 -0700 Subject: [PATCH 096/105] jit: Invalidate blocks even if they end unevenly. This allows blocks to start and end where ever they need, which should be good for replacements and for continuing. --- Core/MIPS/ARM/ArmJit.cpp | 4 +- Core/MIPS/JitCommon/JitBlockCache.cpp | 58 ++++++++++++++++++++------- Core/MIPS/JitCommon/JitBlockCache.h | 3 ++ Core/MIPS/x86/Jit.cpp | 3 +- 4 files changed, 49 insertions(+), 19 deletions(-) diff --git a/Core/MIPS/ARM/ArmJit.cpp b/Core/MIPS/ARM/ArmJit.cpp index 68f347edbd..8d0b0909b4 100644 --- a/Core/MIPS/ARM/ArmJit.cpp +++ b/Core/MIPS/ARM/ArmJit.cpp @@ -21,6 +21,7 @@ #include "Core/Config.h" #include "Core/Core.h" #include "Core/CoreTiming.h" +#include "Core/Debugger/SymbolMap.h" #include "Core/MemMap.h" #include "Core/MIPS/MIPS.h" #include "Core/MIPS/MIPSCodeUtils.h" @@ -417,8 +418,7 @@ bool Jit::ReplaceJalTo(u32 dest) { // No writing exits, keep going! // Add a trigger so that if the inlined code changes, we invalidate this block. - // TODO: Correctly determine the size of this block. - blocks.ProxyBlock(js.blockStart, dest, 4, GetCodePtr()); + blocks.ProxyBlock(js.blockStart, dest, symbolMap.GetFunctionSize(dest) / sizeof(u32), GetCodePtr()); return true; } diff --git a/Core/MIPS/JitCommon/JitBlockCache.cpp b/Core/MIPS/JitCommon/JitBlockCache.cpp index e34207c727..5e6aca1ed8 100644 --- a/Core/MIPS/JitCommon/JitBlockCache.cpp +++ b/Core/MIPS/JitCommon/JitBlockCache.cpp @@ -149,6 +149,7 @@ int JitBlockCache::AllocateBlock(u32 startAddress) { int num = GetBlockNumberFromStartAddress(startAddress, false); if (num >= 0) { if (blocks_[num].IsPureProxy()) { + RemoveBlockMap(num); blocks_[num].invalid = true; b.proxyFor = new std::vector(); *b.proxyFor = *blocks_[num].proxyFor; @@ -200,9 +201,25 @@ void JitBlockCache::ProxyBlock(u32 rootAddress, u32 startAddress, u32 size, cons b.normalEntry = codePtr; b.checkedEntry = codePtr; proxyBlockIndices_.push_back(num_blocks_); + AddBlockMap(num_blocks_); + num_blocks_++; //commit the current block } +void JitBlockCache::AddBlockMap(int block_num) { + const JitBlock &b = blocks_[block_num]; + // Convert the logical address to a physical address for the block map + // Yeah, this'll work fine for PSP too I think. + u32 pAddr = b.originalAddress & 0x1FFFFFFF; + block_map_[std::make_pair(pAddr + 4 * b.originalSize, pAddr)] = block_num; +} + +void JitBlockCache::RemoveBlockMap(int block_num) { + const JitBlock &b = blocks_[block_num]; + u32 pAddr = b.originalAddress & 0x1FFFFFFF; + block_map_.erase(std::make_pair(pAddr + 4 * b.originalSize - 1, pAddr)); +} + static void ExpandRange(std::pair &range, u32 newStart, u32 newEnd) { range.first = std::min(range.first, newStart); range.second = std::max(range.second, newEnd); @@ -215,12 +232,9 @@ void JitBlockCache::FinalizeBlock(int block_num, bool block_link) { MIPSOpcode opcode = GetEmuHackOpForBlock(block_num); Memory::Write_Opcode_JIT(b.originalAddress, opcode); - // Convert the logical address to a physical address for the block map - // Yeah, this'll work fine for PSP too I think. - u32 pAddr = b.originalAddress & 0x1FFFFFFF; + AddBlockMap(block_num); u32 latestExit = 0; - block_map_[std::make_pair(pAddr + 4 * b.originalSize - 1, pAddr)] = block_num; if (block_link) { for (int i = 0; i < MAX_JIT_BLOCK_EXITS; i++) { if (b.exitAddress[i] != INVALID_EXIT) { @@ -479,6 +493,8 @@ void JitBlockCache::DestroyBlock(int block_num, bool invalidate) { return; } JitBlock *b = &blocks_[block_num]; + // No point it being in there anymore. + RemoveBlockMap(block_num); // Pure proxy blocks always point directly to a real block, there should be no chains of // proxy-only blocks pointing to proxy-only blocks. @@ -548,19 +564,31 @@ void JitBlockCache::DestroyBlock(int block_num, bool invalidate) { void JitBlockCache::InvalidateICache(u32 address, const u32 length) { // Convert the logical address to a physical address for the block map - u32 pAddr = address & 0x1FFFFFFF; + const u32 pAddr = address & 0x1FFFFFFF; + const u32 pEnd = pAddr + length; - // destroy JIT blocks - // !! this works correctly under assumption that any two overlapping blocks end at the same address - // TODO: This may not be a safe assumption with jit continuing enabled. - std::map, u32>::iterator it1 = block_map_.lower_bound(std::make_pair(pAddr, 0)), it2 = it1; - while (it2 != block_map_.end() && it2->first.second < pAddr + length) { - DestroyBlock(it2->second, true); - it2++; + // Blocks may start and end in overlapping ways, and destroying one invalidates iterators. + // So after destroying one, we start over. + while (true) { + auto next = block_map_.lower_bound(std::make_pair(pAddr, 0)); + // End is inclusive, so a matching end won't be included. + auto last = block_map_.lower_bound(std::make_pair(pEnd, 0)); + if (next == last) { + // It wasn't in the map at all (or anymore.) + // This includes if both were end(), which should be uncommon. + break; + } + for (; next != last; ++next) { + const u32 blockStart = next->first.second; + const u32 blockEnd = next->first.first; + if (blockStart < pEnd && blockEnd > pAddr) { + DestroyBlock(next->second, true); + // Our iterator is now invalid. Break and search again. + // Most of the time there shouldn't be a bunch of matching blocks. + break; + } + } } - - if (it1 != it2) - block_map_.erase(it1, it2); } int JitBlockCache::GetBlockExitSize() { diff --git a/Core/MIPS/JitCommon/JitBlockCache.h b/Core/MIPS/JitCommon/JitBlockCache.h index 9154dbbe3a..52403764a6 100644 --- a/Core/MIPS/JitCommon/JitBlockCache.h +++ b/Core/MIPS/JitCommon/JitBlockCache.h @@ -148,6 +148,9 @@ private: void LinkBlock(int i); void UnlinkBlock(int i); + void AddBlockMap(int block_num); + void RemoveBlockMap(int block_num); + MIPSOpcode GetEmuHackOpForBlock(int block_num) const; MIPSState *mips_; diff --git a/Core/MIPS/x86/Jit.cpp b/Core/MIPS/x86/Jit.cpp index 49d7108c56..acdf203074 100644 --- a/Core/MIPS/x86/Jit.cpp +++ b/Core/MIPS/x86/Jit.cpp @@ -551,8 +551,7 @@ bool Jit::ReplaceJalTo(u32 dest) { // No writing exits, keep going! // Add a trigger so that if the inlined code changes, we invalidate this block. - // TODO: Correctly determine the size of this block. - blocks.ProxyBlock(js.blockStart, dest, 4, GetCodePtr()); + blocks.ProxyBlock(js.blockStart, dest, symbolMap.GetFunctionSize(dest) / sizeof(u32), GetCodePtr()); return true; } From d98adf27d6148d1ff99af47d86db5523ca26cdcc Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Sun, 12 Oct 2014 17:15:31 -0700 Subject: [PATCH 097/105] x86jit: Add proxy blocks for continuing. --- Core/MIPS/JitCommon/JitState.h | 2 ++ Core/MIPS/x86/CompBranch.cpp | 6 ++++++ Core/MIPS/x86/Jit.cpp | 23 ++++++++++++++++++++--- Core/MIPS/x86/Jit.h | 1 + 4 files changed, 29 insertions(+), 3 deletions(-) diff --git a/Core/MIPS/JitCommon/JitState.h b/Core/MIPS/JitCommon/JitState.h index c77481499d..0bb7007a6e 100644 --- a/Core/MIPS/JitCommon/JitState.h +++ b/Core/MIPS/JitCommon/JitState.h @@ -64,6 +64,8 @@ namespace MIPSComp { u32 compilerPC; u32 blockStart; + u32 lastContinuedPC; + u32 initialBlockSize; int nextExit; bool cancel; bool inDelaySlot; diff --git a/Core/MIPS/x86/CompBranch.cpp b/Core/MIPS/x86/CompBranch.cpp index 0f391805db..ec8979122c 100644 --- a/Core/MIPS/x86/CompBranch.cpp +++ b/Core/MIPS/x86/CompBranch.cpp @@ -215,6 +215,7 @@ void Jit::CompBranchExits(CCFlags cc, u32 targetAddr, u32 notTakenAddr, bool del if (likely) CompileDelaySlot(DELAYSLOT_NICE); + AddContinuedBlock(targetAddr); // Account for the increment in the loop. js.compilerPC = targetAddr - 4; // In case the delay slot was a break or something. @@ -328,6 +329,7 @@ void Jit::BranchRSRTComp(MIPSOpcode op, Gen::CCFlags cc, bool likely) // Branch taken. Always compile the delay slot, and then go to dest. CompileDelaySlot(DELAYSLOT_NICE); + AddContinuedBlock(targetAddr); // Account for the increment in the loop. js.compilerPC = targetAddr - 4; // In case the delay slot was a break or something. @@ -406,6 +408,7 @@ void Jit::BranchRSZeroComp(MIPSOpcode op, Gen::CCFlags cc, bool andLink, bool li if (andLink) gpr.SetImm(MIPS_REG_RA, js.compilerPC + 8); + AddContinuedBlock(targetAddr); // Account for the increment in the loop. js.compilerPC = targetAddr - 4; // In case the delay slot was a break or something. @@ -585,6 +588,7 @@ void Jit::Comp_Jump(MIPSOpcode op) { CompileDelaySlot(DELAYSLOT_NICE); if (jo.continueJumps && js.numInstructions < jo.continueMaxInstructions) { + AddContinuedBlock(targetAddr); // Account for the increment in the loop. js.compilerPC = targetAddr - 4; // In case the delay slot was a break or something. @@ -609,6 +613,7 @@ void Jit::Comp_Jump(MIPSOpcode op) { CompileDelaySlot(DELAYSLOT_NICE); if (jo.continueJumps && js.numInstructions < jo.continueMaxInstructions) { + AddContinuedBlock(targetAddr); // Account for the increment in the loop. js.compilerPC = targetAddr - 4; // In case the delay slot was a break or something. @@ -679,6 +684,7 @@ void Jit::Comp_JumpReg(MIPSOpcode op) if (jo.continueJumps && gpr.IsImm(rs) && js.numInstructions < jo.continueMaxInstructions) { + AddContinuedBlock(gpr.GetImm(rs)); // Account for the increment in the loop. js.compilerPC = gpr.GetImm(rs) - 4; // In case the delay slot was a break or something. diff --git a/Core/MIPS/x86/Jit.cpp b/Core/MIPS/x86/Jit.cpp index acdf203074..6d3cf01e75 100644 --- a/Core/MIPS/x86/Jit.cpp +++ b/Core/MIPS/x86/Jit.cpp @@ -116,8 +116,6 @@ static void JitLogMiss(MIPSOpcode op) JitOptions::JitOptions() { enableBlocklink = true; - // WARNING: These options don't work properly with cache clearing. - // Need to find a smart way to handle before enabling. immBranches = false; continueBranches = false; continueJumps = false; @@ -396,6 +394,8 @@ const u8 *Jit::DoJit(u32 em_address, JitBlock *b) { js.cancel = false; js.blockStart = js.compilerPC = mips_->pc; + js.lastContinuedPC = 0; + js.initialBlockSize = 0; js.nextExit = 0; js.downcountAmount = 0; js.curBlock = b; @@ -466,10 +466,27 @@ const u8 *Jit::DoJit(u32 em_address, JitBlock *b) b->codeSize = (u32)(GetCodePtr() - b->normalEntry); NOP(); AlignCode4(); - b->originalSize = js.numInstructions; + if (js.lastContinuedPC == 0) + b->originalSize = js.numInstructions; + else + { + // We continued at least once. Add the last proxy and set the originalSize correctly. + blocks.ProxyBlock(js.blockStart, js.lastContinuedPC, (js.compilerPC - js.lastContinuedPC) / sizeof(u32), GetCodePtr()); + b->originalSize = js.initialBlockSize; + } return b->normalEntry; } +void Jit::AddContinuedBlock(u32 dest) +{ + // The first block is the root block. When we continue, we create proxy blocks after that. + if (js.lastContinuedPC == 0) + js.initialBlockSize = js.numInstructions; + else + blocks.ProxyBlock(js.blockStart, js.lastContinuedPC, (js.compilerPC - js.lastContinuedPC) / sizeof(u32), GetCodePtr()); + js.lastContinuedPC = dest; +} + bool Jit::DescribeCodePtr(const u8 *ptr, std::string &name) { u32 jitAddr = blocks.GetAddressFromBlockPtr(ptr); diff --git a/Core/MIPS/x86/Jit.h b/Core/MIPS/x86/Jit.h index bc3e7f0f0b..a273e23ba3 100644 --- a/Core/MIPS/x86/Jit.h +++ b/Core/MIPS/x86/Jit.h @@ -193,6 +193,7 @@ private: CompileDelaySlot(flags, &state); } void EatInstruction(MIPSOpcode op); + void AddContinuedBlock(u32 dest); void WriteExit(u32 destination, int exit_num); void WriteExitDestInReg(X64Reg reg); From 1064f580e4fc9c40d306accc722934f22e9f3c61 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Sun, 12 Oct 2014 17:20:26 -0700 Subject: [PATCH 098/105] armjit: Add proxy blocks for continuing. --- Core/MIPS/ARM/ArmCompBranch.cpp | 5 +++++ Core/MIPS/ARM/ArmJit.cpp | 23 ++++++++++++++++++++--- Core/MIPS/ARM/ArmJit.h | 2 ++ 3 files changed, 27 insertions(+), 3 deletions(-) diff --git a/Core/MIPS/ARM/ArmCompBranch.cpp b/Core/MIPS/ARM/ArmCompBranch.cpp index 0f6de65645..85c15f3675 100644 --- a/Core/MIPS/ARM/ArmCompBranch.cpp +++ b/Core/MIPS/ARM/ArmCompBranch.cpp @@ -94,6 +94,7 @@ void Jit::BranchRSRTComp(MIPSOpcode op, ArmGen::CCFlags cc, bool likely) // Branch taken. Always compile the delay slot, and then go to dest. CompileDelaySlot(DELAYSLOT_NICE); + AddContinuedBlock(targetAddr); // Account for the increment in the loop. js.compilerPC = targetAddr - 4; // In case the delay slot was a break or something. @@ -209,6 +210,7 @@ void Jit::BranchRSZeroComp(MIPSOpcode op, ArmGen::CCFlags cc, bool andLink, bool if (andLink) gpr.SetImm(MIPS_REG_RA, js.compilerPC + 8); + AddContinuedBlock(targetAddr); // Account for the increment in the loop. js.compilerPC = targetAddr - 4; // In case the delay slot was a break or something. @@ -461,6 +463,7 @@ void Jit::Comp_Jump(MIPSOpcode op) { case 2: //j CompileDelaySlot(DELAYSLOT_NICE); if (jo.continueJumps && js.numInstructions < jo.continueMaxInstructions) { + AddContinuedBlock(targetAddr); // Account for the increment in the loop. js.compilerPC = targetAddr - 4; // In case the delay slot was a break or something. @@ -478,6 +481,7 @@ void Jit::Comp_Jump(MIPSOpcode op) { gpr.SetImm(MIPS_REG_RA, js.compilerPC + 8); CompileDelaySlot(DELAYSLOT_NICE); if (jo.continueJumps && js.numInstructions < jo.continueMaxInstructions) { + AddContinuedBlock(targetAddr); // Account for the increment in the loop. js.compilerPC = targetAddr - 4; // In case the delay slot was a break or something. @@ -537,6 +541,7 @@ void Jit::Comp_JumpReg(MIPSOpcode op) } if (jo.continueJumps && gpr.IsImm(rs) && js.numInstructions < jo.continueMaxInstructions) { + AddContinuedBlock(gpr.GetImm(rs)); // Account for the increment in the loop. js.compilerPC = gpr.GetImm(rs) - 4; // In case the delay slot was a break or something. diff --git a/Core/MIPS/ARM/ArmJit.cpp b/Core/MIPS/ARM/ArmJit.cpp index 8d0b0909b4..5f503ee29a 100644 --- a/Core/MIPS/ARM/ArmJit.cpp +++ b/Core/MIPS/ARM/ArmJit.cpp @@ -68,8 +68,6 @@ ArmJitOptions::ArmJitOptions() { useBackJump = false; useForwardJump = false; cachePointers = true; - // WARNING: These options don't work properly with cache clearing or jit compare. - // Need to find a smart way to handle before enabling. immBranches = false; continueBranches = false; continueJumps = false; @@ -247,6 +245,8 @@ const u8 *Jit::DoJit(u32 em_address, JitBlock *b) { js.cancel = false; js.blockStart = js.compilerPC = mips_->pc; + js.lastContinuedPC = 0; + js.initialBlockSize = 0; js.nextExit = 0; js.downcountAmount = 0; js.curBlock = b; @@ -356,10 +356,27 @@ const u8 *Jit::DoJit(u32 em_address, JitBlock *b) // Don't forget to zap the newly written instructions in the instruction cache! FlushIcache(); - b->originalSize = js.numInstructions; + if (js.lastContinuedPC == 0) + b->originalSize = js.numInstructions; + else + { + // We continued at least once. Add the last proxy and set the originalSize correctly. + blocks.ProxyBlock(js.blockStart, js.lastContinuedPC, (js.compilerPC - js.lastContinuedPC) / sizeof(u32), GetCodePtr()); + b->originalSize = js.initialBlockSize; + } return b->normalEntry; } +void Jit::AddContinuedBlock(u32 dest) +{ + // The first block is the root block. When we continue, we create proxy blocks after that. + if (js.lastContinuedPC == 0) + js.initialBlockSize = js.numInstructions; + else + blocks.ProxyBlock(js.blockStart, js.lastContinuedPC, (js.compilerPC - js.lastContinuedPC) / sizeof(u32), GetCodePtr()); + js.lastContinuedPC = dest; +} + bool Jit::DescribeCodePtr(const u8 *ptr, std::string &name) { // TODO: Not used by anything yet. diff --git a/Core/MIPS/ARM/ArmJit.h b/Core/MIPS/ARM/ArmJit.h index 95ba4ec867..8a47e8e965 100644 --- a/Core/MIPS/ARM/ArmJit.h +++ b/Core/MIPS/ARM/ArmJit.h @@ -68,6 +68,8 @@ public: void CompileDelaySlot(int flags); void EatInstruction(MIPSOpcode op); + void AddContinuedBlock(u32 dest); + void Comp_RunBlock(MIPSOpcode op); void Comp_ReplacementFunc(MIPSOpcode op); From 4853a1b7a09d8b7132196f606e9b742c40e06777 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Sun, 12 Oct 2014 17:34:59 -0700 Subject: [PATCH 099/105] jit: Optimize proxy block lookup from address. It was really slow before with enough proxy blocks. --- Core/MIPS/JitCommon/JitBlockCache.cpp | 22 +++++++++++++++------- Core/MIPS/JitCommon/JitBlockCache.h | 2 +- 2 files changed, 16 insertions(+), 8 deletions(-) diff --git a/Core/MIPS/JitCommon/JitBlockCache.cpp b/Core/MIPS/JitCommon/JitBlockCache.cpp index 5e6aca1ed8..986dec5d26 100644 --- a/Core/MIPS/JitCommon/JitBlockCache.cpp +++ b/Core/MIPS/JitCommon/JitBlockCache.cpp @@ -123,7 +123,7 @@ void JitBlockCache::Clear() { DestroyBlock(i, false); links_to_.clear(); block_map_.clear(); - proxyBlockIndices_.clear(); + proxyBlockMap_.clear(); num_blocks_ = 0; blockMemRanges_[JITBLOCK_RANGE_SCRATCH] = std::make_pair(0xFFFFFFFF, 0x00000000); @@ -176,7 +176,7 @@ void JitBlockCache::ProxyBlock(u32 rootAddress, u32 startAddress, u32 size, cons // instead of creating a new block. int num = GetBlockNumberFromStartAddress(startAddress, false); if (num != -1) { - INFO_LOG(HLE, "Adding proxy root %08x to block at %08x", rootAddress, startAddress); + DEBUG_LOG(HLE, "Adding proxy root %08x to block at %08x", rootAddress, startAddress); if (!blocks_[num].proxyFor) { blocks_[num].proxyFor = new std::vector(); } @@ -200,7 +200,7 @@ void JitBlockCache::ProxyBlock(u32 rootAddress, u32 startAddress, u32 size, cons // Make binary searches and stuff work ok b.normalEntry = codePtr; b.checkedEntry = codePtr; - proxyBlockIndices_.push_back(num_blocks_); + proxyBlockMap_.insert(std::make_pair(startAddress, num_blocks_)); AddBlockMap(num_blocks_); num_blocks_++; //commit the current block @@ -337,9 +337,10 @@ int JitBlockCache::GetBlockNumberFromStartAddress(u32 addr, bool realBlocksOnly) int bl = GetBlockNumberFromEmuHackOp(inst); if (bl < 0) { if (!realBlocksOnly) { - // Wasn't an emu hack op, look through proxyBlockIndices_. - for (size_t i = 0; i < proxyBlockIndices_.size(); i++) { - int blockIndex = proxyBlockIndices_[i]; + // Wasn't an emu hack op, look through proxyBlockMap_. + auto range = proxyBlockMap_.equal_range(addr); + for (auto it = range.first; it != range.second; ++it) { + const int blockIndex = it->second; if (blocks_[blockIndex].originalAddress == addr && !blocks_[blockIndex].proxyFor && !blocks_[blockIndex].invalid) return blockIndex; } @@ -513,7 +514,14 @@ void JitBlockCache::DestroyBlock(int block_num, bool invalidate) { delete b->proxyFor; b->proxyFor = 0; } - // TODO: Remove from proxyBlockIndices_. + auto range = proxyBlockMap_.equal_range(b->originalAddress); + for (auto it = range.first; it != range.second; ++it) { + if (it->second == block_num) { + // Found it. Delete and bail. + proxyBlockMap_.erase(it); + break; + } + } // TODO: Handle the case when there's a proxy block and a regular JIT block at the same location. // In this case we probably "leak" the proxy block currently (no memory leak but it'll stay enabled). diff --git a/Core/MIPS/JitCommon/JitBlockCache.h b/Core/MIPS/JitCommon/JitBlockCache.h index 52403764a6..cadc1ddb2d 100644 --- a/Core/MIPS/JitCommon/JitBlockCache.h +++ b/Core/MIPS/JitCommon/JitBlockCache.h @@ -156,7 +156,7 @@ private: MIPSState *mips_; CodeBlock *codeBlock_; JitBlock *blocks_; - std::vector proxyBlockIndices_; + std::multimap proxyBlockMap_; int num_blocks_; std::multimap links_to_; From 2e81a388920a5907e7d1c2a03ef27605409f7371 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Sun, 12 Oct 2014 17:46:54 -0700 Subject: [PATCH 100/105] jit: Fix a possible infinite loop in invalidation. --- Core/MIPS/JitCommon/JitBlockCache.cpp | 3 +++ 1 file changed, 3 insertions(+) diff --git a/Core/MIPS/JitCommon/JitBlockCache.cpp b/Core/MIPS/JitCommon/JitBlockCache.cpp index 986dec5d26..7973995ef3 100644 --- a/Core/MIPS/JitCommon/JitBlockCache.cpp +++ b/Core/MIPS/JitCommon/JitBlockCache.cpp @@ -596,6 +596,9 @@ void JitBlockCache::InvalidateICache(u32 address, const u32 length) { break; } } + if (next == last) { + break; + } } } From e6373aaed9b151aadaa739969cb44f0f3f256997 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Sun, 12 Oct 2014 17:47:07 -0700 Subject: [PATCH 101/105] jit: Remove from the block map more carefully. --- Core/MIPS/JitCommon/JitBlockCache.cpp | 13 ++++++++++++- 1 file changed, 12 insertions(+), 1 deletion(-) diff --git a/Core/MIPS/JitCommon/JitBlockCache.cpp b/Core/MIPS/JitCommon/JitBlockCache.cpp index 7973995ef3..b0a8e28219 100644 --- a/Core/MIPS/JitCommon/JitBlockCache.cpp +++ b/Core/MIPS/JitCommon/JitBlockCache.cpp @@ -217,7 +217,18 @@ void JitBlockCache::AddBlockMap(int block_num) { void JitBlockCache::RemoveBlockMap(int block_num) { const JitBlock &b = blocks_[block_num]; u32 pAddr = b.originalAddress & 0x1FFFFFFF; - block_map_.erase(std::make_pair(pAddr + 4 * b.originalSize - 1, pAddr)); + auto it = block_map_.find(std::make_pair(pAddr + 4 * b.originalSize - 1, pAddr)); + if (it != block_map_.end() && it->second == block_num) { + block_map_.erase(it); + } else { + // It wasn't in there, or it has the wrong key. Let's search... + for (auto it = block_map_.begin(); it != block_map_.end(); ++it) { + if (it->second == block_num) { + block_map_.erase(it); + break; + } + } + } } static void ExpandRange(std::pair &range, u32 newStart, u32 newEnd) { From 040a6d1745749ced4662d02fa5b9e4e2ea820f35 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Sun, 12 Oct 2014 18:57:56 -0700 Subject: [PATCH 102/105] jit: Improve performance of clearing jit. --- Core/MIPS/JitCommon/JitBlockCache.cpp | 10 +++++++--- 1 file changed, 7 insertions(+), 3 deletions(-) diff --git a/Core/MIPS/JitCommon/JitBlockCache.cpp b/Core/MIPS/JitCommon/JitBlockCache.cpp index b0a8e28219..63f1e56e70 100644 --- a/Core/MIPS/JitCommon/JitBlockCache.cpp +++ b/Core/MIPS/JitCommon/JitBlockCache.cpp @@ -119,11 +119,11 @@ void JitBlockCache::Shutdown() { // This clears the JIT cache. It's called from JitCache.cpp when the JIT cache // is full and when saving and loading states. void JitBlockCache::Clear() { + block_map_.clear(); + proxyBlockMap_.clear(); for (int i = 0; i < num_blocks_; i++) DestroyBlock(i, false); links_to_.clear(); - block_map_.clear(); - proxyBlockMap_.clear(); num_blocks_ = 0; blockMemRanges_[JITBLOCK_RANGE_SCRATCH] = std::make_pair(0xFFFFFFFF, 0x00000000); @@ -216,7 +216,11 @@ void JitBlockCache::AddBlockMap(int block_num) { void JitBlockCache::RemoveBlockMap(int block_num) { const JitBlock &b = blocks_[block_num]; - u32 pAddr = b.originalAddress & 0x1FFFFFFF; + if (b.invalid) { + return; + } + + const u32 pAddr = b.originalAddress & 0x1FFFFFFF; auto it = block_map_.find(std::make_pair(pAddr + 4 * b.originalSize - 1, pAddr)); if (it != block_map_.end() && it->second == block_num) { block_map_.erase(it); From b53f13480acdb281ce203d4a9f4a67a959763212 Mon Sep 17 00:00:00 2001 From: "Unknown W. Brackets" Date: Sun, 12 Oct 2014 19:01:04 -0700 Subject: [PATCH 103/105] x86jit: Centralize continuing logic. --- Core/MIPS/x86/CompBranch.cpp | 10 +++++----- Core/MIPS/x86/Jit.h | 21 ++++++++++++++++++++- 2 files changed, 25 insertions(+), 6 deletions(-) diff --git a/Core/MIPS/x86/CompBranch.cpp b/Core/MIPS/x86/CompBranch.cpp index ec8979122c..e66f277086 100644 --- a/Core/MIPS/x86/CompBranch.cpp +++ b/Core/MIPS/x86/CompBranch.cpp @@ -168,9 +168,9 @@ bool Jit::PredictTakeBranch(u32 targetAddr, bool likely) { void Jit::CompBranchExits(CCFlags cc, u32 targetAddr, u32 notTakenAddr, bool delaySlotIsNice, bool likely, bool andLink) { // We may want to try to continue along this branch a little while, to reduce reg flushing. - if (CanContinueBranch()) + bool predictTakeBranch = PredictTakeBranch(targetAddr, likely); + if (CanContinueBranch(predictTakeBranch ? targetAddr : notTakenAddr)) { - bool predictTakeBranch = PredictTakeBranch(targetAddr, likely); if (predictTakeBranch) cc = FlipCCFlag(cc); @@ -586,7 +586,7 @@ void Jit::Comp_Jump(MIPSOpcode op) { switch (op >> 26) { case 2: //j CompileDelaySlot(DELAYSLOT_NICE); - if (jo.continueJumps && js.numInstructions < jo.continueMaxInstructions) + if (CanContinueJump(targetAddr)) { AddContinuedBlock(targetAddr); // Account for the increment in the loop. @@ -611,7 +611,7 @@ void Jit::Comp_Jump(MIPSOpcode op) { // Save return address - might be overwritten by delay slot. gpr.SetImm(MIPS_REG_RA, js.compilerPC + 8); CompileDelaySlot(DELAYSLOT_NICE); - if (jo.continueJumps && js.numInstructions < jo.continueMaxInstructions) + if (CanContinueJump(targetAddr)) { AddContinuedBlock(targetAddr); // Account for the increment in the loop. @@ -682,7 +682,7 @@ void Jit::Comp_JumpReg(MIPSOpcode op) gpr.DiscardRegContentsIfCached(MIPS_REG_T9); } - if (jo.continueJumps && gpr.IsImm(rs) && js.numInstructions < jo.continueMaxInstructions) + if (gpr.IsImm(rs) && CanContinueJump(gpr.GetImm(rs))) { AddContinuedBlock(gpr.GetImm(rs)); // Account for the increment in the loop. diff --git a/Core/MIPS/x86/Jit.h b/Core/MIPS/x86/Jit.h index a273e23ba3..8a5e75cf92 100644 --- a/Core/MIPS/x86/Jit.h +++ b/Core/MIPS/x86/Jit.h @@ -260,7 +260,7 @@ private: } bool PredictTakeBranch(u32 targetAddr, bool likely); - bool CanContinueBranch() { + bool CanContinueBranch(u32 targetAddr) { if (!jo.continueBranches || js.numInstructions >= jo.continueMaxInstructions) { return false; } @@ -268,6 +268,25 @@ private: if (js.nextExit >= MAX_JIT_BLOCK_EXITS - 2) { return false; } + // Sometimes we predict wrong and get into impossible conditions where games have jumps to 0. + if (!targetAddr) { + return false; + } + return true; + } + bool CanContinueJump(u32 targetAddr) { + if (!jo.continueJumps || js.numInstructions >= jo.continueMaxInstructions) { + return false; + } + if (!targetAddr) { + return false; + } + return true; + } + bool CanContinueImmBranch(u32 targetAddr) { + if (!jo.immBranches || js.numInstructions >= jo.continueMaxInstructions) { + return false; + } return true; } From b8d6a64f7a61e856629921c3015fff15e7051e0f Mon Sep 17 00:00:00 2001 From: Martin Foo Date: Tue, 14 Oct 2014 20:24:09 +0800 Subject: [PATCH 104/105] Compatibility ndk r9d armeabi-v7a-hard APP_ABI building. --- android/jni/Android.mk | 3 ++- android/jni/Locals.mk | 3 ++- 2 files changed, 4 insertions(+), 2 deletions(-) diff --git a/android/jni/Android.mk b/android/jni/Android.mk index 90df331681..7a33723437 100644 --- a/android/jni/Android.mk +++ b/android/jni/Android.mk @@ -51,7 +51,8 @@ ARCH_FILES := \ $(SRC)/GPU/Common/VertexDecoderX86.cpp endif -ifeq ($(TARGET_ARCH_ABI),armeabi-v7a) +# ifeq ($(TARGET_ARCH_ABI),armeabi-v7a) +ifeq ($(findstring armeabi-v7a,$(TARGET_ARCH_ABI)),armeabi-v7a) ARCH_FILES := \ $(SRC)/GPU/Common/TextureDecoderNEON.cpp.neon \ $(SRC)/Common/ArmEmitter.cpp \ diff --git a/android/jni/Locals.mk b/android/jni/Locals.mk index f3608378d8..d2844bafba 100644 --- a/android/jni/Locals.mk +++ b/android/jni/Locals.mk @@ -17,7 +17,8 @@ LOCAL_C_INCLUDES := \ LOCAL_STATIC_LIBRARIES := native libzip LOCAL_LDLIBS := -lz -lGLESv2 -lEGL -ldl -llog -ifeq ($(TARGET_ARCH_ABI),armeabi-v7a) +# ifeq ($(TARGET_ARCH_ABI),armeabi-v7a) +ifeq ($(findstring armeabi-v7a,$(TARGET_ARCH_ABI)),armeabi-v7a) LOCAL_LDLIBS += $(LOCAL_PATH)/../../ffmpeg/android/armv7/lib/libavformat.a LOCAL_LDLIBS += $(LOCAL_PATH)/../../ffmpeg/android/armv7/lib/libavcodec.a LOCAL_LDLIBS += $(LOCAL_PATH)/../../ffmpeg/android/armv7/lib/libswresample.a From 00173b7aee860b37f8ebdcb264a505fe976b326f Mon Sep 17 00:00:00 2001 From: mgaver Date: Thu, 16 Oct 2014 01:38:17 +0900 Subject: [PATCH 105/105] Update ViewController.mm Fix scale for iPhone 6 Plus --- ios/ViewController.mm | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/ios/ViewController.mm b/ios/ViewController.mm index b1944b6b78..333d360807 100644 --- a/ios/ViewController.mm +++ b/ios/ViewController.mm @@ -21,6 +21,8 @@ #include "gfx_es2/fbo.h" #define IS_IPAD() ([UIDevice currentDevice].userInterfaceIdiom == UIUserInterfaceIdiomPad) +#define IS_IPHONE() ([UIDevice currentDevice].userInterfaceIdiom == UIUserInterfaceIdiomPhone) +#define IS_IPHONE_6P() (IS_IPHONE() && [[UIScreen mainScreen] bounds].size.height == 736.0) float dp_xscale = 1.0f; float dp_yscale = 1.0f; @@ -130,7 +132,7 @@ ViewController* sharedViewController; [EAGLContext setCurrentContext:self.context]; self.preferredFramesPerSecond = 60; - float scale = [UIScreen mainScreen].scale; + float scale = (IS_IPHONE_6P() ? 3.0f : [UIScreen mainScreen].scale); CGSize size = [[UIApplication sharedApplication].delegate window].frame.size; if (size.height > size.width) @@ -227,7 +229,7 @@ ViewController* sharedViewController; { lock_guard guard(input_state.lock); - float scale = [UIScreen mainScreen].scale; + float scale = (IS_IPHONE_6P() ? 3.0f : [UIScreen mainScreen].scale); float scaledX = (int)(x * dp_xscale) * scale; float scaledY = (int)(y * dp_yscale) * scale;