2V0-g: port the .fnt font pipeline over a FreeType GDI emulation
CGraphicText, CGraphicFontTexture and CGraphicDib are the 40250 bodies verbatim; CGraphicTextInstance is verbatim except that its D3D draws become UI render commands (one image quad per glyph, Bar for cursor/underline). - platform/Win32Gdi.cpp: the GDI subset those files call, over vendored FreeType 2.13.3, following GDI's LOGFONT cell height, win metrics and gasp rules; system font lookup with substitute and CJK fallback faces - Win32Crt: MultiByteToWideChar / WideCharToMultiByte (UTF-8 and 1252 by hand, other code pages through iconv) - CGraphicImageTexture memory textures: glyph pages reach Godot as "mem:<id>@<revision>"; Metin2PythonHost.memory_texture + Canvas cache - Util.cpp code page / font face table, SetDefaultCodePage at boot, DefaultFont_* block, fnt resource type, GetMaxTextureWidth/Height - TextTag, StringCodec, Arabic, Vietnamese copied from 40250 - tests: port.text (Tahoma 12 metrics, alignment, CP949), port.app_loop and python_host_test.gd now check glyph quads instead of text commands Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5.5
parent
141206cb69
commit
afe63bfa61
@@ -31,6 +31,10 @@ target_include_directories(port_logic PUBLIC ${MT_PORT_SHIMS})
|
||||
# Libraries 40250 links that are vendored as-is: <lzo/lzo1x.h> (EterBase/lzo.h), <cryptopp/*>
|
||||
# (EterBase/cipher.h), <Python-2.7/*> (ScriptLib).
|
||||
target_link_libraries(port_logic PUBLIC mt3p::minilzo mt3p::cryptopp)
|
||||
# common/Win32Crt.cpp converts non-UTF-8 code pages with iconv (MT_ICONV_LIBRARY, extension/CMakeLists.txt).
|
||||
if(MT_ICONV_LIBRARY)
|
||||
target_link_libraries(port_logic PUBLIC ${MT_ICONV_LIBRARY})
|
||||
endif()
|
||||
if(TARGET mtpython)
|
||||
target_link_libraries(port_logic PUBLIC mt3p::python)
|
||||
endif()
|
||||
@@ -164,6 +168,12 @@ if(BUILD_TESTING AND CMAKE_SYSTEM_NAME STREQUAL CMAKE_HOST_SYSTEM_NAME)
|
||||
add_test(NAME port.eterpack COMMAND $<TARGET_FILE:port_eterpack_test> ${MT_40250_CLIENT})
|
||||
set_tests_properties(port.eterpack PROPERTIES SKIP_RETURN_CODE 77)
|
||||
|
||||
# 批次 2V0-g: the 40250 font path over the FreeType GDI emulation; needs a system Tahoma.
|
||||
add_executable(port_text_test ${CMAKE_CURRENT_SOURCE_DIR}/../../tests/port_text_test.cpp)
|
||||
target_link_libraries(port_text_test PRIVATE port_platform)
|
||||
add_test(NAME port.text COMMAND $<TARGET_FILE:port_text_test>)
|
||||
set_tests_properties(port.text PROPERTIES SKIP_RETURN_CODE 77)
|
||||
|
||||
if(TARGET mtpython)
|
||||
add_executable(port_python_launcher_test ${CMAKE_CURRENT_SOURCE_DIR}/../../tests/port_python_launcher_test.cpp)
|
||||
target_link_libraries(port_python_launcher_test PRIVATE port_platform)
|
||||
|
||||
@@ -0,0 +1,209 @@
|
||||
#include "StdAfx.h"
|
||||
#include "TextTag.h"
|
||||
|
||||
int GetTextTag(const wchar_t * src, int maxLen, int & tagLen, std::wstring & extraInfo)
|
||||
{
|
||||
tagLen = 1;
|
||||
|
||||
if (maxLen < 2 || *src != L'|')
|
||||
return TEXT_TAG_PLAIN;
|
||||
|
||||
const wchar_t * cur = ++src;
|
||||
|
||||
if (*cur == L'c') // color
|
||||
{
|
||||
if (maxLen < 10)
|
||||
return TEXT_TAG_PLAIN;
|
||||
|
||||
tagLen = 10;
|
||||
extraInfo.assign(++cur, 8);
|
||||
return TEXT_TAG_COLOR;
|
||||
}
|
||||
else if (*cur == L'|') // ||는 |로 표시한다.
|
||||
{
|
||||
tagLen = 2;
|
||||
return TEXT_TAG_TAG;
|
||||
}
|
||||
else if (*cur == L'r') // restore color
|
||||
{
|
||||
tagLen = 2;
|
||||
return TEXT_TAG_RESTORE_COLOR;
|
||||
}
|
||||
else if (*cur == L'H') // hyperlink |Hitem:10000:0:0:0:0|h[이름]|h
|
||||
{
|
||||
tagLen = 2;
|
||||
return TEXT_TAG_HYPERLINK_START;
|
||||
}
|
||||
else if (*cur == L'h') // end of hyperlink
|
||||
{
|
||||
tagLen = 2;
|
||||
return TEXT_TAG_HYPERLINK_END;
|
||||
}
|
||||
|
||||
return TEXT_TAG_PLAIN;
|
||||
}
|
||||
|
||||
std::wstring GetTextTagOutputString(const wchar_t * src, int src_len)
|
||||
{
|
||||
int len;
|
||||
std::wstring dst;
|
||||
std::wstring extraInfo;
|
||||
int output_len = 0;
|
||||
int hyperlinkStep = 0;
|
||||
|
||||
for (int i = 0; i < src_len; )
|
||||
{
|
||||
int tag = GetTextTag(&src[i], src_len - i, len, extraInfo);
|
||||
|
||||
if (tag == TEXT_TAG_PLAIN || tag == TEXT_TAG_TAG)
|
||||
{
|
||||
if (hyperlinkStep == 0)
|
||||
{
|
||||
++output_len;
|
||||
dst += src[i];
|
||||
}
|
||||
}
|
||||
else if (tag == TEXT_TAG_HYPERLINK_START)
|
||||
hyperlinkStep = 1;
|
||||
else if (tag == TEXT_TAG_HYPERLINK_END)
|
||||
hyperlinkStep = 0;
|
||||
|
||||
i += len;
|
||||
}
|
||||
return dst;
|
||||
}
|
||||
|
||||
int GetTextTagInternalPosFromRenderPos(const wchar_t * src, int src_len, int offset)
|
||||
{
|
||||
int len;
|
||||
std::wstring dst;
|
||||
std::wstring extraInfo;
|
||||
int output_len = 0;
|
||||
int hyperlinkStep = 0;
|
||||
bool color_tag = false;
|
||||
int internal_offset = 0;
|
||||
|
||||
for (int i = 0; i < src_len; )
|
||||
{
|
||||
int tag = GetTextTag(&src[i], src_len - i, len, extraInfo);
|
||||
|
||||
if (tag == TEXT_TAG_COLOR)
|
||||
{
|
||||
color_tag = true;
|
||||
internal_offset = i;
|
||||
}
|
||||
else if (tag == TEXT_TAG_RESTORE_COLOR)
|
||||
{
|
||||
color_tag = false;
|
||||
}
|
||||
else if (tag == TEXT_TAG_PLAIN || tag == TEXT_TAG_TAG)
|
||||
{
|
||||
if (hyperlinkStep == 0)
|
||||
{
|
||||
if (!color_tag)
|
||||
internal_offset = i;
|
||||
|
||||
if (offset <= output_len)
|
||||
return internal_offset;
|
||||
|
||||
++output_len;
|
||||
dst += src[i];
|
||||
}
|
||||
}
|
||||
else if (tag == TEXT_TAG_HYPERLINK_START)
|
||||
hyperlinkStep = 1;
|
||||
else if (tag == TEXT_TAG_HYPERLINK_END)
|
||||
hyperlinkStep = 0;
|
||||
|
||||
i += len;
|
||||
}
|
||||
|
||||
return internal_offset;
|
||||
}
|
||||
|
||||
int GetTextTagOutputLen(const wchar_t * src, int src_len)
|
||||
{
|
||||
int len;
|
||||
std::wstring extraInfo;
|
||||
int output_len = 0;
|
||||
int hyperlinkStep = 0;
|
||||
|
||||
for (int i = 0; i < src_len; )
|
||||
{
|
||||
int tag = GetTextTag(&src[i], src_len - i, len, extraInfo);
|
||||
|
||||
if (tag == TEXT_TAG_PLAIN || tag == TEXT_TAG_TAG)
|
||||
{
|
||||
if (hyperlinkStep == 0)
|
||||
++output_len;
|
||||
}
|
||||
else if (tag == TEXT_TAG_HYPERLINK_START)
|
||||
hyperlinkStep = 1;
|
||||
else if (tag == TEXT_TAG_HYPERLINK_END)
|
||||
hyperlinkStep = 0;
|
||||
|
||||
i += len;
|
||||
}
|
||||
return output_len;
|
||||
}
|
||||
|
||||
int FindColorTagStartPosition(const wchar_t * src, int src_len)
|
||||
{
|
||||
if (src_len < 2)
|
||||
return 0;
|
||||
|
||||
const wchar_t * cur = src;
|
||||
|
||||
// |r의 경우
|
||||
if (*cur == L'r' && *(cur - 1) == L'|')
|
||||
{
|
||||
int len = src_len;
|
||||
|
||||
// ||r은 무시
|
||||
if (len >= 2 && *(cur - 2) == L'|')
|
||||
return 1;
|
||||
|
||||
cur -= 2;
|
||||
len -= 2;
|
||||
|
||||
// |c까지 찾아서 |위치까지 리턴한다.
|
||||
while (len > 1) // 최소 2자를 검사해야 된다.
|
||||
{
|
||||
if (*cur == L'c' && *(cur - 1) == L'|')
|
||||
return (src - cur) + 1;
|
||||
|
||||
--cur;
|
||||
--len;
|
||||
}
|
||||
return (src_len); // 못찾으면 전부;;
|
||||
}
|
||||
// ||의 경우
|
||||
else if (*cur == L'|' && *(cur - 1) == L'|')
|
||||
return 1;
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
int FindColorTagEndPosition(const wchar_t * src, int src_len)
|
||||
{
|
||||
const wchar_t * cur = src;
|
||||
|
||||
if (src_len >= 4 && *cur == L'|' && *(cur + 1) == L'c')
|
||||
{
|
||||
int left = src_len - 2;
|
||||
cur += 2;
|
||||
|
||||
while (left > 1)
|
||||
{
|
||||
if (*cur == L'|' && *(cur + 1) == L'r')
|
||||
return (cur - src) + 1;
|
||||
|
||||
--left;
|
||||
++cur;
|
||||
}
|
||||
}
|
||||
else if (src_len >= 2 && *cur == L'|' && *(cur + 1) == L'|')
|
||||
return 1;
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,17 @@
|
||||
#pragma once // PORT: 40250 TextTag.h has no include guard; the header gate includes every header twice.
|
||||
enum
|
||||
{
|
||||
TEXT_TAG_PLAIN,
|
||||
TEXT_TAG_TAG, // ||
|
||||
TEXT_TAG_COLOR, // |cffffffff
|
||||
TEXT_TAG_HYPERLINK_START, // |H
|
||||
TEXT_TAG_HYPERLINK_END, // |h ex) |Hitem:1234:1:1:1|h
|
||||
TEXT_TAG_RESTORE_COLOR,
|
||||
};
|
||||
|
||||
extern int GetTextTag(const wchar_t * src, int maxLen, int & tagLen, std::wstring & extraInfo);
|
||||
extern std::wstring GetTextTagOutputString(const wchar_t * src, int src_len);
|
||||
extern int GetTextTagOutputLen(const wchar_t * src, int len);
|
||||
extern int FindColorTagEndPosition(const wchar_t * src, int src_len);
|
||||
extern int FindColorTagStartPosition(const wchar_t * src, int src_len);
|
||||
extern int GetTextTagInternalPosFromRenderPos(const wchar_t * src, int src_len, int offset);
|
||||
@@ -0,0 +1,366 @@
|
||||
#include "StdAfx.h"
|
||||
#include "Arabic.h"
|
||||
#include <assert.h>
|
||||
|
||||
enum ARABIC_CODE
|
||||
{
|
||||
ARABIC_CODE_BASE = 0x0621,
|
||||
ARABIC_CODE_LAST = 0x064a,
|
||||
ARABIC_CODE_COUNT = ARABIC_CODE_LAST - ARABIC_CODE_BASE + 1,
|
||||
};
|
||||
|
||||
enum ARABIC_FORM_TYPE
|
||||
{
|
||||
DEBUG_CODE,
|
||||
ISOLATED,
|
||||
INITIAL,
|
||||
MEDIAL,
|
||||
FINAL,
|
||||
ARABIC_FORM_TYPE_NUM,
|
||||
};
|
||||
|
||||
bool Arabic_IsInSpace(wchar_t code)
|
||||
{
|
||||
switch (code)
|
||||
{
|
||||
case ' ':
|
||||
case '\t':
|
||||
return true;
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
bool Arabic_IsInSymbol(wchar_t code)
|
||||
{
|
||||
return (code >= 0x20 && code <= 0x2f) ||
|
||||
(code >= 0x3a && code <= 0x40) ||
|
||||
(code >= 0x5b && code <= 0x60) ||
|
||||
(code >= 0x7b && code <= 0x7e);
|
||||
}
|
||||
|
||||
|
||||
bool Arabic_IsInPresentation(wchar_t code)
|
||||
{
|
||||
return (code >= 0xfb50 && code <= 0xfdff) || (code >= 0xfe70 && code <= 0xfeff) || (code == 0x08);
|
||||
}
|
||||
|
||||
bool Arabic_HasPresentation(wchar_t* codes, int last)
|
||||
{
|
||||
while (last > 0)
|
||||
{
|
||||
if (Arabic_IsInSpace(codes[last]))
|
||||
{
|
||||
last--;
|
||||
}
|
||||
else
|
||||
{
|
||||
if (Arabic_IsInPresentation(codes[last]))
|
||||
return true;
|
||||
else
|
||||
break;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
wchar_t Arabic_GetComposition(wchar_t cur, wchar_t next, ARABIC_FORM_TYPE pos)
|
||||
{
|
||||
static wchar_t LAM_ALEF_MADDA[ARABIC_FORM_TYPE_NUM] = {0x0622, 0xFEF5, 0, 0, 0xFEF6};
|
||||
static wchar_t LAM_ALEF_HAMZA_ABOVE[ARABIC_FORM_TYPE_NUM] = {0x0623, 0xFEF7, 0, 0, 0xFEF8};
|
||||
static wchar_t LAM_ALEF_HAMZA_BELOW[ARABIC_FORM_TYPE_NUM] = {0x0625, 0xFEF9, 0, 0, 0xFEFA};
|
||||
static wchar_t LAM_ALEF[ARABIC_FORM_TYPE_NUM] = {0x0627, 0xFEFB, 0, 0, 0xFEFC};
|
||||
|
||||
switch (cur)
|
||||
{
|
||||
case 0x0644:
|
||||
switch (next)
|
||||
{
|
||||
case 0x0622:return LAM_ALEF_MADDA[pos];break;
|
||||
case 0x0623:return LAM_ALEF_HAMZA_ABOVE[pos];break;
|
||||
case 0x0625:return LAM_ALEF_HAMZA_BELOW[pos];break;
|
||||
case 0x0627:return LAM_ALEF[pos];break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
wchar_t Arabic_GetMap(wchar_t code, ARABIC_FORM_TYPE pos)
|
||||
{
|
||||
static wchar_t HAMZA[ARABIC_FORM_TYPE_NUM] = {0x0621 , 0xFE80, 0, 0, 0};
|
||||
static wchar_t ALEF_MADDA[ARABIC_FORM_TYPE_NUM] = {0x0622 , 0xFE81, 0, 0, 0xFE82};
|
||||
static wchar_t ALEF_HAMZA_ABOVE[ARABIC_FORM_TYPE_NUM] = {0x0623 , 0xFE83, 0, 0, 0xFE84};
|
||||
static wchar_t WAW_HAMZA[ARABIC_FORM_TYPE_NUM] = {0x0624 , 0xFE85, 0, 0, 0xFE86};
|
||||
static wchar_t ALEF_HAMZA_BELOW[ARABIC_FORM_TYPE_NUM] = {0x0625 , 0xFE87, 0, 0, 0xFE88};
|
||||
static wchar_t YEH_HAMZA[ARABIC_FORM_TYPE_NUM] = {0x0626 , 0xFE89, 0xFE8B, 0xFE8C, 0xFE8A};
|
||||
static wchar_t ALEF[ARABIC_FORM_TYPE_NUM] = {0x0627 , 0xFE8D, 0, 0, 0xFE8E};
|
||||
static wchar_t BEH[ARABIC_FORM_TYPE_NUM] = {0x0628 , 0xFE8F, 0xFE91, 0xFE92, 0xFE90};
|
||||
static wchar_t TEH_MARBUTA[ARABIC_FORM_TYPE_NUM] = {0x0629 , 0xFE93, 0, 0, 0xFE94};
|
||||
static wchar_t TEH[ARABIC_FORM_TYPE_NUM] = {0x062A , 0xFE95, 0xFE97, 0xFE98, 0xFE96};
|
||||
static wchar_t THEH[ARABIC_FORM_TYPE_NUM] = {0x062B , 0xFE99, 0xFE9B, 0xFE9C, 0xFE9A};
|
||||
static wchar_t JEEM[ARABIC_FORM_TYPE_NUM] = {0x062C , 0xFE9D, 0xFE9F, 0xFEA0, 0xFE9E};
|
||||
static wchar_t HAH[ARABIC_FORM_TYPE_NUM] = {0x062D , 0xFEA1, 0xFEA3, 0xFEA4, 0xFEA2};
|
||||
static wchar_t KHAH[ARABIC_FORM_TYPE_NUM] = {0x062E , 0xFEA5, 0xFEA7, 0xFEA8, 0xFEA6};
|
||||
static wchar_t DAL[ARABIC_FORM_TYPE_NUM] = {0x062F , 0xFEA9, 0, 0, 0xFEAA};
|
||||
static wchar_t THAL[ARABIC_FORM_TYPE_NUM] = {0x0630 , 0xFEAB, 0, 0, 0xFEAC};
|
||||
static wchar_t REH[ARABIC_FORM_TYPE_NUM] = {0x0631 , 0xFEAD, 0, 0, 0xFEAE};
|
||||
static wchar_t ZAIN[ARABIC_FORM_TYPE_NUM] = {0x0632 , 0xFEAF, 0, 0, 0xFEB0};
|
||||
static wchar_t SEEN[ARABIC_FORM_TYPE_NUM] = {0x0633 , 0xFEB1, 0xFEB3, 0xFEB4, 0xFEB2};
|
||||
static wchar_t SHEEN[ARABIC_FORM_TYPE_NUM] = {0x0634 , 0xFEB5, 0xFEB7, 0xFEB8, 0xFEB6};
|
||||
static wchar_t SAD[ARABIC_FORM_TYPE_NUM] = {0x0635 , 0xFEB9, 0xFEBB, 0xFEBC, 0xFEBA};
|
||||
static wchar_t DAD[ARABIC_FORM_TYPE_NUM] = {0x0636 , 0xFEBD, 0xFEBF, 0xFEC0, 0xFEBE};
|
||||
static wchar_t TAH[ARABIC_FORM_TYPE_NUM] = {0x0637 , 0xFEC1, 0xFEC3, 0xFEC4, 0xFEC2};
|
||||
static wchar_t ZAH[ARABIC_FORM_TYPE_NUM] = {0x0638 , 0xFEC5, 0xFEC7, 0xFEC8, 0xFEC6};
|
||||
static wchar_t AIN[ARABIC_FORM_TYPE_NUM] = {0x0639 , 0xFEC9, 0xFECB, 0xFECC, 0xFECA};
|
||||
static wchar_t GHAIN[ARABIC_FORM_TYPE_NUM] = {0x063A , 0xFECD, 0xFECF, 0xFED0, 0xFECE};
|
||||
static wchar_t TATWEEL[ARABIC_FORM_TYPE_NUM] = {0x0640 , 0x0640, 0, 0, 0};
|
||||
static wchar_t OX_FEH[ARABIC_FORM_TYPE_NUM] = {0x0641 , 0xFED1, 0xFED3, 0xFED4, 0xFED2};
|
||||
static wchar_t QAF[ARABIC_FORM_TYPE_NUM] = {0x0642 , 0xFED5, 0xFED7, 0xFED8, 0xFED6};
|
||||
static wchar_t KAF[ARABIC_FORM_TYPE_NUM] = {0x0643 , 0xFED9, 0xFEDB, 0xFEDC, 0xFEDA};
|
||||
static wchar_t LAM[ARABIC_FORM_TYPE_NUM] = {0x0644 , 0xFEDD, 0xFEDF, 0xFEE0, 0xFEDE};
|
||||
static wchar_t MEEM[ARABIC_FORM_TYPE_NUM] = {0x0645 , 0xFEE1, 0xFEE3, 0xFEE4, 0xFEE2};
|
||||
static wchar_t NOON[ARABIC_FORM_TYPE_NUM] = {0x0646 , 0xFEE5, 0xFEE7, 0xFEE8, 0xFEE6};
|
||||
static wchar_t HEH[ARABIC_FORM_TYPE_NUM] = {0x0647 , 0xFEE9, 0xFEEB, 0xFEEC, 0xFEEA};
|
||||
static wchar_t WAW[ARABIC_FORM_TYPE_NUM] = {0x0648 , 0xFEED, 0, 0, 0xFEEE};
|
||||
static wchar_t ALEF_MAKSURA[ARABIC_FORM_TYPE_NUM] = {0x0649 , 0xFEEF, 0, 0, 0xFEF0};
|
||||
static wchar_t YEH[ARABIC_FORM_TYPE_NUM] = {0x064A , 0xFEF1, 0xFEF3, 0xFEF4, 0xFEF2};
|
||||
|
||||
switch (code)
|
||||
{
|
||||
case 0x0621: return HAMZA[pos];break;
|
||||
case 0x0622: return ALEF_MADDA[pos];break;
|
||||
case 0x0623: return ALEF_HAMZA_ABOVE[pos];break;
|
||||
case 0x0624: return WAW_HAMZA[pos];break;
|
||||
case 0x0625: return ALEF_HAMZA_BELOW[pos];break;
|
||||
case 0x0626: return YEH_HAMZA[pos];break;
|
||||
case 0x0627: return ALEF[pos];break;
|
||||
case 0x0628: return BEH[pos];break;
|
||||
case 0x0629: return TEH_MARBUTA[pos];break;
|
||||
case 0x062A: return TEH[pos];break;
|
||||
case 0x062B: return THEH[pos];break;
|
||||
case 0x062C: return JEEM[pos];break;
|
||||
case 0x062D: return HAH[pos];break;
|
||||
case 0x062E: return KHAH[pos];break;
|
||||
case 0x062F: return DAL[pos];break;
|
||||
case 0x0630: return THAL[pos];break;
|
||||
case 0x0631: return REH[pos];break;
|
||||
case 0x0632: return ZAIN[pos];break;
|
||||
case 0x0633: return SEEN[pos];break;
|
||||
case 0x0634: return SHEEN[pos];break;
|
||||
case 0x0635: return SAD[pos];break;
|
||||
case 0x0636: return DAD[pos];break;
|
||||
case 0x0637: return TAH[pos];break;
|
||||
case 0x0638: return ZAH[pos];break;
|
||||
case 0x0639: return AIN[pos];break;
|
||||
case 0x063A: return GHAIN[pos];break;
|
||||
case 0x0640: return TATWEEL[pos];break;
|
||||
case 0x0641: return OX_FEH[pos];break;
|
||||
case 0x0642: return QAF[pos];break;
|
||||
case 0x0643: return KAF[pos];break;
|
||||
case 0x0644: return LAM[pos];break;
|
||||
case 0x0645: return MEEM[pos];break;
|
||||
case 0x0646: return NOON[pos];break;
|
||||
case 0x0647: return HEH[pos];break;
|
||||
case 0x0648: return WAW[pos];break;
|
||||
case 0x0649: return ALEF_MAKSURA[pos];break;
|
||||
case 0x064A: return YEH[pos];break;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
bool Arabic_IsInMap(wchar_t code)
|
||||
{
|
||||
return (code >= ARABIC_CODE_BASE && code <= ARABIC_CODE_LAST);
|
||||
}
|
||||
|
||||
bool Arabic_IsInComposing(wchar_t code)
|
||||
{
|
||||
switch (code)
|
||||
{
|
||||
case 0x064B: // FATHATAN
|
||||
case 0x064C: // DAMMATAN
|
||||
case 0x064D: // KASRATAN
|
||||
case 0x064E: // FATHA
|
||||
case 0x064F: // DAMMA
|
||||
case 0x0650: // KASRA
|
||||
case 0x0651: // SHADDA
|
||||
case 0x0652: // SUKUN
|
||||
case 0x0653: // MADDAH ABOVE
|
||||
case 0x0654: // HAMZA ABOVE
|
||||
case 0x0655: // HAMZA BELOW
|
||||
case 0x0670: // SUPERSCRIPT ALEF
|
||||
case 0x06D6: // HIGH LIG. SAD WITH LAM WITH ALEF MAKSURA
|
||||
case 0x06D7: // HIGH LIG. QAF WITH LAM WITH ALEF MAKSURA
|
||||
case 0x06D8: // HIGH MEEM INITIAL FORM
|
||||
case 0x06D9: // HIGH LAM ALEF
|
||||
case 0x06DA: // HIGH JEEM
|
||||
case 0x06DB: // HIGH THREE DOTS
|
||||
case 0x06DC: // HIGH SEEN
|
||||
|
||||
// The 2 entires below should not be here - contact unicode.org !!
|
||||
// case 0x06DD: // END OF AYAH
|
||||
// case 0x06DE: // START OF RUB EL HIZB
|
||||
|
||||
case 0x06DF: // HIGH ROUNDED ZERO
|
||||
case 0x06E0: // HIGH UPRIGHT RECTANGULAR ZERO
|
||||
case 0x06E1: // HIGH DOTLESS HEAD OF KHAH
|
||||
case 0x06E2: // HIGH MEEM ISOLATED FORM
|
||||
case 0x06E3: // LOW SEEN
|
||||
case 0x06E4: // HIGH MADDA
|
||||
case 0x06E7: // HIGH YEH
|
||||
case 0x06E8: // HIGH NOON
|
||||
case 0x06EA: // EMPTY CENTRE LOW STOP
|
||||
case 0x06EB: // EMPTY CENTRE HIGH STOP
|
||||
case 0x06EC: // HIGH STOP WITH FILLED CENTRE
|
||||
case 0x06ED: // LOW MEEM
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
bool Arabic_IsNext(wchar_t code)
|
||||
{
|
||||
return (code == 0x0640);
|
||||
}
|
||||
|
||||
bool Arabic_IsComb1(wchar_t code)
|
||||
{
|
||||
return (code == 0x0644);
|
||||
}
|
||||
|
||||
bool Arabic_IsComb2(wchar_t code)
|
||||
{
|
||||
switch (code)
|
||||
{
|
||||
case 0x0622:
|
||||
case 0x0623:
|
||||
case 0x0625:
|
||||
case 0x0627:
|
||||
return true;
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
size_t Arabic_MakeShape(wchar_t* src, size_t srcLen, wchar_t* dst, size_t dstLen)
|
||||
{
|
||||
assert(dstLen >= srcLen);
|
||||
|
||||
const size_t srcLastIndex = srcLen - 1;
|
||||
|
||||
unsigned dstIndex = 0;
|
||||
for (size_t srcIndex = 0; srcIndex < srcLen; ++srcIndex)
|
||||
{
|
||||
wchar_t cur = src[srcIndex];
|
||||
|
||||
//printf("now %x\n", cur);
|
||||
|
||||
if (Arabic_IsInMap(cur))
|
||||
{
|
||||
// 이전 글자 얻어내기
|
||||
wchar_t prev = 0;
|
||||
{
|
||||
size_t prevIndex = srcIndex;
|
||||
while (prevIndex > 0)
|
||||
{
|
||||
prevIndex--;
|
||||
prev = src[prevIndex];
|
||||
//printf("\tprev %d:%x\n", prevIndex, cur);
|
||||
if (Arabic_IsInComposing(prev))
|
||||
continue;
|
||||
else
|
||||
break;
|
||||
}
|
||||
|
||||
if ((srcIndex == 0) ||
|
||||
(!Arabic_IsInMap(prev)) ||
|
||||
(!Arabic_GetMap(prev, INITIAL) && !Arabic_GetMap(prev, MEDIAL)))
|
||||
{
|
||||
//printf("\tprev not defined\n");
|
||||
prev = 0;
|
||||
}
|
||||
}
|
||||
|
||||
// 다음 글자 얻어내기
|
||||
wchar_t next = 0;
|
||||
{
|
||||
size_t nextIndex = srcIndex;
|
||||
while (nextIndex < srcLastIndex)
|
||||
{
|
||||
nextIndex++;
|
||||
next = src[nextIndex];
|
||||
if (Arabic_IsInComposing(next))
|
||||
continue;
|
||||
else
|
||||
break;
|
||||
}
|
||||
|
||||
if ((nextIndex == srcLen) ||
|
||||
(!Arabic_IsInMap(next)) ||
|
||||
(!Arabic_GetMap(next, MEDIAL) && !Arabic_GetMap(next, FINAL) && !Arabic_IsNext(next)))
|
||||
{
|
||||
//printf("\tnext not defined\n");
|
||||
next = 0;
|
||||
}
|
||||
}
|
||||
|
||||
if (Arabic_IsComb1(cur) && Arabic_IsComb2(next))
|
||||
{
|
||||
if (prev)
|
||||
dst[dstIndex] = Arabic_GetComposition(cur, next, FINAL);
|
||||
else
|
||||
dst[dstIndex] = Arabic_GetComposition(cur, next, ISOLATED);
|
||||
|
||||
//printf("\tGot me a complex:%x\n", dst[dstIndex]);
|
||||
|
||||
srcIndex++;
|
||||
dstIndex++;
|
||||
}
|
||||
else if (prev && next && (dst[dstIndex] = Arabic_GetMap(cur, MEDIAL)))
|
||||
{
|
||||
//printf("\tGot prev & next:%x\n", dst[dstIndex]);
|
||||
dstIndex++;
|
||||
}
|
||||
else if (prev && (dst[dstIndex] = Arabic_GetMap(cur, FINAL)))
|
||||
{
|
||||
//printf("\tGot prev:%x\n", dst[dstIndex]);
|
||||
dstIndex++;
|
||||
}
|
||||
else if (next && (dst[dstIndex] = Arabic_GetMap(cur, INITIAL)))
|
||||
{
|
||||
//printf("\tGot next:%x\n", dst[dstIndex]);
|
||||
dstIndex++;
|
||||
}
|
||||
else
|
||||
{
|
||||
dst[dstIndex] = Arabic_GetMap(cur, ISOLATED);
|
||||
//printf("\tGot nothing:%x\n", dst[dstIndex]);
|
||||
dstIndex++;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
dst[dstIndex] = cur;
|
||||
dstIndex++;
|
||||
}
|
||||
}
|
||||
return dstIndex;
|
||||
}
|
||||
|
||||
wchar_t Arabic_ConvSymbol(wchar_t c)
|
||||
{
|
||||
switch (c)
|
||||
{
|
||||
case '(': return')';break;
|
||||
case ')': return'(';break;
|
||||
case '<': return'>';break;
|
||||
case '>': return'<';break;
|
||||
case '{': return'}';break;
|
||||
case '}': return'{';break;
|
||||
case '[': return']';break;
|
||||
case ']': return '[';break;
|
||||
default:
|
||||
return c;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,11 @@
|
||||
#pragma once
|
||||
|
||||
wchar_t Arabic_ConvSymbol(wchar_t c);
|
||||
|
||||
bool Arabic_IsInSpace(wchar_t code);
|
||||
bool Arabic_IsInSymbol(wchar_t code);
|
||||
bool Arabic_IsInPresentation(wchar_t code);
|
||||
bool Arabic_HasPresentation(wchar_t* codes, int last);
|
||||
size_t Arabic_MakeShape(wchar_t* src, size_t srcLen, wchar_t* dst, size_t dstLen);
|
||||
wchar_t Arabic_ConvEnglishModeSymbol(wchar_t code);
|
||||
|
||||
@@ -0,0 +1,43 @@
|
||||
#include "StdAfx.h"
|
||||
#include "StringCodec.h"
|
||||
#include "StringCodec_Vietnamese.h"
|
||||
|
||||
int Ymir_WideCharToMultiByte(
|
||||
UINT CodePage,
|
||||
DWORD dwFlags,
|
||||
LPCWSTR lpWideCharStr,
|
||||
int cchWideChar,
|
||||
LPSTR lpMultiByteStr,
|
||||
int cbMultiByte,
|
||||
LPCSTR lpDefaultChar,
|
||||
LPBOOL lpUsedDefaultChar
|
||||
)
|
||||
{
|
||||
if (CodePage == CP_1258)
|
||||
{
|
||||
return EL_String_Encode_Vietnamese(lpWideCharStr, cchWideChar, lpMultiByteStr, cbMultiByte);
|
||||
}
|
||||
else
|
||||
{
|
||||
return WideCharToMultiByte(CodePage, dwFlags, lpWideCharStr, cchWideChar, lpMultiByteStr, cbMultiByte, lpDefaultChar, lpUsedDefaultChar);
|
||||
}
|
||||
}
|
||||
|
||||
int Ymir_MultiByteToWideChar(
|
||||
UINT CodePage,
|
||||
DWORD dwFlags,
|
||||
LPCSTR lpMultiByteStr,
|
||||
int cbMultiByte,
|
||||
LPWSTR lpWideCharStr,
|
||||
int cchWideChar
|
||||
)
|
||||
{
|
||||
if (CodePage == CP_1258)
|
||||
{
|
||||
return EL_String_Decode_Vietnamese(lpMultiByteStr, cbMultiByte, lpWideCharStr, cchWideChar);
|
||||
}
|
||||
else
|
||||
{
|
||||
return MultiByteToWideChar(CodePage, dwFlags, lpMultiByteStr, cbMultiByte, lpWideCharStr, cchWideChar);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,21 @@
|
||||
#pragma once
|
||||
|
||||
int Ymir_WideCharToMultiByte(
|
||||
UINT CodePage,
|
||||
DWORD dwFlags,
|
||||
LPCWSTR lpWideCharStr,
|
||||
int cchWideChar,
|
||||
LPSTR lpMultiByteStr,
|
||||
int cbMultiByte,
|
||||
LPCSTR lpDefaultChar,
|
||||
LPBOOL lpUsedDefaultChar
|
||||
);
|
||||
|
||||
int Ymir_MultiByteToWideChar(
|
||||
UINT CodePage,
|
||||
DWORD dwFlags,
|
||||
LPCSTR lpMultiByteStr,
|
||||
int cbMultiByte,
|
||||
LPWSTR lpWideCharStr,
|
||||
int cchWideChar
|
||||
);
|
||||
@@ -0,0 +1,554 @@
|
||||
#include "StdAfx.h"
|
||||
#include "StringCodec_Vietnamese.h"
|
||||
|
||||
#pragma warning(disable: 4310) // char 짤림 경고 무시
|
||||
|
||||
static wchar_t cp1258_to_unicode[256] = {
|
||||
0x0000, 0x0001, 0x0002, 0x0003, 0x0004, 0x0005, 0x0006, 0x0007,
|
||||
0x0008, 0x0009, 0x000a, 0x000b, 0x000c, 0x000d, 0x000e, 0x000f,
|
||||
0x0010, 0x0011, 0x0012, 0x0013, 0x0014, 0x0015, 0x0016, 0x0017,
|
||||
0x0018, 0x0019, 0x001a, 0x001b, 0x001c, 0x001d, 0x001e, 0x001f,
|
||||
0x0020, 0x0021, 0x0022, 0x0023, 0x0024, 0x0025, 0x0026, 0x0027,
|
||||
0x0028, 0x0029, 0x002a, 0x002b, 0x002c, 0x002d, 0x002e, 0x002f,
|
||||
0x0030, 0x0031, 0x0032, 0x0033, 0x0034, 0x0035, 0x0036, 0x0037,
|
||||
0x0038, 0x0039, 0x003a, 0x003b, 0x003c, 0x003d, 0x003e, 0x003f,
|
||||
0x0040, 0x0041, 0x0042, 0x0043, 0x0044, 0x0045, 0x0046, 0x0047,
|
||||
0x0048, 0x0049, 0x004a, 0x004b, 0x004c, 0x004d, 0x004e, 0x004f,
|
||||
0x0050, 0x0051, 0x0052, 0x0053, 0x0054, 0x0055, 0x0056, 0x0057,
|
||||
0x0058, 0x0059, 0x005a, 0x005b, 0x005c, 0x005d, 0x005e, 0x005f,
|
||||
0x0060, 0x0061, 0x0062, 0x0063, 0x0064, 0x0065, 0x0066, 0x0067,
|
||||
0x0068, 0x0069, 0x006a, 0x006b, 0x006c, 0x006d, 0x006e, 0x006f,
|
||||
0x0070, 0x0071, 0x0072, 0x0073, 0x0074, 0x0075, 0x0076, 0x0077,
|
||||
0x0078, 0x0079, 0x007a, 0x007b, 0x007c, 0x007d, 0x007e, 0x007f,
|
||||
0x20ac, 0x0081, 0x201a, 0x0192, 0x201e, 0x2026, 0x2020, 0x2021, //0x80
|
||||
0x02c6, 0x2030, 0x008a, 0x2039, 0x0152, 0x008d, 0x008e, 0x008f, //
|
||||
0x0090, 0x2018, 0x2019, 0x201c, 0x201d, 0x2022, 0x2013, 0x2014, //0x90
|
||||
0x02dc, 0x2122, 0x009a, 0x203a, 0x0153, 0x009d, 0x009e, 0x0178, //
|
||||
0x00a0, 0x00a1, 0x00a2, 0x00a3, 0x00a4, 0x00a5, 0x00a6, 0x00a7, //0xa0
|
||||
0x00a8, 0x00a9, 0x00aa, 0x00ab, 0x00ac, 0x00ad, 0x00ae, 0x00af, //
|
||||
0x00b0, 0x00b1, 0x00b2, 0x00b3, 0x00b4, 0x00b5, 0x00b6, 0x00b7, //0xb0
|
||||
0x00b8, 0x00b9, 0x00ba, 0x00bb, 0x00bc, 0x00bd, 0x00be, 0x00bf, //
|
||||
0x00c0, 0x00c1, 0x00c2, 0x0102, 0x00c4, 0x00c5, 0x00c6, 0x00c7, //0xc0
|
||||
0x00c8, 0x00c9, 0x00ca, 0x00cb, 0x0300, 0x00cd, 0x00ce, 0x00cf, //
|
||||
0x0110, 0x00d1, 0x0309, 0x00d3, 0x00d4, 0x01a0, 0x00d6, 0x00d7, //0xd0
|
||||
0x00d8, 0x00d9, 0x00da, 0x00db, 0x00dc, 0x01af, 0x0303, 0x00df, //
|
||||
0x00e0, 0x00e1, 0x00e2, 0x0103, 0x00e4, 0x00e5, 0x00e6, 0x00e7, //0xe0
|
||||
0x00e8, 0x00e9, 0x00ea, 0x00eb, 0x0301, 0x00ed, 0x00ee, 0x00ef, //
|
||||
0x0111, 0x00f1, 0x0323, 0x00f3, 0x00f4, 0x01a1, 0x00f6, 0x00f7, //0xf0
|
||||
0x00f8, 0x00f9, 0x00fa, 0x00fb, 0x00fc, 0x01b0, 0x20ab, 0x00ff, //
|
||||
};
|
||||
|
||||
static wchar_t cp1258_composed_table[][5] = {
|
||||
{ 0x00c1, 0x00c0, 0x1ea2, 0x00c3, 0x1ea0 },
|
||||
{ 0x00e1, 0x00e0, 0x1ea3, 0x00e3, 0x1ea1 },
|
||||
{ 0x1eae, 0x1eb0, 0x1eb2, 0x1eb4, 0x1eb6 },
|
||||
{ 0x1eaf, 0x1eb1, 0x1eb3, 0x1eb5, 0x1eb7 },
|
||||
{ 0x1ea4, 0x1ea6, 0x1ea8, 0x1eaa, 0x1eac },
|
||||
{ 0x1ea5, 0x1ea7, 0x1ea9, 0x1eab, 0x1ead },
|
||||
{ 0x00c9, 0x00c8, 0x1eba, 0x1ebc, 0x1eb8 },
|
||||
{ 0x00e9, 0x00e8, 0x1ebb, 0x1ebd, 0x1eb9 },
|
||||
{ 0x1ebe, 0x1ec0, 0x1ec2, 0x1ec4, 0x1ec6 },
|
||||
{ 0x1ebf, 0x1ec1, 0x1ec3, 0x1ec5, 0x1ec7 },
|
||||
{ 0x00cd, 0x00cc, 0x1ec8, 0x0128, 0x1eca },
|
||||
{ 0x00ed, 0x00ec, 0x1ec9, 0x0129, 0x1ecb },
|
||||
{ 0x00d3, 0x00d2, 0x1ece, 0x00d5, 0x1ecc },
|
||||
{ 0x00f3, 0x00f2, 0x1ecf, 0x00f5, 0x1ecd },
|
||||
{ 0x1ed0, 0x1ed2, 0x1ed4, 0x1ed6, 0x1ed8 },
|
||||
{ 0x1ed1, 0x1ed3, 0x1ed5, 0x1ed7, 0x1ed9 },
|
||||
{ 0x1eda, 0x1edc, 0x1ede, 0x1ee0, 0x1ee2 },
|
||||
{ 0x1edb, 0x1edd, 0x1edf, 0x1ee1, 0x1ee3 },
|
||||
{ 0x00da, 0x00d9, 0x1ee6, 0x0168, 0x1ee4 },
|
||||
{ 0x00fa, 0x00f9, 0x1ee7, 0x0169, 0x1ee5 },
|
||||
{ 0x1ee8, 0x1eea, 0x1eec, 0x1eee, 0x1ef0 },
|
||||
{ 0x1ee9, 0x1eeb, 0x1eed, 0x1eef, 0x1ef1 },
|
||||
{ 0x00dd, 0x1ef2, 0x1ef6, 0x1ef8, 0x1ef4 },
|
||||
{ 0x00fd, 0x1ef3, 0x1ef7, 0x1ef9, 0x1ef5 },
|
||||
};
|
||||
|
||||
static bool IsTone(wchar_t tone)
|
||||
{
|
||||
switch(tone)
|
||||
{
|
||||
case 0x0300:
|
||||
case 0x0301:
|
||||
case 0x0309:
|
||||
case 0x0303:
|
||||
case 0x0323:
|
||||
return true;
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
static wchar_t ComposeTone(wchar_t prev, wchar_t tone)
|
||||
{
|
||||
int col, row;
|
||||
|
||||
switch(tone)
|
||||
{
|
||||
case 0x0301: col = 0; break;
|
||||
case 0x0300: col = 1; break;
|
||||
case 0x0309: col = 2; break;
|
||||
case 0x0303: col = 3; break;
|
||||
case 0x0323: col = 4; break;
|
||||
default:
|
||||
return prev;
|
||||
}
|
||||
|
||||
switch(prev)
|
||||
{
|
||||
case 0x0041: row = 0; break;
|
||||
case 0x0061: row = 1; break;
|
||||
case 0x0102: row = 2; break;
|
||||
case 0x0103: row = 3; break;
|
||||
case 0x00C2: row = 4; break;
|
||||
case 0x00E2: row = 5; break;
|
||||
case 0x0045: row = 6; break;
|
||||
case 0x0065: row = 7; break;
|
||||
case 0x00CA: row = 8; break;
|
||||
case 0x00EA: row = 9; break;
|
||||
case 0x0049: row = 10; break;
|
||||
case 0x0069: row = 11; break;
|
||||
case 0x004F: row = 12; break;
|
||||
case 0x006F: row = 13; break;
|
||||
case 0x00D4: row = 14; break;
|
||||
case 0x00F4: row = 15; break;
|
||||
case 0x01A0: row = 16; break;
|
||||
case 0x01A1: row = 17; break;
|
||||
case 0x0055: row = 18; break;
|
||||
case 0x0075: row = 19; break;
|
||||
case 0x01AF: row = 20; break;
|
||||
case 0x01B0: row = 21; break;
|
||||
case 0x0059: row = 22; break;
|
||||
case 0x0079: row = 23; break;
|
||||
default:
|
||||
return prev;
|
||||
}
|
||||
|
||||
return cp1258_composed_table[row][col];
|
||||
}
|
||||
|
||||
int EL_String_Decode_Vietnamese(const char* multi, int multiLen, wchar_t* wide, int wideLen)
|
||||
{
|
||||
if(multiLen < 0)
|
||||
multiLen = (int)strlen(multi) + 1;
|
||||
|
||||
int src = 0;
|
||||
int dest = 0;
|
||||
|
||||
if(multiLen > 0)
|
||||
{
|
||||
/* 첫글자는 무조건 변경 */
|
||||
wchar_t prev = cp1258_to_unicode[(BYTE)multi[src++]];
|
||||
|
||||
while(src < multiLen)
|
||||
{
|
||||
wchar_t unicode = cp1258_to_unicode[(BYTE)multi[src]];
|
||||
|
||||
/* 다음 문자가 Tone 인가? */
|
||||
if(IsTone(unicode))
|
||||
{
|
||||
/* 앞의 문자와 합하자. */
|
||||
prev = ComposeTone(prev, unicode);
|
||||
}
|
||||
else
|
||||
{
|
||||
/* 일반 문자가 왔다. 앞 문자를 변환 */
|
||||
if(dest < wideLen)
|
||||
wide[dest++] = prev;
|
||||
prev = unicode;
|
||||
}
|
||||
++src;
|
||||
}
|
||||
|
||||
if(dest < wideLen)
|
||||
wide[dest++] = prev;
|
||||
}
|
||||
|
||||
return dest;
|
||||
}
|
||||
|
||||
static bool DecomposeLetter(wchar_t input, char* letter)
|
||||
{
|
||||
switch(input)
|
||||
{
|
||||
case 0x00c1: // L'Á'
|
||||
case 0x00c0: // L'À'
|
||||
case 0x1ea2: // L'Ả'
|
||||
case 0x00c3: // L'Ã'
|
||||
case 0x1ea0: // L'Ạ'
|
||||
*letter = 'A';
|
||||
return true;
|
||||
case 0x00e1: // L'á'
|
||||
case 0x00e0: // L'à'
|
||||
case 0x1ea3: // L'ả'
|
||||
case 0x00e3: // L'ã'
|
||||
case 0x1ea1: // L'ạ'
|
||||
*letter = 'a';
|
||||
return true;
|
||||
case 0x1eae: // L'Ắ'
|
||||
case 0x1eb0: // L'Ằ'
|
||||
case 0x1eb2: // L'Ẳ'
|
||||
case 0x1eb4: // L'Ẵ'
|
||||
case 0x1eb6: // L'Ặ'
|
||||
*letter = (char)0xc3;
|
||||
return true;
|
||||
case 0x1eaf: // L'ắ'
|
||||
case 0x1eb1: // L'ằ'
|
||||
case 0x1eb3: // L'ẳ'
|
||||
case 0x1eb5: // L'ẵ'
|
||||
case 0x1eb7: // L'ặ'
|
||||
*letter = (char)0xe3;
|
||||
return true;
|
||||
case 0x1ea4: // L'Ấ'
|
||||
case 0x1ea6: // L'Ầ'
|
||||
case 0x1ea8: // L'Ẩ'
|
||||
case 0x1eaa: // L'Ẫ'
|
||||
case 0x1eac: // L'Ậ'
|
||||
*letter = (char)0xc2;
|
||||
return true;
|
||||
case 0x1ea5: // L'ấ'
|
||||
case 0x1ea7: // L'ầ'
|
||||
case 0x1ea9: // L'ẩ'
|
||||
case 0x1eab: // L'ẫ'
|
||||
case 0x1ead: // L'ậ'
|
||||
*letter = (char)0xe2;
|
||||
return true;
|
||||
case 0x00c9: // L'É'
|
||||
case 0x00c8: // L'È'
|
||||
case 0x1eba: // L'Ẻ'
|
||||
case 0x1ebc: // L'Ẽ'
|
||||
case 0x1eb8: // L'Ẹ'
|
||||
*letter = (char)'E';
|
||||
return true;
|
||||
case 0x00e9: // L'é'
|
||||
case 0x00e8: // L'è'
|
||||
case 0x1ebb: // L'ẻ'
|
||||
case 0x1ebd: // L'ẽ'
|
||||
case 0x1eb9: // L'ẹ'
|
||||
*letter = (char)'e';
|
||||
return true;
|
||||
case 0x1ebe: // L'Ế'
|
||||
case 0x1ec0: // L'Ề'
|
||||
case 0x1ec2: // L'Ể'
|
||||
case 0x1ec4: // L'Ễ'
|
||||
case 0x1ec6: // L'Ệ'
|
||||
*letter = (char)0xca;
|
||||
return true;
|
||||
case 0x1ebf: // L'ế'
|
||||
case 0x1ec1: // L'ề'
|
||||
case 0x1ec3: // L'ể'
|
||||
case 0x1ec5: // L'ễ'
|
||||
case 0x1ec7: // L'ệ'
|
||||
*letter = (char)0xea;
|
||||
return true;
|
||||
case 0x00cd: // L'Í'
|
||||
case 0x00cc: // L'Ì'
|
||||
case 0x1ec8: // L'Ỉ'
|
||||
case 0x0128: // L'Ĩ'
|
||||
case 0x1eca: // L'Ị'
|
||||
*letter = (char)'I';
|
||||
return true;
|
||||
case 0x00ed: // L'í'
|
||||
case 0x00ec: // L'ì'
|
||||
case 0x1ec9: // L'ỉ'
|
||||
case 0x0129: // L'ĩ'
|
||||
case 0x1ecb: // L'ị'
|
||||
*letter = (char)'i';
|
||||
return true;
|
||||
case 0x00d3: // L'Ó'
|
||||
case 0x00d2: // L'Ò'
|
||||
case 0x1ece: // L'Ỏ'
|
||||
case 0x00d5: // L'Õ'
|
||||
case 0x1ecc: // L'Ọ'
|
||||
*letter = (char)'O';
|
||||
return true;
|
||||
case 0x00f3: // L'ó'
|
||||
case 0x00f2: // L'ò'
|
||||
case 0x1ecf: // L'ỏ'
|
||||
case 0x00f5: // L'õ'
|
||||
case 0x1ecd: // L'ọ'
|
||||
*letter = (char)'o';
|
||||
return true;
|
||||
case 0x1ed0: // L'Ố'
|
||||
case 0x1ed2: // L'Ồ'
|
||||
case 0x1ed4: // L'Ổ'
|
||||
case 0x1ed6: // L'Ỗ'
|
||||
case 0x1ed8: // L'Ộ'
|
||||
*letter = (char)0xd4;
|
||||
return true;
|
||||
case 0x1ed1: // L'ố'
|
||||
case 0x1ed3: // L'ồ'
|
||||
case 0x1ed5: // L'ổ'
|
||||
case 0x1ed7: // L'ỗ'
|
||||
case 0x1ed9: // L'ộ'
|
||||
*letter = (char)0xf4;
|
||||
return true;
|
||||
case 0x1eda: // L'Ớ'
|
||||
case 0x1edc: // L'Ờ'
|
||||
case 0x1ede: // L'Ở'
|
||||
case 0x1ee0: // L'Ỡ'
|
||||
case 0x1ee2: // L'Ợ'
|
||||
*letter = (char)0xd5;
|
||||
return true;
|
||||
case 0x1edb: // L'ớ'
|
||||
case 0x1edd: // L'ờ'
|
||||
case 0x1edf: // L'ở'
|
||||
case 0x1ee1: // L'ỡ'
|
||||
case 0x1ee3: // L'ợ'
|
||||
*letter = (char)0xf5;
|
||||
return true;
|
||||
case 0x00da: // L'Ú'
|
||||
case 0x00d9: // L'Ù'
|
||||
case 0x1ee6: // L'Ủ'
|
||||
case 0x0168: // L'Ũ'
|
||||
case 0x1ee4: // L'Ụ'
|
||||
*letter = (char)'U';
|
||||
return true;
|
||||
case 0x00fa: // L'ú'
|
||||
case 0x00f9: // L'ù'
|
||||
case 0x1ee7: // L'ủ'
|
||||
case 0x0169: // L'ũ'
|
||||
case 0x1ee5: // L'ụ'
|
||||
*letter = (char)'u';
|
||||
return true;
|
||||
case 0x1ee8: // L'Ứ'
|
||||
case 0x1eea: // L'Ừ'
|
||||
case 0x1eec: // L'Ử'
|
||||
case 0x1eee: // L'Ữ'
|
||||
case 0x1ef0: // L'Ự'
|
||||
*letter = (char)0xdd;
|
||||
return true;
|
||||
case 0x1ee9: // L'ứ'
|
||||
case 0x1eeb: // L'ừ'
|
||||
case 0x1eed: // L'ử'
|
||||
case 0x1eef: // L'ữ'
|
||||
case 0x1ef1: // L'ự'
|
||||
*letter = (char)0xfd;
|
||||
return true;
|
||||
case 0x1ef2: // L'Ỳ'
|
||||
case 0x00dd: // L'Ý'
|
||||
case 0x1ef6: // L'Ỷ'
|
||||
case 0x1ef8: // L'Ỹ'
|
||||
case 0x1ef4: // L'Ỵ'
|
||||
*letter = (char)'Y';
|
||||
return true;
|
||||
case 0x1ef3: // L'ỳ'
|
||||
case 0x00fd: // L'ý'
|
||||
case 0x1ef7: // L'ỷ'
|
||||
case 0x1ef9: // L'ỹ'
|
||||
case 0x1ef5: // L'ỵ'
|
||||
*letter = (char)'y';
|
||||
return true;
|
||||
case 0x0102: // L'Ă'
|
||||
*letter = (char)0xc3;
|
||||
return true;
|
||||
case 0x0103: // L'ă'
|
||||
*letter = (char)0xe3;
|
||||
return true;
|
||||
case 0x0110: // L'Đ'
|
||||
*letter = (char)0xd0;
|
||||
return true;
|
||||
case 0x0111: // L'đ'
|
||||
*letter = (char)0xf0;
|
||||
return true;
|
||||
case 0x01a0: // L'Ơ'
|
||||
*letter = (char)0xd5;
|
||||
return true;
|
||||
case 0x01a1: // L'ơ'
|
||||
*letter = (char)0xf5;
|
||||
return true;
|
||||
case 0x01af: // L'Ư'
|
||||
*letter = (char)0xdd;
|
||||
return true;
|
||||
case 0x01b0: // L'ư'
|
||||
*letter = (char)0xfd;
|
||||
return true;
|
||||
case 0x20ab: // L'₫'
|
||||
*letter = (char)0xfe;
|
||||
return true;
|
||||
case 0x201c: // L'“'
|
||||
*letter = (char)'"';
|
||||
return true;
|
||||
case 0x201d: // L'”'
|
||||
*letter = (char)'"';
|
||||
return true;
|
||||
}
|
||||
|
||||
if(input < 256)
|
||||
{
|
||||
*letter = (char)input;
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
static bool DecomposeTone(wchar_t input, char* tone)
|
||||
{
|
||||
switch(input)
|
||||
{
|
||||
case 0x00c1: // L'Á'
|
||||
case 0x00e1: // L'á'
|
||||
case 0x1eae: // L'Ắ'
|
||||
case 0x1eaf: // L'ắ'
|
||||
case 0x1ea4: // L'Ấ'
|
||||
case 0x1ea5: // L'ấ'
|
||||
case 0x00c9: // L'É'
|
||||
case 0x00e9: // L'é'
|
||||
case 0x1ebe: // L'Ế'
|
||||
case 0x1ebf: // L'ế'
|
||||
case 0x00cd: // L'Í'
|
||||
case 0x00ed: // L'í'
|
||||
case 0x00d3: // L'Ó'
|
||||
case 0x00f3: // L'ó'
|
||||
case 0x1ed0: // L'Ố'
|
||||
case 0x1ed1: // L'ố'
|
||||
case 0x1eda: // L'Ớ'
|
||||
case 0x1edb: // L'ớ'
|
||||
case 0x00da: // L'Ú'
|
||||
case 0x00fa: // L'ú'
|
||||
case 0x1ee8: // L'Ứ'
|
||||
case 0x1ee9: // L'ứ'
|
||||
case 0x00dd: // L'Ý'
|
||||
case 0x00fd: // L'ý'
|
||||
*tone = (char)0xec;
|
||||
return true;
|
||||
case 0x00c0: // L'À'
|
||||
case 0x00e0: // L'à'
|
||||
case 0x1eb0: // L'Ằ'
|
||||
case 0x1eb1: // L'ằ'
|
||||
case 0x1ea6: // L'Ầ'
|
||||
case 0x1ea7: // L'ầ'
|
||||
case 0x00c8: // L'È'
|
||||
case 0x00e8: // L'è'
|
||||
case 0x1ec0: // L'Ề'
|
||||
case 0x1ec1: // L'ề'
|
||||
case 0x00cc: // L'Ì'
|
||||
case 0x00ec: // L'ì'
|
||||
case 0x00d2: // L'Ò'
|
||||
case 0x00f2: // L'ò'
|
||||
case 0x1ed2: // L'Ồ'
|
||||
case 0x1ed3: // L'ồ'
|
||||
case 0x1edc: // L'Ờ'
|
||||
case 0x1edd: // L'ờ'
|
||||
case 0x00d9: // L'Ù'
|
||||
case 0x00f9: // L'ù'
|
||||
case 0x1eea: // L'Ừ'
|
||||
case 0x1eeb: // L'ừ'
|
||||
case 0x1ef2: // L'Ỳ'
|
||||
case 0x1ef3: // L'ỳ'
|
||||
*tone = (char)0xcc;
|
||||
return true;
|
||||
case 0x1ea2: // L'Ả'
|
||||
case 0x1ea3: // L'ả'
|
||||
case 0x1eb2: // L'Ẳ'
|
||||
case 0x1eb3: // L'ẳ'
|
||||
case 0x1ea8: // L'Ẩ'
|
||||
case 0x1ea9: // L'ẩ'
|
||||
case 0x1eba: // L'Ẻ'
|
||||
case 0x1ebb: // L'ẻ'
|
||||
case 0x1ec2: // L'Ể'
|
||||
case 0x1ec3: // L'ể'
|
||||
case 0x1ec8: // L'Ỉ'
|
||||
case 0x1ec9: // L'ỉ'
|
||||
case 0x1ece: // L'Ỏ'
|
||||
case 0x1ecf: // L'ỏ'
|
||||
case 0x1ed4: // L'Ổ'
|
||||
case 0x1ed5: // L'ổ'
|
||||
case 0x1ede: // L'Ở'
|
||||
case 0x1edf: // L'ở'
|
||||
case 0x1ee6: // L'Ủ'
|
||||
case 0x1ee7: // L'ủ'
|
||||
case 0x1eec: // L'Ử'
|
||||
case 0x1eed: // L'ử'
|
||||
case 0x1ef6: // L'Ỷ'
|
||||
case 0x1ef7: // L'ỷ'
|
||||
*tone = (char)0xd2;
|
||||
return true;
|
||||
case 0x00c3: // L'Ã'
|
||||
case 0x00e3: // L'ã'
|
||||
case 0x1eb4: // L'Ẵ'
|
||||
case 0x1eb5: // L'ẵ'
|
||||
case 0x1eaa: // L'Ẫ'
|
||||
case 0x1eab: // L'ẫ'
|
||||
case 0x1ebc: // L'Ẽ'
|
||||
case 0x1ebd: // L'ẽ'
|
||||
case 0x1ec4: // L'Ễ'
|
||||
case 0x1ec5: // L'ễ'
|
||||
case 0x0128: // L'Ĩ'
|
||||
case 0x0129: // L'ĩ'
|
||||
case 0x00d5: // L'Õ'
|
||||
case 0x00f5: // L'õ'
|
||||
case 0x1ed6: // L'Ỗ'
|
||||
case 0x1ed7: // L'ỗ'
|
||||
case 0x1ee0: // L'Ỡ'
|
||||
case 0x1ee1: // L'ỡ'
|
||||
case 0x0169: // L'ũ'
|
||||
case 0x0168: // L'Ũ'
|
||||
case 0x1eee: // L'Ữ'
|
||||
case 0x1eef: // L'ữ'
|
||||
case 0x1ef8: // L'Ỹ'
|
||||
case 0x1ef9: // L'ỹ'
|
||||
*tone = (char)0xde;
|
||||
return true;
|
||||
case 0x1ea0: // L'Ạ'
|
||||
case 0x1ea1: // L'ạ'
|
||||
case 0x1eb6: // L'Ặ'
|
||||
case 0x1eb7: // L'ặ'
|
||||
case 0x1eac: // L'Ậ'
|
||||
case 0x1ead: // L'ậ'
|
||||
case 0x1eb8: // L'Ẹ'
|
||||
case 0x1eb9: // L'ẹ'
|
||||
case 0x1ec6: // L'Ệ'
|
||||
case 0x1ec7: // L'ệ'
|
||||
case 0x1eca: // L'Ị'
|
||||
case 0x1ecb: // L'ị'
|
||||
case 0x1ecc: // L'Ọ'
|
||||
case 0x1ecd: // L'ọ'
|
||||
case 0x1ed8: // L'Ộ'
|
||||
case 0x1ed9: // L'ộ'
|
||||
case 0x1ee2: // L'Ợ'
|
||||
case 0x1ee3: // L'ợ'
|
||||
case 0x1ee4: // L'Ụ'
|
||||
case 0x1ee5: // L'ụ'
|
||||
case 0x1ef0: // L'Ự'
|
||||
case 0x1ef1: // L'ự'
|
||||
case 0x1ef4: // L'Ỵ'
|
||||
case 0x1ef5: // L'ỵ'
|
||||
*tone = (char)0xf2;
|
||||
return true;
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
int EL_String_Encode_Vietnamese(const wchar_t* wide, int wideLen, char* multi, int multiLen)
|
||||
{
|
||||
if(wideLen < 0)
|
||||
wideLen = (int)wcslen(wide) + 1;
|
||||
|
||||
int src = 0;
|
||||
int dest = 0;
|
||||
|
||||
if(wideLen > 0)
|
||||
{
|
||||
while(src < wideLen && dest < multiLen)
|
||||
{
|
||||
char letter;
|
||||
if(DecomposeLetter(wide[src], &letter))
|
||||
{
|
||||
multi[dest++] = letter;
|
||||
}
|
||||
|
||||
char tone;
|
||||
if(DecomposeTone(wide[src], &tone) && dest < multiLen)
|
||||
{
|
||||
multi[dest++] = tone;
|
||||
}
|
||||
|
||||
++src;
|
||||
}
|
||||
}
|
||||
|
||||
return dest;
|
||||
}
|
||||
@@ -0,0 +1,4 @@
|
||||
#pragma once
|
||||
|
||||
int EL_String_Decode_Vietnamese(const char* multi, int multiLen, wchar_t* wide, int wideLen);
|
||||
int EL_String_Encode_Vietnamese(const wchar_t* wide, int wideLen, char* multi, int multiLen);
|
||||
@@ -11,6 +11,10 @@
|
||||
#include <thread>
|
||||
#include <vector>
|
||||
|
||||
#if __has_include(<iconv.h>)
|
||||
#include <iconv.h>
|
||||
#define MT_WIN32CRT_HAS_ICONV 1
|
||||
#endif
|
||||
#include <glob.h>
|
||||
#include <sys/stat.h>
|
||||
#include <unistd.h>
|
||||
@@ -165,4 +169,198 @@ DWORD GetTickCount()
|
||||
return timeGetTime();
|
||||
}
|
||||
|
||||
namespace {
|
||||
|
||||
// Windows-1252 0x80..0x9F; the other bytes map to the same code point (0 marks the undefined ones).
|
||||
const char32_t kCp1252High[32] = {
|
||||
0x20AC, 0, 0x201A, 0x0192, 0x201E, 0x2026, 0x2020, 0x2021, 0x02C6, 0x2030, 0x0160, 0x2039, 0x0152, 0, 0x017D, 0,
|
||||
0, 0x2018, 0x2019, 0x201C, 0x201D, 0x2022, 0x2013, 0x2014, 0x02DC, 0x2122, 0x0161, 0x203A, 0x0153, 0, 0x017E, 0x0178,
|
||||
};
|
||||
|
||||
bool is_cp1252(UINT cp) { return cp == CP_ACP || cp == 1252; }
|
||||
|
||||
bool decode_utf8(const unsigned char* p, size_t n, std::u32string& out)
|
||||
{
|
||||
bool ok = true;
|
||||
for (size_t i = 0; i < n;) {
|
||||
const unsigned char c = p[i];
|
||||
size_t len = c < 0x80 ? 1 : c >= 0xC2 && c <= 0xDF ? 2 : c >= 0xE0 && c <= 0xEF ? 3 : c >= 0xF0 && c <= 0xF4 ? 4 : 0;
|
||||
char32_t cp = len == 1 ? c : len == 2 ? (c & 0x1F) : len == 3 ? (c & 0x0F) : (c & 0x07);
|
||||
if (len == 0 || i + len > n) { out.push_back(0xFFFD); ok = false; ++i; continue; }
|
||||
bool valid = true;
|
||||
for (size_t k = 1; k < len; ++k) {
|
||||
if ((p[i + k] & 0xC0) != 0x80) { valid = false; break; }
|
||||
cp = (cp << 6) | (p[i + k] & 0x3F);
|
||||
}
|
||||
if (!valid || (len == 3 && cp < 0x800) || (len == 4 && (cp < 0x10000 || cp > 0x10FFFF)) || (cp >= 0xD800 && cp <= 0xDFFF)) {
|
||||
out.push_back(0xFFFD); ok = false; ++i; continue;
|
||||
}
|
||||
out.push_back(cp);
|
||||
i += len;
|
||||
}
|
||||
return ok;
|
||||
}
|
||||
|
||||
void encode_utf8(char32_t cp, std::string& out)
|
||||
{
|
||||
if (cp < 0x80) out.push_back(char(cp));
|
||||
else if (cp < 0x800) { out.push_back(char(0xC0 | (cp >> 6))); out.push_back(char(0x80 | (cp & 0x3F))); }
|
||||
else if (cp < 0x10000) { out.push_back(char(0xE0 | (cp >> 12))); out.push_back(char(0x80 | ((cp >> 6) & 0x3F))); out.push_back(char(0x80 | (cp & 0x3F))); }
|
||||
else { out.push_back(char(0xF0 | (cp >> 18))); out.push_back(char(0x80 | ((cp >> 12) & 0x3F))); out.push_back(char(0x80 | ((cp >> 6) & 0x3F))); out.push_back(char(0x80 | (cp & 0x3F))); }
|
||||
}
|
||||
|
||||
#if defined(MT_WIN32CRT_HAS_ICONV)
|
||||
std::string iconv_name(UINT cp)
|
||||
{
|
||||
switch (cp) {
|
||||
case 949: return "CP949";
|
||||
case 936: return "GBK";
|
||||
case 950: return "BIG5";
|
||||
case 932: return "SHIFT_JIS";
|
||||
case 874: return "CP874";
|
||||
default: return "CP" + std::to_string(cp);
|
||||
}
|
||||
}
|
||||
|
||||
// One character at a time, so an undecodable byte becomes one default char like Windows does.
|
||||
bool iconv_decode(UINT cp, const unsigned char* p, size_t n, std::u32string& out)
|
||||
{
|
||||
iconv_t cd = iconv_open("UTF-32LE", iconv_name(cp).c_str());
|
||||
if (cd == reinterpret_cast<iconv_t>(-1)) return false;
|
||||
for (size_t i = 0; i < n;) {
|
||||
unsigned char wide[16];
|
||||
bool done = false;
|
||||
for (size_t len = 1; len <= 4 && i + len <= n && !done; ++len) {
|
||||
iconv(cd, nullptr, nullptr, nullptr, nullptr);
|
||||
char* in = const_cast<char*>(reinterpret_cast<const char*>(p + i));
|
||||
size_t in_left = len;
|
||||
char* o = reinterpret_cast<char*>(wide);
|
||||
size_t o_left = sizeof(wide);
|
||||
if (iconv(cd, &in, &in_left, &o, &o_left) != size_t(-1) && in_left == 0 && o_left < sizeof(wide)) {
|
||||
for (size_t k = 0; k + 4 <= sizeof(wide) - o_left; k += 4)
|
||||
out.push_back(char32_t(wide[k] | (wide[k + 1] << 8) | (wide[k + 2] << 16) | (unsigned(wide[k + 3]) << 24)));
|
||||
i += len;
|
||||
done = true;
|
||||
}
|
||||
}
|
||||
if (!done) { out.push_back(0xFFFD); ++i; }
|
||||
}
|
||||
iconv_close(cd);
|
||||
return true;
|
||||
}
|
||||
|
||||
bool iconv_encode(UINT cp, const std::u32string& in, std::string& out, bool* used_default)
|
||||
{
|
||||
iconv_t cd = iconv_open(iconv_name(cp).c_str(), "UTF-32LE");
|
||||
if (cd == reinterpret_cast<iconv_t>(-1)) return false;
|
||||
for (char32_t c : in) {
|
||||
unsigned char src[4] = {static_cast<unsigned char>(c), static_cast<unsigned char>(c >> 8), static_cast<unsigned char>(c >> 16), static_cast<unsigned char>(c >> 24)};
|
||||
char buf[16];
|
||||
char* ip = reinterpret_cast<char*>(src);
|
||||
size_t il = 4;
|
||||
char* op = buf;
|
||||
size_t ol = sizeof(buf);
|
||||
iconv(cd, nullptr, nullptr, nullptr, nullptr);
|
||||
if (iconv(cd, &ip, &il, &op, &ol) != size_t(-1) && il == 0) {
|
||||
out.append(buf, sizeof(buf) - ol);
|
||||
} else {
|
||||
out.push_back('?');
|
||||
*used_default = true;
|
||||
}
|
||||
}
|
||||
iconv_close(cd);
|
||||
return true;
|
||||
}
|
||||
#endif
|
||||
|
||||
bool decode(UINT cp, const unsigned char* p, size_t n, std::u32string& out)
|
||||
{
|
||||
if (cp == CP_UTF8) return decode_utf8(p, n, out);
|
||||
if (is_cp1252(cp)) {
|
||||
for (size_t i = 0; i < n; ++i) {
|
||||
const unsigned char c = p[i];
|
||||
out.push_back(c >= 0x80 && c < 0xA0 ? (kCp1252High[c - 0x80] ? kCp1252High[c - 0x80] : char32_t(c)) : char32_t(c));
|
||||
}
|
||||
return true;
|
||||
}
|
||||
#if defined(MT_WIN32CRT_HAS_ICONV)
|
||||
if (iconv_decode(cp, p, n, out)) return true;
|
||||
#endif
|
||||
// No converter for this code page: ASCII passes, the rest becomes U+FFFD.
|
||||
for (size_t i = 0; i < n; ++i) out.push_back(p[i] < 0x80 ? char32_t(p[i]) : char32_t(0xFFFD));
|
||||
return true;
|
||||
}
|
||||
|
||||
void encode(UINT cp, const std::u32string& in, std::string& out, bool* used_default)
|
||||
{
|
||||
if (cp == CP_UTF8) {
|
||||
for (char32_t c : in) encode_utf8(c, out);
|
||||
return;
|
||||
}
|
||||
if (is_cp1252(cp)) {
|
||||
for (char32_t c : in) {
|
||||
if (c < 0x80 || (c >= 0xA0 && c < 0x100)) { out.push_back(char(c)); continue; }
|
||||
int hit = -1;
|
||||
for (int k = 0; k < 32; ++k)
|
||||
if (kCp1252High[k] == c) { hit = k; break; }
|
||||
if (hit >= 0) out.push_back(char(0x80 + hit));
|
||||
else { out.push_back('?'); *used_default = true; }
|
||||
}
|
||||
return;
|
||||
}
|
||||
#if defined(MT_WIN32CRT_HAS_ICONV)
|
||||
if (iconv_encode(cp, in, out, used_default)) return;
|
||||
#endif
|
||||
for (char32_t c : in) {
|
||||
if (c < 0x80) out.push_back(char(c));
|
||||
else { out.push_back('?'); *used_default = true; }
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
int MultiByteToWideChar(UINT CodePage, DWORD dwFlags, LPCSTR lpMultiByteStr, int cbMultiByte,
|
||||
LPWSTR lpWideCharStr, int cchWideChar)
|
||||
{
|
||||
if (!lpMultiByteStr || cbMultiByte == 0 || cchWideChar < 0) return 0;
|
||||
const size_t n = cbMultiByte < 0 ? std::strlen(lpMultiByteStr) + 1 : size_t(cbMultiByte);
|
||||
std::u32string wide;
|
||||
const bool ok = decode(CodePage, reinterpret_cast<const unsigned char*>(lpMultiByteStr), n, wide);
|
||||
if (!ok && (dwFlags & MB_ERR_INVALID_CHARS)) return 0;
|
||||
if (cchWideChar == 0) return int(wide.size());
|
||||
if (wide.size() > size_t(cchWideChar)) return 0;
|
||||
for (size_t i = 0; i < wide.size(); ++i) lpWideCharStr[i] = WCHAR(wide[i]);
|
||||
return int(wide.size());
|
||||
}
|
||||
|
||||
int WideCharToMultiByte(UINT CodePage, DWORD, LPCWSTR lpWideCharStr, int cchWideChar,
|
||||
LPSTR lpMultiByteStr, int cbMultiByte, LPCSTR lpDefaultChar, LPBOOL lpUsedDefaultChar)
|
||||
{
|
||||
if (!lpWideCharStr || cchWideChar == 0 || cbMultiByte < 0) return 0;
|
||||
size_t n = 0;
|
||||
if (cchWideChar < 0) { while (lpWideCharStr[n]) ++n; ++n; }
|
||||
else n = size_t(cchWideChar);
|
||||
std::u32string wide(n, 0);
|
||||
for (size_t i = 0; i < n; ++i) wide[i] = char32_t(lpWideCharStr[i]);
|
||||
std::string bytes;
|
||||
bool used_default = false;
|
||||
encode(CodePage, wide, bytes, &used_default);
|
||||
if (used_default && lpDefaultChar && CodePage != CP_UTF8) {
|
||||
// '?' was the placeholder; the caller's default char replaces it only where a fallback happened.
|
||||
std::string redo;
|
||||
for (char32_t c : wide) {
|
||||
std::string one;
|
||||
bool miss = false;
|
||||
encode(CodePage, std::u32string(1, c), one, &miss);
|
||||
redo += miss ? std::string(1, *lpDefaultChar) : one;
|
||||
}
|
||||
bytes.swap(redo);
|
||||
}
|
||||
if (lpUsedDefaultChar) *lpUsedDefaultChar = used_default ? TRUE : FALSE;
|
||||
if (cbMultiByte == 0) return int(bytes.size());
|
||||
if (bytes.size() > size_t(cbMultiByte)) return 0;
|
||||
std::memcpy(lpMultiByteStr, bytes.data(), bytes.size());
|
||||
return int(bytes.size());
|
||||
}
|
||||
|
||||
#endif // !_WIN32
|
||||
|
||||
@@ -83,4 +83,16 @@ void DeleteCriticalSection(LPCRITICAL_SECTION cs);
|
||||
void EnterCriticalSection(LPCRITICAL_SECTION cs);
|
||||
void LeaveCriticalSection(LPCRITICAL_SECTION cs);
|
||||
|
||||
// stringapiset.h. WCHAR is 32-bit here, so wide strings hold UTF-32 code points. CP_UTF8 and
|
||||
// CP_ACP/1252 are converted directly; other code pages go through iconv where the platform has it.
|
||||
#define CP_ACP 0
|
||||
#define CP_UTF8 65001
|
||||
#define MB_PRECOMPOSED 0x00000001
|
||||
#define MB_ERR_INVALID_CHARS 0x00000008
|
||||
#define WC_NO_BEST_FIT_CHARS 0x00000400
|
||||
int MultiByteToWideChar(UINT CodePage, DWORD dwFlags, LPCSTR lpMultiByteStr, int cbMultiByte,
|
||||
LPWSTR lpWideCharStr, int cchWideChar);
|
||||
int WideCharToMultiByte(UINT CodePage, DWORD dwFlags, LPCWSTR lpWideCharStr, int cchWideChar,
|
||||
LPSTR lpMultiByteStr, int cbMultiByte, LPCSTR lpDefaultChar, LPBOOL lpUsedDefaultChar);
|
||||
|
||||
#endif // !_WIN32
|
||||
|
||||
@@ -1,10 +1,13 @@
|
||||
#pragma once
|
||||
// Shim for <windows.h> on non-Windows targets: the Win32 scalar types and CRT spellings
|
||||
// 40250 sources use (common/Win32Types.h, common/Win32Crt.h), plus the headers the real one pulls
|
||||
// in that 40250 relies on transitively (winnt.h brings <string.h> for memcpy/memset).
|
||||
// in that 40250 relies on transitively (winnt.h brings <string.h> for memcpy/memset, and <wchar.h>
|
||||
// for wcslen).
|
||||
#include <string.h>
|
||||
#include <wchar.h>
|
||||
#include "../../Win32Types.h"
|
||||
#include "../../Win32Crt.h"
|
||||
#include "wingdi.h"
|
||||
|
||||
// winuser.h / wingdi.h constants named in 40250 declarations (default arguments).
|
||||
#define WS_OVERLAPPED 0x00000000L
|
||||
|
||||
@@ -0,0 +1,64 @@
|
||||
#pragma once
|
||||
// Shim for the <wingdi.h> subset the 40250 font path uses (EterLib/GrpDIB, GrpFontTexture, Util).
|
||||
// The calls are implemented by the platform layer over FreeType (platform/Win32Gdi.cpp): one DIB
|
||||
// per memory DC, glyphs rendered the way GDI lays out a TrueType face for a LOGFONT cell height.
|
||||
#include "../../Win32Types.h"
|
||||
|
||||
typedef DWORD COLORREF;
|
||||
typedef void* HGDIOBJ;
|
||||
#define RGB(r, g, b) ((COLORREF)(((BYTE)(r) | ((WORD)((BYTE)(g)) << 8)) | (((DWORD)(BYTE)(b)) << 16)))
|
||||
|
||||
typedef struct _ABCFLOAT
|
||||
{
|
||||
FLOAT abcfA;
|
||||
FLOAT abcfB;
|
||||
FLOAT abcfC;
|
||||
} ABCFLOAT, *LPABCFLOAT;
|
||||
|
||||
#define TRANSPARENT 1
|
||||
#define OPAQUE 2
|
||||
|
||||
#define FW_NORMAL 400
|
||||
#define FW_BOLD 700
|
||||
#define OUT_DEFAULT_PRECIS 0
|
||||
#define CLIP_DEFAULT_PRECIS 0
|
||||
#define DEFAULT_QUALITY 0
|
||||
#define NONANTIALIASED_QUALITY 3
|
||||
#define ANTIALIASED_QUALITY 4
|
||||
#define DEFAULT_PITCH 0
|
||||
|
||||
#define ANSI_CHARSET 0
|
||||
#define DEFAULT_CHARSET 1
|
||||
#define SHIFTJIS_CHARSET 128
|
||||
#define HANGUL_CHARSET 129
|
||||
#define GB2312_CHARSET 134
|
||||
#define CHINESEBIG5_CHARSET 136
|
||||
#define GREEK_CHARSET 161
|
||||
#define TURKISH_CHARSET 162
|
||||
#define VIETNAMESE_CHARSET 163
|
||||
#define HEBREW_CHARSET 177
|
||||
#define ARABIC_CHARSET 178
|
||||
#define BALTIC_CHARSET 186
|
||||
#define RUSSIAN_CHARSET 204
|
||||
#define THAI_CHARSET 222
|
||||
#define EASTEUROPE_CHARSET 238
|
||||
|
||||
#define BI_RGB 0L
|
||||
#define DIB_RGB_COLORS 0
|
||||
|
||||
HDC CreateCompatibleDC(HDC hdc);
|
||||
BOOL DeleteDC(HDC hdc);
|
||||
HBITMAP CreateDIBSection(HDC hdc, const BITMAPINFO* pbmi, UINT usage, void** ppvBits, HANDLE hSection, DWORD offset);
|
||||
int SetDIBitsToDevice(HDC hdc, int xDest, int yDest, DWORD w, DWORD h, int xSrc, int ySrc, UINT StartScan,
|
||||
UINT cLines, const void* lpvBits, const BITMAPINFO* lpbmi, UINT ColorUse);
|
||||
HFONT CreateFontIndirect(const LOGFONT* lplf);
|
||||
BOOL DeleteObject(HGDIOBJ ho);
|
||||
HGDIOBJ SelectObject(HDC hdc, HGDIOBJ h);
|
||||
COLORREF SetTextColor(HDC hdc, COLORREF color);
|
||||
COLORREF SetBkColor(HDC hdc, COLORREF color);
|
||||
int SetBkMode(HDC hdc, int mode);
|
||||
BOOL GetTextExtentPoint32W(HDC hdc, LPCWSTR lpString, int c, LPSIZE psizl);
|
||||
BOOL GetCharABCWidthsFloatW(HDC hdc, UINT iFirst, UINT iLast, LPABCFLOAT lpABC);
|
||||
BOOL TextOutW(HDC hdc, int x, int y, LPCWSTR lpString, int c);
|
||||
BOOL TextOutA(HDC hdc, int x, int y, LPCSTR lpString, int c);
|
||||
#define TextOut TextOutA
|
||||
Reference in New Issue
Block a user