/* bit flag for UConverter.options indicating GB 18030 special handling */ #define _MBCS_OPTION_GB18030 0x8000
/* bit flag for UConverter.options indicating KEIS,JEF,JIF special handling */ #define _MBCS_OPTION_KEIS 0x01000 #define _MBCS_OPTION_JEF 0x02000 #define _MBCS_OPTION_JIPS 0x04000
range=gb18030Ranges[0]; for(i=0; i<UPRV_LENGTHOF(gb18030Ranges); range+=4, ++i) { if (range[0] <= static_cast<uint32_t>(cp) && static_cast<uint32_t>(cp) <= range[1]) { /* found the Unicode code point, output the four-byte sequence for it */
uint32_t linear; char bytes[4];
/* get the linear value of the first GB 18030 code in this range */
linear=range[2]-LINEAR_18030_BASE;
/* add the offset from the beginning of the range */
linear += (static_cast<uint32_t>(cp) - range[0]);
/* turn this into a four-byte sequence */
bytes[3] = static_cast<char>(0x30 + linear % 10); linear /= 10;
bytes[2] = static_cast<char>(0x81 + linear % 126); linear /= 126;
bytes[1] = static_cast<char>(0x30 + linear % 10); linear /= 10;
bytes[0] = static_cast<char>(0x81 + linear);
linear=LINEAR_18030(cnv->toUBytes[0], cnv->toUBytes[1], cnv->toUBytes[2], cnv->toUBytes[3]);
range=gb18030Ranges[0]; for(i=0; i<UPRV_LENGTHOF(gb18030Ranges); range+=4, ++i) { if(range[2]<=linear && linear<=range[3]) { /* found the sequence, output the Unicode code point for it */
*pErrorCode=U_ZERO_ERROR;
/* add the linear difference between the input and start sequences to the start code point */
linear=range[0]+(linear-range[2]);
/* output this code point */
ucnv_toUWriteCodePoint(cnv, linear, target, targetLimit, offsets, sourceIndex, pErrorCode);
return0;
}
}
}
/* no mapping */
*pErrorCode=U_INVALID_CHAR_FOUND; return length;
}
/* copy and modify the to-Unicode state table */
newStateTable = reinterpret_cast<int32_t(*)[256]>(p);
uprv_memcpy(newStateTable, mbcsTable->stateTable, mbcsTable->countStates*1024);
/* copy and modify the from-Unicode result table */
newResults = reinterpret_cast<uint16_t*>(newStateTable[mbcsTable->countStates]);
uprv_memcpy(newResults, bytes, sizeofFromUBytes);
/* conveniently, the table access macros work on the left side of expressions */ if(mbcsTable->outputType==MBCS_OUTPUT_1) {
MBCS_SINGLE_RESULT_FROM_U(table, newResults, U_LF)=EBCDIC_RT_NL;
MBCS_SINGLE_RESULT_FROM_U(table, newResults, U_NL)=EBCDIC_RT_LF;
} else/* MBCS_OUTPUT_2_SISO */ {
stage2Entry=MBCS_STAGE_2_FROM_U(table, U_LF);
MBCS_VALUE_2_FROM_STAGE_2(newResults, stage2Entry, U_LF)=EBCDIC_NL;
/* set the canonical converter name */
name = reinterpret_cast<char*>(newResults) + sizeofFromUBytes;
uprv_strcpy(name, sharedData->staticData->name);
uprv_strcat(name, UCNV_SWAP_LFNL_OPTION_STRING);
/* set the pointers */
icu::umtx_lock(nullptr); if(mbcsTable->swapLFNLStateTable==nullptr) {
mbcsTable->swapLFNLStateTable=newStateTable;
mbcsTable->swapLFNLFromUnicodeBytes = reinterpret_cast<uint8_t*>(newResults);
mbcsTable->swapLFNLName=name;
/* for EUC outputTypes, modify the value like genmbcs.c's transformEUC() */ switch(mbcsTable->outputType) { case MBCS_OUTPUT_3_EUC: if(value<=0xffff) { /* short sequences are stored directly */ /* code set 0 or 1 */
} elseif(value<=0x8effff) { /* code set 2 */
value&=0x7fff;
} else/* first byte is 0x8f */ { /* code set 3 */
value&=0xff7f;
} break; case MBCS_OUTPUT_4_EUC: if(value<=0xffffff) { /* short sequences are stored directly */ /* code set 0 or 1 */
} elseif(value<=0x8effffff) { /* code set 2 */
value&=0x7fffff;
} else/* first byte is 0x8f */ { /* code set 3 */
value&=0xff7fff;
} break; default: break;
}
/* copy existing data and reroute the pointers */
stage1 = reinterpret_cast<uint16_t*>(mbcsTable->reconstitutedData);
uprv_memcpy(stage1, mbcsTable->fromUnicodeTable, stage1Length*2);
/* indexes into stage 2 count from the bottom of the fromUnicodeTable */
stage2 = reinterpret_cast<uint32_t*>(stage1);
/* reconstitute the initial part of stage 2 from the mbcsIndex */
{
int32_t stageUTF8Length = (static_cast<int32_t>(mbcsTable->maxFastUChar) + 1) >> 6;
int32_t stageUTF8Index=0;
int32_t st1, st2, st3, i;
for(st1=0; stageUTF8Index<stageUTF8Length; ++st1) {
st2=stage1[st1]; if (st2 != static_cast<int32_t>(stage1Length) / 2) { /* each stage 2 block has 64 entries corresponding to 16 entries in the mbcsIndex */ for(i=0; i<16; ++i) {
st3=mbcsTable->mbcsIndex[stageUTF8Index++]; if(st3!=0) { /* an stage 2 entry's index is per stage 3 16-block, not per stage 3 entry */
st3>>=4; /* *4stage2entriespointto4consecutivestage316-blockswhichare *allocatedtogetherasasingle64-blockforaccessfromthembcsIndex
*/
stage2[st2++]=st3++;
stage2[st2++]=st3++;
stage2[st2++]=st3++;
stage2[st2++]=st3;
} else { /* no stage 3 block, skip */
st2+=4;
}
}
} else { /* no stage 2 block, skip */
stageUTF8Index+=16;
}
}
}
/* reconstitute fromUnicodeBytes with roundtrips from toUnicode data */
ucnv_MBCSEnumToUnicode(mbcsTable, writeStage3Roundtrip, mbcsTable, pErrorCode);
}
/* extension-only file, load the base table and set values appropriately */ if((extIndexes=mbcsTable->extIndexes)==nullptr) { /* extension-only file without extension */
*pErrorCode=U_INVALID_TABLE_FORMAT; return;
}
if(pArgs->nestedLoads!=1) { /* an extension table must not be loaded as a base table */
*pErrorCode=U_INVALID_TABLE_FILE; return;
}
/* load the base table */
baseName = reinterpret_cast<constchar*>(header) + headerLength * 4; if(0==uprv_strcmp(baseName, sharedData->staticData->name)) { /* forbid loading this same extension-only file */
*pErrorCode=U_INVALID_TABLE_FORMAT; return;
}
/* TODO parse package name out of the prefix of the base name in the extension .cnv file? */
args.size=sizeof(UConverterLoadArgs);
args.nestedLoads=2;
args.onlyTestIsLoadable=pArgs->onlyTestIsLoadable;
args.reserved=pArgs->reserved;
args.options=pArgs->options;
args.pkg=pArgs->pkg;
args.name=baseName;
baseSharedData=ucnv_load(&args, pErrorCode); if(U_FAILURE(*pErrorCode)) { return;
} if( baseSharedData->staticData->conversionType!=UCNV_MBCS ||
baseSharedData->mbcs.baseSharedData!=nullptr
) {
ucnv_unload(baseSharedData);
*pErrorCode=U_INVALID_TABLE_FORMAT; return;
} if(pArgs->onlyTestIsLoadable) { /* *Exitassoonasweknowthatwecanloadtheconverter *andtheformatisvalidandsupported. *Theworstthatcanhappeninthefollowingcodeisamemory *allocationerror.
*/
ucnv_unload(baseSharedData); return;
}
/* copy the base table data */
uprv_memcpy(mbcsTable, &baseSharedData->mbcs, sizeof(UConverterMBCSTable));
/* overwrite values with relevant ones for the extension converter */
mbcsTable->baseSharedData=baseSharedData;
mbcsTable->extIndexes=extIndexes;
/* *Setaspecial,runtime-onlyoutputTypeiftheextensionconverter *isaDBCSversionofabaseconverterthatalsomapssinglebytes.
*/ if( sharedData->staticData->conversionType==UCNV_DBCS ||
(sharedData->staticData->conversionType==UCNV_MBCS &&
sharedData->staticData->minBytesPerChar>=2)
) { if(baseSharedData->mbcs.outputType==MBCS_OUTPUT_2_SISO) { /* the base converter is SI/SO-stateful */
int32_t entry;
/* get the dbcs state from the state table entry for SO=0x0e */
entry=mbcsTable->stateTable[0][0xe]; if( MBCS_ENTRY_IS_FINAL(entry) &&
MBCS_ENTRY_FINAL_ACTION(entry)==MBCS_STATE_CHANGE_ONLY &&
MBCS_ENTRY_FINAL_STATE(entry)!=0
) {
mbcsTable->dbcsOnlyState = static_cast<uint8_t>(MBCS_ENTRY_FINAL_STATE(entry));
mbcsTable->outputType=MBCS_OUTPUT_DBCS_ONLY;
}
} elseif(
baseSharedData->staticData->conversionType==UCNV_MBCS &&
baseSharedData->staticData->minBytesPerChar==1 &&
baseSharedData->staticData->maxBytesPerChar==2 &&
mbcsTable->countStates<=127
) { /* non-stateful base converter, need to modify the state table */
int32_t (*newStateTable)[256];
int32_t *state;
int32_t i, count;
/* allocate a new state table and copy the base state table contents */
count=mbcsTable->countStates;
newStateTable = static_cast<int32_t(*)[256]>(uprv_malloc((count + 1) * 1024)); if(newStateTable==nullptr) {
ucnv_unload(baseSharedData);
*pErrorCode=U_MEMORY_ALLOCATION_ERROR; return;
}
/* change all final single-byte entries to go to a new all-illegal state */
state=newStateTable[0]; for(i=0; i<256; ++i) { if(MBCS_ENTRY_IS_FINAL(state[i])) {
state[i]=MBCS_ENTRY_TRANSITION(count, 0);
}
}
/* build the new all-illegal state */
state=newStateTable[count];
for(i=0; i<256; ++i) {
state[i]=MBCS_ENTRY_FINAL(0, MBCS_STATE_ILLEGAL, 0);
}
mbcsTable->stateTable=(const int32_t (*)[256])newStateTable;
mbcsTable->countStates = static_cast<uint8_t>(count + 1);
mbcsTable->stateTableOwned=true;
mbcsTable->outputType=MBCS_OUTPUT_DBCS_ONLY;
}
}
/*
* unlike below for files with base tables, do not get the unicodeMask
* from the sharedData; instead, use the base table's unicodeMask,
* which we copied in the memcpy above;
* this is necessary because the static data unicodeMask, especially
* the UCNV_HAS_SUPPLEMENTARY flag, is part of the base table data
*/
} else {
/* conversion file with a base table; an additional extension table is optional */
/* make sure that the output type is known */
switch(mbcsTable->outputType) {
case MBCS_OUTPUT_1:
case MBCS_OUTPUT_2:
case MBCS_OUTPUT_3:
case MBCS_OUTPUT_4:
case MBCS_OUTPUT_3_EUC:
case MBCS_OUTPUT_4_EUC:
case MBCS_OUTPUT_2_SISO:
/* OK */
break;
default:
*pErrorCode=U_INVALID_TABLE_FORMAT;
return;
}
if(pArgs->onlyTestIsLoadable) {
/*
* Exit as soon as we know that we can load the converter
* and the format is valid and supported.
* The worst that can happen in the following code is a memory
* allocation error.
*/
return;
}
/*
* converter versions 6.1 and up contain a unicodeMask that is
* used here to select the most efficient function implementations
*/
info.size=sizeof(UDataInfo);
udata_getInfo((UDataMemory *)sharedData->dataMemory, &info);
if(info.formatVersion[0]>6 || (info.formatVersion[0]==6 && info.formatVersion[1]>=1)) {
/* mask off possible future extensions to be safe */
mbcsTable->unicodeMask = static_cast<uint8_t>(sharedData->staticData->unicodeMask & 3);
} else {
/* for older versions, assume worst case: contains anything possible (prevent over-optimizations) */
mbcsTable->unicodeMask=UCNV_HAS_SUPPLEMENTARY|UCNV_HAS_SURROGATES;
}
/*
* _MBCSHeader.version 4.3 adds utf8Friendly data structures.
* Check for the header version, SBCS vs. MBCS, and for whether the
* data structures are optimized for code points as high as what the
* runtime code is designed for.
* The implementation does not handle mapping tables with entries for
* unpaired surrogates.
*/
if( header->version[1]>=3 &&
(mbcsTable->unicodeMask&UCNV_HAS_SURROGATES)==0 &&
(mbcsTable->countStates==1 ?
(header->version[2]>=(SBCS_FAST_MAX>>8)) :
(header->version[2]>=(MBCS_FAST_MAX>>8))
)
) {
mbcsTable->utf8Friendly=true;
if(mbcsTable->countStates==1) {
/*
* SBCS: Stage 3 is allocated in 64-entry blocks for U+0000..SBCS_FAST_MAX or higher.
* Build a table with indexes to each block, to be used instead of
* the regular stage 1/2 table.
*/
int32_t i;
for(i=0; i<(SBCS_FAST_LIMIT>>6); ++i) {
mbcsTable->sbcsIndex[i]=mbcsTable->fromUnicodeTable[mbcsTable->fromUnicodeTable[i>>4]+((i<<2)&0x3c)];
}
/* set SBCS_FAST_MAX to reflect the reach of sbcsIndex[] even if header->version[2]>(SBCS_FAST_MAX>>8) */
mbcsTable->maxFastUChar=SBCS_FAST_MAX;
} else {
/*
* MBCS: Stage 3 is allocated in 64-entry blocks for U+0000..MBCS_FAST_MAX or higher.
* The .cnv file is prebuilt with an additional stage table with indexes
* to each block.
*/
mbcsTable->mbcsIndex = reinterpret_cast<const uint16_t*>(
mbcsTable->fromUnicodeBytes +
(noFromU ? 0 : mbcsTable->fromUBytesLength));
mbcsTable->maxFastUChar = (static_cast<char16_t>(header->version[2]) << 8) | 0xff;
}
}
/* calculate a bit set of 4 ASCII characters per bit that round-trip to ASCII bytes */
{
uint32_t asciiRoundtrips=0xffffffff;
int32_t i;
/* Set the impl pointer here so that it is set for both extension-only and base tables. */
if(mbcsTable->utf8Friendly) {
if(mbcsTable->countStates==1) {
sharedData->impl=&_SBCSUTF8Impl;
} else {
if(mbcsTable->outputType==MBCS_OUTPUT_2) {
sharedData->impl=&_DBCSUTF8Impl;
}
}
}
if(mbcsTable->outputType==MBCS_OUTPUT_DBCS_ONLY || mbcsTable->outputType==MBCS_OUTPUT_2_SISO) {
/*
* MBCS_OUTPUT_DBCS_ONLY: No SBCS mappings, therefore ASCII does not roundtrip.
* MBCS_OUTPUT_2_SISO: Bypass the ASCII fastpath to handle prevLength correctly.
*/
mbcsTable->asciiRoundtrips=0;
}
}
/* the option does not apply, remove it */
cnv->options=pArgs->options&=~UCNV_OPTION_SWAP_LFNL;
}
}
}
if(uprv_strstr(pArgs->name, "18030")!=nullptr) {
if(uprv_strstr(pArgs->name, "gb18030")!=nullptr || uprv_strstr(pArgs->name, "GB18030")!=nullptr) {
/* set a flag for GB 18030 mode, which changes the callback behavior */
cnv->options|=_MBCS_OPTION_GB18030;
}
} else if((uprv_strstr(pArgs->name, "KEIS")!=nullptr) || (uprv_strstr(pArgs->name, "keis")!=nullptr)) {
/* set a flag for KEIS converter, which changes the SI/SO character sequence */
cnv->options|=_MBCS_OPTION_KEIS;
} else if((uprv_strstr(pArgs->name, "JEF")!=nullptr) || (uprv_strstr(pArgs->name, "jef")!=nullptr)) {
/* set a flag for JEF converter, which changes the SI/SO character sequence */
cnv->options|=_MBCS_OPTION_JEF;
} else if((uprv_strstr(pArgs->name, "JIPS")!=nullptr) || (uprv_strstr(pArgs->name, "jips")!=nullptr)) {
/* set a flag for JIPS converter, which changes the SI/SO character sequence */
cnv->options|=_MBCS_OPTION_JIPS;
}
/* fix maxBytesPerUChar depending on outputType and options etc. */
if(outputType==MBCS_OUTPUT_2_SISO) {
cnv->maxBytesPerUChar=3; /* SO+DBCS */
}
limit=mbcsTable->countToUFallbacks;
if(limit>0) {
/* do a binary search for the fallback mapping */
toUFallbacks=mbcsTable->toUFallbacks;
start=0;
while(start<limit-1) {
i=(start+limit)/2;
if(offset<toUFallbacks[i].offset) {
limit=i;
} else {
start=i;
}
}
/* did we really find it? */
if(offset==toUFallbacks[start].offset) {
return toUFallbacks[start].codePoint;
}
}
return 0xfffe;
}
/* This version of ucnv_MBCSToUnicodeWithOffsets() is optimized for single-byte, single-state codepages. */
static void
ucnv_MBCSSingleToUnicodeWithOffsets(UConverterToUnicodeArgs *pArgs,
UErrorCode *pErrorCode) {
UConverter *cnv;
const uint8_t *source, *sourceLimit;
char16_t *target;
const char16_t *targetLimit;
int32_t *offsets;
const int32_t (*stateTable)[256];
int32_t sourceIndex;
int32_t entry;
char16_t c;
uint8_t action;
/* set up the local pointers */
cnv=pArgs->converter;
source = reinterpret_cast<const uint8_t*>(pArgs->source);
sourceLimit = reinterpret_cast<const uint8_t*>(pArgs->sourceLimit);
target=pArgs->target;
targetLimit=pArgs->targetLimit;
offsets=pArgs->offsets;
/* sourceIndex=-1 if the current character began in the previous buffer */
sourceIndex=0;
/* conversion loop */
while(source<sourceLimit) {
/*
* This following test is to see if available input would overflow the output.
* It does not catch output of more than one code unit that
* overflows as a result of a surrogate pair or callback output
* from the last source byte.
* Therefore, those situations also test for overflows and will
* then break the loop, too.
*/
if(target>=targetLimit) {
/* target is full */
*pErrorCode=U_BUFFER_OVERFLOW_ERROR;
break;
}
/* test the most common case first */
if(MBCS_ENTRY_FINAL_IS_VALID_DIRECT_16(entry)) {
/* output BMP code point */
*target++ = static_cast<char16_t>(MBCS_ENTRY_FINAL_VALUE_16(entry));
if(offsets!=nullptr) {
*offsets++=sourceIndex;
}
/* normal end of action codes: prepare for a new character */
++sourceIndex;
continue;
}
/*
* An if-else-if chain provides more reliable performance for
* the most common cases compared to a switch.
*/
action = static_cast<uint8_t>(MBCS_ENTRY_FINAL_ACTION(entry));
if(action==MBCS_STATE_VALID_DIRECT_20 ||
(action==MBCS_STATE_FALLBACK_DIRECT_20 && UCNV_TO_U_USE_FALLBACK(cnv))
) {
entry=MBCS_ENTRY_FINAL_VALUE(entry);
/* output surrogate pair */
*target++ = static_cast<char16_t>(0xd800 | static_cast<char16_t>(entry >> 10));
if(offsets!=nullptr) {
*offsets++=sourceIndex;
}
c = static_cast<char16_t>(0xdc00 | static_cast<char16_t>(entry & 0x3ff));
if(target<targetLimit) {
*target++=c;
if(offsets!=nullptr) {
*offsets++=sourceIndex;
}
} else {
/* target overflow */
cnv->UCharErrorBuffer[0]=c;
cnv->UCharErrorBufferLength=1;
*pErrorCode=U_BUFFER_OVERFLOW_ERROR;
break;
}
if(U_FAILURE(*pErrorCode)) {
/* not mappable or buffer overflow */
break;
}
}
}
/* write back the updated pointers */
pArgs->source = reinterpret_cast<const char*>(source);
pArgs->target=target;
pArgs->offsets=offsets;
}
/*
* This version of ucnv_MBCSSingleToUnicodeWithOffsets() is optimized for single-byte, single-state codepages
* that only map to and from the BMP.
* In addition to single-byte optimizations, the offset calculations
* become much easier.
*/
static void
ucnv_MBCSSingleToBMPWithOffsets(UConverterToUnicodeArgs *pArgs,
UErrorCode *pErrorCode) {
UConverter *cnv;
const uint8_t *source, *sourceLimit, *lastSource;
char16_t *target;
int32_t targetCapacity, length;
int32_t *offsets;
const int32_t (*stateTable)[256];
int32_t sourceIndex;
int32_t entry;
uint8_t action;
/* set up the local pointers */
cnv=pArgs->converter;
source = reinterpret_cast<const uint8_t*>(pArgs->source);
sourceLimit = reinterpret_cast<const uint8_t*>(pArgs->sourceLimit);
target=pArgs->target;
targetCapacity = static_cast<int32_t>(pArgs->targetLimit - pArgs->target);
offsets=pArgs->offsets;
/* sourceIndex=-1 if the current character began in the previous buffer */
sourceIndex=0;
lastSource=source;
/*
* since the conversion here is 1:1 char16_t:uint8_t, we need only one counter
* for the minimum of the sourceLength and targetCapacity
*/
length = static_cast<int32_t>(sourceLimit - source);
if(length<targetCapacity) {
targetCapacity=length;
}
#if MBCS_UNROLL_SINGLE_TO_BMP
/* unrolling makes it faster on Pentium III/Windows 2000 */
/* unroll the loop with the most common case */
unrolled:
if(targetCapacity>=16) {
int32_t count, loops, oredEntries;
/* were all 16 entries really valid? */
if(!MBCS_ENTRY_FINAL_IS_VALID_DIRECT_16(oredEntries)) {
/* no, return to the first of these 16 */
source-=16;
target-=16;
break;
}
} while(--count>0);
count=loops-count;
targetCapacity-=16*count;
/* test the most common case first */
if(MBCS_ENTRY_FINAL_IS_VALID_DIRECT_16(entry)) {
/* output BMP code point */
*target++ = static_cast<char16_t>(MBCS_ENTRY_FINAL_VALUE_16(entry));
--targetCapacity;
continue;
}
/*
* An if-else-if chain provides more reliable performance for
* the most common cases compared to a switch.
*/
action = static_cast<uint8_t>(MBCS_ENTRY_FINAL_ACTION(entry));
if(action==MBCS_STATE_FALLBACK_DIRECT_16) {
if(UCNV_TO_U_USE_FALLBACK(cnv)) {
/* output BMP code point */
*target++ = static_cast<char16_t>(MBCS_ENTRY_FINAL_VALUE_16(entry));
--targetCapacity;
continue;
}
} else if(action==MBCS_STATE_UNASSIGNED) {
/* just fall through */
} else if(action==MBCS_STATE_ILLEGAL) {
/* callback(illegal) */
*pErrorCode=U_ILLEGAL_CHAR_FOUND;
} else {
/* reserved, must never occur */
continue;
}
/* set offsets since the start or the last extension */
if(offsets!=nullptr) {
int32_t count = static_cast<int32_t>(source - lastSource);
/* predecrement: do not set the offset for the callback-causing character */
while(--count>0) {
*offsets++=sourceIndex++;
}
/* offset and sourceIndex are now set for the current character */
}
if(U_FAILURE(*pErrorCode)) {
/* not mappable or buffer overflow */
break;
}
/* recalculate the targetCapacity after an extension mapping */
targetCapacity = static_cast<int32_t>(pArgs->targetLimit - target);
length = static_cast<int32_t>(sourceLimit - source);
if(length<targetCapacity) {
targetCapacity=length;
}
}
#if MBCS_UNROLL_SINGLE_TO_BMP
/* unrolling makes it faster on Pentium III/Windows 2000 */
goto unrolled;
#endif
}
if(U_SUCCESS(*pErrorCode) && source<sourceLimit && target>=pArgs->targetLimit) {
/* target is full */
*pErrorCode=U_BUFFER_OVERFLOW_ERROR;
}
/* set offsets since the start or the last callback */
if(offsets!=nullptr) {
size_t count=source-lastSource;
while(count>0) {
*offsets++=sourceIndex++;
--count;
}
}
/* write back the updated pointers */
pArgs->source = reinterpret_cast<const char*>(source);
pArgs->target=target;
pArgs->offsets=offsets;
}
static UBool
hasValidTrailBytes(const int32_t (*stateTable)[256], uint8_t state) {
const int32_t *row=stateTable[state];
int32_t b, entry;
/* First test for final entries in this state for some commonly valid byte values. */
entry=row[0xa1];
if( !MBCS_ENTRY_IS_TRANSITION(entry) &&
MBCS_ENTRY_FINAL_ACTION(entry)!=MBCS_STATE_ILLEGAL
) {
return true;
}
entry=row[0x41];
if( !MBCS_ENTRY_IS_TRANSITION(entry) &&
MBCS_ENTRY_FINAL_ACTION(entry)!=MBCS_STATE_ILLEGAL
) {
return true;
}
/* Then test for final entries in this state. */
for(b=0; b<=0xff; ++b) {
entry=row[b];
if( !MBCS_ENTRY_IS_TRANSITION(entry) &&
MBCS_ENTRY_FINAL_ACTION(entry)!=MBCS_STATE_ILLEGAL
) {
return true;
}
}
/* Then recurse for transition entries. */
for(b=0; b<=0xff; ++b) {
entry=row[b];
if( MBCS_ENTRY_IS_TRANSITION(entry) &&
hasValidTrailBytes(stateTable, static_cast<uint8_t>(MBCS_ENTRY_TRANSITION_STATE(entry)))
) {
return true;
}
}
return false;
}
/*
* Is byte b a single/lead byte in this state?
* Recurse for transition states, because here we don't want to say that
* b is a lead byte if all byte sequences that start with b are illegal.
*/
static UBool
isSingleOrLead(const int32_t (*stateTable)[256], uint8_t state, UBool isDBCSOnly, uint8_t b) {
const int32_t *row=stateTable[state];
int32_t entry=row[b];
if(MBCS_ENTRY_IS_TRANSITION(entry)) { /* lead byte */
return hasValidTrailBytes(stateTable, static_cast<uint8_t>(MBCS_ENTRY_TRANSITION_STATE(entry)));
} else {
uint8_t action = static_cast<uint8_t>(MBCS_ENTRY_FINAL_ACTION(entry));
if(action==MBCS_STATE_CHANGE_ONLY && isDBCSOnly) {
return false; /* SI/SO are illegal for DBCS-only conversion */
} else {
return action!=MBCS_STATE_ILLEGAL;
}
}
}
/* use optimized function if possible */
cnv=pArgs->converter;
if(cnv->preToULength>0) {
/*
* pass sourceIndex=-1 because we continue from an earlier buffer
* in the future, this may change with continuous offsets
*/
ucnv_extContinueMatchToU(cnv, pArgs, -1, pErrorCode);
/* set up the local pointers */
source=(const uint8_t *)pArgs->source;
sourceLimit=(const uint8_t *)pArgs->sourceLimit;
target=pArgs->target;
targetLimit=pArgs->targetLimit;
offsets=pArgs->offsets;
/* get the converter state from UConverter */
offset=cnv->toUnicodeStatus;
byteIndex=cnv->toULength;
bytes=cnv->toUBytes;
/*
* if we are in the SBCS state for a DBCS-only converter,
* then load the DBCS state from the MBCS data
* (dbcsOnlyState==0 if it is not a DBCS-only converter)
*/
if((state=(uint8_t)(cnv->mode))==0) {
state=cnv->sharedData->mbcs.dbcsOnlyState;
}
/* sourceIndex=-1 if the current character began in the previous buffer */
sourceIndex=byteIndex==0 ? 0 : -1;
nextSourceIndex=0;
/* conversion loop */
while(source<sourceLimit) {
/*
* This following test is to see if available input would overflow the output.
* It does not catch output of more than one code unit that
* overflows as a result of a surrogate pair or callback output
* from the last source byte.
* Therefore, those situations also test for overflows and will
* then break the loop, too.
*/
if(target>=targetLimit) {
/* target is full */
*pErrorCode=U_BUFFER_OVERFLOW_ERROR;
break;
}
if(byteIndex==0) {
/* optimized loop for 1/2-byte input and BMP output */
if(offsets==nullptr) {
do {
entry=stateTable[state][*source];
if(MBCS_ENTRY_IS_TRANSITION(entry)) {
state=(uint8_t)MBCS_ENTRY_TRANSITION_STATE(entry);
offset=MBCS_ENTRY_TRANSITION_OFFSET(entry);
++source;
if( source<sourceLimit &&
MBCS_ENTRY_IS_FINAL(entry=stateTable[state][*source]) &&
MBCS_ENTRY_FINAL_ACTION(entry)==MBCS_STATE_VALID_16 &&
(c=unicodeCodeUnits[offset+MBCS_ENTRY_FINAL_VALUE_16(entry)])<0xfffe
) {
++source;
*target++=c;
state=(uint8_t)MBCS_ENTRY_FINAL_STATE(entry); /* typically 0 */
offset=0;
} else {
/* set the state and leave the optimized loop */
bytes[0]=*(source-1);
byteIndex=1;
break;
}
} else {
if(MBCS_ENTRY_FINAL_IS_VALID_DIRECT_16(entry)) {
/* output BMP code point */
++source;
*target++=(char16_t)MBCS_ENTRY_FINAL_VALUE_16(entry);
state=(uint8_t)MBCS_ENTRY_FINAL_STATE(entry); /* typically 0 */
} else {
/* leave the optimized loop */
break;
}
}
} while(source<sourceLimit && target<targetLimit);
} else /* offsets!=nullptr */ {
do {
entry=stateTable[state][*source];
if(MBCS_ENTRY_IS_TRANSITION(entry)) {
state=(uint8_t)MBCS_ENTRY_TRANSITION_STATE(entry);
offset=MBCS_ENTRY_TRANSITION_OFFSET(entry);
++source;
if( source<sourceLimit &&
MBCS_ENTRY_IS_FINAL(entry=stateTable[state][*source]) &&
MBCS_ENTRY_FINAL_ACTION(entry)==MBCS_STATE_VALID_16 &&
(c=unicodeCodeUnits[offset+MBCS_ENTRY_FINAL_VALUE_16(entry)])<0xfffe
) {
++source;
*target++=c;
if(offsets!=nullptr) {
*offsets++=sourceIndex;
sourceIndex=(nextSourceIndex+=2);
}
state=(uint8_t)MBCS_ENTRY_FINAL_STATE(entry); /* typically 0 */
offset=0;
} else {
/* set the state and leave the optimized loop */
++nextSourceIndex;
bytes[0]=*(source-1);
byteIndex=1;
break;
}
} else {
if(MBCS_ENTRY_FINAL_IS_VALID_DIRECT_16(entry)) {
/* output BMP code point */
++source;
*target++=(char16_t)MBCS_ENTRY_FINAL_VALUE_16(entry);
if(offsets!=nullptr) {
*offsets++=sourceIndex;
sourceIndex=++nextSourceIndex;
}
state=(uint8_t)MBCS_ENTRY_FINAL_STATE(entry); /* typically 0 */
} else {
/* leave the optimized loop */
break;
}
}
} while(source<sourceLimit && target<targetLimit);
}
/*
* these tests and break statements could be put inside the loop
* if C had "break outerLoop" like Java
*/
if(source>=sourceLimit) {
break;
}
if(target>=targetLimit) {
/* target is full */
*pErrorCode=U_BUFFER_OVERFLOW_ERROR;
break;
}
offset=0;
break;
}
} else if(action==MBCS_STATE_CHANGE_ONLY) {
/*
* This serves as a state change without any output.
* It is useful for reading simple stateful encodings,
* for example using just Shift-In/Shift-Out codes.
* The 21 unused bits may later be used for more sophisticated
* state transitions.
*/
if(cnv->sharedData->mbcs.dbcsOnlyState==0) {
byteIndex=0;
} else {
/* SI/SO are illegal for DBCS-only conversion */
state=(uint8_t)(cnv->mode); /* restore the previous state */
/* callback(illegal) */
*pErrorCode=U_ILLEGAL_CHAR_FOUND;
}
} else if(action==MBCS_STATE_FALLBACK_DIRECT_16) {
if(UCNV_TO_U_USE_FALLBACK(cnv)) {
/* output BMP code point */
*target++=(char16_t)MBCS_ENTRY_FINAL_VALUE_16(entry);
if(offsets!=nullptr) {
*offsets++=sourceIndex;
}
byteIndex=0;
}
} else if(action==MBCS_STATE_UNASSIGNED) {
/* just fall through */
} else if(action==MBCS_STATE_ILLEGAL) {
/* callback(illegal) */
*pErrorCode=U_ILLEGAL_CHAR_FOUND;
} else {
/* reserved, must never occur */
byteIndex=0;
}
/* end of action codes: prepare for a new character */
offset=0;
if(byteIndex==0) {
sourceIndex=nextSourceIndex;
} else if(U_FAILURE(*pErrorCode)) {
/* callback(illegal) */
if(byteIndex>1) {
/*
* Ticket 5691: consistent illegal sequences:
* - We include at least the first byte in the illegal sequence.
* - If any of the non-initial bytes could be the start of a character,
* we stop the illegal sequence before the first one of those.
*/
UBool isDBCSOnly = cnv->sharedData->mbcs.dbcsOnlyState != 0;
int8_t i;
for(i=1;
i<byteIndex && !isSingleOrLead(stateTable, state, isDBCSOnly, bytes[i]);
++i) {}
if(i<byteIndex) {
/* Back out some bytes. */
int8_t backOutDistance=byteIndex-i;
int32_t bytesFromThisBuffer=(int32_t)(source-(const uint8_t *)pArgs->source);
byteIndex=i; /* length of reported illegal byte sequence */
if(backOutDistance<=bytesFromThisBuffer) {
source-=backOutDistance;
} else {
/* Back out bytes from the previous buffer: Need to replay them. */
cnv->preToULength=(int8_t)(bytesFromThisBuffer-backOutDistance);
/* preToULength is negative! */
uprv_memcpy(cnv->preToU, bytes+i, -cnv->preToULength);
source=(const uint8_t *)pArgs->source;
}
}
}
break;
} else /* unassigned sequences indicated with byteIndex>0 */ {
/* try an extension mapping */
pArgs->source=(const char *)source;
byteIndex=_extToU(cnv, cnv->sharedData,
byteIndex, &source, sourceLimit,
&target, targetLimit,
&offsets, sourceIndex,
pArgs->flush,
pErrorCode);
sourceIndex=nextSourceIndex+=(int32_t)(source-(const uint8_t *)pArgs->source);
if(U_FAILURE(*pErrorCode)) {
/* not mappable or buffer overflow */
break;
}
}
}
/* set the converter state back into UConverter */
cnv->toUnicodeStatus=offset;
cnv->mode=state;
cnv->toULength=byteIndex;
/* write back the updated pointers */
pArgs->source=(const char *)source;
pArgs->target=target;
pArgs->offsets=offsets;
}
/*
* This version of ucnv_MBCSGetNextUChar() is optimized for single-byte, single-state codepages.
* We still need a conversion loop in case we find reserved action codes, which are to be ignored.
*/
static UChar32
ucnv_MBCSSingleGetNextUChar(UConverterToUnicodeArgs *pArgs,
UErrorCode *pErrorCode) {
UConverter *cnv;
const int32_t (*stateTable)[256];
const uint8_t *source, *sourceLimit;
int32_t entry;
uint8_t action;
/* set up the local pointers */
cnv=pArgs->converter;
source = reinterpret_cast<const uint8_t*>(pArgs->source);
sourceLimit = reinterpret_cast<const uint8_t*>(pArgs->sourceLimit);
if((cnv->options&UCNV_OPTION_SWAP_LFNL)!=0) {
stateTable=(const int32_t (*)[256])cnv->sharedData->mbcs.swapLFNLStateTable;
} else {
stateTable=cnv->sharedData->mbcs.stateTable;
}
/* no output because of empty input or only state changes */
*pErrorCode=U_INDEX_OUTOFBOUNDS_ERROR;
return 0xffff;
}
/*
* Version of _MBCSToUnicodeWithOffsets() optimized for single-character
* conversion without offset handling.
*
* When a character does not have a mapping to Unicode, then we return to the
* generic ucnv_getNextUChar() code for extension/GB 18030 and error/callback
* handling.
* We also defer to the generic code in other complicated cases and have them
* ultimately handled by _MBCSToUnicodeWithOffsets() itself.
*
* All normal mappings and errors are handled here.
*/
static UChar32 U_CALLCONV
ucnv_MBCSGetNextUChar(UConverterToUnicodeArgs *pArgs,
UErrorCode *pErrorCode) {
UConverter *cnv;
const uint8_t *source, *sourceLimit, *lastSource;
/* use optimized function if possible */
cnv=pArgs->converter;
if(cnv->preToULength>0) {
/* use the generic code in ucnv_getNextUChar() to continue with a partial match */
return UCNV_GET_NEXT_UCHAR_USE_TO_U;
}
if(cnv->sharedData->mbcs.unicodeMask&UCNV_HAS_SURROGATES) {
/*
* Using the generic ucnv_getNextUChar() code lets us deal correctly
* with the rare case of a codepage that maps single surrogates
* without adding the complexity to this already complicated function here.
*/
return UCNV_GET_NEXT_UCHAR_USE_TO_U;
} else if(cnv->sharedData->mbcs.countStates==1) {
return ucnv_MBCSSingleGetNextUChar(pArgs, pErrorCode);
}
/* set up the local pointers */
source = lastSource = reinterpret_cast<const uint8_t*>(pArgs->source);
sourceLimit = reinterpret_cast<const uint8_t*>(pArgs->sourceLimit);
/* get the converter state from UConverter */
offset=cnv->toUnicodeStatus;
/*
* if we are in the SBCS state for a DBCS-only converter,
* then load the DBCS state from the MBCS data
* (dbcsOnlyState==0 if it is not a DBCS-only converter)
*/
if ((state = static_cast<uint8_t>(cnv->mode)) == 0) {
state=cnv->sharedData->mbcs.dbcsOnlyState;
}
/* optimization for 1/2-byte input and BMP output */
if( source<sourceLimit &&
MBCS_ENTRY_IS_FINAL(entry=stateTable[state][*source]) &&
MBCS_ENTRY_FINAL_ACTION(entry)==MBCS_STATE_VALID_16 &&
(c=unicodeCodeUnits[offset+MBCS_ENTRY_FINAL_VALUE_16(entry)])<0xfffe
) {
++source;
state = static_cast<uint8_t>(MBCS_ENTRY_FINAL_STATE(entry)); /* typically 0 */
/* output BMP code point */
break;
}
} else {
/* save the previous state for proper extension mapping with SI/SO-stateful converters */
cnv->mode=state;
/* set the next state early so that we can reuse the entry variable */
state = static_cast<uint8_t>(MBCS_ENTRY_FINAL_STATE(entry)); /* typically 0 */
/*
* An if-else-if chain provides more reliable performance for
* the most common cases compared to a switch.
*/
action = static_cast<uint8_t>(MBCS_ENTRY_FINAL_ACTION(entry));
if(action==MBCS_STATE_VALID_DIRECT_16) {
/* output BMP code point */
c = static_cast<char16_t>(MBCS_ENTRY_FINAL_VALUE_16(entry));
break;
} else if(action==MBCS_STATE_VALID_16) {
offset+=MBCS_ENTRY_FINAL_VALUE_16(entry);
c=unicodeCodeUnits[offset];
if(c<0xfffe) {
/* output BMP code point */
break;
} else if(c==0xfffe) {
if(UCNV_TO_U_USE_FALLBACK(cnv) && (c=ucnv_MBCSGetFallback(&cnv->sharedData->mbcs, offset))!=0xfffe) {
break;
}
} else {
/* callback(illegal) */
*pErrorCode=U_ILLEGAL_CHAR_FOUND;
}
} else if(action==MBCS_STATE_VALID_16_PAIR) {
offset+=MBCS_ENTRY_FINAL_VALUE_16(entry);
c=unicodeCodeUnits[offset++];
if(c<0xd800) {
/* output BMP code point below 0xd800 */
break;
} else if(UCNV_TO_U_USE_FALLBACK(cnv) ? c<=0xdfff : c<=0xdbff) {
/* output roundtrip or fallback supplementary code point */
c=((c&0x3ff)<<10)+unicodeCodeUnits[offset]+(0x10000-0xdc00);
break;
} else if(UCNV_TO_U_USE_FALLBACK(cnv) ? (c&0xfffe)==0xe000 : c==0xe000) {
/* output roundtrip BMP code point above 0xd800 or fallback BMP code point */
c=unicodeCodeUnits[offset];
break;
} else if(c==0xffff) {
/* callback(illegal) */
*pErrorCode=U_ILLEGAL_CHAR_FOUND;
}
} else if(action==MBCS_STATE_VALID_DIRECT_20 ||
(action==MBCS_STATE_FALLBACK_DIRECT_20 && UCNV_TO_U_USE_FALLBACK(cnv))
) {
/* output supplementary code point */
c = static_cast<UChar32>(MBCS_ENTRY_FINAL_VALUE(entry) + 0x10000);
break;
} else if(action==MBCS_STATE_CHANGE_ONLY) {
/*
* This serves as a state change without any output.
* It is useful for reading simple stateful encodings,
* for example using just Shift-In/Shift-Out codes.
* The 21 unused bits may later be used for more sophisticated
* state transitions.
*/
if(cnv->sharedData->mbcs.dbcsOnlyState!=0) {
/* SI/SO are illegal for DBCS-only conversion */
state = static_cast<uint8_t>(cnv->mode); /* restore the previous state */
/* callback(illegal) */
*pErrorCode=U_ILLEGAL_CHAR_FOUND;
}
} else if(action==MBCS_STATE_FALLBACK_DIRECT_16) {
if(UCNV_TO_U_USE_FALLBACK(cnv)) {
/* output BMP code point */
c = static_cast<char16_t>(MBCS_ENTRY_FINAL_VALUE_16(entry));
break;
}
} else if(action==MBCS_STATE_UNASSIGNED) {
/* just fall through */
} else if(action==MBCS_STATE_ILLEGAL) {
/* callback(illegal) */
*pErrorCode=U_ILLEGAL_CHAR_FOUND;
} else {
/* reserved (must never occur), or only state change */
offset=0;
lastSource=source;
continue;
}
/* end of action codes: prepare for a new character */
offset=0;
if(c<0) {
if(U_SUCCESS(*pErrorCode) && source==sourceLimit && lastSource<source) {
/* incomplete character byte sequence */
cnv->toULength = static_cast<int8_t>(source - lastSource);
uprv_memcpy(cnv->toUBytes, lastSource, cnv->toULength);
*pErrorCode=U_TRUNCATED_CHAR_FOUND;
} else if(U_FAILURE(*pErrorCode)) {
/* callback(illegal) */
/*
* Ticket 5691: consistent illegal sequences:
* - We include at least the first byte in the illegal sequence.
* - If any of the non-initial bytes could be the start of a character,
* we stop the illegal sequence before the first one of those.
*/
UBool isDBCSOnly = static_cast<UBool>(cnv->sharedData->mbcs.dbcsOnlyState != 0);
uint8_t *bytes=cnv->toUBytes;
*bytes++=*lastSource++; /* first byte */
if(lastSource==source) {
cnv->toULength=1;
} else /* lastSource<source: multi-byte character */ {
int8_t i;
for(i=1;
lastSource<source && !isSingleOrLead(stateTable, state, isDBCSOnly, *lastSource);
++i
) {
*bytes++=*lastSource++;
}
cnv->toULength=i;
source=lastSource;
}
} else {
/* no output because of empty input or only state changes */
*pErrorCode=U_INDEX_OUTOFBOUNDS_ERROR;
}
c=0xffff;
}
/* set the converter state back into UConverter, ready for a new character */
cnv->toUnicodeStatus=0;
cnv->mode=state;
/* write back the updated pointer */
pArgs->source = reinterpret_cast<const char*>(source);
return c;
}
#if 0
/*
* Code disabled 2002dec09 (ICU 2.4) because it is not currently used in ICU. markus
* Removal improves code coverage.
*/
/**
* This version of ucnv_MBCSSimpleGetNextUChar() is optimized for single-byte, single-state codepages.
* It does not handle the EBCDIC swaplfnl option (set in UConverter).
* It does not handle conversion extensions (_extToU()).
*/
U_CFUNC UChar32
ucnv_MBCSSingleSimpleGetNextUChar(UConverterSharedData *sharedData,
uint8_t b, UBool useFallback) {
int32_t entry;
uint8_t action;
/*
* An if-else-if chain provides more reliable performance for
* the most common cases compared to a switch.
*/
action=(uint8_t)(MBCS_ENTRY_FINAL_ACTION(entry));
if(action==MBCS_STATE_VALID_DIRECT_20) {
/* output supplementary code point */
return 0x10000+MBCS_ENTRY_FINAL_VALUE(entry);
} else if(action==MBCS_STATE_FALLBACK_DIRECT_16) {
if(!TO_U_USE_FALLBACK(useFallback)) {
return 0xfffe;
}
/* output BMP code point */
return (char16_t)MBCS_ENTRY_FINAL_VALUE_16(entry);
} else if(action==MBCS_STATE_FALLBACK_DIRECT_20) {
if(!TO_U_USE_FALLBACK(useFallback)) {
return 0xfffe;
}
/* output supplementary code point */
return 0x10000+MBCS_ENTRY_FINAL_VALUE(entry);
} else if(action==MBCS_STATE_UNASSIGNED) {
return 0xfffe;
} else if(action==MBCS_STATE_ILLEGAL) {
return 0xffff;
} else {
/* reserved, must never occur */
return 0xffff;
}
}
#endif
/*
* This is a simple version of _MBCSGetNextUChar() that is used
* by other converter implementations.
* It only returns an "assigned" result if it consumes the entire input.
* It does not use state from the converter, nor error codes.
* It does not handle the EBCDIC swaplfnl option (set in UConverter).
* It handles conversion extensions but not GB 18030.
*
* Return value:
* U+fffe unassigned
* U+ffff illegal
* otherwise the Unicode code point
*/
U_CFUNC UChar32
ucnv_MBCSSimpleGetNextUChar(UConverterSharedData *sharedData,
const char *source, int32_t length,
UBool useFallback) {
const int32_t (*stateTable)[256];
const uint16_t *unicodeCodeUnits;
uint32_t offset;
uint8_t state, action;
UChar32 c;
int32_t i, entry;
if(length<=0) {
/* no input at all: "illegal" */
return 0xffff;
}
#if 0
/*
* Code disabled 2002dec09 (ICU 2.4) because it is not currently used in ICU. markus
* TODO In future releases, verify that this function is never called for SBCS
* conversions, i.e., that sharedData->mbcs.countStates==1 is still true.
* Removal improves code coverage.
*/
/* use optimized function if possible */
if(sharedData->mbcs.countStates==1) {
if(length==1) {
return ucnv_MBCSSingleSimpleGetNextUChar(sharedData, (uint8_t)*source, useFallback);
} else {
return 0xffff; /* illegal: more than a single byte for an SBCS converter */
}
}
#endif
/* set up the local pointers */
stateTable=sharedData->mbcs.stateTable;
unicodeCodeUnits=sharedData->mbcs.unicodeCodeUnits;
/* converter state */
offset=0;
state=sharedData->mbcs.dbcsOnlyState;
/* use optimized function if possible */
cnv=pArgs->converter;
unicodeMask=cnv->sharedData->mbcs.unicodeMask;
/* set up the local pointers */
source=pArgs->source;
sourceLimit=pArgs->sourceLimit;
target = reinterpret_cast<uint8_t*>(pArgs->target);
targetCapacity = static_cast<int32_t>(pArgs->targetLimit - pArgs->target);
offsets=pArgs->offsets;
while(source<sourceLimit) {
/*
* This following test is to see if available input would overflow the output.
* It does not catch output of more than one byte that
* overflows as a result of a multi-byte character or callback output
* from the last source character.
* Therefore, those situations also test for overflows and will
* then break the loop, too.
*/
if(targetCapacity>0) {
/*
* Get a correct Unicode code point:
* a single char16_t for a BMP code point or
* a matched surrogate pair for a "supplementary code point".
*/
c=*source++;
++nextSourceIndex;
if(c<=0x7f && IS_ASCII_ROUNDTRIP(c, asciiRoundtrips)) {
*target++ = static_cast<uint8_t>(c);
if(offsets!=nullptr) {
*offsets++=sourceIndex;
sourceIndex=nextSourceIndex;
}
--targetCapacity;
c=0;
continue;
}
/*
* utf8Friendly table: Test for <=0xd7ff rather than <=MBCS_FAST_MAX
* to avoid dealing with surrogates.
* MBCS_FAST_MAX must be >=0xd7ff.
*/
if(c<=0xd7ff) {
value=DBCS_RESULT_FROM_MOST_BMP(mbcsIndex, (const uint16_t *)bytes, c);
/* There are only roundtrips (!=0) and no-mapping (==0) entries. */
if(value==0) {
goto unassigned;
}
/* output the value */
} else {
/*
* This also tests if the codepage maps single surrogates.
* If it does, then surrogates are not paired but mapped separately.
* Note that in this case unmatched surrogates are not detected.
*/
if(U16_IS_SURROGATE(c) && !(unicodeMask&UCNV_HAS_SURROGATES)) {
if(U16_IS_SURROGATE_LEAD(c)) {
getTrail:
if(source<sourceLimit) {
/* test the following code unit */
char16_t trail=*source;
if(U16_IS_TRAIL(trail)) {
++source;
++nextSourceIndex;
c=U16_GET_SUPPLEMENTARY(c, trail);
if(!(unicodeMask&UCNV_HAS_SUPPLEMENTARY)) {
/* BMP-only codepages are stored without stage 1 entries for supplementary code points */
/* callback(unassigned) */
goto unassigned;
}
/* convert this supplementary code point */
/* exit this condition tree */
} else {
/* this is an unmatched lead code unit (1st surrogate) */
/* callback(illegal) */
*pErrorCode=U_ILLEGAL_CHAR_FOUND;
break;
}
} else {
/* no more input */
break;
}
} else {
/* this is an unmatched trail code unit (2nd surrogate) */
/* callback(illegal) */
*pErrorCode=U_ILLEGAL_CHAR_FOUND;
break;
}
}
/* convert the Unicode code point in c into codepage bytes */
stage2Entry=MBCS_STAGE_2_FROM_U(table, c);
/* get the bytes and the length for the output */
/* MBCS_OUTPUT_2 */
value=MBCS_VALUE_2_FROM_STAGE_2(bytes, stage2Entry, c);
/* is this code point assigned, or do we use fallbacks? */
if(!(MBCS_FROM_U_IS_ROUNDTRIP(stage2Entry, c) ||
(UCNV_FROM_U_USE_FALLBACK(cnv, c) && value!=0))
) {
/*
* We allow a0 byte output if the "assigned" bit is set for this entry.
* There is no way with this data structure for fallback output
* to be a zero byte.
*/
if(U_FAILURE(*pErrorCode)) {
/* not mappable or buffer overflow */
break;
} else {
/* a mapping was written to the target, continue */
/* recalculate the targetCapacity after an extension mapping */
targetCapacity = static_cast<int32_t>(pArgs->targetLimit - reinterpret_cast<char*>(target));
/* normal end of conversion: prepare for a new character */
sourceIndex=nextSourceIndex;
continue;
}
}
}
/* write the output character bytes from value and length */
/* from the first if in the loop we know that targetCapacity>0 */
if(value<=0xff) {
/* this is easy because we know that there is enough space */
*target++ = static_cast<uint8_t>(value);
if(offsets!=nullptr) {
*offsets++=sourceIndex;
}
--targetCapacity;
} else /* length==2 */ {
*target++ = static_cast<uint8_t>(value >> 8);
if(2<=targetCapacity) {
*target++ = static_cast<uint8_t>(value);
if(offsets!=nullptr) {
*offsets++=sourceIndex;
*offsets++=sourceIndex;
}
targetCapacity-=2;
} else {
if(offsets!=nullptr) {
*offsets++=sourceIndex;
}
cnv->charErrorBuffer[0] = static_cast<char>(value);
cnv->charErrorBufferLength=1;
/* normal end of conversion: prepare for a new character */
c=0;
sourceIndex=nextSourceIndex;
continue;
} else {
/* target is full */
*pErrorCode=U_BUFFER_OVERFLOW_ERROR;
break;
}
}
/* set the converter state back into UConverter */
cnv->fromUChar32=c;
/* write back the updated pointers */
pArgs->source=source;
pArgs->target = reinterpret_cast<char*>(target);
pArgs->offsets=offsets;
}
/* This version of ucnv_MBCSFromUnicodeWithOffsets() is optimized for single-byte codepages. */
static void
ucnv_MBCSSingleFromUnicodeWithOffsets(UConverterFromUnicodeArgs *pArgs,
UErrorCode *pErrorCode) {
UConverter *cnv;
const char16_t *source, *sourceLimit;
uint8_t *target;
int32_t targetCapacity;
int32_t *offsets;
const uint16_t *table;
const uint16_t *results;
UChar32 c;
int32_t sourceIndex, nextSourceIndex;
uint16_t value, minValue;
UBool hasSupplementary;
/* set up the local pointers */
cnv=pArgs->converter; source=pArgs->source;
sourceLimit=pArgs->sourceLimit;
target = reinterpret_cast<uint8_t*>(pArgs->target);
targetCapacity = static_cast<int32_t>(pArgs->targetLimit - pArgs->target);
offsets=pArgs->offsets;
if(cnv->useFallback) {
/* use all roundtrip and fallback results */
minValue=0x800;
} else {
/* use only roundtrips and fallbacks from private-use characters */
minValue=0xc00;
}
hasSupplementary = static_cast<UBool>(cnv->sharedData->mbcs.unicodeMask & UCNV_HAS_SUPPLEMENTARY);
/* get the converter state from UConverter */
c=cnv->fromUChar32;
/* sourceIndex=-1 if the current character began in the previous buffer */
sourceIndex= c==0 ? 0 : -1;
nextSourceIndex=0;
while(source<sourceLimit) {
/*
* This following test is to see if available input would overflow the output.
* It does not catch output of more than one byte that
* overflows as a result of a multi-byte character or callback output
* from the last source character.
* Therefore, those situations also test for overflows and will
* then break the loop, too.
*/
if(targetCapacity>0) {
/*
* Get a correct Unicode code point:
* a single char16_t for a BMP code point or
* a matched surrogate pair for a"supplementary code point".
*/
c=*source++;
++nextSourceIndex;
if(U16_IS_SURROGATE(c)) {
if(U16_IS_SURROGATE_LEAD(c)) {
getTrail:
if(source<sourceLimit) {
/* test the following code unit */
char16_t trail=*source;
if(U16_IS_TRAIL(trail)) {
++source;
++nextSourceIndex;
c=U16_GET_SUPPLEMENTARY(c, trail);
if(!hasSupplementary) {
/* BMP-only codepages are stored without stage 1 entries for supplementary code points */
/* callback(unassigned) */
goto unassigned;
}
/* convert this supplementary code point */
/* exit this condition tree */
} else {
/* this is an unmatched lead code unit (1st surrogate) */
/* callback(illegal) */
*pErrorCode=U_ILLEGAL_CHAR_FOUND;
break;
}
} else {
/* no more input */
break;
}
} else {
/* this is an unmatched trail code unit (2nd surrogate) */
/* callback(illegal) */
*pErrorCode=U_ILLEGAL_CHAR_FOUND;
break;
}
}
/* convert the Unicode code point in c into codepage bytes */
value=MBCS_SINGLE_RESULT_FROM_U(table, results, c);
/* is this code point assigned, or do we use fallbacks? */
if(value>=minValue) {
/* assigned, write the output character bytes from value and length */
/* length==1 */
/* this is easy because we know that there is enough space */
*target++ = static_cast<uint8_t>(value);
if(offsets!=nullptr) {
*offsets++=sourceIndex;
}
--targetCapacity;
/* normal end of conversion: prepare for a new character */
c=0;
sourceIndex=nextSourceIndex;
} else { /* unassigned */
unassigned:
/* try an extension mapping */
pArgs->source=source;
c=_extFromU(cnv, cnv->sharedData,
c, &source, sourceLimit,
&target, target+targetCapacity,
&offsets, sourceIndex,
pArgs->flush,
pErrorCode);
nextSourceIndex += static_cast<int32_t>(source - pArgs->source);
if(U_FAILURE(*pErrorCode)) {
/* not mappable or buffer overflow */
break;
} else {
/* a mapping was written to the target, continue */
/* recalculate the targetCapacity after an extension mapping */
targetCapacity = static_cast<int32_t>(pArgs->targetLimit - reinterpret_cast<char*>(target));
/* normal end of conversion: prepare for a new character */
sourceIndex=nextSourceIndex;
}
}
} else {
/* target is full */
*pErrorCode=U_BUFFER_OVERFLOW_ERROR;
break;
}
}
/* set the converter state back into UConverter */
cnv->fromUChar32=c;
/* write back the updated pointers */
pArgs->source=source;
pArgs->target = reinterpret_cast<char*>(target);
pArgs->offsets=offsets;
}
/*
* This version of ucnv_MBCSFromUnicode() is optimized for single-byte codepages
* that map only to and from the BMP.
* In addition to single-byte/state optimizations, the offset calculations
* become much easier.
* It would be possible to use the sbcsIndex for UTF-8-friendly tables,
* but measurements have shown that this diminishes performance
* in more cases than it improves it.
* See SVN revision 21013 (2007-feb-06) for the last version with #if switches
* for various MBCS and SBCS optimizations.
*/
static void
ucnv_MBCSSingleFromBMPWithOffsets(UConverterFromUnicodeArgs *pArgs,
UErrorCode *pErrorCode) {
UConverter *cnv;
const char16_t *source, *sourceLimit, *lastSource;
uint8_t *target;
int32_t targetCapacity, length;
int32_t *offsets;
/* set up the local pointers */
cnv=pArgs->converter; source=pArgs->source;
sourceLimit=pArgs->sourceLimit;
target = reinterpret_cast<uint8_t*>(pArgs->target);
targetCapacity = static_cast<int32_t>(pArgs->targetLimit - pArgs->target);
offsets=pArgs->offsets;
if(cnv->useFallback) {
/* use all roundtrip and fallback results */
minValue=0x800;
} else {
/* use only roundtrips and fallbacks from private-use characters */
minValue=0xc00;
}
/* get the converter state from UConverter */
c=cnv->fromUChar32;
/* sourceIndex=-1 if the current character began in the previous buffer */
sourceIndex= c==0 ? 0 : -1;
lastSource=source;
/*
* since the conversion here is 1:1 char16_t:uint8_t, we need only one counter
* for the minimum of the sourceLength and targetCapacity
*/
length = static_cast<int32_t>(sourceLimit - source);
if(length<targetCapacity) {
targetCapacity=length;
}
#if MBCS_UNROLL_SINGLE_FROM_BMP
/* unrolling makes it slower on Pentium III/Windows 2000?! */
/* unroll the loop with the most common case */
unrolled:
if(targetCapacity>=4) {
int32_t count, loops;
uint16_t andedValues;
/* were all 4 entries really valid? */
if(andedValues<minValue) {
/* no, return to the first of these 4 */ source-=4;
target-=4;
break;
}
} while(--count>0);
count=loops-count;
targetCapacity-=4*count;
while(targetCapacity>0) {
/*
* Get a correct Unicode code point:
* a single char16_t for a BMP code point or
* a matched surrogate pair for a"supplementary code point".
*/
c=*source++;
/*
* Do not immediately check for single surrogates:
* Assume that they are unassigned and check for them in that case.
* This speeds up the conversion of assigned characters.
*/
/* convert the Unicode code point in c into codepage bytes */
if(c<=0x7f && IS_ASCII_ROUNDTRIP(c, asciiRoundtrips)) {
*target++ = static_cast<uint8_t>(c);
--targetCapacity;
c=0;
continue;
}
value=MBCS_SINGLE_RESULT_FROM_U(table, results, c);
/* is this code point assigned, or do we use fallbacks? */
if(value>=minValue) {
/* assigned, write the output character bytes from value and length */
/* length==1 */
/* this is easy because we know that there is enough space */
*target++ = static_cast<uint8_t>(value);
--targetCapacity;
/* normal end of conversion: prepare for a new character */
c=0;
continue;
} else if(!U16_IS_SURROGATE(c)) {
/* normal, unassigned BMP character */
} else if(U16_IS_SURROGATE_LEAD(c)) {
getTrail:
if(source<sourceLimit) {
/* test the following code unit */
char16_t trail=*source;
if(U16_IS_TRAIL(trail)) {
++source;
c=U16_GET_SUPPLEMENTARY(c, trail);
/* this codepage does not map supplementary code points */
/* callback(unassigned) */
} else {
/* this is an unmatched lead code unit (1st surrogate) */
/* callback(illegal) */
*pErrorCode=U_ILLEGAL_CHAR_FOUND;
break;
}
} else {
/* no more input */
if (pArgs->flush) {
*pErrorCode=U_TRUNCATED_CHAR_FOUND;
}
break;
}
} else {
/* this is an unmatched trail code unit (2nd surrogate) */
/* callback(illegal) */
*pErrorCode=U_ILLEGAL_CHAR_FOUND;
break;
}
/* c does not have a mapping */
/* get the number of code units for c to correctly advance sourceIndex */
length=U16_LENGTH(c);
/* set offsets since the start or the last extension */
if(offsets!=nullptr) {
int32_t count = static_cast<int32_t>(source - lastSource);
/* do not set the offset for this character */
count-=length;
while(count>0) {
*offsets++=sourceIndex++;
--count;
}
/* offsets and sourceIndex are now set for the current character */
}
if(U_FAILURE(*pErrorCode)) {
/* not mappable or buffer overflow */
break;
} else {
/* a mapping was written to the target, continue */
/* recalculate the targetCapacity after an extension mapping */
targetCapacity = static_cast<int32_t>(pArgs->targetLimit - reinterpret_cast<char*>(target));
length = static_cast<int32_t>(sourceLimit - source);
if(length<targetCapacity) {
targetCapacity=length;
}
}
#if MBCS_UNROLL_SINGLE_FROM_BMP
/* unrolling makes it slower on Pentium III/Windows 2000?! */
goto unrolled;
#endif
}
if(U_SUCCESS(*pErrorCode) && source<sourceLimit && target>=(uint8_t *)pArgs->targetLimit) {
/* target is full */
*pErrorCode=U_BUFFER_OVERFLOW_ERROR;
}
/* set offsets since the start or the last callback */
if(offsets!=nullptr) {
size_t count=source-lastSource;
if (count > 0 && *pErrorCode == U_TRUNCATED_CHAR_FOUND) {
/*
Caller gave us a partial supplementary character,
which this function couldn't convert in any case.
The callback will handle the offset.
*/
count--;
}
while(count>0) {
*offsets++=sourceIndex++;
--count;
}
}
/* set the converter state back into UConverter */
cnv->fromUChar32=c;
/* write back the updated pointers */
pArgs->source=source;
pArgs->target = reinterpret_cast<char*>(target);
pArgs->offsets=offsets;
}
if(cnv->preFromUFirstCP>=0) {
/*
* pass sourceIndex=-1 because we continue from an earlier buffer
* in the future, this may change with continuous offsets
*/
ucnv_extContinueMatchFromU(cnv, pArgs, -1, pErrorCode);
/* use optimized function if possible */
outputType=cnv->sharedData->mbcs.outputType;
unicodeMask=cnv->sharedData->mbcs.unicodeMask;
if(outputType==MBCS_OUTPUT_1 && !(unicodeMask&UCNV_HAS_SURROGATES)) {
if(!(unicodeMask&UCNV_HAS_SUPPLEMENTARY)) {
ucnv_MBCSSingleFromBMPWithOffsets(pArgs, pErrorCode);
} else {
ucnv_MBCSSingleFromUnicodeWithOffsets(pArgs, pErrorCode);
}
return;
} else if(outputType==MBCS_OUTPUT_2 && cnv->sharedData->mbcs.utf8Friendly) {
ucnv_MBCSDoubleFromUnicodeWithOffsets(pArgs, pErrorCode);
return;
}
/* set up the local pointers */ source=pArgs->source;
sourceLimit=pArgs->sourceLimit;
target=(uint8_t *)pArgs->target;
targetCapacity=(int32_t)(pArgs->targetLimit-pArgs->target);
offsets=pArgs->offsets;
/* get the converter state from UConverter */
c=cnv->fromUChar32;
SMTP(8( =""NAME>/
prevLength=cnv->The Postfix SMTP+LMTPimplements andLMTP
berun the< href"8.">b<b>(8</a processmanager.The process name, <b>smtp</b> or
/* set the real value */
=1;
} else {
preventjava.lang.StringIndexOutOfBoundsException: Range [47, 36) out of bounds for length 73
prevLength=0;
}
/* sourceIndex=-1 if (:<><b
java.lang.StringIndexOutOfBoundsException: Range [22, 19) out of bounds for length 23
sourceIndex=pathname interpreted java.lang.StringIndexOutOfBoundsException: Range [47, 46) out of bounds for length 78
nextSourceIndex0;
/* Get the SI/SO character for the converter */
siLength = static_cast<( b<b) IPv6 address beformattedas
soLengthstatic_cast<>getSISOBytesSO >options,)java.lang.StringIndexOutOfBoundsException: Index 77 out of bounds for length 77
/* conversion loop */
/*
* This is another piece of ugly code:
* A goto address the is unde
call
* It saves me to check in each loop iteration a check of if(c==0)
* and duplicating the trail-surrogate-handling code in the else
* branch of that check.
* I could not find any other way to get around this other than
* using a function java.lang.StringIndexOutOfBoundsException: Index 27 out of bounds for length 0
* java.lang.StringIndexOutOfBoundsException: Range [44, 42) out of bounds for length 78
*
*Markus 2000-jul19
*/
if(c!=0 && targetCapacity>0) {
java.lang.StringIndexOutOfBoundsException: Range [13, 12) out of bounds for length 22
while(<sourceLimit){
/a"//.etf/html/"3207/a(TARTTLScommand)
low the output.
* It does not catcha=https/toolsietforghtmlrfc6533"RFC </ Internationalized Delivery Status Notifications)
f amultibyte orcallback
* from the last source character.
Thereforethose alsooverflowswill
* then tified of bounces, protocol problems, and of other trouble
*/
the sameprogram,and protocol and parameters
/*
*GetaUnicode point
BMP point a surrogatepair "code "java.lang.StringIndexOutOfBoundsException: Index 73 out of bounds for length 73
*/NeversendEHLOat start ofanSMTPsession.
c=*source++;
++nextSourceIndex;.ltCRgt<LF" workaround
*target++=(uint8_t)c;
if(offsets!=Lookup tables, indexed by the remote SMTP server address, with
*offsets++=sourceIndex;
=sourceIndex;
SMTP greet with a5 status code.
}
java.lang.StringIndexOutOfBoundsException: Range [33, 32) out of bounds for length 33
caseinsensitive lists ofEHLOkeywords( starttls
continue;
/*
* utf8Friendly table: Test
to dealingwith .
* Availablein Postfixversion .and
/
<b><a href.5.java.lang.StringIndexOutOfBoundsException: Range [35, 34) out of bounds for length 105
=java.lang.StringIndexOutOfBoundsException: Range [32, 31) out of bounds for length 38
java.lang.StringIndexOutOfBoundsException: Range [63, 62) out of bounds for length 119
/* There are
java.lang.StringIndexOutOfBoundsException: Index 18 out of bounds for length 0
MBCS_OUTPUT_2:
value=
if
if(value==0) {
goto unassigned;
} else {
length=1;
} else {
=2;
P version33 andlater
break
case MBCS_OUTPUT_2_SISOin Postfix3.andlater
/* 1/2-java.lang.StringIndexOutOfBoundsException: Range [7, 1) out of bounds for length 42
/*
* Save the old Optional setting that avoids services<bdata
the the _time,a
* Thenjava.lang.StringIndexOutOfBoundsException: Range [20, 19) out of bounds for length 78
* java.lang.StringIndexOutOfBoundsException: Range [30, 29) out of bounds for length 99
becausethe state foracharacterthatis .
* Enables discovery for the specified service(s) using DNS SRV
* incase the functionchangeditfor itsoutput.
*/
cnv->to orIP lookupasif SRVrecord lookup
value=((const uint16_t
<href5htmlmime_boundary_length_limitmime_boundary_length_limit<a>(2048<b
if(value==0) {
goto unassigned;
} <b><a href="postconf.5.smtp_send_xforward_command>smtp_send_xforward_command</> (o/b
length=1<>a =""SASL AUTHENTICATION </a><b
<<href"postconf..html#smtp_sasl_password_maps"smtp_sasl_password_maps</a> (empty)</b>
*change double-byte mode to - /
if(siLength= ){
uint32_t)siBytes[0<<8;
java.lang.StringIndexOutOfBoundsException: Range [19, 17) out of bounds for length 105
} else if (siLength == 2Enable sender-dependentauthentication the PostfixSMTP
value|=(uint32_t)siBytes[1]<<8;
disables SMTP connectioncaching ensure that fromdif-
length = 3;
;
}
} else {
if(When a SMTP rejects SASL authentication request
length=2;
} else {
/* change from single-byte mode to double-byte */
if (soLength == 1) {
value|=(uint32_t)soBytes[0]<<16;
length = 3;
=)java.lang.StringIndexOutOfBoundsException: Index 55 out of bounds for length 55
value|=(b><a href="postconf.5.html#smtp_tls_CAfile</>empty/b
value|=(uint32_t)soBytes[0]<<24;
lengththat Postfix to aremote SMTPserver
}
prevLength=2;
}
break;
case MBCS_OUTPUT_DBCS_ONLY:
/* table with single-byte results, but only DBCS mappings used */
value=((const uint16_t *)bytes)[value +(c&0x3f)];
(=0xff
/* no mapping or SBCS result, not taken for DBCS-only */
goto unassigned;
, obsoletea=postconf5.#"smtp_tls_per_site/>parameter.
length=2;
break;
case : p=bytes+(value+(c&<ba =postconf5.tmlsmtp_tls_session_cache_timeoutsmtp_tls_session_cache_timeout/>(s)b>
value=((uint32_t)*p<<16)|((uint32_t)p[1]<<8)|p[2];
if(value<=0xff) {
if(value==0) {
goto unassigned;
} else {
length=1;
OpenSSL for"ULL" grade
} else if(value<=0xffff)<><href"5html#ls_low_cipherlist"><a(' -d' )<>
length=2;
} bapostconf.sjava.lang.StringIndexOutOfBoundsException: Range [76, 74) out of bounds for length 216
length=3;
case MBCS_OUTPUT_4 java.lang.StringIndexOutOfBoundsException: Range [33, 32) out of bounds for length 46 32t)[ (&x3f]
if(value<=0xff) {
if(value==0) {
goto unassigned;
} else {
length=1;
else (<=xffff)
=2
} else if(value<=0xffffff) {
length=3;
} else {
length=4;
}
java.lang.StringIndexOutOfBoundsException: Index 26 out of bounds for length 26
case :
value
/* EUC 16-bit TheTLS MX hostswith" recordswhenthe
ifAvailable 3and later
=)
;
} else multipledeliveriesperencryptedconnection.
java.lang.StringIndexOutOfBoundsException: Range [28, 27) out of bounds for length 74
}
} else if((value&0x8000)==0) {
value|0x8e8000;
length=3;
} else if((value&0x80)==0) {
value|=0x8f0080;
length=3;
} else {
length=2;
}
break;
if(value<=0xff) {
ifvalue=0
goto unassigned;
} else {
length=1;
}
} else if(valueanhost if itsnamematches any policyMX pat-
=2;
}(valuex800000=0 java.lang.StringIndexOutOfBoundsException: Index 52 out of bounds for length 52
value|=0
length=4;
} else if((value&0x8000)==0) {
value|=0x8f008000;
length=4;
}
length=3;
}
break;
java.lang.StringIndexOutOfBoundsException: Range [24, 23) out of bounds for length 24
/ "html></a>(empty)</>
java.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
java.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
SMTP limit forcompleting a TCP connec-
* In reality, this is b><a href="postconf.5.html#smtp_helo_timeout"300)
* Not having a default branch also causes warnings withcommandandjava.lang.StringIndexOutOfBoundsException: Range [45, 43) out of bounds for length 78
* some compilers.
*/ 0
length=0;
break;
/* output the value */
} else {
/*
* This also tests if thejava.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
it but
* and for r theremote serverresponsejava.lang.StringIndexOutOfBoundsException: Index 64 out of bounds for length 64
*/
(c &(java.lang.StringIndexOutOfBoundsException: Range [76, 75) out of bounds for length 79
if(U16_IS_SURROGATE_LEAD(c)) {
getTrail:
(source) {
/* test thea=postconf..#relayhost"relayhost</,or zero (no ).
char16_t SMTP timelimit forsendingthe commandjava.lang.StringIndexOutOfBoundsException: Index 78 out of bounds for length 78
if(U16_IS_TRAIL(trail)) {
++source;
++nextSourceIndex<>".5java.lang.StringIndexOutOfBoundsException: Range [35, 34) out of bounds for length 121
c=U16_GET_SUPPLEMENTARY(c, trail);
if(!(unicodeMaskU)) {
/ only are 1java.lang.StringIndexOutOfBoundsException: Range [93, 92) out of bounds for length 125
*/
/* callback(unassigned) */
goto unassigned;
}
/* convert this supplementary tions.
tree */
e{
(1stsurrogate)*
/* callback
*pErrorCode;
break;
}
} java.lang.StringIndexOutOfBoundsException: Range [0, 30) out of bounds for length 0
/* no more input */
b><a href="postconf.5.html><()/
}
} else {
/* this is an unmatched trail code unit (2cy_limit"transport_destination_concurrency_limit> $ahref=postconf..#default_destination_concurrency_limitdefault_destination_concur</a></java.lang.StringIndexOutOfBoundsException: Index 223 out of bounds for length 223
/* callback(illegal) */
*pErrorCode=U_ILLEGAL_CHAR_FOUND;
break;
}
}
/* convert the Unicode code point in c into codepage me ofthe message delivery transport.
/
*java.lang.StringIndexOutOfBoundsException: Range [29, 28) out of bounds for length 82
the file
*
java.lang.StringIndexOutOfBoundsException: Range [45, 44) out of bounds for length 84 form isused forDNSlookups
*
* The result consists of a32-bit value from stage 2 and
given withthe<hrefp..java.lang.StringIndexOutOfBoundsException: Range [79, 77) out of bounds for length 109
* The pointer points to the character's bytes in stage Optional list of nexthop destination, remote client or server
* Bits 15..0 of the verboselogging levelto increase by the amount specified in
* 3116 flags java.lang.StringIndexOutOfBoundsException: Range [75, 74) out of bounds for length 77
* 16 in the areroundtrip-assignedjava.lang.StringIndexOutOfBoundsException: Index 73 out of bounds for length 73
*
* For 2-byte and 4-byte codepages, the bytes are stored as uint16_t
* respectively as uint32_t, in <hrefpostconf.5.header_checks</> a=p5htmlb>body_checks</.
* For 3-byte codepages, list that to .
*
* For encodings that use only either 0x8e or 0x8f as the first
of theirbyte sequences, the first two bytes in
* this third stage indicate with their 7th bits whether these bytes
* are to
* one of the two Single-Shift codes. With this, the How much time Postfix daemon process may take handlea
ytefewer characterthan actual length
* EUC <b><a href="postconf.#java.lang.StringIndexOutOfBoundsException: Range [66, 65) out of bounds for length 109
*
* Other than that, leading zero bytes are removed and the other
* bytes output. A single zero byte may be output if the "assigned"
* bit in stage 2 was on.
* The data structure java.lang.StringIndexOutOfBoundsException: Index 42 out of bounds for length 0
* and also does not allow output of leading zeros.
*/
stage2Entry=MBCS_STAGE_2_FROM_U(table, c);
/* get the bytes and the length for the output */
switch(outputType) {
case MBCS_OUTPUT_2:
=(bytes , c)
(<0 java.lang.StringIndexOutOfBoundsException: Index 37 out of bounds for length 37
length=1;
} else {
length=2;
}
break;
case MBCS_OUTPUT_2_SISO:
/* 1/2-byte stateful with Shift-In/Shift-Out<> =.5.html></>(</
/*
old state in object
* numericalnetwork thatthePostfix
* Then, if this b><a href="postconf.5.html#smtp_bind_address6 e)>
* is not taken, the callback code must not save the new state in the converter
* because the newstate isfora that isnot output.
* mustrestore
* in case the callback function changed it for its output.
*/
cnv->fromUnicodeStatus the of-java.lang.StringIndexOutOfBoundsException: Range [54, 53) out of bounds for length 72
valueMBCS_VALUE_2_FROM_STAGE_2, stage2Entry java.lang.StringIndexOutOfBoundsException: Index 75 out of bounds for length 75
if(value<=0xff) {
if(value==0 && MBCS_FROM_U_IS_ROUNDTRIP(stage2Entry, c)==0) {
no mapping leave value=0*
length=0;
if(<1)java.lang.StringIndexOutOfBoundsException: Index 50 out of bounds for length 50
length=1;
} else {
/* change from double-byte mode to single-byte *
if (siLength == 1) {
value|=(uint32_t)siBytes[0]<<8;
length = 2;
ifjava.lang.StringIndexOutOfBoundsException: Range [48, 47) out of bounds for length 55
value|=(b><a href="postconf.5.html#smtp_tcp_port">smtp_tcp_port</a> (smtp)</b>
value| Postfix.and:
length = 3;
}
prevLength
java.lang.StringIndexOutOfBoundsException: Index 25 out of bounds for length 25
} else {
ifprevLength=){
length=2;
} else {
/* change from single-byte mode to double-byte */
if (soLength == 1) {
value|=(uint32_t)soBytes[0]<<16;
length = 3;
} else if (a href="master.5.html">master(5)<
;
/></
length = 4;
}
prevLength=2;
}
}
break;
MBCS_OUTPUT_DBCS_ONLY
/* table with single-byte results, but only DBCS java.lang.StringIndexOutOfBoundsException: Index 72 out of bounds for length 18
value=MBCS_VALUE_2_FROM_STAGE_2(bytes, stage2Entry, c);
f(<=0ff {
/* no mapping or SBCS result, not taken for DBCS-only */
value=java.lang.StringIndexOutOfBoundsException: Index 39 out of bounds for length 25
length=0;
} else {
length=2java.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
}
break;
case MBCS_OUTPUT_3: p=MBCS_POINTER_3_FROM_STAGE_2(bytes, stage2Entry, c);
value=((uint32_t)*p<<16)|((uint32_t)p[1]<<8)|p[2];
if(value<=0xff) {
length=1;
} else if(value<=0xffff) {
length=2;
} else {
length=3;
}
break;
case MBCS_OUTPUT_4:
value=MBCS_VALUE_4_FROM_STAGE_2(bytes, stage2Entry, c);
if(value<=0xff) {
length=1;
} else if(value<=0xffff) {
length=2;
} else if(value<=0xffffff) {
length=3;
} else {
length=4;
}
break;
case MBCS_OUTPUT_3_EUC:
value=MBCS_VALUE_2_FROM_STAGE_2(bytes, stage2Entry, c);
/* EUC 16-bit fixed-length representation */
if(value<=0xff) {
length=1;
} else if((value&0x8000)==0) {
value|=0x8e8000;
length=3;
} else if((value&0x80)==0) {
value|=0x8f0080;
length=3;
} else {
length=2;
}
break;
case MBCS_OUTPUT_4_EUC: p=MBCS_POINTER_3_FROM_STAGE_2(bytes, stage2Entry, c);
value=((uint32_t)*p<<16)|((uint32_t)p[1]<<8)|p[2];
/* EUC 16-bit fixed-length representation applied to the first two bytes */
if(value<=0xff) {
length=1;
} else if(value<=0xffff) {
length=2;
} else if((value&0x800000)==0) {
value|=0x8e800000;
length=4;
} else if((value&0x8000)==0) {
value|=0x8f008000;
length=4;
} else {
length=3;
}
break;
default:
/* must not occur */
/*
* To avoid compiler warnings that value & length may be
* used without having been initialized, we set them here.
* In reality, this is unreachable code.
* Not having a default branch also causes warnings with
* some compilers.
*/
value=stage2Entry=0; /* stage2Entry=0 to reset roundtrip flags */
length=0;
break;
}
/* is this code point assigned, or do we use fallbacks? */
if(!(MBCS_FROM_U_IS_ROUNDTRIP(stage2Entry, c)!=0 ||
(UCNV_FROM_U_USE_FALLBACK(cnv, c) && value!=0))
) {
/*
* We allow a0 byte output if the "assigned" bit is set for this entry.
* There is no way with this data structure for fallback output
* to be a zero byte.
*/
if(U_FAILURE(*pErrorCode)) {
/* not mappable or buffer overflow */
break;
} else {
/* a mapping was written to the target, continue */
/* recalculate the targetCapacity after an extension mapping */
targetCapacity=(int32_t)(pArgs->targetLimit-(char *)target);
/* normal end of conversion: prepare for a new character */
if(offsets!=nullptr) {
prevSourceIndex=sourceIndex;
sourceIndex=nextSourceIndex;
}
continue;
}
}
}
/* write the output character bytes from value and length */
/* from the first if in the loop we know that targetCapacity>0 */
if(length<=targetCapacity) {
if(offsets==nullptr) {
switch(length) {
/* each branch falls through to the next one */
case 4:
*target++=(uint8_t)(value>>24);
U_FALLTHROUGH;
case 3:
*target++=(uint8_t)(value>>16);
U_FALLTHROUGH;
case 2:
*target++=(uint8_t)(value>>8);
U_FALLTHROUGH;
case 1:
*target++=(uint8_t)value;
U_FALLTHROUGH;
default:
/* will never occur */
break;
}
} else {
switch(length) {
/* each branch falls through to the next one */
case 4:
*target++=(uint8_t)(value>>24);
*offsets++=sourceIndex;
U_FALLTHROUGH;
case 3:
*target++=(uint8_t)(value>>16);
*offsets++=sourceIndex;
U_FALLTHROUGH;
case 2:
*target++=(uint8_t)(value>>8);
*offsets++=sourceIndex;
U_FALLTHROUGH;
case 1:
*target++=(uint8_t)value;
*offsets++=sourceIndex;
U_FALLTHROUGH;
default:
/* will never occur */
break;
}
}
targetCapacity-=length;
} else {
uint8_t *charErrorBuffer;
/*
* We actually do this backwards here:
* In order to save an intermediate variable, we output
* first to the overflow buffer what does not fit into the
* regular target.
*/
/* we know that 1<=targetCapacity<length<=4 */
length-=targetCapacity;
charErrorBuffer=(uint8_t *)cnv->charErrorBuffer;
switch(length) {
/* each branch falls through to the next one */
case 3:
*charErrorBuffer++=(uint8_t)(value>>16);
U_FALLTHROUGH;
case 2:
*charErrorBuffer++=(uint8_t)(value>>8);
U_FALLTHROUGH;
case 1:
*charErrorBuffer=(uint8_t)value;
U_FALLTHROUGH;
default:
/* will never occur */
break;
}
cnv->charErrorBufferLength=(int8_t)length;
/* now output what fits into the regular target */
value>>=8*length; /* length was reduced by targetCapacity */
switch(targetCapacity) {
/* each branch falls through to the next one */
case 3:
*target++=(uint8_t)(value>>16);
if(offsets!=nullptr) {
*offsets++=sourceIndex;
}
U_FALLTHROUGH;
case 2:
*target++=(uint8_t)(value>>8);
if(offsets!=nullptr) {
*offsets++=sourceIndex;
}
U_FALLTHROUGH;
case 1:
*target++=(uint8_t)value;
if(offsets!=nullptr) {
*offsets++=sourceIndex;
}
U_FALLTHROUGH;
default:
/* will never occur */
break;
}
/* normal end of conversion: prepare for a new character */
c=0;
if(offsets!=nullptr) {
prevSourceIndex=sourceIndex;
sourceIndex=nextSourceIndex;
}
continue;
} else {
/* target is full */
*pErrorCode=U_BUFFER_OVERFLOW_ERROR;
break;
}
}
/*
* the end of the input stream and detection of truncated input
* are handled by the framework, but for EBCDIC_STATEFUL conversion
* we need to emit an SI at the very end
*
* conditions:
* successful
* EBCDIC_STATEFUL in DBCS mode
* end of input and no truncated input
*/
if( U_SUCCESS(*pErrorCode) &&
outputType==MBCS_OUTPUT_2_SISO && prevLength==2 &&
pArgs->flush && source>=sourceLimit && c==0
) {
/* EBCDIC_STATEFUL ending with DBCS: emit an SI to return the output stream to SBCS */
if(targetCapacity>0) {
*target++ = siBytes[0];
if (siLength == 2) {
if (targetCapacity<2) {
cnv->charErrorBuffer[0] = siBytes[1];
cnv->charErrorBufferLength=1;
*pErrorCode=U_BUFFER_OVERFLOW_ERROR;
} else {
*target++ = siBytes[1];
}
}
if(offsets!=nullptr) {
/* set the last source character's index (sourceIndex points at sourceLimit now) */
*offsets++=prevSourceIndex;
}
} else {
/* target is full */
cnv->charErrorBuffer[0] = siBytes[0];
if (siLength == 2) {
cnv->charErrorBuffer[1] = siBytes[1];
}
cnv->charErrorBufferLength=siLength;
*pErrorCode=U_BUFFER_OVERFLOW_ERROR;
}
prevLength=1; /* we switched into SBCS */
}
/* set the converter state back into UConverter */
cnv->fromUChar32=c;
cnv->fromUnicodeStatus=prevLength;
/* write back the updated pointers */
pArgs->source=source;
pArgs->target=(char *)target;
pArgs->offsets=offsets;
}
/*
* This is another simple conversion function for internal use by other
* conversion implementations.
* It does not use the converter state nor call callbacks.
* It does not handle the EBCDIC swaplfnl option (set in UConverter).
* It handles conversion extensions but not GB 18030.
*
* It converts one single Unicode code point into codepage bytes, encoded
* as one 32-bit value. The function returns the number of bytes in *pValue:
* 1..4 the number of bytes in *pValue
* 0 unassigned (*pValue undefined)
* -1 illegal (currently not used, *pValue undefined)
*
* *pValue will contain the resulting bytes with the last byte in bits 7..0,
* the second to last byte in bits 15..8, etc.
* Currently, the function assumes but does not check that 0<=c<=0x10ffff.
*/
U_CFUNC int32_t
ucnv_MBCSFromUChar32(UConverterSharedData *sharedData,
UChar32 c, uint32_t *pValue,
UBool useFallback) {
const int32_t *cx;
const uint16_t *table;
#if 0
/* #if 0 because this is not currently used in ICU - reduce code, increase code coverage */
const uint8_t *p;
#endif
uint32_t stage2Entry;
uint32_t value;
int32_t length;
/* BMP-only codepages are stored without stage 1 entries for supplementary code points */
if(c<=0xffff || (sharedData->mbcs.unicodeMask&UCNV_HAS_SUPPLEMENTARY)) { table=sharedData->mbcs.fromUnicodeTable;
/* convert the Unicode code point in c into codepage bytes (same as in _MBCSFromUnicodeWithOffsets) */
if(sharedData->mbcs.outputType==MBCS_OUTPUT_1) {
value=MBCS_SINGLE_RESULT_FROM_U(table, (uint16_t *)sharedData->mbcs.fromUnicodeBytes, c);
/* is this code point assigned, or do we use fallbacks? */
if(useFallback ? value>=0x800 : value>=0xc00) {
*pValue=value&0xff;
return 1;
}
} else /* outputType!=MBCS_OUTPUT_1 */ {
stage2Entry=MBCS_STAGE_2_FROM_U(table, c);
/* get the bytes and the length for the output */
switch(sharedData->mbcs.outputType) {
case MBCS_OUTPUT_2:
value=MBCS_VALUE_2_FROM_STAGE_2(sharedData->mbcs.fromUnicodeBytes, stage2Entry, c);
if(value<=0xff) {
length=1;
} else {
length=2;
}
break;
#if 0
/* #if 0 because this is not currently used in ICU - reduce code, increase code coverage */
case MBCS_OUTPUT_DBCS_ONLY:
/* table with single-byte results, but only DBCS mappings used */
value=MBCS_VALUE_2_FROM_STAGE_2(sharedData->mbcs.fromUnicodeBytes, stage2Entry, c);
if(value<=0xff) {
/* no mapping or SBCS result, not taken for DBCS-only */
value=stage2Entry=0; /* stage2Entry=0 to reset roundtrip flags */
length=0;
} else {
length=2;
}
break;
case MBCS_OUTPUT_3: p=MBCS_POINTER_3_FROM_STAGE_2(sharedData->mbcs.fromUnicodeBytes, stage2Entry, c);
value=((uint32_t)*p<<16)|((uint32_t)p[1]<<8)|p[2];
if(value<=0xff) {
length=1;
} else if(value<=0xffff) {
length=2;
} else {
length=3;
}
break;
case MBCS_OUTPUT_4:
value=MBCS_VALUE_4_FROM_STAGE_2(sharedData->mbcs.fromUnicodeBytes, stage2Entry, c);
if(value<=0xff) {
length=1;
} else if(value<=0xffff) {
length=2;
} else if(value<=0xffffff) {
length=3;
} else {
length=4;
}
break;
case MBCS_OUTPUT_3_EUC:
value=MBCS_VALUE_2_FROM_STAGE_2(sharedData->mbcs.fromUnicodeBytes, stage2Entry, c);
/* EUC 16-bit fixed-length representation */
if(value<=0xff) {
length=1;
} else if((value&0x8000)==0) {
value|=0x8e8000;
length=3;
} else if((value&0x80)==0) {
value|=0x8f0080;
length=3;
} else {
length=2;
}
break;
case MBCS_OUTPUT_4_EUC: p=MBCS_POINTER_3_FROM_STAGE_2(sharedData->mbcs.fromUnicodeBytes, stage2Entry, c);
value=((uint32_t)*p<<16)|((uint32_t)p[1]<<8)|p[2];
/* EUC 16-bit fixed-length representation applied to the first two bytes */
if(value<=0xff) {
length=1;
} else if(value<=0xffff) {
length=2;
} else if((value&0x800000)==0) {
value|=0x8e800000;
length=4;
} else if((value&0x8000)==0) {
value|=0x8f008000;
length=4;
} else {
length=3;
}
break;
#endif
default:
/* must not occur */
return -1;
}
/* is this code point assigned, or do we use fallbacks? */
if( MBCS_FROM_U_IS_ROUNDTRIP(stage2Entry, c) ||
(FROM_U_USE_FALLBACK(useFallback, c) && value!=0)
) {
/*
* We allow a0 byte output if the "assigned" bit is set for this entry.
* There is no way with this data structure for fallback output
* to be a zero byte.
*/
/* assigned */
*pValue=value;
return length;
}
}
}
#if 0
/*
* This function has been moved to ucnv2022.c for inlining.
* This implementation is here only for documentation purposes
*/
/**
* This version of ucnv_MBCSFromUChar32() is optimized for single-byte codepages.
* It does not handle the EBCDIC swaplfnl option (set in UConverter).
* It does not handle conversion extensions (_extFromU()).
*
* It returns the codepage byte for the code point, or -1 if it is unassigned.
*/
U_CFUNC int32_t
ucnv_MBCSSingleFromUChar32(UConverterSharedData *sharedData,
UChar32 c,
UBool useFallback) {
const uint16_t *table;
int32_t value;
/* BMP-only codepages are stored without stage 1 entries for supplementary code points */
if(c>=0x10000 && !(sharedData->mbcs.unicodeMask&UCNV_HAS_SUPPLEMENTARY)) {
return -1;
}
/* convert the Unicode code point in c into codepage bytes (same as in _MBCSFromUnicodeWithOffsets) */ table=sharedData->mbcs.fromUnicodeTable;
/* get the byte for the output */
value=MBCS_SINGLE_RESULT_FROM_U(table, (uint16_t *)sharedData->mbcs.fromUnicodeBytes, c);
/* is this code point assigned, or do we use fallbacks? */
if(useFallback ? value>=0x800 : value>=0xc00) {
return value&0xff;
} else {
return -1;
}
}
#endif
if(cnv->useFallback) {
/* use all roundtrip and fallback results */
minValue=0x800;
} else {
/* use only roundtrips and fallbacks from private-use characters */
minValue=0xc00;
}
hasSupplementary = static_cast<UBool>(cnv->sharedData->mbcs.unicodeMask & UCNV_HAS_SUPPLEMENTARY);
/* get the converter state from the UTF-8 UConverter */
if(utf8->toULength > 0) {
toULength=oldToULength=utf8->toULength;
toULimit = static_cast<int8_t>(utf8->mode);
c = static_cast<UChar32>(utf8->toUnicodeStatus);
} else {
toULength=oldToULength=toULimit=0;
c = 0;
}
// The conversion loop checks source<sourceLimit only once per 1/2/3-byte character.
// If the buffer ends with a truncated 2- or 3-byte sequence,
// then we reduce the sourceLimit to before that,
// and collect the remaining bytes after the conversion loop.
{
// Do not go back into the bytes that will be read for finishing a partial
// sequence from the previous buffer.
int32_t length = static_cast<int32_t>(sourceLimit - source) - (toULimit - oldToULength);
if(length>0) {
uint8_t b1=*(sourceLimit-1);
if(U8_IS_SINGLE(b1)) {
// common ASCII character
} else if(U8_IS_TRAIL(b1) && length>=2) {
uint8_t b2=*(sourceLimit-2);
if(0xe0<=b2 && b2<0xf0 && U8_IS_VALID_LEAD3_AND_T1(b2, b1)) {
// truncated 3-byte sequence
sourceLimit-=2;
}
} else if(0xc2<=b1 && b1<0xf0) {
// truncated 2- or 3-byte sequence
--sourceLimit;
}
}
}
if(c!=0 && targetCapacity>0) {
utf8->toUnicodeStatus=0;
utf8->toULength=0;
goto moreBytes;
/*
* Note: We could avoid the goto by duplicating some of the moreBytes
* code, but only up to the point of collecting a complete UTF-8
* sequence; then recurse for the toUBytes[toULength]
* and then continue with normal conversion.
*
* If so, move this code to just after initializing the minimum
* set of local variables for reading the UTF-8input
* (utf8, source, target, limits but not cnv, table, minValue, etc.).
*
* Potential advantages:
* - avoid the goto
* - oldToULength could become a local variable in just those code blocks
* that deal with buffer boundaries
* - possibly faster if the goto prevents some compiler optimizations
* (this would need measuring to confirm)
* Disadvantage:
* - code duplication
*/
}
if(c<0) {
/* handle "complicated" and error cases, and continuing partial characters */
oldToULength=0;
toULength=1;
toULimit=U8_COUNT_BYTES_NON_ASCII(b);
c=b;
moreBytes:
while(toULength<toULimit) {
/*
* The sourceLimit may have been adjusted before the conversion loop
* to stop before a truncated sequence.
* Here we need to use the real limit in case we have two truncated
* sequences at the end.
* See ticket #7492.
*/
if(source<(uint8_t *)pToUArgs->sourceLimit) { b=*source;
if(icu::UTF8::isValidTrail(c, b, toULength, toULimit)) {
++source;
++toULength;
c=(c<<6)+b;
} else {
break; /* sequence too short, stop with toULength<toULimit */
}
} else {
/* store the partial UTF-8 character, compatible with the regular UTF-8 converter */ source-=(toULength-oldToULength);
while(oldToULength<toULength) {
utf8->toUBytes[oldToULength++]=*source++;
}
utf8->toUnicodeStatus=c;
utf8->toULength=toULength;
utf8->mode=toULimit;
pToUArgs->source=(char *)source;
pFromUArgs->target = reinterpret_cast<char*>(target);
return;
}
}
if(value>=minValue) {
/* output the mapping for c */
*target++ = static_cast<uint8_t>(value);
--targetCapacity;
} else {
/* value<minValue means c is unassigned (unmappable) */
/*
* Try an extension mapping.
* Pass in no source because we don't have UTF-16 input.
* If we have a partial match on c, we will return and revert
* to UTF-8->UTF-16->charset conversion.
*/
static const char16_t nul=0;
const char16_t *noSource=&nul;
c=_extFromU(cnv, cnv->sharedData,
c, &noSource, noSource,
&target, target+targetCapacity,
nullptr, -1,
pFromUArgs->flush,
pErrorCode);
if(U_FAILURE(*pErrorCode)) {
/* not mappable or buffer overflow */
cnv->fromUChar32=c;
break;
} else if(cnv->preFromUFirstCP>=0) {
/*
* Partial match, return and revert to pivoting.
* In normal from-UTF-16 conversion, we would just continue
* but then exit the loop because the extension match would
* have consumed the source.
*/
*pErrorCode=U_USING_DEFAULT_WARNING;
break;
} else {
/* a mapping was written to the target, continue */
/* recalculate the targetCapacity after an extension mapping */
targetCapacity = static_cast<int32_t>(pFromUArgs->targetLimit - reinterpret_cast<char*>(target));
}
}
} else {
/* target is full */
*pErrorCode=U_BUFFER_OVERFLOW_ERROR;
break;
}
}
/*
* The sourceLimit may have been adjusted before the conversion loop
* to stop before a truncated sequence.
* If so, then collect the truncated sequence now.
*/
if(U_SUCCESS(*pErrorCode) &&
cnv->preFromUFirstCP<0 && source<(sourceLimit=(uint8_t *)pToUArgs->sourceLimit)) {
c=utf8->toUBytes[0]=b=*source++;
toULength=1;
toULimit=U8_COUNT_BYTES(b);
while(source<sourceLimit) {
utf8->toUBytes[toULength++]=b=*source++;
c=(c<<6)+b;
}
utf8->toUnicodeStatus=c;
utf8->toULength=toULength;
utf8->mode=toULimit;
}
/* write back the updated pointers */
pToUArgs->source=(char *)source;
pFromUArgs->target = reinterpret_cast<char*>(target);
}
/* get the converter state from the UTF-8 UConverter */
if(utf8->toULength > 0) {
toULength=oldToULength=utf8->toULength;
toULimit = static_cast<int8_t>(utf8->mode);
c = static_cast<UChar32>(utf8->toUnicodeStatus);
} else {
toULength=oldToULength=toULimit=0;
c = 0;
}
// The conversion loop checks source<sourceLimit only once per 1/2/3-byte character.
// If the buffer ends with a truncated 2- or 3-byte sequence,
// then we reduce the sourceLimit to before that,
// and collect the remaining bytes after the conversion loop.
{
// Do not go back into the bytes that will be read for finishing a partial
// sequence from the previous buffer.
int32_t length = static_cast<int32_t>(sourceLimit - source) - (toULimit - oldToULength);
if(length>0) {
uint8_t b1=*(sourceLimit-1);
if(U8_IS_SINGLE(b1)) {
// common ASCII character
} else if(U8_IS_TRAIL(b1) && length>=2) {
uint8_t b2=*(sourceLimit-2);
if(0xe0<=b2 && b2<0xf0 && U8_IS_VALID_LEAD3_AND_T1(b2, b1)) {
// truncated 3-byte sequence
sourceLimit-=2;
}
} else if(0xc2<=b1 && b1<0xf0) {
// truncated 2- or 3-byte sequence
--sourceLimit;
}
}
}
if(c!=0 && targetCapacity>0) {
utf8->toUnicodeStatus=0;
utf8->toULength=0;
goto moreBytes;
/* See note in ucnv_SBCSFromUTF8() about this goto. */
}
if(c<0) {
/* handle "complicated" and error cases, and continuing partial characters */
oldToULength=0;
toULength=1;
toULimit=U8_COUNT_BYTES_NON_ASCII(b);
c=b;
moreBytes:
while(toULength<toULimit) {
/*
* The sourceLimit may have been adjusted before the conversion loop
* to stop before a truncated sequence.
* Here we need to use the real limit in case we have two truncated
* sequences at the end.
* See ticket #7492.
*/
if(source<(uint8_t *)pToUArgs->sourceLimit) { b=*source;
if(icu::UTF8::isValidTrail(c, b, toULength, toULimit)) {
++source;
++toULength;
c=(c<<6)+b;
} else {
break; /* sequence too short, stop with toULength<toULimit */
}
} else {
/* store the partial UTF-8 character, compatible with the regular UTF-8 converter */ source-=(toULength-oldToULength);
while(oldToULength<toULength) {
utf8->toUBytes[oldToULength++]=*source++;
}
utf8->toUnicodeStatus=c;
utf8->toULength=toULength;
utf8->mode=toULimit;
pToUArgs->source=(char *)source;
pFromUArgs->target = reinterpret_cast<char*>(target);
return;
}
}
/* get the bytes and the length for the output */
/* MBCS_OUTPUT_2 */
value=MBCS_VALUE_2_FROM_STAGE_2(results, stage2Entry, c);
/* is this code point assigned, or do we use fallbacks? */
if(!(MBCS_FROM_U_IS_ROUNDTRIP(stage2Entry, c) ||
(UCNV_FROM_U_USE_FALLBACK(cnv, c) && value!=0))
) {
goto unassigned;
}
}
}
/* write the output character bytes from value and length */
/* from the first if in the loop we know that targetCapacity>0 */
if(value<=0xff) {
/* this is easy because we know that there is enough space */
*target++ = static_cast<uint8_t>(value);
--targetCapacity;
} else /* length==2 */ {
*target++ = static_cast<uint8_t>(value >> 8);
if(2<=targetCapacity) {
*target++ = static_cast<uint8_t>(value);
targetCapacity-=2;
} else {
cnv->charErrorBuffer[0] = static_cast<char>(value);
cnv->charErrorBufferLength=1;
unassigned:
{
/*
* Try an extension mapping.
* Pass in no source because we don't have UTF-16 input.
* If we have a partial match on c, we will return and revert
* to UTF-8->UTF-16->charset conversion.
*/
static const char16_t nul=0;
const char16_t *noSource=&nul;
c=_extFromU(cnv, cnv->sharedData,
c, &noSource, noSource,
&target, target+targetCapacity,
nullptr, -1,
pFromUArgs->flush,
pErrorCode);
if(U_FAILURE(*pErrorCode)) {
/* not mappable or buffer overflow */
cnv->fromUChar32=c;
break;
} else if(cnv->preFromUFirstCP>=0) {
/*
* Partial match, return and revert to pivoting.
* In normal from-UTF-16 conversion, we would just continue
* but then exit the loop because the extension match would
* have consumed the source.
*/
*pErrorCode=U_USING_DEFAULT_WARNING;
break;
} else {
/* a mapping was written to the target, continue */
/* recalculate the targetCapacity after an extension mapping */
targetCapacity = static_cast<int32_t>(pFromUArgs->targetLimit - reinterpret_cast<char*>(target));
continue;
}
}
} else {
/* target is full */
*pErrorCode=U_BUFFER_OVERFLOW_ERROR;
break;
}
}
/*
* The sourceLimit may have been adjusted before the conversion loop
* to stop before a truncated sequence.
* If so, then collect the truncated sequence now.
*/
if(U_SUCCESS(*pErrorCode) &&
cnv->preFromUFirstCP<0 && source<(sourceLimit=(uint8_t *)pToUArgs->sourceLimit)) {
c=utf8->toUBytes[0]=b=*source++;
toULength=1;
toULimit=U8_COUNT_BYTES(b);
while(source<sourceLimit) {
utf8->toUBytes[toULength++]=b=*source++;
c=(c<<6)+b;
}
utf8->toUnicodeStatus=c;
utf8->toULength=toULength;
utf8->mode=toULimit;
}
/* write back the updated pointers */
pToUArgs->source=(char *)source;
pFromUArgs->target = reinterpret_cast<char*>(target);
}
state0=cnv->sharedData->mbcs.stateTable[cnv->sharedData->mbcs.dbcsOnlyState];
for(i=0; i<256; ++i) {
/* all bytes that cause a state transition from state 0 are lead bytes */
starters[i] = static_cast<UBool>(MBCS_ENTRY_IS_TRANSITION(state0[i]));
}
}
/*
* This is an internal function that allows other converter implementations
* to check whether a byte is a lead byte.
*/
U_CFUNC UBool
ucnv_MBCSIsLeadByte(UConverterSharedData *sharedData, char byte) {
return MBCS_ENTRY_IS_TRANSITION(sharedData->mbcs.stateTable[0][(uint8_t)byte]);
}
/* first, select between subChar and subChar1 */
if( cnv->subChar1!=0 &&
(cnv->sharedData->mbcs.extIndexes!=nullptr ?
cnv->useSubChar1 :
(cnv->invalidUCharBuffer[0]<=0xff))
) {
/* select subChar1 if it is set (not 0) and the unmappable Unicode code point is up to U+00ff (IBM MBCS behavior) */
subchar = reinterpret_cast<char*>(&cnv->subChar1);
length=1;
} else {
/* select subChar in all other cases */
subchar = reinterpret_cast<char*>(cnv->subChars);
length=cnv->subCharLen;
}
/* reset the selector for the next code point */
cnv->useSubChar1=false;
if (cnv->sharedData->mbcs.outputType == MBCS_OUTPUT_2_SISO) { p=buffer;
/* fromUnicodeStatus contains prevLength */
switch(length) {
case 1:
if(cnv->fromUnicodeStatus==2) {
/* DBCS mode and SBCS sub char: change to SBCS */
cnv->fromUnicodeStatus=1;
*p++=UCNV_SI;
}
*p++=subchar[0];
break;
case 2:
if(cnv->fromUnicodeStatus<=1) {
/* SBCS mode and DBCS sub char: change to DBCS */
cnv->fromUnicodeStatus=2;
*p++=UCNV_SO;
}
*p++=subchar[0];
*p++=subchar[1];
break;
default:
*pErrorCode=U_ILLEGAL_ARGUMENT_ERROR;
return;
}
subchar=buffer;
length = static_cast<int32_t>(p - buffer);
}
Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.
Bemerkung:
Die farbliche Syntaxdarstellung und die Messung sind noch experimentell.