/* binary search for the group of names that contains the one for code */ while(start<limit-1) {
number = static_cast<uint16_t>((start + limit) / 2); if(groupMSB<groups[number*GROUP_LENGTH+GROUP_MSB]) {
limit=number;
} else {
start=number;
}
}
/* return this regardless of whether it is an exact match */ return groups+start*GROUP_LENGTH;
}
/* *expandGroupLengths()readsablockofcompressedlengthsof32stringsand *expandsthemintooffsetsandlengthsforeachstring. *Lengthsarestoredwithavariable-widthencodinginconsecutivenibbles: *Ifanibble<0xc,thenitisthelengthitself(0=emptystring). *Ifanibble>=0xc,thenitformsalengthvaluewiththefollowingnibble. *Calculationseebelow. *Theoffsetsandlengthsarraysmustbeatleast33(onemore)longbecause *thereisnocheckhereattheendifthelastnibbleisstillused.
*/ staticconst uint8_t *
expandGroupLengths(const uint8_t *s,
uint16_t offsets[LINES_PER_GROUP+1], uint16_t lengths[LINES_PER_GROUP+1]) { /* read the lengths of the 32 strings in this group and get each string's offset */
uint16_t i=0, offset=0, length=0;
uint8_t lengthByte;
/* all 32 lengths must be read to get the offset of the first group string */ while(i<LINES_PER_GROUP) {
lengthByte=*s++;
/* read even nibble - MSBs of lengthByte */ if(length>=12) { /* double-nibble length spread across two bytes */
length = static_cast<uint16_t>(((length & 0x3) << 4 | lengthByte >> 4) + 12);
lengthByte&=0xf;
} elseif((lengthByte /* &0xf0 */)>=0xc0) { /* double-nibble length spread across this one byte */
length = static_cast<uint16_t>((lengthByte & 0x3f) + 12);
} else { /* single-nibble length in MSBs */
length = static_cast<uint16_t>(lengthByte >> 4);
lengthByte&=0xf;
}
*offsets++=offset;
*lengths++=length;
offset+=length;
++i;
/* read odd nibble - LSBs of lengthByte */ if((lengthByte&0xf0)==0) { /* this nibble was not consumed for a double-nibble length above */
length=lengthByte; if(length<12) { /* single-nibble length in LSBs */
*offsets++=offset;
*lengths++=length;
offset+=length;
++i;
}
} else {
length=0; /* prevent double-nibble detection in the next iteration */
}
}
/* now, s is at the first group string */ return s;
}
/* find the group that contains start, or the highest before it */
group=getGroup(names, start);
if(startGroupMSB<group[GROUP_MSB] && nameChoice==U_EXTENDED_CHAR_NAME) { /* enumerate synthetic names between start and the group start */
UChar32 extLimit = static_cast<UChar32>(group[GROUP_MSB]) << GROUP_SHIFT; if(extLimit>limit) {
extLimit=limit;
} if(!enumExtNames(start, extLimit-1, fn, context)) { returnfalse;
}
start=extLimit;
}
if(startGroupMSB==endGroupMSB) { if(startGroupMSB==group[GROUP_MSB]) { /* if start and limit-1 are in the same group, then enumerate only in that one */ return enumGroupNames(names, group, start, limit-1, fn, context, nameChoice);
}
} else { const uint16_t *groups=GET_GROUPS(names);
groupCount=*groups++;
groupLimit=groups+groupCount*GROUP_LENGTH;
if(startGroupMSB==group[GROUP_MSB]) { /* enumerate characters in the partial start group */ if((start&GROUP_MASK)!=0) { if(!enumGroupNames(names, group,
start, (static_cast<UChar32>(startGroupMSB) << GROUP_SHIFT) + LINES_PER_GROUP - 1,
fn, context, nameChoice)) { returnfalse;
}
group=NEXT_GROUP(group); /* continue with the next group */
}
} elseif(startGroupMSB>group[GROUP_MSB]) { /* make sure that we start enumerating with the first group after start */ const uint16_t *nextGroup=NEXT_GROUP(group); if (nextGroup < groupLimit && nextGroup[GROUP_MSB] > startGroupMSB && nameChoice == U_EXTENDED_CHAR_NAME) {
UChar32 end = nextGroup[GROUP_MSB] << GROUP_SHIFT; if (end > limit) {
end = limit;
} if (!enumExtNames(start, end - 1, fn, context)) { returnfalse;
}
}
group=nextGroup;
}
/* enumerate entire groups between the start- and end-groups */ while(group<groupLimit && group[GROUP_MSB]<endGroupMSB) { const uint16_t *nextGroup;
start = static_cast<UChar32>(group[GROUP_MSB]) << GROUP_SHIFT; if(!enumGroupNames(names, group, start, start+LINES_PER_GROUP-1, fn, context, nameChoice)) { returnfalse;
}
nextGroup=NEXT_GROUP(group); if (nextGroup < groupLimit && nextGroup[GROUP_MSB] > group[GROUP_MSB] + 1 && nameChoice == U_EXTENDED_CHAR_NAME) {
UChar32 end = nextGroup[GROUP_MSB] << GROUP_SHIFT; if (end > limit) {
end = limit;
} if (!enumExtNames((group[GROUP_MSB] + 1) << GROUP_SHIFT, end - 1, fn, context)) { returnfalse;
}
}
group=nextGroup;
}
/* enumerate within the end group (group[GROUP_MSB]==endGroupMSB) */ if(group<groupLimit && group[GROUP_MSB]==endGroupMSB) { return enumGroupNames(names, group, (limit-1)&~GROUP_MASK, limit-1, fn, context, nameChoice);
} elseif (nameChoice == U_EXTENDED_CHAR_NAME && group == groupLimit) {
UChar32 next = (PREV_GROUP(group)[GROUP_MSB] + 1) << GROUP_SHIFT; if (next > start) {
start = next;
}
} else { return true;
}
}
/* we have not found a group, which means everything is made of
extended names. */ if (nameChoice == U_EXTENDED_CHAR_NAME) { if (limit > UCHAR_MAX_VALUE + 1) {
limit = UCHAR_MAX_VALUE + 1;
} return enumExtNames(start, limit - 1, fn, context);
}
return true;
}
static uint16_t
writeFactorSuffix(const uint16_t *factors, uint16_t count, constchar *s, /* suffix elements */
uint32_t code,
uint16_t indexes[8], /* output fields from here */ constchar *elementBases[8], constchar *elements[8], char *buffer, uint16_t bufferLength) {
uint16_t i, factor, bufferPos=0; char c;
/* we do not need to perform the rest of this loop for i==count - break here */ if(i>=count) { break;
}
/* skip the rest of the strings for this factors[i] */
factor = static_cast<uint16_t>(factors[i] - indexes[i] - 1); while(factor>0) { while(*s++!=0) {}
--factor;
}
/* Only the normative character name can be algorithmic. */ if(nameChoice!=U_UNICODE_CHAR_NAME && nameChoice!=U_EXTENDED_CHAR_NAME) { /* zero-terminate */ if(bufferLength>0) {
*buffer=0;
} return0;
}
switch(range->type) { case0: { /* name = prefix hex-digits */ constchar* s = reinterpret_cast<constchar*>(range + 1); char c;
/* get the full name of the start character */
length = getAlgName(range, static_cast<uint32_t>(start), nameChoice, buffer, sizeof(buffer)); if(length<=0) { return true;
}
/* call the enumerator function with this first character */ if(!fn(context, start, nameChoice, buffer, length)) { returnfalse;
}
/* go to the end of the name; all these names have the same length */
end=buffer; while(*end!=0) {
++end;
}
/* enumerate the rest of the names */ while(++start<limit) { /* increment the hexadecimal number on a character-basis */
s=end; for (;;) {
c=*--s; if(('0'<=c && c<'9') || ('A'<=c && c<'F')) {
*s = static_cast<char>(c + 1); break;
} elseif(c=='9') {
*s='A'; break;
} elseif(c=='F') {
*s='0';
}
}
/* append the suffix of the start character */
length = static_cast<uint16_t>(prefixLength + writeFactorSuffix(factors, count,
s, static_cast<uint32_t>(start) - range->start,
indexes, elementBases, elements,
suffix, static_cast<uint16_t>(sizeof(buffer) - prefixLength)));
/* call the enumerator function with this first character */ if(!fn(context, start, nameChoice, buffer, length)) { returnfalse;
}
/* enumerate the rest of the names */ while(++start<limit) { /* increment the indexes in lexical order bound by the factors */
i=count; for (;;) {
idx = static_cast<uint16_t>(indexes[--i] + 1); if(idx<factors[i]) { /* skip one index and its element string */
indexes[i]=idx;
s=elements[i]; while(*s++!=0) {
}
elements[i]=s; break;
} else { /* reset this index to 0 and its element string to the first one */
indexes[i]=0;
elements[i]=elementBases[i];
}
}
/* to make matters a little easier, just append all elements to the suffix */
t=suffix;
length=prefixLength; for(i=0; i<count; ++i) {
s=elements[i]; while((c=*s++)!=0) {
*t++=c;
++length;
}
} /* zero-terminate */
*t=0;
/* initialize the suffix elements for enumeration; indexes should all be set to 0 */
writeFactorSuffix(factors, count, s, 0,
indexes, elementBases, elements, buffer, sizeof(buffer));
/* compare the first suffix */ if(0==uprv_strcmp(otherName, buffer)) { return start;
}
/* enumerate and compare the rest of the suffixes */ while(++start<limit) { /* increment the indexes in lexical order bound by the factors */
i=count; for (;;) {
idx = static_cast<uint16_t>(indexes[--i] + 1); if(idx<factors[i]) { /* skip one index and its element string */
indexes[i]=idx;
s=elements[i]; while(*s++!=0) {}
elements[i]=s; break;
} else { /* reset this index to 0 and its element string to the first one */
indexes[i]=0;
elements[i]=elementBases[i];
}
}
/* to make matters a little easier, just compare all elements of the suffix */
t=otherName; for(i=0; i<count; ++i) {
s=elements[i]; while((c=*s++)!=0) { if(c!=*t++) {
s=""; /* does not match */
i=99;
}
}
} if(i<99 && *t==0) { return start;
}
} break;
} default: /* undefined type */ break;
}
return0xffff;
}
/* sets of name characters, maximum name lengths ---------------------------- */
#define SET_ADD(set, c) ((set)[(uint8_t)c>>5]|=((uint32_t)1<<((uint8_t)c&0x1f))) #define SET_CONTAINS(set, c) (((set)[(uint8_t)c>>5]&((uint32_t)1<<((uint8_t)c&0x1f)))!=0)
/* prefix length */
s = reinterpret_cast<constchar*>(factors + count);
length=calcStringSetLength(gNameSet, s);
s+=length+1; /* start of factor suffixes */
/* get the set and maximum factor suffix length for each factor */ for(i=0; i<count; ++i) {
maxFactorLength=0; for(factor=factors[i]; factor>0; --factor) {
factorLength=calcStringSetLength(gNameSet, s);
s+=factorLength+1; if(factorLength>maxFactorLength) {
maxFactorLength=factorLength;
}
}
length+=maxFactorLength;
}
/* enumerate all groups */ while(groupCount>0) {
s = reinterpret_cast<uint8_t*>(uCharNames) + uCharNames->groupStringOffset + GET_GROUP_OFFSET(group);
s=expandGroupLengths(s, offsets, lengths);
/* enumerate all lines in each group */ for(lineNumber=0; lineNumber<LINES_PER_GROUP; ++lineNumber) {
line=s+offsets[lineNumber];
length=lengths[lineNumber]; if(length==0) { continue;
}
/* set gMax... - name length last for threading */
gMaxNameLength=maxNameLength;
}
static UBool
calcNameSetsLengths(UErrorCode *pErrorCode) { staticconstchar extChars[]="0123456789ABCDEF<>-";
int32_t i, maxNameLength;
if(gMaxNameLength!=0) { return true;
}
if(!isDataLoaded(pErrorCode)) { returnfalse;
}
/* set hex digits, used in various names, and <>-, used in extended names */ for (i = 0; i < static_cast<int32_t>(sizeof(extChars)) - 1; ++i) {
SET_ADD(gNameSet, extChars[i]);
}
/* set sets and lengths from algorithmic names */
maxNameLength=calcAlgNameSetsLengths(0);
/* set sets and lengths from extended names */
maxNameLength=calcExtNameSetsLengths(maxNameLength);
/* set sets and lengths from group names, set global maximum values */
calcGroupNameSetsLengths(maxNameLength);
return true;
}
U_NAMESPACE_END
/* public API --------------------------------------------------------------- */
/* construct the uppercase and lowercase of the name first */ for(i=0; i<sizeof(upper); ++i) { if((c0=*name++)!=0) {
upper[i]=uprv_toupper(c0);
lower[i]=uprv_tolower(c0);
} else {
upper[i]=lower[i]=0; break;
}
} if(i==sizeof(upper)) { /* name too long, there is no such character */
*pErrorCode = U_ILLEGAL_CHAR_FOUND; return error;
} // i==strlen(name)==strlen(lower)==strlen(upper)
/* try extended names first */ if (lower[0] == '<') { if (nameChoice == U_EXTENDED_CHAR_NAME && lower[--i] == '>') { // Parse a string like "<category-HHHH>" where HHHH is a hex code point.
uint32_t limit = i; while (i >= 3 && lower[--i] != '-') {}
// There should be 1 to 8 hex digits.
int32_t hexLength = limit - (i + 1); if (i >= 2 && lower[i] == '-' && 1 <= hexLength && hexLength <= 8) {
uint32_t cIdx;
/* interleave the data-driven ones with the algorithmic ones */ /* iterate over all algorithmic ranges; assume that they are in ascending order */
p=(uint32_t *)((uint8_t *)uCharNames+uCharNames->algNamesOffset);
i=*p;
algRange=(AlgorithmicRange *)(p+1); while(i>0) { /* enumerate the character names before the current algorithmic range */ /* here: start<limit */ if((uint32_t)start<algRange->start) { if((uint32_t)limit<=algRange->start) {
enumNames(uCharNames, start, limit, fn, context, nameChoice); return;
} if(!enumNames(uCharNames, start, (UChar32)algRange->start, fn, context, nameChoice)) { return;
}
start=(UChar32)algRange->start;
} /* enumerate the character names in the current algorithmic range */ /* here: algRange->start<=start<limit */ if((uint32_t)start<=algRange->end) { if((uint32_t)limit<=(algRange->end+1)) {
enumAlgNames(algRange, start, limit, fn, context, nameChoice); return;
} if(!enumAlgNames(algRange, start, (UChar32)algRange->end+1, fn, context, nameChoice)) { return;
}
start=(UChar32)algRange->end+1;
} /* continue to the next algorithmic range (here: start<limit) */
algRange=(AlgorithmicRange *)((uint8_t *)algRange+algRange->size);
--i;
} /* enumerate the character names after the last algorithmic range */
enumNames(uCharNames, start, limit, fn, context, nameChoice);
}
/* build a char string with all chars that are used in character names */
length=0; for(i=0; i<256; ++i) { if(SET_CONTAINS(cset, i)) {
cs[length++] = static_cast<char>(i);
}
}
/* convert the char string to a char16_t string */
u_charsToUChars(cs, us, length);
/* add each char16_t to the USet */ for(i=0; i<length; ++i) { if(us[i]!=0 || cs[i]==0) { /* non-invariant chars become (char16_t)0 */
sa->add(sa->set, us[i]);
}
}
}
/* set the direct bytes (byte 0 always maps to itself) */ for(i=1; i<tokenCount; ++i) { if(tokens[i]==-1) { /* convert the direct byte character */
c1 = static_cast<uint8_t>(i);
ds->swapInvChars(ds, &c1, 1, &c2, pErrorCode); if(U_FAILURE(*pErrorCode)) {
udata_printError(ds, "unames/makeTokenMap() finds variant character 0x%02x used (input charset family %d)\n",
i, ds->inCharset); return;
}
/* enter the converted character into the map and mark it used */
map[c1]=c2;
usedOutChar[c2]=true;
}
}
/* set the mappings for the rest of the permutation */ for(i=j=1; i<tokenCount; ++i) { /* set mappings that were not set for direct bytes */ if(map[i]==0) { /* set an output byte value that was not used as an output byte above */ while(usedOutChar[j]) {
++j;
}
map[i] = static_cast<uint8_t>(j++);
}
}
/* check data format and format version */
pInfo=(const UDataInfo *)((constchar *)inData+4); if(!(
pInfo->dataFormat[0]==0x75 && /* dataFormat="unam" */
pInfo->dataFormat[1]==0x6e &&
pInfo->dataFormat[2]==0x61 &&
pInfo->dataFormat[3]==0x6d &&
pInfo->formatVersion[0]==1
)) {
udata_printError(ds, "uchar_swapNames(): data format %02x.%02x.%02x.%02x (format version %02x) is not recognized as unames.icu\n",
pInfo->dataFormat[0], pInfo->dataFormat[1],
pInfo->dataFormat[2], pInfo->dataFormat[3],
pInfo->formatVersion[0]);
*pErrorCode=U_UNSUPPORTED_ERROR; return0;
}
/* iterate through string groups until only a few padding bytes are left */ while(stringsCount>32) {
nextInStrings=expandGroupLengths(inStrings, offsets, lengths);
/* move past the length bytes */
stringsCount-=(uint32_t)(nextInStrings-inStrings);
outStrings+=nextInStrings-inStrings;
inStrings=nextInStrings;
count=offsets[31]+lengths[31]; /* total number of string bytes in this group */
stringsCount-=count;
/* swap the string bytes using map[] and trailMap[] */ while(count>0) {
c=*inStrings++;
*outStrings++=map[c]; if(tokens[c]!=-2) {
--count;
} else { /* token lead byte: swap the trail byte, too */
*outStrings++=trailMap[*inStrings++];
count-=2;
}
}
}
}
for(i=0; i<count; ++i) { if(offset>(uint32_t)length) {
udata_printError(ds, "uchar_swapNames(): too few bytes (%d after header) for unames.icu algorithmic range %u\n",
length, i);
*pErrorCode=U_INDEX_OUTOFBOUNDS_ERROR; return0;
}
Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.
Bemerkung:
Die farbliche Syntaxdarstellung und die Messung sind noch experimentell.