U_CFUNC void U_EXPORT2
ucase_addCaseClosure(UChar32 c, const USetAdder *sa) {
uint16_t props=UTRIE2_GET16(&ucase_props_singleton.trie, c); if(!UCASE_HAS_EXCEPTION(props)) { if(UCASE_GET_TYPE(props)!=UCASE_NONE) { /* add the one simple case mapping, no matter what type it is */
int32_t delta=UCASE_GET_DELTA(props); if(delta!=0) {
sa->add(sa->set, c+delta);
}
}
} else { /* *chasexceptions,sotheremaybemultiplesimpleand/or *fullcasemappings.Addthemall.
*/ const uint16_t *pe=GET_EXCEPTIONS(&ucase_props_singleton, props);
uint16_t excWord=*pe++; const uint16_t *pe0=pe;
// Hardcode the case closure of i and its relatives and ignore the // data file data for these characters. // The Turkic dotless i and dotted I with their case mapping conditions // and case folding option make the related characters behave specially. // This code matches their closure behavior to their case folding behavior. if (excWord&UCASE_EXC_CONDITIONAL_FOLD) { // These characters have Turkic case foldings. Hardcode their closure. if (c == 0x49) { // Regular i and I are in one equivalence class.
sa->add(sa->set, 0x69); return;
} elseif (c == 0x130) { // Dotted I is in a class with <0069 0307> // (for canonical equivalence with <0049 0307>).
sa->addString(sa->set, iDot, 2); return;
}
} elseif (c == 0x69) {
sa->add(sa->set, 0x49); return;
} elseif (c == 0x131) { // Dotless i is in a class by itself. return;
}
/* get the closure string pointer & length */ const char16_t *closure;
int32_t closureLength; if(HAS_SLOT(excWord, UCASE_EXC_CLOSURE)) {
pe=pe0;
GET_SLOT_VALUE(excWord, UCASE_EXC_CLOSURE, pe, closureLength);
closureLength&=UCASE_CLOSURE_MAX_LENGTH; /* higher bits are reserved */
closure=(const char16_t *)pe+1; /* behind this slot, unless there are full case mappings */
} else {
closureLength=0;
closure=nullptr;
}
/* add the full case folding */ if(HAS_SLOT(excWord, UCASE_EXC_FULL_MAPPINGS)) {
pe=pe0;
int32_t fullLength;
GET_SLOT_VALUE(excWord, UCASE_EXC_FULL_MAPPINGS, pe, fullLength);
/* start of full case mapping strings */
++pe;
fullLength&=0xffff; /* bits 16 and higher are reserved */
/* skip the lowercase result string */
pe+=fullLength&UCASE_FULL_LOWER;
fullLength>>=4;
/* add the full case folding string */
int32_t length=fullLength&0xf; if(length!=0) {
sa->addString(sa->set, (const char16_t *)pe, length);
pe+=length;
}
/* skip the uppercase and titlecase strings */
fullLength>>=4;
pe+=fullLength&0xf;
fullLength>>=4;
pe+=fullLength;
closure=(const char16_t *)pe; /* behind full case mappings */
}
/* add each code point in the closure string */ for(int32_t idx=0; idx<closureLength;) {
UChar32 mapping;
U16_NEXT_UNSAFE(closure, idx, mapping);
sa->add(sa->set, mapping);
}
}
}
U_CFUNC void U_EXPORT2
ucase_addSimpleCaseClosure(UChar32 c, const USetAdder *sa) {
uint16_t props=UTRIE2_GET16(&ucase_props_singleton.trie, c); if(!UCASE_HAS_EXCEPTION(props)) { if(UCASE_GET_TYPE(props)!=UCASE_NONE) { /* add the one simple case mapping, no matter what type it is */
int32_t delta=UCASE_GET_DELTA(props); if(delta!=0) {
sa->add(sa->set, c+delta);
}
}
} else { // c has exceptions. Add the mappings relevant for scf=Simple_Case_Folding. const uint16_t *pe=GET_EXCEPTIONS(&ucase_props_singleton, props);
uint16_t excWord=*pe++; const uint16_t *pe0=pe;
// Hardcode the case closure of i and its relatives and ignore the // data file data for these characters, like in ucase_addCaseClosure(). if (excWord&UCASE_EXC_CONDITIONAL_FOLD) { // These characters have Turkic case foldings. Hardcode their closure. if (c == 0x49) { // Regular i and I are in one equivalence class.
sa->add(sa->set, 0x69); return;
} elseif (c == 0x130) { // For scf=Simple_Case_Folding, dotted I is in a class by itself. return;
}
} elseif (c == 0x69) {
sa->add(sa->set, 0x49); return;
} elseif (c == 0x131) { // Dotless i is in a class by itself. return;
}
/* get the closure string pointer & length */ const char16_t *closure;
int32_t closureLength; if(HAS_SLOT(excWord, UCASE_EXC_CLOSURE)) {
pe=pe0;
GET_SLOT_VALUE(excWord, UCASE_EXC_CLOSURE, pe, closureLength);
closureLength&=UCASE_CLOSURE_MAX_LENGTH; /* higher bits are reserved */
closure=(const char16_t *)pe+1; /* behind this slot, unless there are full case mappings */
} else {
closureLength=0;
closure=nullptr;
}
// Skip the full case mappings. if(closureLength > 0 && HAS_SLOT(excWord, UCASE_EXC_FULL_MAPPINGS)) {
pe=pe0;
int32_t fullLength;
GET_SLOT_VALUE(excWord, UCASE_EXC_FULL_MAPPINGS, pe, fullLength);
/* start of full case mapping strings */
++pe;
fullLength&=0xffff; /* bits 16 and higher are reserved */
// Skip all 4 full case mappings.
pe+=fullLength&UCASE_FULL_LOWER;
fullLength>>=4;
pe+=fullLength&0xf;
fullLength>>=4;
pe+=fullLength&0xf;
fullLength>>=4;
pe+=fullLength;
closure=(const char16_t *)pe; /* behind full case mappings */
}
// Add each code point in the closure string whose scf maps back to c. for(int32_t idx=0; idx<closureLength;) {
UChar32 mapping;
U16_NEXT_UNSAFE(closure, idx, mapping);
sa->add(sa->set, mapping);
}
}
}
max-=length; /* we require length<=max, so no need to decrement max in the loop */ do {
c1=*s++;
c2=*t++; if(c2==0) { return1; /* reached the end of t but not of s */
}
c1-=c2; if(c1!=0) { return c1; /* return difference result */
}
} while(--length>0); /* ends with length==0 */
if(max==0 || *t==0) { return0; /* equal to length of both strings */
} else { return -max; /* return length difference */
}
}
if(ucase_props_singleton.unfold==nullptr || s==nullptr) { returnfalse; /* no reverse case folding data, or no string */
} if(length<=1) { /* the string is too short to find any match */ /* *moreprecisewouldbe: *if(!u_strHasMoreChar32Than(s,length,1)) *butthisdoesnotmakemuchpracticaldifferencebecause *asinglesupplementarycodepointwouldjustnotbefound
*/ returnfalse;
}
/** @return same as ucase_getType() and set bit 2 if c is case-ignorable */
U_CAPI int32_t U_EXPORT2
ucase_getTypeOrIgnorable(UChar32 c) {
uint16_t props=UTRIE2_GET16(&ucase_props_singleton.trie, c); return UCASE_GET_TYPE_AND_IGNORABLE(props);
}
for(/* dir!=0 sets direction */; (c=iter(context, dir))>=0; dir=0) {
int32_t type=ucase_getTypeOrIgnorable(c); if(type&4) { /* case-ignorable, continue with the loop */
} elseif(type!=UCASE_NONE) { returntrue; /* followed by cased letter */
} else { returnfalse; /* uncased and not case-ignorable */
}
}
returnfalse; /* not followed by cased letter */
}
/* Is preceded by Soft_Dotted character with no intervening cc=230 ? */ static UBool
isPrecededBySoftDotted(UCaseContextIterator *iter, void *context) {
UChar32 c;
int32_t dotType;
int8_t dir;
if(iter==nullptr) { returnfalse;
}
for(dir=-1; (c=iter(context, dir))>=0; dir=0) {
dotType=getDotType(c); if(dotType==UCASE_SOFT_DOTTED) { returntrue; /* preceded by TYPE_i */
} elseif(dotType!=UCASE_OTHER_ACCENT) { returnfalse; /* preceded by different base character (not TYPE_i), or intervening cc==230 */
}
}
/* Is preceded by base character 'I' with no intervening cc=230 ? */ static UBool
isPrecededBy_I(UCaseContextIterator *iter, void *context) {
UChar32 c;
int32_t dotType;
int8_t dir;
if(iter==nullptr) { returnfalse;
}
for(dir=-1; (c=iter(context, dir))>=0; dir=0) { if(c==0x49) { returntrue; /* preceded by I */
}
dotType=getDotType(c); if(dotType!=UCASE_OTHER_ACCENT) { returnfalse; /* preceded by different base character (not I), or intervening cc==230 */
}
}
returnfalse; /* not preceded by I */
}
/* Is followed by one or more cc==230 ? */ static UBool
isFollowedByMoreAbove(UCaseContextIterator *iter, void *context) {
UChar32 c;
int32_t dotType;
int8_t dir;
if(iter==nullptr) { returnfalse;
}
for(dir=1; (c=iter(context, dir))>=0; dir=0) {
dotType=getDotType(c); if(dotType==UCASE_ABOVE) { returntrue; /* at least one cc==230 following */
} elseif(dotType!=UCASE_OTHER_ACCENT) { returnfalse; /* next base character, no more cc==230 following */
}
}
returnfalse; /* no more cc==230 following */
}
/* Is followed by a dot above (without cc==230 in between) ? */ static UBool
isFollowedByDotAbove(UCaseContextIterator *iter, void *context) {
UChar32 c;
int32_t dotType;
int8_t dir;
if(iter==nullptr) { returnfalse;
}
for(dir=1; (c=iter(context, dir))>=0; dir=0) { if(c==0x307) { returntrue;
}
dotType=getDotType(c); if(dotType!=UCASE_OTHER_ACCENT) { returnfalse; /* next base character or cc==230 in between */
}
}
returnfalse; /* no dot above following */
}
U_CAPI int32_t U_EXPORT2
ucase_toFullLower(UChar32 c,
UCaseContextIterator *iter, void *context, const char16_t **pString,
int32_t loc) { // The sign of the result has meaning, input must be non-negative so that it can be returned as is.
U_ASSERT(c >= 0);
UChar32 result=c; // Reset the output pointer in case it was uninitialized.
*pString=nullptr;
uint16_t props=UTRIE2_GET16(&ucase_props_singleton.trie, c); if(!UCASE_HAS_EXCEPTION(props)) { if(UCASE_IS_UPPER_OR_TITLE(props)) {
result=c+UCASE_GET_DELTA(props);
}
} else { const uint16_t *pe=GET_EXCEPTIONS(&ucase_props_singleton, props), *pe2;
uint16_t excWord=*pe++;
int32_t full;
pe2=pe;
if(excWord&UCASE_EXC_CONDITIONAL_SPECIAL) { /* use hardcoded conditions and mappings */
/* *Testforconditionalmappingsfirst *(otherwisetheunconditionaldefaultmappingsarealwaystaken), *thentestforcharactersthathaveunconditionalmappingsinSpecialCasing.txt, *thengettheUnicodeData.txtmappings.
*/ if( loc==UCASE_LOC_LITHUANIAN && /* base characters, find accents above */
(((c==0x49 || c==0x4a || c==0x12e) &&
isFollowedByMoreAbove(iter, context)) || /* precomposed with accent above, no need to find one */
(c==0xcc || c==0xcd || c==0x128))
) { /* #Lithuanian
0049;00690307;0049;0049;ltMore_Above;#LATINCAPITALLETTERI 004A;006A0307;004A;004A;ltMore_Above;#LATINCAPITALLETTERJ 012E;012F0307;012E;012E;ltMore_Above;#LATINCAPITALLETTERIWITHOGONEK 00CC;006903070300;00CC;00CC;lt;#LATINCAPITALLETTERIWITHGRAVE 00CD;006903070301;00CD;00CD;lt;#LATINCAPITALLETTERIWITHACUTE 0128;006903070303;0128;0128;lt;#LATINCAPITALLETTERIWITHTILDE
*/ switch(c) { case0x49: /* LATIN CAPITAL LETTER I */
*pString=iDot; return2; case0x4a: /* LATIN CAPITAL LETTER J */
*pString=jDot; return2; case0x12e: /* LATIN CAPITAL LETTER I WITH OGONEK */
*pString=iOgonekDot; return2; case0xcc: /* LATIN CAPITAL LETTER I WITH GRAVE */
*pString=iDotGrave; return3; case0xcd: /* LATIN CAPITAL LETTER I WITH ACUTE */
*pString=iDotAcute; return3; case0x128: /* LATIN CAPITAL LETTER I WITH TILDE */
*pString=iDotTilde; return3; default: return0; /* will not occur */
} /* # Turkish and Azeri */
} elseif(loc==UCASE_LOC_TURKISH && c==0x130) { /* #Iandi-dotless;I-dotandiarecasepairsinTurkishandAzeri #Thefollowingruleshandlethosecases.
03A3;03C2;03A3;03A3;Final_Sigma;#GREEKCAPITALLETTERSIGMA
*/ return0x3c2; /* greek small final sigma */
} else { /* no known conditional special case mapping, use a normal mapping */
}
} elseif(HAS_SLOT(excWord, UCASE_EXC_FULL_MAPPINGS)) {
GET_SLOT_VALUE(excWord, UCASE_EXC_FULL_MAPPINGS, pe, full);
full&=UCASE_FULL_LOWER; if(full!=0) { /* set the output pointer to the lowercase mapping */
*pString=reinterpret_cast<const char16_t *>(pe+1);
/* internal */ static int32_t
toUpperOrTitle(UChar32 c,
UCaseContextIterator *iter, void *context, const char16_t **pString,
int32_t loc,
UBool upperNotTitle) { // The sign of the result has meaning, input must be non-negative so that it can be returned as is.
U_ASSERT(c >= 0);
UChar32 result=c; // Reset the output pointer in case it was uninitialized.
*pString=nullptr;
uint16_t props=UTRIE2_GET16(&ucase_props_singleton.trie, c); if(!UCASE_HAS_EXCEPTION(props)) { if(UCASE_GET_TYPE(props)==UCASE_LOWER) {
result=c+UCASE_GET_DELTA(props);
}
} else { const uint16_t *pe=GET_EXCEPTIONS(&ucase_props_singleton, props), *pe2;
uint16_t excWord=*pe++;
int32_t full, idx;
pe2=pe;
if(excWord&UCASE_EXC_CONDITIONAL_SPECIAL) { /* use hardcoded conditions and mappings */ if(loc==UCASE_LOC_TURKISH && c==0x69) { /* #TurkishandAzeri
0307;0307;;;ltAfter_Soft_Dotted;#COMBININGDOTABOVE
*/ return0; /* remove the dot (continue without output) */
} elseif(c==0x0587) { // See ICU-13416: // և ligature ech-yiwn // uppercases to ԵՒ=ech+yiwn by default and in Western Armenian, // but to ԵՎ=ech+vew in Eastern Armenian. if(loc==UCASE_LOC_ARMENIAN) {
*pString=upperNotTitle ? u"ԵՎ" : u"Եվ";
} else {
*pString=upperNotTitle ? u"ԵՒ" : u"Եւ";
} return2;
} else { /* no known conditional special case mapping, use a normal mapping */
}
} elseif(HAS_SLOT(excWord, UCASE_EXC_FULL_MAPPINGS)) {
GET_SLOT_VALUE(excWord, UCASE_EXC_FULL_MAPPINGS, pe, full);
/* start of full case mapping strings */
++pe;
/* skip the lowercase and case-folding result strings */
pe+=full&UCASE_FULL_LOWER;
full>>=4;
pe+=full&0xf;
full>>=4;
if(upperNotTitle) {
full&=0xf;
} else { /* skip the uppercase result string */
pe+=full&0xf;
full=(full>>4)&0xf;
}
if(full!=0) { /* set the output pointer to the result string */
*pString=reinterpret_cast<const char16_t *>(pe);
U_CAPI int32_t U_EXPORT2
ucase_toFullFolding(UChar32 c, const char16_t **pString,
uint32_t options) { // The sign of the result has meaning, input must be non-negative so that it can be returned as is.
U_ASSERT(c >= 0);
UChar32 result=c; // Reset the output pointer in case it was uninitialized.
*pString=nullptr;
uint16_t props=UTRIE2_GET16(&ucase_props_singleton.trie, c); if(!UCASE_HAS_EXCEPTION(props)) { if(UCASE_IS_UPPER_OR_TITLE(props)) {
result=c+UCASE_GET_DELTA(props);
}
} else { const uint16_t *pe=GET_EXCEPTIONS(&ucase_props_singleton, props), *pe2;
uint16_t excWord=*pe++;
int32_t full, idx;
pe2=pe;
if(excWord&UCASE_EXC_CONDITIONAL_FOLD) { /* use hardcoded conditions and mappings */ if((options&_FOLD_CASE_OPTIONS_MASK)==U_FOLD_CASE_DEFAULT) { /* default mappings */ if(c==0x49) { /* 0049; C; 0069; # LATIN CAPITAL LETTER I */ return0x69;
} elseif(c==0x130) { /* 0130; F; 0069 0307; # LATIN CAPITAL LETTER I WITH DOT ABOVE */
*pString=iDot; return2;
}
} else { /* Turkic mappings */ if(c==0x49) { /* 0049; T; 0131; # LATIN CAPITAL LETTER I */ return0x131;
} elseif(c==0x130) { /* 0130; T; 0069; # LATIN CAPITAL LETTER I WITH DOT ABOVE */ return0x69;
}
}
} elseif(HAS_SLOT(excWord, UCASE_EXC_FULL_MAPPINGS)) {
GET_SLOT_VALUE(excWord, UCASE_EXC_FULL_MAPPINGS, pe, full);
/* start of full case mapping strings */
++pe;
/* skip the lowercase result string */
pe+=full&UCASE_FULL_LOWER;
full=(full>>4)&0xf;
if(full!=0) { /* set the output pointer to the result string */
*pString=reinterpret_cast<const char16_t *>(pe);
Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.
Bemerkung:
Die farbliche Syntaxdarstellung und die Messung sind noch experimentell.