/** * Return the script code for a given name, or -1 if not found. */ UScriptCode AnyTransliterator::scriptNameToCode(const UnicodeString& name) { char buf[128]; UScriptCode code; UErrorCode ec = U_ZERO_ERROR; name.extract(0, 128, buf, 128, ""); if (uscript_getCode(buf, &code, 1, &ec) != 1 || U_FAILURE(ec)) { code = USCRIPT_INVALID_CODE; } return code; }
/** * Return the script code for a given name, or -1 if not found. */ static UScriptCode scriptNameToCode(const UnicodeString& name) { char buf[128]; UScriptCode code; UErrorCode ec = U_ZERO_ERROR; int32_t nameLen = name.length(); UBool isInvariant = uprv_isInvariantUString(name.getBuffer(), nameLen); if (isInvariant) { name.extract(0, nameLen, buf, (int32_t)sizeof(buf), US_INV); buf[127] = 0; // Make sure that we NULL terminate the string. } if (!isInvariant || uscript_getCode(buf, &code, 1, &ec) != 1 || U_FAILURE(ec)) { code = USCRIPT_INVALID_CODE; } return code; }
TransliteratorSpec::TransliteratorSpec(const UnicodeString& theSpec) : top(theSpec), res(0) { UErrorCode status = U_ZERO_ERROR; Locale topLoc(""); LocaleUtility::initLocaleFromName(theSpec, topLoc); if (!topLoc.isBogus()) { res = new ResourceBundle(U_ICUDATA_TRANSLIT, topLoc, status); /* test for NULL */ if (res == 0) { return; } if (U_FAILURE(status) || status == U_USING_DEFAULT_WARNING) { delete res; res = 0; } } // Canonicalize script name -or- do locale->script mapping status = U_ZERO_ERROR; static const int32_t capacity = 10; UScriptCode script[capacity]={USCRIPT_INVALID_CODE}; int32_t num = uscript_getCode(CharString().appendInvariantChars(theSpec, status).data(), script, capacity, &status); if (num > 0 && script[0] != USCRIPT_INVALID_CODE) { scriptName = UnicodeString(uscript_getName(script[0]), -1, US_INV); } // Canonicalize top if (res != 0) { // Canonicalize locale name UnicodeString locStr; LocaleUtility::initNameFromLocale(topLoc, locStr); if (!locStr.isBogus()) { top = locStr; } } else if (scriptName.length() != 0) { // We are a script; use canonical name top = scriptName; } // assert(spec != top); reset(); }
void SpoofImpl::addScriptChars(const char *locale, UnicodeSet *allowedChars, UErrorCode &status) { UScriptCode scripts[30]; int32_t numScripts = uscript_getCode(locale, scripts, sizeof(scripts)/sizeof(UScriptCode), &status); if (U_FAILURE(status)) { return; } if (status == U_USING_DEFAULT_WARNING) { status = U_ILLEGAL_ARGUMENT_ERROR; return; } UnicodeSet tmpSet; int32_t i; for (i=0; i<numScripts; i++) { tmpSet.applyIntPropertyValue(UCHAR_SCRIPT, scripts[i], status); allowedChars->addAll(tmpSet); } }
FontMap::FontMap(const char *fileName, le_int16 pointSize, GUISupport *guiSupport, LEErrorCode &status) : fPointSize(pointSize), fFontCount(0), fAscent(0), fDescent(0), fLeading(0), fGUISupport(guiSupport) { le_int32 defaultFont = -1, i, script; le_bool haveFonts = FALSE; for (i = 0; i < scriptCodeCount; i += 1) { fFontIndices[i] = -1; fFontNames[i] = NULL; fFontInstances[i] = NULL; } if (LE_FAILURE(status)) { return; } char *c, *scriptName, *fontName, *line, buffer[BUFFER_SIZE]; FILE *file; file = fopen(fileName, "r"); if (file == NULL) { sprintf(errorMessage, "Could not open the font map file: %s.", fileName); fGUISupport->postErrorMessage(errorMessage, "Font Map Error"); status = LE_FONT_FILE_NOT_FOUND_ERROR; return; } while (fgets(buffer, BUFFER_SIZE, file) != NULL) { UScriptCode scriptCode; UErrorCode scriptStatus = U_ZERO_ERROR; line = strip(buffer); if (line[0] == '#' || line[0] == 0) { continue; } c = strchr(line, ':'); c[0] = 0; fontName = strip(&c[1]); scriptName = strip(line); if (strcmp(scriptName, "DEFAULT") == 0) { defaultFont = getFontIndex(fontName); haveFonts = TRUE; continue; } le_int32 fillCount = uscript_getCode(scriptName, &scriptCode, 1, &scriptStatus); if (U_FAILURE(scriptStatus) || fillCount <= 0 || scriptStatus == U_USING_FALLBACK_WARNING || scriptStatus == U_USING_DEFAULT_WARNING) { sprintf(errorMessage, "The script name %s is invalid.", line); fGUISupport->postErrorMessage(errorMessage, "Font Map Error"); continue; } script = (le_int32) scriptCode; if (fFontIndices[script] >= 0) { // FIXME: complain that this is a duplicate entry and bail (?) fFontIndices[script] = -1; } fFontIndices[script] = getFontIndex(fontName); haveFonts = TRUE; } if (defaultFont >= 0) { for (script = 0; script < scriptCodeCount; script += 1) { if (fFontIndices[script] < 0) { fFontIndices[script] = defaultFont; } } } if (! haveFonts) { sprintf(errorMessage, "The font map file %s does not contain any valid scripts.", fileName); fGUISupport->postErrorMessage(errorMessage, "Font Map Error"); status = LE_ILLEGAL_ARGUMENT_ERROR; } fclose(file); }
void TestUScriptCodeAPI(){ int i =0; int numErrors =0; { const char* testNames[]={ /* test locale */ "en", "en_US", "sr", "ta" , "te_IN", "hi", "he", "ar", /* test abbr */ "Hani", "Hang","Hebr","Hira", "Knda","Kana","Khmr","Lao", "Latn",/*"Latf","Latg",*/ "Mlym", "Mong", /* test names */ "CYRILLIC","DESERET","DEVANAGARI","ETHIOPIC","GEORGIAN", "GOTHIC", "GREEK", "GUJARATI", "COMMON", "INHERITED", /* test lower case names */ "malayalam", "mongolian", "myanmar", "ogham", "old-italic", "oriya", "runic", "sinhala", "syriac","tamil", "telugu", "thaana", "thai", "tibetan", /* test the bounds*/ "tagb", "arabic", /* test bogus */ "asfdasd", "5464", "12235", /* test the last index */ "zyyy", "YI", '\0' }; UScriptCode expected[] ={ /* locales should return */ USCRIPT_LATIN, USCRIPT_LATIN, USCRIPT_CYRILLIC, USCRIPT_TAMIL, USCRIPT_TELUGU, USCRIPT_DEVANAGARI, USCRIPT_HEBREW, USCRIPT_ARABIC, /* abbr should return */ USCRIPT_HAN, USCRIPT_HANGUL, USCRIPT_HEBREW, USCRIPT_HIRAGANA, USCRIPT_KANNADA, USCRIPT_KATAKANA, USCRIPT_KHMER, USCRIPT_LAO, USCRIPT_LATIN,/* USCRIPT_LATIN, USCRIPT_LATIN,*/ USCRIPT_MALAYALAM, USCRIPT_MONGOLIAN, /* names should return */ USCRIPT_CYRILLIC, USCRIPT_DESERET, USCRIPT_DEVANAGARI, USCRIPT_ETHIOPIC, USCRIPT_GEORGIAN, USCRIPT_GOTHIC, USCRIPT_GREEK, USCRIPT_GUJARATI, USCRIPT_COMMON, USCRIPT_INHERITED, /* lower case names should return */ USCRIPT_MALAYALAM, USCRIPT_MONGOLIAN, USCRIPT_MYANMAR, USCRIPT_OGHAM, USCRIPT_OLD_ITALIC, USCRIPT_ORIYA, USCRIPT_RUNIC, USCRIPT_SINHALA, USCRIPT_SYRIAC, USCRIPT_TAMIL, USCRIPT_TELUGU, USCRIPT_THAANA, USCRIPT_THAI, USCRIPT_TIBETAN, /* bounds */ USCRIPT_TAGBANWA, USCRIPT_ARABIC, /* bogus names should return invalid code */ USCRIPT_INVALID_CODE, USCRIPT_INVALID_CODE, USCRIPT_INVALID_CODE, USCRIPT_COMMON, USCRIPT_YI, }; UErrorCode err = U_ZERO_ERROR; const int32_t capacity = 10; for( ; testNames[i]!='\0'; i++){ UScriptCode script[10]={USCRIPT_INVALID_CODE}; uscript_getCode(testNames[i],script,capacity, &err); if( script[0] != expected[i]){ log_data_err("Error getting script code Got: %i Expected: %i for name %s (Error code does not propagate if data is not present. Are you missing data?)\n", script[0],expected[i],testNames[i]); numErrors++; } } if(numErrors >0 ){ log_data_err("Errors uchar_getScriptCode() : %i \n",numErrors); } } { UErrorCode err = U_ZERO_ERROR; int32_t capacity=0; int32_t j; UScriptCode jaCode[]={USCRIPT_KATAKANA, USCRIPT_HIRAGANA, USCRIPT_HAN }; UScriptCode script[10]={USCRIPT_INVALID_CODE}; int32_t num = uscript_getCode("ja",script,capacity, &err); /* preflight */ if(err==U_BUFFER_OVERFLOW_ERROR){ err = U_ZERO_ERROR; capacity = 10; num = uscript_getCode("ja",script,capacity, &err); if(num!=(sizeof(jaCode)/sizeof(UScriptCode))){ log_err("Errors uscript_getScriptCode() for Japanese locale: num=%d, expected %d \n", num, (sizeof(jaCode)/sizeof(UScriptCode))); } for(j=0;j<sizeof(jaCode)/sizeof(UScriptCode);j++) { if(script[j]!=jaCode[j]) { log_err("Japanese locale: code #%d was %d (%s) but expected %d (%s)\n", j, script[j], uscript_getName(script[j]), jaCode[j], uscript_getName(jaCode[j])); } } }else{ log_data_err("Errors in uscript_getScriptCode() expected error : %s got: %s \n", "U_BUFFER_OVERFLOW_ERROR", u_errorName(err)); } } { UScriptCode testAbbr[]={ /* names should return */ USCRIPT_CYRILLIC, USCRIPT_DESERET, USCRIPT_DEVANAGARI, USCRIPT_ETHIOPIC, USCRIPT_GEORGIAN, USCRIPT_GOTHIC, USCRIPT_GREEK, USCRIPT_GUJARATI, }; const char* expectedNames[]={ /* test names */ "Cyrillic","Deseret","Devanagari","Ethiopic","Georgian", "Gothic", "Greek", "Gujarati", '\0' }; i=0; while(i<sizeof(testAbbr)/sizeof(UScriptCode)){ const char* name = uscript_getName(testAbbr[i]); if(name == NULL) { log_data_err("Couldn't get script name\n"); return; } numErrors=0; if(strcmp(expectedNames[i],name)!=0){ log_err("Error getting abbreviations Got: %s Expected: %s\n",name,expectedNames[i]); numErrors++; } if(numErrors > 0){ if(numErrors >0 ){ log_err("Errors uchar_getScriptAbbr() : %i \n",numErrors); } } i++; } } { UScriptCode testAbbr[]={ /* abbr should return */ USCRIPT_HAN, USCRIPT_HANGUL, USCRIPT_HEBREW, USCRIPT_HIRAGANA, USCRIPT_KANNADA, USCRIPT_KATAKANA, USCRIPT_KHMER, USCRIPT_LAO, USCRIPT_LATIN, USCRIPT_MALAYALAM, USCRIPT_MONGOLIAN, }; const char* expectedAbbr[]={ /* test abbr */ "Hani", "Hang","Hebr","Hira", "Knda","Kana","Khmr","Laoo", "Latn", "Mlym", "Mong", '\0' }; i=0; while(i<sizeof(testAbbr)/sizeof(UScriptCode)){ const char* name = uscript_getShortName(testAbbr[i]); numErrors=0; if(strcmp(expectedAbbr[i],name)!=0){ log_err("Error getting abbreviations Got: %s Expected: %s\n",name,expectedAbbr[i]); numErrors++; } if(numErrors > 0){ if(numErrors >0 ){ log_err("Errors uchar_getScriptAbbr() : %i \n",numErrors); } } i++; } } /* now test uscript_getScript() API */ { uint32_t codepoints[] = { 0x0000FF9D, /* USCRIPT_KATAKANA*/ 0x0000FFBE, /* USCRIPT_HANGUL*/ 0x0000FFC7, /* USCRIPT_HANGUL*/ 0x0000FFCF, /* USCRIPT_HANGUL*/ 0x0000FFD7, /* USCRIPT_HANGUL*/ 0x0000FFDC, /* USCRIPT_HANGUL*/ 0x00010300, /* USCRIPT_OLD_ITALIC*/ 0x00010330, /* USCRIPT_GOTHIC*/ 0x0001034A, /* USCRIPT_GOTHIC*/ 0x00010400, /* USCRIPT_DESERET*/ 0x00010428, /* USCRIPT_DESERET*/ 0x0001D167, /* USCRIPT_INHERITED*/ 0x0001D17B, /* USCRIPT_INHERITED*/ 0x0001D185, /* USCRIPT_INHERITED*/ 0x0001D1AA, /* USCRIPT_INHERITED*/ 0x00020000, /* USCRIPT_HAN*/ 0x00000D02, /* USCRIPT_MALAYALAM*/ 0x00000D00, /* USCRIPT_UNKNOWN (new Zzzz value in Unicode 5.0) */ 0x00000000, /* USCRIPT_COMMON*/ 0x0001D169, /* USCRIPT_INHERITED*/ 0x0001D182, /* USCRIPT_INHERITED*/ 0x0001D18B, /* USCRIPT_INHERITED*/ 0x0001D1AD, /* USCRIPT_INHERITED*/ }; UScriptCode expected[] = { USCRIPT_KATAKANA , USCRIPT_HANGUL , USCRIPT_HANGUL , USCRIPT_HANGUL , USCRIPT_HANGUL , USCRIPT_HANGUL , USCRIPT_OLD_ITALIC, USCRIPT_GOTHIC , USCRIPT_GOTHIC , USCRIPT_DESERET , USCRIPT_DESERET , USCRIPT_INHERITED, USCRIPT_INHERITED, USCRIPT_INHERITED, USCRIPT_INHERITED, USCRIPT_HAN , USCRIPT_MALAYALAM, USCRIPT_UNKNOWN, USCRIPT_COMMON, USCRIPT_INHERITED , USCRIPT_INHERITED , USCRIPT_INHERITED , USCRIPT_INHERITED , }; UScriptCode code = USCRIPT_INVALID_CODE; UErrorCode status = U_ZERO_ERROR; UBool passed = TRUE; for(i=0; i<LENGTHOF(codepoints); ++i){ code = uscript_getScript(codepoints[i],&status); if(U_SUCCESS(status)){ if( code != expected[i] || code != (UScriptCode)u_getIntPropertyValue(codepoints[i], UCHAR_SCRIPT) ) { log_err("uscript_getScript for codepoint \\U%08X failed\n",codepoints[i]); passed = FALSE; } }else{ log_err("uscript_getScript for codepoint \\U%08X failed. Error: %s\n", codepoints[i],u_errorName(status)); break; } } if(passed==FALSE){ log_err("uscript_getScript failed.\n"); } } { UScriptCode code= USCRIPT_INVALID_CODE; UErrorCode status = U_ZERO_ERROR; code = uscript_getScript(0x001D169,&status); if(code != USCRIPT_INHERITED){ log_err("\\U001D169 is not contained in USCRIPT_INHERITED"); } } { UScriptCode code= USCRIPT_INVALID_CODE; UErrorCode status = U_ZERO_ERROR; int32_t err = 0; for(i = 0; i<=0x10ffff; i++){ code = uscript_getScript(i,&status); if(code == USCRIPT_INVALID_CODE){ err++; log_err("uscript_getScript for codepoint \\U%08X failed.\n", i); } } if(err>0){ log_err("uscript_getScript failed for %d codepoints\n", err); } } { for(i=0; (UScriptCode)i< USCRIPT_CODE_LIMIT; i++){ const char* name = uscript_getName((UScriptCode)i); if(name==NULL || strcmp(name,"")==0){ log_err("uscript_getName failed for code %i: name is NULL or \"\"\n",i); } } } { /* * These script codes were originally added to ICU pre-3.6, so that ICU would * have all ISO 15924 script codes. ICU was then based on Unicode 4.1. * These script codes were added with only short names because we don't * want to invent long names ourselves. * Unicode 5 and later encode some of these scripts and give them long names. * Whenever this happens, the long script names here need to be updated. */ static const char* expectedLong[] = { "Balinese", "Batk", "Blis", "Brah", "Cham", "Cirt", "Cyrs", "Egyd", "Egyh", "Egyp", "Geok", "Hans", "Hant", "Hmng", "Hung", "Inds", "Java", "Kayah_Li", "Latf", "Latg", "Lepcha", "Lina", "Mand", "Maya", "Mero", "Nko", "Orkh", "Perm", "Phags_Pa", "Phoenician", "Plrd", "Roro", "Sara", "Syre", "Syrj", "Syrn", "Teng", "Vai", "Visp", "Cuneiform", "Zxxx", "Unknown", "Carian", "Jpan", "Lana", "Lycian", "Lydian", "Ol_Chiki", "Rejang", "Saurashtra", "Sgnw", "Sundanese", "Moon", "Mtei", /* new in ICU 4.0 */ "Armi", "Avst", "Cakm", "Kore", "Kthi", "Mani", "Phli", "Phlp", "Phlv", "Prti", "Samr", "Tavt", "Zmth", "Zsym", }; static const char* expectedShort[] = { "Bali", "Batk", "Blis", "Brah", "Cham", "Cirt", "Cyrs", "Egyd", "Egyh", "Egyp", "Geok", "Hans", "Hant", "Hmng", "Hung", "Inds", "Java", "Kali", "Latf", "Latg", "Lepc", "Lina", "Mand", "Maya", "Mero", "Nkoo", "Orkh", "Perm", "Phag", "Phnx", "Plrd", "Roro", "Sara", "Syre", "Syrj", "Syrn", "Teng", "Vaii", "Visp", "Xsux", "Zxxx", "Zzzz", "Cari", "Jpan", "Lana", "Lyci", "Lydi", "Olck", "Rjng", "Saur", "Sgnw", "Sund", "Moon", "Mtei", /* new in ICU 4.0 */ "Armi", "Avst", "Cakm", "Kore", "Kthi", "Mani", "Phli", "Phlp", "Phlv", "Prti", "Samr", "Tavt", "Zmth", "Zsym", }; int32_t j = 0; for(i=USCRIPT_BALINESE; (UScriptCode)i<USCRIPT_CODE_LIMIT; i++, j++){ const char* name = uscript_getName((UScriptCode)i); if(name==NULL || strcmp(name,expectedLong[j])!=0){ log_err("uscript_getName failed for code %i: %s!=%s\n", i, name, expectedLong[j]); } name = uscript_getShortName((UScriptCode)i); if(name==NULL || strcmp(name,expectedShort[j])!=0){ log_err("uscript_getShortName failed for code %i: %s!=%s\n", i, name, expectedShort[j]); } } for(i=0; i<LENGTHOF(expectedLong); i++){ UScriptCode fillIn[5] = {USCRIPT_INVALID_CODE}; UErrorCode status = U_ZERO_ERROR; int32_t len = 0; len = uscript_getCode(expectedShort[i], fillIn, LENGTHOF(fillIn), &status); if(U_FAILURE(status)){ log_err("uscript_getCode failed for script name %s. Error: %s\n",expectedShort[i], u_errorName(status)); } if(len>1){ log_err("uscript_getCode did not return expected number of codes for script %s. EXPECTED: 1 GOT: %i\n", expectedShort[i], len); } if(fillIn[0]!= (UScriptCode)(USCRIPT_BALINESE+i)){ log_err("uscript_getCode did not return expected code for script %s. EXPECTED: %i GOT: %i\n", expectedShort[i], (USCRIPT_BALINESE+i), fillIn[0] ); } } } }
void TestUScriptCodeAPI(){ int i =0; int numErrors =0; { const char* testNames[]={ /* test locale */ "en", "en_US", "sr", "ta" , "te_IN", "hi", "he", "ar", /* test abbr */ "Hani", "Hang","Hebr","Hira", "Knda","Kana","Khmr","Lao", "Latn",/*"Latf","Latg",*/ "Mlym", "Mong", /* test names */ "CYRILLIC","DESERET","DEVANAGARI","ETHIOPIC","GEORGIAN", "GOTHIC", "GREEK", "GUJARATI", "COMMON", "INHERITED", /* test lower case names */ "malayalam", "mongolian", "myanmar", "ogham", "old-italic", "oriya", "runic", "sinhala", "syriac","tamil", "telugu", "thaana", "thai", "tibetan", /* test the bounds*/ "tagb", "arabic", /* test bogus */ "asfdasd", "5464", "12235", /* test the last index */ "zyyy", "YI", NULL }; UScriptCode expected[] ={ /* locales should return */ USCRIPT_LATIN, USCRIPT_LATIN, USCRIPT_CYRILLIC, USCRIPT_TAMIL, USCRIPT_TELUGU, USCRIPT_DEVANAGARI, USCRIPT_HEBREW, USCRIPT_ARABIC, /* abbr should return */ USCRIPT_HAN, USCRIPT_HANGUL, USCRIPT_HEBREW, USCRIPT_HIRAGANA, USCRIPT_KANNADA, USCRIPT_KATAKANA, USCRIPT_KHMER, USCRIPT_LAO, USCRIPT_LATIN,/* USCRIPT_LATIN, USCRIPT_LATIN,*/ USCRIPT_MALAYALAM, USCRIPT_MONGOLIAN, /* names should return */ USCRIPT_CYRILLIC, USCRIPT_DESERET, USCRIPT_DEVANAGARI, USCRIPT_ETHIOPIC, USCRIPT_GEORGIAN, USCRIPT_GOTHIC, USCRIPT_GREEK, USCRIPT_GUJARATI, USCRIPT_COMMON, USCRIPT_INHERITED, /* lower case names should return */ USCRIPT_MALAYALAM, USCRIPT_MONGOLIAN, USCRIPT_MYANMAR, USCRIPT_OGHAM, USCRIPT_OLD_ITALIC, USCRIPT_ORIYA, USCRIPT_RUNIC, USCRIPT_SINHALA, USCRIPT_SYRIAC, USCRIPT_TAMIL, USCRIPT_TELUGU, USCRIPT_THAANA, USCRIPT_THAI, USCRIPT_TIBETAN, /* bounds */ USCRIPT_TAGBANWA, USCRIPT_ARABIC, /* bogus names should return invalid code */ USCRIPT_INVALID_CODE, USCRIPT_INVALID_CODE, USCRIPT_INVALID_CODE, USCRIPT_COMMON, USCRIPT_YI, }; UErrorCode err = U_ZERO_ERROR; const int32_t capacity = 10; for( ; testNames[i]!=NULL; i++){ UScriptCode script[10]={USCRIPT_INVALID_CODE}; uscript_getCode(testNames[i],script,capacity, &err); if( script[0] != expected[i]){ log_data_err("Error getting script code Got: %i Expected: %i for name %s (Error code does not propagate if data is not present. Are you missing data?)\n", script[0],expected[i],testNames[i]); numErrors++; } } if(numErrors >0 ){ log_data_err("Errors uchar_getScriptCode() : %i \n",numErrors); } } { UErrorCode err = U_ZERO_ERROR; int32_t capacity=0; int32_t j; UScriptCode jaCode[]={USCRIPT_KATAKANA, USCRIPT_HIRAGANA, USCRIPT_HAN }; UScriptCode script[10]={USCRIPT_INVALID_CODE}; int32_t num = uscript_getCode("ja",script,capacity, &err); /* preflight */ if(err==U_BUFFER_OVERFLOW_ERROR){ err = U_ZERO_ERROR; capacity = 10; num = uscript_getCode("ja",script,capacity, &err); if(num!=UPRV_LENGTHOF(jaCode)){ log_err("Errors uscript_getScriptCode() for Japanese locale: num=%d, expected %d \n", num, UPRV_LENGTHOF(jaCode)); } for(j=0;j<UPRV_LENGTHOF(jaCode);j++) { if(script[j]!=jaCode[j]) { log_err("Japanese locale: code #%d was %d (%s) but expected %d (%s)\n", j, script[j], uscript_getName(script[j]), jaCode[j], uscript_getName(jaCode[j])); } } }else{ log_data_err("Errors in uscript_getScriptCode() expected error : %s got: %s \n", "U_BUFFER_OVERFLOW_ERROR", u_errorName(err)); } } { static const UScriptCode LATIN[1] = { USCRIPT_LATIN }; static const UScriptCode CYRILLIC[1] = { USCRIPT_CYRILLIC }; static const UScriptCode DEVANAGARI[1] = { USCRIPT_DEVANAGARI }; static const UScriptCode HAN[1] = { USCRIPT_HAN }; static const UScriptCode JAPANESE[3] = { USCRIPT_KATAKANA, USCRIPT_HIRAGANA, USCRIPT_HAN }; static const UScriptCode KOREAN[2] = { USCRIPT_HANGUL, USCRIPT_HAN }; static const UScriptCode HAN_BOPO[2] = { USCRIPT_HAN, USCRIPT_BOPOMOFO }; UScriptCode scripts[5]; UErrorCode err; int32_t num; // Should work regardless of whether we have locale data for the language. err = U_ZERO_ERROR; num = uscript_getCode("tg", scripts, UPRV_LENGTHOF(scripts), &err); assertEqualScripts("tg script: Cyrl", CYRILLIC, 1, scripts, num, err); // Tajik err = U_ZERO_ERROR; num = uscript_getCode("xsr", scripts, UPRV_LENGTHOF(scripts), &err); assertEqualScripts("xsr script: Deva", DEVANAGARI, 1, scripts, num, err); // Sherpa // Multi-script languages. err = U_ZERO_ERROR; num = uscript_getCode("ja", scripts, UPRV_LENGTHOF(scripts), &err); assertEqualScripts("ja scripts: Kana Hira Hani", JAPANESE, UPRV_LENGTHOF(JAPANESE), scripts, num, err); err = U_ZERO_ERROR; num = uscript_getCode("ko", scripts, UPRV_LENGTHOF(scripts), &err); assertEqualScripts("ko scripts: Hang Hani", KOREAN, UPRV_LENGTHOF(KOREAN), scripts, num, err); err = U_ZERO_ERROR; num = uscript_getCode("zh", scripts, UPRV_LENGTHOF(scripts), &err); assertEqualScripts("zh script: Hani", HAN, 1, scripts, num, err); err = U_ZERO_ERROR; num = uscript_getCode("zh-Hant", scripts, UPRV_LENGTHOF(scripts), &err); assertEqualScripts("zh-Hant scripts: Hani Bopo", HAN_BOPO, 2, scripts, num, err); err = U_ZERO_ERROR; num = uscript_getCode("zh-TW", scripts, UPRV_LENGTHOF(scripts), &err); assertEqualScripts("zh-TW scripts: Hani Bopo", HAN_BOPO, 2, scripts, num, err); // Ambiguous API, but this probably wants to return Latin rather than Rongorongo (Roro). err = U_ZERO_ERROR; num = uscript_getCode("ro-RO", scripts, UPRV_LENGTHOF(scripts), &err); assertEqualScripts("ro-RO script: Latn", LATIN, 1, scripts, num, err); } { UScriptCode testAbbr[]={ /* names should return */ USCRIPT_CYRILLIC, USCRIPT_DESERET, USCRIPT_DEVANAGARI, USCRIPT_ETHIOPIC, USCRIPT_GEORGIAN, USCRIPT_GOTHIC, USCRIPT_GREEK, USCRIPT_GUJARATI, }; const char* expectedNames[]={ /* test names */ "Cyrillic","Deseret","Devanagari","Ethiopic","Georgian", "Gothic", "Greek", "Gujarati", NULL }; i=0; while(i<UPRV_LENGTHOF(testAbbr)){ const char* name = uscript_getName(testAbbr[i]); if(name == NULL) { log_data_err("Couldn't get script name\n"); return; } numErrors=0; if(strcmp(expectedNames[i],name)!=0){ log_err("Error getting abbreviations Got: %s Expected: %s\n",name,expectedNames[i]); numErrors++; } if(numErrors > 0){ if(numErrors >0 ){ log_err("Errors uchar_getScriptAbbr() : %i \n",numErrors); } } i++; } } { UScriptCode testAbbr[]={ /* abbr should return */ USCRIPT_HAN, USCRIPT_HANGUL, USCRIPT_HEBREW, USCRIPT_HIRAGANA, USCRIPT_KANNADA, USCRIPT_KATAKANA, USCRIPT_KHMER, USCRIPT_LAO, USCRIPT_LATIN, USCRIPT_MALAYALAM, USCRIPT_MONGOLIAN, }; const char* expectedAbbr[]={ /* test abbr */ "Hani", "Hang","Hebr","Hira", "Knda","Kana","Khmr","Laoo", "Latn", "Mlym", "Mong", NULL }; i=0; while(i<UPRV_LENGTHOF(testAbbr)){ const char* name = uscript_getShortName(testAbbr[i]); numErrors=0; if(strcmp(expectedAbbr[i],name)!=0){ log_err("Error getting abbreviations Got: %s Expected: %s\n",name,expectedAbbr[i]); numErrors++; } if(numErrors > 0){ if(numErrors >0 ){ log_err("Errors uchar_getScriptAbbr() : %i \n",numErrors); } } i++; } } /* now test uscript_getScript() API */ { uint32_t codepoints[] = { 0x0000FF9D, /* USCRIPT_KATAKANA*/ 0x0000FFBE, /* USCRIPT_HANGUL*/ 0x0000FFC7, /* USCRIPT_HANGUL*/ 0x0000FFCF, /* USCRIPT_HANGUL*/ 0x0000FFD7, /* USCRIPT_HANGUL*/ 0x0000FFDC, /* USCRIPT_HANGUL*/ 0x00010300, /* USCRIPT_OLD_ITALIC*/ 0x00010330, /* USCRIPT_GOTHIC*/ 0x0001034A, /* USCRIPT_GOTHIC*/ 0x00010400, /* USCRIPT_DESERET*/ 0x00010428, /* USCRIPT_DESERET*/ 0x0001D167, /* USCRIPT_INHERITED*/ 0x0001D17B, /* USCRIPT_INHERITED*/ 0x0001D185, /* USCRIPT_INHERITED*/ 0x0001D1AA, /* USCRIPT_INHERITED*/ 0x00020000, /* USCRIPT_HAN*/ 0x00000D02, /* USCRIPT_MALAYALAM*/ 0x00000D00, /* USCRIPT_UNKNOWN (new Zzzz value in Unicode 5.0) */ 0x00000000, /* USCRIPT_COMMON*/ 0x0001D169, /* USCRIPT_INHERITED*/ 0x0001D182, /* USCRIPT_INHERITED*/ 0x0001D18B, /* USCRIPT_INHERITED*/ 0x0001D1AD, /* USCRIPT_INHERITED*/ }; UScriptCode expected[] = { USCRIPT_KATAKANA , USCRIPT_HANGUL , USCRIPT_HANGUL , USCRIPT_HANGUL , USCRIPT_HANGUL , USCRIPT_HANGUL , USCRIPT_OLD_ITALIC, USCRIPT_GOTHIC , USCRIPT_GOTHIC , USCRIPT_DESERET , USCRIPT_DESERET , USCRIPT_INHERITED, USCRIPT_INHERITED, USCRIPT_INHERITED, USCRIPT_INHERITED, USCRIPT_HAN , USCRIPT_MALAYALAM, USCRIPT_UNKNOWN, USCRIPT_COMMON, USCRIPT_INHERITED , USCRIPT_INHERITED , USCRIPT_INHERITED , USCRIPT_INHERITED , }; UScriptCode code = USCRIPT_INVALID_CODE; UErrorCode status = U_ZERO_ERROR; UBool passed = TRUE; for(i=0; i<UPRV_LENGTHOF(codepoints); ++i){ code = uscript_getScript(codepoints[i],&status); if(U_SUCCESS(status)){ if( code != expected[i] || code != (UScriptCode)u_getIntPropertyValue(codepoints[i], UCHAR_SCRIPT) ) { log_err("uscript_getScript for codepoint \\U%08X failed\n",codepoints[i]); passed = FALSE; } }else{ log_err("uscript_getScript for codepoint \\U%08X failed. Error: %s\n", codepoints[i],u_errorName(status)); break; } } if(passed==FALSE){ log_err("uscript_getScript failed.\n"); } } { UScriptCode code= USCRIPT_INVALID_CODE; UErrorCode status = U_ZERO_ERROR; code = uscript_getScript(0x001D169,&status); if(code != USCRIPT_INHERITED){ log_err("\\U001D169 is not contained in USCRIPT_INHERITED"); } } { UScriptCode code= USCRIPT_INVALID_CODE; UErrorCode status = U_ZERO_ERROR; int32_t err = 0; for(i = 0; i<=0x10ffff; i++){ code = uscript_getScript(i,&status); if(code == USCRIPT_INVALID_CODE){ err++; log_err("uscript_getScript for codepoint \\U%08X failed.\n", i); } } if(err>0){ log_err("uscript_getScript failed for %d codepoints\n", err); } } { for(i=0; (UScriptCode)i< USCRIPT_CODE_LIMIT; i++){ const char* name = uscript_getName((UScriptCode)i); if(name==NULL || strcmp(name,"")==0){ log_err("uscript_getName failed for code %i: name is NULL or \"\"\n",i); } } } { /* * These script codes were originally added to ICU pre-3.6, so that ICU would * have all ISO 15924 script codes. ICU was then based on Unicode 4.1. * These script codes were added with only short names because we don't * want to invent long names ourselves. * Unicode 5 and later encode some of these scripts and give them long names. * Whenever this happens, the long script names here need to be updated. */ static const char* expectedLong[] = { "Balinese", "Batak", "Blis", "Brahmi", "Cham", "Cirt", "Cyrs", "Egyd", "Egyh", "Egyptian_Hieroglyphs", "Geok", "Hans", "Hant", "Pahawh_Hmong", "Old_Hungarian", "Inds", "Javanese", "Kayah_Li", "Latf", "Latg", "Lepcha", "Linear_A", "Mandaic", "Maya", "Meroitic_Hieroglyphs", "Nko", "Old_Turkic", "Old_Permic", "Phags_Pa", "Phoenician", "Miao", "Roro", "Sara", "Syre", "Syrj", "Syrn", "Teng", "Vai", "Visp", "Cuneiform", "Zxxx", "Unknown", "Carian", "Jpan", "Tai_Tham", "Lycian", "Lydian", "Ol_Chiki", "Rejang", "Saurashtra", "SignWriting", "Sundanese", "Moon", "Meetei_Mayek", /* new in ICU 4.0 */ "Imperial_Aramaic", "Avestan", "Chakma", "Kore", "Kaithi", "Manichaean", "Inscriptional_Pahlavi", "Psalter_Pahlavi", "Phlv", "Inscriptional_Parthian", "Samaritan", "Tai_Viet", "Zmth", "Zsym", /* new in ICU 4.4 */ "Bamum", "Lisu", "Nkgb", "Old_South_Arabian", /* new in ICU 4.6 */ "Bassa_Vah", "Duployan", "Elbasan", "Grantha", "Kpel", "Loma", "Mende_Kikakui", "Meroitic_Cursive", "Old_North_Arabian", "Nabataean", "Palmyrene", "Khudawadi", "Warang_Citi", /* new in ICU 4.8 */ "Afak", "Jurc", "Mro", "Nshu", "Sharada", "Sora_Sompeng", "Takri", "Tangut", "Wole", /* new in ICU 49 */ "Anatolian_Hieroglyphs", "Khojki", "Tirhuta", /* new in ICU 52 */ "Caucasian_Albanian", "Mahajani", /* new in ICU 54 */ "Ahom", "Hatran", "Modi", "Multani", "Pau_Cin_Hau", "Siddham", // new in ICU 58 "Adlam", "Bhaiksuki", "Marchen", "Newa", "Osage", "Hanb", "Jamo", "Zsye" }; static const char* expectedShort[] = { "Bali", "Batk", "Blis", "Brah", "Cham", "Cirt", "Cyrs", "Egyd", "Egyh", "Egyp", "Geok", "Hans", "Hant", "Hmng", "Hung", "Inds", "Java", "Kali", "Latf", "Latg", "Lepc", "Lina", "Mand", "Maya", "Mero", "Nkoo", "Orkh", "Perm", "Phag", "Phnx", "Plrd", "Roro", "Sara", "Syre", "Syrj", "Syrn", "Teng", "Vaii", "Visp", "Xsux", "Zxxx", "Zzzz", "Cari", "Jpan", "Lana", "Lyci", "Lydi", "Olck", "Rjng", "Saur", "Sgnw", "Sund", "Moon", "Mtei", /* new in ICU 4.0 */ "Armi", "Avst", "Cakm", "Kore", "Kthi", "Mani", "Phli", "Phlp", "Phlv", "Prti", "Samr", "Tavt", "Zmth", "Zsym", /* new in ICU 4.4 */ "Bamu", "Lisu", "Nkgb", "Sarb", /* new in ICU 4.6 */ "Bass", "Dupl", "Elba", "Gran", "Kpel", "Loma", "Mend", "Merc", "Narb", "Nbat", "Palm", "Sind", "Wara", /* new in ICU 4.8 */ "Afak", "Jurc", "Mroo", "Nshu", "Shrd", "Sora", "Takr", "Tang", "Wole", /* new in ICU 49 */ "Hluw", "Khoj", "Tirh", /* new in ICU 52 */ "Aghb", "Mahj", /* new in ICU 54 */ "Ahom", "Hatr", "Modi", "Mult", "Pauc", "Sidd", // new in ICU 58 "Adlm", "Bhks", "Marc", "Newa", "Osge", "Hanb", "Jamo", "Zsye" }; int32_t j = 0; if(UPRV_LENGTHOF(expectedLong)!=(USCRIPT_CODE_LIMIT-USCRIPT_BALINESE)) { log_err("need to add new script codes in cucdapi.c!\n"); return; } for(i=USCRIPT_BALINESE; (UScriptCode)i<USCRIPT_CODE_LIMIT; i++, j++){ const char* name = uscript_getName((UScriptCode)i); if(name==NULL || strcmp(name,expectedLong[j])!=0){ log_err("uscript_getName failed for code %i: %s!=%s\n", i, name, expectedLong[j]); } name = uscript_getShortName((UScriptCode)i); if(name==NULL || strcmp(name,expectedShort[j])!=0){ log_err("uscript_getShortName failed for code %i: %s!=%s\n", i, name, expectedShort[j]); } } for(i=0; i<UPRV_LENGTHOF(expectedLong); i++){ UScriptCode fillIn[5] = {USCRIPT_INVALID_CODE}; UErrorCode status = U_ZERO_ERROR; int32_t len = 0; len = uscript_getCode(expectedShort[i], fillIn, UPRV_LENGTHOF(fillIn), &status); if(U_FAILURE(status)){ log_err("uscript_getCode failed for script name %s. Error: %s\n",expectedShort[i], u_errorName(status)); } if(len>1){ log_err("uscript_getCode did not return expected number of codes for script %s. EXPECTED: 1 GOT: %i\n", expectedShort[i], len); } if(fillIn[0]!= (UScriptCode)(USCRIPT_BALINESE+i)){ log_err("uscript_getCode did not return expected code for script %s. EXPECTED: %i GOT: %i\n", expectedShort[i], (USCRIPT_BALINESE+i), fillIn[0] ); } } } { /* test characters which have Script_Extensions */ UErrorCode errorCode=U_ZERO_ERROR; if(!( USCRIPT_COMMON==uscript_getScript(0x0640, &errorCode) && USCRIPT_INHERITED==uscript_getScript(0x0650, &errorCode) && USCRIPT_ARABIC==uscript_getScript(0xfdf2, &errorCode)) || U_FAILURE(errorCode) ) { log_err("uscript_getScript(character with Script_Extensions) failed\n"); } } }
U_CDECL_BEGIN void readTestFile(const char *testFilePath, TestCaseCallback callback) { #if !UCONFIG_NO_REGULAR_EXPRESSIONS UErrorCode status = U_ZERO_ERROR; UXMLParser *parser = UXMLParser::createParser(status); UXMLElement *root = parser->parseFile(testFilePath, status); if (root == NULL) { log_err("Could not open the test data file: %s\n", testFilePath); delete parser; return; } UnicodeString test_case = UNICODE_STRING_SIMPLE("test-case"); UnicodeString test_text = UNICODE_STRING_SIMPLE("test-text"); UnicodeString test_font = UNICODE_STRING_SIMPLE("test-font"); UnicodeString result_glyphs = UNICODE_STRING_SIMPLE("result-glyphs"); UnicodeString result_indices = UNICODE_STRING_SIMPLE("result-indices"); UnicodeString result_positions = UNICODE_STRING_SIMPLE("result-positions"); // test-case attributes UnicodeString id_attr = UNICODE_STRING_SIMPLE("id"); UnicodeString script_attr = UNICODE_STRING_SIMPLE("script"); UnicodeString lang_attr = UNICODE_STRING_SIMPLE("lang"); // test-font attributes UnicodeString name_attr = UNICODE_STRING_SIMPLE("name"); UnicodeString ver_attr = UNICODE_STRING_SIMPLE("version"); UnicodeString cksum_attr = UNICODE_STRING_SIMPLE("checksum"); const UXMLElement *testCase; int32_t tc = 0; while((testCase = root->nextChildElement(tc)) != NULL) { if (testCase->getTagName().compare(test_case) == 0) { char *id = getCString(testCase->getAttribute(id_attr)); char *script = getCString(testCase->getAttribute(script_attr)); char *lang = getCString(testCase->getAttribute(lang_attr)); char *fontName = NULL; char *fontVer = NULL; char *fontCksum = NULL; const UXMLElement *element; int32_t ec = 0; int32_t charCount = 0; // int32_t typoFlags = 3; // kerning + ligatures... UScriptCode scriptCode; le_int32 languageCode = -1; UnicodeString text, glyphs, indices, positions; int32_t glyphCount = 0, indexCount = 0, positionCount = 0; TestResult expected = {0, NULL, NULL, NULL}; uscript_getCode(script, &scriptCode, 1, &status); if (LE_FAILURE(status)) { log_err("invalid script name: %s.\n", script); goto free_c_strings; } if (lang != NULL) { languageCode = getLanguageCode(lang); if (languageCode < 0) { log_err("invalid language name: %s.\n", lang); goto free_c_strings; } } while((element = testCase->nextChildElement(ec)) != NULL) { UnicodeString tag = element->getTagName(); // TODO: make sure that each element is only used once. if (tag.compare(test_font) == 0) { fontName = getCString(element->getAttribute(name_attr)); fontVer = getCString(element->getAttribute(ver_attr)); fontCksum = getCString(element->getAttribute(cksum_attr)); } else if (tag.compare(test_text) == 0) { text = element->getText(TRUE); charCount = text.length(); } else if (tag.compare(result_glyphs) == 0) { glyphs = element->getText(TRUE); } else if (tag.compare(result_indices) == 0) { indices = element->getText(TRUE); } else if (tag.compare(result_positions) == 0) { positions = element->getText(TRUE); } else { // an unknown tag... char *cTag = getCString(&tag); log_info("Test %s: unknown element with tag \"%s\"\n", id, cTag); freeCString(cTag); } } expected.glyphs = (LEGlyphID *) getHexArray(glyphs, glyphCount); expected.indices = (le_int32 *) getHexArray(indices, indexCount); expected.positions = getFloatArray(positions, positionCount); expected.glyphCount = glyphCount; if (glyphCount < charCount || indexCount != glyphCount || positionCount < glyphCount * 2 + 2) { log_err("Test %s: inconsistent input data: charCount = %d, glyphCount = %d, indexCount = %d, positionCount = %d\n", id, charCount, glyphCount, indexCount, positionCount); goto free_expected; }; (*callback)(id, fontName, fontVer, fontCksum, scriptCode, languageCode, text.getBuffer(), charCount, &expected); free_expected: DELETE_ARRAY(expected.positions); DELETE_ARRAY(expected.indices); DELETE_ARRAY(expected.glyphs); free_c_strings: freeCString(fontCksum); freeCString(fontVer); freeCString(fontName); freeCString(lang); freeCString(script); freeCString(id); } } delete root; delete parser; #endif }