class HunspellImpl
{ public:
HunspellImpl(constchar* affpath, constchar* dpath, constchar* key = nullptr);
HunspellImpl(const HunspellImpl&) = delete;
HunspellImpl& operator=(const HunspellImpl&) = delete;
~HunspellImpl(); int add_dic(constchar* dpath, constchar* key = nullptr);
std::vector<std::string> suffix_suggest(const std::string& root_word);
std::vector<std::string> generate(const std::string& word, const std::vector<std::string>& pl);
std::vector<std::string> generate(const std::string& word, const std::string& pattern);
std::vector<std::string> stem(const std::string& word);
std::vector<std::string> stem(const std::vector<std::string>& morph);
std::vector<std::string> analyze(const std::string& word); int get_langnum() const; bool input_conv(const std::string& word, std::string& dest); bool spell(const std::string& word,
std::vector<std::string>& candidate_stack, int* info = nullptr,
std::string* root = nullptr,
std::chrono::steady_clock::time_point suggest_start = std::chrono::steady_clock::time_point::max());
std::vector<std::string> suggest(const std::string& word);
std::vector<std::string> suggest(const std::string& word,
std::vector<std::string>& suggest_candidate_stack,
std::chrono::steady_clock::time_point suggest_start); const std::string& get_wordchars_cpp() const; const std::vector<w_char>& get_wordchars_utf16() const; const std::string& get_dict_encoding() const; int add(const std::string& word); int add_with_flags(const std::string& word, const std::string& flags, const std::string& desc = ""); int add_with_affix(const std::string& word, const std::string& example); int remove(const std::string& word); const std::string& get_version_cpp() const; struct cs_info* get_csconv();
int spell(constchar* word, int* info = nullptr, char** root = nullptr); int suggest(char*** slst, constchar* word); int suffix_suggest(char*** slst, constchar* root_word); void free_list(char*** slst, int n); char* get_dic_encoding(); int analyze(char*** slst, constchar* word); int stem(char*** slst, constchar* word); int stem(char*** slst, char** morph, int n); int generate(char*** slst, constchar* word, constchar* word2); int generate(char*** slst, constchar* word, char** desc, int n); constchar* get_wordchars() const; constchar* get_version() const; int input_conv(constchar* word, char* dest, size_t destsize);
private:
std::vector<std::unique_ptr<HashMgr>> m_HMgrs;
std::unique_ptr<AffixMgr> pAMgr; // pAMgr depends on m_HMgrs
std::unique_ptr<SuggestMgr> pSMgr; // pSMgr depends on pAMgr
std::string affixpath;
std::string encoding; conststruct cs_info* csconv; int langnum; int utf8; int complexprefixes;
std::vector<std::string> wordbreak;
/* first set up the hash manager */
m_HMgrs.push_back(std::unique_ptr<HashMgr>(new HashMgr(dpath, affpath, key)));
/* next set up the affix manager */ /* it needs access to the hash manager lookup methods */
pAMgr.reset(new AffixMgr(affpath, m_HMgrs, key));
/* get the preferred try string and the dictionary */ /* encoding from the Affix Manager for that dictionary */
std::string try_string = pAMgr->get_try_string();
encoding = pAMgr->get_encoding();
langnum = pAMgr->get_langnum();
utf8 = pAMgr->get_utf8(); if (!utf8)
csconv = get_current_cs(encoding);
complexprefixes = pAMgr->get_complexprefixes();
wordbreak = pAMgr->get_breaktable();
/* and finally set up the suggestion manager */
pSMgr.reset(new SuggestMgr(try_string, MAXSUGGESTION, pAMgr.get()));
}
// load extra dictionaries int HunspellImpl::add_dic(constchar* dpath, constchar* key) {
m_HMgrs.push_back(std::unique_ptr<HashMgr>(new HashMgr(dpath, affixpath.c_str(), key))); return0;
}
// make a copy of src at dest while removing all characters // specified in IGNORE rule void HunspellImpl::clean_ignore(std::string& dest, const std::string& src) {
dest.clear();
dest.assign(src); constchar* ignoredchars = pAMgr ? pAMgr->get_ignore() : nullptr; if (ignoredchars != nullptr) { if (utf8) { const std::vector<w_char>& ignoredchars_utf16 =
pAMgr->get_ignore_utf16();
remove_ignored_chars_utf(dest, ignoredchars_utf16);
} else {
remove_ignored_chars(dest, ignoredchars);
}
}
}
// make a copy of src at destination while removing all leading // blanks and removing any trailing periods after recording // their presence with the abbreviation flag // also since already going through character by character, // set the capitalization type // return the length of the "cleaned" (and UTF-8 encoded) word
/* insert a word to the beginning of the suggestion array */ void HunspellImpl::insert_sug(std::vector<std::string>& slst, const std::string& word) {
slst.insert(slst.begin(), word);
}
bool HunspellImpl::spell(const std::string& word, std::vector<std::string>& candidate_stack, int* info, std::string* root,
std::chrono::steady_clock::time_point suggest_start) { // something very broken if spell ends up calling itself with the same word if (std::find(candidate_stack.begin(), candidate_stack.end(), word) != candidate_stack.end()) returnfalse;
int info2 = 0; if (!info)
info = &info2; else
*info = 0;
// Hunspell supports XML input of the simplified API (see manual) if (word == SPELL_XML) returntrue; if (utf8) { if (word.size() >= MAXWORDUTF8LEN) returnfalse;
} else { if (word.size() >= MAXWORDLEN) returnfalse;
} int captype = NOCAP;
size_t abbv = 0;
size_t wl = 0;
// other patterns for (auto& j : wordbreak) {
size_t plen = j.size();
size_t found = scw.find(j); if ((found > 0) && (found < wl - plen)) {
size_t found2 = scw.find(j, found + 1); // try to break at the second occurance // to recognize dictionary words with wordbreak if (found2 > 0 && (found2 < wl - plen))
found = found2;
std::string substring(scw.substr(found + plen)); if (!spell(substring, candidate_stack, nullptr, nullptr, suggest_start)) continue;
std::string suffix(scw.substr(found));
scw.resize(found); // examine 2 sides of the break point if (spell(scw, candidate_stack, nullptr, nullptr, suggest_start)) {
*info |= SPELL_COMPOUND; returntrue;
}
scw.append(suffix);
// LANG_hu: spec. dash rule if (langnum == LANG_hu && j == "-") {
suffix = scw.substr(found + 1);
scw.resize(found + 1); if (spell(scw, candidate_stack, nullptr, nullptr, suggest_start)) {
*info |= SPELL_COMPOUND; returntrue; // check the first part with dash
}
scw.append(suffix);
} // end of LANG specific region
}
}
// other patterns (break at first break point) for (auto& j : wordbreak) {
size_t plen = j.size(), found = scw.find(j); if ((found > 0) && (found < wl - plen)) { if (!spell(scw.substr(found + plen), candidate_stack, nullptr, nullptr, suggest_start)) continue;
std::string suffix(scw.substr(found));
scw.resize(found); // examine 2 sides of the break point if (spell(scw, candidate_stack, nullptr, nullptr, suggest_start)) {
*info |= SPELL_COMPOUND; returntrue;
}
scw.append(suffix);
// LANG_hu: spec. dash rule if (langnum == LANG_hu && j == "-") {
suffix = scw.substr(found + 1);
scw.resize(found + 1); if (spell(scw, candidate_stack, nullptr, nullptr, suggest_start)) {
*info |= SPELL_COMPOUND; returntrue; // check the first part with dash
}
scw.append(suffix);
} // end of LANG specific region
}
}
}
// remove IGNORE characters from the string
clean_ignore(word, w);
if (word.empty()) return nullptr;
// word reversing wrapper for complex prefixes if (complexprefixes) { if (utf8)
reverseword_utf(word); else
reverseword(word);
}
int len = word.size();
// look word in hash table struct hentry* he = nullptr; for (size_t i = 0; (i < m_HMgrs.size()) && !he; ++i) {
he = m_HMgrs[i]->lookup(word.c_str(), word.size());
// check forbidden and onlyincompound words if ((he) && (he->astr) && (pAMgr) &&
TESTAFF(he->astr, pAMgr->get_forbiddenword(), he->alen)) { if (info)
*info |= SPELL_FORBIDDEN; // LANG_hu section: set dash information for suggestions if (langnum == LANG_hu) { if (pAMgr->get_compoundflag() &&
TESTAFF(he->astr, pAMgr->get_compoundflag(), he->alen)) { if (info)
*info |= SPELL_COMPOUND;
}
} return nullptr;
}
// he = next not needaffix, onlyincompound homonym or onlyupcase word while (he && (he->astr) && pAMgr &&
((pAMgr->get_needaffix() &&
TESTAFF(he->astr, pAMgr->get_needaffix(), he->alen)) ||
(pAMgr->get_onlyincompound() &&
TESTAFF(he->astr, pAMgr->get_onlyincompound(), he->alen)) ||
(info && (*info & SPELL_INITCAP) &&
TESTAFF(he->astr, ONLYUPCASEFLAG, he->alen))))
he = he->next_homonym;
}
// check with affixes if (!he && pAMgr) { // try stripping off affixes
he = pAMgr->affix_check(word, 0, len, 0);
if (he) { if ((he->astr) && (pAMgr) &&
TESTAFF(he->astr, pAMgr->get_forbiddenword(), he->alen)) { if (info)
*info |= SPELL_FORBIDDEN; return nullptr;
} if (root) {
root->assign(he->word); if (complexprefixes) { if (utf8)
reverseword_utf(*root); else
reverseword(*root);
}
} // try check compound word
} elseif (pAMgr->get_compound()) { struct hentry* rwords[100] = {}; // buffer for COMPOUND pattern checking
// first allow only 2 words in the compound int setinfo = SPELL_COMPOUND_2; if (info)
setinfo |= *info;
he = pAMgr->compound_check(word, 0, 0, 100, 0, nullptr, (hentry**)&rwords, 0, 0, &setinfo); if (info)
*info = setinfo & ~SPELL_COMPOUND_2; // if not 2-word compoud word, try with 3 or more words // (only if original info didn't forbid it) if (!he && info && !(*info & SPELL_COMPOUND_2)) {
*info &= ~SPELL_COMPOUND_2;
he = pAMgr->compound_check(word, 0, 0, 100, 0, nullptr, (hentry**)&rwords, 0, 0, info); // accept the compound with 3 or more words only if it is // - not a dictionary word with a typo and // - not two words written separately, // - or if it's an arbitrary number accepted by compound rules (e.g. 999%) if (he && !isdigit(word[0]))
{
std::vector<std::string> slst; if (pSMgr->suggest(slst, word, nullptr, /*test_simplesug=*/true))
he = nullptr;
}
}
// LANG_hu section: `moving rule' with last dash if ((!he) && (langnum == LANG_hu) && (word[len - 1] == '-')) {
std::string dup(word, 0, len - 1);
he = pAMgr->compound_check(dup, -5, 0, 100, 0, nullptr, (hentry**)&rwords, 1, 0, info);
} // end of LANG specific region if (he) { if (root) {
root->assign(he->word); if (complexprefixes) { if (utf8)
reverseword_utf(*root); else
reverseword(*root);
}
} if (info)
*info |= SPELL_COMPOUND;
}
}
}
std::vector<std::string> HunspellImpl::suggest(const std::string& word, std::vector<std::string>& suggest_candidate_stack, std::chrono::steady_clock::time_point suggest_start) {
if (suggest_candidate_stack.size() > MAX_CANDIDATE_STACK_DEPTH || // apply a fairly arbitrary depth limit // something very broken if suggest ends up calling itself with the same word
std::find(suggest_candidate_stack.begin(), suggest_candidate_stack.end(), word) != suggest_candidate_stack.end()) { return { };
}
if (std::chrono::steady_clock::now() - suggest_start > TIMELIMIT_GLOBAL_MS) return { };
bool capwords;
size_t abbv; int captype;
std::vector<std::string> spell_candidate_stack;
suggest_candidate_stack.push_back(word);
std::vector<std::string> slst = suggest_internal(word, spell_candidate_stack, suggest_candidate_stack,
capwords, abbv, captype, suggest_start);
suggest_candidate_stack.pop_back(); // word reversing wrapper for complex prefixes if (complexprefixes) { for (auto& j : slst) { if (utf8)
reverseword_utf(j); else
reverseword(j);
}
}
// capitalize if (capwords) { for (auto& j : slst) {
std::string capitalized(j);
mkinitcap(capitalized); if (capitalized == word) continue; // capitalizing would just reproduce the misspelled word
j = std::move(capitalized);
}
}
// expand suggestions with dot(s) if (abbv && pAMgr && pAMgr->get_sugswithdots() && word.size() >= abbv) { for (auto& j : slst) {
j.append(word.substr(word.size() - abbv));
}
}
// remove bad capitalized and forbidden forms if (pAMgr && (pAMgr->get_keepcase() || pAMgr->get_forbiddenword())) { switch (captype) { case INITCAP: case ALLCAP: {
size_t l = 0; for (size_t j = 0; j < slst.size(); ++j) { if (slst[j].find(' ') == std::string::npos && !spell(slst[j], spell_candidate_stack, nullptr, nullptr, suggest_start)) {
std::string s;
std::vector<w_char> w; if (utf8) {
u8_u16(w, slst[j]);
} else {
s = slst[j];
}
mkallsmall2(s, w); if (spell(s, spell_candidate_stack, nullptr, nullptr, suggest_start)) {
slst[l] = std::move(s);
++l;
} else {
mkinitcap2(s, w); if (spell(s, spell_candidate_stack, nullptr, nullptr, suggest_start)) {
slst[l] = std::move(s);
++l;
}
}
} else {
slst[l] = slst[j];
++l;
}
}
slst.resize(l);
}
}
}
// remove duplications
size_t l = 0; for (size_t j = 0; j < slst.size(); ++j) {
slst[l] = slst[j]; for (size_t k = 0; k < l; ++k) { if (slst[k] == slst[j]) {
--l; break;
}
}
++l;
}
slst.resize(l);
// output conversion
RepList* rl = (pAMgr) ? pAMgr->get_oconvtable() : nullptr; if (rl) {
size_t l = 0; for (size_t i = 0; i < slst.size(); ++i) {
std::string wspace; if (rl->conv(slst[i], wspace)) {
slst[i] = std::move(wspace);
} // gh#1002: OCONV can map a generated form back to the input word // (e.g. "románórum" -> "romanórum" when the user typed "romanórum"), // leaving the misspelled word as its own suggestion. if (slst[i] == word) continue; if (l != i)
slst[l] = std::move(slst[i]);
++l;
}
slst.resize(l);
} return slst;
}
int onlycmpdsug = 0; if (!pSMgr || m_HMgrs.empty()) return slst;
// process XML input of the simplified API (see manual) if (word.compare(0, sizeof(SPELL_XML) - 3, SPELL_XML, sizeof(SPELL_XML) - 3) == 0) { return spellml(word);
} if (utf8) { if (word.size() >= MAXWORDUTF8LEN) return slst;
} else { if (word.size() >= MAXWORDLEN) return slst;
}
size_t wl = 0;
#ifdefined(FUZZING_BUILD_MODE_UNSAFE_FOR_PRODUCTION) if (wl > 32768) return slst; #endif
}
bool good = false;
// check capitalized form for FORCEUCASE if (pAMgr && captype == NOCAP && pAMgr->get_forceucase()) { int info = SPELL_ORIGCAP; if (checkword(scw, &info, nullptr, suggest_start)) {
std::string form(std::move(scw));
mkinitcap(form);
slst.push_back(std::move(form)); return slst;
}
}
switch (captype) { case NOCAP: {
good |= pSMgr->suggest(slst, scw, &onlycmpdsug); if (std::chrono::steady_clock::now() - suggest_start > TIMELIMIT_GLOBAL_MS) return slst; if (abbv) {
std::string wspace(scw);
wspace.push_back('.');
good |= pSMgr->suggest(slst, wspace, &onlycmpdsug); if (std::chrono::steady_clock::now() - suggest_start > TIMELIMIT_GLOBAL_MS) return slst;
} break;
}
case INITCAP: {
capwords = true;
good |= pSMgr->suggest(slst, scw, &onlycmpdsug); if (std::chrono::steady_clock::now() - suggest_start > TIMELIMIT_GLOBAL_MS) return slst;
std::string wspace(scw);
mkallsmall2(wspace, sunicw);
good |= pSMgr->suggest(slst, wspace, &onlycmpdsug); if (std::chrono::steady_clock::now() - suggest_start > TIMELIMIT_GLOBAL_MS) return slst; break;
} case HUHINITCAP:
capwords = true; /* FALLTHROUGH */ case HUHCAP: {
good |= pSMgr->suggest(slst, scw, &onlycmpdsug); if (std::chrono::steady_clock::now() - suggest_start > TIMELIMIT_GLOBAL_MS) return slst; // something.The -> something. The
size_t dot_pos = scw.find('.'); if (dot_pos != std::string::npos) {
std::string postdot = scw.substr(dot_pos + 1); int captype_; if (utf8) {
std::vector<w_char> postdotu;
u8_u16(postdotu, postdot);
captype_ = get_captype_utf8(postdotu, langnum);
} else {
captype_ = get_captype(postdot, csconv);
} if (captype_ == INITCAP) {
std::string str(scw);
str.insert(dot_pos + 1, 1, ' ');
insert_sug(slst, str);
}
}
std::string wspace;
if (captype == HUHINITCAP) { // TheOpenOffice.org -> The OpenOffice.org
wspace = scw;
mkinitsmall2(wspace, sunicw);
good |= pSMgr->suggest(slst, wspace, &onlycmpdsug); if (std::chrono::steady_clock::now() - suggest_start > TIMELIMIT_GLOBAL_MS) return slst;
}
wspace = scw;
mkallsmall2(wspace, sunicw); if (spell(wspace, spell_candidate_stack, nullptr, nullptr, suggest_start))
insert_sug(slst, wspace);
size_t prevns = slst.size();
good |= pSMgr->suggest(slst, wspace, &onlycmpdsug); if (std::chrono::steady_clock::now() - suggest_start > TIMELIMIT_GLOBAL_MS) return slst; if (captype == HUHINITCAP) {
mkinitcap2(wspace, sunicw); if (spell(wspace, spell_candidate_stack, nullptr, nullptr, suggest_start))
insert_sug(slst, wspace);
good |= pSMgr->suggest(slst, wspace, &onlycmpdsug); if (std::chrono::steady_clock::now() - suggest_start > TIMELIMIT_GLOBAL_MS) return slst;
} // aNew -> "a New" (instead of "a new") for (size_t j = prevns; j < slst.size(); ++j) { constchar* space = strchr(slst[j].c_str(), ' '); if (space) {
size_t slen = strlen(space + 1); // different case after space (need capitalisation) if ((slen < wl) && strcmp(scw.c_str() + wl - slen, space + 1) != 0) {
std::string first(slst[j].c_str(), space + 1);
std::string second(space + 1);
std::vector<w_char> w; if (utf8)
u8_u16(w, second);
mkinitcap2(second, w); // set as first suggestion
slst.erase(slst.begin() + j);
slst.insert(slst.begin(), first + second);
}
}
} break;
}
case ALLCAP: {
std::string wspace(scw);
mkallsmall2(wspace, sunicw);
good |= pSMgr->suggest(slst, wspace, &onlycmpdsug); if (std::chrono::steady_clock::now() - suggest_start > TIMELIMIT_GLOBAL_MS) return slst; if (pAMgr && pAMgr->get_keepcase() && spell(wspace, spell_candidate_stack, nullptr, nullptr, suggest_start))
insert_sug(slst, wspace);
mkinitcap2(wspace, sunicw);
good |= pSMgr->suggest(slst, wspace, &onlycmpdsug); if (std::chrono::steady_clock::now() - suggest_start > TIMELIMIT_GLOBAL_MS) return slst; for (auto& j : slst) {
mkallcap(j); if (pAMgr && pAMgr->get_checksharps()) { if (utf8) {
mystrrep(j, "\xC3\x9F", "SS");
} else {
mystrrep(j, "\xDF", "SS");
}
}
} break;
}
}
// LANG_hu section: replace '-' with ' ' in Hungarian if (langnum == LANG_hu) { for (auto& j : slst) {
size_t pos = j.find('-'); if (pos != std::string::npos) { int info;
std::string w(j.substr(0, pos));
w.append(j.substr(pos + 1));
(void)spell(w, spell_candidate_stack, &info, nullptr, suggest_start); if ((info & SPELL_COMPOUND) && (info & SPELL_FORBIDDEN)) {
j[pos] = ' ';
} else
j[pos] = '-';
}
}
} // END OF LANG_hu section // try ngram approach since found nothing good suggestion if (!good && pAMgr && (slst.empty() || onlycmpdsug) && (pAMgr->get_maxngramsugs() != 0)) { switch (captype) { case NOCAP: {
pSMgr->ngsuggest(slst, scw.c_str(), m_HMgrs, NOCAP); if (std::chrono::steady_clock::now() - suggest_start > TIMELIMIT_GLOBAL_MS) return slst; break;
} /* FALLTHROUGH */ case HUHINITCAP:
capwords = true; /* FALLTHROUGH */ case HUHCAP: {
std::string wspace(scw);
mkallsmall2(wspace, sunicw);
pSMgr->ngsuggest(slst, wspace.c_str(), m_HMgrs, HUHCAP); if (std::chrono::steady_clock::now() - suggest_start > TIMELIMIT_GLOBAL_MS) return slst; break;
} case INITCAP: {
capwords = true;
std::string wspace(scw);
mkallsmall2(wspace, sunicw);
pSMgr->ngsuggest(slst, wspace.c_str(), m_HMgrs, INITCAP); if (std::chrono::steady_clock::now() - suggest_start > TIMELIMIT_GLOBAL_MS) return slst; break;
} case ALLCAP: {
std::string wspace(scw);
mkallsmall2(wspace, sunicw);
size_t oldns = slst.size();
pSMgr->ngsuggest(slst, wspace.c_str(), m_HMgrs, ALLCAP); if (std::chrono::steady_clock::now() - suggest_start > TIMELIMIT_GLOBAL_MS) return slst; for (size_t j = oldns; j < slst.size(); ++j) {
mkallcap(slst[j]);
} break;
}
}
}
// try dash suggestion (Afo-American -> Afro-American) // Note: LibreOffice was modified to treat dashes as word // characters to check "scot-free" etc. word forms, but // we need to handle suggestions for "Afo-American", etc., // while "Afro-American" is missing from the dictionary. // TODO avoid possible overgeneration
size_t dash_pos = scw.find('-'); if (dash_pos != std::string::npos) { int nodashsug = 1; for (size_t j = 0; j < slst.size() && nodashsug == 1; ++j) { if (slst[j].find('-') != std::string::npos)
nodashsug = 0;
}
size_t prev_pos = 0; bool last = false;
while (!good && nodashsug && !last) { if (dash_pos == scw.size())
last = true;
std::string chunk = scw.substr(prev_pos, dash_pos - prev_pos); if (chunk != word && !spell(chunk, spell_candidate_stack, nullptr, nullptr, suggest_start)) {
std::vector<std::string> nlst = suggest(chunk, suggest_candidate_stack, suggest_start); if (std::chrono::steady_clock::now() - suggest_start > TIMELIMIT_GLOBAL_MS) return slst; for (auto j = nlst.rbegin(); j != nlst.rend(); ++j) {
std::string wspace = scw.substr(0, prev_pos);
wspace.append(*j); if (!last) {
wspace.append("-");
wspace.append(scw.substr(dash_pos + 1));
} int info = 0; if (pAMgr && pAMgr->get_forbiddenword())
checkword(wspace, &info, nullptr, suggest_start); if (!(info & SPELL_FORBIDDEN))
insert_sug(slst, wspace);
}
nodashsug = 0;
} if (!last) {
prev_pos = dash_pos + 1;
dash_pos = scw.find('-', prev_pos);
} if (dash_pos == std::string::npos)
dash_pos = scw.size();
}
} return slst;
}
int HunspellImpl::add(const std::string& word) { if (!m_HMgrs.empty()) return m_HMgrs[0]->add(word); return0;
}
int HunspellImpl::add_with_flags(const std::string& word, const std::string& flags, const std::string& desc) { if (!m_HMgrs.empty()) return m_HMgrs[0]->add_with_flags(word, flags, desc); return0;
}
int HunspellImpl::add_with_affix(const std::string& word, const std::string& example) { if (!m_HMgrs.empty()) return m_HMgrs[0]->add_with_affix(word, example); return0;
}
int HunspellImpl::remove(const std::string& word) { if (!m_HMgrs.empty()) return m_HMgrs[0]->remove(word); return0;
}
struct cs_info* HunspellImpl::get_csconv() { // Preserve pre-1.7.3 ABI: returned pointer is now to read-only data, // but the public signature still says non-const. Callers must not // write through it. returnconst_cast<struct cs_info*>(csconv);
}
void HunspellImpl::cat_result(std::string& result, const std::string& st) { if (!st.empty()) { if (!result.empty())
result.append("\n");
result.append(st);
}
}
std::vector<std::string> HunspellImpl::analyze(const std::string& word) {
std::vector<std::string> slst = analyze_internal(word); // output conversion
RepList* rl = (pAMgr) ? pAMgr->get_oconvtable() : nullptr; if (rl) { for (size_t i = 0; rl && i < slst.size(); ++i) {
std::string wspace; if (rl->conv(slst[i], wspace)) {
slst[i] = std::move(wspace);
}
}
} return slst;
}
std::vector<std::string> HunspellImpl::analyze_internal(const std::string& word) {
std::vector<std::string> candidate_stack, slst; if (!pSMgr || m_HMgrs.empty()) return slst; if (utf8) { if (word.size() >= MAXWORDUTF8LEN) return slst;
} else { if (word.size() >= MAXWORDLEN) return slst;
} int captype = NOCAP;
size_t abbv = 0;
size_t wl = 0;
if (!result.empty()) { // word reversing wrapper for complex prefixes if (complexprefixes) { if (utf8)
reverseword_utf(result); else
reverseword(result);
} return line_tok(result, MSEP_REC);
}
// compound word with dash (HU) I18n // LANG_hu section: set dash information for suggestions
size_t dash_pos = langnum == LANG_hu ? scw.find('-') : std::string::npos; if (dash_pos != std::string::npos) { int nresult = 0;
// return the beginning of the element (attr == NULL) or the attribute
std::string::size_type HunspellImpl::get_xml_pos(const std::string& s, std::string::size_type pos, constchar* attr) { if (pos == std::string::npos) return std::string::npos;
std::string::size_type qpos = in_word.find("<query"); if (qpos == std::string::npos) return slst; // bad XML input
std::string::size_type q2pos = in_word.find('>', qpos); if (q2pos == std::string::npos) return slst; // bad XML input
q2pos = in_word.find("<word", q2pos); if (q2pos == std::string::npos) return slst; // bad XML input
if (check_xml_par(in_word, qpos, "type=", "analyze")) {
std::string cw = get_xml_par(in_word, in_word.find('>', q2pos)); if (!cw.empty())
slst = analyze(cw); if (slst.empty()) return slst; // convert the result to <code><a>ana1</a><a>ana2</a></code> format
std::string r;
r.append("<code>"); for (auto entry : slst) {
r.append("<a>");
std::vector<std::string> HunspellImpl::suffix_suggest(const std::string& root_word) {
std::vector<std::string> slst; struct hentry* he = nullptr; int len;
std::string w2; constchar* word; constchar* ignoredchars = pAMgr->get_ignore(); if (ignoredchars != nullptr) {
w2.assign(root_word); if (utf8) { const std::vector<w_char>& ignoredchars_utf16 =
pAMgr->get_ignore_utf16();
remove_ignored_chars_utf(w2, ignoredchars_utf16);
} else {
remove_ignored_chars(w2, ignoredchars);
}
word = w2.c_str();
len = (int)w2.size();
} else {
word = root_word.c_str();
len = (int)root_word.size();
}
if (!len) return slst;
for (size_t i = 0; (i < m_HMgrs.size()) && !he; ++i) {
he = m_HMgrs[i]->lookup(word, len);
} if (he) {
slst = pAMgr->get_suffix_words(he->astr, he->alen, root_word);
} return slst;
}
namespace { // using malloc because this is for the c-api where the callers // expect to be able to use free char* stringdup(const std::string& s) {
size_t sl = s.size() + 1; char* d = (char*)malloc(sl); if (d)
memcpy(d, s.c_str(), sl); return d;
}
int munge_vector(char*** slst, const std::vector<std::string>& items) { if (items.empty()) {
*slst = nullptr; return0;
} else {
*slst = newchar*[items.size()]; for (size_t i = 0; i < items.size(); ++i)
(*slst)[i] = stringdup(items[i]);
} return items.size();
}
}
int HunspellImpl::spell(constchar* word, int* info, char** root) {
std::string sroot;
std::vector<std::string> candidate_stack; bool ret = spell(word, candidate_stack, info, root ? &sroot : nullptr); if (root) { if (sroot.empty()) {
*root = nullptr;
} else {
*root = stringdup(sroot);
}
} return ret;
}
int HunspellImpl::stem(char*** slst, char** desc, int n) {
std::vector<std::string> morph;
morph.reserve(n); for (int i = 0; i < n; ++i) morph.emplace_back(desc[i]);
int HunspellImpl::generate(char*** slst, constchar* word, constchar* pattern) {
std::vector<std::string> stems = generate(word, pattern); return munge_vector(slst, stems);
}
int HunspellImpl::generate(char*** slst, constchar* word, char** pl, int pln) {
std::vector<std::string> morph;
morph.reserve(pln); for (int i = 0; i < pln; ++i) morph.emplace_back(pl[i]);
int Hunspell_stem2(Hunhandle* pHunspell, char*** slst, char** desc, int n) { returnreinterpret_cast<HunspellImpl*>(pHunspell)->stem(slst, desc, n);
}
int Hunspell_generate(Hunhandle* pHunspell, char*** slst, constchar* word, constchar* pattern)
{ returnreinterpret_cast<HunspellImpl*>(pHunspell)->generate(slst, word, pattern);
}
int Hunspell_generate2(Hunhandle* pHunspell, char*** slst, constchar* word, char** desc, int n)
{ returnreinterpret_cast<HunspellImpl*>(pHunspell)->generate(slst, word, desc, n);
}
/* functions for run-time modification of the dictionary */
/* add word to the run-time dictionary */
int Hunspell_add(Hunhandle* pHunspell, constchar* word) { returnreinterpret_cast<HunspellImpl*>(pHunspell)->add(word);
}
int Hunspell_add_with_flags(Hunhandle* pHunspell, constchar* word, constchar* flags, constchar* desc) { returnreinterpret_cast<HunspellImpl*>(pHunspell)->add_with_flags(word, flags, desc);
}
/* add word to the run-time dictionary with affix flags of *theexample(adictionaryword):Hunspellwillrecognize *affixedformsofthenewword,too.
*/
int Hunspell_add_with_affix(Hunhandle* pHunspell, constchar* word, constchar* example) { returnreinterpret_cast<HunspellImpl*>(pHunspell)->add_with_affix(word, example);
}
/* remove word from the run-time dictionary */
int Hunspell_remove(Hunhandle* pHunspell, constchar* word) { returnreinterpret_cast<HunspellImpl*>(pHunspell)->remove(word);
}
¤ Diese beiden folgenden Angebotsgruppen bietet das Unternehmen0.37Angebot
(Wie Sie bei der Firma Beratungs- und Dienstleistungen beauftragen können 2026-09-28)
¤
Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.
Bemerkung:
Die farbliche Syntaxdarstellung und die Messung sind noch experimentell.