Bug Summary

File:root/firefox-clang/obj-x86_64-pc-linux-gnu/extensions/spellcheck/hunspell/src/./../../../../../extensions/spellcheck/hunspell/src/affixmgr.cxx
Warning:line 4085, column 13
Value stored to 'numdefcpd' is never read

Annotated Source Code

Press '?' to see keyboard shortcuts

clang -cc1 -cc1 -triple x86_64-pc-linux-gnu -O2 -analyze -disable-free -clear-ast-before-backend -disable-llvm-verifier -discard-value-names -main-file-name Unified_cpp_hunspell_src0.cpp -analyzer-checker=core -analyzer-checker=apiModeling -analyzer-checker=unix -analyzer-checker=deadcode -analyzer-checker=cplusplus -analyzer-checker=security.insecureAPI.UncheckedReturn -analyzer-checker=security.insecureAPI.getpw -analyzer-checker=security.insecureAPI.gets -analyzer-checker=security.insecureAPI.mktemp -analyzer-checker=security.insecureAPI.mkstemp -analyzer-checker=security.insecureAPI.vfork -analyzer-checker=nullability.NullPassedToNonnull -analyzer-checker=nullability.NullReturnedFromNonnull -analyzer-output plist -w -setup-static-analyzer -analyzer-config-compatibility-mode=true -mrelocation-model pic -pic-level 2 -fhalf-no-semantic-interposition -mframe-pointer=all -relaxed-aliasing -ffp-contract=off -fno-rounding-math -mconstructor-aliases -funwind-tables=2 -target-cpu x86-64 -tune-cpu generic -debugger-tuning=gdb -fdebug-compilation-dir=/root/firefox-clang/obj-x86_64-pc-linux-gnu/extensions/spellcheck/hunspell/src -fcoverage-compilation-dir=/root/firefox-clang/obj-x86_64-pc-linux-gnu/extensions/spellcheck/hunspell/src -resource-dir /usr/lib/llvm-23/lib/clang/23 -include /root/firefox-clang/config/gcc_hidden.h -include /root/firefox-clang/obj-x86_64-pc-linux-gnu/mozilla-config.h -include hunspell_alloc_hooks.h -I /root/firefox-clang/obj-x86_64-pc-linux-gnu/dist/stl_wrappers -D _GLIBCXX_ASSERTIONS=1 -I /root/firefox-clang/obj-x86_64-pc-linux-gnu/dist/system_wrappers -U _FORTIFY_SOURCE -D _FORTIFY_SOURCE=2 -D DEBUG=1 -D HUNSPELL_STATIC -D MOZ_HAS_MOZGLUE -D MOZILLA_INTERNAL_API -D IMPL_LIBXUL -D MOZ_SUPPORT_LEAKCHECKING -D STATIC_EXPORTABLE_JS_API -I /root/firefox-clang/extensions/spellcheck/hunspell/src -I /root/firefox-clang/obj-x86_64-pc-linux-gnu/extensions/spellcheck/hunspell/src -I /root/firefox-clang/extensions/spellcheck/hunspell/glue -I /root/firefox-clang/obj-x86_64-pc-linux-gnu/ipc/ipdl/_ipdlheaders -I /root/firefox-clang/ipc/chromium/src -I /root/firefox-clang/third_party/abseil-cpp -I /root/firefox-clang/toolkit/components/telemetry -I /root/firefox-clang/xpcom/base -I /root/firefox-clang/obj-x86_64-pc-linux-gnu/dist/include -I /root/firefox-clang/obj-x86_64-pc-linux-gnu/dist/include/nspr -I /root/firefox-clang/obj-x86_64-pc-linux-gnu/dist/include/nss -D MOZILLA_CLIENT -internal-isystem /usr/lib/gcc/x86_64-linux-gnu/16/../../../../include/c++/16 -internal-isystem /usr/lib/gcc/x86_64-linux-gnu/16/../../../../include/x86_64-linux-gnu/c++/16 -internal-isystem /usr/lib/gcc/x86_64-linux-gnu/16/../../../../include/c++/16/backward -internal-isystem /usr/lib/llvm-23/lib/clang/23/include -internal-isystem /usr/local/include -internal-isystem /usr/lib/gcc/x86_64-linux-gnu/16/../../../../x86_64-linux-gnu/include -internal-externc-isystem /usr/include/x86_64-linux-gnu -internal-externc-isystem /include -internal-externc-isystem /usr/include -Wno-error=pessimizing-move -Wno-error=large-by-value-copy=128 -Wno-error=implicit-int-float-conversion -Wno-error=thread-safety-analysis -Wno-error=tautological-type-limit-compare -Wno-invalid-offsetof -Wno-range-loop-analysis -Wno-deprecated-anon-enum-enum-conversion -Wno-deprecated-enum-enum-conversion -Wno-inline-new-delete -Wno-error=deprecated-declarations -Wno-error=array-bounds -Wno-error=free-nonheap-object -Wno-error=atomic-alignment -Wno-error=deprecated-builtins -Wno-psabi -Wno-error=builtin-macro-redefined -Wno-vla-cxx-extension -Wno-unknown-warning-option -Wno-character-conversion -Wno-implicit-fallthrough -std=gnu++20 -fdeprecated-macro -ferror-limit 19 -fstrict-flex-arrays=1 -stack-protector 2 -fstack-clash-protection -ftrivial-auto-var-init=pattern -fno-rtti -fgnuc-version=4.2.1 -fno-implicit-modules -fskip-odr-check-in-gmf -fno-sized-deallocation -fno-aligned-allocation -fdiagnostics-absolute-paths -vectorize-loops -vectorize-slp -analyzer-checker optin.performance.Padding -analyzer-output=html -analyzer-config stable-report-filename=true -mllvm -dwarf-linkage-names=Abstract -faddrsig -fdwarf2-cfi-asm -o /tmp/scan-build-2026-09-01-224014-2642839-1 -x c++ Unified_cpp_hunspell_src0.cpp
1/* ***** BEGIN LICENSE BLOCK *****
2 * Version: MPL 1.1/GPL 2.0/LGPL 2.1
3 *
4 * Copyright (C) 2002-2022 Németh László
5 *
6 * The contents of this file are subject to the Mozilla Public License Version
7 * 1.1 (the "License"); you may not use this file except in compliance with
8 * the License. You may obtain a copy of the License at
9 * http://www.mozilla.org/MPL/
10 *
11 * Software distributed under the License is distributed on an "AS IS" basis,
12 * WITHOUT WARRANTY OF ANY KIND, either express or implied. See the License
13 * for the specific language governing rights and limitations under the
14 * License.
15 *
16 * Hunspell is based on MySpell which is Copyright (C) 2002 Kevin Hendricks.
17 *
18 * Contributor(s): David Einstein, Davide Prina, Giuseppe Modugno,
19 * Gianluca Turconi, Simon Brouwer, Noll János, Bíró Árpád,
20 * Goldman Eleonóra, Sarlós Tamás, Bencsáth Boldizsár, Halácsy Péter,
21 * Dvornik László, Gefferth András, Nagy Viktor, Varga Dániel, Chris Halls,
22 * Rene Engelhard, Bram Moolenaar, Dafydd Jones, Harri Pitkänen
23 *
24 * Alternatively, the contents of this file may be used under the terms of
25 * either the GNU General Public License Version 2 or later (the "GPL"), or
26 * the GNU Lesser General Public License Version 2.1 or later (the "LGPL"),
27 * in which case the provisions of the GPL or the LGPL are applicable instead
28 * of those above. If you wish to allow use of your version of this file only
29 * under the terms of either the GPL or the LGPL, and not to allow others to
30 * use your version of this file under the terms of the MPL, indicate your
31 * decision by deleting the provisions above and replace them with the notice
32 * and other provisions required by the GPL or the LGPL. If you do not delete
33 * the provisions above, a recipient may use your version of this file under
34 * the terms of any one of the MPL, the GPL or the LGPL.
35 *
36 * ***** END LICENSE BLOCK ***** */
37/*
38 * Copyright 2002 Kevin B. Hendricks, Stratford, Ontario, Canada
39 * And Contributors. All rights reserved.
40 *
41 * Redistribution and use in source and binary forms, with or without
42 * modification, are permitted provided that the following conditions
43 * are met:
44 *
45 * 1. Redistributions of source code must retain the above copyright
46 * notice, this list of conditions and the following disclaimer.
47 *
48 * 2. Redistributions in binary form must reproduce the above copyright
49 * notice, this list of conditions and the following disclaimer in the
50 * documentation and/or other materials provided with the distribution.
51 *
52 * 3. All modifications to the source code must be clearly marked as
53 * such. Binary redistributions based on modified source code
54 * must be clearly marked as modified versions in the documentation
55 * and/or other materials provided with the distribution.
56 *
57 * THIS SOFTWARE IS PROVIDED BY KEVIN B. HENDRICKS AND CONTRIBUTORS
58 * ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT
59 * LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS
60 * FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL
61 * KEVIN B. HENDRICKS OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT,
62 * INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING,
63 * BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES;
64 * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
65 * HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
66 * LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY
67 * OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
68 * SUCH DAMAGE.
69 */
70
71#include <cstdlib>
72#include <cstring>
73#include <cstdio>
74#include <cctype>
75#include <ctime>
76
77#include <algorithm>
78#include <chrono>
79#include <memory>
80#include <limits>
81#include <string>
82#include <vector>
83
84#include "affixmgr.hxx"
85#include "affentry.hxx"
86#include "langnum.hxx"
87
88#include "csutil.hxx"
89
90AffixMgr::AffixMgr(const char* affpath,
91 const std::vector<std::unique_ptr<HashMgr>>& ptr,
92 const char* key)
93 : alldic(ptr)
94 , pHMgr(ptr[0].get()) {
95
96 // register hash manager and load affix data from aff file
97 csconv = nullptr;
98 utf8 = 0;
99 complexprefixes = 0;
100 parsedmaptable = false;
101 parsedbreaktable = false;
102 iconvtable = nullptr;
103 oconvtable = nullptr;
104 // allow simplified compound forms (see 3rd field of CHECKCOMPOUNDPATTERN)
105 simplifiedcpd = 0;
106 parsedcheckcpd = false;
107 parseddefcpd = false;
108 phone = nullptr;
109 compoundflag = FLAG_NULL0x00; // permits word in compound forms
110 compoundbegin = FLAG_NULL0x00; // may be first word in compound forms
111 compoundmiddle = FLAG_NULL0x00; // may be middle word in compound forms
112 compoundend = FLAG_NULL0x00; // may be last word in compound forms
113 compoundroot = FLAG_NULL0x00; // compound word signing flag
114 compoundpermitflag = FLAG_NULL0x00; // compound permitting flag for suffixed word
115 compoundforbidflag = FLAG_NULL0x00; // compound fordidden flag for suffixed word
116 compoundmoresuffixes = 0; // allow more suffixes within compound words
117 checkcompounddup = 0; // forbid double words in compounds
118 checkcompoundrep = 0; // forbid bad compounds (may be non-compound word with
119 // a REP substitution)
120 checkcompoundcase =
121 0; // forbid upper and lowercase combinations at word bounds
122 checkcompoundtriple = 0; // forbid compounds with triple letters
123 simplifiedtriple = 0; // allow simplified triple letters in compounds
124 // (Schiff+fahrt -> Schiffahrt)
125 forbiddenword = FORBIDDENWORD65510; // forbidden word signing flag
126 nosuggest = FLAG_NULL0x00; // don't suggest words signed with NOSUGGEST flag
127 nongramsuggest = FLAG_NULL0x00;
128 langnum = 0; // language code (see http://l10n.openoffice.org/languages.html)
129 needaffix = FLAG_NULL0x00; // forbidden root, allowed only with suffixes
130 cpdwordmax = -1; // default: unlimited wordcount in compound words
131 cpdmin = -1; // undefined
132 cpdmaxsyllable = 0; // default: unlimited syllablecount in compound words
133 pfxappnd = nullptr; // previous prefix for counting syllables of the prefix BUG
134 sfxappnd = nullptr; // previous suffix for counting syllables of the suffix BUG
135 sfxextra = 0; // modifier for syllable count of sfxappnd BUG
136 checknum = 0; // checking numbers, and word with numbers
137 havecontclass = 0; // flags of possible continuing classes (double affix)
138 // LEMMA_PRESENT: not put root into the morphological output. Lemma presents
139 // in morhological description in dictionary file. It's often combined with
140 // PSEUDOROOT.
141 lemma_present = FLAG_NULL0x00;
142 circumfix = FLAG_NULL0x00;
143 onlyincompound = FLAG_NULL0x00;
144 maxngramsugs = -1; // undefined
145 maxdiff = -1; // undefined
146 onlymaxdiff = 0;
147 maxcpdsugs = -1; // undefined
148 nosplitsugs = 0;
149 sugswithdots = 0;
150 keepcase = 0;
151 forceucase = 0;
152 warn = 0;
153 forbidwarn = 0;
154 checksharps = 0;
155 substandard = FLAG_NULL0x00;
156 fullstrip = 0;
157
158 sfx = nullptr;
159 pfx = nullptr;
160
161 for (int i = 0; i < SETSIZE256; i++) {
162 pStart[i] = nullptr;
163 sStart[i] = nullptr;
164 pFlag[i] = nullptr;
165 sFlag[i] = nullptr;
166 }
167
168 memset(contclasses, 0, CONTSIZE65536 * sizeof(char));
169
170 if (parse_file(affpath, key)) {
171 fprintf(stderrstderr, "Failure loading aff file %s\n", affpath);
172 }
173
174 /* get encoding for CHECKCOMPOUNDCASE */
175 if (!utf8) {
176 csconv = get_current_cs(get_encoding());
177 for (int i = 0; i <= 255; i++) {
178 if ((csconv[i].cupper != csconv[i].clower) &&
179 (wordchars.find((char)i) == std::string::npos)) {
180 wordchars.push_back((char)i);
181 }
182 }
183 }
184
185 // default BREAK definition
186 if (!parsedbreaktable) {
187 breaktable.emplace_back("-");
188 breaktable.emplace_back("^-");
189 breaktable.emplace_back("-$");
190 parsedbreaktable = true;
191 }
192
193#if defined(FUZZING_BUILD_MODE_UNSAFE_FOR_PRODUCTION)
194 // not entirely sure this is invalid, so only for fuzzing for now
195 if (iconvtable && !iconvtable->check_against_breaktable(breaktable)) {
196 delete iconvtable;
197 iconvtable = nullptr;
198 }
199#endif
200
201 if (cpdmin == -1)
202 cpdmin = MINCPDLEN3;
203}
204
205AffixMgr::~AffixMgr() {
206 // pass through linked prefix entries and clean up
207 for (int i = 0; i < SETSIZE256; i++) {
208 pFlag[i] = nullptr;
209 PfxEntry* ptr = pStart[i];
210 PfxEntry* nptr = nullptr;
211 while (ptr) {
212 nptr = ptr->getNext();
213 delete (ptr);
214 ptr = nptr;
215 }
216 }
217
218 // pass through linked suffix entries and clean up
219 for (int j = 0; j < SETSIZE256; j++) {
220 sFlag[j] = nullptr;
221 SfxEntry* ptr = sStart[j];
222 SfxEntry* nptr = nullptr;
223 while (ptr) {
224 nptr = ptr->getNext();
225 delete (ptr);
226 ptr = nptr;
227 }
228 sStart[j] = nullptr;
229 }
230
231 delete iconvtable;
232 delete oconvtable;
233 delete phone;
234
235 FREE_FLAG(compoundflag)compoundflag = 0;
236 FREE_FLAG(compoundbegin)compoundbegin = 0;
237 FREE_FLAG(compoundmiddle)compoundmiddle = 0;
238 FREE_FLAG(compoundend)compoundend = 0;
239 FREE_FLAG(compoundpermitflag)compoundpermitflag = 0;
240 FREE_FLAG(compoundforbidflag)compoundforbidflag = 0;
241 FREE_FLAG(compoundroot)compoundroot = 0;
242 FREE_FLAG(forbiddenword)forbiddenword = 0;
243 FREE_FLAG(nosuggest)nosuggest = 0;
244 FREE_FLAG(nongramsuggest)nongramsuggest = 0;
245 FREE_FLAG(needaffix)needaffix = 0;
246 FREE_FLAG(lemma_present)lemma_present = 0;
247 FREE_FLAG(circumfix)circumfix = 0;
248 FREE_FLAG(onlyincompound)onlyincompound = 0;
249
250 cpdwordmax = 0;
251 pHMgr = nullptr;
252 cpdmin = 0;
253 cpdmaxsyllable = 0;
254 checknum = 0;
255#ifdef MOZILLA_CLIENT1
256 delete[] csconv;
257#endif
258}
259
260void AffixMgr::finishFileMgr(FileMgr* afflst) {
261 delete afflst;
262
263 // convert affix trees to sorted list
264 process_pfx_tree_to_list();
265 process_sfx_tree_to_list();
266}
267
268// read in aff file and build up prefix and suffix entry objects
269int AffixMgr::parse_file(const char* affpath, const char* key) {
270
271 // checking flag duplication
272 char dupflags[CONTSIZE65536];
273 char dupflags_ini = 1;
274
275 // first line indicator for removing byte order mark
276 int firstline = 1;
277
278 // open the affix file
279 FileMgr* afflst = new FileMgr(affpath, key);
280
281 // step one is to parse the affix file building up the internal
282 // affix data structures
283
284 // read in each line ignoring any that do not
285 // start with a known line type indicator
286 std::string line;
287 while (afflst->getline(line)) {
288 mychomp(line);
289
290 /* remove byte order mark */
291 if (firstline) {
292 firstline = 0;
293 // Affix file begins with byte order mark: possible incompatibility with
294 // old Hunspell versions
295 if (line.compare(0, 3, "\xEF\xBB\xBF", 3) == 0) {
296 line.erase(0, 3);
297 }
298 }
299
300 /* parse in the keyboard string */
301 if (line.compare(0, 3, "KEY", 3) == 0) {
302 if (!parse_string(line, keystring, afflst->getlinenum())) {
303 finishFileMgr(afflst);
304 return 1;
305 }
306 }
307
308 /* parse in the try string */
309 if (line.compare(0, 3, "TRY", 3) == 0) {
310 if (!parse_string(line, trystring, afflst->getlinenum())) {
311 finishFileMgr(afflst);
312 return 1;
313 }
314 }
315
316 /* parse in the name of the character set used by the .dict and .aff */
317 if (line.compare(0, 3, "SET", 3) == 0) {
318 if (!parse_string(line, encoding, afflst->getlinenum())) {
319 finishFileMgr(afflst);
320 return 1;
321 }
322 if (encoding == "UTF-8") {
323 utf8 = 1;
324 }
325 }
326
327 /* parse COMPLEXPREFIXES for agglutinative languages with right-to-left
328 * writing system */
329 if (line.compare(0, 15, "COMPLEXPREFIXES", 15) == 0)
330 complexprefixes = 1;
331
332 /* parse in the flag used by the controlled compound words */
333 if (line.compare(0, 12, "COMPOUNDFLAG", 12) == 0) {
334 if (!parse_flag(line, &compoundflag, afflst)) {
335 finishFileMgr(afflst);
336 return 1;
337 }
338 }
339
340 /* parse in the flag used by compound words */
341 if (line.compare(0, 13, "COMPOUNDBEGIN", 13) == 0) {
342 if (complexprefixes) {
343 if (!parse_flag(line, &compoundend, afflst)) {
344 finishFileMgr(afflst);
345 return 1;
346 }
347 } else {
348 if (!parse_flag(line, &compoundbegin, afflst)) {
349 finishFileMgr(afflst);
350 return 1;
351 }
352 }
353 }
354
355 /* parse in the flag used by compound words */
356 if (line.compare(0, 14, "COMPOUNDMIDDLE", 14) == 0) {
357 if (!parse_flag(line, &compoundmiddle, afflst)) {
358 finishFileMgr(afflst);
359 return 1;
360 }
361 }
362
363 /* parse in the flag used by compound words */
364 if (line.compare(0, 11, "COMPOUNDEND", 11) == 0) {
365 if (complexprefixes) {
366 if (!parse_flag(line, &compoundbegin, afflst)) {
367 finishFileMgr(afflst);
368 return 1;
369 }
370 } else {
371 if (!parse_flag(line, &compoundend, afflst)) {
372 finishFileMgr(afflst);
373 return 1;
374 }
375 }
376 }
377
378 /* parse in the data used by compound_check() method */
379 if (line.compare(0, 15, "COMPOUNDWORDMAX", 15) == 0) {
380 if (!parse_num(line, &cpdwordmax, afflst)) {
381 finishFileMgr(afflst);
382 return 1;
383 }
384 }
385
386 /* parse in the flag sign compounds in dictionary */
387 if (line.compare(0, 12, "COMPOUNDROOT", 12) == 0) {
388 if (!parse_flag(line, &compoundroot, afflst)) {
389 finishFileMgr(afflst);
390 return 1;
391 }
392 }
393
394 /* parse in the flag used by compound_check() method */
395 if (line.compare(0, 18, "COMPOUNDPERMITFLAG", 18) == 0) {
396 if (!parse_flag(line, &compoundpermitflag, afflst)) {
397 finishFileMgr(afflst);
398 return 1;
399 }
400 }
401
402 /* parse in the flag used by compound_check() method */
403 if (line.compare(0, 18, "COMPOUNDFORBIDFLAG", 18) == 0) {
404 if (!parse_flag(line, &compoundforbidflag, afflst)) {
405 finishFileMgr(afflst);
406 return 1;
407 }
408 }
409
410 if (line.compare(0, 20, "COMPOUNDMORESUFFIXES", 20) == 0) {
411 compoundmoresuffixes = 1;
412 }
413
414 if (line.compare(0, 16, "CHECKCOMPOUNDDUP", 16) == 0) {
415 checkcompounddup = 1;
416 }
417
418 if (line.compare(0, 16, "CHECKCOMPOUNDREP", 16) == 0) {
419 checkcompoundrep = 1;
420 }
421
422 if (line.compare(0, 19, "CHECKCOMPOUNDTRIPLE", 19) == 0) {
423 checkcompoundtriple = 1;
424 }
425
426 if (line.compare(0, 16, "SIMPLIFIEDTRIPLE", 16) == 0) {
427 simplifiedtriple = 1;
428 }
429
430 if (line.compare(0, 17, "CHECKCOMPOUNDCASE", 17) == 0) {
431 checkcompoundcase = 1;
432 }
433
434 if (line.compare(0, 9, "NOSUGGEST", 9) == 0) {
435 if (!parse_flag(line, &nosuggest, afflst)) {
436 finishFileMgr(afflst);
437 return 1;
438 }
439 }
440
441 if (line.compare(0, 14, "NONGRAMSUGGEST", 14) == 0) {
442 if (!parse_flag(line, &nongramsuggest, afflst)) {
443 finishFileMgr(afflst);
444 return 1;
445 }
446 }
447
448 /* parse in the flag used by forbidden words */
449 if (line.compare(0, 13, "FORBIDDENWORD", 13) == 0) {
450 if (!parse_flag(line, &forbiddenword, afflst)) {
451 finishFileMgr(afflst);
452 return 1;
453 }
454 }
455
456 /* parse in the flag used by forbidden words (is deprecated) */
457 if (line.compare(0, 13, "LEMMA_PRESENT", 13) == 0) {
458 if (!parse_flag(line, &lemma_present, afflst)) {
459 finishFileMgr(afflst);
460 return 1;
461 }
462 }
463
464 /* parse in the flag used by circumfixes */
465 if (line.compare(0, 9, "CIRCUMFIX", 9) == 0) {
466 if (!parse_flag(line, &circumfix, afflst)) {
467 finishFileMgr(afflst);
468 return 1;
469 }
470 }
471
472 /* parse in the flag used by fogemorphemes */
473 if (line.compare(0, 14, "ONLYINCOMPOUND", 14) == 0) {
474 if (!parse_flag(line, &onlyincompound, afflst)) {
475 finishFileMgr(afflst);
476 return 1;
477 }
478 }
479
480 /* parse in the flag used by `needaffixs' (is deprecated) */
481 if (line.compare(0, 10, "PSEUDOROOT", 10) == 0) {
482 if (!parse_flag(line, &needaffix, afflst)) {
483 finishFileMgr(afflst);
484 return 1;
485 }
486 }
487
488 /* parse in the flag used by `needaffixs' */
489 if (line.compare(0, 9, "NEEDAFFIX", 9) == 0) {
490 if (!parse_flag(line, &needaffix, afflst)) {
491 finishFileMgr(afflst);
492 return 1;
493 }
494 }
495
496 /* parse in the minimal length for words in compounds */
497 if (line.compare(0, 11, "COMPOUNDMIN", 11) == 0) {
498 if (!parse_num(line, &cpdmin, afflst)) {
499 finishFileMgr(afflst);
500 return 1;
501 }
502 if (cpdmin < 1)
503 cpdmin = 1;
504 }
505
506 /* parse in the max. words and syllables in compounds */
507 if (line.compare(0, 16, "COMPOUNDSYLLABLE", 16) == 0) {
508 if (!parse_cpdsyllable(line, afflst)) {
509 finishFileMgr(afflst);
510 return 1;
511 }
512 }
513
514 /* parse in the flag used by compound_check() method */
515 if (line.compare(0, 11, "SYLLABLENUM", 11) == 0) {
516 if (!parse_string(line, cpdsyllablenum, afflst->getlinenum())) {
517 finishFileMgr(afflst);
518 return 1;
519 }
520 }
521
522 /* parse in the flag used by the controlled compound words */
523 if (line.compare(0, 8, "CHECKNUM", 8) == 0) {
524 checknum = 1;
525 }
526
527 /* parse in the extra word characters */
528 if (line.compare(0, 9, "WORDCHARS", 9) == 0) {
529 if (!parse_array(line, wordchars, wordchars_utf16,
530 utf8, afflst->getlinenum())) {
531 finishFileMgr(afflst);
532 return 1;
533 }
534 }
535
536 /* parse in the ignored characters (for example, Arabic optional diacretics
537 * charachters */
538 if (line.compare(0, 6, "IGNORE", 6) == 0) {
539 if (!parse_array(line, ignorechars, ignorechars_utf16,
540 utf8, afflst->getlinenum())) {
541 finishFileMgr(afflst);
542 return 1;
543 }
544 }
545
546 /* parse in the input conversion table */
547 if (line.compare(0, 5, "ICONV", 5) == 0) {
548 if (!parse_convtable(line, afflst, &iconvtable, "ICONV")) {
549 finishFileMgr(afflst);
550 return 1;
551 }
552 }
553
554 /* parse in the output conversion table */
555 if (line.compare(0, 5, "OCONV", 5) == 0) {
556 if (!parse_convtable(line, afflst, &oconvtable, "OCONV")) {
557 finishFileMgr(afflst);
558 return 1;
559 }
560 }
561
562 /* parse in the phonetic translation table */
563 if (line.compare(0, 5, "PHONE", 5) == 0) {
564 if (!parse_phonetable(line, afflst)) {
565 finishFileMgr(afflst);
566 return 1;
567 }
568 }
569
570 /* parse in the checkcompoundpattern table */
571 if (line.compare(0, 20, "CHECKCOMPOUNDPATTERN", 20) == 0) {
572 if (!parse_checkcpdtable(line, afflst)) {
573 finishFileMgr(afflst);
574 return 1;
575 }
576 }
577
578 /* parse in the defcompound table */
579 if (line.compare(0, 12, "COMPOUNDRULE", 12) == 0) {
580 if (!parse_defcpdtable(line, afflst)) {
581 finishFileMgr(afflst);
582 return 1;
583 }
584 }
585
586 /* parse in the related character map table */
587 if (line.compare(0, 3, "MAP", 3) == 0) {
588 if (!parse_maptable(line, afflst)) {
589 finishFileMgr(afflst);
590 return 1;
591 }
592 }
593
594 /* parse in the word breakpoints table */
595 if (line.compare(0, 5, "BREAK", 5) == 0) {
596 if (!parse_breaktable(line, afflst)) {
597 finishFileMgr(afflst);
598 return 1;
599 }
600 }
601
602 /* parse in the language for language specific codes */
603 if (line.compare(0, 4, "LANG", 4) == 0) {
604 if (!parse_string(line, lang, afflst->getlinenum())) {
605 finishFileMgr(afflst);
606 return 1;
607 }
608 langnum = get_lang_num(lang);
609 }
610
611 if (line.compare(0, 7, "VERSION", 7) == 0) {
612 size_t startpos = line.find_first_not_of(" \t", 7);
613 if (startpos != std::string::npos) {
614 version = line.substr(startpos);
615 }
616 }
617
618 if (line.compare(0, 12, "MAXNGRAMSUGS", 12) == 0) {
619 if (!parse_num(line, &maxngramsugs, afflst)) {
620 finishFileMgr(afflst);
621 return 1;
622 }
623 }
624
625 if (line.compare(0, 11, "ONLYMAXDIFF", 11) == 0)
626 onlymaxdiff = 1;
627
628 if (line.compare(0, 7, "MAXDIFF", 7) == 0) {
629 if (!parse_num(line, &maxdiff, afflst)) {
630 finishFileMgr(afflst);
631 return 1;
632 }
633 }
634
635 if (line.compare(0, 10, "MAXCPDSUGS", 10) == 0) {
636 if (!parse_num(line, &maxcpdsugs, afflst)) {
637 finishFileMgr(afflst);
638 return 1;
639 }
640 }
641
642 if (line.compare(0, 11, "NOSPLITSUGS", 11) == 0) {
643 nosplitsugs = 1;
644 }
645
646 if (line.compare(0, 9, "FULLSTRIP", 9) == 0) {
647 fullstrip = 1;
648 }
649
650 if (line.compare(0, 12, "SUGSWITHDOTS", 12) == 0) {
651 sugswithdots = 1;
652 }
653
654 /* parse in the flag used by forbidden words */
655 if (line.compare(0, 8, "KEEPCASE", 8) == 0) {
656 if (!parse_flag(line, &keepcase, afflst)) {
657 finishFileMgr(afflst);
658 return 1;
659 }
660 }
661
662 /* parse in the flag used by `forceucase' */
663 if (line.compare(0, 10, "FORCEUCASE", 10) == 0) {
664 if (!parse_flag(line, &forceucase, afflst)) {
665 finishFileMgr(afflst);
666 return 1;
667 }
668 }
669
670 /* parse in the flag used by `warn' */
671 if (line.compare(0, 4, "WARN", 4) == 0) {
672 if (!parse_flag(line, &warn, afflst)) {
673 finishFileMgr(afflst);
674 return 1;
675 }
676 }
677
678 if (line.compare(0, 10, "FORBIDWARN", 10) == 0) {
679 forbidwarn = 1;
680 }
681
682 /* parse in the flag used by the affix generator */
683 if (line.compare(0, 11, "SUBSTANDARD", 11) == 0) {
684 if (!parse_flag(line, &substandard, afflst)) {
685 finishFileMgr(afflst);
686 return 1;
687 }
688 }
689
690 if (line.compare(0, 11, "CHECKSHARPS", 11) == 0) {
691 checksharps = 1;
692 }
693
694 /* parse this affix: P - prefix, S - suffix */
695 // affix type
696 char ft = ' ';
697 if (line.compare(0, 3, "PFX", 3) == 0)
698 ft = complexprefixes ? 'S' : 'P';
699 if (line.compare(0, 3, "SFX", 3) == 0)
700 ft = complexprefixes ? 'P' : 'S';
701 if (ft != ' ') {
702 if (dupflags_ini) {
703 memset(dupflags, 0, sizeof(dupflags));
704 dupflags_ini = 0;
705 }
706 if (!parse_affix(line, ft, afflst, dupflags)) {
707 finishFileMgr(afflst);
708 return 1;
709 }
710 }
711 }
712
713 finishFileMgr(afflst);
714 // affix trees are sorted now
715
716 // now we can speed up performance greatly taking advantage of the
717 // relationship between the affixes and the idea of "subsets".
718
719 // View each prefix as a potential leading subset of another and view
720 // each suffix (reversed) as a potential trailing subset of another.
721
722 // To illustrate this relationship if we know the prefix "ab" is found in the
723 // word to examine, only prefixes that "ab" is a leading subset of need be
724 // examined.
725 // Furthermore is "ab" is not present then none of the prefixes that "ab" is
726 // is a subset need be examined.
727 // The same argument goes for suffix string that are reversed.
728
729 // Then to top this off why not examine the first char of the word to quickly
730 // limit the set of prefixes to examine (i.e. the prefixes to examine must
731 // be leading supersets of the first character of the word (if they exist)
732
733 // To take advantage of this "subset" relationship, we need to add two links
734 // from entry. One to take next if the current prefix is found (call it
735 // nexteq)
736 // and one to take next if the current prefix is not found (call it nextne).
737
738 // Since we have built ordered lists, all that remains is to properly
739 // initialize
740 // the nextne and nexteq pointers that relate them
741
742 process_pfx_order();
743 process_sfx_order();
744
745 return 0;
746}
747
748// we want to be able to quickly access prefix information
749// both by prefix flag, and sorted by prefix string itself
750// so we need to set up two indexes
751
752int AffixMgr::build_pfxtree(PfxEntry* pfxptr) {
753 PfxEntry* ptr;
754 PfxEntry* pptr;
755 PfxEntry* ep = pfxptr;
756
757 // get the right starting points
758 const char* key = ep->getKey();
759 const auto flg = (unsigned char)(ep->getFlag() & 0x00FF);
760
761 // first index by flag which must exist
762 ptr = pFlag[flg];
763 ep->setFlgNxt(ptr);
764 pFlag[flg] = ep;
765
766 // handle the special case of null affix string
767 if (*key == '\0') {
768 // always inset them at head of list at element 0
769 ptr = pStart[0];
770 ep->setNext(ptr);
771 pStart[0] = ep;
772 return 0;
773 }
774
775 // now handle the normal case
776 ep->setNextEQ(nullptr);
777 ep->setNextNE(nullptr);
778
779 unsigned char sp = *((const unsigned char*)key);
780 ptr = pStart[sp];
781
782 // handle the first insert
783 if (!ptr) {
784 pStart[sp] = ep;
785 return 0;
786 }
787
788 // otherwise use binary tree insertion so that a sorted
789 // list can easily be generated later
790 pptr = nullptr;
791 for (;;) {
792 pptr = ptr;
793 if (strcmp(ep->getKey(), ptr->getKey()) <= 0) {
794 ptr = ptr->getNextEQ();
795 if (!ptr) {
796 pptr->setNextEQ(ep);
797 break;
798 }
799 } else {
800 ptr = ptr->getNextNE();
801 if (!ptr) {
802 pptr->setNextNE(ep);
803 break;
804 }
805 }
806 }
807 return 0;
808}
809
810// we want to be able to quickly access suffix information
811// both by suffix flag, and sorted by the reverse of the
812// suffix string itself; so we need to set up two indexes
813int AffixMgr::build_sfxtree(SfxEntry* sfxptr) {
814
815 sfxptr->initReverseWord();
816
817 SfxEntry* ptr;
818 SfxEntry* pptr;
819 SfxEntry* ep = sfxptr;
820
821 /* get the right starting point */
822 const char* key = ep->getKey();
823 const auto flg = (unsigned char)(ep->getFlag() & 0x00FF);
824
825 // first index by flag which must exist
826 ptr = sFlag[flg];
827 ep->setFlgNxt(ptr);
828 sFlag[flg] = ep;
829
830 // next index by affix string
831
832 // handle the special case of null affix string
833 if (*key == '\0') {
834 // always inset them at head of list at element 0
835 ptr = sStart[0];
836 ep->setNext(ptr);
837 sStart[0] = ep;
838 return 0;
839 }
840
841 // now handle the normal case
842 ep->setNextEQ(nullptr);
843 ep->setNextNE(nullptr);
844
845 unsigned char sp = *((const unsigned char*)key);
846 ptr = sStart[sp];
847
848 // handle the first insert
849 if (!ptr) {
850 sStart[sp] = ep;
851 return 0;
852 }
853
854 // otherwise use binary tree insertion so that a sorted
855 // list can easily be generated later
856 pptr = nullptr;
857 for (;;) {
858 pptr = ptr;
859 if (strcmp(ep->getKey(), ptr->getKey()) <= 0) {
860 ptr = ptr->getNextEQ();
861 if (!ptr) {
862 pptr->setNextEQ(ep);
863 break;
864 }
865 } else {
866 ptr = ptr->getNextNE();
867 if (!ptr) {
868 pptr->setNextNE(ep);
869 break;
870 }
871 }
872 }
873 return 0;
874}
875
876// convert from binary tree to sorted list
877int AffixMgr::process_pfx_tree_to_list() {
878 for (int i = 1; i < SETSIZE256; i++) {
879 pStart[i] = process_pfx_in_order(pStart[i], nullptr);
880 }
881 return 0;
882}
883
884PfxEntry* AffixMgr::process_pfx_in_order(PfxEntry* ptr, PfxEntry* nptr) {
885 if (ptr) {
886 nptr = process_pfx_in_order(ptr->getNextNE(), nptr);
887 ptr->setNext(nptr);
888 nptr = process_pfx_in_order(ptr->getNextEQ(), ptr);
889 }
890 return nptr;
891}
892
893// convert from binary tree to sorted list
894int AffixMgr::process_sfx_tree_to_list() {
895 for (int i = 1; i < SETSIZE256; i++) {
896 sStart[i] = process_sfx_in_order(sStart[i], nullptr);
897 }
898 return 0;
899}
900
901SfxEntry* AffixMgr::process_sfx_in_order(SfxEntry* ptr, SfxEntry* nptr) {
902 if (ptr) {
903 nptr = process_sfx_in_order(ptr->getNextNE(), nptr);
904 ptr->setNext(nptr);
905 nptr = process_sfx_in_order(ptr->getNextEQ(), ptr);
906 }
907 return nptr;
908}
909
910// reinitialize the PfxEntry links NextEQ and NextNE to speed searching
911// using the idea of leading subsets this time
912int AffixMgr::process_pfx_order() {
913 PfxEntry* ptr;
914
915 // loop through each prefix list starting point
916 for (int i = 1; i < SETSIZE256; i++) {
917 ptr = pStart[i];
918
919 // look through the remainder of the list
920 // and find next entry with affix that
921 // the current one is not a subset of
922 // mark that as destination for NextNE
923 // use next in list that you are a subset
924 // of as NextEQ
925
926 for (; ptr != nullptr; ptr = ptr->getNext()) {
927 PfxEntry* nptr = ptr->getNext();
928 for (; nptr != nullptr; nptr = nptr->getNext()) {
929 if (!isSubset(ptr->getKey(), nptr->getKey()))
930 break;
931 }
932 ptr->setNextNE(nptr);
933 ptr->setNextEQ(nullptr);
934 if ((ptr->getNext()) &&
935 isSubset(ptr->getKey(), (ptr->getNext())->getKey()))
936 ptr->setNextEQ(ptr->getNext());
937 }
938
939 // now clean up by adding smart search termination strings:
940 // if you are already a superset of the previous prefix
941 // but not a subset of the next, search can end here
942 // so set NextNE properly
943
944 ptr = pStart[i];
945 for (; ptr != nullptr; ptr = ptr->getNext()) {
946 PfxEntry* nptr = ptr->getNext();
947 PfxEntry* mptr = nullptr;
948 for (; nptr != nullptr; nptr = nptr->getNext()) {
949 if (!isSubset(ptr->getKey(), nptr->getKey()))
950 break;
951 mptr = nptr;
952 }
953 if (mptr)
954 mptr->setNextNE(nullptr);
955 }
956 }
957 return 0;
958}
959
960// initialize the SfxEntry links NextEQ and NextNE to speed searching
961// using the idea of leading subsets this time
962int AffixMgr::process_sfx_order() {
963 SfxEntry* ptr;
964
965 // loop through each prefix list starting point
966 for (int i = 1; i < SETSIZE256; i++) {
967 ptr = sStart[i];
968
969 // look through the remainder of the list
970 // and find next entry with affix that
971 // the current one is not a subset of
972 // mark that as destination for NextNE
973 // use next in list that you are a subset
974 // of as NextEQ
975
976 for (; ptr != nullptr; ptr = ptr->getNext()) {
977 SfxEntry* nptr = ptr->getNext();
978 for (; nptr != nullptr; nptr = nptr->getNext()) {
979 if (!isSubset(ptr->getKey(), nptr->getKey()))
980 break;
981 }
982 ptr->setNextNE(nptr);
983 ptr->setNextEQ(nullptr);
984 if ((ptr->getNext()) &&
985 isSubset(ptr->getKey(), (ptr->getNext())->getKey()))
986 ptr->setNextEQ(ptr->getNext());
987 }
988
989 // now clean up by adding smart search termination strings:
990 // if you are already a superset of the previous suffix
991 // but not a subset of the next, search can end here
992 // so set NextNE properly
993
994 ptr = sStart[i];
995 for (; ptr != nullptr; ptr = ptr->getNext()) {
996 SfxEntry* nptr = ptr->getNext();
997 SfxEntry* mptr = nullptr;
998 for (; nptr != nullptr; nptr = nptr->getNext()) {
999 if (!isSubset(ptr->getKey(), nptr->getKey()))
1000 break;
1001 mptr = nptr;
1002 }
1003 if (mptr)
1004 mptr->setNextNE(nullptr);
1005 }
1006 }
1007 return 0;
1008}
1009
1010// add flags to the result for dictionary debugging
1011std::string& AffixMgr::debugflag(std::string& result, unsigned short flag) {
1012 std::string st = encode_flag(flag);
1013 result.push_back(MSEP_FLD' ');
1014 result.append(MORPH_FLAG"fl:");
1015 result.append(st);
1016 return result;
1017}
1018
1019// calculate the character length of the condition
1020int AffixMgr::condlen(const std::string& s) {
1021 int l = 0;
1022 bool group = false;
1023 auto st = s.begin(), end = s.end();
1024 while (st != end) {
1025 if (*st == '[') {
1026 group = true;
1027 l++;
1028 } else if (*st == ']')
1029 group = false;
1030 else if (!group && (!utf8 || (!(*st & 0x80) || ((*st & 0xc0) == 0x80))))
1031 l++;
1032 ++st;
1033 }
1034 return l;
1035}
1036
1037int AffixMgr::encodeit(AffEntry& entry, const std::string& cs) {
1038 if (cs.compare(".") != 0) {
1039 entry.numconds = (char)condlen(cs);
1040 const size_t cslen = cs.size();
1041 const size_t short_part = std::min<size_t>(MAXCONDLEN20, cslen);
1042 memcpy(entry.c.conds, cs.data(), short_part);
1043 if (short_part < MAXCONDLEN20) {
1044 //blank out the remaining space
1045 memset(entry.c.conds + short_part, 0, MAXCONDLEN20 - short_part);
1046 } else if (cs[MAXCONDLEN20]) {
1047 //there is more conditions than fit in fixed space, so its
1048 //a long condition
1049 entry.opts |= aeLONGCOND(1 << 4);
1050 size_t remaining = cs.size() - MAXCONDLEN_1(20 - sizeof(char*));
1051 entry.c.l.conds2 = new char[1 + remaining];
1052 memcpy(entry.c.l.conds2, cs.data() + MAXCONDLEN_1(20 - sizeof(char*)), remaining);
1053 entry.c.l.conds2[remaining] = 0;
1054 }
1055 } else {
1056 entry.numconds = 0;
1057 entry.c.conds[0] = '\0';
1058 }
1059 return 0;
1060}
1061
1062// return 1 if s1 is a leading subset of s2 (dots are for infixes)
1063inline int AffixMgr::isSubset(const char* s1, const char* s2) {
1064 while (((*s1 == *s2) || (*s1 == '.')) && (*s1 != '\0') && (*s2 != '\0')) {
1065 s1++;
1066 s2++;
1067 }
1068 return (*s1 == '\0');
1069}
1070
1071// check word for prefixes
1072struct hentry* AffixMgr::prefix_check(const std::string& word,
1073 int start,
1074 int len,
1075 char in_compound,
1076 const FLAGunsigned short needflag) {
1077 struct hentry* rv = nullptr;
1078
1079 pfx = nullptr;
1080 pfxappnd = nullptr;
1081 sfxappnd = nullptr;
1082 sfxextra = 0;
1083
1084 // first handle the special case of 0 length prefixes
1085 PfxEntry* pe = pStart[0];
1086 while (pe) {
1087 if (
1088 // fogemorpheme
1089 ((in_compound != IN_CPD_NOT0) ||
1090 !(pe->getCont() &&
1091 (TESTAFF(pe->getCont(), onlyincompound, pe->getContLen())(std::binary_search(pe->getCont(), pe->getCont() + pe->
getContLen(), onlyincompound))
))) &&
1092 // permit prefixes in compounds
1093 ((in_compound != IN_CPD_END2) ||
1094 (pe->getCont() &&
1095 (TESTAFF(pe->getCont(), compoundpermitflag, pe->getContLen())(std::binary_search(pe->getCont(), pe->getCont() + pe->
getContLen(), compoundpermitflag))
)))) {
1096 // check prefix
1097 rv = pe->checkword(word, start, len, in_compound, needflag);
1098 if (rv) {
1099 pfx = pe; // BUG: pfx not stateless
1100 return rv;
1101 }
1102 }
1103 pe = pe->getNext();
1104 }
1105
1106 // now handle the general case
1107 unsigned char sp = word[start];
1108 PfxEntry* pptr = pStart[sp];
1109
1110 while (pptr) {
1111 if (isSubset(pptr->getKey(), word.c_str() + start)) {
1112 if (
1113 // fogemorpheme
1114 ((in_compound != IN_CPD_NOT0) ||
1115 !(pptr->getCont() &&
1116 (TESTAFF(pptr->getCont(), onlyincompound, pptr->getContLen())(std::binary_search(pptr->getCont(), pptr->getCont() + pptr
->getContLen(), onlyincompound))
))) &&
1117 // permit prefixes in compounds
1118 ((in_compound != IN_CPD_END2) ||
1119 (pptr->getCont() && (TESTAFF(pptr->getCont(), compoundpermitflag,(std::binary_search(pptr->getCont(), pptr->getCont() + pptr
->getContLen(), compoundpermitflag))
1120 pptr->getContLen())(std::binary_search(pptr->getCont(), pptr->getCont() + pptr
->getContLen(), compoundpermitflag))
)))) {
1121 // check prefix
1122 rv = pptr->checkword(word, start, len, in_compound, needflag);
1123 if (rv) {
1124 pfx = pptr; // BUG: pfx not stateless
1125 return rv;
1126 }
1127 }
1128 pptr = pptr->getNextEQ();
1129 } else {
1130 pptr = pptr->getNextNE();
1131 }
1132 }
1133
1134 return nullptr;
1135}
1136
1137// check word for prefixes and two-level suffixes
1138struct hentry* AffixMgr::prefix_check_twosfx(const std::string& word,
1139 int start,
1140 int len,
1141 char in_compound,
1142 const FLAGunsigned short needflag) {
1143 struct hentry* rv = nullptr;
1144
1145 pfx = nullptr;
1146 sfxappnd = nullptr;
1147 sfxextra = 0;
1148
1149 // first handle the special case of 0 length prefixes
1150 PfxEntry* pe = pStart[0];
1151
1152 while (pe) {
1153 rv = pe->check_twosfx(word, start, len, in_compound, needflag);
1154 if (rv)
1155 return rv;
1156 pe = pe->getNext();
1157 }
1158
1159 // now handle the general case
1160 unsigned char sp = word[start];
1161 PfxEntry* pptr = pStart[sp];
1162
1163 while (pptr) {
1164 if (isSubset(pptr->getKey(), word.c_str() + start)) {
1165 rv = pptr->check_twosfx(word, start, len, in_compound, needflag);
1166 if (rv) {
1167 pfx = pptr;
1168 return rv;
1169 }
1170 pptr = pptr->getNextEQ();
1171 } else {
1172 pptr = pptr->getNextNE();
1173 }
1174 }
1175
1176 return nullptr;
1177}
1178
1179// check word for prefixes and morph
1180std::string AffixMgr::prefix_check_morph(const std::string& word,
1181 int start,
1182 int len,
1183 char in_compound,
1184 const FLAGunsigned short needflag) {
1185
1186 std::string result;
1187
1188 pfx = nullptr;
1189 sfxappnd = nullptr;
1190 sfxextra = 0;
1191
1192 // first handle the special case of 0 length prefixes
1193 PfxEntry* pe = pStart[0];
1194 while (pe) {
1195 std::string st = pe->check_morph(word, start, len, in_compound, needflag);
1196 if (!st.empty()) {
1197 result.append(st);
1198 }
1199 pe = pe->getNext();
1200 }
1201
1202 // now handle the general case
1203 unsigned char sp = word[start];
1204 PfxEntry* pptr = pStart[sp];
1205
1206 while (pptr) {
1207 if (isSubset(pptr->getKey(), word.c_str() + start)) {
1208 std::string st = pptr->check_morph(word, start, len, in_compound, needflag);
1209 if (!st.empty()) {
1210 // fogemorpheme
1211 if ((in_compound != IN_CPD_NOT0) ||
1212 !((pptr->getCont() && (TESTAFF(pptr->getCont(), onlyincompound,(std::binary_search(pptr->getCont(), pptr->getCont() + pptr
->getContLen(), onlyincompound))
1213 pptr->getContLen())(std::binary_search(pptr->getCont(), pptr->getCont() + pptr
->getContLen(), onlyincompound))
)))) {
1214 result.append(st);
1215 pfx = pptr;
1216 }
1217 }
1218 pptr = pptr->getNextEQ();
1219 } else {
1220 pptr = pptr->getNextNE();
1221 }
1222 }
1223
1224 return result;
1225}
1226
1227// check word for prefixes and morph and two-level suffixes
1228std::string AffixMgr::prefix_check_twosfx_morph(const std::string& word,
1229 int start,
1230 int len,
1231 char in_compound,
1232 const FLAGunsigned short needflag) {
1233 std::string result;
1234
1235 pfx = nullptr;
1236 sfxappnd = nullptr;
1237 sfxextra = 0;
1238
1239 // first handle the special case of 0 length prefixes
1240 PfxEntry* pe = pStart[0];
1241 while (pe) {
1242 std::string st = pe->check_twosfx_morph(word, start, len, in_compound, needflag);
1243 if (!st.empty()) {
1244 result.append(st);
1245 }
1246 pe = pe->getNext();
1247 }
1248
1249 // now handle the general case
1250 unsigned char sp = word[start];
1251 PfxEntry* pptr = pStart[sp];
1252
1253 while (pptr) {
1254 if (isSubset(pptr->getKey(), word.c_str() + start)) {
1255 std::string st = pptr->check_twosfx_morph(word, start, len, in_compound, needflag);
1256 if (!st.empty()) {
1257 result.append(st);
1258 pfx = pptr;
1259 }
1260 pptr = pptr->getNextEQ();
1261 } else {
1262 pptr = pptr->getNextNE();
1263 }
1264 }
1265
1266 return result;
1267}
1268
1269// Is word a non-compound with a REP substitution (see checkcompoundrep)?
1270int AffixMgr::cpdrep_check(const std::string& in_word, int wl) {
1271
1272 if ((wl < 2) || get_reptable().empty())
1273 return 0;
1274
1275 std::string word(in_word, 0, wl);
1276
1277 for (const auto& i : get_reptable()) {
1278 // use only available mid patterns
1279 if (!i.outstrings[0].empty()) {
1280 size_t r = 0;
1281 const size_t lenp = i.pattern.size();
1282 // search every occurence of the pattern in the word
1283 while ((r = word.find(i.pattern, r)) != std::string::npos) {
1284 std::string candidate(word);
1285 candidate.replace(r, lenp, i.outstrings[0]);
1286 if (candidate_check(candidate))
1287 return 1;
1288 ++r; // search for the next letter
1289 }
1290 }
1291 }
1292
1293 return 0;
1294}
1295
1296// forbid compound words, if they are in the dictionary as a
1297// word pair separated by space
1298int AffixMgr::cpdwordpair_check(const std::string& word, int wl) {
1299 if (wl > 2) {
1300 std::string candidate(word, 0, wl);
1301 for (size_t i = 1; i < candidate.size(); i++) {
1302 // go to end of the UTF-8 character
1303 if (utf8 && ((candidate[i] & 0xc0) == 0x80))
1304 continue;
1305 candidate.insert(i, 1, ' ');
1306 if (candidate_check(candidate))
1307 return 1;
1308 candidate.erase(i, 1);
1309 }
1310 }
1311
1312 return 0;
1313}
1314
1315// forbid compoundings when there are special patterns at word bound
1316int AffixMgr::cpdpat_check(const std::string& word,
1317 size_t pos,
1318 hentry* r1,
1319 hentry* r2,
1320 const char /*affixed*/) {
1321 for (auto& i : checkcpdtable) {
1322 size_t len;
1323 if (isSubset(i.pattern2.c_str(), word.c_str() + pos) &&
1324 (!r1 || !i.cond ||
1325 (r1->astr && TESTAFF(r1->astr, i.cond, r1->alen)(std::binary_search(r1->astr, r1->astr + r1->alen, i
.cond))
)) &&
1326 (!r2 || !i.cond2 ||
1327 (r2->astr && TESTAFF(r2->astr, i.cond2, r2->alen)(std::binary_search(r2->astr, r2->astr + r2->alen, i
.cond2))
)) &&
1328 // zero length pattern => only TESTAFF
1329 // zero pattern (0/flag) => unmodified stem (zero affixes allowed)
1330 (i.pattern.empty() ||
1331 ((i.pattern[0] == '0' && r1->blen <= pos &&
1332 strncmp(word.c_str() + pos - r1->blen, r1->word, r1->blen) == 0) ||
1333 (i.pattern[0] != '0' &&
1334 ((len = i.pattern.size()) != 0) && len <= pos &&
1335 strncmp(word.c_str() + pos - len, i.pattern.c_str(), len) == 0)))) {
1336 return 1;
1337 }
1338 }
1339 return 0;
1340}
1341
1342// forbid compounding with neighbouring upper and lower case characters at word
1343// bounds
1344int AffixMgr::cpdcase_check(const std::string& word, int pos) {
1345 if (utf8) {
1346 const char* p;
1347 const char* wordp = word.c_str();
1348 for (p = wordp + pos - 1; p > wordp && (*p & 0xc0) == 0x80; p--)
1349 ;
1350 std::string pair(p);
1351 std::vector<w_char> pair_u;
1352 u8_u16(pair_u, pair);
1353 unsigned short a = pair_u.size() > 1 ? (unsigned short)pair_u[1] : 0,
1354 b = !pair_u.empty() ? (unsigned short)pair_u[0] : 0;
1355 if (((unicodetoupper(a, langnum) == a && unicodetolower(a, langnum) != a) ||
1356 (unicodetoupper(b, langnum) == b && unicodetolower(b, langnum) != b)) &&
1357 (a != '-') && (b != '-'))
1358 return 1;
1359 } else {
1360 const unsigned char a = word[pos - 1], b = word[pos];
1361 if ((csconv[a].ccase || csconv[b].ccase) && (a != '-') && (b != '-'))
1362 return 1;
1363 }
1364 return 0;
1365}
1366
1367struct metachar_data {
1368 signed short btpp; // metacharacter (*, ?) position for backtracking
1369 signed short btwp; // word position for metacharacters
1370 int btnum; // number of matched characters in metacharacter
1371};
1372
1373// check compound patterns
1374int AffixMgr::defcpd_check(hentry*** words,
1375 short wnum,
1376 short maxwordnum,
1377 hentry* rv,
1378 hentry** def,
1379 char all) {
1380 int w = 0;
1381
1382 if (!*words) {
1383 w = 1;
1384 *words = def;
1385 }
1386
1387 if (!*words) {
1388 return 0;
1389 }
1390
1391 if (wnum >= maxwordnum) {
1392 if (w)
1393 *words = nullptr;
1394 return 0;
1395 }
1396
1397 std::vector<metachar_data> btinfo(1);
1398
1399 short bt = 0;
1400
1401 (*words)[wnum] = rv;
1402
1403 // has the last word COMPOUNDRULE flag?
1404 if (rv->alen == 0) {
1405 (*words)[wnum] = nullptr;
1406 if (w)
1407 *words = nullptr;
1408 return 0;
1409 }
1410 int ok = 0;
1411 for (auto& i : defcpdtable) {
1412 for (auto& j : i) {
1413 if (j != '*' && j != '?' &&
1414 TESTAFF(rv->astr, j, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, j
))
) {
1415 ok = 1;
1416 break;
1417 }
1418 }
1419 }
1420 if (ok == 0) {
1421 (*words)[wnum] = nullptr;
1422 if (w)
1423 *words = nullptr;
1424 return 0;
1425 }
1426
1427 for (auto& i : defcpdtable) {
1428 size_t pp = 0; // pattern position
1429 signed short wp = 0; // "words" position
1430 int ok2 = 1;
1431 ok = 1;
1432 do {
1433 while ((pp < i.size()) && (wp <= wnum)) {
1434 if (((pp + 1) < i.size()) &&
1435 ((i[pp + 1] == '*') ||
1436 (i[pp + 1] == '?'))) {
1437 int wend = (i[pp + 1] == '?') ? wp : wnum;
1438 ok2 = 1;
1439 pp += 2;
1440 btinfo[bt].btpp = pp;
1441 btinfo[bt].btwp = wp;
1442 while (wp <= wend) {
1443 if (!(*words)[wp] ||
1444 !(*words)[wp]->alen ||
1445 !TESTAFF((*words)[wp]->astr, i[pp - 2],(std::binary_search((*words)[wp]->astr, (*words)[wp]->astr
+ (*words)[wp]->alen, i[pp - 2]))
1446 (*words)[wp]->alen)(std::binary_search((*words)[wp]->astr, (*words)[wp]->astr
+ (*words)[wp]->alen, i[pp - 2]))
) {
1447 ok2 = 0;
1448 break;
1449 }
1450 wp++;
1451 }
1452 if (wp <= wnum)
1453 ok2 = 0;
1454 btinfo[bt].btnum = wp - btinfo[bt].btwp;
1455 if (btinfo[bt].btnum > 0) {
1456 ++bt;
1457 btinfo.resize(bt+1);
1458 }
1459 if (ok2)
1460 break;
1461 } else {
1462 ok2 = 1;
1463 if (!(*words)[wp] || !(*words)[wp]->alen ||
1464 !TESTAFF((*words)[wp]->astr, i[pp],(std::binary_search((*words)[wp]->astr, (*words)[wp]->astr
+ (*words)[wp]->alen, i[pp]))
1465 (*words)[wp]->alen)(std::binary_search((*words)[wp]->astr, (*words)[wp]->astr
+ (*words)[wp]->alen, i[pp]))
) {
1466 ok = 0;
1467 break;
1468 }
1469 pp++;
1470 wp++;
1471 if ((i.size() == pp) && !(wp > wnum))
1472 ok = 0;
1473 }
1474 }
1475 if (ok && ok2) {
1476 size_t r = pp;
1477 while ((i.size() > r) && ((r + 1) < i.size()) &&
1478 ((i[r + 1] == '*') ||
1479 (i[r + 1] == '?')))
1480 r += 2;
1481 if (i.size() <= r)
1482 return 1;
1483 }
1484 // backtrack
1485 if (bt)
1486 do {
1487 ok = 1;
1488 btinfo[bt - 1].btnum--;
1489 pp = btinfo[bt - 1].btpp;
1490 wp = btinfo[bt - 1].btwp + (signed short)btinfo[bt - 1].btnum;
1491 } while ((btinfo[bt - 1].btnum < 0) && --bt);
1492 } while (bt);
1493
1494 if (ok && ok2 && (!all || (i.size() <= pp)))
1495 return 1;
1496
1497 // check zero ending
1498 while (ok && ok2 && (i.size() > pp) &&
1499 ((pp + 1) < i.size()) &&
1500 ((i[pp + 1] == '*') ||
1501 (i[pp + 1] == '?')))
1502 pp += 2;
1503 if (ok && ok2 && (i.size() <= pp))
1504 return 1;
1505 }
1506 (*words)[wnum] = nullptr;
1507 if (w)
1508 *words = nullptr;
1509 return 0;
1510}
1511
1512inline int AffixMgr::candidate_check(const std::string& word) {
1513
1514 struct hentry* rv = lookup(word.c_str(), word.size());
1515 if (rv)
1516 return 1;
1517
1518 // rv = prefix_check(word,0,len,1);
1519 // if (rv) return 1;
1520
1521 rv = affix_check(word, 0, word.size());
1522 if (rv)
1523 return 1;
1524 return 0;
1525}
1526
1527// calculate number of syllable for compound-checking
1528short AffixMgr::get_syllable(const std::string& word) {
1529 if (cpdmaxsyllable == 0)
1530 return 0;
1531
1532 short num = 0;
1533
1534 if (!utf8) {
1535 num = (short)std::count_if(word.begin(), word.end(),
1536 [&](char c) {
1537 return std::binary_search(cpdvowels.begin(), cpdvowels.end(), c);
1538 });
1539 } else if (!cpdvowels_utf16.empty()) {
1540 std::vector<w_char> w;
1541 u8_u16(w, word);
1542 num = (short)std::count_if(w.begin(), w.end(),
1543 [&](w_char wc) {
1544 return std::binary_search(cpdvowels_utf16.begin(), cpdvowels_utf16.end(), wc);
1545 });
1546 }
1547
1548 return num;
1549}
1550
1551void AffixMgr::setcminmax(size_t* cmin, size_t* cmax, const char* word, size_t len) {
1552 if (utf8) {
1553 int i;
1554 for (*cmin = 0, i = 0; (i < cpdmin) && *cmin < len; i++) {
1555 for ((*cmin)++; *cmin < len && (word[*cmin] & 0xc0) == 0x80; (*cmin)++)
1556 ;
1557 }
1558 for (*cmax = len, i = 0; (i < (cpdmin - 1)) && *cmax > 0; i++) {
1559 for ((*cmax)--; *cmax > 0 && (word[*cmax] & 0xc0) == 0x80; (*cmax)--)
1560 ;
1561 }
1562 } else {
1563 *cmin = cpdmin;
1564 *cmax = len - cpdmin + 1;
1565 }
1566}
1567
1568// check if compound word is correctly spelled
1569// hu_mov_rule = spec. Hungarian rule (XXX)
1570struct hentry* AffixMgr::compound_check(const std::string& word,
1571 short wordnum,
1572 short numsyllable,
1573 short maxwordnum,
1574 short wnum,
1575 hentry** words = nullptr,
1576 hentry** rwords = nullptr,
1577 char hu_mov_rule = 0,
1578 char is_sug = 0,
1579 int* info = nullptr) {
1580 short oldnumsyllable, oldnumsyllable2, oldwordnum, oldwordnum2;
1581 hentry *rv = nullptr, *rv_first;
1582 std::string st;
1583 char ch = '\0', affixed;
1584 size_t cmin, cmax;
1585 int striple = 0, soldi = 0, oldcmin = 0, oldcmax = 0, oldlen = 0, checkedstriple = 0;
1586 hentry** oldwords = words;
1587 size_t scpd = 0, len = word.size();
1588
1589 // protect subsequent words[wnum + 1] reads and any recursion
1590 if (wnum + 1 >= maxwordnum)
1591 return nullptr;
1592
1593 int checked_prefix;
1594
1595 // add a time limit to handle possible
1596 // combinatorical explosion of the overlapping words
1597
1598 HUNSPELL_THREAD_LOCALthread_local std::chrono::steady_clock::time_point clock_time_start;
1599 HUNSPELL_THREAD_LOCALthread_local bool timelimit_exceeded;
1600
1601 // get the current time
1602 std::chrono::steady_clock::time_point clock_now = std::chrono::steady_clock::now();
1603
1604 if (wnum == 0) {
1605 // set the start time
1606 clock_time_start = clock_now;
1607 timelimit_exceeded = false;
1608 }
1609 else if (clock_now - clock_time_start > TIMELIMIT_MSstd::chrono::milliseconds(50))
1610 timelimit_exceeded = true;
1611
1612 setcminmax(&cmin, &cmax, word.c_str(), len);
1613
1614 st.assign(word);
1615
1616 for (size_t i = cmin; i < cmax; ++i) {
1617 // go to end of the UTF-8 character
1618 if (utf8) {
1619 for (; (st[i] & 0xc0) == 0x80; i++)
1620 ;
1621 if (i >= cmax)
1622 return nullptr;
1623 }
1624
1625 words = oldwords;
1626 int onlycpdrule = (words) ? 1 : 0;
1627
1628 do { // onlycpdrule loop
1629
1630 oldnumsyllable = numsyllable;
1631 oldwordnum = wordnum;
1632 checked_prefix = 0;
1633
1634 do { // simplified checkcompoundpattern loop
1635
1636 if (timelimit_exceeded ||
1637 std::chrono::steady_clock::now() - clock_time_start > TIMELIMIT_MSstd::chrono::milliseconds(50)) {
1638 timelimit_exceeded = true;
1639 return nullptr;
1640 }
1641
1642 if (scpd > 0) {
1643 for (; scpd <= checkcpdtable.size() &&
1644 (checkcpdtable[scpd - 1].pattern3.empty() ||
1645 i > word.size() ||
1646 word.compare(i, checkcpdtable[scpd - 1].pattern3.size(), checkcpdtable[scpd - 1].pattern3) != 0);
1647 scpd++)
1648 ;
1649
1650 if (scpd > checkcpdtable.size())
1651 break; // break simplified checkcompoundpattern loop
1652 st.replace(i, std::string::npos, checkcpdtable[scpd - 1].pattern);
1653 soldi = i;
1654 i += checkcpdtable[scpd - 1].pattern.size();
1655 st.replace(i, std::string::npos, checkcpdtable[scpd - 1].pattern2);
1656 st.replace(i + checkcpdtable[scpd - 1].pattern2.size(), std::string::npos,
1657 word.substr(soldi + checkcpdtable[scpd - 1].pattern3.size()));
1658
1659 oldlen = len;
1660 len += checkcpdtable[scpd - 1].pattern.size() +
1661 checkcpdtable[scpd - 1].pattern2.size() -
1662 checkcpdtable[scpd - 1].pattern3.size();
1663 oldcmin = cmin;
1664 oldcmax = cmax;
1665 setcminmax(&cmin, &cmax, st.c_str(), len);
1666
1667 cmax = len - cpdmin + 1;
1668 }
1669
1670 if (i >= st.size())
1671 return nullptr;
1672
1673 ch = st[i];
1674 st[i] = '\0';
1675
1676 sfx = nullptr;
1677 pfx = nullptr;
1678
1679 // FIRST WORD
1680
1681 affixed = 1;
1682 rv = lookup(st.c_str(), i); // perhaps without prefix
1683
1684 // forbid dictionary stems with COMPOUNDFORBIDFLAG in
1685 // compound words, overriding the effect of COMPOUNDPERMITFLAG
1686 if ((rv) && compoundforbidflag &&
1687 TESTAFF(rv->astr, compoundforbidflag, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, compoundforbidflag
))
&& !hu_mov_rule) {
1688 bool would_continue = !onlycpdrule && simplifiedcpd;
1689 if (!scpd && would_continue) {
1690 // given the while conditions that continue jumps to, this situation
1691 // never ends
1692 HUNSPELL_WARNING(stderrstderr, "break infinite loop\n");
1693 break;
1694 }
1695
1696 if (scpd > 0 && would_continue) {
1697 // under these conditions we loop again, but the assumption above
1698 // appears to be that cmin and cmax are the original values they
1699 // had in the outside loop
1700 cmin = oldcmin;
1701 cmax = oldcmax;
1702 }
1703 continue;
1704 }
1705
1706 // search homonym with compound flag
1707 while ((rv) && !hu_mov_rule &&
1708 ((needaffix && TESTAFF(rv->astr, needaffix, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, needaffix
))
) ||
1709 !((compoundflag && !words && !onlycpdrule &&
1710 TESTAFF(rv->astr, compoundflag, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, compoundflag
))
) ||
1711 (compoundbegin && !wordnum && !onlycpdrule &&
1712 TESTAFF(rv->astr, compoundbegin, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, compoundbegin
))
) ||
1713 (compoundmiddle && wordnum && !words && !onlycpdrule &&
1714 TESTAFF(rv->astr, compoundmiddle, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, compoundmiddle
))
) ||
1715 (!defcpdtable.empty() && onlycpdrule &&
1716 ((!words && !wordnum &&
1717 defcpd_check(&words, wnum, maxwordnum, rv, rwords, 0)) ||
1718 (words &&
1719 defcpd_check(&words, wnum, maxwordnum, rv, rwords, 0))))) ||
1720 (scpd != 0 && checkcpdtable[scpd - 1].cond != FLAG_NULL0x00 &&
1721 !TESTAFF(rv->astr, checkcpdtable[scpd - 1].cond, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, checkcpdtable
[scpd - 1].cond))
))) {
1722 rv = rv->next_homonym;
1723 }
1724
1725 if (rv)
1726 affixed = 0;
1727
1728 if (!rv) {
1729 if (onlycpdrule)
1730 break;
1731 if (compoundflag &&
1732 !(rv = prefix_check(st, 0, i,
1733 hu_mov_rule ? IN_CPD_OTHER3 : IN_CPD_BEGIN1,
1734 compoundflag))) {
1735 if (((rv = suffix_check(st, 0, i, 0, nullptr, FLAG_NULL0x00, compoundflag, hu_mov_rule ? IN_CPD_OTHER3 : IN_CPD_BEGIN1)) ||
1736 (compoundmoresuffixes && (rv = suffix_check_twosfx(st, 0, i, 0, nullptr, compoundflag)))) &&
1737 !hu_mov_rule && sfx->getCont() &&
1738 ((compoundforbidflag && TESTAFF(sfx->getCont(), compoundforbidflag, sfx->getContLen())(std::binary_search(sfx->getCont(), sfx->getCont() + sfx
->getContLen(), compoundforbidflag))
) ||
1739 (compoundend && TESTAFF(sfx->getCont(), compoundend, sfx->getContLen())(std::binary_search(sfx->getCont(), sfx->getCont() + sfx
->getContLen(), compoundend))
))) {
1740 rv = nullptr;
1741 }
1742 }
1743
1744 if (rv ||
1745 (((wordnum == 0) && compoundbegin &&
1746 ((rv = suffix_check(st, 0, i, 0, nullptr, FLAG_NULL0x00, compoundbegin, hu_mov_rule ? IN_CPD_OTHER3 : IN_CPD_BEGIN1)) ||
1747 (compoundmoresuffixes && (rv = suffix_check_twosfx(st, 0, i, 0, nullptr,
1748 compoundbegin))) || // twofold suffixes + compound
1749 (rv = prefix_check(st, 0, i, hu_mov_rule ? IN_CPD_OTHER3 : IN_CPD_BEGIN1, compoundbegin)))) ||
1750 ((wordnum > 0) && compoundmiddle &&
1751 ((rv = suffix_check(st, 0, i, 0, nullptr, FLAG_NULL0x00, compoundmiddle, hu_mov_rule ? IN_CPD_OTHER3 : IN_CPD_BEGIN1)) ||
1752 (compoundmoresuffixes && (rv = suffix_check_twosfx(st, 0, i, 0, nullptr,
1753 compoundmiddle))) || // twofold suffixes + compound
1754 (rv = prefix_check(st, 0, i, hu_mov_rule ? IN_CPD_OTHER3 : IN_CPD_BEGIN1, compoundmiddle))))))
1755 checked_prefix = 1;
1756 // else check forbiddenwords and needaffix
1757 } else if (rv->astr && (TESTAFF(rv->astr, forbiddenword, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, forbiddenword
))
||
1758 TESTAFF(rv->astr, needaffix, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, needaffix
))
||
1759 TESTAFF(rv->astr, ONLYUPCASEFLAG, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, 65511
))
||
1760 (is_sug && nosuggest &&
1761 TESTAFF(rv->astr, nosuggest, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, nosuggest
))
))) {
1762 st[i] = ch;
1763 // continue;
1764 break;
1765 }
1766
1767 // check non_compound flag in suffix and prefix
1768 if ((rv) && !hu_mov_rule &&
1769 ((pfx && pfx->getCont() &&
1770 TESTAFF(pfx->getCont(), compoundforbidflag, pfx->getContLen())(std::binary_search(pfx->getCont(), pfx->getCont() + pfx
->getContLen(), compoundforbidflag))
) ||
1771 (sfx && sfx->getCont() &&
1772 TESTAFF(sfx->getCont(), compoundforbidflag,(std::binary_search(sfx->getCont(), sfx->getCont() + sfx
->getContLen(), compoundforbidflag))
1773 sfx->getContLen())(std::binary_search(sfx->getCont(), sfx->getCont() + sfx
->getContLen(), compoundforbidflag))
))) {
1774 rv = nullptr;
1775 }
1776
1777 // check compoundend flag in suffix and prefix
1778 if ((rv) && !checked_prefix && compoundend && !hu_mov_rule &&
1779 ((pfx && pfx->getCont() &&
1780 TESTAFF(pfx->getCont(), compoundend, pfx->getContLen())(std::binary_search(pfx->getCont(), pfx->getCont() + pfx
->getContLen(), compoundend))
) ||
1781 (sfx && sfx->getCont() &&
1782 TESTAFF(sfx->getCont(), compoundend, sfx->getContLen())(std::binary_search(sfx->getCont(), sfx->getCont() + sfx
->getContLen(), compoundend))
))) {
1783 rv = nullptr;
1784 }
1785
1786 // check compoundmiddle flag in suffix and prefix
1787 if ((rv) && !checked_prefix && (wordnum == 0) && compoundmiddle &&
1788 !hu_mov_rule &&
1789 ((pfx && pfx->getCont() &&
1790 TESTAFF(pfx->getCont(), compoundmiddle, pfx->getContLen())(std::binary_search(pfx->getCont(), pfx->getCont() + pfx
->getContLen(), compoundmiddle))
) ||
1791 (sfx && sfx->getCont() &&
1792 TESTAFF(sfx->getCont(), compoundmiddle, sfx->getContLen())(std::binary_search(sfx->getCont(), sfx->getCont() + sfx
->getContLen(), compoundmiddle))
))) {
1793 rv = nullptr;
1794 }
1795
1796 // check forbiddenwords
1797 if ((rv) && (rv->astr) &&
1798 (TESTAFF(rv->astr, forbiddenword, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, forbiddenword
))
||
1799 TESTAFF(rv->astr, ONLYUPCASEFLAG, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, 65511
))
||
1800 (is_sug && nosuggest && TESTAFF(rv->astr, nosuggest, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, nosuggest
))
))) {
1801 return nullptr;
1802 }
1803
1804 // increment word number, if the second root has a compoundroot flag
1805 if ((rv) && compoundroot &&
1806 (TESTAFF(rv->astr, compoundroot, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, compoundroot
))
)) {
1807 wordnum++;
1808 }
1809
1810 // first word is acceptable in compound words?
1811 if (((rv) &&
1812 (checked_prefix || (words && words[wnum]) ||
1813 (compoundflag && TESTAFF(rv->astr, compoundflag, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, compoundflag
))
) ||
1814 ((oldwordnum == 0) && compoundbegin &&
1815 TESTAFF(rv->astr, compoundbegin, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, compoundbegin
))
) ||
1816 ((oldwordnum > 0) && compoundmiddle &&
1817 TESTAFF(rv->astr, compoundmiddle, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, compoundmiddle
))
)
1818
1819 // LANG_hu section: spec. Hungarian rule
1820 || ((langnum == LANG_hu) && hu_mov_rule &&
1821 (TESTAFF((std::binary_search(rv->astr, rv->astr + rv->alen, 'F'
))
1822 rv->astr, 'F',(std::binary_search(rv->astr, rv->astr + rv->alen, 'F'
))
1823 rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, 'F'
))
|| // XXX hardwired Hungarian dictionary codes
1824 TESTAFF(rv->astr, 'G', rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, 'G'
))
||
1825 TESTAFF(rv->astr, 'H', rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, 'H'
))
))
1826 // END of LANG_hu section
1827 ) &&
1828 (
1829 // test CHECKCOMPOUNDPATTERN conditions
1830 scpd == 0 || checkcpdtable[scpd - 1].cond == FLAG_NULL0x00 ||
1831 TESTAFF(rv->astr, checkcpdtable[scpd - 1].cond, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, checkcpdtable
[scpd - 1].cond))
) &&
1832 !((checkcompoundtriple && scpd == 0 &&
1833 !words && i < word.size() && // test triple letters
1834 (word[i - 1] == word[i]) &&
1835 (((i > 1) && (word[i - 1] == word[i - 2])) ||
1836 ((word[i - 1] == word[i + 1])) // may be word[i+1] == '\0'
1837 )) ||
1838 (checkcompoundcase && scpd == 0 && !words && i < word.size() &&
1839 cpdcase_check(word, i))))
1840 // LANG_hu section: spec. Hungarian rule
1841 || ((!rv) && (langnum == LANG_hu) && hu_mov_rule &&
1842 (rv = affix_check(st, 0, i)) &&
1843 (sfx && sfx->getCont() &&
1844 ( // XXX hardwired Hungarian dic. codes
1845 TESTAFF(sfx->getCont(), (unsigned short)'x',(std::binary_search(sfx->getCont(), sfx->getCont() + sfx
->getContLen(), (unsigned short)'x'))
1846 sfx->getContLen())(std::binary_search(sfx->getCont(), sfx->getCont() + sfx
->getContLen(), (unsigned short)'x'))
||
1847 TESTAFF((std::binary_search(sfx->getCont(), sfx->getCont() + sfx
->getContLen(), (unsigned short)'%'))
1848 sfx->getCont(), (unsigned short)'%',(std::binary_search(sfx->getCont(), sfx->getCont() + sfx
->getContLen(), (unsigned short)'%'))
1849 sfx->getContLen())(std::binary_search(sfx->getCont(), sfx->getCont() + sfx
->getContLen(), (unsigned short)'%'))
)))) { // first word is ok condition
1850
1851 // LANG_hu section: spec. Hungarian rule
1852 if (langnum == LANG_hu) {
1853 // calculate syllable number of the word
1854 numsyllable += get_syllable(st.substr(0, i));
1855 // + 1 word, if syllable number of the prefix > 1 (hungarian
1856 // convention)
1857 if (pfx && (get_syllable(pfx->getKey()) > 1))
1858 wordnum++;
1859 }
1860 // END of LANG_hu section
1861
1862 // NEXT WORD(S)
1863 rv_first = rv;
1864 st[i] = ch;
1865
1866 do { // striple loop
1867
1868 // check simplifiedtriple
1869 if (simplifiedtriple) {
1870 if (striple) {
1871 checkedstriple = 1;
1872 i--; // check "fahrt" instead of "ahrt" in "Schiffahrt"
1873 } else if (i > 2 && i <= word.size() && word[i - 1] == word[i - 2])
1874 striple = 1;
1875 }
1876
1877 rv = lookup(st.c_str() + i, st.size() - i); // perhaps without prefix
1878
1879 // search homonym with compound flag
1880 while ((rv) && ((needaffix && TESTAFF(rv->astr, needaffix, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, needaffix
))
) ||
1881 !((compoundflag && !words && TESTAFF(rv->astr, compoundflag, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, compoundflag
))
) ||
1882 (compoundend && !words && TESTAFF(rv->astr, compoundend, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, compoundend
))
) ||
1883 (!defcpdtable.empty() && words && defcpd_check(&words, wnum + 1, maxwordnum, rv, nullptr, 1))) ||
1884 (scpd != 0 && checkcpdtable[scpd - 1].cond2 != FLAG_NULL0x00 &&
1885 !TESTAFF(rv->astr, checkcpdtable[scpd - 1].cond2, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, checkcpdtable
[scpd - 1].cond2))
))) {
1886 rv = rv->next_homonym;
1887 }
1888
1889 // check FORCEUCASE
1890 if (rv && forceucase &&
1891 (TESTAFF(rv->astr, forceucase, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, forceucase
))
) &&
1892 !(info && *info & SPELL_ORIGCAP(1 << 5)))
1893 rv = nullptr;
1894
1895 if (rv && words && words[wnum + 1])
1896 return rv_first;
1897
1898 oldnumsyllable2 = numsyllable;
1899 oldwordnum2 = wordnum;
1900
1901 // LANG_hu section: spec. Hungarian rule, XXX hardwired dictionary
1902 // code
1903 if ((rv) && (langnum == LANG_hu) &&
1904 (TESTAFF(rv->astr, 'I', rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, 'I'
))
) &&
1905 !(TESTAFF(rv->astr, 'J', rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, 'J'
))
)) {
1906 numsyllable--;
1907 }
1908 // END of LANG_hu section
1909
1910 // increment word number, if the second root has a compoundroot flag
1911 if ((rv) && (compoundroot) &&
1912 (TESTAFF(rv->astr, compoundroot, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, compoundroot
))
)) {
1913 wordnum++;
1914 }
1915
1916 // check forbiddenwords
1917 if ((rv) && (rv->astr) &&
1918 (TESTAFF(rv->astr, forbiddenword, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, forbiddenword
))
||
1919 TESTAFF(rv->astr, ONLYUPCASEFLAG, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, 65511
))
||
1920 (is_sug && nosuggest &&
1921 TESTAFF(rv->astr, nosuggest, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, nosuggest
))
)))
1922 return nullptr;
1923
1924 // second word is acceptable, as a root?
1925 // hungarian conventions: compounding is acceptable,
1926 // when compound forms consist of 2 words, or if more,
1927 // then the syllable number of root words must be 6, or lesser.
1928
1929 if ((rv) &&
1930 ((compoundflag && TESTAFF(rv->astr, compoundflag, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, compoundflag
))
) ||
1931 (compoundend && TESTAFF(rv->astr, compoundend, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, compoundend
))
)) &&
1932 (((cpdwordmax == -1) || (wordnum + 1 < cpdwordmax)) ||
1933 ((cpdmaxsyllable != 0) &&
1934 (numsyllable + get_syllable(std::string(HENTRY_WORD(rv)&(rv->word[0]), rv->blen)) <=
1935 cpdmaxsyllable))) &&
1936 (
1937 // test CHECKCOMPOUNDPATTERN
1938 checkcpdtable.empty() || scpd != 0 ||
1939 (i < word.size() && !cpdpat_check(word, i, rv_first, rv, 0))) &&
1940 ((!checkcompounddup || (rv != rv_first)))
1941 // test CHECKCOMPOUNDPATTERN conditions
1942 &&
1943 (scpd == 0 || checkcpdtable[scpd - 1].cond2 == FLAG_NULL0x00 ||
1944 TESTAFF(rv->astr, checkcpdtable[scpd - 1].cond2, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, checkcpdtable
[scpd - 1].cond2))
)) {
1945 // forbid compound word, if it is a non-compound word with typical
1946 // fault
1947 if ((checkcompoundrep && cpdrep_check(word, len)) ||
1948 cpdwordpair_check(word, len))
1949 return nullptr;
1950 return rv_first;
1951 }
1952
1953 numsyllable = oldnumsyllable2;
1954 wordnum = oldwordnum2;
1955
1956 // perhaps second word has prefix or/and suffix
1957 sfx = nullptr;
1958 sfxflag = FLAG_NULL0x00;
1959 rv = (compoundflag && !onlycpdrule && i < word.size()) ? affix_check(word, i, word.size() - i, compoundflag, IN_CPD_END2)
1960 : nullptr;
1961 if (!rv && compoundend && !onlycpdrule) {
1962 sfx = nullptr;
1963 pfx = nullptr;
1964 if (i < word.size())
1965 rv = affix_check(word, i, word.size() - i, compoundend, IN_CPD_END2);
1966 }
1967
1968 if (!rv && !defcpdtable.empty() && words) {
1969 if (i < word.size())
1970 rv = affix_check(word, i, word.size() - i, 0, IN_CPD_END2);
1971 if (rv && defcpd_check(&words, wnum + 1, maxwordnum, rv, nullptr, 1))
1972 return rv_first;
1973 rv = nullptr;
1974 }
1975
1976 // test CHECKCOMPOUNDPATTERN conditions (allowed forms)
1977 if (rv &&
1978 !(scpd == 0 || checkcpdtable[scpd - 1].cond2 == FLAG_NULL0x00 ||
1979 TESTAFF(rv->astr, checkcpdtable[scpd - 1].cond2, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, checkcpdtable
[scpd - 1].cond2))
))
1980 rv = nullptr;
1981
1982 // test CHECKCOMPOUNDPATTERN conditions (forbidden compounds)
1983 if (rv && !checkcpdtable.empty() && scpd == 0 &&
1984 cpdpat_check(word, i, rv_first, rv, affixed))
1985 rv = nullptr;
1986
1987 // check non_compound flag in suffix and prefix
1988 if ((rv) && ((pfx && pfx->getCont() &&
1989 TESTAFF(pfx->getCont(), compoundforbidflag,(std::binary_search(pfx->getCont(), pfx->getCont() + pfx
->getContLen(), compoundforbidflag))
1990 pfx->getContLen())(std::binary_search(pfx->getCont(), pfx->getCont() + pfx
->getContLen(), compoundforbidflag))
) ||
1991 (sfx && sfx->getCont() &&
1992 TESTAFF(sfx->getCont(), compoundforbidflag,(std::binary_search(sfx->getCont(), sfx->getCont() + sfx
->getContLen(), compoundforbidflag))
1993 sfx->getContLen())(std::binary_search(sfx->getCont(), sfx->getCont() + sfx
->getContLen(), compoundforbidflag))
))) {
1994 rv = nullptr;
1995 }
1996
1997 // check FORCEUCASE
1998 if (rv && forceucase &&
1999 (TESTAFF(rv->astr, forceucase, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, forceucase
))
) &&
2000 !(info && *info & SPELL_ORIGCAP(1 << 5)))
2001 rv = nullptr;
2002
2003 // check forbiddenwords
2004 if ((rv) && (rv->astr) &&
2005 (TESTAFF(rv->astr, forbiddenword, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, forbiddenword
))
||
2006 TESTAFF(rv->astr, ONLYUPCASEFLAG, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, 65511
))
||
2007 (is_sug && nosuggest &&
2008 TESTAFF(rv->astr, nosuggest, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, nosuggest
))
)))
2009 return nullptr;
2010
2011 // pfxappnd = prefix of word+i, or NULL
2012 // calculate syllable number of prefix.
2013 // hungarian convention: when syllable number of prefix is more,
2014 // than 1, the prefix+word counts as two words.
2015
2016 if (langnum == LANG_hu) {
2017 if (i < word.size()) {
2018 // calculate syllable number of the word
2019 numsyllable += get_syllable(word.substr(i));
2020 }
2021
2022 // - affix syllable num.
2023 // XXX only second suffix (inflections, not derivations)
2024 if (sfxappnd) {
2025 std::string tmp(sfxappnd);
2026 reverseword(tmp);
2027 numsyllable -= short(get_syllable(tmp) + sfxextra);
2028 } else {
2029 numsyllable -= short(sfxextra);
2030 }
2031
2032 // + 1 word, if syllable number of the prefix > 1 (hungarian
2033 // convention)
2034 if (pfx && (get_syllable(pfx->getKey()) > 1))
2035 wordnum++;
2036
2037 // increment syllable num, if last word has a SYLLABLENUM flag
2038 // and the suffix is beginning `s'
2039
2040 if (!cpdsyllablenum.empty()) {
2041 switch (sfxflag) {
2042 case 'c': {
2043 numsyllable += 2;
2044 break;
2045 }
2046 case 'J': {
2047 numsyllable += 1;
2048 break;
2049 }
2050 case 'I': {
2051 if (rv && TESTAFF(rv->astr, 'J', rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, 'J'
))
)
2052 numsyllable += 1;
2053 break;
2054 }
2055 }
2056 }
2057 }
2058
2059 // increment word number, if the second word has a compoundroot flag
2060 if ((rv) && (compoundroot) &&
2061 (TESTAFF(rv->astr, compoundroot, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, compoundroot
))
)) {
2062 wordnum++;
2063 }
2064 // second word is acceptable, as a word with prefix or/and suffix?
2065 // hungarian conventions: compounding is acceptable,
2066 // when compound forms consist 2 word, otherwise
2067 // the syllable number of root words is 6, or lesser.
2068 if ((rv) &&
2069 (((cpdwordmax == -1) || (wordnum + 1 < cpdwordmax)) ||
2070 ((cpdmaxsyllable != 0) && (numsyllable <= cpdmaxsyllable))) &&
2071 ((!checkcompounddup || (rv != rv_first)))) {
2072 // forbid compound word, if it is a non-compound word with typical
2073 // fault
2074 if ((checkcompoundrep && cpdrep_check(word, len)) ||
2075 cpdwordpair_check(word, len))
2076 return nullptr;
2077 return rv_first;
2078 }
2079
2080 numsyllable = oldnumsyllable2;
2081 wordnum = oldwordnum2;
2082
2083 // perhaps second word is a compound word (recursive call)
2084 // (only if SPELL_COMPOUND_2 is not set and maxwordnum is not exceeded)
2085 if ((!info || !(*info & SPELL_COMPOUND_2(1 << 7))) && wordnum + 2 < maxwordnum && wnum + 1 < maxwordnum) {
2086 rv = compound_check(st.substr(i), wordnum + 1,
2087 numsyllable, maxwordnum, wnum + 1, words, rwords, 0,
2088 is_sug, info);
2089
2090 if (rv && !checkcpdtable.empty() && i < word.size() &&
2091 ((scpd == 0 &&
2092 cpdpat_check(word, i, rv_first, rv, affixed)) ||
2093 (scpd != 0 &&
2094 !cpdpat_check(word, i, rv_first, rv, affixed))))
2095 rv = nullptr;
2096 } else {
2097 rv = nullptr;
2098 }
2099 if (rv) {
2100 // forbid compound word, if it is a non-compound word with typical
2101 // fault, or a dictionary word pair
2102
2103 if (cpdwordpair_check(word, len))
2104 return nullptr;
2105
2106 if (checkcompoundrep || forbiddenword) {
2107
2108 if (checkcompoundrep && cpdrep_check(word, len))
2109 return nullptr;
2110
2111 // check first part
2112 if (i < word.size() && word.compare(i, rv->blen, rv->word, rv->blen) == 0) {
2113 char r = st[i + rv->blen];
2114 st[i + rv->blen] = '\0';
2115
2116 if ((checkcompoundrep && cpdrep_check(st, i + rv->blen)) ||
2117 cpdwordpair_check(st, i + rv->blen)) {
2118 st[ + i + rv->blen] = r;
2119 continue;
2120 }
2121
2122 if (forbiddenword) {
2123 struct hentry* rv2 = lookup(word.c_str(), word.size());
2124 if (!rv2 && len <= word.size())
2125 rv2 = affix_check(word, 0, len);
2126 if (rv2 && rv2->astr &&
2127 TESTAFF(rv2->astr, forbiddenword, rv2->alen)(std::binary_search(rv2->astr, rv2->astr + rv2->alen
, forbiddenword))
&&
2128 (strncmp(rv2->word, st.c_str(), i + rv->blen) == 0)) {
2129 return nullptr;
2130 }
2131 }
2132 st[i + rv->blen] = r;
2133 }
2134 }
2135 return rv_first;
2136 }
2137 } while (striple && !checkedstriple); // end of striple loop
2138
2139 if (checkedstriple) {
2140 i++;
2141 checkedstriple = 0;
2142 striple = 0;
2143 }
2144
2145 } // first word is ok condition
2146
2147 if (soldi != 0) {
2148 i = soldi;
2149 soldi = 0;
2150 len = oldlen;
2151 cmin = oldcmin;
2152 cmax = oldcmax;
2153 }
2154 scpd++;
2155
2156 } while (!onlycpdrule && simplifiedcpd &&
2157 scpd <= checkcpdtable.size()); // end of simplifiedcpd loop
2158
2159 scpd = 0;
2160 wordnum = oldwordnum;
2161 numsyllable = oldnumsyllable;
2162
2163 if (soldi != 0) {
2164 i = soldi;
2165 st.assign(word); // XXX add more optim.
2166 soldi = 0;
2167 len = oldlen;
2168 cmin = oldcmin;
2169 cmax = oldcmax;
2170 } else
2171 st[i] = ch;
2172
2173 } while (!defcpdtable.empty() && oldwordnum == 0 &&
2174 onlycpdrule++ < 1); // end of onlycpd loop
2175 }
2176
2177 return nullptr;
2178}
2179
2180// check if compound word is correctly spelled
2181// hu_mov_rule = spec. Hungarian rule (XXX)
2182int AffixMgr::compound_check_morph(const std::string& word,
2183 short wordnum,
2184 short numsyllable,
2185 short maxwordnum,
2186 short wnum,
2187 hentry** words,
2188 hentry** rwords,
2189 char hu_mov_rule,
2190 std::string& result,
2191 const std::string* partresult) {
2192 short oldnumsyllable, oldnumsyllable2, oldwordnum, oldwordnum2;
2193 hentry *rv = nullptr, *rv_first;
2194 std::string st, presult;
2195 char ch, affixed = 0;
2196 int checked_prefix, ok = 0;
2197 size_t cmin, cmax;
2198 hentry** oldwords = words;
2199 size_t len = word.size();
2200
2201 // protect subsequent words[wnum + 1] reads and any recursion
2202 if (wnum + 1 >= maxwordnum)
2203 return 0;
2204
2205 // add a time limit to handle possible
2206 // combinatorical explosion of the overlapping words
2207
2208 HUNSPELL_THREAD_LOCALthread_local std::chrono::steady_clock::time_point clock_time_start;
2209 HUNSPELL_THREAD_LOCALthread_local bool timelimit_exceeded;
2210
2211 // get the current time
2212 std::chrono::steady_clock::time_point clock_now = std::chrono::steady_clock::now();
2213
2214 if (wnum == 0) {
2215 // set the start time
2216 clock_time_start = clock_now;
2217 timelimit_exceeded = false;
2218 }
2219 else if (clock_now - clock_time_start > TIMELIMIT_MSstd::chrono::milliseconds(50))
2220 timelimit_exceeded = true;
2221
2222 setcminmax(&cmin, &cmax, word.c_str(), len);
2223
2224 st.assign(word);
2225
2226 for (size_t i = cmin; i < cmax; ++i) {
2227 // go to end of the UTF-8 character
2228 if (utf8) {
2229 for (; (st[i] & 0xc0) == 0x80; i++)
2230 ;
2231 if (i >= cmax)
2232 return 0;
2233 }
2234
2235 words = oldwords;
2236 int onlycpdrule = (words) ? 1 : 0;
2237
2238 do { // onlycpdrule loop
2239
2240 if (timelimit_exceeded)
2241 return 0;
2242
2243 oldnumsyllable = numsyllable;
2244 oldwordnum = wordnum;
2245 checked_prefix = 0;
2246
2247 if (i >= st.size())
2248 return 0;
2249
2250 ch = st[i];
2251 st[i] = '\0';
2252 sfx = nullptr;
2253
2254 // FIRST WORD
2255
2256 affixed = 1;
2257
2258 presult.clear();
2259 if (partresult)
2260 presult.append(*partresult);
2261
2262 rv = lookup(st.c_str(), i); // perhaps without prefix
2263
2264 // forbid dictionary stems with COMPOUNDFORBIDFLAG in
2265 // compound words, overriding the effect of COMPOUNDPERMITFLAG
2266 if ((rv) && compoundforbidflag &&
2267 TESTAFF(rv->astr, compoundforbidflag, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, compoundforbidflag
))
&& !hu_mov_rule)
2268 continue;
2269
2270 // search homonym with compound flag
2271 while ((rv) && !hu_mov_rule &&
2272 ((needaffix && TESTAFF(rv->astr, needaffix, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, needaffix
))
) ||
2273 !((compoundflag && !words && !onlycpdrule &&
2274 TESTAFF(rv->astr, compoundflag, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, compoundflag
))
) ||
2275 (compoundbegin && !wordnum && !onlycpdrule &&
2276 TESTAFF(rv->astr, compoundbegin, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, compoundbegin
))
) ||
2277 (compoundmiddle && wordnum && !words && !onlycpdrule &&
2278 TESTAFF(rv->astr, compoundmiddle, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, compoundmiddle
))
) ||
2279 (!defcpdtable.empty() && onlycpdrule &&
2280 ((!words && !wordnum &&
2281 defcpd_check(&words, wnum, maxwordnum, rv, rwords, 0)) ||
2282 (words &&
2283 defcpd_check(&words, wnum, maxwordnum, rv, rwords, 0))))))) {
2284 rv = rv->next_homonym;
2285 }
2286
2287
2288 if (rv)
2289 affixed = 0;
2290
2291 if (rv) {
2292 presult.push_back(MSEP_FLD' ');
2293 presult.append(MORPH_PART"pa:");
2294 presult.append(st, 0, i);
2295 if (!HENTRY_FIND(rv, MORPH_STEM"st:")) {
2296 presult.push_back(MSEP_FLD' ');
2297 presult.append(MORPH_STEM"st:");
2298 presult.append(st, 0, i);
2299 }
2300 if (HENTRY_DATA(rv)) {
2301 presult.push_back(MSEP_FLD' ');
2302 presult.append(HENTRY_DATA2(rv));
2303 }
2304 }
2305
2306 if (!rv) {
2307 if (compoundflag &&
2308 !(rv =
2309 prefix_check(st, 0, i, hu_mov_rule ? IN_CPD_OTHER3 : IN_CPD_BEGIN1,
2310 compoundflag))) {
2311 if (((rv = suffix_check(st, 0, i, 0, nullptr, FLAG_NULL0x00, compoundflag, hu_mov_rule ? IN_CPD_OTHER3 : IN_CPD_BEGIN1)) ||
2312 (compoundmoresuffixes && (rv = suffix_check_twosfx(st, 0, i, 0, nullptr, compoundflag)))) &&
2313 !hu_mov_rule && sfx->getCont() &&
2314 ((compoundforbidflag && TESTAFF(sfx->getCont(), compoundforbidflag, sfx->getContLen())(std::binary_search(sfx->getCont(), sfx->getCont() + sfx
->getContLen(), compoundforbidflag))
) ||
2315 (compoundend && TESTAFF(sfx->getCont(), compoundend, sfx->getContLen())(std::binary_search(sfx->getCont(), sfx->getCont() + sfx
->getContLen(), compoundend))
))) {
2316 rv = nullptr;
2317 }
2318 }
2319
2320 if (rv ||
2321 (((wordnum == 0) && compoundbegin &&
2322 ((rv = suffix_check(st, 0, i, 0, nullptr, FLAG_NULL0x00, compoundbegin, hu_mov_rule ? IN_CPD_OTHER3 : IN_CPD_BEGIN1)) ||
2323 (compoundmoresuffixes && (rv = suffix_check_twosfx(st, 0, i, 0, nullptr,
2324 compoundbegin))) || // twofold suffix+compound
2325 (rv = prefix_check(st, 0, i, hu_mov_rule ? IN_CPD_OTHER3 : IN_CPD_BEGIN1, compoundbegin)))) ||
2326 ((wordnum > 0) && compoundmiddle &&
2327 ((rv = suffix_check(st, 0, i, 0, nullptr, FLAG_NULL0x00, compoundmiddle, hu_mov_rule ? IN_CPD_OTHER3 : IN_CPD_BEGIN1)) ||
2328 (compoundmoresuffixes && (rv = suffix_check_twosfx(st, 0, i, 0, nullptr,
2329 compoundmiddle))) || // twofold suffix+compound
2330 (rv = prefix_check(st, 0, i, hu_mov_rule ? IN_CPD_OTHER3 : IN_CPD_BEGIN1, compoundmiddle)))))) {
2331 std::string p;
2332 if (compoundflag)
2333 p = affix_check_morph(st, 0, i, compoundflag);
2334 if (p.empty()) {
2335 if ((wordnum == 0) && compoundbegin) {
2336 p = affix_check_morph(st, 0, i, compoundbegin);
2337 } else if ((wordnum > 0) && compoundmiddle) {
2338 p = affix_check_morph(st, 0, i, compoundmiddle);
2339 }
2340 }
2341 if (!p.empty()) {
2342 presult.push_back(MSEP_FLD' ');
2343 presult.append(MORPH_PART"pa:");
2344 presult.append(st, 0, i);
2345 line_uniq_app(p, MSEP_REC'\n');
2346 presult.append(p);
2347 }
2348 checked_prefix = 1;
2349 }
2350 // else check forbiddenwords
2351 } else if (rv->astr && (TESTAFF(rv->astr, forbiddenword, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, forbiddenword
))
||
2352 TESTAFF(rv->astr, ONLYUPCASEFLAG, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, 65511
))
||
2353 TESTAFF(rv->astr, needaffix, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, needaffix
))
)) {
2354 st[i] = ch;
2355 continue;
2356 }
2357
2358 // check non_compound flag in suffix and prefix
2359 if ((rv) && !hu_mov_rule &&
2360 ((pfx && pfx->getCont() &&
2361 TESTAFF(pfx->getCont(), compoundforbidflag, pfx->getContLen())(std::binary_search(pfx->getCont(), pfx->getCont() + pfx
->getContLen(), compoundforbidflag))
) ||
2362 (sfx && sfx->getCont() &&
2363 TESTAFF(sfx->getCont(), compoundforbidflag, sfx->getContLen())(std::binary_search(sfx->getCont(), sfx->getCont() + sfx
->getContLen(), compoundforbidflag))
))) {
2364 continue;
2365 }
2366
2367 // check compoundend flag in suffix and prefix
2368 if ((rv) && !checked_prefix && compoundend && !hu_mov_rule &&
2369 ((pfx && pfx->getCont() &&
2370 TESTAFF(pfx->getCont(), compoundend, pfx->getContLen())(std::binary_search(pfx->getCont(), pfx->getCont() + pfx
->getContLen(), compoundend))
) ||
2371 (sfx && sfx->getCont() &&
2372 TESTAFF(sfx->getCont(), compoundend, sfx->getContLen())(std::binary_search(sfx->getCont(), sfx->getCont() + sfx
->getContLen(), compoundend))
))) {
2373 continue;
2374 }
2375
2376 // check compoundmiddle flag in suffix and prefix
2377 if ((rv) && !checked_prefix && (wordnum == 0) && compoundmiddle &&
2378 !hu_mov_rule &&
2379 ((pfx && pfx->getCont() &&
2380 TESTAFF(pfx->getCont(), compoundmiddle, pfx->getContLen())(std::binary_search(pfx->getCont(), pfx->getCont() + pfx
->getContLen(), compoundmiddle))
) ||
2381 (sfx && sfx->getCont() &&
2382 TESTAFF(sfx->getCont(), compoundmiddle, sfx->getContLen())(std::binary_search(sfx->getCont(), sfx->getCont() + sfx
->getContLen(), compoundmiddle))
))) {
2383 rv = nullptr;
2384 }
2385
2386 // check forbiddenwords
2387 if ((rv) && (rv->astr) && (TESTAFF(rv->astr, forbiddenword, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, forbiddenword
))
||
2388 TESTAFF(rv->astr, ONLYUPCASEFLAG, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, 65511
))
))
2389 continue;
2390
2391 // increment word number, if the second root has a compoundroot flag
2392 if ((rv) && (compoundroot) &&
2393 (TESTAFF(rv->astr, compoundroot, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, compoundroot
))
)) {
2394 wordnum++;
2395 }
2396
2397 // first word is acceptable in compound words?
2398 if (((rv) &&
2399 (checked_prefix || (words && words[wnum]) || (compoundflag && TESTAFF(rv->astr, compoundflag, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, compoundflag
))
) ||
2400 ((oldwordnum == 0) && compoundbegin && TESTAFF(rv->astr, compoundbegin, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, compoundbegin
))
) ||
2401 ((oldwordnum > 0) && compoundmiddle && TESTAFF(rv->astr, compoundmiddle, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, compoundmiddle
))
)
2402 // LANG_hu section: spec. Hungarian rule
2403 || ((langnum == LANG_hu) && // hu_mov_rule
2404 hu_mov_rule &&
2405 (TESTAFF(rv->astr, 'F', rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, 'F'
))
|| TESTAFF(rv->astr, 'G', rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, 'G'
))
|| TESTAFF(rv->astr, 'H', rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, 'H'
))
))
2406 // END of LANG_hu section
2407 ) &&
2408 !((checkcompoundtriple && !words && // test triple letters
2409 (word[i - 1] == word[i]) &&
2410 (((i > 1) && (word[i - 1] == word[i - 2])) || ((word[i - 1] == word[i + 1])) // may be word[i+1] == '\0'
2411 )) ||
2412 (
2413 // test CHECKCOMPOUNDPATTERN
2414 !checkcpdtable.empty() && !words && cpdpat_check(word, i, rv, nullptr, affixed)) ||
2415 (checkcompoundcase && !words && cpdcase_check(word, i))))
2416 // LANG_hu section: spec. Hungarian rule
2417 || ((!rv) && (langnum == LANG_hu) && hu_mov_rule && (rv = affix_check(st, 0, i)) &&
2418 (sfx && sfx->getCont() &&
2419 (TESTAFF(sfx->getCont(), (unsigned short)'x', sfx->getContLen())(std::binary_search(sfx->getCont(), sfx->getCont() + sfx
->getContLen(), (unsigned short)'x'))
||
2420 TESTAFF(sfx->getCont(), (unsigned short)'%', sfx->getContLen())(std::binary_search(sfx->getCont(), sfx->getCont() + sfx
->getContLen(), (unsigned short)'%'))
)))
2421 // END of LANG_hu section
2422 ) {
2423 // LANG_hu section: spec. Hungarian rule
2424 if (langnum == LANG_hu) {
2425 // calculate syllable number of the word
2426 numsyllable += get_syllable(st.substr(0, i));
2427
2428 // + 1 word, if syllable number of the prefix > 1 (hungarian
2429 // convention)
2430 if (pfx && (get_syllable(pfx->getKey()) > 1))
2431 wordnum++;
2432 }
2433 // END of LANG_hu section
2434
2435 // NEXT WORD(S)
2436 rv_first = rv;
2437 rv = lookup(word.c_str() + i, word.size() - i); // perhaps without prefix
2438
2439 // search homonym with compound flag
2440 while ((rv) && ((needaffix && TESTAFF(rv->astr, needaffix, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, needaffix
))
) ||
2441 !((compoundflag && !words && TESTAFF(rv->astr, compoundflag, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, compoundflag
))
) ||
2442 (compoundend && !words && TESTAFF(rv->astr, compoundend, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, compoundend
))
) ||
2443 (!defcpdtable.empty() && words && defcpd_check(&words, wnum + 1, maxwordnum, rv, nullptr, 1))))) {
2444 rv = rv->next_homonym;
2445 }
2446
2447 if (rv && words && words[wnum + 1]) {
2448 result.append(presult);
2449 result.push_back(MSEP_FLD' ');
2450 result.append(MORPH_PART"pa:");
2451 result.append(word, i, word.size());
2452 if (complexprefixes && HENTRY_DATA(rv))
2453 result.append(HENTRY_DATA2(rv));
2454 if (!HENTRY_FIND(rv, MORPH_STEM"st:")) {
2455 result.push_back(MSEP_FLD' ');
2456 result.append(MORPH_STEM"st:");
2457 result.append(HENTRY_WORD(rv)&(rv->word[0]));
2458 }
2459 // store the pointer of the hash entry
2460 if (!complexprefixes && HENTRY_DATA(rv)) {
2461 result.push_back(MSEP_FLD' ');
2462 result.append(HENTRY_DATA2(rv));
2463 }
2464 result.push_back(MSEP_REC'\n');
2465 return 0;
2466 }
2467
2468 oldnumsyllable2 = numsyllable;
2469 oldwordnum2 = wordnum;
2470
2471 // LANG_hu section: spec. Hungarian rule
2472 if ((rv) && (langnum == LANG_hu) &&
2473 (TESTAFF(rv->astr, 'I', rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, 'I'
))
) &&
2474 !(TESTAFF(rv->astr, 'J', rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, 'J'
))
)) {
2475 numsyllable--;
2476 }
2477 // END of LANG_hu section
2478 // increment word number, if the second root has a compoundroot flag
2479 if ((rv) && (compoundroot) &&
2480 (TESTAFF(rv->astr, compoundroot, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, compoundroot
))
)) {
2481 wordnum++;
2482 }
2483
2484 // check forbiddenwords
2485 if ((rv) && (rv->astr) &&
2486 (TESTAFF(rv->astr, forbiddenword, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, forbiddenword
))
||
2487 TESTAFF(rv->astr, ONLYUPCASEFLAG, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, 65511
))
)) {
2488 st[i] = ch;
2489 continue;
2490 }
2491
2492 // second word is acceptable, as a root?
2493 // hungarian conventions: compounding is acceptable,
2494 // when compound forms consist of 2 words, or if more,
2495 // then the syllable number of root words must be 6, or lesser.
2496 if ((rv) &&
2497 ((compoundflag && TESTAFF(rv->astr, compoundflag, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, compoundflag
))
) ||
2498 (compoundend && TESTAFF(rv->astr, compoundend, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, compoundend
))
)) &&
2499 (((cpdwordmax == -1) || (wordnum + 1 < cpdwordmax)) ||
2500 ((cpdmaxsyllable != 0) &&
2501 (numsyllable + get_syllable(std::string(HENTRY_WORD(rv)&(rv->word[0]), rv->blen)) <=
2502 cpdmaxsyllable))) &&
2503 ((!checkcompounddup || (rv != rv_first)))) {
2504 // bad compound word
2505 result.append(presult);
2506 result.push_back(MSEP_FLD' ');
2507 result.append(MORPH_PART"pa:");
2508 result.append(word, i, word.size());
2509
2510 if (HENTRY_DATA(rv)) {
2511 if (complexprefixes)
2512 result.append(HENTRY_DATA2(rv));
2513 if (!HENTRY_FIND(rv, MORPH_STEM"st:")) {
2514 result.push_back(MSEP_FLD' ');
2515 result.append(MORPH_STEM"st:");
2516 result.append(HENTRY_WORD(rv)&(rv->word[0]));
2517 }
2518 // store the pointer of the hash entry
2519 if (!complexprefixes) {
2520 result.push_back(MSEP_FLD' ');
2521 result.append(HENTRY_DATA2(rv));
2522 }
2523 }
2524 result.push_back(MSEP_REC'\n');
2525 ok = 1;
2526 }
2527
2528 numsyllable = oldnumsyllable2;
2529 wordnum = oldwordnum2;
2530
2531 // perhaps second word has prefix or/and suffix
2532 sfx = nullptr;
2533 sfxflag = FLAG_NULL0x00;
2534
2535 if (compoundflag && !onlycpdrule)
2536 rv = affix_check(word, i, word.size() - i, compoundflag);
2537 else
2538 rv = nullptr;
2539
2540 if (!rv && compoundend && !onlycpdrule) {
2541 sfx = nullptr;
2542 pfx = nullptr;
2543 rv = affix_check(word, i, word.size() - i, compoundend);
2544 }
2545
2546 if (!rv && !defcpdtable.empty() && words) {
2547 rv = affix_check(word, i, word.size() - i, 0, IN_CPD_END2);
2548 if (rv && words && defcpd_check(&words, wnum + 1, maxwordnum, rv, nullptr, 1)) {
2549 std::string m;
2550 if (compoundflag)
2551 m = affix_check_morph(word, i, word.size() - i, compoundflag);
2552 if (m.empty() && compoundend) {
2553 m = affix_check_morph(word, i, word.size() - i, compoundend);
2554 }
2555 result.append(presult);
2556 if (!m.empty()) {
2557 result.push_back(MSEP_FLD' ');
2558 result.append(MORPH_PART"pa:");
2559 result.append(word, i, word.size());
2560 line_uniq_app(m, MSEP_REC'\n');
2561 result.append(m);
2562 }
2563 result.push_back(MSEP_REC'\n');
2564 ok = 1;
2565 }
2566 }
2567
2568 // check non_compound flag in suffix and prefix
2569 if ((rv) &&
2570 ((pfx && pfx->getCont() &&
2571 TESTAFF(pfx->getCont(), compoundforbidflag, pfx->getContLen())(std::binary_search(pfx->getCont(), pfx->getCont() + pfx
->getContLen(), compoundforbidflag))
) ||
2572 (sfx && sfx->getCont() &&
2573 TESTAFF(sfx->getCont(), compoundforbidflag,(std::binary_search(sfx->getCont(), sfx->getCont() + sfx
->getContLen(), compoundforbidflag))
2574 sfx->getContLen())(std::binary_search(sfx->getCont(), sfx->getCont() + sfx
->getContLen(), compoundforbidflag))
))) {
2575 rv = nullptr;
2576 }
2577
2578 // check forbiddenwords
2579 if ((rv) && (rv->astr) &&
2580 (TESTAFF(rv->astr, forbiddenword, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, forbiddenword
))
||
2581 TESTAFF(rv->astr, ONLYUPCASEFLAG, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, 65511
))
) &&
2582 (!TESTAFF(rv->astr, needaffix, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, needaffix
))
)) {
2583 st[i] = ch;
2584 continue;
2585 }
2586
2587 if (langnum == LANG_hu) {
2588 // calculate syllable number of the word
2589 numsyllable += get_syllable(word.c_str() + i);
2590
2591 // - affix syllable num.
2592 // XXX only second suffix (inflections, not derivations)
2593 if (sfxappnd) {
2594 std::string tmp(sfxappnd);
2595 reverseword(tmp);
2596 numsyllable -= short(get_syllable(tmp) + sfxextra);
2597 } else {
2598 numsyllable -= short(sfxextra);
2599 }
2600
2601 // + 1 word, if syllable number of the prefix > 1 (hungarian
2602 // convention)
2603 if (pfx && (get_syllable(pfx->getKey()) > 1))
2604 wordnum++;
2605
2606 // increment syllable num, if last word has a SYLLABLENUM flag
2607 // and the suffix is beginning `s'
2608
2609 if (!cpdsyllablenum.empty()) {
2610 switch (sfxflag) {
2611 case 'c': {
2612 numsyllable += 2;
2613 break;
2614 }
2615 case 'J': {
2616 numsyllable += 1;
2617 break;
2618 }
2619 case 'I': {
2620 if (rv && TESTAFF(rv->astr, 'J', rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, 'J'
))
)
2621 numsyllable += 1;
2622 break;
2623 }
2624 }
2625 }
2626 }
2627
2628 // increment word number, if the second word has a compoundroot flag
2629 if ((rv) && (compoundroot) &&
2630 (TESTAFF(rv->astr, compoundroot, rv->alen)(std::binary_search(rv->astr, rv->astr + rv->alen, compoundroot
))
)) {
2631 wordnum++;
2632 }
2633 // second word is acceptable, as a word with prefix or/and suffix?
2634 // hungarian conventions: compounding is acceptable,
2635 // when compound forms consist 2 word, otherwise
2636 // the syllable number of root words is 6, or lesser.
2637 if ((rv) &&
2638 (((cpdwordmax == -1) || (wordnum + 1 < cpdwordmax)) ||
2639 ((cpdmaxsyllable != 0) && (numsyllable <= cpdmaxsyllable))) &&
2640 ((!checkcompounddup || (rv != rv_first)))) {
2641 std::string m;
2642 if (compoundflag)
2643 m = affix_check_morph(word, i, word.size() - i, compoundflag);
2644 if (m.empty() && compoundend) {
2645 m = affix_check_morph(word, i, word.size() - i, compoundend);
2646 }
2647 result.append(presult);
2648 if (!m.empty()) {
2649 result.push_back(MSEP_FLD' ');
2650 result.append(MORPH_PART"pa:");
2651 result.append(word, i, word.size());
2652 line_uniq_app(m, MSEP_REC'\n');
2653 result.push_back(MSEP_FLD' ');
2654 result.append(m);
2655 }
2656 result.push_back(MSEP_REC'\n');
2657 ok = 1;
2658 }
2659
2660 numsyllable = oldnumsyllable2;
2661 wordnum = oldwordnum2;
2662
2663 // perhaps second word is a compound word (recursive call)
2664 if ((wordnum + 2 < maxwordnum) && (wnum + 1 < maxwordnum) && (ok == 0)) {
2665 compound_check_morph(word.substr(i), wordnum + 1,
2666 numsyllable, maxwordnum, wnum + 1, words, rwords, 0,
2667 result, &presult);
2668 } else {
2669 rv = nullptr;
2670 }
2671 }
2672 st[i] = ch;
2673 wordnum = oldwordnum;
2674 numsyllable = oldnumsyllable;
2675
2676 } while (!defcpdtable.empty() && oldwordnum == 0 &&
2677 onlycpdrule++ < 1); // end of onlycpd loop
2678 }
2679 return 0;
2680}
2681
2682
2683inline int AffixMgr::isRevSubset(const char* s1,
2684 const char* end_of_s2,
2685 int len) {
2686 while ((len > 0) && (*s1 != '\0') && ((*s1 == *end_of_s2) || (*s1 == '.'))) {
2687 s1++;
2688 end_of_s2--;
2689 len--;
2690 }
2691 return (*s1 == '\0');
2692}
2693
2694// check word for suffixes
2695struct hentry* AffixMgr::suffix_check(const std::string& word,
2696 int start,
2697 int len,
2698 int sfxopts,
2699 PfxEntry* ppfx,
2700 const FLAGunsigned short cclass,
2701 const FLAGunsigned short needflag,
2702 char in_compound) {
2703 struct hentry* rv = nullptr;
2704 PfxEntry* ep = ppfx;
2705
2706 // first handle the special case of 0 length suffixes
2707 SfxEntry* se = sStart[0];
2708
2709 while (se) {
2710 if (!cclass || se->getCont()) {
2711 // suffixes are not allowed in beginning of compounds
2712 if ((((in_compound != IN_CPD_BEGIN1)) || // && !cclass
2713 // except when signed with compoundpermitflag flag
2714 (se->getCont() && compoundpermitflag &&
2715 TESTAFF(se->getCont(), compoundpermitflag, se->getContLen())(std::binary_search(se->getCont(), se->getCont() + se->
getContLen(), compoundpermitflag))
)) &&
2716 (!circumfix ||
2717 // no circumfix flag in prefix and suffix
2718 ((!ppfx || !(ep->getCont()) ||
2719 !TESTAFF(ep->getCont(), circumfix, ep->getContLen())(std::binary_search(ep->getCont(), ep->getCont() + ep->
getContLen(), circumfix))
) &&
2720 (!se->getCont() ||
2721 !(TESTAFF(se->getCont(), circumfix, se->getContLen())(std::binary_search(se->getCont(), se->getCont() + se->
getContLen(), circumfix))
))) ||
2722 // circumfix flag in prefix AND suffix
2723 ((ppfx && (ep->getCont()) &&
2724 TESTAFF(ep->getCont(), circumfix, ep->getContLen())(std::binary_search(ep->getCont(), ep->getCont() + ep->
getContLen(), circumfix))
) &&
2725 (se->getCont() &&
2726 (TESTAFF(se->getCont(), circumfix, se->getContLen())(std::binary_search(se->getCont(), se->getCont() + se->
getContLen(), circumfix))
)))) &&
2727 // fogemorpheme
2728 (in_compound ||
2729 !(se->getCont() &&
2730 (TESTAFF(se->getCont(), onlyincompound, se->getContLen())(std::binary_search(se->getCont(), se->getCont() + se->
getContLen(), onlyincompound))
))) &&
2731 // needaffix on prefix or first suffix
2732 (cclass ||
2733 !(se->getCont() &&
2734 TESTAFF(se->getCont(), needaffix, se->getContLen())(std::binary_search(se->getCont(), se->getCont() + se->
getContLen(), needaffix))
) ||
2735 (ppfx &&
2736 !((ep->getCont()) &&
2737 TESTAFF(ep->getCont(), needaffix, ep->getContLen())(std::binary_search(ep->getCont(), ep->getCont() + ep->
getContLen(), needaffix))
)))) {
2738 rv = se->checkword(word, start, len, sfxopts, ppfx,
2739 (FLAGunsigned short)cclass, needflag,
2740 (in_compound ? 0 : onlyincompound));
2741 if (rv) {
2742 sfx = se; // BUG: sfx not stateless
2743 return rv;
2744 }
2745 }
2746 }
2747 se = se->getNext();
2748 }
2749
2750 // now handle the general case
2751 if (len == 0)
2752 return nullptr; // FULLSTRIP
2753 unsigned char sp = word[start + len - 1];
2754 SfxEntry* sptr = sStart[sp];
2755
2756 while (sptr) {
2757 if (isRevSubset(sptr->getKey(), word.c_str() + start + len - 1, len)) {
2758 // suffixes are not allowed in beginning of compounds
2759 if ((((in_compound != IN_CPD_BEGIN1)) || // && !cclass
2760 // except when signed with compoundpermitflag flag
2761 (sptr->getCont() && compoundpermitflag &&
2762 TESTAFF(sptr->getCont(), compoundpermitflag,(std::binary_search(sptr->getCont(), sptr->getCont() + sptr
->getContLen(), compoundpermitflag))
2763 sptr->getContLen())(std::binary_search(sptr->getCont(), sptr->getCont() + sptr
->getContLen(), compoundpermitflag))
)) &&
2764 (!circumfix ||
2765 // no circumfix flag in prefix and suffix
2766 ((!ppfx || !(ep->getCont()) ||
2767 !TESTAFF(ep->getCont(), circumfix, ep->getContLen())(std::binary_search(ep->getCont(), ep->getCont() + ep->
getContLen(), circumfix))
) &&
2768 (!sptr->getCont() ||
2769 !(TESTAFF(sptr->getCont(), circumfix, sptr->getContLen())(std::binary_search(sptr->getCont(), sptr->getCont() + sptr
->getContLen(), circumfix))
))) ||
2770 // circumfix flag in prefix AND suffix
2771 ((ppfx && (ep->getCont()) &&
2772 TESTAFF(ep->getCont(), circumfix, ep->getContLen())(std::binary_search(ep->getCont(), ep->getCont() + ep->
getContLen(), circumfix))
) &&
2773 (sptr->getCont() &&
2774 (TESTAFF(sptr->getCont(), circumfix, sptr->getContLen())(std::binary_search(sptr->getCont(), sptr->getCont() + sptr
->getContLen(), circumfix))
)))) &&
2775 // fogemorpheme
2776 (in_compound ||
2777 !((sptr->getCont() && (TESTAFF(sptr->getCont(), onlyincompound,(std::binary_search(sptr->getCont(), sptr->getCont() + sptr
->getContLen(), onlyincompound))
2778 sptr->getContLen())(std::binary_search(sptr->getCont(), sptr->getCont() + sptr
->getContLen(), onlyincompound))
)))) &&
2779 // needaffix on prefix or first suffix
2780 (cclass ||
2781 !(sptr->getCont() &&
2782 TESTAFF(sptr->getCont(), needaffix, sptr->getContLen())(std::binary_search(sptr->getCont(), sptr->getCont() + sptr
->getContLen(), needaffix))
) ||
2783 (ppfx &&
2784 !((ep->getCont()) &&
2785 TESTAFF(ep->getCont(), needaffix, ep->getContLen())(std::binary_search(ep->getCont(), ep->getCont() + ep->
getContLen(), needaffix))
))))
2786 if (in_compound != IN_CPD_END2 || ppfx ||
2787 !(sptr->getCont() &&
2788 TESTAFF(sptr->getCont(), onlyincompound, sptr->getContLen())(std::binary_search(sptr->getCont(), sptr->getCont() + sptr
->getContLen(), onlyincompound))
)) {
2789 rv = sptr->checkword(word, start, len, sfxopts, ppfx,
2790 cclass, needflag,
2791 (in_compound ? 0 : onlyincompound));
2792 if (rv) {
2793 sfx = sptr; // BUG: sfx not stateless
2794 sfxflag = sptr->getFlag(); // BUG: sfxflag not stateless
2795 if (!sptr->getCont())
2796 sfxappnd = sptr->getKey(); // BUG: sfxappnd not stateless
2797 // LANG_hu section: spec. Hungarian rule
2798 else if (langnum == LANG_hu && sptr->getKeyLen() &&
2799 sptr->getKey()[0] == 'i' && sptr->getKey()[1] != 'y' &&
2800 sptr->getKey()[1] != 't') {
2801 sfxextra = 1;
2802 }
2803 // END of LANG_hu section
2804 return rv;
2805 }
2806 }
2807 sptr = sptr->getNextEQ();
2808 } else {
2809 sptr = sptr->getNextNE();
2810 }
2811 }
2812
2813 return nullptr;
2814}
2815
2816// check word for two-level suffixes
2817struct hentry* AffixMgr::suffix_check_twosfx(const std::string& word,
2818 int start,
2819 int len,
2820 int sfxopts,
2821 PfxEntry* ppfx,
2822 const FLAGunsigned short needflag) {
2823 struct hentry* rv = nullptr;
2824
2825 // first handle the special case of 0 length suffixes
2826 SfxEntry* se = sStart[0];
2827 while (se) {
2828 if (contclasses[se->getFlag()]) {
2829 rv = se->check_twosfx(word, start, len, sfxopts, ppfx, needflag);
2830 if (rv)
2831 return rv;
2832 }
2833 se = se->getNext();
2834 }
2835
2836 // now handle the general case
2837 if (len == 0)
2838 return nullptr; // FULLSTRIP
2839 unsigned char sp = word[start + len - 1];
2840 SfxEntry* sptr = sStart[sp];
2841
2842 while (sptr) {
2843 if (isRevSubset(sptr->getKey(), word.c_str() + start + len - 1, len)) {
2844 if (contclasses[sptr->getFlag()]) {
2845 rv = sptr->check_twosfx(word, start, len, sfxopts, ppfx, needflag);
2846 if (rv) {
2847 sfxflag = sptr->getFlag(); // BUG: sfxflag not stateless
2848 if (!sptr->getCont())
2849 sfxappnd = sptr->getKey(); // BUG: sfxappnd not stateless
2850 return rv;
2851 }
2852 }
2853 sptr = sptr->getNextEQ();
2854 } else {
2855 sptr = sptr->getNextNE();
2856 }
2857 }
2858
2859 return nullptr;
2860}
2861
2862// check word for two-level suffixes and morph
2863std::string AffixMgr::suffix_check_twosfx_morph(const std::string& word,
2864 int start,
2865 int len,
2866 int sfxopts,
2867 PfxEntry* ppfx,
2868 const FLAGunsigned short needflag) {
2869 std::string result;
2870 std::string result2;
2871 std::string result3;
2872
2873 // first handle the special case of 0 length suffixes
2874 SfxEntry* se = sStart[0];
2875 while (se) {
2876 if (contclasses[se->getFlag()]) {
2877 std::string st = se->check_twosfx_morph(word, start, len, sfxopts, ppfx, needflag);
2878 if (!st.empty()) {
2879 if (ppfx) {
2880 if (ppfx->getMorph()) {
2881 result.append(ppfx->getMorph());
2882 result.push_back(MSEP_FLD' ');
2883 } else
2884 debugflag(result, ppfx->getFlag());
2885 }
2886 result.append(st);
2887 if (se->getMorph()) {
2888 result.push_back(MSEP_FLD' ');
2889 result.append(se->getMorph());
2890 } else
2891 debugflag(result, se->getFlag());
2892 result.push_back(MSEP_REC'\n');
2893 }
2894 }
2895 se = se->getNext();
2896 }
2897
2898 // now handle the general case
2899 if (len == 0)
2900 return { }; // FULLSTRIP
2901 unsigned char sp = word[start + len - 1];
2902 SfxEntry* sptr = sStart[sp];
2903
2904 while (sptr) {
2905 if (isRevSubset(sptr->getKey(), word.c_str() + start + len - 1, len)) {
2906 if (contclasses[sptr->getFlag()]) {
2907 std::string st = sptr->check_twosfx_morph(word, start, len, sfxopts, ppfx, needflag);
2908 if (!st.empty()) {
2909 sfxflag = sptr->getFlag(); // BUG: sfxflag not stateless
2910 if (!sptr->getCont())
2911 sfxappnd = sptr->getKey(); // BUG: sfxappnd not stateless
2912 result2.assign(st);
2913
2914 result3.clear();
2915
2916 if (sptr->getMorph()) {
2917 result3.push_back(MSEP_FLD' ');
2918 result3.append(sptr->getMorph());
2919 } else
2920 debugflag(result3, sptr->getFlag());
2921 strlinecat(result2, result3);
2922 result2.push_back(MSEP_REC'\n');
2923 result.append(result2);
2924 }
2925 }
2926 sptr = sptr->getNextEQ();
2927 } else {
2928 sptr = sptr->getNextNE();
2929 }
2930 }
2931
2932 return result;
2933}
2934
2935std::string AffixMgr::suffix_check_morph(const std::string& word,
2936 int start,
2937 int len,
2938 int sfxopts,
2939 PfxEntry* ppfx,
2940 const FLAGunsigned short cclass,
2941 const FLAGunsigned short needflag,
2942 char in_compound) {
2943 std::string result;
2944
2945 struct hentry* rv = nullptr;
2946
2947 PfxEntry* ep = ppfx;
2948
2949 // first handle the special case of 0 length suffixes
2950 SfxEntry* se = sStart[0];
2951 while (se) {
2952 if (!cclass || se->getCont()) {
2953 // suffixes are not allowed in beginning of compounds
2954 if (((((in_compound != IN_CPD_BEGIN1)) || // && !cclass
2955 // except when signed with compoundpermitflag flag
2956 (se->getCont() && compoundpermitflag &&
2957 TESTAFF(se->getCont(), compoundpermitflag, se->getContLen())(std::binary_search(se->getCont(), se->getCont() + se->
getContLen(), compoundpermitflag))
)) &&
2958 (!circumfix ||
2959 // no circumfix flag in prefix and suffix
2960 ((!ppfx || !(ep->getCont()) ||
2961 !TESTAFF(ep->getCont(), circumfix, ep->getContLen())(std::binary_search(ep->getCont(), ep->getCont() + ep->
getContLen(), circumfix))
) &&
2962 (!se->getCont() ||
2963 !(TESTAFF(se->getCont(), circumfix, se->getContLen())(std::binary_search(se->getCont(), se->getCont() + se->
getContLen(), circumfix))
))) ||
2964 // circumfix flag in prefix AND suffix
2965 ((ppfx && (ep->getCont()) &&
2966 TESTAFF(ep->getCont(), circumfix, ep->getContLen())(std::binary_search(ep->getCont(), ep->getCont() + ep->
getContLen(), circumfix))
) &&
2967 (se->getCont() &&
2968 (TESTAFF(se->getCont(), circumfix, se->getContLen())(std::binary_search(se->getCont(), se->getCont() + se->
getContLen(), circumfix))
)))) &&
2969 // fogemorpheme
2970 (in_compound ||
2971 !((se->getCont() &&
2972 (TESTAFF(se->getCont(), onlyincompound, se->getContLen())(std::binary_search(se->getCont(), se->getCont() + se->
getContLen(), onlyincompound))
)))) &&
2973 // needaffix on prefix or first suffix
2974 (cclass ||
2975 !(se->getCont() &&
2976 TESTAFF(se->getCont(), needaffix, se->getContLen())(std::binary_search(se->getCont(), se->getCont() + se->
getContLen(), needaffix))
) ||
2977 (ppfx &&
2978 !((ep->getCont()) &&
2979 TESTAFF(ep->getCont(), needaffix, ep->getContLen())(std::binary_search(ep->getCont(), ep->getCont() + ep->
getContLen(), needaffix))
)))))
2980 rv = se->checkword(word, start, len, sfxopts, ppfx, cclass,
2981 needflag, FLAG_NULL0x00);
2982 while (rv) {
2983 if (ppfx) {
2984 if (ppfx->getMorph()) {
2985 result.append(ppfx->getMorph());
2986 result.push_back(MSEP_FLD' ');
2987 } else
2988 debugflag(result, ppfx->getFlag());
2989 }
2990 if (complexprefixes && HENTRY_DATA(rv))
2991 result.append(HENTRY_DATA2(rv));
2992 if (!HENTRY_FIND(rv, MORPH_STEM"st:")) {
2993 result.push_back(MSEP_FLD' ');
2994 result.append(MORPH_STEM"st:");
2995 result.append(HENTRY_WORD(rv)&(rv->word[0]));
2996 }
2997
2998 if (!complexprefixes && HENTRY_DATA(rv)) {
2999 result.push_back(MSEP_FLD' ');
3000 result.append(HENTRY_DATA2(rv));
3001 }
3002 if (se->getMorph()) {
3003 result.push_back(MSEP_FLD' ');
3004 result.append(se->getMorph());
3005 } else
3006 debugflag(result, se->getFlag());
3007 result.push_back(MSEP_REC'\n');
3008 rv = se->get_next_homonym(rv, sfxopts, ppfx, cclass, needflag);
3009 }
3010 }
3011 se = se->getNext();
3012 }
3013
3014 // now handle the general case
3015 if (len == 0)
3016 return { }; // FULLSTRIP
3017 unsigned char sp = word[start + len - 1];
3018 SfxEntry* sptr = sStart[sp];
3019
3020 while (sptr) {
3021 if (isRevSubset(sptr->getKey(), word.c_str() + start + len - 1, len)) {
3022 // suffixes are not allowed in beginning of compounds
3023 if (((((in_compound != IN_CPD_BEGIN1)) || // && !cclass
3024 // except when signed with compoundpermitflag flag
3025 (sptr->getCont() && compoundpermitflag &&
3026 TESTAFF(sptr->getCont(), compoundpermitflag,(std::binary_search(sptr->getCont(), sptr->getCont() + sptr
->getContLen(), compoundpermitflag))
3027 sptr->getContLen())(std::binary_search(sptr->getCont(), sptr->getCont() + sptr
->getContLen(), compoundpermitflag))
)) &&
3028 (!circumfix ||
3029 // no circumfix flag in prefix and suffix
3030 ((!ppfx || !(ep->getCont()) ||
3031 !TESTAFF(ep->getCont(), circumfix, ep->getContLen())(std::binary_search(ep->getCont(), ep->getCont() + ep->
getContLen(), circumfix))
) &&
3032 (!sptr->getCont() ||
3033 !(TESTAFF(sptr->getCont(), circumfix, sptr->getContLen())(std::binary_search(sptr->getCont(), sptr->getCont() + sptr
->getContLen(), circumfix))
))) ||
3034 // circumfix flag in prefix AND suffix
3035 ((ppfx && (ep->getCont()) &&
3036 TESTAFF(ep->getCont(), circumfix, ep->getContLen())(std::binary_search(ep->getCont(), ep->getCont() + ep->
getContLen(), circumfix))
) &&
3037 (sptr->getCont() &&
3038 (TESTAFF(sptr->getCont(), circumfix, sptr->getContLen())(std::binary_search(sptr->getCont(), sptr->getCont() + sptr
->getContLen(), circumfix))
)))) &&
3039 // fogemorpheme
3040 (in_compound ||
3041 !((sptr->getCont() && (TESTAFF(sptr->getCont(), onlyincompound,(std::binary_search(sptr->getCont(), sptr->getCont() + sptr
->getContLen(), onlyincompound))
3042 sptr->getContLen())(std::binary_search(sptr->getCont(), sptr->getCont() + sptr
->getContLen(), onlyincompound))
)))) &&
3043 // needaffix on first suffix
3044 (cclass ||
3045 !(sptr->getCont() &&
3046 TESTAFF(sptr->getCont(), needaffix, sptr->getContLen())(std::binary_search(sptr->getCont(), sptr->getCont() + sptr
->getContLen(), needaffix))
))))
3047 rv = sptr->checkword(word, start, len, sfxopts, ppfx, cclass,
3048 needflag, FLAG_NULL0x00);
3049 while (rv) {
3050 if (ppfx) {
3051 if (ppfx->getMorph()) {
3052 result.append(ppfx->getMorph());
3053 result.push_back(MSEP_FLD' ');
3054 } else
3055 debugflag(result, ppfx->getFlag());
3056 }
3057 if (complexprefixes && HENTRY_DATA(rv))
3058 result.append(HENTRY_DATA2(rv));
3059 if (!HENTRY_FIND(rv, MORPH_STEM"st:")) {
3060 result.push_back(MSEP_FLD' ');
3061 result.append(MORPH_STEM"st:");
3062 result.append(HENTRY_WORD(rv)&(rv->word[0]));
3063 }
3064
3065 if (!complexprefixes && HENTRY_DATA(rv)) {
3066 result.push_back(MSEP_FLD' ');
3067 result.append(HENTRY_DATA2(rv));
3068 }
3069
3070 if (sptr->getMorph()) {
3071 result.push_back(MSEP_FLD' ');
3072 result.append(sptr->getMorph());
3073 } else
3074 debugflag(result, sptr->getFlag());
3075 result.push_back(MSEP_REC'\n');
3076 rv = sptr->get_next_homonym(rv, sfxopts, ppfx, cclass, needflag);
3077 }
3078 sptr = sptr->getNextEQ();
3079 } else {
3080 sptr = sptr->getNextNE();
3081 }
3082 }
3083
3084 return result;
3085}
3086
3087// check if word with affixes is correctly spelled
3088struct hentry* AffixMgr::affix_check(const std::string& word,
3089 int start,
3090 int len,
3091 const FLAGunsigned short needflag,
3092 char in_compound) {
3093
3094 // check all prefixes (also crossed with suffixes if allowed)
3095 struct hentry* rv = prefix_check(word, start, len, in_compound, needflag);
3096 if (rv)
3097 return rv;
3098
3099 // if still not found check all suffixes
3100 rv = suffix_check(word, start, len, 0, nullptr, FLAG_NULL0x00, needflag, in_compound);
3101
3102 if (havecontclass) {
3103 sfx = nullptr;
3104 pfx = nullptr;
3105
3106 if (rv)
3107 return rv;
3108 // if still not found check all two-level suffixes
3109 rv = suffix_check_twosfx(word, start, len, 0, nullptr, needflag);
3110
3111 if (rv)
3112 return rv;
3113 // if still not found check all two-level suffixes
3114 rv = prefix_check_twosfx(word, start, len, IN_CPD_NOT0, needflag);
3115 }
3116
3117 return rv;
3118}
3119
3120// check if word with affixes is correctly spelled
3121std::string AffixMgr::affix_check_morph(const std::string& word,
3122 int start,
3123 int len,
3124 const FLAGunsigned short needflag,
3125 char in_compound) {
3126 std::string result;
3127
3128 // check all prefixes (also crossed with suffixes if allowed)
3129 std::string st = prefix_check_morph(word, start, len, in_compound);
3130 if (!st.empty()) {
3131 result.append(st);
3132 }
3133
3134 // if still not found check all suffixes
3135 st = suffix_check_morph(word, start, len, 0, nullptr, '\0', needflag, in_compound);
3136 if (!st.empty()) {
3137 result.append(st);
3138 }
3139
3140 if (havecontclass) {
3141 sfx = nullptr;
3142 pfx = nullptr;
3143 // if still not found check all two-level suffixes
3144 st = suffix_check_twosfx_morph(word, start, len, 0, nullptr, needflag);
3145 if (!st.empty()) {
3146 result.append(st);
3147 }
3148
3149 // if still not found check all two-level suffixes
3150 st = prefix_check_twosfx_morph(word, start, len, IN_CPD_NOT0, needflag);
3151 if (!st.empty()) {
3152 result.append(st);
3153 }
3154 }
3155
3156 return result;
3157}
3158
3159// morphcmp(): compare MORPH_DERI_SFX, MORPH_INFL_SFX and MORPH_TERM_SFX fields
3160// in the first line of the inputs
3161// return 0, if inputs equal
3162// return 1, if inputs may equal with a secondary suffix
3163// otherwise return -1
3164static int morphcmp(const char* s, const char* t) {
3165 int se = 0, te = 0;
3166 const char* sl;
3167 const char* tl;
3168 const char* olds;
3169 const char* oldt;
3170 if (!s || !t)
3171 return 1;
3172 olds = s;
3173 sl = strchr(s, '\n');
3174 s = strstr(s, MORPH_DERI_SFX"ds:");
3175 if (!s || (sl && sl < s))
3176 s = strstr(olds, MORPH_INFL_SFX"is:");
3177 if (!s || (sl && sl < s)) {
3178 s = strstr(olds, MORPH_TERM_SFX"ts:");
3179 olds = nullptr;
3180 }
3181 oldt = t;
3182 tl = strchr(t, '\n');
3183 t = strstr(t, MORPH_DERI_SFX"ds:");
3184 if (!t || (tl && tl < t))
3185 t = strstr(oldt, MORPH_INFL_SFX"is:");
3186 if (!t || (tl && tl < t))
3187 t = strstr(oldt, MORPH_TERM_SFX"ts:");
3188 while (s && t && (!sl || sl > s) && (!tl || tl > t)) {
3189 s += MORPH_TAG_LENstrlen("st:");
3190 t += MORPH_TAG_LENstrlen("st:");
3191 se = 0;
3192 te = 0;
3193 while ((*s == *t) && !se && !te) {
3194 s++;
3195 t++;
3196 switch (*s) {
3197 case ' ':
3198 case '\n':
3199 case '\t':
3200 case '\0':
3201 se = 1;
3202 }
3203 switch (*t) {
3204 case ' ':
3205 case '\n':
3206 case '\t':
3207 case '\0':
3208 te = 1;
3209 }
3210 }
3211 if (!se || !te) {
3212 // not terminal suffix difference
3213 if (olds)
3214 return -1;
3215 return 1;
3216 }
3217 olds = s;
3218 s = strstr(s, MORPH_DERI_SFX"ds:");
3219 if (!s || (sl && sl < s))
3220 s = strstr(olds, MORPH_INFL_SFX"is:");
3221 if (!s || (sl && sl < s)) {
3222 s = strstr(olds, MORPH_TERM_SFX"ts:");
3223 olds = nullptr;
3224 }
3225 oldt = t;
3226 t = strstr(t, MORPH_DERI_SFX"ds:");
3227 if (!t || (tl && tl < t))
3228 t = strstr(oldt, MORPH_INFL_SFX"is:");
3229 if (!t || (tl && tl < t))
3230 t = strstr(oldt, MORPH_TERM_SFX"ts:");
3231 }
3232 if (!s && !t && se && te)
3233 return 0;
3234 return 1;
3235}
3236
3237std::string AffixMgr::morphgen(const char* ts,
3238 int wl,
3239 const unsigned short* ap,
3240 unsigned short al,
3241 const char* morph,
3242 const char* targetmorph,
3243 int level) {
3244 // handle suffixes
3245 if (!morph)
3246 return {};
3247
3248 // check substandard flag
3249 if (TESTAFF(ap, substandard, al)(std::binary_search(ap, ap + al, substandard)))
3250 return {};
3251
3252 if (morphcmp(morph, targetmorph) == 0)
3253 return ts;
3254
3255 size_t stemmorphcatpos;
3256 std::string mymorph;
3257
3258 // use input suffix fields, if exist
3259 if (strstr(morph, MORPH_INFL_SFX"is:") || strstr(morph, MORPH_DERI_SFX"ds:")) {
3260 mymorph.assign(morph);
3261 mymorph.push_back(MSEP_FLD' ');
3262 stemmorphcatpos = mymorph.size();
3263 } else {
3264 stemmorphcatpos = std::string::npos;
3265 }
3266
3267 for (int i = 0; i < al; i++) {
3268 const auto c = (unsigned char)(ap[i] & 0x00FF);
3269 SfxEntry* sptr = sFlag[c];
3270 while (sptr) {
3271 if (sptr->getFlag() == ap[i] && sptr->getMorph() &&
3272 ((sptr->getContLen() == 0) ||
3273 // don't generate forms with substandard affixes
3274 !TESTAFF(sptr->getCont(), substandard, sptr->getContLen())(std::binary_search(sptr->getCont(), sptr->getCont() + sptr
->getContLen(), substandard))
)) {
3275 const char* stemmorph;
3276 if (stemmorphcatpos != std::string::npos) {
3277 mymorph.replace(stemmorphcatpos, std::string::npos, sptr->getMorph());
3278 stemmorph = mymorph.c_str();
3279 } else {
3280 stemmorph = sptr->getMorph();
3281 }
3282
3283 int cmp = morphcmp(stemmorph, targetmorph);
3284
3285 if (cmp == 0) {
3286 std::string newword = sptr->add(ts, wl);
3287 if (!newword.empty()) {
3288 hentry* check = pHMgr->lookup(newword.c_str(), newword.size()); // XXX extra dic
3289 if (!check || !check->astr ||
3290 !(TESTAFF(check->astr, forbiddenword, check->alen)(std::binary_search(check->astr, check->astr + check->
alen, forbiddenword))
||
3291 TESTAFF(check->astr, ONLYUPCASEFLAG, check->alen)(std::binary_search(check->astr, check->astr + check->
alen, 65511))
)) {
3292 return newword;
3293 }
3294 }
3295 }
3296
3297 // recursive call for secondary suffixes
3298 if ((level == 0) && (cmp == 1) && (sptr->getContLen() > 0) &&
3299 !TESTAFF(sptr->getCont(), substandard, sptr->getContLen())(std::binary_search(sptr->getCont(), sptr->getCont() + sptr
->getContLen(), substandard))
) {
3300 std::string newword = sptr->add(ts, wl);
3301 if (!newword.empty()) {
3302 std::string newword2 =
3303 morphgen(newword.c_str(), newword.size(), sptr->getCont(),
3304 sptr->getContLen(), stemmorph, targetmorph, 1);
3305
3306 if (!newword2.empty()) {
3307 return newword2;
3308 }
3309 }
3310 }
3311 }
3312 sptr = sptr->getFlgNxt();
3313 }
3314 }
3315 return { };
3316}
3317
3318namespace {
3319 // replaces strdup with ansi version
3320 char* mystrdup(const char* s) {
3321 char* d = nullptr;
3322 if (s) {
3323 size_t sl = strlen(s) + 1;
3324 d = new char[sl];
3325 memcpy(d, s, sl);
3326 }
3327 return d;
3328 }
3329}
3330
3331int AffixMgr::expand_rootword(struct guessword* wlst,
3332 int maxn,
3333 const char* ts,
3334 int wl,
3335 const unsigned short* ap,
3336 unsigned short al,
3337 const char* bad,
3338 int badl,
3339 const char* phon) {
3340 int nh = 0;
3341 // first add root word to list
3342 if ((nh < maxn) &&
3343 !(al && ((needaffix && TESTAFF(ap, needaffix, al)(std::binary_search(ap, ap + al, needaffix))) ||
3344 (onlyincompound && TESTAFF(ap, onlyincompound, al)(std::binary_search(ap, ap + al, onlyincompound)))))) {
3345 wlst[nh].word = mystrdup(ts);
3346 wlst[nh].allow = false;
3347 wlst[nh].orig = nullptr;
3348 nh++;
3349 // add special phonetic version
3350 if (phon && (nh < maxn)) {
3351 wlst[nh].word = mystrdup(phon);
3352 wlst[nh].allow = false;
3353 wlst[nh].orig = mystrdup(ts);
3354 nh++;
3355 }
3356 }
3357
3358 // handle suffixes
3359 for (int i = 0; i < al; i++) {
3360 const auto c = (unsigned char)(ap[i] & 0x00FF);
3361 SfxEntry* sptr = sFlag[c];
3362 while (sptr) {
3363 if ((sptr->getFlag() == ap[i]) &&
3364 (!sptr->getKeyLen() ||
3365 ((badl > sptr->getKeyLen()) &&
3366 (strcmp(sptr->getAffix(), bad + badl - sptr->getKeyLen()) == 0))) &&
3367 // check needaffix flag
3368 !(sptr->getCont() &&
3369 ((needaffix &&
3370 TESTAFF(sptr->getCont(), needaffix, sptr->getContLen())(std::binary_search(sptr->getCont(), sptr->getCont() + sptr
->getContLen(), needaffix))
) ||
3371 (circumfix &&
3372 TESTAFF(sptr->getCont(), circumfix, sptr->getContLen())(std::binary_search(sptr->getCont(), sptr->getCont() + sptr
->getContLen(), circumfix))
) ||
3373 (onlyincompound &&
3374 TESTAFF(sptr->getCont(), onlyincompound, sptr->getContLen())(std::binary_search(sptr->getCont(), sptr->getCont() + sptr
->getContLen(), onlyincompound))
)))) {
3375 std::string newword = sptr->add(ts, wl);
3376 if (!newword.empty()) {
3377 if (nh < maxn) {
3378 wlst[nh].word = mystrdup(newword.c_str());
3379 wlst[nh].allow = sptr->allowCross();
3380 wlst[nh].orig = nullptr;
3381 nh++;
3382 // add special phonetic version
3383 if (phon && (nh < maxn)) {
3384 std::string prefix(phon);
3385 std::string key(sptr->getKey());
3386 reverseword(key);
3387 prefix.append(key);
3388 wlst[nh].word = mystrdup(prefix.c_str());
3389 wlst[nh].allow = false;
3390 wlst[nh].orig = mystrdup(newword.c_str());
3391 nh++;
3392 }
3393 }
3394 }
3395 }
3396 sptr = sptr->getFlgNxt();
3397 }
3398 }
3399
3400 int n = nh;
3401
3402 // handle cross products of prefixes and suffixes
3403 for (int j = 1; j < n; j++)
3404 if (wlst[j].allow) {
3405 for (int k = 0; k < al; k++) {
3406 const auto c = (unsigned char)(ap[k] & 0x00FF);
3407 PfxEntry* cptr = pFlag[c];
3408 while (cptr) {
3409 if ((cptr->getFlag() == ap[k]) && cptr->allowCross() &&
3410 (!cptr->getKeyLen() ||
3411 ((badl > cptr->getKeyLen()) &&
3412 (strncmp(cptr->getKey(), bad, cptr->getKeyLen()) == 0)))) {
3413 int l1 = strlen(wlst[j].word);
3414 std::string newword = cptr->add(wlst[j].word, l1);
3415 if (!newword.empty()) {
3416 if (nh < maxn) {
3417 wlst[nh].word = mystrdup(newword.c_str());
3418 wlst[nh].allow = cptr->allowCross();
3419 wlst[nh].orig = nullptr;
3420 nh++;
3421 }
3422 }
3423 }
3424 cptr = cptr->getFlgNxt();
3425 }
3426 }
3427 }
3428
3429 // now handle pure prefixes
3430 for (int m = 0; m < al; m++) {
3431 const auto c = (unsigned char)(ap[m] & 0x00FF);
3432 PfxEntry* ptr = pFlag[c];
3433 while (ptr) {
3434 if ((ptr->getFlag() == ap[m]) &&
3435 (!ptr->getKeyLen() ||
3436 ((badl > ptr->getKeyLen()) &&
3437 (strncmp(ptr->getKey(), bad, ptr->getKeyLen()) == 0))) &&
3438 // check needaffix flag
3439 !(ptr->getCont() &&
3440 ((needaffix &&
3441 TESTAFF(ptr->getCont(), needaffix, ptr->getContLen())(std::binary_search(ptr->getCont(), ptr->getCont() + ptr
->getContLen(), needaffix))
) ||
3442 (circumfix &&
3443 TESTAFF(ptr->getCont(), circumfix, ptr->getContLen())(std::binary_search(ptr->getCont(), ptr->getCont() + ptr
->getContLen(), circumfix))
) ||
3444 (onlyincompound &&
3445 TESTAFF(ptr->getCont(), onlyincompound, ptr->getContLen())(std::binary_search(ptr->getCont(), ptr->getCont() + ptr
->getContLen(), onlyincompound))
)))) {
3446 std::string newword = ptr->add(ts, wl);
3447 if (!newword.empty()) {
3448 if (nh < maxn) {
3449 wlst[nh].word = mystrdup(newword.c_str());
3450 wlst[nh].allow = ptr->allowCross();
3451 wlst[nh].orig = nullptr;
3452 nh++;
3453 }
3454 }
3455 }
3456 ptr = ptr->getFlgNxt();
3457 }
3458 }
3459
3460 return nh;
3461}
3462
3463// return replacing table
3464const std::vector<replentry>& AffixMgr::get_reptable() const {
3465 return pHMgr->get_reptable();
3466}
3467
3468// return iconv table
3469RepList* AffixMgr::get_iconvtable() const {
3470 if (!iconvtable)
3471 return nullptr;
3472 return iconvtable;
3473}
3474
3475// return oconv table
3476RepList* AffixMgr::get_oconvtable() const {
3477 if (!oconvtable)
3478 return nullptr;
3479 return oconvtable;
3480}
3481
3482// return replacing table
3483struct phonetable* AffixMgr::get_phonetable() const {
3484 if (!phone)
3485 return nullptr;
3486 return phone;
3487}
3488
3489// return character map table
3490const std::vector<mapentry>& AffixMgr::get_maptable() const {
3491 return maptable;
3492}
3493
3494// return character map table
3495const std::vector<std::string>& AffixMgr::get_breaktable() const {
3496 return breaktable;
3497}
3498
3499// return text encoding of dictionary
3500const std::string& AffixMgr::get_encoding() {
3501 if (encoding.empty())
3502 encoding = SPELL_ENCODING"ISO8859-1";
3503 return encoding;
3504}
3505
3506// return text encoding of dictionary
3507int AffixMgr::get_langnum() const {
3508 return langnum;
3509}
3510
3511// return double prefix option
3512int AffixMgr::get_complexprefixes() const {
3513 return complexprefixes;
3514}
3515
3516// return FULLSTRIP option
3517int AffixMgr::get_fullstrip() const {
3518 return fullstrip;
3519}
3520
3521FLAGunsigned short AffixMgr::get_keepcase() const {
3522 return keepcase;
3523}
3524
3525FLAGunsigned short AffixMgr::get_forceucase() const {
3526 return forceucase;
3527}
3528
3529FLAGunsigned short AffixMgr::get_warn() const {
3530 return warn;
3531}
3532
3533int AffixMgr::get_forbidwarn() const {
3534 return forbidwarn;
3535}
3536
3537int AffixMgr::get_checksharps() const {
3538 return checksharps;
3539}
3540
3541std::string AffixMgr::encode_flag(unsigned short aflag) const {
3542 return pHMgr->encode_flag(aflag);
3543}
3544
3545// return the preferred ignore string for suggestions
3546const char* AffixMgr::get_ignore() const {
3547 if (ignorechars.empty())
3548 return nullptr;
3549 return ignorechars.c_str();
3550}
3551
3552// return the preferred ignore string for suggestions
3553const std::vector<w_char>& AffixMgr::get_ignore_utf16() const {
3554 return ignorechars_utf16;
3555}
3556
3557// return the keyboard string for suggestions
3558const std::string& AffixMgr::get_key_string() {
3559 if (keystring.empty())
3560 keystring = SPELL_KEYSTRING"qwertyuiop|asdfghjkl|zxcvbnm";
3561 return keystring;
3562}
3563
3564// return the preferred try string for suggestions
3565const std::string& AffixMgr::get_try_string() const {
3566 return trystring;
3567}
3568
3569// return the preferred try string for suggestions
3570const std::string& AffixMgr::get_wordchars() const {
3571 return wordchars;
3572}
3573
3574const std::vector<w_char>& AffixMgr::get_wordchars_utf16() const {
3575 return wordchars_utf16;
3576}
3577
3578// is there compounding?
3579int AffixMgr::get_compound() const {
3580 return compoundflag || compoundbegin || !defcpdtable.empty();
3581}
3582
3583// return the compound words control flag
3584FLAGunsigned short AffixMgr::get_compoundflag() const {
3585 return compoundflag;
3586}
3587
3588// return the forbidden words control flag
3589FLAGunsigned short AffixMgr::get_forbiddenword() const {
3590 return forbiddenword;
3591}
3592
3593// return the forbidden words control flag
3594FLAGunsigned short AffixMgr::get_nosuggest() const {
3595 return nosuggest;
3596}
3597
3598// return the forbidden words control flag
3599FLAGunsigned short AffixMgr::get_nongramsuggest() const {
3600 return nongramsuggest;
3601}
3602
3603// return the substandard root/affix control flag
3604FLAGunsigned short AffixMgr::get_substandard() const {
3605 return substandard;
3606}
3607
3608// return the forbidden words flag modify flag
3609FLAGunsigned short AffixMgr::get_needaffix() const {
3610 return needaffix;
3611}
3612
3613// return the onlyincompound flag
3614FLAGunsigned short AffixMgr::get_onlyincompound() const {
3615 return onlyincompound;
3616}
3617
3618// return the value of suffix
3619const std::string& AffixMgr::get_version() const {
3620 return version;
3621}
3622
3623// utility method to look up root words in hash table
3624struct hentry* AffixMgr::lookup(const char* word, size_t len) {
3625 struct hentry* he = nullptr;
3626 for (size_t i = 0; i < alldic.size() && !he; ++i) {
3627 he = alldic[i]->lookup(word, len);
3628 }
3629 return he;
3630}
3631
3632// return the value of suffix
3633int AffixMgr::have_contclass() const {
3634 return havecontclass;
3635}
3636
3637// return utf8
3638int AffixMgr::get_utf8() const {
3639 return utf8;
3640}
3641
3642int AffixMgr::get_maxngramsugs() const {
3643 return maxngramsugs;
3644}
3645
3646int AffixMgr::get_maxcpdsugs() const {
3647 return maxcpdsugs;
3648}
3649
3650int AffixMgr::get_maxdiff() const {
3651 return maxdiff;
3652}
3653
3654int AffixMgr::get_onlymaxdiff() const {
3655 return onlymaxdiff;
3656}
3657
3658// return nosplitsugs
3659int AffixMgr::get_nosplitsugs() const {
3660 return nosplitsugs;
3661}
3662
3663// return sugswithdots
3664int AffixMgr::get_sugswithdots() const {
3665 return sugswithdots;
3666}
3667
3668/* parse flag */
3669bool AffixMgr::parse_flag(const std::string& line, unsigned short* out, FileMgr* af) {
3670 if (*out != FLAG_NULL0x00 && !(*out >= DEFAULTFLAGS65510)) {
3671 HUNSPELL_WARNING(
3672 stderrstderr,
3673 "error: line %d: multiple definitions of an affix file parameter\n",
3674 af->getlinenum());
3675 return false;
3676 }
3677 std::string s;
3678 if (!parse_string(line, s, af->getlinenum()))
3679 return false;
3680 *out = pHMgr->decode_flag(s);
3681 return true;
3682}
3683
3684/* parse num */
3685bool AffixMgr::parse_num(const std::string& line, int* out, FileMgr* af) {
3686 if (*out != -1) {
3687 HUNSPELL_WARNING(
3688 stderrstderr,
3689 "error: line %d: multiple definitions of an affix file parameter\n",
3690 af->getlinenum());
3691 return false;
3692 }
3693 std::string s;
3694 if (!parse_string(line, s, af->getlinenum()))
3695 return false;
3696 *out = atoi(s.c_str());
3697 return true;
3698}
3699
3700/* parse in the max syllablecount of compound words and */
3701bool AffixMgr::parse_cpdsyllable(const std::string& line, FileMgr* af) {
3702 int i = 0;
3703 int np = 0;
3704 auto iter = line.begin(), start_piece = mystrsep(line, iter);
3705 while (start_piece != line.end()) {
3706 switch (i) {
3707 case 0: {
3708 np++;
3709 break;
3710 }
3711 case 1: {
3712 cpdmaxsyllable = atoi(std::string(start_piece, iter).c_str());
3713 np++;
3714 break;
3715 }
3716 case 2: {
3717 if (!utf8) {
3718 cpdvowels.assign(start_piece, iter);
3719 std::sort(cpdvowels.begin(), cpdvowels.end());
3720 } else {
3721 std::string piece(start_piece, iter);
3722 u8_u16(cpdvowels_utf16, piece);
3723 std::sort(cpdvowels_utf16.begin(), cpdvowels_utf16.end());
3724 }
3725 np++;
3726 break;
3727 }
3728 default:
3729 break;
3730 }
3731 ++i;
3732 start_piece = mystrsep(line, iter);
3733 }
3734 if (np < 2) {
3735 HUNSPELL_WARNING(stderrstderr,
3736 "error: line %d: missing compoundsyllable information\n",
3737 af->getlinenum());
3738 return false;
3739 }
3740 if (np == 2)
3741 cpdvowels = "AEIOUaeiou";
3742 return true;
3743}
3744
3745bool AffixMgr::parse_convtable(const std::string& line,
3746 FileMgr* af,
3747 RepList** rl,
3748 const std::string& keyword) {
3749 if (*rl) {
3750 HUNSPELL_WARNING(stderrstderr, "error: line %d: multiple table definitions\n",
3751 af->getlinenum());
3752 return false;
3753 }
3754 int i = 0;
3755 int np = 0;
3756 int numrl = 0;
3757 auto iter = line.begin(), start_piece = mystrsep(line, iter);
3758 while (start_piece != line.end()) {
3759 switch (i) {
3760 case 0: {
3761 np++;
3762 break;
3763 }
3764 case 1: {
3765 numrl = atoi(std::string(start_piece, iter).c_str());
3766 if (numrl < 1) {
3767 HUNSPELL_WARNING(stderrstderr, "error: line %d: incorrect entry number\n",
3768 af->getlinenum());
3769 return false;
3770 }
3771 *rl = new RepList(numrl);
3772 if (!*rl)
3773 return false;
3774 np++;
3775 break;
3776 }
3777 default:
3778 break;
3779 }
3780 ++i;
3781 start_piece = mystrsep(line, iter);
3782 }
3783 if (np != 2) {
3784 HUNSPELL_WARNING(stderrstderr, "error: line %d: missing data\n",
3785 af->getlinenum());
3786 return false;
3787 }
3788
3789 /* now parse the num lines to read in the remainder of the table */
3790 for (int j = 0; j < numrl; j++) {
3791 std::string nl;
3792 if (!af->getline(nl))
3793 return false;
3794 mychomp(nl);
3795 i = 0;
3796 std::string pattern;
3797 std::string pattern2;
3798 iter = nl.begin();
3799 start_piece = mystrsep(nl, iter);
3800 while (start_piece != nl.end()) {
3801 {
3802 switch (i) {
3803 case 0: {
3804 if (nl.compare(start_piece - nl.begin(), keyword.size(), keyword, 0, keyword.size()) != 0) {
3805 HUNSPELL_WARNING(stderrstderr, "error: line %d: table is corrupt\n",
3806 af->getlinenum());
3807 delete *rl;
3808 *rl = nullptr;
3809 return false;
3810 }
3811 break;
3812 }
3813 case 1: {
3814 pattern.assign(start_piece, iter);
3815 break;
3816 }
3817 case 2: {
3818 pattern2.assign(start_piece, iter);
3819 break;
3820 }
3821 default:
3822 break;
3823 }
3824 ++i;
3825 }
3826 start_piece = mystrsep(nl, iter);
3827 }
3828 if (pattern.empty() || pattern2.empty()) {
3829 HUNSPELL_WARNING(stderrstderr, "error: line %d: table is corrupt\n",
3830 af->getlinenum());
3831 return false;
3832 }
3833
3834 (*rl)->add(pattern, pattern2);
3835 }
3836 return true;
3837}
3838
3839/* parse in the typical fault correcting table */
3840bool AffixMgr::parse_phonetable(const std::string& line, FileMgr* af) {
3841 if (phone) {
3842 HUNSPELL_WARNING(stderrstderr, "error: line %d: multiple table definitions\n",
3843 af->getlinenum());
3844 return false;
3845 }
3846 std::unique_ptr<phonetable> new_phone;
3847 int num = -1;
3848 int i = 0;
3849 int np = 0;
3850 auto iter = line.begin(), start_piece = mystrsep(line, iter);
3851 while (start_piece != line.end()) {
3852 switch (i) {
3853 case 0: {
3854 np++;
3855 break;
3856 }
3857 case 1: {
3858 num = atoi(std::string(start_piece, iter).c_str());
3859 if (num < 1) {
3860 HUNSPELL_WARNING(stderrstderr, "error: line %d: bad entry number\n",
3861 af->getlinenum());
3862 return false;
3863 }
3864 new_phone.reset(new phonetable);
3865 new_phone->utf8 = (char)utf8;
3866 np++;
3867 break;
3868 }
3869 default:
3870 break;
3871 }
3872 ++i;
3873 start_piece = mystrsep(line, iter);
3874 }
3875 if (np != 2) {
3876 HUNSPELL_WARNING(stderrstderr, "error: line %d: missing data\n",
3877 af->getlinenum());
3878 return false;
3879 }
3880
3881 /* now parse the phone->num lines to read in the remainder of the table */
3882 for (int j = 0; j < num; ++j) {
3883 std::string nl;
3884 if (!af->getline(nl))
3885 return false;
3886 mychomp(nl);
3887 i = 0;
3888 const size_t old_size = new_phone->rules.size();
3889 iter = nl.begin();
3890 start_piece = mystrsep(nl, iter);
3891 while (start_piece != nl.end()) {
3892 {
3893 switch (i) {
3894 case 0: {
3895 if (nl.compare(start_piece - nl.begin(), 5, "PHONE", 5) != 0) {
3896 HUNSPELL_WARNING(stderrstderr, "error: line %d: table is corrupt\n",
3897 af->getlinenum());
3898 return false;
3899 }
3900 break;
3901 }
3902 case 1: {
3903 new_phone->rules.emplace_back(start_piece, iter);
3904 break;
3905 }
3906 case 2: {
3907 new_phone->rules.emplace_back(start_piece, iter);
3908 mystrrep(new_phone->rules.back(), "_", "");
3909 break;
3910 }
3911 default:
3912 break;
3913 }
3914 ++i;
3915 }
3916 start_piece = mystrsep(nl, iter);
3917 }
3918 if (new_phone->rules.size() != old_size + 2) {
3919 HUNSPELL_WARNING(stderrstderr, "error: line %d: table is corrupt\n",
3920 af->getlinenum());
3921 return false;
3922 }
3923 }
3924 new_phone->rules.emplace_back("");
3925 new_phone->rules.emplace_back("");
3926 init_phonet_hash(*new_phone);
3927 phone = new_phone.release();
3928 return true;
3929}
3930
3931/* parse in the checkcompoundpattern table */
3932bool AffixMgr::parse_checkcpdtable(const std::string& line, FileMgr* af) {
3933 if (parsedcheckcpd) {
3934 HUNSPELL_WARNING(stderrstderr, "error: line %d: multiple table definitions\n",
3935 af->getlinenum());
3936 return false;
3937 }
3938 parsedcheckcpd = true;
3939 int numcheckcpd = -1;
3940 int i = 0;
3941 int np = 0;
3942 auto iter = line.begin(), start_piece = mystrsep(line, iter);
3943 while (start_piece != line.end()) {
3944 switch (i) {
3945 case 0: {
3946 np++;
3947 break;
3948 }
3949 case 1: {
3950 numcheckcpd = atoi(std::string(start_piece, iter).c_str());
3951 if (numcheckcpd < 1) {
3952 HUNSPELL_WARNING(stderrstderr, "error: line %d: bad entry number\n",
3953 af->getlinenum());
3954 return false;
3955 }
3956 checkcpdtable.reserve(std::min(numcheckcpd, 16384));
3957 np++;
3958 break;
3959 }
3960 default:
3961 break;
3962 }
3963 ++i;
3964 start_piece = mystrsep(line, iter);
3965 }
3966 if (np != 2) {
3967 HUNSPELL_WARNING(stderrstderr, "error: line %d: missing data\n",
3968 af->getlinenum());
3969 return false;
3970 }
3971
3972 /* now parse the numcheckcpd lines to read in the remainder of the table */
3973 for (int j = 0; j < numcheckcpd; ++j) {
3974 std::string nl;
3975 if (!af->getline(nl))
3976 return false;
3977 mychomp(nl);
3978 i = 0;
3979 checkcpdtable.emplace_back();
3980 iter = nl.begin();
3981 start_piece = mystrsep(nl, iter);
3982 while (start_piece != nl.end()) {
3983 switch (i) {
3984 case 0: {
3985 if (nl.compare(start_piece - nl.begin(), 20, "CHECKCOMPOUNDPATTERN", 20) != 0) {
3986 HUNSPELL_WARNING(stderrstderr, "error: line %d: table is corrupt\n",
3987 af->getlinenum());
3988 checkcpdtable.clear();
3989 return false;
3990 }
3991 break;
3992 }
3993 case 1: {
3994 checkcpdtable.back().pattern.assign(start_piece, iter);
3995 size_t slash_pos = checkcpdtable.back().pattern.find('/');
3996 if (slash_pos != std::string::npos) {
3997 std::string chunk(checkcpdtable.back().pattern, slash_pos + 1);
3998 checkcpdtable.back().pattern.resize(slash_pos);
3999 checkcpdtable.back().cond = pHMgr->decode_flag(chunk);
4000 }
4001 break;
4002 }
4003 case 2: {
4004 checkcpdtable.back().pattern2.assign(start_piece, iter);
4005 size_t slash_pos = checkcpdtable.back().pattern2.find('/');
4006 if (slash_pos != std::string::npos) {
4007 std::string chunk(checkcpdtable.back().pattern2, slash_pos + 1);
4008 checkcpdtable.back().pattern2.resize(slash_pos);
4009 checkcpdtable.back().cond2 = pHMgr->decode_flag(chunk);
4010 }
4011 break;
4012 }
4013 case 3: {
4014 checkcpdtable.back().pattern3.assign(start_piece, iter);
4015 simplifiedcpd = 1;
4016 break;
4017 }
4018 default:
4019 break;
4020 }
4021 i++;
4022 start_piece = mystrsep(nl, iter);
4023 }
4024 }
4025 return true;
4026}
4027
4028/* parse in the compound rule table */
4029bool AffixMgr::parse_defcpdtable(const std::string& line, FileMgr* af) {
4030 if (parseddefcpd) {
4031 HUNSPELL_WARNING(stderrstderr, "error: line %d: multiple table definitions\n",
4032 af->getlinenum());
4033 return false;
4034 }
4035 parseddefcpd = true;
4036 int numdefcpd = -1;
4037 int i = 0;
4038 int np = 0;
4039 auto iter = line.begin(), start_piece = mystrsep(line, iter);
4040 while (start_piece != line.end()) {
4041 switch (i) {
4042 case 0: {
4043 np++;
4044 break;
4045 }
4046 case 1: {
4047 numdefcpd = atoi(std::string(start_piece, iter).c_str());
4048 if (numdefcpd < 1) {
4049 HUNSPELL_WARNING(stderrstderr, "error: line %d: bad entry number\n",
4050 af->getlinenum());
4051 return false;
4052 }
4053 defcpdtable.reserve(std::min(numdefcpd, 16384));
4054 np++;
4055 break;
4056 }
4057 default:
4058 break;
4059 }
4060 ++i;
4061 start_piece = mystrsep(line, iter);
4062 }
4063 if (np != 2) {
4064 HUNSPELL_WARNING(stderrstderr, "error: line %d: missing data\n",
4065 af->getlinenum());
4066 return false;
4067 }
4068
4069 /* now parse the numdefcpd lines to read in the remainder of the table */
4070 for (int j = 0; j < numdefcpd; ++j) {
4071 std::string nl;
4072 if (!af->getline(nl))
4073 return false;
4074 mychomp(nl);
4075 i = 0;
4076 defcpdtable.emplace_back();
4077 iter = nl.begin();
4078 start_piece = mystrsep(nl, iter);
4079 while (start_piece != nl.end()) {
4080 switch (i) {
4081 case 0: {
4082 if (nl.compare(start_piece - nl.begin(), 12, "COMPOUNDRULE", 12) != 0) {
4083 HUNSPELL_WARNING(stderrstderr, "error: line %d: table is corrupt\n",
4084 af->getlinenum());
4085 numdefcpd = 0;
Value stored to 'numdefcpd' is never read
4086 return false;
4087 }
4088 break;
4089 }
4090 case 1: { // handle parenthesized flags
4091 if (std::find(start_piece, iter, '(') != iter) {
4092 for (auto k = start_piece; k != iter; ++k) {
4093 auto chb = k, che = k + 1;
4094 if (*k == '(') {
4095 auto parpos = std::find(k, iter, ')');
4096 if (parpos != iter) {
4097 chb = k + 1;
4098 che = parpos;
4099 k = parpos;
4100 }
4101 }
4102
4103 if (*chb == '*' || *chb == '?') {
4104 defcpdtable.back().push_back((FLAGunsigned short)*chb);
4105 } else {
4106 pHMgr->decode_flags(defcpdtable.back(), std::string(chb, che), af);
4107 }
4108 }
4109 } else {
4110 pHMgr->decode_flags(defcpdtable.back(), std::string(start_piece, iter), af);
4111 }
4112 break;
4113 }
4114 default:
4115 break;
4116 }
4117 ++i;
4118 start_piece = mystrsep(nl, iter);
4119 }
4120 if (defcpdtable.back().empty()) {
4121 HUNSPELL_WARNING(stderrstderr, "error: line %d: table is corrupt\n",
4122 af->getlinenum());
4123 return false;
4124 }
4125 }
4126 return true;
4127}
4128
4129/* parse in the character map table */
4130bool AffixMgr::parse_maptable(const std::string& line, FileMgr* af) {
4131 if (parsedmaptable) {
4132 HUNSPELL_WARNING(stderrstderr, "error: line %d: multiple table definitions\n",
4133 af->getlinenum());
4134 return false;
4135 }
4136 parsedmaptable = true;
4137 int nummap = -1;
4138 int i = 0;
4139 int np = 0;
4140 auto iter = line.begin(), start_piece = mystrsep(line, iter);
4141 while (start_piece != line.end()) {
4142 switch (i) {
4143 case 0: {
4144 np++;
4145 break;
4146 }
4147 case 1: {
4148 nummap = atoi(std::string(start_piece, iter).c_str());
4149 if (nummap < 1) {
4150 HUNSPELL_WARNING(stderrstderr, "error: line %d: bad entry number\n",
4151 af->getlinenum());
4152 return false;
4153 }
4154 maptable.reserve(std::min(nummap, 16384));
4155 np++;
4156 break;
4157 }
4158 default:
4159 break;
4160 }
4161 ++i;
4162 start_piece = mystrsep(line, iter);
4163 }
4164 if (np != 2) {
4165 HUNSPELL_WARNING(stderrstderr, "error: line %d: missing data\n",
4166 af->getlinenum());
4167 return false;
4168 }
4169
4170 /* now parse the nummap lines to read in the remainder of the table */
4171 for (int j = 0; j < nummap; ++j) {
4172 std::string nl;
4173 if (!af->getline(nl))
4174 return false;
4175 mychomp(nl);
4176 i = 0;
4177 maptable.emplace_back();
4178 iter = nl.begin();
4179 start_piece = mystrsep(nl, iter);
4180 while (start_piece != nl.end()) {
4181 switch (i) {
4182 case 0: {
4183 if (nl.compare(start_piece - nl.begin(), 3, "MAP", 3) != 0) {
4184 HUNSPELL_WARNING(stderrstderr, "error: line %d: table is corrupt\n",
4185 af->getlinenum());
4186 nummap = 0;
4187 return false;
4188 }
4189 break;
4190 }
4191 case 1: {
4192 for (auto k = start_piece; k != iter; ++k) {
4193 auto chb = k, che = k + 1;
4194 if (*k == '(') {
4195 auto parpos = std::find(k, iter, ')');
4196 if (parpos != iter) {
4197 chb = k + 1;
4198 che = parpos;
4199 k = parpos;
4200 }
4201 } else {
4202 if (utf8 && (*k & 0xc0) == 0xc0) {
4203 ++k;
4204 while (k != iter && (*k & 0xc0) == 0x80)
4205 ++k;
4206 che = k;
4207 --k;
4208 }
4209 }
4210 if (chb == che) {
4211 HUNSPELL_WARNING(stderrstderr, "error: line %d: table is corrupt\n",
4212 af->getlinenum());
4213 }
4214
4215 maptable.back().emplace_back(chb, che);
4216 }
4217 break;
4218 }
4219 default:
4220 break;
4221 }
4222 ++i;
4223 start_piece = mystrsep(nl, iter);
4224 }
4225 if (maptable.back().empty()) {
4226 HUNSPELL_WARNING(stderrstderr, "error: line %d: table is corrupt\n",
4227 af->getlinenum());
4228 return false;
4229 }
4230 }
4231 return true;
4232}
4233
4234/* parse in the word breakpoint table */
4235bool AffixMgr::parse_breaktable(const std::string& line, FileMgr* af) {
4236 if (parsedbreaktable) {
4237 HUNSPELL_WARNING(stderrstderr, "error: line %d: multiple table definitions\n",
4238 af->getlinenum());
4239 return false;
4240 }
4241 parsedbreaktable = true;
4242 int numbreak = -1;
4243 int i = 0;
4244 int np = 0;
4245 auto iter = line.begin(), start_piece = mystrsep(line, iter);
4246 while (start_piece != line.end()) {
4247 switch (i) {
4248 case 0: {
4249 np++;
4250 break;
4251 }
4252 case 1: {
4253 numbreak = atoi(std::string(start_piece, iter).c_str());
4254 if (numbreak < 0) {
4255 HUNSPELL_WARNING(stderrstderr, "error: line %d: bad entry number\n",
4256 af->getlinenum());
4257 return false;
4258 }
4259 if (numbreak == 0)
4260 return true;
4261 breaktable.reserve(std::min(numbreak, 16384));
4262 np++;
4263 break;
4264 }
4265 default:
4266 break;
4267 }
4268 ++i;
4269 start_piece = mystrsep(line, iter);
4270 }
4271 if (np != 2) {
4272 HUNSPELL_WARNING(stderrstderr, "error: line %d: missing data\n",
4273 af->getlinenum());
4274 return false;
4275 }
4276
4277 /* now parse the numbreak lines to read in the remainder of the table */
4278 for (int j = 0; j < numbreak; ++j) {
4279 std::string nl;
4280 if (!af->getline(nl))
4281 return false;
4282 mychomp(nl);
4283 i = 0;
4284 iter = nl.begin();
4285 start_piece = mystrsep(nl, iter);
4286 while (start_piece != nl.end()) {
4287 switch (i) {
4288 case 0: {
4289 if (nl.compare(start_piece - nl.begin(), 5, "BREAK", 5) != 0) {
4290 HUNSPELL_WARNING(stderrstderr, "error: line %d: table is corrupt\n",
4291 af->getlinenum());
4292 numbreak = 0;
4293 return false;
4294 }
4295 break;
4296 }
4297 case 1: {
4298 breaktable.emplace_back(start_piece, iter);
4299 break;
4300 }
4301 default:
4302 break;
4303 }
4304 ++i;
4305 start_piece = mystrsep(nl, iter);
4306 }
4307 }
4308
4309 if (breaktable.size() != static_cast<size_t>(numbreak)) {
4310 HUNSPELL_WARNING(stderrstderr, "error: line %d: table is corrupt\n",
4311 af->getlinenum());
4312 return false;
4313 }
4314
4315 return true;
4316}
4317
4318void AffixMgr::reverse_condition(std::string& piece) {
4319 if (piece.empty())
4320 return;
4321
4322 int neg = 0;
4323 // iterate backwards; k wraps to npos (SIZE_MAX) when decremented past 0
4324 for (size_t k = piece.size() - 1; k != std::string::npos; --k) {
4325 switch (piece[k]) {
4326 case '[': {
4327 if (neg)
4328 piece[k + 1] = '[';
4329 else
4330 piece[k] = ']';
4331 break;
4332 }
4333 case ']': {
4334 piece[k] = '[';
4335 if (neg)
4336 piece[k + 1] = '^';
4337 neg = 0;
4338 break;
4339 }
4340 case '^': {
4341 if (piece[k + 1] == ']')
4342 neg = 1;
4343 else if (neg)
4344 piece[k + 1] = piece[k];
4345 break;
4346 }
4347 default: {
4348 if (neg)
4349 piece[k + 1] = piece[k];
4350 }
4351 }
4352 }
4353}
4354
4355class entries_container {
4356 std::vector<AffEntry*> entries;
4357 AffixMgr* m_mgr;
4358 char m_at;
4359public:
4360 entries_container(char at, AffixMgr* mgr)
4361 : m_mgr(mgr)
4362 , m_at(at) {
4363 }
4364 void release() {
4365 entries.clear();
4366 }
4367 void initialize(int numents,
4368 char opts, unsigned short aflag) {
4369 entries.reserve(std::min(numents, 16384));
4370
4371 if (m_at == 'P') {
4372 entries.push_back(new PfxEntry(m_mgr));
4373 } else {
4374 entries.push_back(new SfxEntry(m_mgr));
4375 }
4376
4377 entries.back()->opts = opts;
4378 entries.back()->aflag = aflag;
4379 }
4380
4381 AffEntry* add_entry(char opts) {
4382 if (m_at == 'P') {
4383 entries.push_back(new PfxEntry(m_mgr));
4384 } else {
4385 entries.push_back(new SfxEntry(m_mgr));
4386 }
4387 AffEntry* ret = entries.back();
4388 ret->opts = entries[0]->opts & opts;
4389 return ret;
4390 }
4391
4392 AffEntry* first_entry() { return entries.empty() ? nullptr : entries[0]; }
4393
4394 ~entries_container() {
4395 for (auto& entry : entries) {
4396 delete entry;
4397 }
4398 }
4399
4400 std::vector<AffEntry*>::iterator begin() { return entries.begin(); }
4401 std::vector<AffEntry*>::iterator end() { return entries.end(); }
4402};
4403
4404bool AffixMgr::parse_affix(const std::string& line,
4405 const char at,
4406 FileMgr* af,
4407 char* dupflags) {
4408 int numents = 0; // number of AffEntry structures to parse
4409
4410 unsigned short aflag = 0; // affix char identifier
4411
4412 char ff = 0;
4413 entries_container affentries(at, this);
4414
4415 int i = 0;
4416
4417// checking lines with bad syntax
4418#ifdef DEBUG1
4419 int basefieldnum = 0;
4420#endif
4421
4422 // split affix header line into pieces
4423
4424 int np = 0;
4425 auto iter = line.begin(), start_piece = mystrsep(line, iter);
4426 while (start_piece != line.end()) {
4427 switch (i) {
4428 // piece 1 - is type of affix
4429 case 0: {
4430 np++;
4431 break;
4432 }
4433
4434 // piece 2 - is affix char
4435 case 1: {
4436 np++;
4437 aflag = pHMgr->decode_flag(std::string(start_piece, iter));
4438 if (((at == 'S') && (dupflags[aflag] & dupSFX(1 << 0))) ||
4439 ((at == 'P') && (dupflags[aflag] & dupPFX(1 << 1)))) {
4440 HUNSPELL_WARNING(
4441 stderrstderr,
4442 "error: line %d: multiple definitions of an affix flag\n",
4443 af->getlinenum());
4444 }
4445 dupflags[aflag] += (char)((at == 'S') ? dupSFX(1 << 0) : dupPFX(1 << 1));
4446 break;
4447 }
4448 // piece 3 - is cross product indicator
4449 case 2: {
4450 np++;
4451 if (*start_piece == 'Y')
4452 ff = aeXPRODUCT(1 << 0);
4453 break;
4454 }
4455
4456 // piece 4 - is number of affentries
4457 case 3: {
4458 np++;
4459 numents = atoi(std::string(start_piece, iter).c_str());
4460 if ((numents <= 0) || ((std::numeric_limits<size_t>::max() /
4461 sizeof(AffEntry)) < static_cast<size_t>(numents))) {
4462 std::string err = pHMgr->encode_flag(aflag);
4463 HUNSPELL_WARNING(stderrstderr, "error: line %d: affix %s: bad entry number\n",
4464 af->getlinenum(), err.c_str());
4465 return false;
4466 }
4467
4468 char opts = ff;
4469 if (utf8)
4470 opts |= aeUTF8(1 << 1);
4471 if (pHMgr->is_aliasf())
4472 opts |= aeALIASF(1 << 2);
4473 if (pHMgr->is_aliasm())
4474 opts |= aeALIASM(1 << 3);
4475 affentries.initialize(numents, opts, aflag);
4476 }
4477
4478 default:
4479 break;
4480 }
4481 ++i;
4482 start_piece = mystrsep(line, iter);
4483 }
4484 // check to make sure we parsed enough pieces
4485 if (np != 4) {
4486 std::string err = pHMgr->encode_flag(aflag);
4487 HUNSPELL_WARNING(stderrstderr, "error: line %d: affix %s: missing data\n",
4488 af->getlinenum(), err.c_str());
4489 return false;
4490 }
4491
4492 // now parse numents affentries for this affix
4493 AffEntry* entry = affentries.first_entry();
4494 for (int ent = 0; ent < numents; ++ent) {
4495 std::string nl;
4496 if (!af->getline(nl))
4497 return false;
4498 mychomp(nl);
4499
4500 iter = nl.begin();
4501 i = 0;
4502 np = 0;
4503
4504 // split line into pieces
4505 start_piece = mystrsep(nl, iter);
4506 while (start_piece != nl.end()) {
4507 switch (i) {
4508 // piece 1 - is type
4509 case 0: {
4510 np++;
4511 if (ent != 0)
4512 entry = affentries.add_entry((char)(aeXPRODUCT(1 << 0) | aeUTF8(1 << 1) | aeALIASF(1 << 2) | aeALIASM(1 << 3)));
4513 break;
4514 }
4515
4516 // piece 2 - is affix char
4517 case 1: {
4518 np++;
4519 std::string chunk(start_piece, iter);
4520 if (pHMgr->decode_flag(chunk) != aflag) {
4521 std::string err = pHMgr->encode_flag(aflag);
4522 HUNSPELL_WARNING(stderrstderr,
4523 "error: line %d: affix %s is corrupt\n",
4524 af->getlinenum(), err.c_str());
4525 return false;
4526 }
4527
4528 if (ent != 0) {
4529 AffEntry* start_entry = affentries.first_entry();
4530 entry->aflag = start_entry->aflag;
4531 }
4532 break;
4533 }
4534
4535 // piece 3 - is string to strip or 0 for null
4536 case 2: {
4537 np++;
4538 entry->strip = std::string(start_piece, iter);
4539 if (complexprefixes) {
4540 if (utf8)
4541 reverseword_utf(entry->strip);
4542 else
4543 reverseword(entry->strip);
4544 }
4545 if (entry->strip.compare("0") == 0) {
4546 entry->strip.clear();
4547 }
4548 break;
4549 }
4550
4551 // piece 4 - is affix string or 0 for null
4552 case 3: {
4553 entry->morphcode = nullptr;
4554 entry->contclass = nullptr;
4555 entry->contclasslen = 0;
4556 np++;
4557 std::string::const_iterator dash = std::find(start_piece, iter, '/');
4558 if (dash != iter) {
4559 entry->appnd = std::string(start_piece, dash);
4560 std::string dash_str(dash + 1, iter);
4561
4562 if (!ignorechars.empty() && !has_no_ignored_chars(entry->appnd, ignorechars)) {
4563 if (utf8) {
4564 remove_ignored_chars_utf(entry->appnd, ignorechars_utf16);
4565 } else {
4566 remove_ignored_chars(entry->appnd, ignorechars);
4567 }
4568 }
4569
4570 if (complexprefixes) {
4571 if (utf8)
4572 reverseword_utf(entry->appnd);
4573 else
4574 reverseword(entry->appnd);
4575 }
4576
4577 if (pHMgr->is_aliasf()) {
4578 int index = atoi(dash_str.c_str());
4579 entry->contclasslen = (unsigned short)pHMgr->get_aliasf(
4580 index, &(entry->contclass), af);
4581 if (!entry->contclasslen)
4582 HUNSPELL_WARNING(stderrstderr,
4583 "error: bad affix flag alias: \"%s\"\n",
4584 dash_str.c_str());
4585 } else {
4586 entry->contclasslen = (unsigned short)pHMgr->decode_flags(
4587 &(entry->contclass), dash_str, af);
4588 std::sort(entry->contclass, entry->contclass + entry->contclasslen);
4589 }
4590
4591 havecontclass = 1;
4592 for (unsigned short _i = 0; _i < entry->contclasslen; _i++) {
4593 contclasses[(entry->contclass)[_i]] = 1;
4594 }
4595 } else {
4596 entry->appnd = std::string(start_piece, iter);
4597
4598 if (!ignorechars.empty() && !has_no_ignored_chars(entry->appnd, ignorechars)) {
4599 if (utf8) {
4600 remove_ignored_chars_utf(entry->appnd, ignorechars_utf16);
4601 } else {
4602 remove_ignored_chars(entry->appnd, ignorechars);
4603 }
4604 }
4605
4606 if (complexprefixes) {
4607 if (utf8)
4608 reverseword_utf(entry->appnd);
4609 else
4610 reverseword(entry->appnd);
4611 }
4612 }
4613
4614 if (entry->appnd.compare("0") == 0) {
4615 entry->appnd.clear();
4616 }
4617 break;
4618 }
4619
4620 // piece 5 - is the conditions descriptions
4621 case 4: {
4622 std::string chunk(start_piece, iter);
4623 np++;
4624 if (complexprefixes) {
4625 if (utf8)
4626 reverseword_utf(chunk);
4627 else
4628 reverseword(chunk);
4629 reverse_condition(chunk);
4630 }
4631 if (!entry->strip.empty() && chunk != "." &&
4632 redundant_condition(at, entry->strip, chunk,
4633 af->getlinenum()))
4634 chunk = ".";
4635 if (at == 'S') {
4636 reverseword(chunk);
4637 reverse_condition(chunk);
4638 }
4639 if (encodeit(*entry, chunk))
4640 return false;
4641 break;
4642 }
4643
4644 case 5: {
4645 std::string chunk(start_piece, iter);
4646 np++;
4647 if (pHMgr->is_aliasm()) {
4648 int index = atoi(chunk.c_str());
4649 entry->morphcode = pHMgr->get_aliasm(index);
4650 } else {
4651 if (complexprefixes) { // XXX - fix me for morph. gen.
4652 if (utf8)
4653 reverseword_utf(chunk);
4654 else
4655 reverseword(chunk);
4656 }
4657 // add the remaining of the line
4658 std::string::const_iterator end = nl.end();
4659 if (iter != end) {
4660 chunk.append(iter, end);
4661 }
4662 entry->morphcode = mystrdup(chunk.c_str());
4663 }
4664 break;
4665 }
4666 default:
4667 break;
4668 }
4669 i++;
4670 start_piece = mystrsep(nl, iter);
4671 }
4672 // check to make sure we parsed enough pieces
4673 if (np < 4) {
4674 std::string err = pHMgr->encode_flag(aflag);
4675 HUNSPELL_WARNING(stderrstderr, "error: line %d: affix %s is corrupt\n",
4676 af->getlinenum(), err.c_str());
4677 return false;
4678 }
4679
4680#ifdef DEBUG1
4681 // detect unnecessary fields, excepting comments
4682 if (basefieldnum) {
4683 int fieldnum =
4684 !(entry->morphcode) ? 5 : ((*(entry->morphcode) == '#') ? 5 : 6);
4685 if (fieldnum != basefieldnum)
4686 HUNSPELL_WARNING(stderrstderr, "warning: line %d: bad field number\n",
4687 af->getlinenum());
4688 } else {
4689 basefieldnum =
4690 !(entry->morphcode) ? 5 : ((*(entry->morphcode) == '#') ? 5 : 6);
4691 }
4692#endif
4693 }
4694
4695 // now create SfxEntry or PfxEntry objects and use links to
4696 // build an ordered (sorted by affix string) list
4697 auto start = affentries.begin(), end = affentries.end();
4698 for (auto affentry = start; affentry != end; ++affentry) {
4699 if (at == 'P') {
4700 build_pfxtree(static_cast<PfxEntry*>(*affentry));
4701 } else {
4702 build_sfxtree(static_cast<SfxEntry*>(*affentry));
4703 }
4704 }
4705
4706 //contents belong to AffixMgr now
4707 affentries.release();
4708
4709 return true;
4710}
4711
4712int AffixMgr::redundant_condition(char ft,
4713 const std::string& strip,
4714 const std::string& cond,
4715 int linenum) {
4716 int stripl = strip.size(), condl = cond.size(), i, j, neg, in;
4717 if (ft == 'P') { // prefix
4718 if (strip.compare(0, condl, cond) == 0)
4719 return 1;
4720 if (utf8) {
4721 } else {
4722 for (i = 0, j = 0; (i < stripl) && (j < condl); i++, j++) {
4723 if (cond[j] != '[') {
4724 if (cond[j] != strip[i]) {
4725 HUNSPELL_WARNING(stderrstderr,
4726 "warning: line %d: incompatible stripping "
4727 "characters and condition\n",
4728 linenum);
4729 return 0;
4730 }
4731 } else {
4732 neg = (cond[j + 1] == '^') ? 1 : 0;
4733 in = 0;
4734 do {
4735 j++;
4736 if (strip[i] == cond[j])
4737 in = 1;
4738 } while ((j < (condl - 1)) && (cond[j] != ']'));
4739 if (j == (condl - 1) && (cond[j] != ']')) {
4740 HUNSPELL_WARNING(stderrstderr,
4741 "error: line %d: missing ] in condition:\n%s\n",
4742 linenum, cond.c_str());
4743 return 0;
4744 }
4745 if ((!neg && !in) || (neg && in)) {
4746 HUNSPELL_WARNING(stderrstderr,
4747 "warning: line %d: incompatible stripping "
4748 "characters and condition\n",
4749 linenum);
4750 return 0;
4751 }
4752 }
4753 }
4754 if (j >= condl)
4755 return 1;
4756 }
4757 } else { // suffix
4758 if ((stripl >= condl) && strip.compare(stripl - condl, std::string::npos, cond) == 0)
4759 return 1;
4760 if (utf8) {
4761 } else {
4762 for (i = stripl - 1, j = condl - 1; (i >= 0) && (j >= 0); i--, j--) {
4763 if (cond[j] != ']') {
4764 if (cond[j] != strip[i]) {
4765 HUNSPELL_WARNING(stderrstderr,
4766 "warning: line %d: incompatible stripping "
4767 "characters and condition\n",
4768 linenum);
4769 return 0;
4770 }
4771 } else if (j > 0) {
4772 in = 0;
4773 do {
4774 j--;
4775 if (strip[i] == cond[j])
4776 in = 1;
4777 } while ((j > 0) && (cond[j] != '['));
4778 if ((j == 0) && (cond[j] != '[')) {
4779 HUNSPELL_WARNING(stderrstderr,
4780 "error: line: %d: missing ] in condition:\n%s\n",
4781 linenum, cond.c_str());
4782 return 0;
4783 }
4784 neg = (cond[j + 1] == '^') ? 1 : 0;
4785 if ((!neg && !in) || (neg && in)) {
4786 HUNSPELL_WARNING(stderrstderr,
4787 "warning: line %d: incompatible stripping "
4788 "characters and condition\n",
4789 linenum);
4790 return 0;
4791 }
4792 }
4793 }
4794 if (j < 0)
4795 return 1;
4796 }
4797 }
4798 return 0;
4799}
4800
4801std::vector<std::string> AffixMgr::get_suffix_words(short unsigned* suff,
4802 int len,
4803 const std::string& root_word) {
4804 std::vector<std::string> slst;
4805 short unsigned* start_ptr = suff;
4806 for (auto ptr : sStart) {
4807 while (ptr) {
4808 suff = start_ptr;
4809 for (int i = 0; i < len; i++) {
4810 if ((*suff) == ptr->getFlag()) {
4811 std::string nw(root_word);
4812 nw.append(ptr->getAffix());
4813 hentry* ht = ptr->checkword(nw, 0, nw.size(), 0, nullptr, 0, 0, 0);
4814 if (ht) {
4815 slst.push_back(std::move(nw));
4816 }
4817 }
4818 suff++;
4819 }
4820 ptr = ptr->getNext();
4821 }
4822 }
4823 return slst;
4824}