From 8f70dfcf69d313a5fe4b208bdef0c71eebdee381 Mon Sep 17 00:00:00 2001 From: Vincent Wei Date: Wed, 27 Mar 2019 16:46:42 +0800 Subject: [PATCH] UStrTailorBreaks --- include/gdi.h | 18 ++- src/font/Makefile.am | 2 +- src/font/unicode-break-arabic.c | 109 +++++++++++++++ src/font/unicode-break-indic.c | 233 ++++++++++++++++++++++++++++++++ src/font/unicode-break-thai.c | 65 +++++++++ src/font/unicode-break.c | 27 ++++ src/include/unicode-ops.h | 7 + src/newgdi/textrunsinfo.c | 12 +- 8 files changed, 467 insertions(+), 6 deletions(-) create mode 100644 src/font/unicode-break-arabic.c create mode 100644 src/font/unicode-break-indic.c create mode 100644 src/font/unicode-break-thai.c diff --git a/include/gdi.h b/include/gdi.h index fc984890..a5663ed4 100644 --- a/include/gdi.h +++ b/include/gdi.h @@ -8813,6 +8813,9 @@ MG_EXPORT int GUIAPI UStrGetBreaks(ScriptType writing_system, Uint8 ctr, Uint8 wbr, Uint8 lbp, Uchar32* ucs, int nr_ucs, BreakOppo** break_oppos); +MG_EXPORT void GUIAPI UStrTailorBreaks(ScriptType writing_system, + const Uchar32* ucs, int nr_ucs, BreakOppo* break_oppos); + /** The function determines whether a character is alphanumeric. */ MG_EXPORT BOOL GUIAPI IsUCharAlnum(Uchar32 uc); @@ -12554,7 +12557,8 @@ typedef struct _TEXTRUNSINFO TEXTRUNSINFO; * \fn TEXTRUNSINFO* GUIAPI CreateTextRunsInfo(const Uchar32* ucs, int nr_ucs, * LanguageCode lang_code, ParagraphDir base_dir, GlyphRunDir run_dir, * GlyphOrient glyph_orient, GlyphOrientPolicy orient_policy, - * const char* logfont_name, RGBCOLOR color); + * const char* logfont_name, RGBCOLOR color, + * BreakOppo* break_oppos) * * \brief Split a Uchar32 paragraph string in mixed scripts into text runs. * @@ -12563,6 +12567,7 @@ typedef struct _TEXTRUNSINFO TEXTRUNSINFO; * * \param ucs The Uchar32 string returned by \a GetUCharsUntilParagraphBoundary. * \param nr_ucs The length of the Uchar32 string. + * \param * \param lang_code The language code. * \param base_dir The base direction of the paragraph. * \param run_dir The writing direction of the paragraph. @@ -12572,21 +12577,26 @@ typedef struct _TEXTRUNSINFO TEXTRUNSINFO; * of some text in the paragraph by calling \a SetFontInTextRuns. * \param color The default text color. You can change the text color * of some text in the paragraph by calling \a SetTextColorInTextRuns + * \param break_oppos If not NULL, the break opportunities will be tailored + * according to the script type of every text run. Please skip the first + * entry when you pass the pointer. * * \return The TEXTRUNSINFO object create; NULL for failure. * * \note This function assumes that you passed one paragraph of * the logical Unicode string. Therefore, you'd better to call this - * function after calling GetUCharsUntilParagraphBoundary. + * function after calling \a GetUCharsUntilParagraphBoundary. * * \note Only available when support for UNICODE is enabled. * - * \sa GetUCharsUntilParagraphBoundary, SetFontInTextRuns, SetTextColorInTextRuns + * \sa GetUCharsUntilParagraphBoundary, UStrGetBreaks, + * SetFontInTextRuns, SetTextColorInTextRuns */ MG_EXPORT TEXTRUNSINFO* GUIAPI CreateTextRunsInfo(const Uchar32* ucs, int nr_ucs, LanguageCode lang_code, ParagraphDir base_dir, GlyphRunDir run_dir, GlyphOrient glyph_orient, GlyphOrientPolicy orient_policy, - const char* logfont_name, RGBCOLOR color); + const char* logfont_name, RGBCOLOR color, + BreakOppo* break_oppos); /** * Set logfont of text runs diff --git a/src/font/Makefile.am b/src/font/Makefile.am index cc3d3f2b..1a1f0e41 100644 --- a/src/font/Makefile.am +++ b/src/font/Makefile.am @@ -14,7 +14,7 @@ SRC_FILES = charset.c charset-arabic.c legacy-bidi.c charset-unicode.c \ bitmapfont.c scripteasy.c \ unicode-emoji.c unicode-vop.c \ unicode-bidi.c \ - unicode-break.c \ + unicode-break.c unicode-break-arabic.c unicode-break-indic.c unicode-break-thai.c \ unicode-joining-types.c unicode-joining.c unicode-shape.c \ unicode-shape-arabic.c harzbuff-minigui-funcs.c \ unicode-iterators.c diff --git a/src/font/unicode-break-arabic.c b/src/font/unicode-break-arabic.c new file mode 100644 index 00000000..06c89b90 --- /dev/null +++ b/src/font/unicode-break-arabic.c @@ -0,0 +1,109 @@ +/* + * This file is part of MiniGUI, a mature cross-platform windowing + * and Graphics User Interface (GUI) support system for embedded systems + * and smart IoT devices. + * + * Copyright (C) 2002~2018, Beijing FMSoft Technologies Co., Ltd. + * Copyright (C) 1998~2002, WEI Yongming + * + * This program is free software: you can redistribute it and/or modify + * it under the terms of the GNU General Public License as published by + * the Free Software Foundation, either version 3 of the License, or + * (at your option) any later version. + * + * This program is distributed in the hope that it will be useful, + * but WITHOUT ANY WARRANTY; without even the implied warranty of + * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the + * GNU General Public License for more details. + * + * You should have received a copy of the GNU General Public License + * along with this program. If not, see . + * + * Or, + * + * As this program is a library, any link to this program must follow + * GNU General Public License version 3 (GPLv3). If you cannot accept + * GPLv3, you need to be licensed from FMSoft. + * + * If you have got a commercial license of this program, please use it + * under the terms and conditions of the commercial license. + * + * For more information about the commercial license, please refer to + * . + */ + +/* +** unicode-break-arabic.c: +** The implementation to tailor Arabic breaks +** +** Created by WEI Yongming at 2019/03/27 +** +** This implementation is derived from LGPL'd Pango. +*/ + +#include +#include +#include +#include +#include + +#include "common.h" + +#ifdef _MGCHARSET_UNICODE + +#include "minigui.h" +#include "gdi.h" +#include "unicode-ops.h" + +#define ALEF_WITH_MADDA_ABOVE 0x0622 +#define YEH_WITH_HAMZA_ABOVE 0x0626 +#define ALEF 0x0627 +#define WAW 0x0648 +#define YEH 0x064A + +#define MADDAH_ABOVE 0x0653 +#define HAMZA_ABOVE 0x0654 +#define HAMZA_BELOW 0x0655 + +/* + * Arabic characters with canonical decompositions that are not just + * ligatures. The characters U+06C0, U+06C2, and U+06D3 are intentionally + * excluded as they are marked as "not an independent letter" in Unicode + * Character Database's NamesList.txt + */ +#define IS_COMPOSITE(c) (ALEF_WITH_MADDA_ABOVE <= (c) && (c) <= YEH_WITH_HAMZA_ABOVE) + +/* If a character is the second part of a composite Arabic character with an Alef */ +#define IS_COMPOSITE_WITH_ALEF(c) (MADDAH_ABOVE <= (c) && (c) <= HAMZA_BELOW) + +void __mg_unicode_break_arabic(const Uchar32* ucs, int nr_ucs, + BreakOppo* break_oppos) +{ + int i; + Uchar32 prev_wc, this_wc; + + for (i = 0, prev_wc = 0; i < nr_ucs; i++, prev_wc = this_wc) { + this_wc = ucs[i]; + + /* + * Unset backspace_deletes_character for various combinations. + * + * A few more combinations may need to be handled here, but are not + * handled yet, as expectations of users is not known or may differ + * among different languages or users: + * some letters combined with U+0658 ARABIC MARK NOON GHUNNA; + * combinations considered one letter in Azerbaijani (WAW+SUKUN and + * FARSI_YEH+HAMZA_ABOVE); combinations of YEH and ALEF_MAKSURA with + * HAMZA_BELOW (Qur'anic); TATWEEL+HAMZA_ABOVE (Qur'anic). + * + * FIXME: Ordering these in some other way may lower the time spent here, or not. + */ + if (IS_COMPOSITE (this_wc) || + (prev_wc == ALEF && IS_COMPOSITE_WITH_ALEF (this_wc)) || + (this_wc == HAMZA_ABOVE && (prev_wc == WAW || prev_wc == YEH))) + break_oppos[i + 1] &= ~BOV_GB_BACKSPACE_DEL_CH; + } +} + +#endif /* _MGCHARSET_UNICODE */ + diff --git a/src/font/unicode-break-indic.c b/src/font/unicode-break-indic.c new file mode 100644 index 00000000..69a7c129 --- /dev/null +++ b/src/font/unicode-break-indic.c @@ -0,0 +1,233 @@ +/* + * This file is part of MiniGUI, a mature cross-platform windowing + * and Graphics User Interface (GUI) support system for embedded systems + * and smart IoT devices. + * + * Copyright (C) 2002~2018, Beijing FMSoft Technologies Co., Ltd. + * Copyright (C) 1998~2002, WEI Yongming + * + * This program is free software: you can redistribute it and/or modify + * it under the terms of the GNU General Public License as published by + * the Free Software Foundation, either version 3 of the License, or + * (at your option) any later version. + * + * This program is distributed in the hope that it will be useful, + * but WITHOUT ANY WARRANTY; without even the implied warranty of + * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the + * GNU General Public License for more details. + * + * You should have received a copy of the GNU General Public License + * along with this program. If not, see . + * + * Or, + * + * As this program is a library, any link to this program must follow + * GNU General Public License version 3 (GPLv3). If you cannot accept + * GPLv3, you need to be licensed from FMSoft. + * + * If you have got a commercial license of this program, please use it + * under the terms and conditions of the commercial license. + * + * For more information about the commercial license, please refer to + * . + */ + +/* +** unicode-break-indic.c: +** The implementation to tailor Indic breaks +** +** Created by WEI Yongming at 2019/03/27 +** +** This implementation is derived from LGPL'd Pango. +*/ + +#include +#include +#include +#include +#include + +#include "common.h" + +#ifdef _MGCHARSET_UNICODE + +#include "minigui.h" +#include "gdi.h" +#include "unicode-ops.h" + +#define DEV_RRA 0x0931 /* 0930 + 093c */ +#define DEV_QA 0x0958 /* 0915 + 093c */ +#define DEV_YA 0x095F /* 092f + 003c */ +#define DEV_KHHA 0x0959 +#define DEV_GHHA 0x095A +#define DEV_ZA 0x095B +#define DEV_DDDHA 0x095C +#define DEV_RHA 0x095D +#define DEV_FA 0x095E +#define DEV_YYA 0x095F + +/* Bengali */ +/* for split matras in all brahmi based script */ +#define BENGALI_SIGN_O 0x09CB /* 09c7 + 09be */ +#define BENGALI_SIGN_AU 0x09CC /* 09c7 + 09d7 */ +#define BENGALI_RRA 0x09DC +#define BENGALI_RHA 0x09DD +#define BENGALI_YYA 0x09DF + +/* Gurumukhi */ +#define GURUMUKHI_LLA 0x0A33 +#define GURUMUKHI_SHA 0x0A36 +#define GURUMUKHI_KHHA 0x0A59 +#define GURUMUKHI_GHHA 0x0A5A +#define GURUMUKHI_ZA 0x0A5B +#define GURUMUKHI_RRA 0x0A5C +#define GURUMUKHI_FA 0x0A5E + +/* Oriya */ +#define ORIYA_AI 0x0B48 +#define ORIYA_O 0x0B4B +#define ORIYA_AU 0x0B4C + +/* Telugu */ +#define TELUGU_EE 0x0C47 +#define TELUGU_AI 0x0C48 + +/* Tamil */ +#define TAMIL_O 0x0BCA +#define TAMIL_OO 0x0BCB +#define TAMIL_AU 0x0BCC + +/* Kannada */ +#define KNDA_EE 0x0CC7 +#define KNDA_AI 0x0CC8 +#define KNDA_O 0x0CCA +#define KNDA_OO 0x0CCB + +/* Malayalam */ +#define MLYM_O 0x0D4A +#define MLYM_OO 0x0D4B +#define MLYM_AU 0x0D4C + +#define IS_COMPOSITE_WITH_BRAHMI_NUKTA(c) ( \ + (c >= BENGALI_RRA && c <= BENGALI_YYA) || \ + (c >= DEV_QA && c <= DEV_YA) || (c == DEV_RRA) || (c >= DEV_KHHA && c <= DEV_YYA) || \ + (c >= KNDA_EE && c <= KNDA_AI) ||(c >= KNDA_O && c <= KNDA_OO) || \ + (c == TAMIL_O) || (c == TAMIL_OO) || (c == TAMIL_AU) || \ + (c == TELUGU_EE) || (c == TELUGU_AI) || \ + (c == ORIYA_AI) || (c == ORIYA_O) || (c == ORIYA_AU) || \ + (c >= GURUMUKHI_KHHA && c <= GURUMUKHI_RRA) || (c == GURUMUKHI_FA)|| (c == GURUMUKHI_LLA)|| (c == GURUMUKHI_SHA) || \ + FALSE) +#define IS_SPLIT_MATRA_BRAHMI(c) ( \ + (c == BENGALI_SIGN_O) || (c == BENGALI_SIGN_AU) || \ + (c >= MLYM_O && c <= MLYM_AU) || \ + FALSE) + +static void +not_cursor_position (BreakOppo *bov) +{ + *bov &= ~BOV_GB_CURSOR_POS; // attr->is_cursor_position = FALSE; + *bov &= ~BOV_GB_CHAR_BREAK; // attr->is_char_break = FALSE; + *bov &= ~BOV_LB_MASK; // attr->is_line_break = FALSE; + *bov |= BOV_LB_NOTALLOWED; // attr->is_mandatory_break = FALSE; + //*bov &= ~BOV_LB_MANDATORY_FLAG; +} + +void __mg_unicode_break_indic(ScriptType writing_system, + const Uchar32 *ucs, int nr_ucs, BreakOppo* bos) +{ + const Uchar32 *p, *next = NULL, *next_next; + Uchar32 prev_wc, this_wc, next_wc, next_next_wc; + BOOL is_conjunct = FALSE; + int i; + + for (p = ucs, prev_wc = 0, i = 0; + p != NULL && p < (ucs + nr_ucs); + p = next, prev_wc = this_wc, i++) + { + this_wc = p[0]; + next = p + 1; + + if (IS_COMPOSITE_WITH_BRAHMI_NUKTA(this_wc) || + IS_SPLIT_MATRA_BRAHMI(this_wc)) { + // attrs[i+1].backspace_deletes_character = FALSE; + bos[i + 1] &= ~BOV_GB_BACKSPACE_DEL_CH; + } + + if (next != NULL && next < (ucs + nr_ucs)) { + next_wc = next[0]; + next_next = next + 1; + } + else { + next_wc = 0; + next_next = NULL; + } + + if (next_next != NULL && next_next < (ucs + nr_ucs)) + next_next_wc = *next_next; + else + next_next_wc = 0; + + switch (writing_system) { + case SCRIPT_SINHALA: + /* + * TODO: The cursor position should be based on the state table. + * This is the wrong place to be doing this. + */ + + /* + * The cursor should treat as a single glyph: + * SINHALA CONS + 0x0DCA + 0x200D + SINHALA CONS + * SINHALA CONS + 0x200D + 0x0DCA + SINHALA CONS + */ + if ((this_wc == 0x0DCA && next_wc == 0x200D) + || (this_wc == 0x200D && next_wc == 0x0DCA)) { + not_cursor_position(bos + i); + not_cursor_position(bos + i + 1); + is_conjunct = TRUE; + } + else if (is_conjunct + && (prev_wc == 0x200D || prev_wc == 0x0DCA) + && this_wc >= 0x0D9A + && this_wc <= 0x0DC6) { + not_cursor_position(bos + i); + is_conjunct = FALSE; + } + /* + * Consonant clusters do NOT result in implicit conjuncts + * in SINHALA orthography. + */ + else if (!is_conjunct && prev_wc == 0x0DCA && this_wc != 0x200D) { + // attrs[i].is_cursor_position = TRUE; + bos[i] |= BOV_GB_CURSOR_POS; + } + + break; + + default: + if (prev_wc != 0 && (this_wc == 0x200D || this_wc == 0x200C)) { + not_cursor_position(bos + i); + if (next_wc != 0) { + not_cursor_position(bos + i+1); + if ((next_next_wc != 0) && + (next_wc == 0x09CD || /* Bengali */ + next_wc == 0x0ACD || /* Gujarati */ + next_wc == 0x094D || /* Hindi */ + next_wc == 0x0CCD || /* Kannada */ + next_wc == 0x0D4D || /* Malayalam */ + next_wc == 0x0B4D || /* Oriya */ + next_wc == 0x0A4D || /* Punjabi */ + next_wc == 0x0BCD || /* Tamil */ + next_wc == 0x0C4D)) /* Telugu */ + { + not_cursor_position(bos + i + 2); + } + } + } + + break; + } + } +} + +#endif /* _MGCHARSET_UNICODE */ + diff --git a/src/font/unicode-break-thai.c b/src/font/unicode-break-thai.c new file mode 100644 index 00000000..42823895 --- /dev/null +++ b/src/font/unicode-break-thai.c @@ -0,0 +1,65 @@ +/* + * This file is part of MiniGUI, a mature cross-platform windowing + * and Graphics User Interface (GUI) support system for embedded systems + * and smart IoT devices. + * + * Copyright (C) 2002~2018, Beijing FMSoft Technologies Co., Ltd. + * Copyright (C) 1998~2002, WEI Yongming + * + * This program is free software: you can redistribute it and/or modify + * it under the terms of the GNU General Public License as published by + * the Free Software Foundation, either version 3 of the License, or + * (at your option) any later version. + * + * This program is distributed in the hope that it will be useful, + * but WITHOUT ANY WARRANTY; without even the implied warranty of + * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the + * GNU General Public License for more details. + * + * You should have received a copy of the GNU General Public License + * along with this program. If not, see . + * + * Or, + * + * As this program is a library, any link to this program must follow + * GNU General Public License version 3 (GPLv3). If you cannot accept + * GPLv3, you need to be licensed from FMSoft. + * + * If you have got a commercial license of this program, please use it + * under the terms and conditions of the commercial license. + * + * For more information about the commercial license, please refer to + * . + */ + +/* +** unicode-break-thai.c: +** The implementation to tailor Thai breaks +** +** Created by WEI Yongming at 2019/03/27 +** +** This implementation is derived from LGPL'd Pango. +*/ + +#include +#include +#include +#include +#include + +#include "common.h" + +#ifdef _MGCHARSET_UNICODE + +#include "minigui.h" +#include "gdi.h" +#include "unicode-ops.h" + +void __mg_unicode_break_thai(const Uchar32* ucs, int nr_ucs, + BreakOppo* break_oppos) +{ + _WRN_PRINTF("NOT IMPLEMENTED"); +} + +#endif /* _MGCHARSET_UNICODE */ + diff --git a/src/font/unicode-break.c b/src/font/unicode-break.c index 83287d67..dd6858ff 100644 --- a/src/font/unicode-break.c +++ b/src/font/unicode-break.c @@ -2994,5 +2994,32 @@ error: return 0; } +void GUIAPI UStrTailorBreaks(ScriptType writing_system, + const Uchar32* ucs, int nr_ucs, BreakOppo* break_oppos) +{ + switch ((int)writing_system) { + case SCRIPT_ARABIC: + __mg_unicode_break_arabic(ucs, nr_ucs, break_oppos); + break; + + case SCRIPT_DEVANAGARI: + case SCRIPT_BENGALI: + case SCRIPT_GURMUKHI: + case SCRIPT_GUJARATI: + case SCRIPT_ORIYA: + case SCRIPT_TAMIL: + case SCRIPT_TELUGU: + case SCRIPT_KANNADA: + case SCRIPT_MALAYALAM: + case SCRIPT_SINHALA: + __mg_unicode_break_indic(writing_system, ucs, nr_ucs, break_oppos); + break; + + case SCRIPT_THAI: + __mg_unicode_break_thai(ucs, nr_ucs, break_oppos); + break; + } +} + #endif /* _MGCHARSET_UNICODE */ diff --git a/src/include/unicode-ops.h b/src/include/unicode-ops.h index 0486f9bd..78c0df2e 100644 --- a/src/include/unicode-ops.h +++ b/src/include/unicode-ops.h @@ -153,6 +153,13 @@ void __mg_emoji_iterator_fini (EmojiIterator *iter); BOOL __mg_language_includes_script(LanguageCode lc, ScriptType script); +void __mg_unicode_break_arabic(const Uchar32* ucs, int nr_ucs, + BreakOppo* break_oppos); +void __mg_unicode_break_indic(ScriptType writing_system, + const Uchar32* ucs, int nr_ucs, BreakOppo* break_oppos); +void __mg_unicode_break_thai(const Uchar32* ucs, int nr_ucs, + BreakOppo* break_oppos); + #ifdef __cplusplus } #endif /* __cplusplus */ diff --git a/src/newgdi/textrunsinfo.c b/src/newgdi/textrunsinfo.c index 0f4d9a26..8e19563d 100644 --- a/src/newgdi/textrunsinfo.c +++ b/src/newgdi/textrunsinfo.c @@ -441,7 +441,8 @@ out: TEXTRUNSINFO* GUIAPI CreateTextRunsInfo(const Uchar32* ucs, int nr_ucs, LanguageCode lang_code, ParagraphDir base_dir, GlyphRunDir run_dir, GlyphOrient glyph_orient, GlyphOrientPolicy orient_policy, - const char* logfont_name, RGBCOLOR color) + const char* logfont_name, RGBCOLOR color, + BreakOppo* break_oppos) { BOOL ok = FALSE; @@ -518,6 +519,15 @@ TEXTRUNSINFO* GUIAPI CreateTextRunsInfo(const Uchar32* ucs, int nr_ucs, goto out; } + if (break_oppos) { + struct list_head* i; + list_for_each(i, &runinfo->truns) { + TextRun* trun = (TextRun*)i; + UStrTailorBreaks(trun->st, runinfo->ucs + trun->si, trun->len, + break_oppos + trun->si); + } + } + ok = TRUE; out: