// This module implements the interface defined in uniprops.h
// in terms of ICU functions.
//
// The ICU documentation is at .
// Includes:
//
// * user guide: https://unicode-org.github.io/icu/userguide/
// * API: https://unicode-org.github.io/icu-docs/apidoc/dev/icu4c/index.html
//
// This file is part of Beastie
// SPDX-FileCopyrightText: 2025 Norman Gray
// SPDX-License-Identifier: BSD-2-Clause
#define HAVE_ICU 1 // ...by hypothesis
#include "uniprops.h"
#include "unicode/uchar.h"
int icu_is_icu_p(void)
{
return 1;
}
#define PROPFUNC(fn, prop) \
int fn(const codepoint_t cp) { \
return u_hasBinaryProperty(cp, prop); \
}
// Lowercase + Uppercase + Lt + Lm + Lo + Nl + Other_Alphabetic
// (note: this is slightly different from u_isalpha, which is all 'L')
//
// I seem to have defined the following two functions differently in mycu.c:
// is that a misunderstanding, or some distinction I've now forgotten
PROPFUNC(icu_letter_p, UCHAR_ALPHABETIC);
PROPFUNC(icu_alphabetic_p, UCHAR_ALPHABETIC); // or u_isUAlphabetic(cp)
PROPFUNC(icu_uppercase_letter_p, UCHAR_UPPERCASE);
PROPFUNC(icu_lowercase_letter_p, UCHAR_LOWERCASE);
int icu_titlecase_letter_p(const codepoint_t cp)
{
return u_istitle(cp);
}
int icu_number_p(const codepoint_t cp)
{
return u_isdigit(cp);
}
int icu_alnum_p(const codepoint_t cp)
{
return u_isalnum(cp);
}
int icu_punctuation_p(const codepoint_t cp)
{
return u_ispunct(cp);
}
int icu_cntrl_p(const codepoint_t cp)
{
//return u_charType(cp) == U_CONTROL_CHAR;
return u_iscntrl(cp);
}
int icu_symbol_p(const codepoint_t cp)
{
int8_t cat = u_charType(cp);
return cat == U_MATH_SYMBOL
|| cat == U_CURRENCY_SYMBOL
|| cat == U_MODIFIER_SYMBOL
|| cat == U_OTHER_SYMBOL;
}
int icu_mark_p(const codepoint_t cp)
{
int8_t cat = u_charType(cp);
return cat == U_NON_SPACING_MARK
|| cat == U_ENCLOSING_MARK
|| cat == U_COMBINING_SPACING_MARK;
}
int icu_space_p(const codepoint_t cp)
{
if (cp < 0x20) {
// Unicode regards all of the codepoints below U+0020 as class
// 'Cc' (‘a C0 or C1 control code’).
// As an exception, deem characters in this range to be spaces
// if they are spaces in the POSIX locale
// (this should match isspace(cp))
return (cp == ' ') // space
|| (cp >= 0x09 && cp <= 0x0d); // to
} else {
// Test whether the codepoint has the Unicode White_Space property.
// This is different from u_isspace (which matches POSIX)
// and u_isWhitespace
// (which matches Java, and excludes non-breaking spaces)
return u_isUWhiteSpace(cp);
}
}
int icu_whitespace_p(const codepoint_t cp)
{
return u_isWhitespace(cp);
}
int icu_nonbreakingspace_p(const codepoint_t cp)
{
return (cp == 0x00a0 || cp == 0x2007 || cp == 0x202f);
}
int icu_wordcharacter_p(const codepoint_t cp)
{
// fast path
if (cp < 0x80) {
return icu_alphabetic_p(cp);
} else {
return icu_alphabetic_p(cp)
|| u_hasBinaryProperty(cp, UCHAR_DIACRITIC)
|| u_hasBinaryProperty(cp, UCHAR_EXTENDER)
|| u_hasBinaryProperty(cp, UCHAR_JOIN_CONTROL);
// return icu_alphabetic_p(cp)
// || icu_diacritic_p(cp)
// || icu_extender_p(cp)
// || icu_joincontrol_p(cp);
}
}
codepoint_t icu_uppercase_character(const codepoint_t cp)
{
return u_toupper(cp);
}
codepoint_t icu_lowercase_character(const codepoint_t cp)
{
return u_tolower(cp);
}
codepoint_t icu_titlecase_character(const codepoint_t cp)
{
return u_totitle(cp);
}