/usr/include/unicode
NameSizeModeActions
alphaindex.h271580644editdlrm
appendable.h87390644editdlrm
basictz.h101630644editdlrm
brkiter.h284780644editdlrm
bytestream.h110130644editdlrm
bytestrie.h212680644editdlrm
bytestriebuilder.h76400644editdlrm
calendar.h1084240644editdlrm
caniter.h76210644editdlrm
casemap.h259330644editdlrm
char16ptr.h73930644editdlrm
chariter.h246380644editdlrm
choicfmt.h245440644editdlrm
coleitr.h141020644editdlrm
coll.h576150644editdlrm
compactdecimalformat.h70470644editdlrm
curramt.h38690644editdlrm
currpinf.h74790644editdlrm
currunit.h41120644editdlrm
datefmt.h416900644editdlrm
dbbi.h12230644editdlrm
dcfmtsym.h211010644editdlrm
decimfmt.h896760644editdlrm
docmain.h73800644editdlrm
dtfmtsym.h391260644editdlrm
dtintrv.h39350644editdlrm
dtitvfmt.h504480644editdlrm
dtitvinf.h190720644editdlrm
dtptngen.h266920644editdlrm
dtrule.h88980644editdlrm
edits.h212370644editdlrm
enumset.h21300644editdlrm
errorcode.h49560644editdlrm
fieldpos.h89060644editdlrm
filteredbrk.h55010644editdlrm
fmtable.h250200644editdlrm
format.h128020644editdlrm
formattedvalue.h99870644editdlrm
fpositer.h31070644editdlrm
gender.h34080644editdlrm
gregocal.h326420644editdlrm
icudataver.h10490644editdlrm
icuplug.h121090644editdlrm
idna.h130000644editdlrm
listformatter.h90040644editdlrm
localebuilder.h113500644editdlrm
localematcher.h274670644editdlrm
localpointer.h198250644editdlrm
locdspnm.h72920644editdlrm
locid.h488040644editdlrm
measfmt.h116900644editdlrm
measunit.h1076130644editdlrm
measure.h44310644editdlrm
messagepattern.h345170644editdlrm
msgfmt.h452780644editdlrm
normalizer2.h344660644editdlrm
normlzr.h316900644editdlrm
nounit.h23050644editdlrm
numberformatter.h945050644editdlrm
numberrangeformatter.h257900644editdlrm
numfmt.h510340644editdlrm
numsys.h75090644editdlrm
parseerr.h31550644editdlrm
parsepos.h56990644editdlrm
platform.h287680644editdlrm
plurfmt.h258470644editdlrm
plurrule.h213160644editdlrm
ptypes.h35770644editdlrm
putil.h64710644editdlrm
rbbi.h284130644editdlrm
rbnf.h500210644editdlrm
rbtz.h162100644editdlrm
regex.h863850644editdlrm
region.h94020644editdlrm
reldatefmt.h227510644editdlrm
rep.h95990644editdlrm
resbund.h185130644editdlrm
schriter.h65110644editdlrm
scientificnumberformatter.h65890644editdlrm
search.h227550644editdlrm
selfmt.h146860644editdlrm
simpleformatter.h128880644editdlrm
simpletz.h467980644editdlrm
smpdtfmt.h727780644editdlrm
sortkey.h114520644editdlrm
std_string.h10760644editdlrm
strenum.h101570644editdlrm
stringoptions.h59260644editdlrm
stringpiece.h102870644editdlrm
stringtriebuilder.h158420644editdlrm
stsearch.h219260644editdlrm
symtable.h43740644editdlrm
tblcoll.h378010644editdlrm
timezone.h448580644editdlrm
tmunit.h34790644editdlrm
tmutamt.h50290644editdlrm
tmutfmt.h76020644editdlrm
translit.h673970644editdlrm
tzfmt.h439610644editdlrm
tznames.h172510644editdlrm
tzrule.h364500644editdlrm
tztrans.h62780644editdlrm
ubidi.h917590644editdlrm
ubiditransform.h130030644editdlrm
ubrk.h250430644editdlrm
ucal.h624630644editdlrm
ucasemap.h155790644editdlrm
ucat.h54780644editdlrm
uchar.h1481110644editdlrm
ucharstrie.h230660644editdlrm
ucharstriebuilder.h76450644editdlrm
uchriter.h137470644editdlrm
uclean.h114680644editdlrm
ucnv.h850510644editdlrm
ucnvsel.h63390644editdlrm
ucnv_cb.h67380644editdlrm
ucnv_err.h214810644editdlrm
ucol.h629870644editdlrm
ucoleitr.h100560644editdlrm
uconfig.h123560644editdlrm
ucpmap.h56630644editdlrm
ucptrie.h230440644editdlrm
ucsdet.h150430644editdlrm
ucurr.h171220644editdlrm
udat.h639020644editdlrm
udata.h160040644editdlrm
udateintervalformat.h122180644editdlrm
udatpg.h274050644editdlrm
udisplaycontext.h60840644editdlrm
uenum.h79810644editdlrm
ufieldpositer.h45130644editdlrm
uformattable.h112330644editdlrm
uformattedvalue.h126000644editdlrm
ugender.h21060644editdlrm
uidna.h342290644editdlrm
uiter.h232990644editdlrm
uldnames.h107330644editdlrm
ulistformatter.h110430644editdlrm
uloc.h559360644editdlrm
ulocdata.h115720644editdlrm
umachine.h164390644editdlrm
umisc.h13650644editdlrm
umsg.h248320644editdlrm
umutablecptrie.h84900644editdlrm
unifilt.h40910644editdlrm
unifunct.h41510644editdlrm
unimatch.h62440644editdlrm
unirepl.h34640644editdlrm
uniset.h681080644editdlrm
unistr.h1746270644editdlrm
unorm.h210420644editdlrm
unorm2.h252690644editdlrm
unum.h562690644editdlrm
unumberformatter.h311770644editdlrm
unumberrangeformatter.h157370644editdlrm
unumsys.h74300644editdlrm
uobject.h109340644editdlrm
upluralrules.h89970644editdlrm
uregex.h737190644editdlrm
uregion.h100470644editdlrm
ureldatefmt.h174480644editdlrm
urename.h1378530644editdlrm
urep.h55070644editdlrm
ures.h374180644editdlrm
uscript.h282910644editdlrm
usearch.h401530644editdlrm
uset.h455030644editdlrm
usetiter.h99100644editdlrm
ushape.h184300644editdlrm
uspoof.h674250644editdlrm
usprep.h83820644editdlrm
ustdio.h394820644editdlrm
ustream.h19340644editdlrm
ustring.h740960644editdlrm
ustringtrie.h32240644editdlrm
utext.h594710644editdlrm
utf.h80570644editdlrm
utf8.h315720644editdlrm
utf16.h239100644editdlrm
utf32.h7630644editdlrm
utf_old.h469290644editdlrm
utmscale.h141070644editdlrm
utrace.h175950644editdlrm
utrans.h261570644editdlrm
utypes.h318070644editdlrm
uvernum.h68390644editdlrm
uversion.h61370644editdlrm
vtzone.h213030644editdlrm
Edit: /usr/include/unicode/utf.h (8057B)
// © 2016 and later: Unicode, Inc. and others. // License & terms of use: http://www.unicode.org/copyright.html /* ******************************************************************************* * * Copyright (C) 1999-2011, International Business Machines * Corporation and others. All Rights Reserved. * ******************************************************************************* * file name: utf.h * encoding: UTF-8 * tab size: 8 (not used) * indentation:4 * * created on: 1999sep09 * created by: Markus W. Scherer */ /** * \file * \brief C API: Code point macros * * This file defines macros for checking whether a code point is * a surrogate or a non-character etc. * * If U_NO_DEFAULT_INCLUDE_UTF_HEADERS is 0 then utf.h is included by utypes.h * and itself includes utf8.h and utf16.h after some * common definitions. * If U_NO_DEFAULT_INCLUDE_UTF_HEADERS is 1 then each of these headers must be * included explicitly if their definitions are used. * * utf8.h and utf16.h define macros for efficiently getting code points * in and out of UTF-8/16 strings. * utf16.h macros have "U16_" prefixes. * utf8.h defines similar macros with "U8_" prefixes for UTF-8 string handling. * * ICU mostly processes 16-bit Unicode strings. * Most of the time, such strings are well-formed UTF-16. * Single, unpaired surrogates must be handled as well, and are treated in ICU * like regular code points where possible. * (Pairs of surrogate code points are indistinguishable from supplementary * code points encoded as pairs of supplementary code units.) * * In fact, almost all Unicode code points in normal text (>99%) * are on the BMP (<=U+ffff) and even <=U+d7ff. * ICU functions handle supplementary code points (U+10000..U+10ffff) * but are optimized for the much more frequently occurring BMP code points. * * umachine.h defines UChar to be an unsigned 16-bit integer. * Since ICU 59, ICU uses char16_t in C++, UChar only in C, * and defines UChar=char16_t by default. See the UChar API docs for details. * * UChar32 is defined to be a signed 32-bit integer (int32_t), large enough for a 21-bit * Unicode code point (Unicode scalar value, 0..0x10ffff) and U_SENTINEL (-1). * Before ICU 2.4, the definition of UChar32 was similarly platform-dependent as * the definition of UChar. For details see the documentation for UChar32 itself. * * utf.h defines a small number of C macros for single Unicode code points. * These are simple checks for surrogates and non-characters. * For actual Unicode character properties see uchar.h. * * By default, string operations must be done with error checking in case * a string is not well-formed UTF-16 or UTF-8. * * The U16_ macros detect if a surrogate code unit is unpaired * (lead unit without trail unit or vice versa) and just return the unit itself * as the code point. * * The U8_ macros detect illegal byte sequences and return a negative value. * Starting with ICU 60, the observable length of a single illegal byte sequence * skipped by one of these macros follows the Unicode 6+ recommendation * which is consistent with the W3C Encoding Standard. * * There are ..._OR_FFFD versions of both U16_ and U8_ macros * that return U+FFFD for illegal code unit sequences. * * The regular "safe" macros require that the initial, passed-in string index * is within bounds. They only check the index when they read more than one * code unit. This is usually done with code similar to the following loop: *
while(i
 *
 * When it is safe to assume that text is well-formed UTF-16
 * (does not contain single, unpaired surrogates), then one can use
 * U16_..._UNSAFE macros.
 * These do not check for proper code unit sequences or truncated text and may
 * yield wrong results or even cause a crash if they are used with "malformed"
 * text.
 * In practice, U16_..._UNSAFE macros will produce slightly less code but
 * should not be faster because the processing is only different when a
 * surrogate code unit is detected, which will be rare.
 *
 * Similarly for UTF-8, there are "safe" macros without a suffix,
 * and U8_..._UNSAFE versions.
 * The performance differences are much larger here because UTF-8 provides so
 * many opportunities for malformed sequences.
 * The unsafe UTF-8 macros are entirely implemented inside the macro definitions
 * and are fast, while the safe UTF-8 macros call functions for some complicated cases.
 *
 * Unlike with UTF-16, malformed sequences cannot be expressed with distinct
 * code point values (0..U+10ffff). They are indicated with negative values instead.
 *
 * For more information see the ICU User Guide Strings chapter
 * (https://unicode-org.github.io/icu/userguide/strings).
 *
 * Usage:
 * ICU coding guidelines for if() statements should be followed when using these macros.
 * Compound statements (curly braces {}) must be used  for if-else-while... 
 * bodies and all macro statements should be terminated with semicolon.
 *
 * @stable ICU 2.4
 */

#ifndef __UTF_H__
#define __UTF_H__

#include "unicode/umachine.h"
/* include the utfXX.h after the following definitions */

/* single-code point definitions -------------------------------------------- */

/**
 * Is this code point a Unicode noncharacter?
 * @param c 32-bit code point
 * @return true or false
 * @stable ICU 2.4
 */
#define U_IS_UNICODE_NONCHAR(c) \
    ((c)>=0xfdd0 && \
     ((c)<=0xfdef || ((c)&0xfffe)==0xfffe) && (c)<=0x10ffff)

/**
 * Is c a Unicode code point value (0..U+10ffff)
 * that can be assigned a character?
 *
 * Code points that are not characters include:
 * - single surrogate code points (U+d800..U+dfff, 2048 code points)
 * - the last two code points on each plane (U+__fffe and U+__ffff, 34 code points)
 * - U+fdd0..U+fdef (new with Unicode 3.1, 32 code points)
 * - the highest Unicode code point value is U+10ffff
 *
 * This means that all code points below U+d800 are character code points,
 * and that boundary is tested first for performance.
 *
 * @param c 32-bit code point
 * @return true or false
 * @stable ICU 2.4
 */
#define U_IS_UNICODE_CHAR(c) \
    ((uint32_t)(c)<0xd800 || \
        (0xdfff<(c) && (c)<=0x10ffff && !U_IS_UNICODE_NONCHAR(c)))

/**
 * Is this code point a BMP code point (U+0000..U+ffff)?
 * @param c 32-bit code point
 * @return true or false
 * @stable ICU 2.8
 */
#define U_IS_BMP(c) ((uint32_t)(c)<=0xffff)

/**
 * Is this code point a supplementary code point (U+10000..U+10ffff)?
 * @param c 32-bit code point
 * @return true or false
 * @stable ICU 2.8
 */
#define U_IS_SUPPLEMENTARY(c) ((uint32_t)((c)-0x10000)<=0xfffff)
 
/**
 * Is this code point a lead surrogate (U+d800..U+dbff)?
 * @param c 32-bit code point
 * @return true or false
 * @stable ICU 2.4
 */
#define U_IS_LEAD(c) (((c)&0xfffffc00)==0xd800)

/**
 * Is this code point a trail surrogate (U+dc00..U+dfff)?
 * @param c 32-bit code point
 * @return true or false
 * @stable ICU 2.4
 */
#define U_IS_TRAIL(c) (((c)&0xfffffc00)==0xdc00)

/**
 * Is this code point a surrogate (U+d800..U+dfff)?
 * @param c 32-bit code point
 * @return true or false
 * @stable ICU 2.4
 */
#define U_IS_SURROGATE(c) (((c)&0xfffff800)==0xd800)

/**
 * Assuming c is a surrogate code point (U_IS_SURROGATE(c)),
 * is it a lead surrogate?
 * @param c 32-bit code point
 * @return true or false
 * @stable ICU 2.4
 */
#define U_IS_SURROGATE_LEAD(c) (((c)&0x400)==0)

/**
 * Assuming c is a surrogate code point (U_IS_SURROGATE(c)),
 * is it a trail surrogate?
 * @param c 32-bit code point
 * @return true or false
 * @stable ICU 4.2
 */
#define U_IS_SURROGATE_TRAIL(c) (((c)&0x400)!=0)

/* include the utfXX.h ------------------------------------------------------ */

#if !U_NO_DEFAULT_INCLUDE_UTF_HEADERS

#include "unicode/utf8.h"
#include "unicode/utf16.h"

/* utf_old.h contains deprecated, pre-ICU 2.4 definitions */
#include "unicode/utf_old.h"

#endif  /* !U_NO_DEFAULT_INCLUDE_UTF_HEADERS */

#endif  /* __UTF_H__ */