diff options
| -rw-r--r-- | include/stc/ccommon.h | 8 | ||||
| -rw-r--r-- | include/stc/utf8.h | 6 | ||||
| -rw-r--r-- | src/utf8code.c | 13 |
3 files changed, 23 insertions, 4 deletions
diff --git a/include/stc/ccommon.h b/include/stc/ccommon.h index 9c8ba414..2ebcf7c1 100644 --- a/include/stc/ccommon.h +++ b/include/stc/ccommon.h @@ -120,7 +120,7 @@ typedef const char* crawstr; #define _c_ROTL(x, k) (x << (k) | x >> (8*sizeof(x) - (k)))
STC_INLINE uint64_t c_fasthash(const void* key, size_t len) {
- const uint8_t *x = (const uint8_t*) key;
+ const uint8_t *x = (const uint8_t*) key;
uint64_t u8, h = 1; size_t n = len >> 3;
uint32_t u4;
while (n--) {
@@ -137,16 +137,16 @@ STC_INLINE uint64_t c_fasthash(const void* key, size_t len) { return _c_ROTL(h, 26) ^ h;
}
-STC_INLINE uint64_t c_strhash(const char *str)
+STC_INLINE uint64_t c_strhash(const char *str)
{ return c_fasthash(str, strlen(str)); }
-STC_INLINE char* c_strnstrn(const char *s, const char *needle,
+STC_INLINE char* c_strnstrn(const char *s, const char *needle,
size_t slen, const size_t nlen) {
if (!nlen) return (char *)s;
if (nlen > slen) return NULL;
slen -= nlen;
do {
- if (*s == *needle && !memcmp(s, needle, nlen))
+ if (*s == *needle && !memcmp(s, needle, nlen))
return (char *)s;
++s;
} while (slen--);
diff --git a/include/stc/utf8.h b/include/stc/utf8.h index 67722345..333a0f92 100644 --- a/include/stc/utf8.h +++ b/include/stc/utf8.h @@ -35,8 +35,10 @@ bool utf8_isalpha(uint32_t c); bool utf8_isalnum(uint32_t c);
uint32_t utf8_tolower(uint32_t c);
uint32_t utf8_toupper(uint32_t c);
+
bool utf8_valid(const char* s);
bool utf8_valid_n(const char* s, size_t n);
+int utf8_icmp_n(const char* s1, const char* s2, size_t u8max);
/* encode/decode next utf8 codepoint. */
enum { UTF8_OK = 0, UTF8_ERROR = 4 };
@@ -46,6 +48,10 @@ void utf8_peek(utf8_decode_t* d, const char *s); unsigned utf8_encode(char *out, uint32_t c);
void utf8_decode(utf8_decode_t *d, const uint8_t b);
+/* case-insensitive utf8 string comparison */
+STC_INLINE int utf8_icmp(const char* s1, const char* s2)
+ { return utf8_icmp_n(s1, s2, (size_t)(-1)); }
+
/* number of characters in the utf8 codepoint from s */
STC_INLINE unsigned utf8_codep_size(const char *s) {
unsigned b = (uint8_t)*s;
diff --git a/src/utf8code.c b/src/utf8code.c index c58f78b6..cef3efca 100644 --- a/src/utf8code.c +++ b/src/utf8code.c @@ -100,6 +100,19 @@ uint32_t utf8_toupper(uint32_t c) { return c; } +int utf8_icmp_n(const char* s1, const char* s2, size_t u8max) { + int ret = 0; + utf8_decode_t d1 = {UTF8_OK}, d2 = {UTF8_OK}; + for (; u8max--; s1 += d1.size, s2 += d2.size) { + utf8_peek(&d1, s1); + utf8_peek(&d2, s2); + ret = utf8_tolower(d1.codep) - utf8_tolower(d2.codep); + if (ret || !*s2) + break; + } + return ret; +} + bool utf8_isupper(uint32_t c) { return utf8_tolower(c) != c; } |
