From 0a92ec2235b5f42e93012be14938bb11e3f3650a Mon Sep 17 00:00:00 2001 From: Tyge Lovset Date: Tue, 31 May 2022 07:06:15 +0200 Subject: cleanup of icmp impl. --- include/stc/cstr.h | 6 ++---- include/stc/utf8.h | 7 ++++--- src/utf8code.c | 24 +++++++++++++++++------- 3 files changed, 23 insertions(+), 14 deletions(-) diff --git a/include/stc/cstr.h b/include/stc/cstr.h index 7bbd9f3b..7c788ce4 100644 --- a/include/stc/cstr.h +++ b/include/stc/cstr.h @@ -240,10 +240,8 @@ STC_INLINE char* cstr_expand_uninit(cstr *self, size_t n) { STC_INLINE int cstr_cmp(const cstr* s1, const cstr* s2) { return strcmp(cstr_str(s1), cstr_str(s2)); } -STC_INLINE int cstr_icmp(const cstr* s1, const cstr* s2) { - csview x = cstr_sv(s1), y = cstr_sv(s2); - return utf8_icmp_n(~(size_t)0, x.str, x.size, y.str, y.size); -} +STC_INLINE int cstr_icmp(const cstr* s1, const cstr* s2) + { return utf8_icmp(cstr_str(s1), cstr_str(s2)); } STC_INLINE bool cstr_eq(const cstr* s1, const cstr* s2) { csview x = cstr_sv(s1), y = cstr_sv(s2); diff --git a/include/stc/utf8.h b/include/stc/utf8.h index a3647e3c..e5fa5894 100644 --- a/include/stc/utf8.h +++ b/include/stc/utf8.h @@ -38,9 +38,9 @@ uint32_t utf8_toupper(uint32_t c); bool utf8_valid(const char* s); bool utf8_valid_n(const char* s, size_t n); + int utf8_icmp_n(size_t u8max, const char* s1, size_t n1, const char* s2, size_t n2); - /* encode/decode next utf8 codepoint. */ enum { UTF8_OK = 0, UTF8_ERROR = 4 }; typedef struct { uint32_t state, codep, size; } utf8_decode_t; @@ -50,8 +50,9 @@ unsigned utf8_encode(char *out, uint32_t c); void utf8_decode(utf8_decode_t *d, const uint8_t b); /* case-insensitive utf8 string comparison */ -STC_INLINE int utf8_icmp(const char* s1, const char* s2) - { return utf8_icmp_n(~(size_t)0, s1, ~(size_t)0, s2, ~(size_t)0); } +STC_INLINE int utf8_icmp(const char* s1, const char* s2) { + return utf8_icmp_n(~(size_t)0, s1, ~(size_t)0, s2, ~(size_t)0); +} /* number of characters in the utf8 codepoint from s */ STC_INLINE unsigned utf8_codep_size(const char *s) { diff --git a/src/utf8code.c b/src/utf8code.c index 2f429541..543070c3 100644 --- a/src/utf8code.c +++ b/src/utf8code.c @@ -99,18 +99,28 @@ uint32_t utf8_toupper(uint32_t c) { } return c; } - +/* +int utf8_icmp(const char* s1, const char* s2) { + utf8_decode_t d1 = {UTF8_OK}, d2 = {UTF8_OK}; + for (;; s1 += d1.size, s2 += d2.size) { + utf8_peek(&d1, s1); + utf8_peek(&d2, s2); + int c = utf8_tolower(d1.codep) - utf8_tolower(d2.codep); + if (c || !*s2) + return c; + } +} +*/ int utf8_icmp_n(size_t u8max, const char* s1, const size_t n1, const char* s2, const size_t n2) { - int ret = 0; utf8_decode_t d1 = {UTF8_OK}, d2 = {UTF8_OK}; size_t j1 = 0, j2 = 0; for (; u8max-- && ((j1 < n1) & (j2 < n2)); j1 += d1.size, j2 += d2.size) { - utf8_peek(&d1, s1+j1); - utf8_peek(&d2, s2+j2); - ret = utf8_tolower(d1.codep) - utf8_tolower(d2.codep); - if (ret || !s2[j2]) - return ret; + utf8_peek(&d1, s1 + j1); + utf8_peek(&d2, s2 + j2); + int c = utf8_tolower(d1.codep) - utf8_tolower(d2.codep); + if (c || !s2[j2]) + return c; } return (j2 < n2) - (j1 < n1); } -- cgit v1.2.3