tree-wide: drop copyright headers from frequent contributors

[elogind.git] / src / basic / utf8.c
diff --git a/src/basic/utf8.c b/src/basic/utf8.c

index 6eae2b983d8dfd7943c479be9918d8539fd538fd..42818358ea828e9f4b0d7c0e6559c2daab385a32 100644 (file)
--- a/src/basic/utf8.c
+++ b/src/basic/utf8.c
@@ -1,22 +1,4 @@
-/***
-  This file is part of systemd.
-
-  Copyright 2008-2011 Kay Sievers
-  Copyright 2012 Lennart Poettering
-
-  systemd is free software; you can redistribute it and/or modify it
-  under the terms of the GNU Lesser General Public License as published by
-  the Free Software Foundation; either version 2.1 of the License, or
-  (at your option) any later version.
-
-  systemd is distributed in the hope that it will be useful, but
-  WITHOUT ANY WARRANTY; without even the implied warranty of
-  MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
-  Lesser General Public License for more details.
-
-  You should have received a copy of the GNU Lesser General Public License
-  along with systemd; If not, see <http://www.gnu.org/licenses/>.
-***/
+/* SPDX-License-Identifier: LGPL-2.1+ */
  
  /* Parts of this file are based on the GLIB utf8 validation functions. The
   * original license text follows. */
@@ -47,6 +29,7 @@
  #include <string.h>
  
  #include "alloc-util.h"
+//#include "gunicode.h"
  #include "hexdecoct.h"
  #include "macro.h"
  #include "utf8.h"
@@ -73,7 +56,7 @@ static bool unichar_is_control(char32_t ch) {
            '\t' is in C0 range, but more or less harmless and commonly used.
          */
  
-        return (ch < ' ' && ch != '\t' && ch != '\n') ||
+        return (ch < ' ' && !IN_SET(ch, '\t', '\n')) ||
                  (0x7F <= ch && ch <= 0x9F);
  }
  
@@ -258,6 +241,9 @@ char *utf8_escape_non_printable(const char *str) {
  char *ascii_is_valid(const char *str) {
          const char *p;
  
+        /* Check whether the string consists of valid ASCII bytes,
+         * i.e values between 0 and 127, inclusive. */
+
          assert(str);
  
          for (p = str; *p; p++)
@@ -267,6 +253,21 @@ char *ascii_is_valid(const char *str) {
          return (char*) str;
  }
  
+char *ascii_is_valid_n(const char *str, size_t len) {
+        size_t i;
+
+        /* Very similar to ascii_is_valid(), but checks exactly len
+         * bytes and rejects any NULs in that range. */
+
+        assert(str);
+
+        for (i = 0; i < len; i++)
+                if ((unsigned char) str[i] >= 128 || str[i] == 0)
+                        return NULL;
+
+        return (char*) str;
+}
+
  /**
   * utf8_encode_unichar() - Encode single UCS-4 character as UTF-8
   * @out_utf8: output buffer of at least 4 bytes or NULL
@@ -407,3 +408,42 @@ int utf8_encoded_valid_unichar(const char *str) {
  
          return len;
  }
+
+size_t utf8_n_codepoints(const char *str) {
+        size_t n = 0;
+
+        /* Returns the number of UTF-8 codepoints in this string, or (size_t) -1 if the string is not valid UTF-8. */
+
+        while (*str != 0) {
+                int k;
+
+                k = utf8_encoded_valid_unichar(str);
+                if (k < 0)
+                        return (size_t) -1;
+
+                str += k;
+                n++;
+        }
+
+        return n;
+}
+
+size_t utf8_console_width(const char *str) {
+        size_t n = 0;
+
+        /* Returns the approximate width a string will take on screen when printed on a character cell
+         * terminal/console. */
+
+        while (*str != 0) {
+                char32_t c;
+
+                if (utf8_encoded_to_unichar(str, &c) < 0)
+                        return (size_t) -1;
+
+                str = utf8_next_char(str);
+
+                n += unichar_iswide(c) ? 2 : 1;
+        }
+
+        return n;
+}