utils: add utf8_clean helper
To morph text into valid utf8 if it isn't already.
This commit is contained in:
@ -164,6 +164,24 @@ static inline bool contains_unbroken_script(const std::string& str) {
|
|||||||
return contains_unbroken_script(str.c_str());
|
return contains_unbroken_script(str.c_str());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* If the string is already valid utf8, return it
|
||||||
|
* otherwise, return a valid utf8 version
|
||||||
|
*
|
||||||
|
* @param str some string
|
||||||
|
*
|
||||||
|
* @return a utf8-string
|
||||||
|
*/
|
||||||
|
static inline std::string utf8_clean(std::string&& str) {
|
||||||
|
if (!g_utf8_validate(str.c_str(), static_cast<gssize>(str.length()), {})) {
|
||||||
|
gchar* clean{g_utf8_make_valid(
|
||||||
|
str.c_str(), static_cast<gssize>(str.length()))};
|
||||||
|
str = clean;
|
||||||
|
g_free(clean);
|
||||||
|
}
|
||||||
|
return std::move(str);
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Flatten a string -- down-case and fold diacritics.
|
* Flatten a string -- down-case and fold diacritics.
|
||||||
*
|
*
|
||||||
@ -172,7 +190,7 @@ static inline bool contains_unbroken_script(const std::string& str) {
|
|||||||
* @return a flattened string
|
* @return a flattened string
|
||||||
*/
|
*/
|
||||||
std::string utf8_flatten(const char* str);
|
std::string utf8_flatten(const char* str);
|
||||||
inline std::string
|
static inline std::string
|
||||||
utf8_flatten(const std::string& s) {
|
utf8_flatten(const std::string& s) {
|
||||||
return utf8_flatten(s.c_str());
|
return utf8_flatten(s.c_str());
|
||||||
}
|
}
|
||||||
|
|||||||
@ -1,5 +1,5 @@
|
|||||||
/*
|
/*
|
||||||
** Copyright (C) 2017-2022 Dirk-Jan C. Binnema <djcb@djcbsoftware.nl>
|
** Copyright (C) 2017-2025 Dirk-Jan C. Binnema <djcb@djcbsoftware.nl>
|
||||||
**
|
**
|
||||||
** This library is free software; you can redistribute it and/or
|
** This library is free software; you can redistribute it and/or
|
||||||
** modify it under the terms of the GNU Lesser General Public License
|
** modify it under the terms of the GNU Lesser General Public License
|
||||||
@ -149,6 +149,22 @@ test_parse_size()
|
|||||||
g_assert_false(!!parse_size("scoobydoobydoo", false));
|
g_assert_false(!!parse_size("scoobydoobydoo", false));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
static void
|
||||||
|
test_utf8_clean()
|
||||||
|
{
|
||||||
|
assert_equal(utf8_clean("James Holden"), "James Holden");
|
||||||
|
|
||||||
|
const uint8_t invalid_bytes[] ={ 'a' , 0xff, 'c', '\0'};
|
||||||
|
std::string invalid{reinterpret_cast<const char*>(invalid_bytes),
|
||||||
|
sizeof(invalid_bytes) - 1};
|
||||||
|
|
||||||
|
g_assert_false(g_utf8_validate(invalid.c_str(), invalid.length(), nullptr));
|
||||||
|
|
||||||
|
const auto valid = utf8_clean(std::move(invalid));
|
||||||
|
g_assert_true(g_utf8_validate(valid.c_str(), valid.length(), {}));
|
||||||
|
assert_equal(valid, "a\357\277\275c"); // replacement char
|
||||||
|
}
|
||||||
|
|
||||||
static void
|
static void
|
||||||
test_flatten()
|
test_flatten()
|
||||||
{
|
{
|
||||||
@ -159,7 +175,7 @@ test_flatten()
|
|||||||
{"đodø", true, "dodo"},
|
{"đodø", true, "dodo"},
|
||||||
|
|
||||||
// don't touch combining characters in CJK etc.
|
// don't touch combining characters in CJK etc.
|
||||||
{"スポンサーシップ募集",true, "スポンサーシップ募集"}
|
{"スポンサーシップ募集", true, "スポンサーシップ募集"}
|
||||||
};
|
};
|
||||||
|
|
||||||
test_cases(cases, [](auto s, auto f) { return utf8_flatten(s); });
|
test_cases(cases, [](auto s, auto f) { return utf8_flatten(s); });
|
||||||
@ -192,7 +208,6 @@ test_clean()
|
|||||||
test_cases(cases, [](auto s, auto f) { return utf8_clean(s); });
|
test_cases(cases, [](auto s, auto f) { return utf8_clean(s); });
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
static void
|
static void
|
||||||
test_word_break()
|
test_word_break()
|
||||||
{
|
{
|
||||||
@ -206,7 +221,6 @@ test_word_break()
|
|||||||
test_cases(cases, [](auto s, auto f) { return utf8_wordbreak(s); });
|
test_cases(cases, [](auto s, auto f) { return utf8_wordbreak(s); });
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
static void
|
static void
|
||||||
test_format()
|
test_format()
|
||||||
{
|
{
|
||||||
@ -327,6 +341,7 @@ main(int argc, char* argv[])
|
|||||||
g_test_add_func("/utils/date-basic", test_date_basic);
|
g_test_add_func("/utils/date-basic", test_date_basic);
|
||||||
g_test_add_func("/utils/date-ymwdhMs", test_date_ymwdhMs);
|
g_test_add_func("/utils/date-ymwdhMs", test_date_ymwdhMs);
|
||||||
g_test_add_func("/utils/parse-size", test_parse_size);
|
g_test_add_func("/utils/parse-size", test_parse_size);
|
||||||
|
g_test_add_func("/utils/utf8-clean", test_utf8_clean);
|
||||||
g_test_add_func("/utils/flatten", test_flatten);
|
g_test_add_func("/utils/flatten", test_flatten);
|
||||||
g_test_add_func("/utils/remove-ctrl", test_remove_ctrl);
|
g_test_add_func("/utils/remove-ctrl", test_remove_ctrl);
|
||||||
g_test_add_func("/utils/clean", test_clean);
|
g_test_add_func("/utils/clean", test_clean);
|
||||||
|
|||||||
Reference in New Issue
Block a user