Replace German characters (umlauts, accents) with english equivalents
The process is known as removing "diacritics" - see Removing diacritics (accents) from strings which uses the following code:
public static String RemoveDiacritics(String s)
{
String normalizedString = s.Normalize(NormalizationForm.FormD);
StringBuilder stringBuilder = new StringBuilder();
for (int i = 0; i < normalizedString.Length; i++)
{
Char c = normalizedString[i];
if (CharUnicodeInfo.GetUnicodeCategory(c) != UnicodeCategory.NonSpacingMark)
stringBuilder.Append(c);
}
return stringBuilder.ToString();
}
Taking inspiration from @Barry Kaye's answer I extended the function a little (and made it a String Extension.
The reason for this is that we need to convert german umlauts into combinations of ascii chars eg. ä = ae
.
It still uses string builder so it should be plenty fast.
You can call it like myStringVariable.RemoveDiacritics();
using System.Collections.Generic;
using System.Globalization;
using System.Text;
using System.Text.RegularExpressions;
namespace Core.Extensions
{
public static class StringExtensions
{
public static IReadOnlyDictionary<string, string> SPECIAL_DIACRITICS = new Dictionary<string, string>
{
{ "ä".Normalize(NormalizationForm.FormD), "ae".Normalize(NormalizationForm.FormD) },
{ "Ä".Normalize(NormalizationForm.FormD), "Ae".Normalize(NormalizationForm.FormD) },
{ "ö".Normalize(NormalizationForm.FormD), "oe".Normalize(NormalizationForm.FormD) },
{ "Ö".Normalize(NormalizationForm.FormD), "Oe".Normalize(NormalizationForm.FormD) },
{ "ü".Normalize(NormalizationForm.FormD), "ue".Normalize(NormalizationForm.FormD) },
{ "Ü".Normalize(NormalizationForm.FormD), "Ue".Normalize(NormalizationForm.FormD) },
{ "ß".Normalize(NormalizationForm.FormD), "ss".Normalize(NormalizationForm.FormD) },
};
public static string RemoveDiacritics(this string s)
{
var stringBuilder = new StringBuilder(s.Normalize(NormalizationForm.FormD));
// Replace certain special chars with special combinations of ascii chars (eg. german umlauts and german double s)
foreach (KeyValuePair<string, string> keyValuePair in SPECIAL_DIACRITICS)
stringBuilder.Replace(keyValuePair.Key, keyValuePair.Value);
// Remove other diacritic chars eg. non spacing marks https://www.compart.com/en/unicode/category/Mn
for (int i = 0; i < stringBuilder.Length; i++)
{
char c = stringBuilder[i];
if (CharUnicodeInfo.GetUnicodeCategory(c) == UnicodeCategory.NonSpacingMark)
stringBuilder.Remove(i, 1);
}
return stringBuilder.ToString();
}
}
}