Skip to content

Commit

Permalink
feat: /text/textutil: add RemoveDiacritics()
Browse files Browse the repository at this point in the history
  • Loading branch information
grokify committed May 4, 2024
1 parent 841fa48 commit e9d2279
Show file tree
Hide file tree
Showing 2 changed files with 45 additions and 0 deletions.
17 changes: 17 additions & 0 deletions text/textutil/text.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,17 @@
package textutil

import (
"strings"
"unicode"

"golang.org/x/text/runes"
"golang.org/x/text/transform"
"golang.org/x/text/unicode/norm"
)

func RemoveDiacritics(s string) (string, error) {
// Should å -> aa: https://stackoverflow.com/questions/11248467/convert-unicode-to-double-ascii-letters-in-python-%C3%9F-ss
t := transform.Chain(norm.NFD, runes.Remove(runes.In(unicode.Mn)), norm.NFC)
result, _, err := transform.String(t, strings.Replace(s, "\u00df", "ss", -1))
return result, err
}
28 changes: 28 additions & 0 deletions text/textutil/text_test.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,28 @@
package textutil

import (
"testing"
)

var removeDiacriticsTests = []struct {
v string
want string
}{
{"å", "a"},
{"ß", "ss"},
{"žůžo", "zuzo"},
}

func TestRemoveDiacritics(t *testing.T) {
for _, tt := range removeDiacriticsTests {
try, err := RemoveDiacritics(tt.v)
if err != nil {
t.Errorf("strconvutil.RemoveDiacritics(\"%s\") Error: [%s]",
tt.v, err.Error())
}
if err == nil && try != tt.want {
t.Errorf("strconvutil.RemoveDiacritics(\"%s\" Error: want [%s], got [%s]",
tt.v, tt.want, try)
}
}
}

0 comments on commit e9d2279

Please sign in to comment.