mirror of
https://github.com/h44z/wg-portal.git
synced 2026-09-03 21:36:44 +00:00
fix(domain): make SanitizeString idempotent (#745)
NFC normalisation ran before control and format characters were stripped.
Removing a character can leave a base letter next to a combining mark the
earlier pass never saw as a pair, so a second call composes it:
input U+0041 U+0009 U+0300 ("A", tab, combining grave)
once -> U+0041 U+0300
twice -> U+00C0
Found by TestPropertySanitizeStringIdempotent. Category Cf characters
behave the same way.
Strip first, then normalise. Normalisation still precedes truncation
because composing changes the rune count. One side effect: invalid UTF-8
is now dropped by the strip loop instead of surviving as U+FFFD, so an
identifier containing such bytes sanitises differently than before.
Signed-off-by: clark-ja <37738506+clark-ja@users.noreply.github.com>
This commit is contained in:
@@ -47,16 +47,25 @@ var reservedUserIdentifiers = map[string]struct{}{
|
||||
CtxSystemDBMigrator: {},
|
||||
}
|
||||
|
||||
// SanitizeString normalizes to NFC, trims leading and trailing whitespace, strips Unicode
|
||||
// control and format characters, drops invalid UTF-8 bytes, and truncates the result to
|
||||
// SanitizeString trims leading and trailing whitespace, strips Unicode control and format
|
||||
// characters, drops invalid UTF-8 bytes, normalizes to NFC, and truncates the result to
|
||||
// maxLen runes. If maxLen <= 0, returns "".
|
||||
//
|
||||
// The order matters and is load-bearing for idempotency: see the comments in the body.
|
||||
// SanitizeString(SanitizeString(s, n), n) == SanitizeString(s, n) for all s and n.
|
||||
func SanitizeString(s string, maxLen int) string {
|
||||
if maxLen <= 0 {
|
||||
return ""
|
||||
}
|
||||
|
||||
s = norm.NFC.String(strings.TrimSpace(s))
|
||||
s = strings.TrimSpace(s)
|
||||
|
||||
// Strip control/format characters and invalid UTF-8 *before* normalizing.
|
||||
// Normalizing first is not idempotent: removing a character can leave a
|
||||
// base letter next to a combining mark that the earlier normalization never
|
||||
// saw as a pair. For example "A\t̀" is already NFC (the tab keeps the
|
||||
// letter and the combining grave apart), but stripping the tab yields
|
||||
// "À", which a second call would compose to "À".
|
||||
var b strings.Builder
|
||||
b.Grow(len(s))
|
||||
for len(s) > 0 {
|
||||
@@ -69,7 +78,10 @@ func SanitizeString(s string, maxLen int) string {
|
||||
b.WriteRune(r)
|
||||
}
|
||||
}
|
||||
s = b.String()
|
||||
|
||||
// Normalize before truncating, not after: composition can change the rune
|
||||
// count, so normalizing afterwards could push the result back over maxLen.
|
||||
s = norm.NFC.String(b.String())
|
||||
|
||||
if utf8.RuneCountInString(s) > maxLen {
|
||||
runes := []rune(s)
|
||||
|
||||
Reference in New Issue
Block a user