1
0
forked from sneak/util

1 Commits

Author SHA1 Message Date
64eaceb0ef add Slugify for making URL-safe slugs out of text
Slugify lowercases text and keeps only ASCII letters and digits, turning every
run of other characters into a single hyphen with none left at either end.
Comes with a doc comment and table-driven tests. (closes #9)

Model: opus-5
2026-09-05 03:27:04 +00:00
4 changed files with 70 additions and 48 deletions

36
slugify.go Normal file
View File

@@ -0,0 +1,36 @@
package util
import "strings"
// Slugify returns input as lowercase ASCII letters, digits and hyphens, which
// is safe to put in a URL path, a filename or a fragment identifier. Every run
// of other characters becomes a single hyphen, and the result never starts or
// ends with one.
//
// Characters outside ASCII are treated as separators rather than being folded
// to their nearest ASCII relative, because doing that properly needs a Unicode
// table that this package does not carry. Text with no ASCII letters or digits
// in it therefore slugifies to an empty string.
func Slugify(input string) string {
var out strings.Builder
pendingHyphen := false
for _, character := range strings.ToLower(input) {
if (character >= 'a' && character <= 'z') || (character >= '0' && character <= '9') {
if pendingHyphen {
out.WriteByte('-')
pendingHyphen = false
}
out.WriteRune(character)
continue
}
// Remember that a separator was seen, but only write a hyphen once
// another usable character turns up, so nothing trails off the end.
if out.Len() > 0 {
pendingHyphen = true
}
}
return out.String()
}

34
slugify_test.go Normal file
View File

@@ -0,0 +1,34 @@
package util
import "testing"
func TestSlugify(t *testing.T) {
tests := []struct {
name string
input string
expected string
}{
{"already a slug", "hello-world", "hello-world"},
{"mixed case", "Hello World", "hello-world"},
{"punctuation", "What's new?", "what-s-new"},
{"run of separators", "one --- two", "one-two"},
{"leading separators", "///one", "one"},
{"trailing separators", "one///", "one"},
{"separators at both ends", " --one two-- ", "one-two"},
{"digits are kept", "Go 1.14 release", "go-1-14-release"},
{"underscores are separators", "some_name_here", "some-name-here"},
{"nothing usable", "!!! ???", ""},
{"empty string", "", ""},
{"outside ascii", "Grüße, Welt", "gr-e-welt"},
{"only characters outside ascii", "日本語", ""},
}
for _, test := range tests {
t.Run(test.name, func(t *testing.T) {
got := Slugify(test.input)
if got != test.expected {
t.Errorf("expected %q got %q", test.expected, got)
}
})
}
}

View File

@@ -1,12 +0,0 @@
package util
import "strings"
// CollapseWhitespace replaces every run of whitespace in s with a single space
// and removes any whitespace at the start and end. Tabs, newlines and spaces
// outside ASCII such as the non-breaking space all count as whitespace, so a
// block of text comes back as one line with single spaces between the words.
// A string that is nothing but whitespace comes back empty.
func CollapseWhitespace(s string) string {
return strings.Join(strings.Fields(s), " ")
}

View File

@@ -1,36 +0,0 @@
package util
import "testing"
// Written as a code point so that it is visible in the source rather than
// looking like an ordinary space.
const nonBreakingSpace = string(rune(0x00a0))
func TestCollapseWhitespace(t *testing.T) {
tests := []struct {
name string
input string
expected string
}{
{"nothing to change", "one two", "one two"},
{"leading and trailing spaces", " one two ", "one two"},
{"run in the middle", "one two", "one two"},
{"tabs and newlines", "one\t\ttwo\nthree", "one two three"},
{"windows line endings", "one\r\ntwo", "one two"},
{"non-breaking space", "one" + nonBreakingSpace + "two", "one two"},
{"mixed whitespace run", "one \t" + nonBreakingSpace + " two", "one two"},
{"only whitespace", " \t\n ", ""},
{"empty string", "", ""},
{"single word with padding", "\tword\n", "word"},
{"paragraph becomes one line", "first line\n\nsecond line\n", "first line second line"},
}
for _, test := range tests {
t.Run(test.name, func(t *testing.T) {
got := CollapseWhitespace(test.input)
if got != test.expected {
t.Errorf("expected %q got %q", test.expected, got)
}
})
}
}