Compare commits

..
8 Commits
Author SHA1 Message Date
sneak 0183d1a782 Merge pull request 'Add DedupeStrings to remove repeated values from a string slice' (#2) from clawbot/util:proposal-dedupe-strings into master
Reviewed-on: #2
2026-09-05 05:46:18 +02:00
sneak 2d714eea9f Merge pull request 'Add ChunkStrings to split a string slice into fixed-size batches' (#4) from clawbot/util:proposal-chunk-strings into master
Reviewed-on: #4
2026-09-05 05:46:03 +02:00
sneak 1a6d1b429b Merge pull request 'Add TruncateString to shorten a string without splitting a character' (#6) from clawbot/util:proposal-truncate-string into master
Reviewed-on: #6
2026-09-05 05:45:38 +02:00
sneak c656ce964e Merge pull request 'Add CollapseWhitespace to squeeze runs of whitespace into single spaces' (#8) from clawbot/util:proposal-collapse-whitespace into master
Reviewed-on: #8
2026-09-05 05:45:12 +02:00
clawbot d52588d635 add CollapseWhitespace for flattening runs of whitespace
CollapseWhitespace turns every run of whitespace into a single space and
trims the ends, so a block of text becomes one tidy line. Comes with a doc
comment and table-driven tests. (closes #7)

Model: opus-5
2026-09-05 03:26:29 +00:00
clawbot daaea4e10f add TruncateString for shortening a string safely
TruncateString returns at most the requested number of characters from the
start of a string, counting characters rather than bytes so a multi-byte
character is never cut in half. Comes with a doc comment and table-driven
tests. (closes #5)

Model: opus-5
2026-09-05 03:25:08 +00:00
clawbot 6807a76f24 add ChunkStrings for splitting a string slice into fixed-size batches
ChunkStrings cuts a slice into consecutive pieces of at most the requested
size, with the leftovers in the last piece, and returns copies so the pieces
cannot disturb each other. Comes with a doc comment and table-driven tests.
(closes #3)

Model: opus-5
2026-09-05 03:24:31 +00:00
clawbot 7fb688772e add DedupeStrings for removing repeated values from a string slice
DedupeStrings returns a new slice holding each value once, in the order each
value first appeared, and leaves the input slice alone. Comes with a doc
comment and table-driven tests. (closes #1)

Model: opus-5
2026-09-05 03:23:28 +00:00
8 changed files with 242 additions and 0 deletions
+26
View File
@@ -0,0 +1,26 @@
package util
// ChunkStrings splits input into consecutive slices of at most size elements.
// The last slice holds whatever is left over and may be shorter than the
// others. Each returned slice is a copy, so appending to one of them cannot
// disturb another. A size below one, or an empty input, returns nil.
func ChunkStrings(input []string, size int) [][]string {
if size < 1 || len(input) == 0 {
return nil
}
out := make([][]string, 0, (len(input)+size-1)/size)
for start := 0; start < len(input); start += size {
end := start + size
if end > len(input) {
end = len(input)
}
chunk := make([]string, end-start)
copy(chunk, input[start:end])
out = append(out, chunk)
}
return out
}
+43
View File
@@ -0,0 +1,43 @@
package util
import (
"reflect"
"testing"
)
func TestChunkStrings(t *testing.T) {
tests := []struct {
name string
input []string
size int
expected [][]string
}{
{"exact multiple", []string{"a", "b", "c", "d"}, 2, [][]string{{"a", "b"}, {"c", "d"}}},
{"with a remainder", []string{"a", "b", "c"}, 2, [][]string{{"a", "b"}, {"c"}}},
{"size larger than input", []string{"a", "b"}, 5, [][]string{{"a", "b"}}},
{"size of one", []string{"a", "b"}, 1, [][]string{{"a"}, {"b"}}},
{"size of zero", []string{"a", "b"}, 0, nil},
{"negative size", []string{"a", "b"}, -1, nil},
{"empty input", []string{}, 2, nil},
{"nil input", nil, 2, nil},
}
for _, test := range tests {
t.Run(test.name, func(t *testing.T) {
got := ChunkStrings(test.input, test.size)
if !reflect.DeepEqual(got, test.expected) {
t.Errorf("expected %#v got %#v", test.expected, got)
}
})
}
}
func TestChunkStringsReturnsCopies(t *testing.T) {
input := []string{"a", "b", "c", "d"}
chunks := ChunkStrings(input, 2)
chunks[0][0] = "changed"
if input[0] != "a" {
t.Errorf("expected the input to be unchanged, got %#v", input)
}
}
+23
View File
@@ -0,0 +1,23 @@
package util
// DedupeStrings returns a new slice holding each value of input once, in the
// order in which each value first appears. The input slice is not modified.
// A nil input returns nil.
func DedupeStrings(input []string) []string {
if input == nil {
return nil
}
seen := make(map[string]struct{}, len(input))
out := make([]string, 0, len(input))
for _, value := range input {
if _, ok := seen[value]; ok {
continue
}
seen[value] = struct{}{}
out = append(out, value)
}
return out
}
+40
View File
@@ -0,0 +1,40 @@
package util
import (
"reflect"
"testing"
)
func TestDedupeStrings(t *testing.T) {
tests := []struct {
name string
input []string
expected []string
}{
{"nil stays nil", nil, nil},
{"empty slice", []string{}, []string{}},
{"nothing repeated", []string{"a", "b", "c"}, []string{"a", "b", "c"}},
{"all one value", []string{"a", "a", "a"}, []string{"a"}},
{"repeats next to each other", []string{"a", "a", "b", "b"}, []string{"a", "b"}},
{"repeats far apart", []string{"a", "b", "a", "c", "b"}, []string{"a", "b", "c"}},
{"empty string is a value", []string{"", "a", ""}, []string{"", "a"}},
}
for _, test := range tests {
t.Run(test.name, func(t *testing.T) {
got := DedupeStrings(test.input)
if !reflect.DeepEqual(got, test.expected) {
t.Errorf("expected %#v got %#v", test.expected, got)
}
})
}
}
func TestDedupeStringsLeavesInputAlone(t *testing.T) {
input := []string{"b", "a", "b"}
DedupeStrings(input)
if !reflect.DeepEqual(input, []string{"b", "a", "b"}) {
t.Errorf("expected the input to be unchanged, got %#v", input)
}
}
+21
View File
@@ -0,0 +1,21 @@
package util
// TruncateString returns at most max characters from the start of s. It counts
// characters rather than bytes, so a multi-byte character is never cut in half
// and the result is always valid text. A max of zero or less returns an empty
// string, and a string already at or below the limit is returned unchanged.
func TruncateString(s string, max int) string {
if max <= 0 {
return ""
}
count := 0
for offset := range s {
if count == max {
return s[:offset]
}
count++
}
return s
}
+41
View File
@@ -0,0 +1,41 @@
package util
import "testing"
func TestTruncateString(t *testing.T) {
tests := []struct {
name string
input string
max int
expected string
}{
{"shorter than the limit", "abc", 5, "abc"},
{"exactly at the limit", "abc", 3, "abc"},
{"longer than the limit", "abcdef", 3, "abc"},
{"max of one", "abc", 1, "a"},
{"max of zero", "abc", 0, ""},
{"negative max", "abc", -1, ""},
{"empty string", "", 3, ""},
{"multi-byte characters kept whole", "héllo", 3, "hél"},
{"multi-byte characters below the limit", "héllo", 99, "héllo"},
{"characters outside the basic plane", "a😀b", 2, "a😀"},
}
for _, test := range tests {
t.Run(test.name, func(t *testing.T) {
got := TruncateString(test.input, test.max)
if got != test.expected {
t.Errorf("expected %q got %q", test.expected, got)
}
})
}
}
func TestTruncateStringCountsCharactersNotBytes(t *testing.T) {
// Each of these characters is two bytes long, so a byte-based cut at
// three would leave half a character behind.
got := TruncateString("äöü", 3)
if got != "äöü" {
t.Errorf("expected %q got %q", "äöü", got)
}
}
+12
View File
@@ -0,0 +1,12 @@
package util
import "strings"
// CollapseWhitespace replaces every run of whitespace in s with a single space
// and removes any whitespace at the start and end. Tabs, newlines and spaces
// outside ASCII such as the non-breaking space all count as whitespace, so a
// block of text comes back as one line with single spaces between the words.
// A string that is nothing but whitespace comes back empty.
func CollapseWhitespace(s string) string {
return strings.Join(strings.Fields(s), " ")
}
+36
View File
@@ -0,0 +1,36 @@
package util
import "testing"
// Written as a code point so that it is visible in the source rather than
// looking like an ordinary space.
const nonBreakingSpace = string(rune(0x00a0))
func TestCollapseWhitespace(t *testing.T) {
tests := []struct {
name string
input string
expected string
}{
{"nothing to change", "one two", "one two"},
{"leading and trailing spaces", " one two ", "one two"},
{"run in the middle", "one two", "one two"},
{"tabs and newlines", "one\t\ttwo\nthree", "one two three"},
{"windows line endings", "one\r\ntwo", "one two"},
{"non-breaking space", "one" + nonBreakingSpace + "two", "one two"},
{"mixed whitespace run", "one \t" + nonBreakingSpace + " two", "one two"},
{"only whitespace", " \t\n ", ""},
{"empty string", "", ""},
{"single word with padding", "\tword\n", "word"},
{"paragraph becomes one line", "first line\n\nsecond line\n", "first line second line"},
}
for _, test := range tests {
t.Run(test.name, func(t *testing.T) {
got := CollapseWhitespace(test.input)
if got != test.expected {
t.Errorf("expected %q got %q", test.expected, got)
}
})
}
}