chore: move runeutil and memoization into internal

This commit is contained in:
Christian Rocha
2024-10-24 15:21:20 -04:00
parent e2033f3460
commit 4e05c24fc4
6 changed files with 7 additions and 5 deletions
+125
View File
@@ -0,0 +1,125 @@
// Package memoization implement a simple memoization cache. It's designed to
// improve performance in textarea.
package memoization
import (
"container/list"
"crypto/sha256"
"fmt"
"sync"
)
// Hasher is an interface that requires a Hash method. The Hash method is
// expected to return a string representation of the hash of the object.
type Hasher interface {
Hash() string
}
// entry is a struct that holds a key-value pair. It is used as an element
// in the evictionList of the MemoCache.
type entry[T any] struct {
key string
value T
}
// MemoCache is a struct that represents a cache with a set capacity. It
// uses an LRU (Least Recently Used) eviction policy. It is safe for
// concurrent use.
type MemoCache[H Hasher, T any] struct {
capacity int
mutex sync.Mutex
cache map[string]*list.Element // The cache holding the results
evictionList *list.List // A list to keep track of the order for LRU
hashableItems map[string]T // This map keeps track of the original hashable items (optional)
}
// NewMemoCache is a function that creates a new MemoCache with a given
// capacity. It returns a pointer to the created MemoCache.
func NewMemoCache[H Hasher, T any](capacity int) *MemoCache[H, T] {
return &MemoCache[H, T]{
capacity: capacity,
cache: make(map[string]*list.Element),
evictionList: list.New(),
hashableItems: make(map[string]T),
}
}
// Capacity is a method that returns the capacity of the MemoCache.
func (m *MemoCache[H, T]) Capacity() int {
return m.capacity
}
// Size is a method that returns the current size of the MemoCache. It is
// the number of items currently stored in the cache.
func (m *MemoCache[H, T]) Size() int {
m.mutex.Lock()
defer m.mutex.Unlock()
return m.evictionList.Len()
}
// Get is a method that returns the value associated with the given
// hashable item in the MemoCache. If there is no corresponding value, the
// method returns nil.
func (m *MemoCache[H, T]) Get(h H) (T, bool) {
m.mutex.Lock()
defer m.mutex.Unlock()
hashedKey := h.Hash()
if element, found := m.cache[hashedKey]; found {
m.evictionList.MoveToFront(element)
return element.Value.(*entry[T]).value, true
}
var result T
return result, false
}
// Set is a method that sets the value for the given hashable item in the
// MemoCache. If the cache is at capacity, it evicts the least recently
// used item before adding the new item.
func (m *MemoCache[H, T]) Set(h H, value T) {
m.mutex.Lock()
defer m.mutex.Unlock()
hashedKey := h.Hash()
if element, found := m.cache[hashedKey]; found {
m.evictionList.MoveToFront(element)
element.Value.(*entry[T]).value = value
return
}
// Check if the cache is at capacity
if m.evictionList.Len() >= m.capacity {
// Evict the least recently used item from the cache
toEvict := m.evictionList.Back()
if toEvict != nil {
evictedEntry := m.evictionList.Remove(toEvict).(*entry[T])
delete(m.cache, evictedEntry.key)
delete(m.hashableItems, evictedEntry.key) // if you're keeping track of original items
}
}
// Add the value to the cache and the evictionList
newEntry := &entry[T]{
key: hashedKey,
value: value,
}
element := m.evictionList.PushFront(newEntry)
m.cache[hashedKey] = element
m.hashableItems[hashedKey] = value // if you're keeping track of original items
}
// HString is a type that implements the Hasher interface for strings.
type HString string
// Hash is a method that returns the hash of the string.
func (h HString) Hash() string {
return fmt.Sprintf("%x", sha256.Sum256([]byte(h)))
}
// HInt is a type that implements the Hasher interface for integers.
type HInt int
// Hash is a method that returns the hash of the integer.
func (h HInt) Hash() string {
return fmt.Sprintf("%x", sha256.Sum256([]byte(fmt.Sprintf("%d", h))))
}
+241
View File
@@ -0,0 +1,241 @@
package memoization
import (
"encoding/binary"
"fmt"
"os"
"testing"
)
type actionType int
const (
set actionType = iota
get
)
type cacheAction struct {
actionType actionType
key HString
value interface{}
expectedValue interface{}
}
type testCase struct {
name string
capacity int
actions []cacheAction
}
func TestCache(t *testing.T) {
tests := []testCase{
{
name: "TestNewMemoCache",
capacity: 5,
actions: []cacheAction{
{actionType: get, expectedValue: nil},
},
},
{
name: "TestSetAndGet",
capacity: 10,
actions: []cacheAction{
{actionType: set, key: "key1", value: "value1"},
{actionType: get, key: "key1", expectedValue: "value1"},
{actionType: set, key: "key1", value: "newValue1"},
{actionType: get, key: "key1", expectedValue: "newValue1"},
{actionType: get, key: "nonExistentKey", expectedValue: nil},
{actionType: set, key: "nilKey", value: ""},
{actionType: get, key: "nilKey", expectedValue: ""},
{actionType: set, key: "keyA", value: "valueA"},
{actionType: set, key: "keyB", value: "valueB"},
{actionType: get, key: "keyA", expectedValue: "valueA"},
{actionType: get, key: "keyB", expectedValue: "valueB"},
},
},
{
name: "TestSetNilValue",
capacity: 10,
actions: []cacheAction{
{actionType: set, key: HString("nilKey"), value: nil},
{actionType: get, key: HString("nilKey"), expectedValue: nil},
},
},
{
name: "TestGetAfterEviction",
capacity: 2,
actions: []cacheAction{
{actionType: set, key: HString("1"), value: 1},
{actionType: set, key: HString("2"), value: 2},
{actionType: set, key: HString("3"), value: 3},
{actionType: get, key: HString("1"), expectedValue: nil},
{actionType: get, key: HString("2"), expectedValue: 2},
},
},
{
name: "TestGetAfterLRU",
capacity: 2,
actions: []cacheAction{
{actionType: set, key: HString("1"), value: 1},
{actionType: set, key: HString("2"), value: 2},
{actionType: get, key: HString("1"), expectedValue: 1},
{actionType: set, key: HString("3"), value: 3},
{actionType: get, key: HString("1"), expectedValue: 1},
{actionType: get, key: HString("3"), expectedValue: 3},
{actionType: get, key: HString("2"), expectedValue: nil},
},
},
{
name: "TestLRU_Capacity3",
capacity: 3,
actions: []cacheAction{
{actionType: set, key: HString("1"), value: 1},
{actionType: set, key: HString("2"), value: 2},
{actionType: set, key: HString("3"), value: 3},
{actionType: get, key: HString("1"), expectedValue: 1}, // Accessing key "1"
{actionType: set, key: HString("4"), value: 4}, // Should evict key "2" since "1" was recently accessed
{actionType: get, key: HString("2"), expectedValue: nil},
{actionType: get, key: HString("1"), expectedValue: 1},
{actionType: get, key: HString("3"), expectedValue: 3},
{actionType: get, key: HString("4"), expectedValue: 4},
},
},
// Test LRU behavior with varying accesses
{
name: "TestLRU_VaryingAccesses",
capacity: 3,
actions: []cacheAction{
{actionType: set, key: HString("1"), value: 1},
{actionType: set, key: HString("2"), value: 2},
{actionType: set, key: HString("3"), value: 3},
{actionType: get, key: HString("1"), expectedValue: 1}, // Accessing key "1"
{actionType: get, key: HString("2"), expectedValue: 2}, // Accessing key "2"
{actionType: set, key: HString("4"), value: 4}, // Should evict key "3"
{actionType: get, key: HString("3"), expectedValue: nil},
{actionType: get, key: HString("1"), expectedValue: 1},
{actionType: get, key: HString("2"), expectedValue: 2},
{actionType: get, key: HString("4"), expectedValue: 4},
},
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
cache := NewMemoCache[HString, interface{}](tt.capacity)
for _, action := range tt.actions {
switch action.actionType {
case set:
cache.Set(action.key, action.value)
case get:
if got, _ := cache.Get(action.key); got != action.expectedValue {
t.Errorf("Get() = %v, want %v", got, action.expectedValue)
}
}
}
})
}
}
func FuzzCache(f *testing.F) {
// Define some seed values for initial scenarios
for _, seed := range [][]byte{
[]byte("7\x010\x0000000020"),
{0, 0, 0, 0}, // Set key 0 to 0
{1, 0, 0, 1}, // Set key 0 to 1
{2, 0}, // Get key 0
} {
f.Add(seed)
}
f.Fuzz(func(t *testing.T, in []byte) {
if len(in) < 1 {
t.Skip() // Skip the test if the input is less than 1 byte
}
cache := NewMemoCache[HInt, int](10) // Initialize a cache with the initial size
expectedValues := make(map[HInt]int) // Map to store expected key-value pairs
accessOrder := make([]HInt, 0) // Slice to store the order of keys accessed
for i := 0; i < len(in); {
opCode := in[i] % 4 // Determine the operation: Set, Get, or Reset (added case for Reset)
i++
switch opCode {
case 0, 1: // Set operation
if i+3 > len(in) {
t.Skip() // Not enough input to continue, so skip
}
key := HInt(binary.BigEndian.Uint16(in[i : i+2]))
value := int(in[i+2])
i += 3
// If the key is already in accessOrder, we remove it and append it again later
for index, accessedKey := range accessOrder {
if accessedKey == key {
accessOrder = append(accessOrder[:index], accessOrder[index+1:]...)
break
}
}
cache.Set(key, value) // Set the value in the cache
expectedValues[key] = value
accessOrder = append(accessOrder, key) // Add the key to the access order slice
// If we exceeded the cache size, we need to evict the least recently used item
if len(accessOrder) > cache.Capacity() {
evictedKey := accessOrder[0]
accessOrder = accessOrder[1:]
delete(expectedValues, evictedKey) // Remove the evicted key from expected values
}
case 2: // Get operation
if i >= len(in) {
t.Skip() // Not enough input to continue, so skip
}
key := HInt(in[i])
i++
expectedValue, ok := expectedValues[key]
if !ok {
// If the key is not found, it means it was either evicted or never added
expectedValue = 0 // The zero value, depends on your cache implementation
} else {
// If the key was accessed, move it to the end of the accessOrder to represent recent use
for index, accessedKey := range accessOrder {
if accessedKey == key {
accessOrder = append(accessOrder[:index], accessOrder[index+1:]...)
accessOrder = append(accessOrder, key)
break
}
}
}
if got, _ := cache.Get(key); got != expectedValue {
fmt.Fprintf(os.Stderr, "cache: capacity: %d, hashable: %v, cache: %v\n", cache.capacity, cache.hashableItems, cache.cache)
t.Fatalf("Get(%v) = %v, want %v", key, got, expectedValue) // The values do not match
}
case 3: // Reset operation
if i >= len(in) {
t.Skip() // Not enough input to continue, so skip
}
newCacheSize := int(in[i]) // Read the new cache size from the input
i++
if newCacheSize == 0 {
t.Skip() // If the size is zero, we skip this test
}
// Create a new cache with the specified size
cache = NewMemoCache[HInt, int](newCacheSize)
// clear and reinitialize the expected values
expectedValues = make(map[HInt]int)
accessOrder = make([]HInt, 0)
}
}
})
}
+102
View File
@@ -0,0 +1,102 @@
// Package runeutil provides utility functions for tidying up incoming runes
// from Key messages.
package runeutil
import (
"unicode"
"unicode/utf8"
)
// Sanitizer is a helper for bubble widgets that want to process
// Runes from input key messages.
type Sanitizer interface {
// Sanitize removes control characters from runes in a KeyRunes
// message, and optionally replaces newline/carriage return/tabs by a
// specified character.
//
// The rune array is modified in-place if possible. In that case, the
// returned slice is the original slice shortened after the control
// characters have been removed/translated.
Sanitize(runes []rune) []rune
}
// NewSanitizer constructs a rune sanitizer.
func NewSanitizer(opts ...Option) Sanitizer {
s := sanitizer{
replaceNewLine: []rune("\n"),
replaceTab: []rune(" "),
}
for _, o := range opts {
s = o(s)
}
return &s
}
// Option is the type of option that can be passed to Sanitize().
type Option func(sanitizer) sanitizer
// ReplaceTabs replaces tabs by the specified string.
func ReplaceTabs(tabRepl string) Option {
return func(s sanitizer) sanitizer {
s.replaceTab = []rune(tabRepl)
return s
}
}
// ReplaceNewlines replaces newline characters by the specified string.
func ReplaceNewlines(nlRepl string) Option {
return func(s sanitizer) sanitizer {
s.replaceNewLine = []rune(nlRepl)
return s
}
}
func (s *sanitizer) Sanitize(runes []rune) []rune {
// dstrunes are where we are storing the result.
dstrunes := runes[:0:len(runes)]
// copied indicates whether dstrunes is an alias of runes
// or a copy. We need a copy when dst moves past src.
// We use this as an optimization to avoid allocating
// a new rune slice in the common case where the output
// is smaller or equal to the input.
copied := false
for src := 0; src < len(runes); src++ {
r := runes[src]
switch {
case r == utf8.RuneError:
// skip
case r == '\r' || r == '\n':
if len(dstrunes)+len(s.replaceNewLine) > src && !copied {
dst := len(dstrunes)
dstrunes = make([]rune, dst, len(runes)+len(s.replaceNewLine))
copy(dstrunes, runes[:dst])
copied = true
}
dstrunes = append(dstrunes, s.replaceNewLine...)
case r == '\t':
if len(dstrunes)+len(s.replaceTab) > src && !copied {
dst := len(dstrunes)
dstrunes = make([]rune, dst, len(runes)+len(s.replaceTab))
copy(dstrunes, runes[:dst])
copied = true
}
dstrunes = append(dstrunes, s.replaceTab...)
case unicode.IsControl(r):
// Other control characters: skip.
default:
// Keep the character.
dstrunes = append(dstrunes, runes[src])
}
}
return dstrunes
}
type sanitizer struct {
replaceNewLine []rune
replaceTab []rune
}
+44
View File
@@ -0,0 +1,44 @@
package runeutil
import (
"testing"
"unicode/utf8"
)
func TestSanitize(t *testing.T) {
td := []struct {
input, output string
}{
{"", ""},
{"x", "x"},
{"\n", "XX"},
{"\na\n", "XXaXX"},
{"\n\n", "XXXX"},
{"\t", ""},
{"hello", "hello"},
{"hel\nlo", "helXXlo"},
{"hel\rlo", "helXXlo"},
{"hel\tlo", "hello"},
{"he\n\nl\tlo", "heXXXXllo"},
{"he\tl\n\nlo", "helXXXXlo"},
{"hel\x1blo", "hello"},
{"hello\xc2", "hello"}, // invalid utf8
}
for _, tc := range td {
runes := make([]rune, 0, len(tc.input))
b := []byte(tc.input)
for i, w := 0, 0; i < len(b); i += w {
var r rune
r, w = utf8.DecodeRune(b[i:])
runes = append(runes, r)
}
t.Logf("input runes: %+v", runes)
s := NewSanitizer(ReplaceNewlines("XX"), ReplaceTabs(""))
result := s.Sanitize(runes)
rs := string(result)
if tc.output != rs {
t.Errorf("%q: expected %q, got %q (%+v)", tc.input, tc.output, rs, result)
}
}
}