Skip to content

字符串与 bytes

本章讲解Go字符串处理:字符串不可变性、拼接方法、bytes/rune操作、字符编码。重点:Go字符串是UTF-8字节序列,理解rune(码点)和字节的区别,掌握高效拼接。

前置知识:第3章 语法基础 学习目标:掌握字符串操作、理解UTF-8编码、会处理中文字符


Go的字符串是只读的字节序列:

s := "hello"
// s[0] = 'H' // 编译错误:无法修改字符串
// 必须创建新字符串
s2 := "H" + s[1:] // "Hello"

Python字符串同样不可变:

s = "hello"
s[0] = "H" # TypeError
s = "H" + s[1:] # 正确,创建新字符串
// 字符串结构(伪代码)
type StringHeader struct {
Data unsafe.Pointer // 指向字节数组
Len int // 长度(字节数)
}

字符串底层是字节数组,UTF-8编码。

str := "hello"
// 字符串转字节切片
bytes := []byte(str)
// 字节切片转字符串
str2 := string(bytes)
// 注意:每次转换都分配新内存

rune是int32的别名,表示Unicode码点:

var r rune = '中' // '中'的码点是U+4E2D
fmt.Printf("%c\n", r) // 中
fmt.Printf("%U\n", r) // U+4E2D
// 遍历字符串获取rune
str := "Hello世界"
for i, r := range str {
fmt.Printf("%d: %c (U+%04X)\n", i, r, r)
}

本节目的:掌握Go字符串拼接的多种方式及性能差异

s1 := "hello"
s2 := "world"
s3 := s1 + " " + s2 // "hello world"

简单场景适用,复杂场景效率低。

name := "Alice"
age := 30
s := fmt.Sprintf("Name: %s, Age: %d", name, age)
// "Name: Alice, Age: 30"

用于格式化复杂场景。

import "strings"
parts := []string{"hello", "world", "go"}
s := strings.Join(parts, " ") // "hello world go"
// 常用场景:路径拼接
paths := []string{"dir1", "dir2", "file.txt"}
s := strings.Join(paths, "/") // "dir1/dir2/file.txt"
import "strings"
var builder strings.Builder
builder.WriteString("hello")
builder.WriteString(" ")
builder.WriteString("world")
result := builder.String() // "hello world"
// 写入格式化内容
fmt.Fprintf(&builder, "Name: %s, Age: %d", name, age)
// 效率低:不推荐循环内使用
s := ""
for _, item := range items {
s += item // 每次都创建新字符串
}
// 效率高:推荐
var parts []string
for _, item := range items {
parts = append(parts, item)
}
s := strings.Join(parts, "")
// 或使用Builder
var builder strings.Builder
for _, item := range items {
builder.WriteString(item)
}
s := builder.String()
# 加号
s = "hello" + " " + "world"
# join方法
parts = ["hello", "world", "go"]
s = " ".join(parts) # "hello world go"
# f-string
s = f"Name: {name}, Age: {age}"

本节目的:掌握bytes包操作字节切片、理解rune(Unicode码点)的处理

处理字节切片的工具库:

import "bytes"
// Buffer:动态增长的字节切片
var buf bytes.Buffer
buf.Write([]byte("hello"))
buf.WriteByte(',')
buf.WriteString(" world")
data := buf.Bytes() // []byte
str := buf.String() // string
// Reader:读取字节切片
reader := bytes.NewReader([]byte("hello world"))
b, _ := reader.ReadByte() // 'h'
// 常用函数
bytes.Split([]byte("a,b,c"), []byte(",")) // [][]byte{a, b, c}
bytes.Contains([]byte("hello"), []byte("ll")) // true
bytes.Count([]byte("hello"), []byte("l")) // 2
bytes.TrimSpace([]byte(" hello ")) // []byte("hello")
import "strings"
// 大小写
strings.ToLower("HELLO") // "hello"
strings.ToUpper("hello") // "HELLO"
strings.Title("hello world") // "Hello World"
// 查找
strings.Index("hello", "ll") // 2
strings.Contains("hello", "ll") // true
strings.HasPrefix("hello", "he") // true
strings.HasSuffix("hello", "lo") // true
// 分割
fields := strings.Fields(" hello world ") // ["hello", "world"]
parts := strings.Split("a,b,c", ",") // ["a", "b", "c"]
parts = strings.SplitN("a,b,c", ",", 2) // ["a", "b,c"]
// 替换
strings.Replace("hello world", "world", "go", 1) // "hello go"
strings.ReplaceAll("aaa", "a", "b") // "bbb"
// 修整
strings.Trim(" hello ", " ") // "hello"
strings.TrimLeft(" hello", " ") // "hello"
strings.TrimRight("hello ", " ") // "hello"
strings.TrimPrefix("hello", "he") // "llo"
strings.TrimSuffix("hello", "lo") // "hel"
str := "Hello世界"
// 统计字符数(rune数)
fmt.Println(len([]rune(str))) // 8
// 提取子串(按字符)
func substringByRune(s string, start, end int) string {
runes := []rune(s)
return string(runes[start:end])
}
sub := substringByRune(str, 6, 8) // "世界"
// 反转字符串
func reverseString(s string) string {
runes := []rune(s)
for i, j := 0, len(runes)-1; i < j; i, j = i+1, j-1 {
runes[i], runes[j] = runes[j], runes[i]
}
return string(runes)
}

本节目的:理解UTF-8编码规则、掌握rune遍历和中文字符处理

Go字符串默认是UTF-8编码:

str := "Hello世界"
fmt.Println(len(str)) // 12(字节),5个ASCII + 3字节(世) + 3字节(界)
// UTF-8编码规则:
// ASCII: 1字节 (0-127)
// 其他: 2-4字节,以特定模式区分
// 验证UTF-8
valid := utf8.ValidString(str) // true
import (
"golang.org/x/text/encoding/simplifiedchinese"
"golang.org/x/text/transform"
)
// GBK转UTF-8
gbkData := []byte{...}
reader := transform.NewReader(bytes.NewReader(gbkData), simplifiedchinese.GBK.NewDecoder())
utf8Data, _ := io.ReadAll(reader)
import "unicode"
c := 'A'
unicode.IsLetter(c) // true
unicode.IsDigit(c) // false
unicode.IsUpper(c) // true
unicode.IsLower(c) // false
unicode.IsSpace(c) // false
unicode.IsPunct(c) // false
// 转换
unicode.ToUpper(c) // 'A'
unicode.ToLower(c) // 'a'
// 判断中文字符
func isChinese(r rune) bool {
return unicode.Han(r)
}
for _, r := range "Hello世界" {
fmt.Printf("%c: %v\n", r, isChinese(r))
}
// H: false
// e: false
// l: false
// l: false
// o: false
// 世: true
// 界: true
import "strconv"
// int转string
s := strconv.Itoa(123) // "123"
s := strconv.FormatInt(123, 10) // "123"
s := strconv.FormatInt(123, 2) // "1111011"
// string转int
i, _ := strconv.Atoi("123") // 123
i, _ := strconv.ParseInt("123", 10, 64) // 123
// 浮点数转string
s := strconv.FormatFloat(3.14, 'f', 2, 64) // "3.14"
s := fmt.Sprintf("%.2f", 3.14) // "3.14"
// string转浮点数
f, _ := strconv.ParseFloat("3.14", 64) // 3.14
import "encoding/json"
// 字符串需要转义
data := map[string]string{"key": "value\nwith\nnewlines"}
jsonBytes, _ := json.Marshal(data)
// {"key":"value\nwith\nnewlines"}
// 解析时自动还原

  1. 频繁拼接用strings.Builder:循环内拼接字符串用strings.Builder,避免+号每次创建新字符串
  2. 遍历中文字符用rune:用for i, r := range str遍历,正确处理多字节UTF-8字符
  3. 字节vs字符:len(str)是字节数,不是字符数;中文字符在UTF-8占3字节
  4. JSON字符串转义:Go的encoding/json会自动处理换行符、双引号等特殊字符
  5. 转换开销:string和[]byte互相转换有内存分配开销,避免在热路径中频繁转换

PythonGo
str不可变string不可变
s.encode('utf-8')[]byte(s)
b.decode('utf-8')string(b)
" ".join(list)strings.Join(list, " ")
f"Name: {n}"fmt.Sprintf("Name: %s", n)
str(x)strconv.Itoa(x)
int(s)strconv.Atoi(s)
Unicode原生rune表示Unicode码点

  1. 实现字符串反转函数
  2. 统计字符串中中文字符的数量
  3. 实现一个函数,将”hello_world”转换为”helloWorld”
  4. 用strings.Builder拼接1000个字符串
  5. 实现字符串截取函数,支持UTF-8字符边界