ci / go (push) Waiting to run
ci / go-db (agent) (push) Waiting to run
ci / go-db (config) (push) Waiting to run
ci / go-db (db) (push) Waiting to run
ci / go-db (evidence) (push) Waiting to run
ci / go-db (llmrec) (push) Waiting to run
ci / go-db (server) (push) Waiting to run
web / web (push) Waiting to run
docs / links (push) Canceled after 0s
detections / detections (push) Canceled after 0s
258 lines
8.7 KiB
Go
258 lines
8.7 KiB
Go
package traffic
|
||
|
||
import (
|
||
"database/sql"
|
||
"fmt"
|
||
"os"
|
||
"path/filepath"
|
||
"strings"
|
||
"testing"
|
||
)
|
||
|
||
// bulkRecord fills the index with inline bodies — the ones that actually make
|
||
// index.sqlite grow. Binary content type keeps them out of the full-text index so
|
||
// the test stays fast; the FTS side is covered by TestReclaimMergesFTSTombstones.
|
||
func bulkRecord(tr *Traffic, host string, n, size int) {
|
||
body := []byte(strings.Repeat("A", size))
|
||
for i := 0; i < n; i++ {
|
||
tr.record(newFlow(host, "GET", fmt.Sprintf("/blob/%d", i), nil, body,
|
||
withRespType("application/octet-stream")))
|
||
}
|
||
}
|
||
|
||
func TestNewIndexEnablesIncrementalVacuum(t *testing.T) {
|
||
tr, _ := openTraffic(t)
|
||
if !tr.incrementalVacuum {
|
||
t.Fatal("新建索引库未启用增量回收")
|
||
}
|
||
var mode int
|
||
if err := tr.DB().QueryRow(`PRAGMA auto_vacuum`).Scan(&mode); err != nil {
|
||
t.Fatal(err)
|
||
}
|
||
if mode != autoVacuumIncremental {
|
||
t.Fatalf("auto_vacuum=%d,应为 %d", mode, autoVacuumIncremental)
|
||
}
|
||
}
|
||
|
||
// TestDeleteReclaimsIndexSpace is the regression: deleting traffic used to leave
|
||
// index.sqlite at its high-water mark forever, because SQLite only chains freed
|
||
// pages onto its freelist and nothing ever returned them to the filesystem.
|
||
func TestDeleteReclaimsIndexSpace(t *testing.T) {
|
||
tr, _ := openTraffic(t)
|
||
const host = "bulk.example.com"
|
||
// 30 × 200KB stays under maxInlineBody, so every body lands in the database
|
||
// itself rather than the blob store — that is where the growth was invisible.
|
||
bulkRecord(tr, host, 30, 200*1024)
|
||
grown := tr.indexBytes()
|
||
if grown < 5<<20 {
|
||
t.Fatalf("索引只有 %d 字节,样本不足以验证回收", grown)
|
||
}
|
||
|
||
if n, err := tr.DeleteHostsExact([]string{host}); err != nil || n != 30 {
|
||
t.Fatalf("DeleteHostsExact=(%d,%v),应为 (30,nil)", n, err)
|
||
}
|
||
tr.reaping.Wait() // 回收在后台分块进行
|
||
|
||
after := tr.indexBytes()
|
||
if after > grown/4 {
|
||
t.Fatalf("删除后索引仍占 %d 字节(删除前 %d),空间没有还给文件系统", after, grown)
|
||
}
|
||
// A handful of pages incremental_vacuum could not move to the end of the file
|
||
// is a normal residual; the ~1500 that the deletion freed must be gone.
|
||
var free int
|
||
if err := tr.DB().QueryRow(`PRAGMA freelist_count`).Scan(&free); err != nil {
|
||
t.Fatal(err)
|
||
}
|
||
if free > 64 {
|
||
t.Fatalf("仍有 %d 个空闲页未回收", free)
|
||
}
|
||
}
|
||
|
||
// TestReclaimMergesFTSTombstones covers the second half of the leak: ex_fts is a
|
||
// contentless_delete index, so a DELETE only writes tombstones. Without a merge
|
||
// the index keeps growing on every deletion — deleting traffic made it bigger.
|
||
func TestReclaimMergesFTSTombstones(t *testing.T) {
|
||
tr, _ := openTraffic(t)
|
||
if !tr.fts {
|
||
t.Skip("驱动未启用 FTS5")
|
||
}
|
||
// Deleted in batches, which is what leaves tombstones spread over many
|
||
// segments rather than emptying the index in one shot.
|
||
for round := 0; round < 4; round++ {
|
||
host := fmt.Sprintf("fts%d.example.com", round)
|
||
for i := 0; i < 20; i++ {
|
||
tr.record(newFlow(host, "GET", fmt.Sprintf("/p/%d", i), nil,
|
||
[]byte(strings.Repeat("secret token 中文正文 padding ", 200))))
|
||
}
|
||
if _, err := tr.DeleteHostsExact([]string{host}); err != nil {
|
||
t.Fatal(err)
|
||
}
|
||
tr.reaping.Wait()
|
||
}
|
||
|
||
var exchanges, segments int
|
||
if err := tr.DB().QueryRow(`SELECT COUNT(*) FROM exchanges`).Scan(&exchanges); err != nil {
|
||
t.Fatal(err)
|
||
}
|
||
if err := tr.DB().QueryRow(`SELECT COUNT(*) FROM ex_fts_data`).Scan(&segments); err != nil {
|
||
t.Fatal(err)
|
||
}
|
||
if exchanges != 0 {
|
||
t.Fatalf("还剩 %d 条流量", exchanges)
|
||
}
|
||
// A fully merged, empty contentless index keeps only its structure rows.
|
||
if segments > 8 {
|
||
t.Fatalf("全文索引残留 %d 行段数据,tombstone 未被合并回收", segments)
|
||
}
|
||
}
|
||
|
||
// TestReclaimOnLegacyIndexIsHarmless covers installs created before
|
||
// auto_vacuum=incremental became the default: incremental_vacuum is a silent
|
||
// no-op there, so reclamation must report the situation and finish rather than
|
||
// spin or fail. Only a full compaction can convert such a file.
|
||
func TestReclaimOnLegacyIndexIsHarmless(t *testing.T) {
|
||
dir := t.TempDir()
|
||
if err := os.MkdirAll(filepath.Join(dir, "_index"), 0o755); err != nil {
|
||
t.Fatal(err)
|
||
}
|
||
// Create the tables first, with auto_vacuum left at its default 0 — exactly the
|
||
// shape Open used to leave behind.
|
||
legacy, err := sql.Open("sqlite", filepath.Join(dir, "_index", "index.sqlite"))
|
||
if err != nil {
|
||
t.Fatal(err)
|
||
}
|
||
if _, err := legacy.Exec(indexSchema); err != nil {
|
||
t.Fatal(err)
|
||
}
|
||
if err := legacy.Close(); err != nil {
|
||
t.Fatal(err)
|
||
}
|
||
|
||
tr, err := Open(dir, "127.0.0.1:0")
|
||
if err != nil {
|
||
t.Fatal(err)
|
||
}
|
||
t.Cleanup(func() { tr.Close() })
|
||
if tr.incrementalVacuum {
|
||
t.Fatal("旧库不应报告已启用增量回收")
|
||
}
|
||
|
||
const host = "legacy.example.com"
|
||
bulkRecord(tr, host, 8, 200*1024)
|
||
if n, err := tr.DeleteHostsExact([]string{host}); err != nil || n != 8 {
|
||
t.Fatalf("DeleteHostsExact=(%d,%v),应为 (8,nil)", n, err)
|
||
}
|
||
tr.reaping.Wait() // 必须收敛,不能卡在预算里
|
||
|
||
// The freelist stays populated: that is the whole reason a compaction entry
|
||
// point is needed for pre-existing databases.
|
||
var free int
|
||
if err := tr.DB().QueryRow(`PRAGMA freelist_count`).Scan(&free); err != nil {
|
||
t.Fatal(err)
|
||
}
|
||
if free == 0 {
|
||
t.Fatal("旧库居然回收了空闲页,说明测试没有真的构造出旧库")
|
||
}
|
||
}
|
||
|
||
// TestDeleteAllPurgesAndCompacts covers the page's clear-everything action: it
|
||
// must leave nothing behind — including host directories the index no longer
|
||
// knows about — and it must hand the index space back, since an emptied index is
|
||
// the one moment a full rewrite is cheap.
|
||
func TestDeleteAllPurgesAndCompacts(t *testing.T) {
|
||
tr, dir := openTraffic(t)
|
||
bulkRecord(tr, "a.example.com", 10, 200*1024)
|
||
bulkRecord(tr, "b.example.com", 10, 200*1024)
|
||
// A text body so the full-text index has real content, and a spilled one so a
|
||
// blob exists to collect.
|
||
tr.record(newFlow("c.example.com", "GET", "/page", nil, []byte(strings.Repeat("secret-token ", 500))))
|
||
tr.record(newFlow("c.example.com", "GET", "/big", nil,
|
||
[]byte(strings.Repeat("B", maxInlineBody+1024)), withRespType("application/sql")))
|
||
// An orphaned legacy directory: no index row points at it, so only a
|
||
// clear-everything should take it.
|
||
orphan := filepath.Join(dir, "orphan.example.com")
|
||
if err := os.MkdirAll(orphan, 0o755); err != nil {
|
||
t.Fatal(err)
|
||
}
|
||
grown := tr.indexBytes()
|
||
if grown < 5<<20 {
|
||
t.Fatalf("索引只有 %d 字节,样本不足", grown)
|
||
}
|
||
|
||
deleted, reclaimed, err := tr.DeleteAll()
|
||
if err != nil {
|
||
t.Fatalf("DeleteAll: %v", err)
|
||
}
|
||
if deleted != 22 {
|
||
t.Fatalf("deleted=%d,应为 22", deleted)
|
||
}
|
||
tr.reaping.Wait()
|
||
|
||
if reclaimed < grown/2 {
|
||
t.Fatalf("只回收了 %d 字节(删除前索引 %d)", reclaimed, grown)
|
||
}
|
||
if after := tr.indexBytes(); after > grown/8 {
|
||
t.Fatalf("清空后索引仍占 %d 字节(删除前 %d)", after, grown)
|
||
}
|
||
for _, q := range []string{
|
||
`SELECT COUNT(*) FROM exchanges`,
|
||
`SELECT COUNT(*) FROM exchange_bodies`,
|
||
`SELECT COUNT(*) FROM blob_refs`,
|
||
} {
|
||
var c int
|
||
if err := tr.DB().QueryRow(q).Scan(&c); err != nil {
|
||
t.Fatal(err)
|
||
}
|
||
if c != 0 {
|
||
t.Fatalf("%s = %d,应为 0", q, c)
|
||
}
|
||
}
|
||
if _, err := os.Stat(orphan); !os.IsNotExist(err) {
|
||
t.Fatalf("孤立的历史 host 目录未被清理:%v", err)
|
||
}
|
||
// Recording must keep working against the freshly rewritten file.
|
||
tr.record(newFlow("d.example.com", "GET", "/after", nil, []byte("清空后仍可录制")))
|
||
if n, err := tr.Count(); err != nil || n != 1 {
|
||
t.Fatalf("清空后 Count=(%d,%v),应为 (1,nil)", n, err)
|
||
}
|
||
}
|
||
|
||
// TestDeleteAllConvertsLegacyIndex is why the purge compacts rather than just
|
||
// deleting: auto_vacuum cannot be switched on after the fact except through a
|
||
// VACUUM, and an emptied index is the cheapest place to pay for one. After this,
|
||
// ordinary deletions reclaim space on their own.
|
||
func TestDeleteAllConvertsLegacyIndex(t *testing.T) {
|
||
dir := t.TempDir()
|
||
old := openLegacyIndex(t, dir)
|
||
if err := old.Close(); err != nil {
|
||
t.Fatal(err)
|
||
}
|
||
tr, err := Open(dir, "127.0.0.1:0")
|
||
if err != nil {
|
||
t.Fatal(err)
|
||
}
|
||
t.Cleanup(func() { tr.Close() })
|
||
if tr.incrementalVacuum {
|
||
t.Fatal("旧库不应报告已启用增量回收")
|
||
}
|
||
|
||
bulkRecord(tr, "legacy.example.com", 10, 200*1024)
|
||
if _, _, err := tr.DeleteAll(); err != nil {
|
||
t.Fatalf("DeleteAll: %v", err)
|
||
}
|
||
if !tr.incrementalVacuum {
|
||
t.Fatal("清空后旧库未被转换为增量回收模式")
|
||
}
|
||
|
||
// The converted database now reclaims on an ordinary host deletion.
|
||
bulkRecord(tr, "again.example.com", 10, 200*1024)
|
||
grown := tr.indexBytes()
|
||
if _, err := tr.DeleteHostsExact([]string{"again.example.com"}); err != nil {
|
||
t.Fatal(err)
|
||
}
|
||
tr.reaping.Wait()
|
||
if after := tr.indexBytes(); after > grown/4 {
|
||
t.Fatalf("转换后普通删除仍未回收:%d 字节(删除前 %d)", after, grown)
|
||
}
|
||
}
|