Files
artex/traffic/reclaim_test.go
T
dela 0335d572de
ci / go (push) Waiting to run
ci / go-db (agent) (push) Waiting to run
ci / go-db (config) (push) Waiting to run
ci / go-db (db) (push) Waiting to run
ci / go-db (evidence) (push) Waiting to run
ci / go-db (llmrec) (push) Waiting to run
ci / go-db (server) (push) Waiting to run
web / web (push) Waiting to run
docs / links (push) Canceled after 0s
detections / detections (push) Canceled after 0s
First Commit
2026-10-09 08:38:16 +08:00

258 lines
8.7 KiB
Go
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
package traffic
import (
"database/sql"
"fmt"
"os"
"path/filepath"
"strings"
"testing"
)
// bulkRecord fills the index with inline bodies — the ones that actually make
// index.sqlite grow. Binary content type keeps them out of the full-text index so
// the test stays fast; the FTS side is covered by TestReclaimMergesFTSTombstones.
func bulkRecord(tr *Traffic, host string, n, size int) {
body := []byte(strings.Repeat("A", size))
for i := 0; i < n; i++ {
tr.record(newFlow(host, "GET", fmt.Sprintf("/blob/%d", i), nil, body,
withRespType("application/octet-stream")))
}
}
func TestNewIndexEnablesIncrementalVacuum(t *testing.T) {
tr, _ := openTraffic(t)
if !tr.incrementalVacuum {
t.Fatal("新建索引库未启用增量回收")
}
var mode int
if err := tr.DB().QueryRow(`PRAGMA auto_vacuum`).Scan(&mode); err != nil {
t.Fatal(err)
}
if mode != autoVacuumIncremental {
t.Fatalf("auto_vacuum=%d,应为 %d", mode, autoVacuumIncremental)
}
}
// TestDeleteReclaimsIndexSpace is the regression: deleting traffic used to leave
// index.sqlite at its high-water mark forever, because SQLite only chains freed
// pages onto its freelist and nothing ever returned them to the filesystem.
func TestDeleteReclaimsIndexSpace(t *testing.T) {
tr, _ := openTraffic(t)
const host = "bulk.example.com"
// 30 × 200KB stays under maxInlineBody, so every body lands in the database
// itself rather than the blob store — that is where the growth was invisible.
bulkRecord(tr, host, 30, 200*1024)
grown := tr.indexBytes()
if grown < 5<<20 {
t.Fatalf("索引只有 %d 字节,样本不足以验证回收", grown)
}
if n, err := tr.DeleteHostsExact([]string{host}); err != nil || n != 30 {
t.Fatalf("DeleteHostsExact=(%d,%v),应为 (30,nil)", n, err)
}
tr.reaping.Wait() // 回收在后台分块进行
after := tr.indexBytes()
if after > grown/4 {
t.Fatalf("删除后索引仍占 %d 字节(删除前 %d),空间没有还给文件系统", after, grown)
}
// A handful of pages incremental_vacuum could not move to the end of the file
// is a normal residual; the ~1500 that the deletion freed must be gone.
var free int
if err := tr.DB().QueryRow(`PRAGMA freelist_count`).Scan(&free); err != nil {
t.Fatal(err)
}
if free > 64 {
t.Fatalf("仍有 %d 个空闲页未回收", free)
}
}
// TestReclaimMergesFTSTombstones covers the second half of the leak: ex_fts is a
// contentless_delete index, so a DELETE only writes tombstones. Without a merge
// the index keeps growing on every deletion — deleting traffic made it bigger.
func TestReclaimMergesFTSTombstones(t *testing.T) {
tr, _ := openTraffic(t)
if !tr.fts {
t.Skip("驱动未启用 FTS5")
}
// Deleted in batches, which is what leaves tombstones spread over many
// segments rather than emptying the index in one shot.
for round := 0; round < 4; round++ {
host := fmt.Sprintf("fts%d.example.com", round)
for i := 0; i < 20; i++ {
tr.record(newFlow(host, "GET", fmt.Sprintf("/p/%d", i), nil,
[]byte(strings.Repeat("secret token 中文正文 padding ", 200))))
}
if _, err := tr.DeleteHostsExact([]string{host}); err != nil {
t.Fatal(err)
}
tr.reaping.Wait()
}
var exchanges, segments int
if err := tr.DB().QueryRow(`SELECT COUNT(*) FROM exchanges`).Scan(&exchanges); err != nil {
t.Fatal(err)
}
if err := tr.DB().QueryRow(`SELECT COUNT(*) FROM ex_fts_data`).Scan(&segments); err != nil {
t.Fatal(err)
}
if exchanges != 0 {
t.Fatalf("还剩 %d 条流量", exchanges)
}
// A fully merged, empty contentless index keeps only its structure rows.
if segments > 8 {
t.Fatalf("全文索引残留 %d 行段数据,tombstone 未被合并回收", segments)
}
}
// TestReclaimOnLegacyIndexIsHarmless covers installs created before
// auto_vacuum=incremental became the default: incremental_vacuum is a silent
// no-op there, so reclamation must report the situation and finish rather than
// spin or fail. Only a full compaction can convert such a file.
func TestReclaimOnLegacyIndexIsHarmless(t *testing.T) {
dir := t.TempDir()
if err := os.MkdirAll(filepath.Join(dir, "_index"), 0o755); err != nil {
t.Fatal(err)
}
// Create the tables first, with auto_vacuum left at its default 0 — exactly the
// shape Open used to leave behind.
legacy, err := sql.Open("sqlite", filepath.Join(dir, "_index", "index.sqlite"))
if err != nil {
t.Fatal(err)
}
if _, err := legacy.Exec(indexSchema); err != nil {
t.Fatal(err)
}
if err := legacy.Close(); err != nil {
t.Fatal(err)
}
tr, err := Open(dir, "127.0.0.1:0")
if err != nil {
t.Fatal(err)
}
t.Cleanup(func() { tr.Close() })
if tr.incrementalVacuum {
t.Fatal("旧库不应报告已启用增量回收")
}
const host = "legacy.example.com"
bulkRecord(tr, host, 8, 200*1024)
if n, err := tr.DeleteHostsExact([]string{host}); err != nil || n != 8 {
t.Fatalf("DeleteHostsExact=(%d,%v),应为 (8,nil)", n, err)
}
tr.reaping.Wait() // 必须收敛,不能卡在预算里
// The freelist stays populated: that is the whole reason a compaction entry
// point is needed for pre-existing databases.
var free int
if err := tr.DB().QueryRow(`PRAGMA freelist_count`).Scan(&free); err != nil {
t.Fatal(err)
}
if free == 0 {
t.Fatal("旧库居然回收了空闲页,说明测试没有真的构造出旧库")
}
}
// TestDeleteAllPurgesAndCompacts covers the page's clear-everything action: it
// must leave nothing behind — including host directories the index no longer
// knows about — and it must hand the index space back, since an emptied index is
// the one moment a full rewrite is cheap.
func TestDeleteAllPurgesAndCompacts(t *testing.T) {
tr, dir := openTraffic(t)
bulkRecord(tr, "a.example.com", 10, 200*1024)
bulkRecord(tr, "b.example.com", 10, 200*1024)
// A text body so the full-text index has real content, and a spilled one so a
// blob exists to collect.
tr.record(newFlow("c.example.com", "GET", "/page", nil, []byte(strings.Repeat("secret-token ", 500))))
tr.record(newFlow("c.example.com", "GET", "/big", nil,
[]byte(strings.Repeat("B", maxInlineBody+1024)), withRespType("application/sql")))
// An orphaned legacy directory: no index row points at it, so only a
// clear-everything should take it.
orphan := filepath.Join(dir, "orphan.example.com")
if err := os.MkdirAll(orphan, 0o755); err != nil {
t.Fatal(err)
}
grown := tr.indexBytes()
if grown < 5<<20 {
t.Fatalf("索引只有 %d 字节,样本不足", grown)
}
deleted, reclaimed, err := tr.DeleteAll()
if err != nil {
t.Fatalf("DeleteAll: %v", err)
}
if deleted != 22 {
t.Fatalf("deleted=%d,应为 22", deleted)
}
tr.reaping.Wait()
if reclaimed < grown/2 {
t.Fatalf("只回收了 %d 字节(删除前索引 %d)", reclaimed, grown)
}
if after := tr.indexBytes(); after > grown/8 {
t.Fatalf("清空后索引仍占 %d 字节(删除前 %d)", after, grown)
}
for _, q := range []string{
`SELECT COUNT(*) FROM exchanges`,
`SELECT COUNT(*) FROM exchange_bodies`,
`SELECT COUNT(*) FROM blob_refs`,
} {
var c int
if err := tr.DB().QueryRow(q).Scan(&c); err != nil {
t.Fatal(err)
}
if c != 0 {
t.Fatalf("%s = %d,应为 0", q, c)
}
}
if _, err := os.Stat(orphan); !os.IsNotExist(err) {
t.Fatalf("孤立的历史 host 目录未被清理:%v", err)
}
// Recording must keep working against the freshly rewritten file.
tr.record(newFlow("d.example.com", "GET", "/after", nil, []byte("清空后仍可录制")))
if n, err := tr.Count(); err != nil || n != 1 {
t.Fatalf("清空后 Count=(%d,%v),应为 (1,nil)", n, err)
}
}
// TestDeleteAllConvertsLegacyIndex is why the purge compacts rather than just
// deleting: auto_vacuum cannot be switched on after the fact except through a
// VACUUM, and an emptied index is the cheapest place to pay for one. After this,
// ordinary deletions reclaim space on their own.
func TestDeleteAllConvertsLegacyIndex(t *testing.T) {
dir := t.TempDir()
old := openLegacyIndex(t, dir)
if err := old.Close(); err != nil {
t.Fatal(err)
}
tr, err := Open(dir, "127.0.0.1:0")
if err != nil {
t.Fatal(err)
}
t.Cleanup(func() { tr.Close() })
if tr.incrementalVacuum {
t.Fatal("旧库不应报告已启用增量回收")
}
bulkRecord(tr, "legacy.example.com", 10, 200*1024)
if _, _, err := tr.DeleteAll(); err != nil {
t.Fatalf("DeleteAll: %v", err)
}
if !tr.incrementalVacuum {
t.Fatal("清空后旧库未被转换为增量回收模式")
}
// The converted database now reclaims on an ordinary host deletion.
bulkRecord(tr, "again.example.com", 10, 200*1024)
grown := tr.indexBytes()
if _, err := tr.DeleteHostsExact([]string{"again.example.com"}); err != nil {
t.Fatal(err)
}
tr.reaping.Wait()
if after := tr.indexBytes(); after > grown/4 {
t.Fatalf("转换后普通删除仍未回收:%d 字节(删除前 %d)", after, grown)
}
}