Initial implementation: NATS->S3 archiver + search/retrieve CLI
ci/woodpecker/pr/build Pipeline was successful
ci/woodpecker/pr/pre-commit Pipeline was successful
ci/woodpecker/pr/test Pipeline was successful

logarchiver replaces the plain Vector archiver leg of the centralized
logging stack (argocd-apps #296) with a Go service that archives raw logs
from NATS JetStream to S3 as zstd-compressed, OpenPGP-encrypted, indexed
objects, plus an operator CLI to search the index and retrieve/decrypt
archived logs. It adds the things that outgrew Vector: zstd compression,
encryption keyed from Ben's Vault GPG secrets engine, a searchable
ClickHouse index, and sink-conditional acks (a batch is acknowledged to
JetStream only after the object is durably in S3 AND indexed).

Service (`logarchiver run`):
- Durable JetStream pull consumer (stream LOGS, durable archiver, subject
  filter default logs.k8s.vault.>), explicit acks, independent offsets.
- Batch per subject by size/count/time -> NDJSON -> zstd -> encrypt -> S3
  PUT -> ClickHouse index row -> ack. On any failure the batch is Nak'd and
  redelivered, so nothing is lost on a sink outage.
- Encryption is a wrapped-DEK envelope (container LARC1): the bulk is
  AES-256-GCM framed under a random data key, and only that 32-byte key is
  OpenPGP-encrypted to the engine's public key. This is because the Vault
  GPG engine does whole-payload decrypt only; retrieval round-trips just the
  tiny wrapped key regardless of object size. Public key fetched from the
  engine or a mounted file (configurable); key fingerprint recorded per
  object; periodic pubkey refresh for rotation.
- Prometheus metrics, structured slog, graceful drain on shutdown.

CLI:
- `search` queries the index (subject/host/time) and lists matching objects.
- `fetch` downloads, decrypts via the Vault GPG engine, unzstds and emits
  NDJSON (optionally re-filtered by host/time).
- `init-schema` creates/prints the ClickHouse archive_index DDL.
- cobra `completion` subcommands.

Config via file+env (k8s-friendly, secrets from env), boundaries (NATS/S3/
ClickHouse/Vault) behind interfaces with unit tests (config, batching,
host/subject extraction, crypto roundtrip with a test key, ack-after-persist
with fakes, search query building). go build/vet/test -race clean;
golangci-lint v2 clean. Woodpecker CI: build/test/pre-commit on PR; on v*
tag a container image plus a Gitea binary release + rpm-internal RPM. Docs
per subcommand + architecture + retrieval runbook + deployment drop-in.

Claude-Session: https://claude.ai/code/session_015ur3i7D2azsMAWTSVABApv
This commit is contained in:
benvin
2026-07-27 22:11:54 +10:00
committed by Ben Vincent
parent d036d31f12
commit c05ccfcb5d
58 changed files with 5946 additions and 1 deletions
+42
View File
@@ -0,0 +1,42 @@
package index
import "fmt"
// CreateDatabaseSQL creates the index database if absent.
func CreateDatabaseSQL(database string) string {
return fmt.Sprintf("CREATE DATABASE IF NOT EXISTS %s", database)
}
// CreateTableSQL returns the DDL for the archive index table. One row is written
// per archived S3 object. In-cluster the argocd bootstrap Job owns table
// creation (like the logging stack's clickhouse-schema PostSync hook); this DDL
// is also shipped as schema/archive_index.sql and applied by `logarchiver
// init-schema`.
//
// PARTITION BY month of min_ts keeps partitions coarse (few objects/day).
// ORDER BY (subject, min_ts) matches the primary search axes. A bloom_filter
// skip index on hosts accelerates host lookups without a per-host column.
func CreateTableSQL(database, table string) string {
return fmt.Sprintf(`CREATE TABLE IF NOT EXISTS %s.%s
(
object_key String,
bucket LowCardinality(String),
subject LowCardinality(String),
hosts Array(LowCardinality(String)),
min_ts DateTime64(3),
max_ts DateTime64(3),
event_count UInt64,
raw_bytes UInt64,
stored_bytes UInt64,
compression LowCardinality(String),
cipher LowCardinality(String),
container_format LowCardinality(String),
key_name LowCardinality(String),
key_fingerprint String,
created_at DateTime64(3) DEFAULT now64(3),
INDEX idx_hosts hosts TYPE bloom_filter GRANULARITY 1
)
ENGINE = MergeTree
PARTITION BY toYYYYMM(min_ts)
ORDER BY (subject, min_ts, object_key)`, database, table)
}
+159
View File
@@ -0,0 +1,159 @@
// Package index writes and queries the ClickHouse archive index — one row per
// stored S3 object — so operators can answer "which objects hold vault logs
// from host X between Y and Z" without scanning S3. The concrete store is
// behind the Index interface so the archiver and CLI test against a fake.
package index
import (
"context"
"crypto/tls"
"fmt"
"time"
"github.com/ClickHouse/clickhouse-go/v2"
"github.com/ClickHouse/clickhouse-go/v2/lib/driver"
)
// Row is one archive-index record.
type Row struct {
ObjectKey string
Bucket string
Subject string
Hosts []string
MinTS time.Time
MaxTS time.Time
EventCount uint64
RawBytes uint64
StoredBytes uint64
Compression string
Cipher string
ContainerFormat string
KeyName string
KeyFingerprint string
}
// Result is one row returned by Search (a subset relevant to retrieval).
type Result struct {
ObjectKey string
Bucket string
Subject string
Hosts []string
MinTS time.Time
MaxTS time.Time
EventCount uint64
RawBytes uint64
StoredBytes uint64
KeyName string
KeyFingerprint string
}
// Index is the archive-index surface.
type Index interface {
Insert(ctx context.Context, row Row) error
Search(ctx context.Context, q SearchQuery) ([]Result, error)
InitSchema(ctx context.Context) error
Ping(ctx context.Context) error
Close() error
}
// Config configures the ClickHouse client.
type Config struct {
Address string // host:port (native protocol, 9000)
Database string
Table string
Username string
Password string
TLS bool
}
// ClickHouse is the ClickHouse-backed Index.
type ClickHouse struct {
conn driver.Conn
database string
table string
}
// NewClickHouse connects to ClickHouse.
func NewClickHouse(ctx context.Context, cfg Config) (*ClickHouse, error) {
opts := &clickhouse.Options{
Addr: []string{cfg.Address},
Auth: clickhouse.Auth{
Database: cfg.Database,
Username: cfg.Username,
Password: cfg.Password,
},
}
if cfg.TLS {
opts.TLS = &tls.Config{MinVersion: tls.VersionTLS12}
}
conn, err := clickhouse.Open(opts)
if err != nil {
return nil, fmt.Errorf("open clickhouse: %w", err)
}
ch := &ClickHouse{conn: conn, database: cfg.Database, table: cfg.Table}
return ch, nil
}
// Ping verifies connectivity.
func (c *ClickHouse) Ping(ctx context.Context) error {
return c.conn.Ping(ctx)
}
// Close closes the connection.
func (c *ClickHouse) Close() error {
return c.conn.Close()
}
// InitSchema creates the database and table if absent (idempotent).
func (c *ClickHouse) InitSchema(ctx context.Context) error {
if err := c.conn.Exec(ctx, CreateDatabaseSQL(c.database)); err != nil {
return fmt.Errorf("create database: %w", err)
}
if err := c.conn.Exec(ctx, CreateTableSQL(c.database, c.table)); err != nil {
return fmt.Errorf("create table: %w", err)
}
return nil
}
// Insert writes one row.
func (c *ClickHouse) Insert(ctx context.Context, row Row) error {
sql := fmt.Sprintf(
"INSERT INTO %s.%s (object_key, bucket, subject, hosts, min_ts, max_ts, event_count, raw_bytes, stored_bytes, compression, cipher, container_format, key_name, key_fingerprint) VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?)",
c.database, c.table)
err := c.conn.Exec(ctx, sql,
row.ObjectKey, row.Bucket, row.Subject, row.Hosts,
row.MinTS, row.MaxTS, row.EventCount, row.RawBytes, row.StoredBytes,
row.Compression, row.Cipher, row.ContainerFormat, row.KeyName, row.KeyFingerprint,
)
if err != nil {
return fmt.Errorf("insert index row: %w", err)
}
return nil
}
// Search runs the parameterized query built from q.
func (c *ClickHouse) Search(ctx context.Context, q SearchQuery) ([]Result, error) {
sql, args := buildSearchSQL(c.database, c.table, q)
rows, err := c.conn.Query(ctx, sql, args...)
if err != nil {
return nil, fmt.Errorf("search index: %w", err)
}
defer func() { _ = rows.Close() }()
var out []Result
for rows.Next() {
var r Result
if err := rows.Scan(
&r.ObjectKey, &r.Bucket, &r.Subject, &r.Hosts,
&r.MinTS, &r.MaxTS, &r.EventCount, &r.RawBytes, &r.StoredBytes,
&r.KeyName, &r.KeyFingerprint,
); err != nil {
return nil, fmt.Errorf("scan result: %w", err)
}
out = append(out, r)
}
if err := rows.Err(); err != nil {
return nil, fmt.Errorf("iterate results: %w", err)
}
return out, nil
}
+111
View File
@@ -0,0 +1,111 @@
package index
import (
"fmt"
"strings"
"time"
)
// SearchQuery describes an index search. Zero-valued fields are omitted.
type SearchQuery struct {
Subject string // NATS-style glob: '*' = one token, '>' = rest. Empty = any.
Host string // exact, or a glob containing '*'. Empty = any.
From time.Time // objects whose range overlaps [From,To]
To time.Time
Limit int
}
// buildSearchSQL renders q into a parameterized ClickHouse SELECT and its args.
// It is pure so it can be unit-tested without a database. Placeholders use the
// clickhouse-go positional style (?), matching Query(ctx, sql, args...).
func buildSearchSQL(database, table string, q SearchQuery) (string, []any) {
var (
where []string
args []any
)
if q.Subject != "" {
where = append(where, "match(subject, ?)")
args = append(args, subjectToRegex(q.Subject))
}
if q.Host != "" {
if strings.Contains(q.Host, "*") {
where = append(where, "arrayExists(h -> match(h, ?), hosts)")
args = append(args, hostGlobToRegex(q.Host))
} else {
where = append(where, "has(hosts, ?)")
args = append(args, q.Host)
}
}
if !q.From.IsZero() {
// object overlaps the window if its max_ts is at/after From.
where = append(where, "max_ts >= ?")
args = append(args, q.From.UTC())
}
if !q.To.IsZero() {
where = append(where, "min_ts <= ?")
args = append(args, q.To.UTC())
}
sql := fmt.Sprintf(
"SELECT object_key, bucket, subject, hosts, min_ts, max_ts, event_count, raw_bytes, stored_bytes, key_name, key_fingerprint FROM %s.%s",
database, table)
if len(where) > 0 {
sql += " WHERE " + strings.Join(where, " AND ")
}
sql += " ORDER BY min_ts, object_key"
if q.Limit > 0 {
sql += " LIMIT ?"
args = append(args, q.Limit)
}
return sql, args
}
// subjectToRegex converts a NATS-style subject glob into an anchored regex for
// ClickHouse match(). '*' matches exactly one dot-delimited token; '>' (only
// meaningful as the final token) matches one or more trailing tokens. Literal
// dots and regex metacharacters are escaped.
func subjectToRegex(glob string) string {
tokens := strings.Split(glob, ".")
var parts []string
for i, tok := range tokens {
switch tok {
case "*":
parts = append(parts, `[^.]+`)
case ">":
// '>' consumes the rest; emit and stop.
if i == 0 {
return "^.+$"
}
return "^" + strings.Join(parts[:i], `\.`) + `(\..+)?$`
default:
parts = append(parts, regexEscape(tok))
}
}
return "^" + strings.Join(parts, `\.`) + "$"
}
// hostGlobToRegex converts a host glob (where '*' matches any run of
// characters, including dots in an FQDN) into an anchored regex for match().
func hostGlobToRegex(glob string) string {
var b strings.Builder
b.WriteByte('^')
for _, seg := range strings.Split(glob, "*") {
b.WriteString(regexEscape(seg))
b.WriteString(".*")
}
// Trim the trailing ".*" added after the last segment, then anchor.
out := strings.TrimSuffix(b.String(), ".*")
return out + "$"
}
func regexEscape(s string) string {
const meta = `\.+*?()|[]{}^$`
var b strings.Builder
for _, r := range s {
if strings.ContainsRune(meta, r) {
b.WriteByte('\\')
}
b.WriteRune(r)
}
return b.String()
}
+110
View File
@@ -0,0 +1,110 @@
package index
import (
"regexp"
"strings"
"testing"
"time"
)
func TestSubjectToRegex(t *testing.T) {
cases := []struct {
glob string
match []string
nomatch []string
}{
{
glob: "logs.vm.*",
match: []string{"logs.vm.db-1", "logs.vm.web"},
nomatch: []string{"logs.vm", "logs.vm.db.1", "logs.k8s.x"},
},
{
glob: "logs.k8s.vault.>",
match: []string{"logs.k8s.vault.audit", "logs.k8s.vault.a.b", "logs.k8s.vault"},
nomatch: []string{"logs.k8s.shop.web", "logs.vm.x"},
},
{
glob: "logs.vm.db-1",
match: []string{"logs.vm.db-1"},
nomatch: []string{"logs.vm.db-2", "logs.vm.db-1.x"},
},
}
for _, c := range cases {
re := regexp.MustCompile(subjectToRegex(c.glob))
for _, s := range c.match {
if !re.MatchString(s) {
t.Errorf("%q -> %q should match %q", c.glob, re.String(), s)
}
}
for _, s := range c.nomatch {
if re.MatchString(s) {
t.Errorf("%q -> %q should NOT match %q", c.glob, re.String(), s)
}
}
}
}
func TestBuildSearchSQLFull(t *testing.T) {
from := time.Date(2026, 7, 26, 0, 0, 0, 0, time.UTC)
to := time.Date(2026, 7, 27, 0, 0, 0, 0, time.UTC)
sql, args := buildSearchSQL("logs", "archive_index", SearchQuery{
Subject: "logs.k8s.vault.>",
Host: "node-1",
From: from,
To: to,
Limit: 50,
})
if !strings.Contains(sql, "FROM logs.archive_index") {
t.Errorf("missing table: %s", sql)
}
for _, want := range []string{"match(subject, ?)", "has(hosts, ?)", "max_ts >= ?", "min_ts <= ?", "ORDER BY min_ts", "LIMIT ?"} {
if !strings.Contains(sql, want) {
t.Errorf("sql missing %q: %s", want, sql)
}
}
if len(args) != 5 {
t.Fatalf("args = %d, want 5: %v", len(args), args)
}
if args[1] != "node-1" {
t.Errorf("host arg = %v", args[1])
}
if args[4] != 50 {
t.Errorf("limit arg = %v", args[4])
}
}
func TestBuildSearchSQLEmpty(t *testing.T) {
sql, args := buildSearchSQL("logs", "archive_index", SearchQuery{})
if strings.Contains(sql, "WHERE") {
t.Errorf("empty query should have no WHERE: %s", sql)
}
if len(args) != 0 {
t.Errorf("args = %v, want none", args)
}
}
func TestBuildSearchSQLHostGlob(t *testing.T) {
sql, args := buildSearchSQL("logs", "archive_index", SearchQuery{Host: "db-*"})
if !strings.Contains(sql, "arrayExists(h -> match(h, ?), hosts)") {
t.Errorf("host glob should use arrayExists/match: %s", sql)
}
if len(args) != 1 {
t.Fatalf("args = %v", args)
}
re := regexp.MustCompile(args[0].(string))
if !re.MatchString("db-1") || re.MatchString("web-1") {
t.Errorf("host glob regex wrong: %q", args[0])
}
}
func TestDDLContainsKeyColumns(t *testing.T) {
ddl := CreateTableSQL("logs", "archive_index")
for _, col := range []string{"object_key", "subject", "hosts", "min_ts", "max_ts", "event_count", "key_fingerprint", "bloom_filter"} {
if !strings.Contains(ddl, col) {
t.Errorf("DDL missing %q", col)
}
}
if !strings.Contains(ddl, "IF NOT EXISTS") {
t.Errorf("DDL should be idempotent")
}
}