Initial implementation: NATS->S3 archiver + search/retrieve CLI
logarchiver replaces the plain Vector archiver leg of the centralized logging stack (argocd-apps #296) with a Go service that archives raw logs from NATS JetStream to S3 as zstd-compressed, OpenPGP-encrypted, indexed objects, plus an operator CLI to search the index and retrieve/decrypt archived logs. It adds the things that outgrew Vector: zstd compression, encryption keyed from Ben's Vault GPG secrets engine, a searchable ClickHouse index, and sink-conditional acks (a batch is acknowledged to JetStream only after the object is durably in S3 AND indexed). Service (`logarchiver run`): - Durable JetStream pull consumer (stream LOGS, durable archiver, subject filter default logs.k8s.vault.>), explicit acks, independent offsets. - Batch per subject by size/count/time -> NDJSON -> zstd -> encrypt -> S3 PUT -> ClickHouse index row -> ack. On any failure the batch is Nak'd and redelivered, so nothing is lost on a sink outage. - Encryption is a wrapped-DEK envelope (container LARC1): the bulk is AES-256-GCM framed under a random data key, and only that 32-byte key is OpenPGP-encrypted to the engine's public key. This is because the Vault GPG engine does whole-payload decrypt only; retrieval round-trips just the tiny wrapped key regardless of object size. Public key fetched from the engine or a mounted file (configurable); key fingerprint recorded per object; periodic pubkey refresh for rotation. - Prometheus metrics, structured slog, graceful drain on shutdown. CLI: - `search` queries the index (subject/host/time) and lists matching objects. - `fetch` downloads, decrypts via the Vault GPG engine, unzstds and emits NDJSON (optionally re-filtered by host/time). - `init-schema` creates/prints the ClickHouse archive_index DDL. - cobra `completion` subcommands. Config via file+env (k8s-friendly, secrets from env), boundaries (NATS/S3/ ClickHouse/Vault) behind interfaces with unit tests (config, batching, host/subject extraction, crypto roundtrip with a test key, ack-after-persist with fakes, search query building). go build/vet/test -race clean; golangci-lint v2 clean. Woodpecker CI: build/test/pre-commit on PR; on v* tag a container image plus a Gitea binary release + rpm-internal RPM. Docs per subcommand + architecture + retrieval runbook + deployment drop-in. Claude-Session: https://claude.ai/code/session_015ur3i7D2azsMAWTSVABApv
This commit is contained in:
@@ -0,0 +1,80 @@
|
||||
package consumer
|
||||
|
||||
import (
|
||||
"context"
|
||||
"crypto/tls"
|
||||
"crypto/x509"
|
||||
"fmt"
|
||||
"os"
|
||||
"time"
|
||||
|
||||
"git.unkin.net/unkin/logarchiver/internal/config"
|
||||
"github.com/nats-io/nats.go"
|
||||
"github.com/nats-io/nats.go/jetstream"
|
||||
)
|
||||
|
||||
// Connect dials NATS as the configured user and returns the connection and a
|
||||
// JetStream context. Callers must Close the returned *nats.Conn.
|
||||
func Connect(cfg config.NATSConfig) (*nats.Conn, jetstream.JetStream, error) {
|
||||
opts := []nats.Option{
|
||||
nats.Name("logarchiver"),
|
||||
nats.MaxReconnects(-1),
|
||||
nats.ReconnectWait(2 * time.Second),
|
||||
}
|
||||
if cfg.User != "" {
|
||||
opts = append(opts, nats.UserInfo(cfg.User, cfg.Password))
|
||||
}
|
||||
if cfg.CAFile != "" {
|
||||
pool := x509.NewCertPool()
|
||||
pem, err := os.ReadFile(cfg.CAFile)
|
||||
if err != nil {
|
||||
return nil, nil, fmt.Errorf("read nats ca %s: %w", cfg.CAFile, err)
|
||||
}
|
||||
if !pool.AppendCertsFromPEM(pem) {
|
||||
return nil, nil, fmt.Errorf("no certs parsed from nats ca %s", cfg.CAFile)
|
||||
}
|
||||
opts = append(opts, nats.Secure(&tls.Config{RootCAs: pool, MinVersion: tls.VersionTLS12}))
|
||||
}
|
||||
nc, err := nats.Connect(cfg.URL, opts...)
|
||||
if err != nil {
|
||||
return nil, nil, fmt.Errorf("connect nats %s: %w", cfg.URL, err)
|
||||
}
|
||||
js, err := jetstream.New(nc)
|
||||
if err != nil {
|
||||
nc.Close()
|
||||
return nil, nil, fmt.Errorf("jetstream context: %w", err)
|
||||
}
|
||||
return nc, js, nil
|
||||
}
|
||||
|
||||
// EnsureConsumer creates or updates the durable pull consumer on the stream with
|
||||
// the configured subject filters. Independent offsets and explicit acks give
|
||||
// logarchiver at-least-once delivery decoupled from the transform tier.
|
||||
func EnsureConsumer(ctx context.Context, js jetstream.JetStream, cfg config.NATSConfig) (jetstream.Consumer, error) {
|
||||
ackWait := cfg.AckWait
|
||||
if ackWait <= 0 {
|
||||
ackWait = 2 * time.Minute
|
||||
}
|
||||
consCfg := jetstream.ConsumerConfig{
|
||||
Durable: cfg.Durable,
|
||||
Name: cfg.Durable,
|
||||
AckPolicy: jetstream.AckExplicitPolicy,
|
||||
DeliverPolicy: jetstream.DeliverAllPolicy,
|
||||
AckWait: ackWait,
|
||||
MaxDeliver: -1,
|
||||
ReplayPolicy: jetstream.ReplayInstantPolicy,
|
||||
}
|
||||
switch len(cfg.Subjects) {
|
||||
case 0:
|
||||
return nil, fmt.Errorf("no subject filters configured")
|
||||
case 1:
|
||||
consCfg.FilterSubject = cfg.Subjects[0]
|
||||
default:
|
||||
consCfg.FilterSubjects = cfg.Subjects
|
||||
}
|
||||
cons, err := js.CreateOrUpdateConsumer(ctx, cfg.Stream, consCfg)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("ensure consumer %s on stream %s: %w", cfg.Durable, cfg.Stream, err)
|
||||
}
|
||||
return cons, nil
|
||||
}
|
||||
@@ -0,0 +1,8 @@
|
||||
package consumer
|
||||
|
||||
import "git.unkin.net/unkin/logarchiver/internal/event"
|
||||
|
||||
// extractMeta projects the host/timestamp fields from a raw event payload.
|
||||
func extractMeta(raw []byte) event.Meta {
|
||||
return event.Extract(raw)
|
||||
}
|
||||
@@ -0,0 +1,237 @@
|
||||
// Package consumer binds the JetStream pull consumer and runs the archive loop.
|
||||
//
|
||||
// The core correctness property: a batch's messages are acknowledged ONLY after
|
||||
// the batch has been sealed, uploaded to S3, and indexed in ClickHouse. If any
|
||||
// of those fails the messages are Nak'd (with a backoff) and JetStream
|
||||
// redelivers them, so nothing is lost on a sink outage. This sink-conditional
|
||||
// acking is the main thing logarchiver does that a stock Vector NATS consumer
|
||||
// cannot.
|
||||
package consumer
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"log/slog"
|
||||
"time"
|
||||
|
||||
"git.unkin.net/unkin/logarchiver/internal/batcher"
|
||||
"github.com/nats-io/nats.go/jetstream"
|
||||
)
|
||||
|
||||
// Persister stores a ready batch durably. On success the caller acks.
|
||||
type Persister interface {
|
||||
Store(ctx context.Context, batch *batcher.Batch) (StoreResult, error)
|
||||
}
|
||||
|
||||
// StoreResult mirrors archiver.StoreResult (kept local to avoid an import cycle;
|
||||
// the archiver's result is adapted at the call site).
|
||||
type StoreResult struct {
|
||||
ObjectKey string
|
||||
Events int
|
||||
RawBytes int64
|
||||
StoredBytes int64
|
||||
}
|
||||
|
||||
// Metrics is the optional metrics surface for the loop.
|
||||
type Metrics interface {
|
||||
MessagesFetched(n int)
|
||||
Acked(n int)
|
||||
BatchFlushed(trigger string)
|
||||
SetPending(n int)
|
||||
}
|
||||
|
||||
// Runner drives the fetch → batch → persist → ack loop.
|
||||
type Runner struct {
|
||||
cons jetstream.Consumer
|
||||
batcher *batcher.Batcher
|
||||
persist Persister
|
||||
log *slog.Logger
|
||||
metrics Metrics
|
||||
fetchBatch int
|
||||
pollWait time.Duration
|
||||
nakBackoff time.Duration
|
||||
drainTO time.Duration
|
||||
nowFn func() time.Time
|
||||
}
|
||||
|
||||
// Options configures a Runner.
|
||||
type Options struct {
|
||||
Consumer jetstream.Consumer
|
||||
Batcher *batcher.Batcher
|
||||
Persister Persister
|
||||
Logger *slog.Logger
|
||||
Metrics Metrics
|
||||
FetchBatch int
|
||||
// PollWait bounds each Fetch and thus how often age-based flushes are checked.
|
||||
PollWait time.Duration
|
||||
// NakBackoff delays redelivery after a persist failure.
|
||||
NakBackoff time.Duration
|
||||
// DrainTimeout bounds the shutdown flush.
|
||||
DrainTimeout time.Duration
|
||||
}
|
||||
|
||||
// NewRunner builds a Runner.
|
||||
func NewRunner(o Options) *Runner {
|
||||
fetch := o.FetchBatch
|
||||
if fetch <= 0 {
|
||||
fetch = 512
|
||||
}
|
||||
poll := o.PollWait
|
||||
if poll <= 0 {
|
||||
poll = time.Second
|
||||
}
|
||||
nak := o.NakBackoff
|
||||
if nak <= 0 {
|
||||
nak = 10 * time.Second
|
||||
}
|
||||
drain := o.DrainTimeout
|
||||
if drain <= 0 {
|
||||
drain = 30 * time.Second
|
||||
}
|
||||
log := o.Logger
|
||||
if log == nil {
|
||||
log = slog.Default()
|
||||
}
|
||||
return &Runner{
|
||||
cons: o.Consumer,
|
||||
batcher: o.Batcher,
|
||||
persist: o.Persister,
|
||||
log: log,
|
||||
metrics: o.Metrics,
|
||||
fetchBatch: fetch,
|
||||
pollWait: poll,
|
||||
nakBackoff: nak,
|
||||
drainTO: drain,
|
||||
nowFn: time.Now,
|
||||
}
|
||||
}
|
||||
|
||||
// Run loops until ctx is cancelled, then drains open batches before returning.
|
||||
func (r *Runner) Run(ctx context.Context) error {
|
||||
r.log.Info("archive loop started",
|
||||
"fetch_batch", r.fetchBatch, "poll_wait", r.pollWait.String())
|
||||
for {
|
||||
if ctx.Err() != nil {
|
||||
return r.drain()
|
||||
}
|
||||
|
||||
// Age-based flush before fetching more.
|
||||
r.flushBatches(ctx, r.batcher.DueByAge(r.nowFn()), "age")
|
||||
|
||||
msgs, err := r.cons.Fetch(r.fetchBatch, jetstream.FetchMaxWait(r.pollWait))
|
||||
if err != nil {
|
||||
if errors.Is(err, context.Canceled) || ctx.Err() != nil {
|
||||
return r.drain()
|
||||
}
|
||||
r.log.Warn("fetch failed", "err", err)
|
||||
r.sleep(ctx, r.pollWait)
|
||||
continue
|
||||
}
|
||||
|
||||
n := 0
|
||||
for msg := range msgs.Messages() {
|
||||
n++
|
||||
r.route(ctx, msg)
|
||||
}
|
||||
if ferr := msgs.Error(); ferr != nil && !errors.Is(ferr, context.Canceled) {
|
||||
r.log.Warn("fetch iteration error", "err", ferr)
|
||||
}
|
||||
if r.metrics != nil {
|
||||
r.metrics.MessagesFetched(n)
|
||||
r.metrics.SetPending(r.batcher.Pending())
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// route decodes a message and adds it to the batcher, flushing if the batch
|
||||
// becomes full.
|
||||
func (r *Runner) route(ctx context.Context, msg jetstream.Msg) {
|
||||
meta := extractMeta(msg.Data())
|
||||
full := r.batcher.Add(batcher.Item{
|
||||
Subject: msg.Subject(),
|
||||
Raw: msg.Data(),
|
||||
Host: meta.Host,
|
||||
Timestamp: meta.Timestamp,
|
||||
HasTS: meta.Ok,
|
||||
Ack: msg,
|
||||
})
|
||||
if full != nil {
|
||||
r.flush(ctx, full, "full")
|
||||
}
|
||||
}
|
||||
|
||||
// flushBatches flushes a slice of batches with the given trigger label.
|
||||
func (r *Runner) flushBatches(ctx context.Context, batches []*batcher.Batch, trigger string) {
|
||||
for _, b := range batches {
|
||||
r.flush(ctx, b, trigger)
|
||||
}
|
||||
}
|
||||
|
||||
// flush persists a batch and, only on success, acks its messages. On failure it
|
||||
// Naks with a backoff so JetStream redelivers.
|
||||
func (r *Runner) flush(ctx context.Context, b *batcher.Batch, trigger string) {
|
||||
if len(b.Items) == 0 {
|
||||
return
|
||||
}
|
||||
res, err := r.persist.Store(ctx, b)
|
||||
if err != nil {
|
||||
r.log.Error("persist failed; batch will be redelivered",
|
||||
"subject", b.Subject, "events", len(b.Items), "trigger", trigger, "err", err)
|
||||
r.nakAll(b)
|
||||
return
|
||||
}
|
||||
acked := r.ackAll(b)
|
||||
if r.metrics != nil {
|
||||
r.metrics.Acked(acked)
|
||||
r.metrics.BatchFlushed(trigger)
|
||||
}
|
||||
r.log.Info("object archived",
|
||||
"subject", b.Subject, "object_key", res.ObjectKey,
|
||||
"events", res.Events, "raw_bytes", res.RawBytes, "stored_bytes", res.StoredBytes,
|
||||
"trigger", trigger)
|
||||
}
|
||||
|
||||
func (r *Runner) ackAll(b *batcher.Batch) int {
|
||||
n := 0
|
||||
for _, it := range b.Items {
|
||||
if msg, ok := it.Ack.(jetstream.Msg); ok {
|
||||
if err := msg.Ack(); err != nil {
|
||||
r.log.Warn("ack failed", "subject", b.Subject, "err", err)
|
||||
continue
|
||||
}
|
||||
n++
|
||||
}
|
||||
}
|
||||
return n
|
||||
}
|
||||
|
||||
func (r *Runner) nakAll(b *batcher.Batch) {
|
||||
for _, it := range b.Items {
|
||||
if msg, ok := it.Ack.(jetstream.Msg); ok {
|
||||
_ = msg.NakWithDelay(r.nakBackoff)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// drain flushes all open batches during shutdown with a bounded timeout.
|
||||
func (r *Runner) drain() error {
|
||||
batches := r.batcher.Drain()
|
||||
if len(batches) == 0 {
|
||||
r.log.Info("archive loop stopped; nothing to drain")
|
||||
return nil
|
||||
}
|
||||
ctx, cancel := context.WithTimeout(context.Background(), r.drainTO)
|
||||
defer cancel()
|
||||
r.log.Info("draining open batches", "batches", len(batches))
|
||||
r.flushBatches(ctx, batches, "shutdown")
|
||||
return nil
|
||||
}
|
||||
|
||||
func (r *Runner) sleep(ctx context.Context, d time.Duration) {
|
||||
t := time.NewTimer(d)
|
||||
defer t.Stop()
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
case <-t.C:
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,151 @@
|
||||
package consumer
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"git.unkin.net/unkin/logarchiver/internal/batcher"
|
||||
"github.com/nats-io/nats.go"
|
||||
"github.com/nats-io/nats.go/jetstream"
|
||||
)
|
||||
|
||||
// fakeMsg is a minimal jetstream.Msg recording ack/nak calls.
|
||||
type fakeMsg struct {
|
||||
subject string
|
||||
data []byte
|
||||
mu sync.Mutex
|
||||
acked bool
|
||||
naked bool
|
||||
}
|
||||
|
||||
func (m *fakeMsg) Metadata() (*jetstream.MsgMetadata, error) { return &jetstream.MsgMetadata{}, nil }
|
||||
func (m *fakeMsg) Data() []byte { return m.data }
|
||||
func (m *fakeMsg) Headers() nats.Header { return nil }
|
||||
func (m *fakeMsg) Subject() string { return m.subject }
|
||||
func (m *fakeMsg) Reply() string { return "" }
|
||||
func (m *fakeMsg) Ack() error {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
m.acked = true
|
||||
return nil
|
||||
}
|
||||
func (m *fakeMsg) DoubleAck(context.Context) error { return nil }
|
||||
func (m *fakeMsg) Nak() error {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
m.naked = true
|
||||
return nil
|
||||
}
|
||||
func (m *fakeMsg) NakWithDelay(time.Duration) error {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
m.naked = true
|
||||
return nil
|
||||
}
|
||||
func (m *fakeMsg) InProgress() error { return nil }
|
||||
func (m *fakeMsg) Term() error { return nil }
|
||||
func (m *fakeMsg) TermWithReason(string) error { return nil }
|
||||
|
||||
func (m *fakeMsg) isAcked() bool {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
return m.acked
|
||||
}
|
||||
func (m *fakeMsg) isNaked() bool {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
return m.naked
|
||||
}
|
||||
|
||||
// fakePersister records calls and can be made to fail.
|
||||
type fakePersister struct {
|
||||
fail bool
|
||||
called int
|
||||
}
|
||||
|
||||
func (p *fakePersister) Store(_ context.Context, b *batcher.Batch) (StoreResult, error) {
|
||||
p.called++
|
||||
if p.fail {
|
||||
return StoreResult{}, errors.New("boom")
|
||||
}
|
||||
return StoreResult{ObjectKey: "k", Events: len(b.Items)}, nil
|
||||
}
|
||||
|
||||
func batchWith(msgs ...*fakeMsg) *batcher.Batch {
|
||||
b := &batcher.Batch{Subject: "s"}
|
||||
for _, m := range msgs {
|
||||
b.Items = append(b.Items, batcher.Item{Subject: "s", Raw: m.data, Ack: jetstream.Msg(m)})
|
||||
}
|
||||
return b
|
||||
}
|
||||
|
||||
// TestFlushAcksOnlyAfterPersist is the core correctness test: messages are acked
|
||||
// exactly when Store succeeds, and Nak'd (never acked) when it fails.
|
||||
func TestFlushAcksOnSuccess(t *testing.T) {
|
||||
p := &fakePersister{}
|
||||
r := NewRunner(Options{Persister: p})
|
||||
m1 := &fakeMsg{subject: "s", data: []byte(`{"host":"h"}`)}
|
||||
m2 := &fakeMsg{subject: "s", data: []byte(`{"host":"h2"}`)}
|
||||
|
||||
r.flush(context.Background(), batchWith(m1, m2), "test")
|
||||
|
||||
if p.called != 1 {
|
||||
t.Fatalf("Store called %d times, want 1", p.called)
|
||||
}
|
||||
if !m1.isAcked() || !m2.isAcked() {
|
||||
t.Errorf("messages should be acked after successful persist")
|
||||
}
|
||||
if m1.isNaked() || m2.isNaked() {
|
||||
t.Errorf("messages must not be naked on success")
|
||||
}
|
||||
}
|
||||
|
||||
func TestFlushNaksOnFailure(t *testing.T) {
|
||||
p := &fakePersister{fail: true}
|
||||
r := NewRunner(Options{Persister: p})
|
||||
m1 := &fakeMsg{subject: "s", data: []byte(`{"host":"h"}`)}
|
||||
|
||||
r.flush(context.Background(), batchWith(m1), "test")
|
||||
|
||||
if m1.isAcked() {
|
||||
t.Errorf("message must NOT be acked when persist fails")
|
||||
}
|
||||
if !m1.isNaked() {
|
||||
t.Errorf("message should be naked so JetStream redelivers")
|
||||
}
|
||||
}
|
||||
|
||||
func TestFlushEmptyBatchNoop(t *testing.T) {
|
||||
p := &fakePersister{}
|
||||
r := NewRunner(Options{Persister: p})
|
||||
r.flush(context.Background(), &batcher.Batch{Subject: "s"}, "test")
|
||||
if p.called != 0 {
|
||||
t.Errorf("empty batch should not call Store")
|
||||
}
|
||||
}
|
||||
|
||||
// TestRouteAndFlushIntegration wires a real batcher: adding enough messages to
|
||||
// fill the batch triggers a full flush that persists and acks exactly those.
|
||||
func TestRouteFlushViaBatcher(t *testing.T) {
|
||||
p := &fakePersister{}
|
||||
bat := batcher.New(batcher.Limits{MaxEvents: 2})
|
||||
r := NewRunner(Options{Persister: p, Batcher: bat})
|
||||
|
||||
m1 := &fakeMsg{subject: "s", data: []byte(`{"host":"a"}`)}
|
||||
m2 := &fakeMsg{subject: "s", data: []byte(`{"host":"b"}`)}
|
||||
r.route(context.Background(), m1)
|
||||
if m1.isAcked() {
|
||||
t.Errorf("first message should not be acked before batch fills")
|
||||
}
|
||||
r.route(context.Background(), m2) // fills batch -> flush
|
||||
|
||||
if p.called != 1 {
|
||||
t.Fatalf("Store called %d times, want 1 after fill", p.called)
|
||||
}
|
||||
if !m1.isAcked() || !m2.isAcked() {
|
||||
t.Errorf("both messages should be acked after the full-batch flush")
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user