Files
logarchiver/internal/cli/filter.go
T
benvin 5ec89b0028 Initial implementation: NATS->S3 archiver + search/retrieve CLI
logarchiver replaces the plain Vector archiver leg of the centralized
logging stack (argocd-apps #296) with a Go service that archives raw logs
from NATS JetStream to S3 as zstd-compressed, OpenPGP-encrypted, indexed
objects, plus an operator CLI to search the index and retrieve/decrypt
archived logs. It adds the things that outgrew Vector: zstd compression,
encryption keyed from Ben's Vault GPG secrets engine, a searchable
ClickHouse index, and sink-conditional acks (a batch is acknowledged to
JetStream only after the object is durably in S3 AND indexed).

Service (`logarchiver run`):
- Durable JetStream pull consumer (stream LOGS, durable archiver, subject
  filter default logs.k8s.vault.>), explicit acks, independent offsets.
- Batch per subject by size/count/time -> NDJSON -> zstd -> encrypt -> S3
  PUT -> ClickHouse index row -> ack. On any failure the batch is Nak'd and
  redelivered, so nothing is lost on a sink outage.
- Encryption is a wrapped-DEK envelope (container LARC1): the bulk is
  AES-256-GCM framed under a random data key, and only that 32-byte key is
  OpenPGP-encrypted to the engine's public key. This is because the Vault
  GPG engine does whole-payload decrypt only; retrieval round-trips just the
  tiny wrapped key regardless of object size. Public key fetched from the
  engine or a mounted file (configurable); key fingerprint recorded per
  object; periodic pubkey refresh for rotation.
- Prometheus metrics, structured slog, graceful drain on shutdown.

CLI:
- `search` queries the index (subject/host/time) and lists matching objects.
- `fetch` downloads, decrypts via the Vault GPG engine, unzstds and emits
  NDJSON (optionally re-filtered by host/time).
- `init-schema` creates/prints the ClickHouse archive_index DDL.
- cobra `completion` subcommands.

Config via file+env (k8s-friendly, secrets from env), boundaries (NATS/S3/
ClickHouse/Vault) behind interfaces with unit tests (config, batching,
host/subject extraction, crypto roundtrip with a test key, ack-after-persist
with fakes, search query building). go build/vet/test -race clean;
golangci-lint v2 clean. Woodpecker CI: build/test/pre-commit on PR; on v*
tag a container image plus a Gitea binary release + rpm-internal RPM. Docs
per subcommand + architecture + retrieval runbook + deployment drop-in.

Claude-Session: https://claude.ai/code/session_015ur3i7D2azsMAWTSVABApv
2026-07-27 22:11:54 +10:00

124 lines
2.9 KiB
Go

package cli
import (
"bytes"
"io"
"strings"
"time"
"git.unkin.net/unkin/logarchiver/internal/event"
)
// lineFilter is a predicate over a single NDJSON event line.
type lineFilter func(raw []byte) bool
// newLineFilter builds a predicate from optional host/time constraints. A nil
// filter (all constraints empty) means "pass everything".
func newLineFilter(host string, from, to time.Time) lineFilter {
if host == "" && from.IsZero() && to.IsZero() {
return nil
}
return func(raw []byte) bool {
meta := event.Extract(raw)
if host != "" && !globMatch(host, meta.Host) {
return false
}
if !from.IsZero() || !to.IsZero() {
// Events without a parseable timestamp are kept (we cannot exclude
// them on time grounds without dropping data).
if meta.Ok {
if !from.IsZero() && meta.Timestamp.Before(from) {
return false
}
if !to.IsZero() && meta.Timestamp.After(to) {
return false
}
}
}
return true
}
}
// filterWriter forwards only complete NDJSON lines that satisfy filter. It
// buffers a trailing partial line across Writes so streaming decryption can feed
// it arbitrary chunks. Flush must be called at end to emit any final unterminated
// line. A nil filter forwards bytes verbatim.
type filterWriter struct {
dst io.Writer
filter lineFilter
buf bytes.Buffer
}
func newFilterWriter(dst io.Writer, filter lineFilter) *filterWriter {
return &filterWriter{dst: dst, filter: filter}
}
func (w *filterWriter) Write(p []byte) (int, error) {
if w.filter == nil {
return w.dst.Write(p)
}
w.buf.Write(p)
for {
data := w.buf.Bytes()
i := bytes.IndexByte(data, '\n')
if i < 0 {
break
}
line := data[:i]
if len(bytes.TrimSpace(line)) > 0 && w.filter(line) {
if _, err := w.dst.Write(line); err != nil {
return 0, err
}
if _, err := w.dst.Write([]byte{'\n'}); err != nil {
return 0, err
}
}
w.buf.Next(i + 1)
}
return len(p), nil
}
// Flush emits a trailing line that had no terminating newline.
func (w *filterWriter) Flush() error {
if w.filter == nil {
return nil
}
line := bytes.TrimRight(w.buf.Bytes(), "\n")
w.buf.Reset()
if len(bytes.TrimSpace(line)) > 0 && w.filter(line) {
if _, err := w.dst.Write(line); err != nil {
return err
}
if _, err := w.dst.Write([]byte{'\n'}); err != nil {
return err
}
}
return nil
}
// globMatch matches pattern against s where '*' matches any run of characters.
// With no '*', it is an exact match.
func globMatch(pattern, s string) bool {
if !strings.Contains(pattern, "*") {
return pattern == s
}
parts := strings.Split(pattern, "*")
// Anchor first part.
if !strings.HasPrefix(s, parts[0]) {
return false
}
s = s[len(parts[0]):]
for _, part := range parts[1 : len(parts)-1] {
if part == "" {
continue
}
idx := strings.Index(s, part)
if idx < 0 {
return false
}
s = s[idx+len(part):]
}
// Anchor last part.
return strings.HasSuffix(s, parts[len(parts)-1])
}