Disk buffer for batches VictoriaLogs cannot take, lossless Docker positions, non-blocking reverse DNS

- Batches that fail go to /data/spool (SPOOL_MAX_MB, 1 GiB by default) and
  are sent again oldest first; retries no longer block the store loop and
  follow the shutdown context.
- Docker and host logs wait for room in a full queue instead of being
  dropped; the Docker position only moves once a line is stored or spooled.
- Reverse DNS no longer holds up the syslog listeners, with an LRU cache
  and a cap on concurrent lookups.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
This commit is contained in:
cedricandClaude Opus 5.5 committed 2026-10-03 16:26:18 +02:00
1 parent 3466a29692
commit 3504263992
14 files changed
+778 -84

No files matched your search

+89 -14
View File
@@ -1,6 +1,7 @@
package main
import (
"container/list"
"context"
"net"
"strings"
@@ -15,17 +16,26 @@ type ReverseDNS struct {
posTTL time.Duration // cache duration of a found name
negTTL time.Duration // cache duration of "no name"
lookups chan struct{} // limits the lookups running at once
waiters chan struct{} // limits the messages waiting for a lookup (Resolve)
mu sync.Mutex
cache map[string]*rdnsEntry
cache map[string]*list.Element // ip -> element of lru
lru *list.List // *rdnsEntry, most recently used first
}
type rdnsEntry struct {
ip string
name string
expires time.Time
done chan struct{} // closed once the lookup has finished
}
const rdnsMaxEntries = 10000
const (
rdnsMaxEntries = 10000
rdnsMaxLookups = 64
rdnsMaxWaiters = 1024
)
// NewReverseDNS uses the system resolver, or `server` ("ip" or "ip:port") when set.
func NewReverseDNS(enabled bool, server string) *ReverseDNS {
@@ -47,7 +57,10 @@ func NewReverseDNS(enabled bool, server string) *ReverseDNS {
r: r,
posTTL: time.Hour,
negTTL: 10 * time.Minute,
cache: make(map[string]*rdnsEntry),
lookups: make(chan struct{}, rdnsMaxLookups),
waiters: make(chan struct{}, rdnsMaxWaiters),
cache: make(map[string]*list.Element),
lru: list.New(),
}
}
@@ -60,6 +73,38 @@ func isClosed(ch chan struct{}) bool {
}
}
// entry returns the cache entry of ip, starting its lookup when it is missing
// or expired, or nil when too many lookups are already running (a flood of
// unknown addresses). The least recently used entry makes room for a new one.
func (d *ReverseDNS) entry(ip string) *rdnsEntry {
d.mu.Lock()
defer d.mu.Unlock()
if el := d.cache[ip]; el != nil {
e := el.Value.(*rdnsEntry)
if !isClosed(e.done) || time.Now().Before(e.expires) {
d.lru.MoveToFront(el)
return e
}
}
select {
case d.lookups <- struct{}{}:
default:
return nil
}
if el := d.cache[ip]; el != nil {
d.lru.Remove(el)
}
for d.lru.Len() >= rdnsMaxEntries {
old := d.lru.Back()
d.lru.Remove(old)
delete(d.cache, old.Value.(*rdnsEntry).ip)
}
e := &rdnsEntry{ip: ip, done: make(chan struct{})}
d.cache[ip] = d.lru.PushFront(e)
go d.resolve(ip, e)
return e
}
// Lookup returns the name of ip, or "" when ip is not an IP address, has no
// PTR record, or is not resolved within `wait`. A lookup that takes longer
// keeps running in the background and fills the cache for later calls.
@@ -67,18 +112,10 @@ func (d *ReverseDNS) Lookup(ip string, wait time.Duration) string {
if !d.enabled || net.ParseIP(ip) == nil {
return ""
}
d.mu.Lock()
e := d.cache[ip]
if e == nil || (isClosed(e.done) && time.Now().After(e.expires)) {
if len(d.cache) >= rdnsMaxEntries {
d.cache = make(map[string]*rdnsEntry)
}
e = &rdnsEntry{done: make(chan struct{})}
d.cache[ip] = e
go d.resolve(ip, e)
e := d.entry(ip)
if e == nil {
return ""
}
d.mu.Unlock()
if !isClosed(e.done) {
timer := time.NewTimer(wait)
defer timer.Stop()
@@ -91,6 +128,43 @@ func (d *ReverseDNS) Lookup(ip string, wait time.Duration) string {
return e.name
}
// Resolve is Lookup without blocking the caller: fn gets the name (or "") at
// once when it is known, otherwise from a goroutine after at most `wait`. The
// syslog listeners use it so that a slow DNS server never delays the reading
// of the next messages.
func (d *ReverseDNS) Resolve(ip string, wait time.Duration, fn func(name string)) {
if !d.enabled || net.ParseIP(ip) == nil {
fn("")
return
}
e := d.entry(ip)
switch {
case e == nil:
fn("")
return
case isClosed(e.done):
fn(e.name)
return
}
select {
case d.waiters <- struct{}{}:
default:
fn("") // too many messages waiting already
return
}
go func() {
defer func() { <-d.waiters }()
timer := time.NewTimer(wait)
defer timer.Stop()
select {
case <-e.done:
fn(e.name)
case <-timer.C:
fn("")
}
}()
}
// LookupMany resolves several addresses in parallel; unresolved ones are absent.
func (d *ReverseDNS) LookupMany(ips []string, wait time.Duration) map[string]string {
out := make(map[string]string)
@@ -112,6 +186,7 @@ func (d *ReverseDNS) LookupMany(ips []string, wait time.Duration) map[string]str
}
func (d *ReverseDNS) resolve(ip string, e *rdnsEntry) {
defer func() { <-d.lookups }()
ctx, cancel := context.WithTimeout(context.Background(), 3*time.Second)
defer cancel()
ttl := d.negTTL