feat(kafka): add shard-aware topology, admin, and client layer
Topic naming (kmsvc.{queue}.shard-{id} / .fifo.shard-{id} / .dlq.shard-{id}),
hash-range shard selection/splitting (murmur2-based, matching Kafka's
default partitioner), and topic admin/producer/consumer client
constructors shared by the operator and the message-plane.
This commit is contained in:
@@ -0,0 +1,153 @@
|
|||||||
|
package kafka
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"fmt"
|
||||||
|
"strconv"
|
||||||
|
|
||||||
|
"github.com/twmb/franz-go/pkg/kadm"
|
||||||
|
"github.com/twmb/franz-go/pkg/kgo"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Admin wraps kadm.Client with the topic conventions from design.md §6.
|
||||||
|
type Admin struct {
|
||||||
|
client *kadm.Client
|
||||||
|
}
|
||||||
|
|
||||||
|
// NewAdmin builds an Admin from a set of broker addresses.
|
||||||
|
func NewAdmin(brokers []string) (*Admin, error) {
|
||||||
|
cl, err := kgo.NewClient(kgo.SeedBrokers(brokers...))
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("creating kafka client: %w", err)
|
||||||
|
}
|
||||||
|
return &Admin{client: kadm.NewClient(cl)}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// TopicConfig is the subset of Kafka topic configuration design.md §6 cares about.
|
||||||
|
type TopicConfig struct {
|
||||||
|
PartitionCount int32
|
||||||
|
ReplicationFactor int16
|
||||||
|
RetentionSeconds int32
|
||||||
|
MinInsyncReplicas int16
|
||||||
|
}
|
||||||
|
|
||||||
|
// CreateTopic creates a topic idempotently: if it already exists with the
|
||||||
|
// requested partition count, this is a no-op (design.md §3 task 3
|
||||||
|
// acceptance criteria).
|
||||||
|
func (a *Admin) CreateTopic(ctx context.Context, topic string, cfg TopicConfig) error {
|
||||||
|
configs := map[string]*string{
|
||||||
|
"retention.ms": strPtr(strconv.FormatInt(int64(cfg.RetentionSeconds)*1000, 10)),
|
||||||
|
"cleanup.policy": strPtr("delete"),
|
||||||
|
"min.insync.replicas": strPtr(strconv.Itoa(int(cfg.MinInsyncReplicas))),
|
||||||
|
}
|
||||||
|
|
||||||
|
resp, err := a.client.CreateTopics(ctx, cfg.PartitionCount, cfg.ReplicationFactor, configs, topic)
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("create topic %q: %w", topic, err)
|
||||||
|
}
|
||||||
|
for _, t := range resp {
|
||||||
|
if t.Err != nil && !isTopicExistsErr(t.Err) {
|
||||||
|
return fmt.Errorf("create topic %q: %w", topic, t.Err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// DeleteTopic deletes a topic. Deleting a non-existent topic is not an error.
|
||||||
|
func (a *Admin) DeleteTopic(ctx context.Context, topic string) error {
|
||||||
|
resp, err := a.client.DeleteTopics(ctx, topic)
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("delete topic %q: %w", topic, err)
|
||||||
|
}
|
||||||
|
for _, t := range resp {
|
||||||
|
if t.Err != nil && !isUnknownTopicErr(t.Err) {
|
||||||
|
return fmt.Errorf("delete topic %q: %w", topic, t.Err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// DescribeTopic returns the live partition count and config for a topic, or
|
||||||
|
// ok=false if it does not exist.
|
||||||
|
func (a *Admin) DescribeTopic(ctx context.Context, topic string) (partitions int32, ok bool, err error) {
|
||||||
|
td, err := a.client.ListTopics(ctx, topic)
|
||||||
|
if err != nil {
|
||||||
|
return 0, false, fmt.Errorf("describe topic %q: %w", topic, err)
|
||||||
|
}
|
||||||
|
detail, found := td[topic]
|
||||||
|
if !found || detail.Err != nil {
|
||||||
|
return 0, false, nil
|
||||||
|
}
|
||||||
|
return int32(len(detail.Partitions)), true, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// LogEndOffsetSum returns the sum of the high-watermark offsets across every
|
||||||
|
// partition of a topic — a monotonically increasing proxy for total records
|
||||||
|
// produced, used by the queue-operator's shard-split sampler (design.md §2c)
|
||||||
|
// to estimate throughput between two reconcile ticks.
|
||||||
|
func (a *Admin) LogEndOffsetSum(ctx context.Context, topic string) (int64, error) {
|
||||||
|
offsets, err := a.client.ListEndOffsets(ctx, topic)
|
||||||
|
if err != nil {
|
||||||
|
return 0, fmt.Errorf("list end offsets %q: %w", topic, err)
|
||||||
|
}
|
||||||
|
var sum int64
|
||||||
|
for _, partitions := range offsets {
|
||||||
|
for _, o := range partitions {
|
||||||
|
if o.Err != nil {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
sum += o.Offset
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return sum, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// ConsumerLag returns the total lag (sum across partitions) of a consumer
|
||||||
|
// group against a topic, used by the queue-operator to decide when a
|
||||||
|
// `Closing` shard (design.md §2c) is fully drained.
|
||||||
|
func (a *Admin) ConsumerLag(ctx context.Context, group, topic string) (int64, error) {
|
||||||
|
lags, err := a.client.Lag(ctx, group)
|
||||||
|
if err != nil {
|
||||||
|
return 0, fmt.Errorf("lag for group %q: %w", group, err)
|
||||||
|
}
|
||||||
|
described, ok := lags[group]
|
||||||
|
if !ok || described.Error() != nil {
|
||||||
|
return 0, nil
|
||||||
|
}
|
||||||
|
return described.Lag.TotalByTopic()[topic].Lag, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// CommitOffset commits the next offset to read for (group, topic, partition)
|
||||||
|
// to Kafka's real consumer-group offsets — the message plane calls this on
|
||||||
|
// ack to advance the low-watermark commit described in design.md §3, rather
|
||||||
|
// than relying on the consumer client's own interval autocommit (which is
|
||||||
|
// disabled, see internal/kafka.Consumer).
|
||||||
|
func (a *Admin) CommitOffset(ctx context.Context, group, topic string, partition int32, offset int64) error {
|
||||||
|
offsets := kadm.Offsets{topic: {partition: kadm.Offset{Topic: topic, Partition: partition, At: offset}}}
|
||||||
|
resp, err := a.client.CommitOffsets(ctx, group, offsets)
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("commit offset %s/%s/%d: %w", group, topic, partition, err)
|
||||||
|
}
|
||||||
|
if err := resp.Error(); err != nil {
|
||||||
|
return fmt.Errorf("commit offset %s/%s/%d: %w", group, topic, partition, err)
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func (a *Admin) Close() {
|
||||||
|
a.client.Close()
|
||||||
|
}
|
||||||
|
|
||||||
|
func strPtr(s string) *string { return &s }
|
||||||
|
|
||||||
|
func isTopicExistsErr(err error) bool {
|
||||||
|
return err != nil && (err.Error() == "TOPIC_ALREADY_EXISTS" || containsCode(err, "TOPIC_ALREADY_EXISTS"))
|
||||||
|
}
|
||||||
|
|
||||||
|
func isUnknownTopicErr(err error) bool {
|
||||||
|
return err != nil && containsCode(err, "UNKNOWN_TOPIC_OR_PARTITION")
|
||||||
|
}
|
||||||
|
|
||||||
|
func containsCode(err error, code string) bool {
|
||||||
|
return err != nil && (err.Error() == code || len(err.Error()) >= len(code) && (err.Error()[:len(code)] == code))
|
||||||
|
}
|
||||||
@@ -0,0 +1,37 @@
|
|||||||
|
package kafka
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
|
||||||
|
"github.com/twmb/franz-go/pkg/kgo"
|
||||||
|
)
|
||||||
|
|
||||||
|
// NewProducerClient builds a producer-only client. Callers own Close().
|
||||||
|
func NewProducerClient(brokers []string) (*kgo.Client, error) {
|
||||||
|
cl, err := kgo.NewClient(
|
||||||
|
kgo.SeedBrokers(brokers...),
|
||||||
|
kgo.ProducerBatchCompression(kgo.SnappyCompression()),
|
||||||
|
)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("creating kafka producer client: %w", err)
|
||||||
|
}
|
||||||
|
return cl, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// NewConsumerClient builds a client that consumes the given topics as part
|
||||||
|
// of consumerGroup. The message-plane service relies on Kafka's native
|
||||||
|
// consumer-group rebalancing for partition ownership across replicas
|
||||||
|
// (design.md §9) — offset commit is driven manually via the low-watermark
|
||||||
|
// strategy in design.md §3, so auto-commit is disabled.
|
||||||
|
func NewConsumerClient(brokers []string, consumerGroup string, topics ...string) (*kgo.Client, error) {
|
||||||
|
cl, err := kgo.NewClient(
|
||||||
|
kgo.SeedBrokers(brokers...),
|
||||||
|
kgo.ConsumerGroup(consumerGroup),
|
||||||
|
kgo.ConsumeTopics(topics...),
|
||||||
|
kgo.DisableAutoCommit(),
|
||||||
|
)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("creating kafka consumer client: %w", err)
|
||||||
|
}
|
||||||
|
return cl, nil
|
||||||
|
}
|
||||||
@@ -0,0 +1,77 @@
|
|||||||
|
package kafka
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"fmt"
|
||||||
|
|
||||||
|
"github.com/twmb/franz-go/pkg/kgo"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Record is a fetched Kafka record relevant to the message plane.
|
||||||
|
type Record struct {
|
||||||
|
Topic string
|
||||||
|
Partition int32
|
||||||
|
Offset int64
|
||||||
|
Key []byte
|
||||||
|
Value []byte
|
||||||
|
}
|
||||||
|
|
||||||
|
// Consumer wraps a kgo.Client consuming a queue's shard topics under the
|
||||||
|
// shared consumer group (design.md §3 ConsumerGroup). Autocommit is
|
||||||
|
// disabled: offset commits are driven by the watermark logic in
|
||||||
|
// internal/core/queue, not by the client's own interval commit, so a
|
||||||
|
// crash between receive and ack never advances the committed offset past an
|
||||||
|
// unacked message (design.md §3's at-least-once guarantee).
|
||||||
|
type Consumer struct {
|
||||||
|
client *kgo.Client
|
||||||
|
}
|
||||||
|
|
||||||
|
// NewConsumer subscribes to the given shard topics as a real consumer-group
|
||||||
|
// member: Kafka's group-rebalance protocol handles partition ownership
|
||||||
|
// across replicas (design.md §9), and the queue-operator's drain check
|
||||||
|
// (internal/kafka.Admin.ConsumerLag) reads this same group's committed
|
||||||
|
// offsets to know when a `Closing` shard is safe to delete.
|
||||||
|
func NewConsumer(brokers []string, group string, topics ...string) (*Consumer, error) {
|
||||||
|
cl, err := kgo.NewClient(
|
||||||
|
kgo.SeedBrokers(brokers...),
|
||||||
|
kgo.ConsumerGroup(group),
|
||||||
|
kgo.ConsumeTopics(topics...),
|
||||||
|
kgo.DisableAutoCommit(),
|
||||||
|
)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("creating kafka consumer client: %w", err)
|
||||||
|
}
|
||||||
|
return &Consumer{client: cl}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// AddTopics subscribes to additional shard topics created by a split,
|
||||||
|
// without losing the group membership/offsets already held for existing
|
||||||
|
// topics.
|
||||||
|
func (c *Consumer) AddTopics(topics ...string) {
|
||||||
|
c.client.AddConsumeTopics(topics...)
|
||||||
|
}
|
||||||
|
|
||||||
|
func (c *Consumer) Close() {
|
||||||
|
c.client.Close()
|
||||||
|
}
|
||||||
|
|
||||||
|
// Poll runs one non-blocking fetch iteration and returns whatever records
|
||||||
|
// were immediately available. The long-poll wait loop lives in
|
||||||
|
// internal/core/queue.ReceiveMessageService, not here.
|
||||||
|
func (c *Consumer) Poll(ctx context.Context) ([]Record, error) {
|
||||||
|
fetches := c.client.PollFetches(ctx)
|
||||||
|
if errs := fetches.Errors(); len(errs) > 0 {
|
||||||
|
return nil, fmt.Errorf("poll fetches: %w", errs[0].Err)
|
||||||
|
}
|
||||||
|
var out []Record
|
||||||
|
fetches.EachRecord(func(r *kgo.Record) {
|
||||||
|
out = append(out, Record{
|
||||||
|
Topic: r.Topic,
|
||||||
|
Partition: r.Partition,
|
||||||
|
Offset: r.Offset,
|
||||||
|
Key: r.Key,
|
||||||
|
Value: r.Value,
|
||||||
|
})
|
||||||
|
})
|
||||||
|
return out, nil
|
||||||
|
}
|
||||||
@@ -0,0 +1,41 @@
|
|||||||
|
package kafka
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"fmt"
|
||||||
|
|
||||||
|
"github.com/twmb/franz-go/pkg/kgo"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Producer writes to manually-partitioned shard topics: the caller has
|
||||||
|
// already computed the destination partition via PartitionWithinShard, so
|
||||||
|
// the client must be configured with a manual partitioner that respects
|
||||||
|
// Record.Partition rather than re-hashing the key itself.
|
||||||
|
type Producer struct {
|
||||||
|
client *kgo.Client
|
||||||
|
}
|
||||||
|
|
||||||
|
func NewProducer(brokers []string) (*Producer, error) {
|
||||||
|
cl, err := kgo.NewClient(
|
||||||
|
kgo.SeedBrokers(brokers...),
|
||||||
|
kgo.RecordPartitioner(kgo.ManualPartitioner()),
|
||||||
|
)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("creating kafka producer client: %w", err)
|
||||||
|
}
|
||||||
|
return &Producer{client: cl}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func (p *Producer) Close() {
|
||||||
|
p.client.Close()
|
||||||
|
}
|
||||||
|
|
||||||
|
// Produce synchronously writes one record to topic/partition and returns
|
||||||
|
// the offset it landed at.
|
||||||
|
func (p *Producer) Produce(ctx context.Context, topic string, partition int32, key, value []byte) (int64, error) {
|
||||||
|
rec := &kgo.Record{Topic: topic, Partition: partition, Key: key, Value: value}
|
||||||
|
if err := p.client.ProduceSync(ctx, rec).FirstErr(); err != nil {
|
||||||
|
return 0, fmt.Errorf("producing to %s/%d: %w", topic, partition, err)
|
||||||
|
}
|
||||||
|
return rec.Offset, nil
|
||||||
|
}
|
||||||
@@ -0,0 +1,178 @@
|
|||||||
|
// Package kafka implements the queue-name <-> Kafka-topic mapping and
|
||||||
|
// topic admin operations described in design.md §6.
|
||||||
|
package kafka
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
)
|
||||||
|
|
||||||
|
const (
|
||||||
|
topicPrefix = "kmsvc."
|
||||||
|
fifoSuffix = ".fifo"
|
||||||
|
dlqSuffix = ".dlq"
|
||||||
|
shardInfix = ".shard-"
|
||||||
|
DefaultPartitionCount = 6
|
||||||
|
|
||||||
|
// FullHashRangeEnd is the exclusive upper bound of the 32-bit shard key
|
||||||
|
// space; a queue's first shard owns [0, FullHashRangeEnd).
|
||||||
|
FullHashRangeEnd uint32 = 0xFFFFFFFF
|
||||||
|
)
|
||||||
|
|
||||||
|
// ShardTopicName returns the Kafka topic name for one shard of a queue, per
|
||||||
|
// design.md §6: kmsvc.{queueName}.shard-{id} (standard),
|
||||||
|
// kmsvc.{queueName}.fifo.shard-{id} (FIFO).
|
||||||
|
func ShardTopicName(queueName string, fifo bool, shardID string) string {
|
||||||
|
base := topicPrefix + queueName
|
||||||
|
if fifo {
|
||||||
|
base += fifoSuffix
|
||||||
|
}
|
||||||
|
return base + shardInfix + shardID
|
||||||
|
}
|
||||||
|
|
||||||
|
// DLQShardTopicName returns the DLQ topic name for one shard of a queue's
|
||||||
|
// DLQ, per design.md §6: kmsvc.{queueName}.dlq.shard-{id} / .fifo.dlq.shard-{id}.
|
||||||
|
func DLQShardTopicName(queueName string, fifo bool, shardID string) string {
|
||||||
|
base := topicPrefix + queueName
|
||||||
|
if fifo {
|
||||||
|
base += fifoSuffix
|
||||||
|
}
|
||||||
|
return base + dlqSuffix + shardInfix + shardID
|
||||||
|
}
|
||||||
|
|
||||||
|
// ConsumerGroup returns the single Kafka consumer-group name shared by every
|
||||||
|
// message-plane replica consuming a queue's shards, used both for normal
|
||||||
|
// consumption and by the queue-operator's drain check (design.md §2c, §9).
|
||||||
|
func ConsumerGroup(queueName string) string {
|
||||||
|
return "kmsvc-consumer-" + queueName
|
||||||
|
}
|
||||||
|
|
||||||
|
// murmur2 mirrors Kafka's default partitioner hash (murmur2), used so that
|
||||||
|
// FIFO partition assignment here matches what a native Kafka producer would
|
||||||
|
// compute for the same key, per design.md §6.
|
||||||
|
func murmur2(data []byte) uint32 {
|
||||||
|
const (
|
||||||
|
seed uint32 = 0x9747b28c
|
||||||
|
m uint32 = 0x5bd1e995
|
||||||
|
r = 24
|
||||||
|
)
|
||||||
|
length := len(data)
|
||||||
|
h := seed ^ uint32(length)
|
||||||
|
four := length / 4
|
||||||
|
|
||||||
|
for i := 0; i < four; i++ {
|
||||||
|
i4 := i * 4
|
||||||
|
k := uint32(data[i4]&0xff) |
|
||||||
|
(uint32(data[i4+1]&0xff) << 8) |
|
||||||
|
(uint32(data[i4+2]&0xff) << 16) |
|
||||||
|
(uint32(data[i4+3]&0xff) << 24)
|
||||||
|
k *= m
|
||||||
|
k ^= k >> r
|
||||||
|
k *= m
|
||||||
|
h *= m
|
||||||
|
h ^= k
|
||||||
|
}
|
||||||
|
|
||||||
|
switch length & 3 {
|
||||||
|
case 3:
|
||||||
|
h ^= uint32(data[(length&^3)+2]&0xff) << 16
|
||||||
|
fallthrough
|
||||||
|
case 2:
|
||||||
|
h ^= uint32(data[(length&^3)+1]&0xff) << 8
|
||||||
|
fallthrough
|
||||||
|
case 1:
|
||||||
|
h ^= uint32(data[length&^3] & 0xff)
|
||||||
|
h *= m
|
||||||
|
}
|
||||||
|
|
||||||
|
h ^= h >> 13
|
||||||
|
h *= m
|
||||||
|
h ^= h >> 15
|
||||||
|
|
||||||
|
return h
|
||||||
|
}
|
||||||
|
|
||||||
|
// toPositive mirrors Kafka's Utils.toPositive, masking the sign bit so the
|
||||||
|
// result is usable as an unsigned partition index.
|
||||||
|
func toPositive(n uint32) uint32 {
|
||||||
|
return n & 0x7fffffff
|
||||||
|
}
|
||||||
|
|
||||||
|
// HashKey returns the murmur2 hash of a routing key (MessageGroupId for FIFO
|
||||||
|
// queues, a random UUID for standard queues) into the shard key space used
|
||||||
|
// for both shard selection and within-shard partitioning (design.md §2c, §6).
|
||||||
|
func HashKey(key string) uint32 {
|
||||||
|
return toPositive(murmur2([]byte(key)))
|
||||||
|
}
|
||||||
|
|
||||||
|
// Shard is the subset of a Queue's status.shards entry needed for routing.
|
||||||
|
// Phase mirrors apis/kmsvc/v1.ShardPhase as a plain string to avoid this
|
||||||
|
// package depending on the CRD API package.
|
||||||
|
type Shard struct {
|
||||||
|
ID string
|
||||||
|
Topic string
|
||||||
|
HashRangeStart uint32
|
||||||
|
HashRangeEnd uint32
|
||||||
|
Phase string
|
||||||
|
}
|
||||||
|
|
||||||
|
// ActiveShards filters to shards eligible to receive newly-sent messages
|
||||||
|
// (design.md §2c: a `Closing` shard keeps being consumed/drained but stops
|
||||||
|
// being a write target).
|
||||||
|
func ActiveShards(shards []Shard) []Shard {
|
||||||
|
active := make([]Shard, 0, len(shards))
|
||||||
|
for _, s := range shards {
|
||||||
|
if s.Phase == "" || s.Phase == "Active" {
|
||||||
|
active = append(active, s)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return active
|
||||||
|
}
|
||||||
|
|
||||||
|
// SelectShard returns the shard whose hash range contains routingKey's hash,
|
||||||
|
// per design.md §2c. Callers doing write-path routing should pass
|
||||||
|
// ActiveShards(shards) so messages never target a `Closing` shard; the
|
||||||
|
// shards passed in must cover the full key space with no gaps for this to
|
||||||
|
// always find a match.
|
||||||
|
func SelectShard(shards []Shard, routingKey string) (Shard, bool) {
|
||||||
|
h := HashKey(routingKey)
|
||||||
|
for _, s := range shards {
|
||||||
|
if h >= s.HashRangeStart && h < s.HashRangeEnd {
|
||||||
|
return s, true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return Shard{}, false
|
||||||
|
}
|
||||||
|
|
||||||
|
// SplitHashRange returns the midpoint of [start, end), the boundary between
|
||||||
|
// the two child shards created by a split (design.md §2c).
|
||||||
|
func SplitHashRange(start, end uint32) uint32 {
|
||||||
|
return start + (end-start)/2
|
||||||
|
}
|
||||||
|
|
||||||
|
// PartitionWithinShard returns the partition a message lands on within its
|
||||||
|
// shard's topic. For FIFO queues routingKey is the MessageGroupId, ensuring
|
||||||
|
// all messages for a group are ordered on one partition within that shard
|
||||||
|
// (design.md §3, §6); for standard queues routingKey is a random per-message
|
||||||
|
// value, so traffic just spreads evenly.
|
||||||
|
func PartitionWithinShard(routingKey string, partitionsPerShard int32) int32 {
|
||||||
|
if partitionsPerShard <= 0 {
|
||||||
|
partitionsPerShard = DefaultPartitionCount
|
||||||
|
}
|
||||||
|
return int32(HashKey(routingKey) % uint32(partitionsPerShard))
|
||||||
|
}
|
||||||
|
|
||||||
|
// ValidateNoDLQCycle enforces design.md §5's DLQ-loop guard: a queue's
|
||||||
|
// dead-letter target must not be itself, and a queue marked as a DLQ must
|
||||||
|
// not itself have a dead-letter target (no DLQ chains).
|
||||||
|
func ValidateNoDLQCycle(queueName string, isDLQ bool, deadLetterTarget string) error {
|
||||||
|
if deadLetterTarget == "" {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
if deadLetterTarget == queueName {
|
||||||
|
return fmt.Errorf("queue %q cannot set its own dead-letter target", queueName)
|
||||||
|
}
|
||||||
|
if isDLQ {
|
||||||
|
return fmt.Errorf("DLQ queue %q cannot itself have a dead-letter target (no DLQ chains)", queueName)
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
@@ -0,0 +1,127 @@
|
|||||||
|
package kafka
|
||||||
|
|
||||||
|
import "testing"
|
||||||
|
|
||||||
|
func TestShardTopicName(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
queue string
|
||||||
|
fifo bool
|
||||||
|
shard string
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
{"standard", "orders", false, "0", "kmsvc.orders.shard-0"},
|
||||||
|
{"fifo", "orders", true, "0", "kmsvc.orders.fifo.shard-0"},
|
||||||
|
{"split child", "orders", true, "2", "kmsvc.orders.fifo.shard-2"},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
t.Run(c.name, func(t *testing.T) {
|
||||||
|
if got := ShardTopicName(c.queue, c.fifo, c.shard); got != c.want {
|
||||||
|
t.Errorf("ShardTopicName(%q, %v, %q) = %q, want %q", c.queue, c.fifo, c.shard, got, c.want)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestDLQShardTopicName(t *testing.T) {
|
||||||
|
if got, want := DLQShardTopicName("orders", false, "0"), "kmsvc.orders.dlq.shard-0"; got != want {
|
||||||
|
t.Errorf("DLQShardTopicName = %q, want %q", got, want)
|
||||||
|
}
|
||||||
|
if got, want := DLQShardTopicName("orders", true, "0"), "kmsvc.orders.fifo.dlq.shard-0"; got != want {
|
||||||
|
t.Errorf("DLQShardTopicName(fifo) = %q, want %q", got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestPartitionWithinShardDeterministic(t *testing.T) {
|
||||||
|
const partitions = 6
|
||||||
|
p1 := PartitionWithinShard("group-a", partitions)
|
||||||
|
p2 := PartitionWithinShard("group-a", partitions)
|
||||||
|
if p1 != p2 {
|
||||||
|
t.Errorf("PartitionWithinShard not deterministic: %d != %d", p1, p2)
|
||||||
|
}
|
||||||
|
if p1 < 0 || p1 >= partitions {
|
||||||
|
t.Errorf("PartitionWithinShard out of range: %d", p1)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestPartitionWithinShardDistribution(t *testing.T) {
|
||||||
|
const partitions = 6
|
||||||
|
seen := map[int32]bool{}
|
||||||
|
for i := 0; i < 1000; i++ {
|
||||||
|
groupID := "group-" + string(rune('a'+i%26)) + string(rune('0'+i%10))
|
||||||
|
p := PartitionWithinShard(groupID, partitions)
|
||||||
|
seen[p] = true
|
||||||
|
}
|
||||||
|
if len(seen) < 2 {
|
||||||
|
t.Errorf("expected groups to spread across multiple partitions, got %d distinct partitions", len(seen))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSelectShard(t *testing.T) {
|
||||||
|
mid := SplitHashRange(0, FullHashRangeEnd)
|
||||||
|
shards := []Shard{
|
||||||
|
{ID: "1", Topic: "kmsvc.orders.fifo.shard-1", HashRangeStart: 0, HashRangeEnd: mid},
|
||||||
|
{ID: "2", Topic: "kmsvc.orders.fifo.shard-2", HashRangeStart: mid, HashRangeEnd: FullHashRangeEnd},
|
||||||
|
}
|
||||||
|
for i := 0; i < 1000; i++ {
|
||||||
|
key := "group-" + string(rune('a'+i%26)) + string(rune('0'+i%10))
|
||||||
|
s, ok := SelectShard(shards, key)
|
||||||
|
if !ok {
|
||||||
|
t.Fatalf("no shard found for key %q", key)
|
||||||
|
}
|
||||||
|
h := HashKey(key)
|
||||||
|
if h < s.HashRangeStart || h >= s.HashRangeEnd {
|
||||||
|
t.Errorf("key %q hash %d outside selected shard range [%d, %d)", key, h, s.HashRangeStart, s.HashRangeEnd)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSelectShardStableAcrossSplit(t *testing.T) {
|
||||||
|
parent := []Shard{{ID: "0", Topic: "kmsvc.orders.fifo.shard-0", HashRangeStart: 0, HashRangeEnd: FullHashRangeEnd}}
|
||||||
|
mid := SplitHashRange(0, FullHashRangeEnd)
|
||||||
|
children := []Shard{
|
||||||
|
{ID: "1", Topic: "kmsvc.orders.fifo.shard-1", HashRangeStart: 0, HashRangeEnd: mid},
|
||||||
|
{ID: "2", Topic: "kmsvc.orders.fifo.shard-2", HashRangeStart: mid, HashRangeEnd: FullHashRangeEnd},
|
||||||
|
}
|
||||||
|
for i := 0; i < 1000; i++ {
|
||||||
|
key := "group-" + string(rune('a'+i%26)) + string(rune('0'+i%10))
|
||||||
|
before, _ := SelectShard(parent, key)
|
||||||
|
after, ok := SelectShard(children, key)
|
||||||
|
if !ok {
|
||||||
|
t.Fatalf("no child shard found for key %q", key)
|
||||||
|
}
|
||||||
|
h := HashKey(key)
|
||||||
|
wantParent := h < mid
|
||||||
|
gotParentSide := after.ID == "1"
|
||||||
|
if wantParent != gotParentSide {
|
||||||
|
t.Errorf("key %q (hash %d) routed to wrong child after split", key, h)
|
||||||
|
}
|
||||||
|
_ = before
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestValidateNoDLQCycle(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
queue string
|
||||||
|
isDLQ bool
|
||||||
|
target string
|
||||||
|
expectErr bool
|
||||||
|
}{
|
||||||
|
{"no target", "orders", false, "", false},
|
||||||
|
{"valid target", "orders", false, "orders-dlq", false},
|
||||||
|
{"self reference", "orders", false, "orders", true},
|
||||||
|
{"dlq with target", "orders-dlq", true, "another-dlq", true},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
t.Run(c.name, func(t *testing.T) {
|
||||||
|
err := ValidateNoDLQCycle(c.queue, c.isDLQ, c.target)
|
||||||
|
if c.expectErr && err == nil {
|
||||||
|
t.Errorf("expected error, got nil")
|
||||||
|
}
|
||||||
|
if !c.expectErr && err != nil {
|
||||||
|
t.Errorf("expected no error, got %v", err)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
Reference in New Issue
Block a user