mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-10-02 20:53:39 +00:00
Compare commits
12
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
e4fe657bd9 | ||
|
|
31473fbe10 | ||
|
|
618235bd72 | ||
|
|
49e1e56aa1 | ||
|
|
ad6a4ef9b2 | ||
|
|
46602a229c | ||
|
|
b0290dc33d | ||
|
|
914810fd13 | ||
|
|
2c6ed03823 | ||
|
|
6a97079454 | ||
|
|
1c879148d8 | ||
|
|
0347212b64 |
@@ -0,0 +1,38 @@
|
||||
FROM golang:1.24-alpine
|
||||
|
||||
# Install necessary tools
|
||||
RUN apk add --no-cache \
|
||||
curl \
|
||||
netcat-openbsd \
|
||||
bash \
|
||||
git \
|
||||
build-base \
|
||||
ca-certificates
|
||||
|
||||
# Set working directory
|
||||
WORKDIR /workspace
|
||||
|
||||
# Copy go mod files first for better caching
|
||||
COPY go.mod go.sum ./
|
||||
RUN go mod download
|
||||
|
||||
# Copy only the necessary source code for building and testing
|
||||
COPY weed/ ./weed/
|
||||
COPY test/mq/integration/ ./test/mq/integration/
|
||||
|
||||
# Build the weed binary for testing (optional - tests can use external cluster)
|
||||
RUN cd weed && CGO_ENABLED=1 GOOS=linux go build -o ../weed .
|
||||
|
||||
# Create test results directory
|
||||
RUN mkdir -p /test-results
|
||||
|
||||
# Set up environment
|
||||
ENV CGO_ENABLED=1
|
||||
ENV GOOS=linux
|
||||
ENV GO111MODULE=on
|
||||
|
||||
# Default working directory for tests
|
||||
WORKDIR /workspace
|
||||
|
||||
# Entry point for running tests - use explicit bash
|
||||
ENTRYPOINT ["/bin/bash", "-c"]
|
||||
@@ -0,0 +1,223 @@
|
||||
.PHONY: help build test test-basic test-performance test-failover test-agent clean up down logs
|
||||
|
||||
# Detect architecture and Docker platform compatibility
|
||||
ARCH := $(shell uname -m)
|
||||
OS := $(shell uname -s)
|
||||
ifeq ($(ARCH),arm64)
|
||||
ifeq ($(OS),Darwin)
|
||||
# On Apple Silicon macOS, use native arm64 for better performance
|
||||
DOCKER_PLATFORM := linux/arm64
|
||||
else
|
||||
DOCKER_PLATFORM := linux/arm64
|
||||
endif
|
||||
else
|
||||
DOCKER_PLATFORM := linux/amd64
|
||||
endif
|
||||
|
||||
# Default target
|
||||
help:
|
||||
@echo "SeaweedMQ Integration Test Suite"
|
||||
@echo ""
|
||||
@echo "Available targets:"
|
||||
@echo " build - Build SeaweedFS Docker images"
|
||||
@echo " test - Run all integration tests (in Docker)"
|
||||
@echo " test-basic - Run basic pub/sub tests (in Docker)"
|
||||
@echo " test-native - Run all tests natively (no Docker test container)"
|
||||
@echo " test-basic-native - Run basic tests natively (recommended for Apple Silicon)"
|
||||
@echo " test-performance - Run performance tests"
|
||||
@echo " test-failover - Run failover tests"
|
||||
@echo " test-agent - Run agent tests"
|
||||
@echo " up - Start test environment (local build)"
|
||||
@echo " up-prod - Start test environment (production images)"
|
||||
@echo " up-cluster - Start cluster only (no test runner)"
|
||||
@echo " down - Stop test environment"
|
||||
@echo " clean - Clean up test environment and results"
|
||||
@echo " logs - Show container logs"
|
||||
|
||||
# Build SeaweedFS Docker images
|
||||
build:
|
||||
@echo "Building SeaweedFS Docker image for $(DOCKER_PLATFORM)..."
|
||||
cd ../.. && docker build --platform $(DOCKER_PLATFORM) -f docker/Dockerfile.go_build -t chrislusf/seaweedfs:local .
|
||||
@echo "Building test runner image for $(DOCKER_PLATFORM)..."
|
||||
cd ../.. && docker build --platform $(DOCKER_PLATFORM) -f test/mq/Dockerfile.test -t seaweedfs-test-runner .
|
||||
|
||||
# Start the test environment
|
||||
up: build
|
||||
@echo "Starting SeaweedMQ test environment..."
|
||||
docker-compose -f docker-compose.test.yml up -d master0 master1 master2
|
||||
@echo "Waiting for masters to be ready..."
|
||||
sleep 10
|
||||
docker-compose -f docker-compose.test.yml up -d volume1 volume2 volume3
|
||||
@echo "Waiting for volumes to be ready..."
|
||||
sleep 10
|
||||
docker-compose -f docker-compose.test.yml up -d filer1 filer2
|
||||
@echo "Waiting for filers to be ready..."
|
||||
sleep 15
|
||||
docker-compose -f docker-compose.test.yml up -d broker1 broker2 broker3
|
||||
@echo "Waiting for brokers to be ready..."
|
||||
sleep 20
|
||||
@echo "Test environment is ready!"
|
||||
|
||||
# Start the test environment with production images (no build required)
|
||||
up-prod: build-test-runner
|
||||
@echo "Starting SeaweedMQ test environment with production images..."
|
||||
docker-compose -f docker-compose.production.yml up -d master0 master1 master2
|
||||
@echo "Waiting for masters to be ready..."
|
||||
sleep 10
|
||||
docker-compose -f docker-compose.production.yml up -d volume1 volume2 volume3
|
||||
@echo "Waiting for volumes to be ready..."
|
||||
sleep 10
|
||||
docker-compose -f docker-compose.production.yml up -d filer1 filer2
|
||||
@echo "Waiting for filers to be ready..."
|
||||
sleep 15
|
||||
docker-compose -f docker-compose.production.yml up -d broker1 broker2 broker3
|
||||
@echo "Waiting for brokers to be ready..."
|
||||
sleep 20
|
||||
@echo "Test environment is ready!"
|
||||
|
||||
# Build only the test runner image (for production setup)
|
||||
build-test-runner:
|
||||
@echo "Building test runner image for $(DOCKER_PLATFORM)..."
|
||||
cd ../.. && docker build --platform $(DOCKER_PLATFORM) -f test/mq/Dockerfile.test -t seaweedfs-test-runner .
|
||||
|
||||
# Start cluster only (no test runner, no build required)
|
||||
up-cluster:
|
||||
@echo "Starting SeaweedMQ cluster only..."
|
||||
docker-compose -f docker-compose.cluster.yml up -d master0 master1 master2
|
||||
@echo "Waiting for masters to be ready..."
|
||||
sleep 10
|
||||
docker-compose -f docker-compose.cluster.yml up -d volume1 volume2 volume3
|
||||
@echo "Waiting for volumes to be ready..."
|
||||
sleep 10
|
||||
docker-compose -f docker-compose.cluster.yml up -d filer1 filer2
|
||||
@echo "Waiting for filers to be ready..."
|
||||
sleep 15
|
||||
docker-compose -f docker-compose.cluster.yml up -d broker1 broker2 broker3
|
||||
@echo "Waiting for brokers to be ready..."
|
||||
sleep 20
|
||||
@echo "SeaweedMQ cluster is ready!"
|
||||
@echo "Masters: http://localhost:19333, http://localhost:19334, http://localhost:19335"
|
||||
@echo "Filers: http://localhost:18888, http://localhost:18889"
|
||||
@echo "Brokers: localhost:17777, localhost:17778, localhost:17779"
|
||||
|
||||
# Stop the test environment
|
||||
down:
|
||||
@echo "Stopping SeaweedMQ test environment..."
|
||||
docker-compose -f docker-compose.test.yml down
|
||||
docker-compose -f docker-compose.production.yml down
|
||||
docker-compose -f docker-compose.cluster.yml down
|
||||
|
||||
# Clean up everything
|
||||
clean:
|
||||
@echo "Cleaning up test environment..."
|
||||
docker-compose -f docker-compose.test.yml down -v
|
||||
docker system prune -f
|
||||
sudo rm -rf /tmp/test-results/*
|
||||
|
||||
# Show container logs
|
||||
logs:
|
||||
docker-compose -f docker-compose.test.yml logs -f
|
||||
|
||||
# Run all integration tests
|
||||
test:
|
||||
@echo "Running all integration tests..."
|
||||
docker-compose -f docker-compose.test.yml run --rm test-runner \
|
||||
sh -c "go test -v -timeout=30m ./test/mq/integration/... -args -test.parallel=4"
|
||||
|
||||
# Run basic pub/sub tests
|
||||
test-basic:
|
||||
@echo "Running basic pub/sub tests natively (no container restart)..."
|
||||
cd ../.. && SEAWEED_MASTERS="localhost:19333,localhost:19334,localhost:19335" \
|
||||
SEAWEED_BROKERS="localhost:17777,localhost:17778,localhost:17779" \
|
||||
SEAWEED_FILERS="localhost:18888,localhost:18889" \
|
||||
go test -v -timeout=10m ./test/mq/integration/ -run TestBasic
|
||||
|
||||
# Run performance tests
|
||||
test-performance:
|
||||
@echo "Running performance tests..."
|
||||
docker-compose -f docker-compose.test.yml run --rm test-runner \
|
||||
sh -c "go test -v -timeout=20m ./test/mq/integration/ -run TestPerformance"
|
||||
|
||||
# Run failover tests
|
||||
test-failover:
|
||||
@echo "Running failover tests..."
|
||||
docker-compose -f docker-compose.test.yml run --rm test-runner \
|
||||
sh -c "go test -v -timeout=15m ./test/mq/integration/ -run TestFailover"
|
||||
|
||||
# Run agent tests
|
||||
test-agent:
|
||||
@echo "Running agent tests..."
|
||||
docker-compose -f docker-compose.test.yml run --rm test-runner \
|
||||
sh -c "go test -v -timeout=10m ./test/mq/integration/ -run TestAgent"
|
||||
|
||||
# Development targets (run tests natively without Docker container)
|
||||
test-dev:
|
||||
@echo "Running tests in development mode (using local binaries)..."
|
||||
SEAWEED_MASTERS="localhost:19333,localhost:19334,localhost:19335" \
|
||||
SEAWEED_BROKERS="localhost:17777,localhost:17778,localhost:17779" \
|
||||
SEAWEED_FILERS="localhost:18888,localhost:18889" \
|
||||
go test -v -timeout=10m ./integration/...
|
||||
|
||||
# Native test running (no Docker container for tests)
|
||||
test-native:
|
||||
@echo "Running tests natively (without Docker container for tests)..."
|
||||
cd ../.. && SEAWEED_MASTERS="localhost:19333,localhost:19334,localhost:19335" \
|
||||
SEAWEED_BROKERS="localhost:17777,localhost:17778,localhost:17779" \
|
||||
SEAWEED_FILERS="localhost:18888,localhost:18889" \
|
||||
go test -v -timeout=10m ./test/mq/integration/...
|
||||
|
||||
# Basic native tests
|
||||
test-basic-native:
|
||||
@echo "Running basic tests natively..."
|
||||
cd ../.. && SEAWEED_MASTERS="localhost:19333,localhost:19334,localhost:19335" \
|
||||
SEAWEED_BROKERS="localhost:17777,localhost:17778,localhost:17779" \
|
||||
SEAWEED_FILERS="localhost:18888,localhost:18889" \
|
||||
go test -v -timeout=10m ./test/mq/integration/ -run TestBasic
|
||||
|
||||
# Quick smoke test
|
||||
smoke-test:
|
||||
@echo "Running smoke test..."
|
||||
docker-compose -f docker-compose.test.yml run --rm test-runner \
|
||||
sh -c "go test -v -timeout=5m ./test/mq/integration/ -run TestBasicPublishSubscribe"
|
||||
|
||||
# Performance benchmarks
|
||||
benchmark:
|
||||
@echo "Running performance benchmarks..."
|
||||
docker-compose -f docker-compose.test.yml run --rm test-runner \
|
||||
sh -c "go test -v -timeout=30m -bench=. ./test/mq/integration/..."
|
||||
|
||||
# Check test environment health
|
||||
health:
|
||||
@echo "Checking test environment health..."
|
||||
@echo "Masters:"
|
||||
@curl -s http://localhost:19333/cluster/status || echo "Master 0 not accessible"
|
||||
@curl -s http://localhost:19334/cluster/status || echo "Master 1 not accessible"
|
||||
@curl -s http://localhost:19335/cluster/status || echo "Master 2 not accessible"
|
||||
@echo ""
|
||||
@echo "Filers:"
|
||||
@curl -s http://localhost:18888/ || echo "Filer 1 not accessible"
|
||||
@curl -s http://localhost:18889/ || echo "Filer 2 not accessible"
|
||||
@echo ""
|
||||
@echo "Brokers:"
|
||||
@nc -z localhost 17777 && echo "Broker 1 accessible" || echo "Broker 1 not accessible"
|
||||
@nc -z localhost 17778 && echo "Broker 2 accessible" || echo "Broker 2 not accessible"
|
||||
@nc -z localhost 17779 && echo "Broker 3 accessible" || echo "Broker 3 not accessible"
|
||||
|
||||
# Generate test reports
|
||||
report:
|
||||
@echo "Generating test reports..."
|
||||
docker-compose -f docker-compose.test.yml run --rm test-runner \
|
||||
sh -c "go test -v -timeout=30m ./test/mq/integration/... -json > /test-results/test-report.json"
|
||||
|
||||
# Load testing
|
||||
load-test:
|
||||
@echo "Running load tests..."
|
||||
docker-compose -f docker-compose.test.yml run --rm test-runner \
|
||||
sh -c "go test -v -timeout=45m ./test/mq/integration/ -run TestLoad"
|
||||
|
||||
# View monitoring dashboards
|
||||
monitoring:
|
||||
@echo "Starting monitoring stack..."
|
||||
docker-compose -f docker-compose.test.yml up -d prometheus grafana
|
||||
@echo "Prometheus: http://localhost:19090"
|
||||
@echo "Grafana: http://localhost:13000 (admin/admin)"
|
||||
@@ -0,0 +1,414 @@
|
||||
# SeaweedMQ Integration Test Suite
|
||||
|
||||
This directory contains a comprehensive integration test suite for SeaweedMQ, designed to validate all critical functionalities from basic pub/sub operations to advanced features like auto-scaling, failover, and performance testing.
|
||||
|
||||
## Overview
|
||||
|
||||
The integration test suite provides:
|
||||
|
||||
- **Automated Environment Setup**: Docker Compose based test clusters
|
||||
- **Comprehensive Test Coverage**: Basic pub/sub, scaling, failover, performance
|
||||
- **Monitoring & Metrics**: Prometheus and Grafana integration
|
||||
- **CI/CD Ready**: Configurable for continuous integration pipelines
|
||||
- **Load Testing**: Performance benchmarks and stress tests
|
||||
|
||||
## Architecture
|
||||
|
||||
The test environment consists of:
|
||||
|
||||
```
|
||||
┌─────────────────┐ ┌─────────────────┐ ┌─────────────────┐
|
||||
│ Master Cluster │ │ Volume Servers │ │ Filer Cluster │
|
||||
│ (3 nodes) │ │ (3 nodes) │ │ (2 nodes) │
|
||||
└─────────────────┘ └─────────────────┘ └─────────────────┘
|
||||
│ │ │
|
||||
└───────────────────────┼───────────────────────┘
|
||||
│
|
||||
┌─────────────────┐ ┌─────────────────┐ ┌─────────────────┐
|
||||
│ Broker Cluster │ │ Test Framework │ │ Monitoring │
|
||||
│ (3 nodes) │ │ (Go Tests) │ │ (Prometheus + │
|
||||
└─────────────────┘ └─────────────────┘ │ Grafana) │
|
||||
└─────────────────┘
|
||||
```
|
||||
|
||||
## Quick Start
|
||||
|
||||
### Prerequisites
|
||||
|
||||
- Docker and Docker Compose
|
||||
- Go 1.21+
|
||||
- Make
|
||||
- 8GB+ RAM recommended
|
||||
- 20GB+ disk space for test data
|
||||
|
||||
### Basic Usage
|
||||
|
||||
Choose one of three approaches:
|
||||
|
||||
#### Option 1: Quick Cluster Setup (Fastest)
|
||||
Just starts a SeaweedMQ cluster - no test runner, no build required.
|
||||
|
||||
1. **Start Cluster**:
|
||||
```bash
|
||||
cd test/mq
|
||||
make up-cluster
|
||||
```
|
||||
|
||||
2. **Test manually** or with your own code against:
|
||||
- Masters: http://localhost:19333, http://localhost:19334, http://localhost:19335
|
||||
- Brokers: localhost:17777, localhost:17778, localhost:17779
|
||||
- Filers: http://localhost:18888, http://localhost:18889
|
||||
|
||||
#### Option 2: Use Production Images (Recommended for testing)
|
||||
Uses official SeaweedFS images from Docker Hub with test runner.
|
||||
|
||||
1. **Start Test Environment**:
|
||||
```bash
|
||||
cd test/mq
|
||||
make up-prod
|
||||
```
|
||||
|
||||
2. **Run Tests**:
|
||||
```bash
|
||||
make test-basic # Basic pub/sub tests
|
||||
```
|
||||
|
||||
#### Option 3: Build from Source (Development)
|
||||
Builds SeaweedFS from your local source code to test latest changes.
|
||||
|
||||
1. **Start Test Environment**:
|
||||
```bash
|
||||
cd test/mq
|
||||
make up # This will build and start
|
||||
```
|
||||
|
||||
2. **Run Tests**:
|
||||
```bash
|
||||
make test-basic # Basic pub/sub tests
|
||||
```
|
||||
|
||||
#### Common Commands
|
||||
|
||||
3. **Run All Tests**:
|
||||
```bash
|
||||
make test
|
||||
```
|
||||
|
||||
4. **Run Specific Test Categories**:
|
||||
```bash
|
||||
make test-performance # Performance tests
|
||||
make test-failover # Failover tests
|
||||
make test-agent # Agent tests
|
||||
```
|
||||
|
||||
5. **Quick Smoke Test**:
|
||||
```bash
|
||||
make smoke-test
|
||||
```
|
||||
|
||||
6. **Check Health**:
|
||||
```bash
|
||||
make health
|
||||
```
|
||||
|
||||
7. **Clean Up**:
|
||||
```bash
|
||||
make down
|
||||
```
|
||||
|
||||
## Test Categories
|
||||
|
||||
### 1. Basic Functionality Tests
|
||||
|
||||
**File**: `integration/basic_pubsub_test.go`
|
||||
|
||||
- **TestBasicPublishSubscribe**: Basic message publishing and consumption
|
||||
- **TestMultipleConsumers**: Load balancing across multiple consumers
|
||||
- **TestMessageOrdering**: FIFO ordering within partitions
|
||||
- **TestSchemaValidation**: Schema validation and complex nested structures
|
||||
|
||||
### 2. Partitioning and Scaling Tests
|
||||
|
||||
**File**: `integration/scaling_test.go` (to be implemented)
|
||||
|
||||
- **TestPartitionDistribution**: Message distribution across partitions
|
||||
- **TestAutoSplitMerge**: Automatic partition split/merge based on load
|
||||
- **TestBrokerScaling**: Adding/removing brokers during operation
|
||||
- **TestLoadBalancing**: Even load distribution verification
|
||||
|
||||
### 3. Failover and Reliability Tests
|
||||
|
||||
**File**: `integration/failover_test.go` (to be implemented)
|
||||
|
||||
- **TestBrokerFailover**: Leader failover scenarios
|
||||
- **TestBrokerRecovery**: Recovery from broker failures
|
||||
- **TestMessagePersistence**: Data durability across restarts
|
||||
- **TestFollowerReplication**: Leader-follower consistency
|
||||
|
||||
### 4. Performance Tests
|
||||
|
||||
**File**: `integration/performance_test.go` (to be implemented)
|
||||
|
||||
- **TestHighThroughputPublish**: High-volume message publishing
|
||||
- **TestHighThroughputSubscribe**: High-volume message consumption
|
||||
- **TestLatencyMeasurement**: End-to-end latency analysis
|
||||
- **TestResourceUtilization**: CPU, memory, and disk usage
|
||||
|
||||
### 5. Agent Tests
|
||||
|
||||
**File**: `integration/agent_test.go` (to be implemented)
|
||||
|
||||
- **TestAgentPublishSessions**: Session management for publishers
|
||||
- **TestAgentSubscribeSessions**: Session management for subscribers
|
||||
- **TestAgentFailover**: Agent reconnection and failover
|
||||
- **TestAgentConcurrency**: Concurrent session handling
|
||||
|
||||
## Configuration
|
||||
|
||||
### Environment Variables
|
||||
|
||||
The test framework supports configuration via environment variables:
|
||||
|
||||
```bash
|
||||
# Cluster endpoints
|
||||
SEAWEED_MASTERS="master0:9333,master1:9334,master2:9335"
|
||||
SEAWEED_BROKERS="broker1:17777,broker2:17778,broker3:17779"
|
||||
SEAWEED_FILERS="filer1:8888,filer2:8889"
|
||||
|
||||
# Test configuration
|
||||
GO_TEST_TIMEOUT="30m"
|
||||
TEST_RESULTS_DIR="/test-results"
|
||||
```
|
||||
|
||||
### Docker Compose Override
|
||||
|
||||
Create `docker-compose.override.yml` to customize the test environment:
|
||||
|
||||
```yaml
|
||||
version: '3.9'
|
||||
services:
|
||||
broker1:
|
||||
environment:
|
||||
- CUSTOM_ENV_VAR=value
|
||||
test-runner:
|
||||
volumes:
|
||||
- ./custom-config:/config
|
||||
```
|
||||
|
||||
## Monitoring and Metrics
|
||||
|
||||
### Prometheus Metrics
|
||||
|
||||
Access Prometheus at: http://localhost:19090
|
||||
|
||||
Key metrics to monitor:
|
||||
- Message throughput: `seaweedmq_messages_published_total`
|
||||
- Consumer lag: `seaweedmq_consumer_lag_seconds`
|
||||
- Broker health: `seaweedmq_broker_health`
|
||||
- Resource usage: `seaweedfs_disk_usage_bytes`
|
||||
|
||||
### Grafana Dashboards
|
||||
|
||||
Access Grafana at: http://localhost:13000 (admin/admin)
|
||||
|
||||
Pre-configured dashboards:
|
||||
- **SeaweedMQ Overview**: System health and throughput
|
||||
- **Performance Metrics**: Latency and resource usage
|
||||
- **Error Analysis**: Error rates and failure patterns
|
||||
|
||||
## Development
|
||||
|
||||
### Writing New Tests
|
||||
|
||||
1. **Create Test File**:
|
||||
```bash
|
||||
touch integration/my_new_test.go
|
||||
```
|
||||
|
||||
2. **Use Test Framework**:
|
||||
```go
|
||||
func TestMyFeature(t *testing.T) {
|
||||
suite := NewIntegrationTestSuite(t)
|
||||
require.NoError(t, suite.Setup())
|
||||
|
||||
// Your test logic here
|
||||
}
|
||||
```
|
||||
|
||||
3. **Run Specific Test**:
|
||||
```bash
|
||||
go test -v ./integration/ -run TestMyFeature
|
||||
```
|
||||
|
||||
### Test Framework Components
|
||||
|
||||
**IntegrationTestSuite**: Base test framework with cluster management
|
||||
**MessageCollector**: Utility for collecting and verifying received messages
|
||||
**TestMessage**: Standard message structure for testing
|
||||
**Schema Builders**: Helpers for creating test schemas
|
||||
|
||||
### Local Development
|
||||
|
||||
Run tests against a local SeaweedMQ cluster:
|
||||
|
||||
```bash
|
||||
make test-dev
|
||||
```
|
||||
|
||||
This uses local binaries instead of Docker containers.
|
||||
|
||||
## Continuous Integration
|
||||
|
||||
### GitHub Actions Example
|
||||
|
||||
```yaml
|
||||
name: Integration Tests
|
||||
on: [push, pull_request]
|
||||
|
||||
jobs:
|
||||
integration-tests:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v3
|
||||
- uses: actions/setup-go@v3
|
||||
with:
|
||||
go-version: 1.21
|
||||
- name: Run Integration Tests
|
||||
run: |
|
||||
cd test/mq
|
||||
make test
|
||||
```
|
||||
|
||||
### Jenkins Pipeline
|
||||
|
||||
```groovy
|
||||
pipeline {
|
||||
agent any
|
||||
stages {
|
||||
stage('Setup') {
|
||||
steps {
|
||||
sh 'cd test/mq && make up'
|
||||
}
|
||||
}
|
||||
stage('Test') {
|
||||
steps {
|
||||
sh 'cd test/mq && make test'
|
||||
}
|
||||
post {
|
||||
always {
|
||||
sh 'cd test/mq && make down'
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
### Common Issues
|
||||
|
||||
1. **Port Conflicts**:
|
||||
```bash
|
||||
# Check port usage
|
||||
netstat -tulpn | grep :19333
|
||||
|
||||
# Kill conflicting processes
|
||||
sudo kill -9 $(lsof -t -i:19333)
|
||||
```
|
||||
|
||||
2. **Docker Resource Issues**:
|
||||
```bash
|
||||
# Increase Docker memory (8GB+)
|
||||
# Clean up Docker resources
|
||||
docker system prune -a
|
||||
```
|
||||
|
||||
3. **Test Timeouts**:
|
||||
```bash
|
||||
# Increase timeout
|
||||
GO_TEST_TIMEOUT=60m make test
|
||||
```
|
||||
|
||||
### Debug Mode
|
||||
|
||||
Run tests with verbose logging:
|
||||
|
||||
```bash
|
||||
docker-compose -f docker-compose.test.yml run --rm test-runner \
|
||||
sh -c "go test -v -race ./test/mq/integration/... -args -test.v"
|
||||
```
|
||||
|
||||
### Container Logs
|
||||
|
||||
View real-time logs:
|
||||
|
||||
```bash
|
||||
make logs
|
||||
|
||||
# Or specific service
|
||||
docker-compose -f docker-compose.test.yml logs -f broker1
|
||||
```
|
||||
|
||||
## Performance Benchmarks
|
||||
|
||||
### Throughput Benchmarks
|
||||
|
||||
```bash
|
||||
make benchmark
|
||||
```
|
||||
|
||||
Expected performance (on 8-core, 16GB RAM):
|
||||
- **Publish Throughput**: 50K+ messages/second/broker
|
||||
- **Subscribe Throughput**: 100K+ messages/second/broker
|
||||
- **End-to-End Latency**: P95 < 100ms
|
||||
- **Storage Efficiency**: < 20% overhead
|
||||
|
||||
### Load Testing
|
||||
|
||||
```bash
|
||||
make load-test
|
||||
```
|
||||
|
||||
Stress tests with:
|
||||
- 1M+ messages
|
||||
- 100+ concurrent producers
|
||||
- 50+ concurrent consumers
|
||||
- Multiple topic scenarios
|
||||
|
||||
## Contributing
|
||||
|
||||
### Test Guidelines
|
||||
|
||||
1. **Test Isolation**: Each test should be independent
|
||||
2. **Resource Cleanup**: Always clean up resources in test teardown
|
||||
3. **Timeouts**: Set appropriate timeouts for operations
|
||||
4. **Error Handling**: Test both success and failure scenarios
|
||||
5. **Documentation**: Document test purpose and expected behavior
|
||||
|
||||
### Code Style
|
||||
|
||||
- Follow Go testing conventions
|
||||
- Use testify for assertions
|
||||
- Include setup/teardown in test functions
|
||||
- Use descriptive test names
|
||||
|
||||
## Future Enhancements
|
||||
|
||||
- [ ] Chaos engineering tests (network partitions, node failures)
|
||||
- [ ] Multi-datacenter deployment testing
|
||||
- [ ] Schema evolution compatibility tests
|
||||
- [ ] Security and authentication tests
|
||||
- [ ] Performance regression detection
|
||||
- [ ] Automated load pattern generation
|
||||
|
||||
## Support
|
||||
|
||||
For issues and questions:
|
||||
- Check existing GitHub issues
|
||||
- Review SeaweedMQ documentation
|
||||
- Join SeaweedFS community discussions
|
||||
|
||||
---
|
||||
|
||||
*This integration test suite ensures SeaweedMQ's reliability, performance, and functionality across all critical use cases and failure scenarios.*
|
||||
@@ -0,0 +1,165 @@
|
||||
services:
|
||||
# Masters
|
||||
master0:
|
||||
image: chrislusf/seaweedfs:latest
|
||||
ports:
|
||||
- "19333:9333"
|
||||
volumes:
|
||||
- /tmp/seaweedfs-test/master0:/data
|
||||
command: "master -port=9333 -mdir=/data -peers=master0:9333,master1:9334,master2:9335 -ip=master0 -defaultReplication=001"
|
||||
networks:
|
||||
- seaweedfs-test
|
||||
healthcheck:
|
||||
test: ["CMD", "wget", "-q", "--spider", "http://master0:9333/cluster/status"]
|
||||
interval: 30s
|
||||
timeout: 10s
|
||||
retries: 5
|
||||
start_period: 30s
|
||||
|
||||
master1:
|
||||
image: chrislusf/seaweedfs:latest
|
||||
ports:
|
||||
- "19334:9334"
|
||||
volumes:
|
||||
- /tmp/seaweedfs-test/master1:/data
|
||||
command: "master -port=9334 -mdir=/data -peers=master0:9333,master1:9334,master2:9335 -ip=master1 -defaultReplication=001"
|
||||
networks:
|
||||
- seaweedfs-test
|
||||
healthcheck:
|
||||
test: ["CMD", "wget", "-q", "--spider", "http://master1:9334/cluster/status"]
|
||||
interval: 30s
|
||||
timeout: 10s
|
||||
retries: 5
|
||||
start_period: 30s
|
||||
|
||||
master2:
|
||||
image: chrislusf/seaweedfs:latest
|
||||
ports:
|
||||
- "19335:9335"
|
||||
volumes:
|
||||
- /tmp/seaweedfs-test/master2:/data
|
||||
command: "master -port=9335 -mdir=/data -peers=master0:9333,master1:9334,master2:9335 -ip=master2 -defaultReplication=001"
|
||||
networks:
|
||||
- seaweedfs-test
|
||||
healthcheck:
|
||||
test: ["CMD", "wget", "-q", "--spider", "http://master2:9335/cluster/status"]
|
||||
interval: 30s
|
||||
timeout: 10s
|
||||
retries: 5
|
||||
start_period: 30s
|
||||
|
||||
# Volume Servers
|
||||
volume1:
|
||||
image: chrislusf/seaweedfs:latest
|
||||
ports:
|
||||
- "18080:8080"
|
||||
volumes:
|
||||
- /tmp/seaweedfs-test/volume1:/data
|
||||
command: "volume -port=8080 -mserver=master0:9333,master1:9334,master2:9335 -dir=/data"
|
||||
depends_on:
|
||||
- master0
|
||||
- master1
|
||||
- master2
|
||||
networks:
|
||||
- seaweedfs-test
|
||||
|
||||
volume2:
|
||||
image: chrislusf/seaweedfs:latest
|
||||
ports:
|
||||
- "18081:8081"
|
||||
volumes:
|
||||
- /tmp/seaweedfs-test/volume2:/data
|
||||
command: "volume -port=8081 -mserver=master0:9333,master1:9334,master2:9335 -dir=/data"
|
||||
depends_on:
|
||||
- master0
|
||||
- master1
|
||||
- master2
|
||||
networks:
|
||||
- seaweedfs-test
|
||||
|
||||
volume3:
|
||||
image: chrislusf/seaweedfs:latest
|
||||
ports:
|
||||
- "18082:8082"
|
||||
volumes:
|
||||
- /tmp/seaweedfs-test/volume3:/data
|
||||
command: "volume -port=8082 -mserver=master0:9333,master1:9334,master2:9335 -dir=/data"
|
||||
depends_on:
|
||||
- master0
|
||||
- master1
|
||||
- master2
|
||||
networks:
|
||||
- seaweedfs-test
|
||||
|
||||
# Filers
|
||||
filer1:
|
||||
image: chrislusf/seaweedfs:latest
|
||||
ports:
|
||||
- "18888:8888"
|
||||
volumes:
|
||||
- /tmp/seaweedfs-test/filer1:/data
|
||||
command: "filer -port=8888 -master=master0:9333,master1:9334,master2:9335"
|
||||
depends_on:
|
||||
- volume1
|
||||
- volume2
|
||||
- volume3
|
||||
networks:
|
||||
- seaweedfs-test
|
||||
|
||||
filer2:
|
||||
image: chrislusf/seaweedfs:latest
|
||||
ports:
|
||||
- "18889:8889"
|
||||
volumes:
|
||||
- /tmp/seaweedfs-test/filer2:/data
|
||||
command: "filer -port=8889 -master=master0:9333,master1:9334,master2:9335"
|
||||
depends_on:
|
||||
- volume1
|
||||
- volume2
|
||||
- volume3
|
||||
networks:
|
||||
- seaweedfs-test
|
||||
|
||||
# Message Queue Brokers
|
||||
broker1:
|
||||
image: chrislusf/seaweedfs:latest
|
||||
ports:
|
||||
- "17777:17777"
|
||||
volumes:
|
||||
- /tmp/seaweedfs-test/broker1:/data
|
||||
command: "mq.broker -port=17777 -master=master0:9333,master1:9334,master2:9335"
|
||||
depends_on:
|
||||
- filer1
|
||||
- filer2
|
||||
networks:
|
||||
- seaweedfs-test
|
||||
|
||||
broker2:
|
||||
image: chrislusf/seaweedfs:latest
|
||||
ports:
|
||||
- "17778:17778"
|
||||
volumes:
|
||||
- /tmp/seaweedfs-test/broker2:/data
|
||||
command: "mq.broker -port=17778 -master=master0:9333,master1:9334,master2:9335"
|
||||
depends_on:
|
||||
- filer1
|
||||
- filer2
|
||||
networks:
|
||||
- seaweedfs-test
|
||||
|
||||
broker3:
|
||||
image: chrislusf/seaweedfs:latest
|
||||
ports:
|
||||
- "17779:17779"
|
||||
volumes:
|
||||
- /tmp/seaweedfs-test/broker3:/data
|
||||
command: "mq.broker -port=17779 -master=master0:9333,master1:9334,master2:9335"
|
||||
depends_on:
|
||||
- filer1
|
||||
- filer2
|
||||
networks:
|
||||
- seaweedfs-test
|
||||
|
||||
networks:
|
||||
seaweedfs-test:
|
||||
driver: bridge
|
||||
@@ -0,0 +1,260 @@
|
||||
services:
|
||||
# Masters
|
||||
master0:
|
||||
image: chrislusf/seaweedfs:latest
|
||||
ports:
|
||||
- "19333:19333"
|
||||
- "9333:9333"
|
||||
volumes:
|
||||
- /tmp/seaweedfs-test/master0:/data
|
||||
command: "master -port=9333 -mdir=/data -peers=master0:9333,master1:9334,master2:9335 -defaultReplication=001"
|
||||
environment:
|
||||
WEED_MASTER_VOLUME_GROWTH_COPY_1: 7
|
||||
WEED_MASTER_VOLUME_GROWTH_COPY_2: 6
|
||||
WEED_MASTER_VOLUME_GROWTH_COPY_OTHER: 3
|
||||
networks:
|
||||
- seaweedfs-test
|
||||
healthcheck:
|
||||
test: ["CMD", "wget", "-q", "--spider", "http://master0:9333/cluster/status"]
|
||||
interval: 10s
|
||||
timeout: 5s
|
||||
retries: 3
|
||||
|
||||
master1:
|
||||
image: chrislusf/seaweedfs:latest
|
||||
ports:
|
||||
- "19334:9334"
|
||||
- "9334:9334"
|
||||
volumes:
|
||||
- /tmp/seaweedfs-test/master1:/data
|
||||
command: "master -port=9334 -mdir=/data -peers=master0:9333,master1:9334,master2:9335 -defaultReplication=001"
|
||||
environment:
|
||||
WEED_MASTER_VOLUME_GROWTH_COPY_1: 7
|
||||
WEED_MASTER_VOLUME_GROWTH_COPY_2: 6
|
||||
WEED_MASTER_VOLUME_GROWTH_COPY_OTHER: 3
|
||||
networks:
|
||||
- seaweedfs-test
|
||||
healthcheck:
|
||||
test: ["CMD", "wget", "-q", "--spider", "http://master1:9334/cluster/status"]
|
||||
interval: 10s
|
||||
timeout: 5s
|
||||
retries: 3
|
||||
|
||||
master2:
|
||||
image: chrislusf/seaweedfs:latest
|
||||
ports:
|
||||
- "19335:9335"
|
||||
- "9335:9335"
|
||||
volumes:
|
||||
- /tmp/seaweedfs-test/master2:/data
|
||||
command: "master -port=9335 -mdir=/data -peers=master0:9333,master1:9334,master2:9335 -defaultReplication=001"
|
||||
environment:
|
||||
WEED_MASTER_VOLUME_GROWTH_COPY_1: 7
|
||||
WEED_MASTER_VOLUME_GROWTH_COPY_2: 6
|
||||
WEED_MASTER_VOLUME_GROWTH_COPY_OTHER: 3
|
||||
networks:
|
||||
- seaweedfs-test
|
||||
healthcheck:
|
||||
test: ["CMD", "wget", "-q", "--spider", "http://master2:9335/cluster/status"]
|
||||
interval: 10s
|
||||
timeout: 5s
|
||||
retries: 3
|
||||
|
||||
# Volume Servers
|
||||
volume1:
|
||||
image: chrislusf/seaweedfs:latest
|
||||
ports:
|
||||
- "18080:8080"
|
||||
- "8080:8080"
|
||||
volumes:
|
||||
- /tmp/seaweedfs-test/volume1:/data
|
||||
command: "volume -port=8080 -mserver=master0:9333,master1:9334,master2:9335 -dir=/data"
|
||||
depends_on:
|
||||
master0:
|
||||
condition: service_healthy
|
||||
master1:
|
||||
condition: service_healthy
|
||||
master2:
|
||||
condition: service_healthy
|
||||
networks:
|
||||
- seaweedfs-test
|
||||
environment:
|
||||
WEED_VOLUME_MAX: 100
|
||||
|
||||
volume2:
|
||||
image: chrislusf/seaweedfs:latest
|
||||
ports:
|
||||
- "18081:8081"
|
||||
- "8081:8081"
|
||||
volumes:
|
||||
- /tmp/seaweedfs-test/volume2:/data
|
||||
command: "volume -port=8081 -mserver=master0:9333,master1:9334,master2:9335 -dir=/data"
|
||||
depends_on:
|
||||
master0:
|
||||
condition: service_healthy
|
||||
master1:
|
||||
condition: service_healthy
|
||||
master2:
|
||||
condition: service_healthy
|
||||
networks:
|
||||
- seaweedfs-test
|
||||
environment:
|
||||
WEED_VOLUME_MAX: 100
|
||||
|
||||
volume3:
|
||||
image: chrislusf/seaweedfs:latest
|
||||
ports:
|
||||
- "18082:8082"
|
||||
- "8082:8082"
|
||||
volumes:
|
||||
- /tmp/seaweedfs-test/volume3:/data
|
||||
command: "volume -port=8082 -mserver=master0:9333,master1:9334,master2:9335 -dir=/data"
|
||||
depends_on:
|
||||
master0:
|
||||
condition: service_healthy
|
||||
master1:
|
||||
condition: service_healthy
|
||||
master2:
|
||||
condition: service_healthy
|
||||
networks:
|
||||
- seaweedfs-test
|
||||
environment:
|
||||
WEED_VOLUME_MAX: 100
|
||||
|
||||
# Filers
|
||||
filer1:
|
||||
image: chrislusf/seaweedfs:latest
|
||||
ports:
|
||||
- "18888:8888"
|
||||
- "8888:8888"
|
||||
volumes:
|
||||
- /tmp/seaweedfs-test/filer1:/data
|
||||
command: "filer -port=8888 -master=master0:9333,master1:9334,master2:9335"
|
||||
depends_on:
|
||||
- volume1
|
||||
- volume2
|
||||
- volume3
|
||||
networks:
|
||||
- seaweedfs-test
|
||||
environment:
|
||||
WEED_LEVELDB2_DIR: "/data/filerldb2"
|
||||
|
||||
filer2:
|
||||
image: chrislusf/seaweedfs:latest
|
||||
ports:
|
||||
- "18889:8889"
|
||||
- "8889:8889"
|
||||
volumes:
|
||||
- /tmp/seaweedfs-test/filer2:/data
|
||||
command: "filer -port=8889 -master=master0:9333,master1:9334,master2:9335"
|
||||
depends_on:
|
||||
- volume1
|
||||
- volume2
|
||||
- volume3
|
||||
networks:
|
||||
- seaweedfs-test
|
||||
environment:
|
||||
WEED_LEVELDB2_DIR: "/data/filerldb2"
|
||||
|
||||
# Message Queue Brokers
|
||||
broker1:
|
||||
image: chrislusf/seaweedfs:latest
|
||||
ports:
|
||||
- "17777:17777"
|
||||
volumes:
|
||||
- /tmp/seaweedfs-test/broker1:/data
|
||||
command: "mq.broker -port=17777 -filer=filer1:8888,filer2:8889"
|
||||
depends_on:
|
||||
- filer1
|
||||
- filer2
|
||||
networks:
|
||||
- seaweedfs-test
|
||||
environment:
|
||||
WEED_MQ_BROKER_CPU_PERCENT: 80
|
||||
|
||||
broker2:
|
||||
image: chrislusf/seaweedfs:latest
|
||||
ports:
|
||||
- "17778:17778"
|
||||
volumes:
|
||||
- /tmp/seaweedfs-test/broker2:/data
|
||||
command: "mq.broker -port=17778 -filer=filer1:8888,filer2:8889"
|
||||
depends_on:
|
||||
- filer1
|
||||
- filer2
|
||||
networks:
|
||||
- seaweedfs-test
|
||||
environment:
|
||||
WEED_MQ_BROKER_CPU_PERCENT: 80
|
||||
|
||||
broker3:
|
||||
image: chrislusf/seaweedfs:latest
|
||||
ports:
|
||||
- "17779:17779"
|
||||
volumes:
|
||||
- /tmp/seaweedfs-test/broker3:/data
|
||||
command: "mq.broker -port=17779 -filer=filer1:8888,filer2:8889"
|
||||
depends_on:
|
||||
- filer1
|
||||
- filer2
|
||||
networks:
|
||||
- seaweedfs-test
|
||||
environment:
|
||||
WEED_MQ_BROKER_CPU_PERCENT: 80
|
||||
|
||||
# Test Runner
|
||||
test-runner:
|
||||
image: seaweedfs-test-runner
|
||||
volumes:
|
||||
- ../..:/workspace
|
||||
- /tmp/test-results:/test-results
|
||||
working_dir: /workspace
|
||||
networks:
|
||||
- seaweedfs-test
|
||||
environment:
|
||||
- SEAWEED_MASTERS=master0:9333,master1:9334,master2:9335
|
||||
- SEAWEED_BROKERS=broker1:17777,broker2:17778,broker3:17779
|
||||
- SEAWEED_FILERS=filer1:8888,filer2:8889
|
||||
- GO111MODULE=on
|
||||
- CGO_ENABLED=0
|
||||
depends_on:
|
||||
- broker1
|
||||
- broker2
|
||||
- broker3
|
||||
|
||||
# Monitoring
|
||||
prometheus:
|
||||
image: prom/prometheus:latest
|
||||
ports:
|
||||
- "19090:9090"
|
||||
volumes:
|
||||
- ./prometheus.yml:/etc/prometheus/prometheus.yml
|
||||
networks:
|
||||
- seaweedfs-test
|
||||
depends_on:
|
||||
- master0
|
||||
- master1
|
||||
- master2
|
||||
- volume1
|
||||
- volume2
|
||||
- volume3
|
||||
- filer1
|
||||
- filer2
|
||||
- broker1
|
||||
- broker2
|
||||
- broker3
|
||||
|
||||
grafana:
|
||||
image: grafana/grafana:latest
|
||||
ports:
|
||||
- "13000:3000"
|
||||
environment:
|
||||
- GF_SECURITY_ADMIN_PASSWORD=admin
|
||||
networks:
|
||||
- seaweedfs-test
|
||||
depends_on:
|
||||
- prometheus
|
||||
|
||||
networks:
|
||||
seaweedfs-test:
|
||||
driver: bridge
|
||||
@@ -0,0 +1,341 @@
|
||||
services:
|
||||
# Master cluster for coordination and metadata
|
||||
master0:
|
||||
image: chrislusf/seaweedfs:local
|
||||
container_name: test-master0
|
||||
ports:
|
||||
- "19333:9333"
|
||||
- "29333:19333"
|
||||
command: >
|
||||
-v=1
|
||||
master
|
||||
-volumeSizeLimitMB=100
|
||||
-resumeState=false
|
||||
-ip=master0
|
||||
-port=9333
|
||||
-peers=master0:9333,master1:9334,master2:9335
|
||||
-mdir=/tmp/master0
|
||||
environment:
|
||||
WEED_MASTER_VOLUME_GROWTH_COPY_1: 1
|
||||
WEED_MASTER_VOLUME_GROWTH_COPY_2: 2
|
||||
WEED_MASTER_VOLUME_GROWTH_COPY_OTHER: 1
|
||||
networks:
|
||||
- seaweedmq-test
|
||||
healthcheck:
|
||||
test: ["CMD", "wget", "-q", "--spider", "http://master0:9333/cluster/status"]
|
||||
interval: 10s
|
||||
timeout: 5s
|
||||
retries: 3
|
||||
|
||||
master1:
|
||||
image: chrislusf/seaweedfs:local
|
||||
container_name: test-master1
|
||||
ports:
|
||||
- "19334:9334"
|
||||
- "29334:19334"
|
||||
command: >
|
||||
-v=1
|
||||
master
|
||||
-volumeSizeLimitMB=100
|
||||
-resumeState=false
|
||||
-ip=master1
|
||||
-port=9334
|
||||
-peers=master0:9333,master1:9334,master2:9335
|
||||
-mdir=/tmp/master1
|
||||
environment:
|
||||
WEED_MASTER_VOLUME_GROWTH_COPY_1: 1
|
||||
WEED_MASTER_VOLUME_GROWTH_COPY_2: 2
|
||||
WEED_MASTER_VOLUME_GROWTH_COPY_OTHER: 1
|
||||
networks:
|
||||
- seaweedmq-test
|
||||
depends_on:
|
||||
- master0
|
||||
|
||||
master2:
|
||||
image: chrislusf/seaweedfs:local
|
||||
container_name: test-master2
|
||||
ports:
|
||||
- "19335:9335"
|
||||
- "29335:19335"
|
||||
command: >
|
||||
-v=1
|
||||
master
|
||||
-volumeSizeLimitMB=100
|
||||
-resumeState=false
|
||||
-ip=master2
|
||||
-port=9335
|
||||
-peers=master0:9333,master1:9334,master2:9335
|
||||
-mdir=/tmp/master2
|
||||
environment:
|
||||
WEED_MASTER_VOLUME_GROWTH_COPY_1: 1
|
||||
WEED_MASTER_VOLUME_GROWTH_COPY_2: 2
|
||||
WEED_MASTER_VOLUME_GROWTH_COPY_OTHER: 1
|
||||
networks:
|
||||
- seaweedmq-test
|
||||
depends_on:
|
||||
- master0
|
||||
|
||||
# Volume servers for data storage
|
||||
volume1:
|
||||
image: chrislusf/seaweedfs:local
|
||||
container_name: test-volume1
|
||||
ports:
|
||||
- "18080:8080"
|
||||
- "28080:18080"
|
||||
volumes:
|
||||
- volume1-data:/data/volume1
|
||||
entrypoint: ["/bin/sh", "-c"]
|
||||
command: >
|
||||
"mkdir -p /data/volume1 && exec weed -v=1
|
||||
volume
|
||||
-dataCenter=dc1
|
||||
-rack=rack1
|
||||
-mserver=master0:9333,master1:9334,master2:9335
|
||||
-port=8080
|
||||
-ip=volume1
|
||||
-publicUrl=localhost:18080
|
||||
-preStopSeconds=1
|
||||
-dir=/data/volume1"
|
||||
networks:
|
||||
- seaweedmq-test
|
||||
depends_on:
|
||||
master0:
|
||||
condition: service_healthy
|
||||
|
||||
volume2:
|
||||
image: chrislusf/seaweedfs:local
|
||||
container_name: test-volume2
|
||||
ports:
|
||||
- "18081:8081"
|
||||
- "28081:18081"
|
||||
volumes:
|
||||
- volume2-data:/data/volume2
|
||||
entrypoint: ["/bin/sh", "-c"]
|
||||
command: >
|
||||
"mkdir -p /data/volume2 && exec weed -v=1
|
||||
volume
|
||||
-dataCenter=dc1
|
||||
-rack=rack2
|
||||
-mserver=master0:9333,master1:9334,master2:9335
|
||||
-port=8081
|
||||
-ip=volume2
|
||||
-publicUrl=localhost:18081
|
||||
-preStopSeconds=1
|
||||
-dir=/data/volume2"
|
||||
networks:
|
||||
- seaweedmq-test
|
||||
depends_on:
|
||||
master0:
|
||||
condition: service_healthy
|
||||
|
||||
volume3:
|
||||
image: chrislusf/seaweedfs:local
|
||||
container_name: test-volume3
|
||||
ports:
|
||||
- "18082:8082"
|
||||
- "28082:18082"
|
||||
volumes:
|
||||
- volume3-data:/data/volume3
|
||||
entrypoint: ["/bin/sh", "-c"]
|
||||
command: >
|
||||
"mkdir -p /data/volume3 && exec weed -v=1
|
||||
volume
|
||||
-dataCenter=dc2
|
||||
-rack=rack1
|
||||
-mserver=master0:9333,master1:9334,master2:9335
|
||||
-port=8082
|
||||
-ip=volume3
|
||||
-publicUrl=localhost:18082
|
||||
-preStopSeconds=1
|
||||
-dir=/data/volume3"
|
||||
networks:
|
||||
- seaweedmq-test
|
||||
depends_on:
|
||||
master0:
|
||||
condition: service_healthy
|
||||
|
||||
# Filer servers for metadata
|
||||
filer1:
|
||||
image: chrislusf/seaweedfs:local
|
||||
container_name: test-filer1
|
||||
ports:
|
||||
- "18888:8888"
|
||||
- "28888:18888"
|
||||
command: >
|
||||
-v=1
|
||||
filer
|
||||
-defaultReplicaPlacement=100
|
||||
-master=master0:9333,master1:9334,master2:9335
|
||||
-port=8888
|
||||
-ip=filer1
|
||||
-dataCenter=dc1
|
||||
networks:
|
||||
- seaweedmq-test
|
||||
depends_on:
|
||||
master0:
|
||||
condition: service_healthy
|
||||
healthcheck:
|
||||
test: ["CMD", "wget", "-q", "--spider", "http://filer1:8888/"]
|
||||
interval: 10s
|
||||
timeout: 5s
|
||||
retries: 3
|
||||
|
||||
filer2:
|
||||
image: chrislusf/seaweedfs:local
|
||||
container_name: test-filer2
|
||||
ports:
|
||||
- "18889:8889"
|
||||
- "28889:18889"
|
||||
command: >
|
||||
-v=1
|
||||
filer
|
||||
-defaultReplicaPlacement=100
|
||||
-master=master0:9333,master1:9334,master2:9335
|
||||
-port=8889
|
||||
-ip=filer2
|
||||
-dataCenter=dc2
|
||||
networks:
|
||||
- seaweedmq-test
|
||||
depends_on:
|
||||
filer1:
|
||||
condition: service_healthy
|
||||
|
||||
# Message Queue Brokers
|
||||
broker1:
|
||||
image: chrislusf/seaweedfs:local
|
||||
container_name: test-broker1
|
||||
ports:
|
||||
- "17777:17777"
|
||||
command: >
|
||||
-v=1
|
||||
mq.broker
|
||||
-master=master0:9333,master1:9334,master2:9335
|
||||
-port=17777
|
||||
-ip=127.0.0.1
|
||||
-dataCenter=dc1
|
||||
-rack=rack1
|
||||
networks:
|
||||
- seaweedmq-test
|
||||
depends_on:
|
||||
filer1:
|
||||
condition: service_healthy
|
||||
healthcheck:
|
||||
test: ["CMD", "nc", "-z", "localhost", "17777"]
|
||||
interval: 10s
|
||||
timeout: 5s
|
||||
retries: 3
|
||||
|
||||
broker2:
|
||||
image: chrislusf/seaweedfs:local
|
||||
container_name: test-broker2
|
||||
ports:
|
||||
- "17778:17778"
|
||||
command: >
|
||||
-v=1
|
||||
mq.broker
|
||||
-master=master0:9333,master1:9334,master2:9335
|
||||
-port=17778
|
||||
-ip=127.0.0.1
|
||||
-dataCenter=dc1
|
||||
-rack=rack2
|
||||
networks:
|
||||
- seaweedmq-test
|
||||
depends_on:
|
||||
broker1:
|
||||
condition: service_healthy
|
||||
|
||||
broker3:
|
||||
image: chrislusf/seaweedfs:local
|
||||
container_name: test-broker3
|
||||
ports:
|
||||
- "17779:17779"
|
||||
command: >
|
||||
-v=1
|
||||
mq.broker
|
||||
-master=master0:9333,master1:9334,master2:9335
|
||||
-port=17779
|
||||
-ip=127.0.0.1
|
||||
-dataCenter=dc2
|
||||
-rack=rack1
|
||||
networks:
|
||||
- seaweedmq-test
|
||||
depends_on:
|
||||
broker1:
|
||||
condition: service_healthy
|
||||
|
||||
# Test runner container
|
||||
test-runner:
|
||||
build:
|
||||
context: ../../
|
||||
dockerfile: test/mq/Dockerfile.test
|
||||
container_name: test-runner
|
||||
volumes:
|
||||
- ../../:/app
|
||||
- /tmp/test-results:/test-results
|
||||
working_dir: /app
|
||||
environment:
|
||||
- SEAWEED_MASTERS=master0:9333,master1:9334,master2:9335
|
||||
- SEAWEED_BROKERS=broker1:17777,broker2:17778,broker3:17779
|
||||
- SEAWEED_FILERS=filer1:8888,filer2:8889
|
||||
- TEST_RESULTS_DIR=/test-results
|
||||
- GO_TEST_TIMEOUT=30m
|
||||
networks:
|
||||
- seaweedmq-test
|
||||
depends_on:
|
||||
broker1:
|
||||
condition: service_healthy
|
||||
broker2:
|
||||
condition: service_started
|
||||
broker3:
|
||||
condition: service_started
|
||||
command: >
|
||||
sh -c "
|
||||
echo 'Waiting for cluster to be ready...' &&
|
||||
sleep 30 &&
|
||||
echo 'Running integration tests...' &&
|
||||
go test -v -timeout=30m ./test/mq/integration/... -args -test.parallel=4
|
||||
"
|
||||
|
||||
# Monitoring and metrics
|
||||
prometheus:
|
||||
image: prom/prometheus:latest
|
||||
container_name: test-prometheus
|
||||
ports:
|
||||
- "19090:9090"
|
||||
volumes:
|
||||
- ./prometheus.yml:/etc/prometheus/prometheus.yml
|
||||
networks:
|
||||
- seaweedmq-test
|
||||
command:
|
||||
- '--config.file=/etc/prometheus/prometheus.yml'
|
||||
- '--storage.tsdb.path=/prometheus'
|
||||
- '--web.console.libraries=/etc/prometheus/console_libraries'
|
||||
- '--web.console.templates=/etc/prometheus/consoles'
|
||||
- '--web.enable-lifecycle'
|
||||
|
||||
grafana:
|
||||
image: grafana/grafana:latest
|
||||
container_name: test-grafana
|
||||
ports:
|
||||
- "13000:3000"
|
||||
environment:
|
||||
- GF_SECURITY_ADMIN_PASSWORD=admin
|
||||
volumes:
|
||||
- grafana-storage:/var/lib/grafana
|
||||
- ./grafana/dashboards:/etc/grafana/provisioning/dashboards
|
||||
- ./grafana/datasources:/etc/grafana/provisioning/datasources
|
||||
networks:
|
||||
- seaweedmq-test
|
||||
|
||||
networks:
|
||||
seaweedmq-test:
|
||||
driver: bridge
|
||||
ipam:
|
||||
config:
|
||||
- subnet: 172.20.0.0/16
|
||||
|
||||
volumes:
|
||||
grafana-storage:
|
||||
volume1-data:
|
||||
volume2-data:
|
||||
volume3-data:
|
||||
@@ -0,0 +1,339 @@
|
||||
package integration
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/mq/schema"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/mq_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/schema_pb"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
func TestBasicPublishSubscribe(t *testing.T) {
|
||||
suite := NewIntegrationTestSuite(t)
|
||||
require.NoError(t, suite.Setup())
|
||||
|
||||
namespace := "test"
|
||||
topicName := fmt.Sprintf("basic-pubsub-%d", time.Now().UnixNano()) // Unique topic name per run
|
||||
testSchema := CreateTestSchema()
|
||||
messageCount := 10
|
||||
|
||||
// Create publisher
|
||||
pubConfig := &PublisherTestConfig{
|
||||
Namespace: namespace,
|
||||
TopicName: topicName,
|
||||
PartitionCount: 1,
|
||||
PublisherName: "basic-publisher",
|
||||
RecordType: testSchema,
|
||||
}
|
||||
|
||||
publisher, err := suite.CreatePublisher(pubConfig)
|
||||
require.NoError(t, err)
|
||||
|
||||
// Create subscriber
|
||||
subConfig := &SubscriberTestConfig{
|
||||
Namespace: namespace,
|
||||
TopicName: topicName,
|
||||
ConsumerGroup: "test-group",
|
||||
ConsumerInstanceId: "consumer-1",
|
||||
MaxPartitionCount: 1,
|
||||
SlidingWindowSize: 10,
|
||||
OffsetType: schema_pb.OffsetType_RESET_TO_EARLIEST,
|
||||
}
|
||||
|
||||
subscriber, err := suite.CreateSubscriber(subConfig)
|
||||
require.NoError(t, err, "Failed to create subscriber")
|
||||
|
||||
// Set up message collector
|
||||
collector := NewMessageCollector(messageCount)
|
||||
subscriber.SetOnDataMessageFn(func(m *mq_pb.SubscribeMessageResponse_Data) {
|
||||
t.Logf("[Subscriber] Received message with key: %s, ts: %d", string(m.Data.Key), m.Data.TsNs)
|
||||
collector.AddMessage(TestMessage{
|
||||
ID: fmt.Sprintf("msg-%d", len(collector.GetMessages())),
|
||||
Content: m.Data.Value,
|
||||
Timestamp: time.Unix(0, m.Data.TsNs),
|
||||
Key: m.Data.Key,
|
||||
})
|
||||
})
|
||||
|
||||
// Start subscriber
|
||||
go func() {
|
||||
err := subscriber.Subscribe()
|
||||
if err != nil {
|
||||
t.Logf("Subscriber error: %v", err)
|
||||
}
|
||||
}()
|
||||
|
||||
// Wait for subscriber to be ready
|
||||
t.Logf("[Test] Waiting for subscriber to be ready...")
|
||||
time.Sleep(2 * time.Second)
|
||||
|
||||
// Publish test messages
|
||||
for i := 0; i < messageCount; i++ {
|
||||
record := schema.RecordBegin().
|
||||
SetString("id", fmt.Sprintf("msg-%d", i)).
|
||||
SetInt64("timestamp", time.Now().UnixNano()).
|
||||
SetString("content", fmt.Sprintf("Test message %d", i)).
|
||||
SetInt32("sequence", int32(i)).
|
||||
RecordEnd()
|
||||
|
||||
key := []byte(fmt.Sprintf("key-%d", i))
|
||||
t.Logf("[Publisher] Publishing message %d with key: %s", i, string(key))
|
||||
err := publisher.PublishRecord(key, record)
|
||||
require.NoError(t, err, "Failed to publish message %d", i)
|
||||
}
|
||||
|
||||
t.Logf("[Test] Waiting for messages to be received...")
|
||||
messages := collector.WaitForMessages(30 * time.Second)
|
||||
t.Logf("[Test] WaitForMessages returned. Received %d messages.", len(messages))
|
||||
|
||||
// Verify all messages were received
|
||||
assert.Len(t, messages, messageCount, "Expected %d messages, got %d", messageCount, len(messages))
|
||||
|
||||
// Verify message content
|
||||
for i, msg := range messages {
|
||||
assert.NotEmpty(t, msg.Content, "Message %d should have content", i)
|
||||
assert.NotEmpty(t, msg.Key, "Message %d should have key", i)
|
||||
}
|
||||
|
||||
t.Logf("[Test] TestBasicPublishSubscribe completed.")
|
||||
}
|
||||
|
||||
func TestMultipleConsumers(t *testing.T) {
|
||||
suite := NewIntegrationTestSuite(t)
|
||||
require.NoError(t, suite.Setup())
|
||||
|
||||
namespace := "test"
|
||||
topicName := "multi-consumer"
|
||||
testSchema := CreateTestSchema()
|
||||
messageCount := 20
|
||||
consumerCount := 3
|
||||
|
||||
// Create publisher
|
||||
pubConfig := &PublisherTestConfig{
|
||||
Namespace: namespace,
|
||||
TopicName: topicName,
|
||||
PartitionCount: 3, // Multiple partitions for load distribution
|
||||
PublisherName: "multi-publisher",
|
||||
RecordType: testSchema,
|
||||
}
|
||||
|
||||
publisher, err := suite.CreatePublisher(pubConfig)
|
||||
require.NoError(t, err)
|
||||
|
||||
// Create multiple consumers
|
||||
collectors := make([]*MessageCollector, consumerCount)
|
||||
for i := 0; i < consumerCount; i++ {
|
||||
collectors[i] = NewMessageCollector(messageCount / consumerCount) // Expect roughly equal distribution
|
||||
|
||||
subConfig := &SubscriberTestConfig{
|
||||
Namespace: namespace,
|
||||
TopicName: topicName,
|
||||
ConsumerGroup: "multi-consumer-group", // Same group for load balancing
|
||||
ConsumerInstanceId: fmt.Sprintf("consumer-%d", i),
|
||||
MaxPartitionCount: 1,
|
||||
SlidingWindowSize: 10,
|
||||
OffsetType: schema_pb.OffsetType_RESET_TO_EARLIEST,
|
||||
}
|
||||
|
||||
subscriber, err := suite.CreateSubscriber(subConfig)
|
||||
require.NoError(t, err)
|
||||
|
||||
// Set up message collection for this consumer
|
||||
collectorIndex := i
|
||||
subscriber.SetOnDataMessageFn(func(m *mq_pb.SubscribeMessageResponse_Data) {
|
||||
collectors[collectorIndex].AddMessage(TestMessage{
|
||||
ID: fmt.Sprintf("consumer-%d-msg-%d", collectorIndex, len(collectors[collectorIndex].GetMessages())),
|
||||
Content: m.Data.Value,
|
||||
Timestamp: time.Unix(0, m.Data.TsNs),
|
||||
Key: m.Data.Key,
|
||||
})
|
||||
})
|
||||
|
||||
// Start subscriber
|
||||
go func() {
|
||||
subscriber.Subscribe()
|
||||
}()
|
||||
}
|
||||
|
||||
// Wait for subscribers to be ready
|
||||
time.Sleep(3 * time.Second)
|
||||
|
||||
// Publish messages with different keys to distribute across partitions
|
||||
for i := 0; i < messageCount; i++ {
|
||||
record := schema.RecordBegin().
|
||||
SetString("id", fmt.Sprintf("multi-msg-%d", i)).
|
||||
SetInt64("timestamp", time.Now().UnixNano()).
|
||||
SetString("content", fmt.Sprintf("Multi consumer test message %d", i)).
|
||||
SetInt32("sequence", int32(i)).
|
||||
RecordEnd()
|
||||
|
||||
key := []byte(fmt.Sprintf("partition-key-%d", i%3)) // Distribute across 3 partitions
|
||||
err := publisher.PublishRecord(key, record)
|
||||
require.NoError(t, err)
|
||||
}
|
||||
|
||||
// Wait for all messages to be consumed
|
||||
time.Sleep(10 * time.Second)
|
||||
|
||||
// Verify message distribution
|
||||
totalReceived := 0
|
||||
for i, collector := range collectors {
|
||||
messages := collector.GetMessages()
|
||||
t.Logf("Consumer %d received %d messages", i, len(messages))
|
||||
totalReceived += len(messages)
|
||||
}
|
||||
|
||||
// All messages should be consumed across all consumers
|
||||
assert.Equal(t, messageCount, totalReceived, "Total messages received should equal messages sent")
|
||||
}
|
||||
|
||||
func TestMessageOrdering(t *testing.T) {
|
||||
suite := NewIntegrationTestSuite(t)
|
||||
require.NoError(t, suite.Setup())
|
||||
|
||||
namespace := "test"
|
||||
topicName := "ordering-test"
|
||||
testSchema := CreateTestSchema()
|
||||
messageCount := 15
|
||||
|
||||
// Create publisher
|
||||
pubConfig := &PublisherTestConfig{
|
||||
Namespace: namespace,
|
||||
TopicName: topicName,
|
||||
PartitionCount: 1, // Single partition to guarantee ordering
|
||||
PublisherName: "ordering-publisher",
|
||||
RecordType: testSchema,
|
||||
}
|
||||
|
||||
publisher, err := suite.CreatePublisher(pubConfig)
|
||||
require.NoError(t, err)
|
||||
|
||||
// Create subscriber
|
||||
subConfig := &SubscriberTestConfig{
|
||||
Namespace: namespace,
|
||||
TopicName: topicName,
|
||||
ConsumerGroup: "ordering-group",
|
||||
ConsumerInstanceId: "ordering-consumer",
|
||||
MaxPartitionCount: 1,
|
||||
SlidingWindowSize: 5,
|
||||
OffsetType: schema_pb.OffsetType_RESET_TO_EARLIEST,
|
||||
}
|
||||
|
||||
subscriber, err := suite.CreateSubscriber(subConfig)
|
||||
require.NoError(t, err)
|
||||
|
||||
// Set up message collector
|
||||
collector := NewMessageCollector(messageCount)
|
||||
subscriber.SetOnDataMessageFn(func(m *mq_pb.SubscribeMessageResponse_Data) {
|
||||
collector.AddMessage(TestMessage{
|
||||
ID: fmt.Sprintf("ordered-msg"),
|
||||
Content: m.Data.Value,
|
||||
Timestamp: time.Unix(0, m.Data.TsNs),
|
||||
Key: m.Data.Key,
|
||||
})
|
||||
})
|
||||
|
||||
// Start subscriber
|
||||
go func() {
|
||||
subscriber.Subscribe()
|
||||
}()
|
||||
|
||||
// Wait for consumer to be ready
|
||||
time.Sleep(2 * time.Second)
|
||||
|
||||
// Publish messages with same key to ensure they go to same partition
|
||||
publishTimes := make([]time.Time, messageCount)
|
||||
for i := 0; i < messageCount; i++ {
|
||||
publishTimes[i] = time.Now()
|
||||
|
||||
record := schema.RecordBegin().
|
||||
SetString("id", fmt.Sprintf("ordered-%d", i)).
|
||||
SetInt64("timestamp", publishTimes[i].UnixNano()).
|
||||
SetString("content", fmt.Sprintf("Ordered message %d", i)).
|
||||
SetInt32("sequence", int32(i)).
|
||||
RecordEnd()
|
||||
|
||||
key := []byte("same-partition-key") // Same key ensures same partition
|
||||
err := publisher.PublishRecord(key, record)
|
||||
require.NoError(t, err)
|
||||
|
||||
// Small delay to ensure different timestamps
|
||||
time.Sleep(10 * time.Millisecond)
|
||||
}
|
||||
|
||||
// Wait for all messages
|
||||
messages := collector.WaitForMessages(30 * time.Second)
|
||||
require.Len(t, messages, messageCount)
|
||||
|
||||
// Verify ordering within the partition
|
||||
suite.AssertMessageOrdering(t, messages)
|
||||
}
|
||||
|
||||
func TestSchemaValidation(t *testing.T) {
|
||||
suite := NewIntegrationTestSuite(t)
|
||||
require.NoError(t, suite.Setup())
|
||||
|
||||
namespace := "test"
|
||||
topicName := "schema-validation"
|
||||
|
||||
// Test with simple schema
|
||||
simpleSchema := CreateTestSchema()
|
||||
|
||||
pubConfig := &PublisherTestConfig{
|
||||
Namespace: namespace,
|
||||
TopicName: topicName,
|
||||
PartitionCount: 1,
|
||||
PublisherName: "schema-publisher",
|
||||
RecordType: simpleSchema,
|
||||
}
|
||||
|
||||
publisher, err := suite.CreatePublisher(pubConfig)
|
||||
require.NoError(t, err)
|
||||
|
||||
// Test valid record
|
||||
validRecord := schema.RecordBegin().
|
||||
SetString("id", "valid-msg").
|
||||
SetInt64("timestamp", time.Now().UnixNano()).
|
||||
SetString("content", "Valid message").
|
||||
SetInt32("sequence", 1).
|
||||
RecordEnd()
|
||||
|
||||
err = publisher.PublishRecord([]byte("test-key"), validRecord)
|
||||
assert.NoError(t, err, "Valid record should be published successfully")
|
||||
|
||||
// Test with complex nested schema
|
||||
complexSchema := CreateComplexTestSchema()
|
||||
|
||||
complexPubConfig := &PublisherTestConfig{
|
||||
Namespace: namespace,
|
||||
TopicName: topicName + "-complex",
|
||||
PartitionCount: 1,
|
||||
PublisherName: "complex-publisher",
|
||||
RecordType: complexSchema,
|
||||
}
|
||||
|
||||
complexPublisher, err := suite.CreatePublisher(complexPubConfig)
|
||||
require.NoError(t, err)
|
||||
|
||||
// Test complex nested record
|
||||
complexRecord := schema.RecordBegin().
|
||||
SetString("user_id", "user123").
|
||||
SetString("name", "John Doe").
|
||||
SetInt32("age", 30).
|
||||
SetStringList("emails", "john@example.com", "john.doe@company.com").
|
||||
SetRecord("address",
|
||||
schema.RecordBegin().
|
||||
SetString("street", "123 Main St").
|
||||
SetString("city", "New York").
|
||||
SetString("zipcode", "10001").
|
||||
RecordEnd()).
|
||||
SetInt64("created_at", time.Now().UnixNano()).
|
||||
RecordEnd()
|
||||
|
||||
err = complexPublisher.PublishRecord([]byte("complex-key"), complexRecord)
|
||||
assert.NoError(t, err, "Complex nested record should be published successfully")
|
||||
}
|
||||
@@ -0,0 +1,379 @@
|
||||
package integration
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"os"
|
||||
"strings"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/mq/agent"
|
||||
"github.com/seaweedfs/seaweedfs/weed/mq/client/pub_client"
|
||||
"github.com/seaweedfs/seaweedfs/weed/mq/client/sub_client"
|
||||
"github.com/seaweedfs/seaweedfs/weed/mq/schema"
|
||||
"github.com/seaweedfs/seaweedfs/weed/mq/topic"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/schema_pb"
|
||||
"github.com/stretchr/testify/require"
|
||||
"google.golang.org/grpc"
|
||||
"google.golang.org/grpc/credentials/insecure"
|
||||
)
|
||||
|
||||
// TestEnvironment holds the configuration for the test environment
|
||||
type TestEnvironment struct {
|
||||
Masters []string
|
||||
Brokers []string
|
||||
Filers []string
|
||||
TestTimeout time.Duration
|
||||
CleanupFuncs []func()
|
||||
mutex sync.Mutex
|
||||
}
|
||||
|
||||
// IntegrationTestSuite provides the base test framework
|
||||
type IntegrationTestSuite struct {
|
||||
env *TestEnvironment
|
||||
agents map[string]*agent.MessageQueueAgent
|
||||
publishers map[string]*pub_client.TopicPublisher
|
||||
subscribers map[string]*sub_client.TopicSubscriber
|
||||
subCancels map[string]context.CancelFunc
|
||||
cleanupOnce sync.Once
|
||||
t *testing.T
|
||||
}
|
||||
|
||||
// NewIntegrationTestSuite creates a new test suite instance
|
||||
func NewIntegrationTestSuite(t *testing.T) *IntegrationTestSuite {
|
||||
env := &TestEnvironment{
|
||||
Masters: getEnvList("SEAWEED_MASTERS", []string{"localhost:19333"}),
|
||||
Brokers: getEnvList("SEAWEED_BROKERS", []string{"localhost:17777"}),
|
||||
Filers: getEnvList("SEAWEED_FILERS", []string{"localhost:18888"}),
|
||||
TestTimeout: getEnvDuration("GO_TEST_TIMEOUT", 30*time.Minute),
|
||||
}
|
||||
|
||||
return &IntegrationTestSuite{
|
||||
env: env,
|
||||
agents: make(map[string]*agent.MessageQueueAgent),
|
||||
publishers: make(map[string]*pub_client.TopicPublisher),
|
||||
subscribers: make(map[string]*sub_client.TopicSubscriber),
|
||||
subCancels: make(map[string]context.CancelFunc),
|
||||
t: t,
|
||||
}
|
||||
}
|
||||
|
||||
// Setup initializes the test environment
|
||||
func (its *IntegrationTestSuite) Setup() error {
|
||||
// Wait for cluster to be ready
|
||||
if err := its.waitForClusterReady(); err != nil {
|
||||
return fmt.Errorf("cluster not ready: %v", err)
|
||||
}
|
||||
|
||||
// Register cleanup
|
||||
its.t.Cleanup(its.Cleanup)
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
// Cleanup performs cleanup operations
|
||||
func (its *IntegrationTestSuite) Cleanup() {
|
||||
its.cleanupOnce.Do(func() {
|
||||
// Close all subscribers first (they use context cancellation)
|
||||
for name := range its.subscribers {
|
||||
if cancel, ok := its.subCancels[name]; ok && cancel != nil {
|
||||
cancel()
|
||||
its.t.Logf("Cancelled subscriber context: %s", name)
|
||||
}
|
||||
its.t.Logf("Cleaned up subscriber: %s", name)
|
||||
}
|
||||
|
||||
// Wait a moment for gRPC connections to close gracefully
|
||||
time.Sleep(1 * time.Second)
|
||||
|
||||
// Close all publishers
|
||||
for name, publisher := range its.publishers {
|
||||
if publisher != nil {
|
||||
// Add timeout to prevent deadlock during shutdown
|
||||
done := make(chan bool, 1)
|
||||
go func(p *pub_client.TopicPublisher, n string) {
|
||||
p.Shutdown()
|
||||
done <- true
|
||||
}(publisher, name)
|
||||
|
||||
select {
|
||||
case <-done:
|
||||
its.t.Logf("Cleaned up publisher: %s", name)
|
||||
case <-time.After(5 * time.Second):
|
||||
its.t.Logf("Publisher shutdown timed out: %s", name)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Execute additional cleanup functions
|
||||
its.env.mutex.Lock()
|
||||
for _, cleanup := range its.env.CleanupFuncs {
|
||||
cleanup()
|
||||
}
|
||||
its.env.mutex.Unlock()
|
||||
})
|
||||
}
|
||||
|
||||
// CreatePublisher creates a new topic publisher
|
||||
func (its *IntegrationTestSuite) CreatePublisher(config *PublisherTestConfig) (*pub_client.TopicPublisher, error) {
|
||||
publisherConfig := &pub_client.PublisherConfiguration{
|
||||
Topic: topic.NewTopic(config.Namespace, config.TopicName),
|
||||
PartitionCount: config.PartitionCount,
|
||||
Brokers: its.env.Brokers,
|
||||
PublisherName: config.PublisherName,
|
||||
RecordType: config.RecordType,
|
||||
}
|
||||
|
||||
publisher, err := pub_client.NewTopicPublisher(publisherConfig)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("failed to create publisher: %v", err)
|
||||
}
|
||||
|
||||
its.publishers[config.PublisherName] = publisher
|
||||
return publisher, nil
|
||||
}
|
||||
|
||||
// CreateSubscriber creates a new topic subscriber
|
||||
func (its *IntegrationTestSuite) CreateSubscriber(config *SubscriberTestConfig) (*sub_client.TopicSubscriber, error) {
|
||||
subscriberConfig := &sub_client.SubscriberConfiguration{
|
||||
ConsumerGroup: config.ConsumerGroup,
|
||||
ConsumerGroupInstanceId: config.ConsumerInstanceId,
|
||||
GrpcDialOption: grpc.WithTransportCredentials(insecure.NewCredentials()),
|
||||
MaxPartitionCount: config.MaxPartitionCount,
|
||||
SlidingWindowSize: config.SlidingWindowSize,
|
||||
}
|
||||
|
||||
contentConfig := &sub_client.ContentConfiguration{
|
||||
Topic: topic.NewTopic(config.Namespace, config.TopicName),
|
||||
Filter: config.Filter,
|
||||
PartitionOffsets: config.PartitionOffsets,
|
||||
OffsetType: config.OffsetType,
|
||||
OffsetTsNs: config.OffsetTsNs,
|
||||
}
|
||||
|
||||
offsetChan := make(chan sub_client.KeyedOffset, 1024)
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
subscriber := sub_client.NewTopicSubscriber(
|
||||
ctx,
|
||||
its.env.Brokers,
|
||||
subscriberConfig,
|
||||
contentConfig,
|
||||
offsetChan,
|
||||
)
|
||||
|
||||
its.subscribers[config.ConsumerInstanceId] = subscriber
|
||||
its.subCancels[config.ConsumerInstanceId] = cancel
|
||||
return subscriber, nil
|
||||
}
|
||||
|
||||
// CreateAgent creates a new message queue agent
|
||||
func (its *IntegrationTestSuite) CreateAgent(name string) (*agent.MessageQueueAgent, error) {
|
||||
var brokerAddresses []pb.ServerAddress
|
||||
for _, broker := range its.env.Brokers {
|
||||
brokerAddresses = append(brokerAddresses, pb.ServerAddress(broker))
|
||||
}
|
||||
|
||||
agentOptions := &agent.MessageQueueAgentOptions{
|
||||
SeedBrokers: brokerAddresses,
|
||||
}
|
||||
|
||||
mqAgent := agent.NewMessageQueueAgent(
|
||||
agentOptions,
|
||||
grpc.WithTransportCredentials(insecure.NewCredentials()),
|
||||
)
|
||||
|
||||
its.agents[name] = mqAgent
|
||||
return mqAgent, nil
|
||||
}
|
||||
|
||||
// PublisherTestConfig holds configuration for creating test publishers
|
||||
type PublisherTestConfig struct {
|
||||
Namespace string
|
||||
TopicName string
|
||||
PartitionCount int32
|
||||
PublisherName string
|
||||
RecordType *schema_pb.RecordType
|
||||
}
|
||||
|
||||
// SubscriberTestConfig holds configuration for creating test subscribers
|
||||
type SubscriberTestConfig struct {
|
||||
Namespace string
|
||||
TopicName string
|
||||
ConsumerGroup string
|
||||
ConsumerInstanceId string
|
||||
MaxPartitionCount int32
|
||||
SlidingWindowSize int32
|
||||
Filter string
|
||||
PartitionOffsets []*schema_pb.PartitionOffset
|
||||
OffsetType schema_pb.OffsetType
|
||||
OffsetTsNs int64
|
||||
}
|
||||
|
||||
// TestMessage represents a test message with metadata
|
||||
type TestMessage struct {
|
||||
ID string
|
||||
Content []byte
|
||||
Timestamp time.Time
|
||||
Key []byte
|
||||
}
|
||||
|
||||
// MessageCollector collects received messages for verification
|
||||
type MessageCollector struct {
|
||||
messages []TestMessage
|
||||
mutex sync.RWMutex
|
||||
waitCh chan struct{}
|
||||
expected int
|
||||
closed bool // protect against closing waitCh multiple times
|
||||
}
|
||||
|
||||
// NewMessageCollector creates a new message collector
|
||||
func NewMessageCollector(expectedCount int) *MessageCollector {
|
||||
return &MessageCollector{
|
||||
messages: make([]TestMessage, 0),
|
||||
waitCh: make(chan struct{}),
|
||||
expected: expectedCount,
|
||||
}
|
||||
}
|
||||
|
||||
// AddMessage adds a received message to the collector
|
||||
func (mc *MessageCollector) AddMessage(msg TestMessage) {
|
||||
mc.mutex.Lock()
|
||||
defer mc.mutex.Unlock()
|
||||
|
||||
mc.messages = append(mc.messages, msg)
|
||||
if len(mc.messages) >= mc.expected && !mc.closed {
|
||||
close(mc.waitCh)
|
||||
mc.closed = true
|
||||
}
|
||||
}
|
||||
|
||||
// WaitForMessages waits for the expected number of messages or timeout
|
||||
func (mc *MessageCollector) WaitForMessages(timeout time.Duration) []TestMessage {
|
||||
select {
|
||||
case <-mc.waitCh:
|
||||
case <-time.After(timeout):
|
||||
}
|
||||
|
||||
mc.mutex.RLock()
|
||||
defer mc.mutex.RUnlock()
|
||||
|
||||
result := make([]TestMessage, len(mc.messages))
|
||||
copy(result, mc.messages)
|
||||
return result
|
||||
}
|
||||
|
||||
// GetMessages returns all collected messages
|
||||
func (mc *MessageCollector) GetMessages() []TestMessage {
|
||||
mc.mutex.RLock()
|
||||
defer mc.mutex.RUnlock()
|
||||
|
||||
result := make([]TestMessage, len(mc.messages))
|
||||
copy(result, mc.messages)
|
||||
return result
|
||||
}
|
||||
|
||||
// CreateTestSchema creates a simple test schema
|
||||
func CreateTestSchema() *schema_pb.RecordType {
|
||||
return schema.RecordTypeBegin().
|
||||
WithField("id", schema.TypeString).
|
||||
WithField("timestamp", schema.TypeInt64).
|
||||
WithField("content", schema.TypeString).
|
||||
WithField("sequence", schema.TypeInt32).
|
||||
RecordTypeEnd()
|
||||
}
|
||||
|
||||
// CreateComplexTestSchema creates a complex test schema with nested structures
|
||||
func CreateComplexTestSchema() *schema_pb.RecordType {
|
||||
addressType := schema.RecordTypeBegin().
|
||||
WithField("street", schema.TypeString).
|
||||
WithField("city", schema.TypeString).
|
||||
WithField("zipcode", schema.TypeString).
|
||||
RecordTypeEnd()
|
||||
|
||||
return schema.RecordTypeBegin().
|
||||
WithField("user_id", schema.TypeString).
|
||||
WithField("name", schema.TypeString).
|
||||
WithField("age", schema.TypeInt32).
|
||||
WithField("emails", schema.ListOf(schema.TypeString)).
|
||||
WithRecordField("address", addressType).
|
||||
WithField("created_at", schema.TypeInt64).
|
||||
RecordTypeEnd()
|
||||
}
|
||||
|
||||
// Helper functions
|
||||
|
||||
func getEnvList(key string, defaultValue []string) []string {
|
||||
value := os.Getenv(key)
|
||||
if value == "" {
|
||||
return defaultValue
|
||||
}
|
||||
return strings.Split(value, ",")
|
||||
}
|
||||
|
||||
func getEnvDuration(key string, defaultValue time.Duration) time.Duration {
|
||||
value := os.Getenv(key)
|
||||
if value == "" {
|
||||
return defaultValue
|
||||
}
|
||||
|
||||
duration, err := time.ParseDuration(value)
|
||||
if err != nil {
|
||||
return defaultValue
|
||||
}
|
||||
return duration
|
||||
}
|
||||
|
||||
func (its *IntegrationTestSuite) waitForClusterReady() error {
|
||||
maxRetries := 30
|
||||
retryInterval := 2 * time.Second
|
||||
|
||||
for i := 0; i < maxRetries; i++ {
|
||||
if its.isClusterReady() {
|
||||
return nil
|
||||
}
|
||||
its.t.Logf("Waiting for cluster to be ready... attempt %d/%d", i+1, maxRetries)
|
||||
time.Sleep(retryInterval)
|
||||
}
|
||||
|
||||
return fmt.Errorf("cluster not ready after %d attempts", maxRetries)
|
||||
}
|
||||
|
||||
func (its *IntegrationTestSuite) isClusterReady() bool {
|
||||
// Check if at least one broker is accessible
|
||||
for _, broker := range its.env.Brokers {
|
||||
if its.isBrokerReady(broker) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func (its *IntegrationTestSuite) isBrokerReady(broker string) bool {
|
||||
// Simple connection test
|
||||
conn, err := grpc.NewClient(broker, grpc.WithTransportCredentials(insecure.NewCredentials()))
|
||||
if err != nil {
|
||||
return false
|
||||
}
|
||||
defer conn.Close()
|
||||
|
||||
// TODO: Add actual health check call here
|
||||
return true
|
||||
}
|
||||
|
||||
// AssertMessagesReceived verifies that expected messages were received
|
||||
func (its *IntegrationTestSuite) AssertMessagesReceived(t *testing.T, collector *MessageCollector, expectedCount int, timeout time.Duration) {
|
||||
messages := collector.WaitForMessages(timeout)
|
||||
require.Len(t, messages, expectedCount, "Expected %d messages, got %d", expectedCount, len(messages))
|
||||
}
|
||||
|
||||
// AssertMessageOrdering verifies that messages are received in the expected order
|
||||
func (its *IntegrationTestSuite) AssertMessageOrdering(t *testing.T, messages []TestMessage) {
|
||||
for i := 1; i < len(messages); i++ {
|
||||
require.True(t, messages[i].Timestamp.After(messages[i-1].Timestamp) || messages[i].Timestamp.Equal(messages[i-1].Timestamp),
|
||||
"Messages not in chronological order: message %d timestamp %v should be >= message %d timestamp %v",
|
||||
i, messages[i].Timestamp, i-1, messages[i-1].Timestamp)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,666 @@
|
||||
package integration
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/mq_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/schema_pb"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
"google.golang.org/grpc/codes"
|
||||
"google.golang.org/grpc/status"
|
||||
)
|
||||
|
||||
// PerformanceMetrics holds test metrics
|
||||
type PerformanceMetrics struct {
|
||||
MessagesPublished int64
|
||||
MessagesConsumed int64
|
||||
PublishLatencies []time.Duration
|
||||
ConsumeLatencies []time.Duration
|
||||
StartTime time.Time
|
||||
EndTime time.Time
|
||||
ErrorCount int64
|
||||
mu sync.RWMutex
|
||||
}
|
||||
|
||||
func (m *PerformanceMetrics) AddPublishLatency(d time.Duration) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
m.PublishLatencies = append(m.PublishLatencies, d)
|
||||
}
|
||||
|
||||
func (m *PerformanceMetrics) AddConsumeLatency(d time.Duration) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
m.ConsumeLatencies = append(m.ConsumeLatencies, d)
|
||||
}
|
||||
|
||||
func (m *PerformanceMetrics) GetThroughput() float64 {
|
||||
duration := m.EndTime.Sub(m.StartTime).Seconds()
|
||||
if duration == 0 {
|
||||
return 0
|
||||
}
|
||||
return float64(atomic.LoadInt64(&m.MessagesPublished)) / duration
|
||||
}
|
||||
|
||||
func (m *PerformanceMetrics) GetP95Latency(latencies []time.Duration) time.Duration {
|
||||
if len(latencies) == 0 {
|
||||
return 0
|
||||
}
|
||||
|
||||
// Simple P95 calculation - in production use proper percentile library
|
||||
index := int(float64(len(latencies)) * 0.95)
|
||||
if index >= len(latencies) {
|
||||
index = len(latencies) - 1
|
||||
}
|
||||
|
||||
// Sort latencies (simplified)
|
||||
for i := 0; i < len(latencies)-1; i++ {
|
||||
for j := 0; j < len(latencies)-i-1; j++ {
|
||||
if latencies[j] > latencies[j+1] {
|
||||
latencies[j], latencies[j+1] = latencies[j+1], latencies[j]
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return latencies[index]
|
||||
}
|
||||
|
||||
// Enhanced performance metrics with connection error tracking
|
||||
type EnhancedPerformanceMetrics struct {
|
||||
MessagesPublished int64
|
||||
MessagesConsumed int64
|
||||
PublishLatencies []time.Duration
|
||||
ConsumeLatencies []time.Duration
|
||||
StartTime time.Time
|
||||
EndTime time.Time
|
||||
ErrorCount int64
|
||||
ConnectionErrors int64
|
||||
ApplicationErrors int64
|
||||
RetryAttempts int64
|
||||
mu sync.RWMutex
|
||||
}
|
||||
|
||||
func (m *EnhancedPerformanceMetrics) AddPublishLatency(d time.Duration) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
m.PublishLatencies = append(m.PublishLatencies, d)
|
||||
}
|
||||
|
||||
func (m *EnhancedPerformanceMetrics) AddConsumeLatency(d time.Duration) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
m.ConsumeLatencies = append(m.ConsumeLatencies, d)
|
||||
}
|
||||
|
||||
func (m *EnhancedPerformanceMetrics) GetThroughput() float64 {
|
||||
duration := m.EndTime.Sub(m.StartTime).Seconds()
|
||||
if duration == 0 {
|
||||
return 0
|
||||
}
|
||||
return float64(atomic.LoadInt64(&m.MessagesPublished)) / duration
|
||||
}
|
||||
|
||||
func (m *EnhancedPerformanceMetrics) GetP95Latency(latencies []time.Duration) time.Duration {
|
||||
if len(latencies) == 0 {
|
||||
return 0
|
||||
}
|
||||
|
||||
// Simple P95 calculation
|
||||
index := int(float64(len(latencies)) * 0.95)
|
||||
if index >= len(latencies) {
|
||||
index = len(latencies) - 1
|
||||
}
|
||||
|
||||
// Sort latencies (simplified bubble sort for small datasets)
|
||||
sorted := make([]time.Duration, len(latencies))
|
||||
copy(sorted, latencies)
|
||||
for i := 0; i < len(sorted)-1; i++ {
|
||||
for j := 0; j < len(sorted)-i-1; j++ {
|
||||
if sorted[j] > sorted[j+1] {
|
||||
sorted[j], sorted[j+1] = sorted[j+1], sorted[j]
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return sorted[index]
|
||||
}
|
||||
|
||||
// isConnectionError determines if an error is a connection-level error
|
||||
func isConnectionError(err error) bool {
|
||||
if err == nil {
|
||||
return false
|
||||
}
|
||||
|
||||
errStr := err.Error()
|
||||
|
||||
// Check for gRPC status codes
|
||||
if st, ok := status.FromError(err); ok {
|
||||
switch st.Code() {
|
||||
case codes.Unavailable, codes.DeadlineExceeded, codes.Canceled, codes.Unknown:
|
||||
return true
|
||||
}
|
||||
}
|
||||
|
||||
// Check for common connection error strings
|
||||
connectionErrorPatterns := []string{
|
||||
"EOF",
|
||||
"error reading server preface",
|
||||
"connection refused",
|
||||
"connection reset",
|
||||
"broken pipe",
|
||||
"network is unreachable",
|
||||
"no route to host",
|
||||
"transport is closing",
|
||||
"connection error",
|
||||
"dial tcp",
|
||||
"context deadline exceeded",
|
||||
}
|
||||
|
||||
for _, pattern := range connectionErrorPatterns {
|
||||
if containsString(errStr, pattern) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
|
||||
return false
|
||||
}
|
||||
|
||||
func containsString(s, substr string) bool {
|
||||
for i := 0; i <= len(s)-len(substr); i++ {
|
||||
if s[i:i+len(substr)] == substr {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func TestPerformanceThroughput(t *testing.T) {
|
||||
suite := NewIntegrationTestSuite(t)
|
||||
require.NoError(t, suite.Setup())
|
||||
|
||||
topicName := "performance-throughput-test"
|
||||
namespace := "perf-test"
|
||||
|
||||
metrics := &PerformanceMetrics{
|
||||
StartTime: time.Now(),
|
||||
}
|
||||
|
||||
// Test parameters
|
||||
numMessages := 50000
|
||||
numPublishers := 5
|
||||
messageSize := 1024 // 1KB messages
|
||||
|
||||
// Create message payload
|
||||
payload := make([]byte, messageSize)
|
||||
for i := range payload {
|
||||
payload[i] = byte(i % 256)
|
||||
}
|
||||
|
||||
t.Logf("Starting throughput test: %d messages, %d publishers, %d bytes per message",
|
||||
numMessages, numPublishers, messageSize)
|
||||
|
||||
// Start publishers
|
||||
var publishWg sync.WaitGroup
|
||||
messagesPerPublisher := numMessages / numPublishers
|
||||
|
||||
for i := 0; i < numPublishers; i++ {
|
||||
publishWg.Add(1)
|
||||
go func(publisherID int) {
|
||||
defer publishWg.Done()
|
||||
|
||||
// Create publisher for this goroutine
|
||||
pubConfig := &PublisherTestConfig{
|
||||
Namespace: namespace,
|
||||
TopicName: topicName,
|
||||
PartitionCount: 4, // Multiple partitions for better throughput
|
||||
PublisherName: fmt.Sprintf("perf-publisher-%d", publisherID),
|
||||
RecordType: nil, // Use raw publish
|
||||
}
|
||||
|
||||
publisher, err := suite.CreatePublisher(pubConfig)
|
||||
if err != nil {
|
||||
atomic.AddInt64(&metrics.ErrorCount, 1)
|
||||
t.Errorf("Failed to create publisher %d: %v", publisherID, err)
|
||||
return
|
||||
}
|
||||
|
||||
for j := 0; j < messagesPerPublisher; j++ {
|
||||
messageKey := fmt.Sprintf("publisher-%d-msg-%d", publisherID, j)
|
||||
|
||||
start := time.Now()
|
||||
err := publisher.Publish([]byte(messageKey), payload)
|
||||
latency := time.Since(start)
|
||||
|
||||
if err != nil {
|
||||
atomic.AddInt64(&metrics.ErrorCount, 1)
|
||||
continue
|
||||
}
|
||||
|
||||
atomic.AddInt64(&metrics.MessagesPublished, 1)
|
||||
metrics.AddPublishLatency(latency)
|
||||
|
||||
// Small delay to prevent overwhelming the system
|
||||
if j%1000 == 0 {
|
||||
time.Sleep(1 * time.Millisecond)
|
||||
}
|
||||
}
|
||||
}(i)
|
||||
}
|
||||
|
||||
// Wait for publishing to complete
|
||||
publishWg.Wait()
|
||||
metrics.EndTime = time.Now()
|
||||
|
||||
// Verify results
|
||||
publishedCount := atomic.LoadInt64(&metrics.MessagesPublished)
|
||||
errorCount := atomic.LoadInt64(&metrics.ErrorCount)
|
||||
throughput := metrics.GetThroughput()
|
||||
|
||||
t.Logf("Performance Results:")
|
||||
t.Logf(" Messages Published: %d", publishedCount)
|
||||
t.Logf(" Errors: %d", errorCount)
|
||||
t.Logf(" Throughput: %.2f messages/second", throughput)
|
||||
t.Logf(" Duration: %v", metrics.EndTime.Sub(metrics.StartTime))
|
||||
|
||||
if len(metrics.PublishLatencies) > 0 {
|
||||
p95Latency := metrics.GetP95Latency(metrics.PublishLatencies)
|
||||
t.Logf(" P95 Publish Latency: %v", p95Latency)
|
||||
|
||||
// Performance assertions
|
||||
assert.Less(t, p95Latency, 100*time.Millisecond, "P95 publish latency should be under 100ms")
|
||||
}
|
||||
|
||||
// Throughput requirements
|
||||
expectedMinThroughput := 5000.0 // 5K messages/sec minimum (relaxed from 10K for initial testing)
|
||||
assert.Greater(t, throughput, expectedMinThroughput,
|
||||
"Throughput should exceed %.0f messages/second", expectedMinThroughput)
|
||||
|
||||
// Error rate should be low
|
||||
errorRate := float64(errorCount) / float64(publishedCount+errorCount)
|
||||
assert.Less(t, errorRate, 0.05, "Error rate should be less than 5%")
|
||||
}
|
||||
|
||||
func TestPerformanceLatency(t *testing.T) {
|
||||
suite := NewIntegrationTestSuite(t)
|
||||
require.NoError(t, suite.Setup())
|
||||
|
||||
topicName := "performance-latency-test"
|
||||
namespace := "perf-test"
|
||||
|
||||
// Create publisher
|
||||
pubConfig := &PublisherTestConfig{
|
||||
Namespace: namespace,
|
||||
TopicName: topicName,
|
||||
PartitionCount: 1, // Single partition for latency testing
|
||||
PublisherName: "latency-publisher",
|
||||
RecordType: nil,
|
||||
}
|
||||
|
||||
publisher, err := suite.CreatePublisher(pubConfig)
|
||||
require.NoError(t, err)
|
||||
|
||||
// Start consumer first
|
||||
subConfig := &SubscriberTestConfig{
|
||||
Namespace: namespace,
|
||||
TopicName: topicName,
|
||||
ConsumerGroup: "latency-test-group",
|
||||
ConsumerInstanceId: "latency-consumer-1",
|
||||
MaxPartitionCount: 1,
|
||||
SlidingWindowSize: 10,
|
||||
OffsetType: schema_pb.OffsetType_RESET_TO_EARLIEST,
|
||||
}
|
||||
|
||||
subscriber, err := suite.CreateSubscriber(subConfig)
|
||||
require.NoError(t, err)
|
||||
|
||||
numMessages := 5000
|
||||
collector := NewMessageCollector(numMessages)
|
||||
|
||||
// Set up message handler
|
||||
subscriber.SetOnDataMessageFn(func(m *mq_pb.SubscribeMessageResponse_Data) {
|
||||
collector.AddMessage(TestMessage{
|
||||
ID: string(m.Data.Key),
|
||||
Content: m.Data.Value,
|
||||
Timestamp: time.Unix(0, m.Data.TsNs),
|
||||
Key: m.Data.Key,
|
||||
})
|
||||
})
|
||||
|
||||
// Start subscriber
|
||||
go func() {
|
||||
err := subscriber.Subscribe()
|
||||
if err != nil {
|
||||
t.Logf("Subscriber error: %v", err)
|
||||
}
|
||||
}()
|
||||
|
||||
// Wait for consumer to be ready
|
||||
time.Sleep(2 * time.Second)
|
||||
|
||||
metrics := &PerformanceMetrics{
|
||||
StartTime: time.Now(),
|
||||
}
|
||||
|
||||
t.Logf("Starting latency test with %d messages", numMessages)
|
||||
|
||||
// Publish messages with controlled timing
|
||||
for i := 0; i < numMessages; i++ {
|
||||
messageKey := fmt.Sprintf("latency-msg-%d", i)
|
||||
payload := fmt.Sprintf("test-payload-data-%d", i)
|
||||
|
||||
start := time.Now()
|
||||
err := publisher.Publish([]byte(messageKey), []byte(payload))
|
||||
publishLatency := time.Since(start)
|
||||
|
||||
require.NoError(t, err)
|
||||
metrics.AddPublishLatency(publishLatency)
|
||||
|
||||
// Controlled rate for latency measurement
|
||||
if i%100 == 0 {
|
||||
time.Sleep(10 * time.Millisecond)
|
||||
}
|
||||
}
|
||||
|
||||
metrics.EndTime = time.Now()
|
||||
|
||||
// Wait for messages to be consumed
|
||||
messages := collector.WaitForMessages(30 * time.Second)
|
||||
|
||||
// Analyze latency results
|
||||
t.Logf("Latency Test Results:")
|
||||
t.Logf(" Messages Published: %d", numMessages)
|
||||
t.Logf(" Messages Consumed: %d", len(messages))
|
||||
|
||||
if len(metrics.PublishLatencies) > 0 {
|
||||
p95PublishLatency := metrics.GetP95Latency(metrics.PublishLatencies)
|
||||
t.Logf(" P95 Publish Latency: %v", p95PublishLatency)
|
||||
|
||||
// Latency assertions
|
||||
assert.Less(t, p95PublishLatency, 50*time.Millisecond,
|
||||
"P95 publish latency should be under 50ms")
|
||||
}
|
||||
|
||||
// Verify message delivery
|
||||
deliveryRate := float64(len(messages)) / float64(numMessages)
|
||||
assert.Greater(t, deliveryRate, 0.80, "Should deliver at least 80% of messages")
|
||||
}
|
||||
|
||||
func TestPerformanceConcurrentConsumers(t *testing.T) {
|
||||
suite := NewIntegrationTestSuite(t)
|
||||
require.NoError(t, suite.Setup())
|
||||
|
||||
topicName := "performance-concurrent-test"
|
||||
namespace := "perf-test"
|
||||
numPartitions := int32(4)
|
||||
|
||||
// Create publisher first
|
||||
pubConfig := &PublisherTestConfig{
|
||||
Namespace: namespace,
|
||||
TopicName: topicName,
|
||||
PartitionCount: numPartitions,
|
||||
PublisherName: "concurrent-publisher",
|
||||
RecordType: nil,
|
||||
}
|
||||
|
||||
publisher, err := suite.CreatePublisher(pubConfig)
|
||||
require.NoError(t, err)
|
||||
|
||||
// Test parameters
|
||||
numConsumers := 8
|
||||
numMessages := 20000
|
||||
consumerGroup := "concurrent-perf-group"
|
||||
|
||||
t.Logf("Starting concurrent consumer test: %d consumers, %d messages, %d partitions",
|
||||
numConsumers, numMessages, numPartitions)
|
||||
|
||||
// Start multiple consumers
|
||||
var collectors []*MessageCollector
|
||||
|
||||
for i := 0; i < numConsumers; i++ {
|
||||
collector := NewMessageCollector(numMessages / numConsumers) // Expected per consumer
|
||||
collectors = append(collectors, collector)
|
||||
|
||||
subConfig := &SubscriberTestConfig{
|
||||
Namespace: namespace,
|
||||
TopicName: topicName,
|
||||
ConsumerGroup: consumerGroup,
|
||||
ConsumerInstanceId: fmt.Sprintf("consumer-%d", i),
|
||||
MaxPartitionCount: numPartitions,
|
||||
SlidingWindowSize: 10,
|
||||
OffsetType: schema_pb.OffsetType_RESET_TO_EARLIEST,
|
||||
}
|
||||
|
||||
subscriber, err := suite.CreateSubscriber(subConfig)
|
||||
require.NoError(t, err)
|
||||
|
||||
// Set up message handler for this consumer
|
||||
func(consumerID int, c *MessageCollector) {
|
||||
subscriber.SetOnDataMessageFn(func(m *mq_pb.SubscribeMessageResponse_Data) {
|
||||
c.AddMessage(TestMessage{
|
||||
ID: fmt.Sprintf("%d-%s", consumerID, string(m.Data.Key)),
|
||||
Content: m.Data.Value,
|
||||
Timestamp: time.Unix(0, m.Data.TsNs),
|
||||
Key: m.Data.Key,
|
||||
})
|
||||
})
|
||||
}(i, collector)
|
||||
|
||||
// Start subscriber
|
||||
go func(consumerID int, s *SubscriberTestConfig) {
|
||||
sub, _ := suite.CreateSubscriber(s)
|
||||
err := sub.Subscribe()
|
||||
if err != nil {
|
||||
t.Logf("Consumer %d error: %v", consumerID, err)
|
||||
}
|
||||
}(i, subConfig)
|
||||
}
|
||||
|
||||
// Wait for consumers to initialize
|
||||
time.Sleep(3 * time.Second)
|
||||
|
||||
// Start publishing
|
||||
startTime := time.Now()
|
||||
|
||||
var publishWg sync.WaitGroup
|
||||
numPublishers := 3
|
||||
messagesPerPublisher := numMessages / numPublishers
|
||||
|
||||
for i := 0; i < numPublishers; i++ {
|
||||
publishWg.Add(1)
|
||||
go func(publisherID int) {
|
||||
defer publishWg.Done()
|
||||
|
||||
for j := 0; j < messagesPerPublisher; j++ {
|
||||
messageKey := fmt.Sprintf("concurrent-msg-%d-%d", publisherID, j)
|
||||
payload := fmt.Sprintf("publisher-%d-message-%d-data", publisherID, j)
|
||||
|
||||
err := publisher.Publish([]byte(messageKey), []byte(payload))
|
||||
|
||||
if err != nil {
|
||||
t.Logf("Publish error: %v", err)
|
||||
}
|
||||
|
||||
// Rate limiting
|
||||
if j%1000 == 0 {
|
||||
time.Sleep(2 * time.Millisecond)
|
||||
}
|
||||
}
|
||||
}(i)
|
||||
}
|
||||
|
||||
publishWg.Wait()
|
||||
publishDuration := time.Since(startTime)
|
||||
|
||||
// Allow time for message consumption
|
||||
time.Sleep(10 * time.Second)
|
||||
|
||||
// Analyze results
|
||||
totalConsumed := int64(0)
|
||||
for i, collector := range collectors {
|
||||
messages := collector.GetMessages()
|
||||
consumed := int64(len(messages))
|
||||
totalConsumed += consumed
|
||||
t.Logf("Consumer %d consumed %d messages", i, consumed)
|
||||
}
|
||||
|
||||
publishThroughput := float64(numMessages) / publishDuration.Seconds()
|
||||
consumeThroughput := float64(totalConsumed) / publishDuration.Seconds()
|
||||
|
||||
t.Logf("Concurrent Consumer Test Results:")
|
||||
t.Logf(" Total Published: %d", numMessages)
|
||||
t.Logf(" Total Consumed: %d", totalConsumed)
|
||||
t.Logf(" Publish Throughput: %.2f msg/sec", publishThroughput)
|
||||
t.Logf(" Consume Throughput: %.2f msg/sec", consumeThroughput)
|
||||
t.Logf(" Test Duration: %v", publishDuration)
|
||||
|
||||
// Performance assertions (relaxed for initial testing)
|
||||
deliveryRate := float64(totalConsumed) / float64(numMessages)
|
||||
assert.Greater(t, deliveryRate, 0.70, "Should consume at least 70% of messages")
|
||||
|
||||
expectedMinThroughput := 2000.0 // 2K messages/sec minimum for concurrent consumption
|
||||
assert.Greater(t, consumeThroughput, expectedMinThroughput,
|
||||
"Consume throughput should exceed %.0f messages/second", expectedMinThroughput)
|
||||
}
|
||||
|
||||
func TestPerformanceWithErrorHandling(t *testing.T) {
|
||||
suite := NewIntegrationTestSuite(t)
|
||||
require.NoError(t, suite.Setup())
|
||||
|
||||
topicName := "performance-error-handling-test"
|
||||
namespace := "perf-test"
|
||||
|
||||
metrics := &EnhancedPerformanceMetrics{
|
||||
StartTime: time.Now(),
|
||||
}
|
||||
|
||||
// Test parameters
|
||||
numMessages := 50000
|
||||
numPublishers := 5
|
||||
messageSize := 1024 // 1KB messages
|
||||
|
||||
// Create message payload
|
||||
payload := make([]byte, messageSize)
|
||||
for i := range payload {
|
||||
payload[i] = byte(i % 256)
|
||||
}
|
||||
|
||||
t.Logf("Starting performance test with enhanced error handling: %d messages, %d publishers, %d bytes per message",
|
||||
numMessages, numPublishers, messageSize)
|
||||
|
||||
// Start publishers
|
||||
var publishWg sync.WaitGroup
|
||||
messagesPerPublisher := numMessages / numPublishers
|
||||
|
||||
for i := 0; i < numPublishers; i++ {
|
||||
publishWg.Add(1)
|
||||
go func(publisherID int) {
|
||||
defer publishWg.Done()
|
||||
|
||||
// Create publisher for this goroutine
|
||||
pubConfig := &PublisherTestConfig{
|
||||
Namespace: namespace,
|
||||
TopicName: topicName,
|
||||
PartitionCount: 4, // Multiple partitions for better throughput
|
||||
PublisherName: fmt.Sprintf("error-aware-publisher-%d", publisherID),
|
||||
RecordType: nil, // Use raw publish
|
||||
}
|
||||
|
||||
publisher, err := suite.CreatePublisher(pubConfig)
|
||||
if err != nil {
|
||||
atomic.AddInt64(&metrics.ErrorCount, 1)
|
||||
t.Errorf("Failed to create publisher %d: %v", publisherID, err)
|
||||
return
|
||||
}
|
||||
|
||||
for j := 0; j < messagesPerPublisher; j++ {
|
||||
messageKey := fmt.Sprintf("publisher-%d-msg-%d", publisherID, j)
|
||||
|
||||
start := time.Now()
|
||||
err := publisher.Publish([]byte(messageKey), payload)
|
||||
latency := time.Since(start)
|
||||
|
||||
if err != nil {
|
||||
atomic.AddInt64(&metrics.ErrorCount, 1)
|
||||
|
||||
// Classify the error type
|
||||
if isConnectionError(err) {
|
||||
atomic.AddInt64(&metrics.ConnectionErrors, 1)
|
||||
t.Logf("Connection error (publisher %d, msg %d): %v", publisherID, j, err)
|
||||
} else {
|
||||
atomic.AddInt64(&metrics.ApplicationErrors, 1)
|
||||
t.Logf("Application error (publisher %d, msg %d): %v", publisherID, j, err)
|
||||
}
|
||||
continue
|
||||
}
|
||||
|
||||
atomic.AddInt64(&metrics.MessagesPublished, 1)
|
||||
metrics.AddPublishLatency(latency)
|
||||
|
||||
// Small delay to prevent overwhelming the system
|
||||
if j%1000 == 0 {
|
||||
time.Sleep(1 * time.Millisecond)
|
||||
}
|
||||
}
|
||||
}(i)
|
||||
}
|
||||
|
||||
// Wait for publishing to complete
|
||||
publishWg.Wait()
|
||||
metrics.EndTime = time.Now()
|
||||
|
||||
// Analyze results with enhanced error reporting
|
||||
publishedCount := atomic.LoadInt64(&metrics.MessagesPublished)
|
||||
totalErrors := atomic.LoadInt64(&metrics.ErrorCount)
|
||||
connectionErrors := atomic.LoadInt64(&metrics.ConnectionErrors)
|
||||
applicationErrors := atomic.LoadInt64(&metrics.ApplicationErrors)
|
||||
throughput := metrics.GetThroughput()
|
||||
|
||||
t.Logf("Enhanced Performance Results:")
|
||||
t.Logf(" Messages Successfully Published: %d", publishedCount)
|
||||
t.Logf(" Total Errors: %d", totalErrors)
|
||||
t.Logf(" Connection-Level Errors: %d", connectionErrors)
|
||||
t.Logf(" Application-Level Errors: %d", applicationErrors)
|
||||
t.Logf(" Throughput: %.2f messages/second", throughput)
|
||||
t.Logf(" Duration: %v", metrics.EndTime.Sub(metrics.StartTime))
|
||||
|
||||
if len(metrics.PublishLatencies) > 0 {
|
||||
p95Latency := metrics.GetP95Latency(metrics.PublishLatencies)
|
||||
t.Logf(" P95 Publish Latency: %v", p95Latency)
|
||||
|
||||
// Performance assertions (adjusted for error handling overhead)
|
||||
assert.Less(t, p95Latency, 100*time.Millisecond, "P95 publish latency should be under 100ms")
|
||||
}
|
||||
|
||||
// Enhanced error analysis
|
||||
totalAttempts := publishedCount + totalErrors
|
||||
if totalAttempts > 0 {
|
||||
successRate := float64(publishedCount) / float64(totalAttempts)
|
||||
connectionErrorRate := float64(connectionErrors) / float64(totalAttempts)
|
||||
applicationErrorRate := float64(applicationErrors) / float64(totalAttempts)
|
||||
|
||||
t.Logf("Error Analysis:")
|
||||
t.Logf(" Success Rate: %.2f%%", successRate*100)
|
||||
t.Logf(" Connection Error Rate: %.2f%%", connectionErrorRate*100)
|
||||
t.Logf(" Application Error Rate: %.2f%%", applicationErrorRate*100)
|
||||
|
||||
// Assertions based on error types
|
||||
assert.Greater(t, successRate, 0.80, "Success rate should be greater than 80%")
|
||||
assert.Less(t, applicationErrorRate, 0.01, "Application error rate should be less than 1%")
|
||||
|
||||
// Connection errors are expected under high load but should be handled
|
||||
if connectionErrors > 0 {
|
||||
t.Logf("Note: %d connection errors detected - this indicates the test is successfully stressing the system", connectionErrors)
|
||||
t.Logf("Recommendation: Implement retry logic for production applications to handle these connection errors")
|
||||
}
|
||||
}
|
||||
|
||||
// Throughput requirements (adjusted for error handling)
|
||||
expectedMinThroughput := 5000.0 // 5K messages/sec minimum
|
||||
assert.Greater(t, throughput, expectedMinThroughput,
|
||||
"Throughput should exceed %.0f messages/second", expectedMinThroughput)
|
||||
}
|
||||
@@ -0,0 +1,336 @@
|
||||
package integration
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"math"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/glog"
|
||||
"github.com/seaweedfs/seaweedfs/weed/mq/client/pub_client"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/schema_pb"
|
||||
"google.golang.org/grpc/codes"
|
||||
"google.golang.org/grpc/status"
|
||||
)
|
||||
|
||||
// ResilientPublisher wraps TopicPublisher with enhanced error handling
|
||||
type ResilientPublisher struct {
|
||||
publisher *pub_client.TopicPublisher
|
||||
config *PublisherTestConfig
|
||||
suite *IntegrationTestSuite
|
||||
|
||||
// Error tracking
|
||||
connectionErrors int64
|
||||
applicationErrors int64
|
||||
retryAttempts int64
|
||||
totalPublishes int64
|
||||
|
||||
// Retry configuration
|
||||
maxRetries int
|
||||
baseDelay time.Duration
|
||||
maxDelay time.Duration
|
||||
backoffFactor float64
|
||||
|
||||
// Circuit breaker
|
||||
circuitOpen bool
|
||||
circuitOpenTime time.Time
|
||||
circuitTimeout time.Duration
|
||||
|
||||
mu sync.RWMutex
|
||||
}
|
||||
|
||||
// RetryConfig holds retry configuration
|
||||
type RetryConfig struct {
|
||||
MaxRetries int
|
||||
BaseDelay time.Duration
|
||||
MaxDelay time.Duration
|
||||
BackoffFactor float64
|
||||
CircuitTimeout time.Duration
|
||||
}
|
||||
|
||||
// DefaultRetryConfig returns sensible defaults for retry configuration
|
||||
func DefaultRetryConfig() *RetryConfig {
|
||||
return &RetryConfig{
|
||||
MaxRetries: 5,
|
||||
BaseDelay: 10 * time.Millisecond,
|
||||
MaxDelay: 5 * time.Second,
|
||||
BackoffFactor: 2.0,
|
||||
CircuitTimeout: 30 * time.Second,
|
||||
}
|
||||
}
|
||||
|
||||
// NewResilientPublisher creates a new resilient publisher
|
||||
func (its *IntegrationTestSuite) CreateResilientPublisher(config *PublisherTestConfig, retryConfig *RetryConfig) (*ResilientPublisher, error) {
|
||||
if retryConfig == nil {
|
||||
retryConfig = DefaultRetryConfig()
|
||||
}
|
||||
|
||||
publisher, err := its.CreatePublisher(config)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("failed to create base publisher: %v", err)
|
||||
}
|
||||
|
||||
return &ResilientPublisher{
|
||||
publisher: publisher,
|
||||
config: config,
|
||||
suite: its,
|
||||
maxRetries: retryConfig.MaxRetries,
|
||||
baseDelay: retryConfig.BaseDelay,
|
||||
maxDelay: retryConfig.MaxDelay,
|
||||
backoffFactor: retryConfig.BackoffFactor,
|
||||
circuitTimeout: retryConfig.CircuitTimeout,
|
||||
}, nil
|
||||
}
|
||||
|
||||
// PublishWithRetry publishes a message with retry logic and error handling
|
||||
func (rp *ResilientPublisher) PublishWithRetry(key, value []byte) error {
|
||||
atomic.AddInt64(&rp.totalPublishes, 1)
|
||||
|
||||
// Check circuit breaker
|
||||
if rp.isCircuitOpen() {
|
||||
atomic.AddInt64(&rp.applicationErrors, 1)
|
||||
return fmt.Errorf("circuit breaker is open")
|
||||
}
|
||||
|
||||
var lastErr error
|
||||
for attempt := 0; attempt <= rp.maxRetries; attempt++ {
|
||||
if attempt > 0 {
|
||||
atomic.AddInt64(&rp.retryAttempts, 1)
|
||||
delay := rp.calculateDelay(attempt)
|
||||
glog.V(1).Infof("Retrying publish after %v (attempt %d/%d)", delay, attempt, rp.maxRetries)
|
||||
time.Sleep(delay)
|
||||
}
|
||||
|
||||
err := rp.publisher.Publish(key, value)
|
||||
if err == nil {
|
||||
// Success - reset circuit breaker if it was open
|
||||
rp.resetCircuitBreaker()
|
||||
return nil
|
||||
}
|
||||
|
||||
lastErr = err
|
||||
|
||||
// Classify error type
|
||||
if rp.isConnectionError(err) {
|
||||
atomic.AddInt64(&rp.connectionErrors, 1)
|
||||
glog.V(1).Infof("Connection error on attempt %d: %v", attempt+1, err)
|
||||
|
||||
// For connection errors, try to recreate the publisher
|
||||
if attempt < rp.maxRetries {
|
||||
if recreateErr := rp.recreatePublisher(); recreateErr != nil {
|
||||
glog.Warningf("Failed to recreate publisher: %v", recreateErr)
|
||||
}
|
||||
}
|
||||
continue
|
||||
} else {
|
||||
// Application error - don't retry
|
||||
atomic.AddInt64(&rp.applicationErrors, 1)
|
||||
glog.Warningf("Application error (not retrying): %v", err)
|
||||
break
|
||||
}
|
||||
}
|
||||
|
||||
// All retries exhausted or non-retryable error
|
||||
rp.openCircuitBreaker()
|
||||
return fmt.Errorf("publish failed after %d attempts, last error: %v", rp.maxRetries+1, lastErr)
|
||||
}
|
||||
|
||||
// PublishRecord publishes a record with retry logic
|
||||
func (rp *ResilientPublisher) PublishRecord(key []byte, record *schema_pb.RecordValue) error {
|
||||
atomic.AddInt64(&rp.totalPublishes, 1)
|
||||
|
||||
if rp.isCircuitOpen() {
|
||||
atomic.AddInt64(&rp.applicationErrors, 1)
|
||||
return fmt.Errorf("circuit breaker is open")
|
||||
}
|
||||
|
||||
var lastErr error
|
||||
for attempt := 0; attempt <= rp.maxRetries; attempt++ {
|
||||
if attempt > 0 {
|
||||
atomic.AddInt64(&rp.retryAttempts, 1)
|
||||
delay := rp.calculateDelay(attempt)
|
||||
time.Sleep(delay)
|
||||
}
|
||||
|
||||
err := rp.publisher.PublishRecord(key, record)
|
||||
if err == nil {
|
||||
rp.resetCircuitBreaker()
|
||||
return nil
|
||||
}
|
||||
|
||||
lastErr = err
|
||||
|
||||
if rp.isConnectionError(err) {
|
||||
atomic.AddInt64(&rp.connectionErrors, 1)
|
||||
if attempt < rp.maxRetries {
|
||||
if recreateErr := rp.recreatePublisher(); recreateErr != nil {
|
||||
glog.Warningf("Failed to recreate publisher: %v", recreateErr)
|
||||
}
|
||||
}
|
||||
continue
|
||||
} else {
|
||||
atomic.AddInt64(&rp.applicationErrors, 1)
|
||||
break
|
||||
}
|
||||
}
|
||||
|
||||
rp.openCircuitBreaker()
|
||||
return fmt.Errorf("publish record failed after %d attempts, last error: %v", rp.maxRetries+1, lastErr)
|
||||
}
|
||||
|
||||
// recreatePublisher attempts to recreate the underlying publisher
|
||||
func (rp *ResilientPublisher) recreatePublisher() error {
|
||||
rp.mu.Lock()
|
||||
defer rp.mu.Unlock()
|
||||
|
||||
// Shutdown old publisher
|
||||
if rp.publisher != nil {
|
||||
rp.publisher.Shutdown()
|
||||
}
|
||||
|
||||
// Create new publisher
|
||||
newPublisher, err := rp.suite.CreatePublisher(rp.config)
|
||||
if err != nil {
|
||||
return fmt.Errorf("failed to recreate publisher: %v", err)
|
||||
}
|
||||
|
||||
rp.publisher = newPublisher
|
||||
glog.V(1).Infof("Successfully recreated publisher")
|
||||
return nil
|
||||
}
|
||||
|
||||
// isConnectionError determines if an error is a connection-level error that should be retried
|
||||
func (rp *ResilientPublisher) isConnectionError(err error) bool {
|
||||
if err == nil {
|
||||
return false
|
||||
}
|
||||
|
||||
errStr := err.Error()
|
||||
|
||||
// Check for gRPC status codes
|
||||
if st, ok := status.FromError(err); ok {
|
||||
switch st.Code() {
|
||||
case codes.Unavailable, codes.DeadlineExceeded, codes.Canceled, codes.Unknown:
|
||||
return true
|
||||
}
|
||||
}
|
||||
|
||||
// Check for common connection error strings
|
||||
connectionErrorPatterns := []string{
|
||||
"EOF",
|
||||
"error reading server preface",
|
||||
"connection refused",
|
||||
"connection reset",
|
||||
"broken pipe",
|
||||
"network is unreachable",
|
||||
"no route to host",
|
||||
"transport is closing",
|
||||
"connection error",
|
||||
"dial tcp",
|
||||
"context deadline exceeded",
|
||||
}
|
||||
|
||||
for _, pattern := range connectionErrorPatterns {
|
||||
if contains(errStr, pattern) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
|
||||
return false
|
||||
}
|
||||
|
||||
// contains checks if a string contains a substring (case-insensitive)
|
||||
func contains(s, substr string) bool {
|
||||
return len(s) >= len(substr) &&
|
||||
(s == substr ||
|
||||
len(s) > len(substr) &&
|
||||
(s[:len(substr)] == substr ||
|
||||
s[len(s)-len(substr):] == substr ||
|
||||
containsSubstring(s, substr)))
|
||||
}
|
||||
|
||||
func containsSubstring(s, substr string) bool {
|
||||
for i := 0; i <= len(s)-len(substr); i++ {
|
||||
if s[i:i+len(substr)] == substr {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// calculateDelay calculates exponential backoff delay
|
||||
func (rp *ResilientPublisher) calculateDelay(attempt int) time.Duration {
|
||||
delay := float64(rp.baseDelay) * math.Pow(rp.backoffFactor, float64(attempt-1))
|
||||
if delay > float64(rp.maxDelay) {
|
||||
delay = float64(rp.maxDelay)
|
||||
}
|
||||
return time.Duration(delay)
|
||||
}
|
||||
|
||||
// Circuit breaker methods
|
||||
func (rp *ResilientPublisher) isCircuitOpen() bool {
|
||||
rp.mu.RLock()
|
||||
defer rp.mu.RUnlock()
|
||||
|
||||
if !rp.circuitOpen {
|
||||
return false
|
||||
}
|
||||
|
||||
// Check if circuit should be reset
|
||||
if time.Since(rp.circuitOpenTime) > rp.circuitTimeout {
|
||||
return false
|
||||
}
|
||||
|
||||
return true
|
||||
}
|
||||
|
||||
func (rp *ResilientPublisher) openCircuitBreaker() {
|
||||
rp.mu.Lock()
|
||||
defer rp.mu.Unlock()
|
||||
|
||||
rp.circuitOpen = true
|
||||
rp.circuitOpenTime = time.Now()
|
||||
glog.Warningf("Circuit breaker opened due to repeated failures")
|
||||
}
|
||||
|
||||
func (rp *ResilientPublisher) resetCircuitBreaker() {
|
||||
rp.mu.Lock()
|
||||
defer rp.mu.Unlock()
|
||||
|
||||
if rp.circuitOpen {
|
||||
rp.circuitOpen = false
|
||||
glog.V(1).Infof("Circuit breaker reset")
|
||||
}
|
||||
}
|
||||
|
||||
// GetErrorStats returns error statistics
|
||||
func (rp *ResilientPublisher) GetErrorStats() ErrorStats {
|
||||
return ErrorStats{
|
||||
ConnectionErrors: atomic.LoadInt64(&rp.connectionErrors),
|
||||
ApplicationErrors: atomic.LoadInt64(&rp.applicationErrors),
|
||||
RetryAttempts: atomic.LoadInt64(&rp.retryAttempts),
|
||||
TotalPublishes: atomic.LoadInt64(&rp.totalPublishes),
|
||||
CircuitOpen: rp.isCircuitOpen(),
|
||||
}
|
||||
}
|
||||
|
||||
// ErrorStats holds error statistics
|
||||
type ErrorStats struct {
|
||||
ConnectionErrors int64
|
||||
ApplicationErrors int64
|
||||
RetryAttempts int64
|
||||
TotalPublishes int64
|
||||
CircuitOpen bool
|
||||
}
|
||||
|
||||
// Shutdown gracefully shuts down the resilient publisher
|
||||
func (rp *ResilientPublisher) Shutdown() error {
|
||||
rp.mu.Lock()
|
||||
defer rp.mu.Unlock()
|
||||
|
||||
if rp.publisher != nil {
|
||||
return rp.publisher.Shutdown()
|
||||
}
|
||||
return nil
|
||||
}
|
||||
@@ -0,0 +1,286 @@
|
||||
# SeaweedMQ Integration Test Design
|
||||
|
||||
## Overview
|
||||
|
||||
This document outlines the comprehensive integration test strategy for SeaweedMQ, covering all critical functionalities from basic pub/sub operations to advanced features like auto-scaling, failover, and performance testing.
|
||||
|
||||
## Architecture Under Test
|
||||
|
||||
SeaweedMQ consists of:
|
||||
- **Masters**: Cluster coordination and metadata management
|
||||
- **Volume Servers**: Storage layer for persistent messages
|
||||
- **Filers**: File system interface for metadata storage
|
||||
- **Brokers**: Message processing and routing (stateless)
|
||||
- **Agents**: Client interface for pub/sub operations
|
||||
- **Schema System**: Protobuf-based message schema management
|
||||
|
||||
## Test Categories
|
||||
|
||||
### 1. Basic Functionality Tests
|
||||
|
||||
#### 1.1 Basic Pub/Sub Operations
|
||||
- **Test**: `TestBasicPublishSubscribe`
|
||||
- Publish messages to a topic
|
||||
- Subscribe and receive messages
|
||||
- Verify message content and ordering
|
||||
- Test with different data types (string, int, bytes, records)
|
||||
|
||||
- **Test**: `TestMultipleConsumers`
|
||||
- Multiple subscribers on same topic
|
||||
- Verify message distribution
|
||||
- Test consumer group functionality
|
||||
|
||||
- **Test**: `TestMessageOrdering`
|
||||
- Publish messages in sequence
|
||||
- Verify FIFO ordering within partitions
|
||||
- Test with different partition keys
|
||||
|
||||
#### 1.2 Schema Management
|
||||
- **Test**: `TestSchemaValidation`
|
||||
- Publish with valid schemas
|
||||
- Reject invalid schema messages
|
||||
- Test schema evolution scenarios
|
||||
|
||||
- **Test**: `TestRecordTypes`
|
||||
- Nested record structures
|
||||
- List types and complex schemas
|
||||
- Schema-to-Parquet conversion
|
||||
|
||||
### 2. Partitioning and Scaling Tests
|
||||
|
||||
#### 2.1 Partition Management
|
||||
- **Test**: `TestPartitionDistribution`
|
||||
- Messages distributed across partitions based on keys
|
||||
- Verify partition assignment logic
|
||||
- Test partition rebalancing
|
||||
|
||||
- **Test**: `TestAutoSplitMerge`
|
||||
- Simulate high load to trigger auto-split
|
||||
- Simulate low load to trigger auto-merge
|
||||
- Verify data consistency during splits/merges
|
||||
|
||||
#### 2.2 Broker Scaling
|
||||
- **Test**: `TestBrokerAddRemove`
|
||||
- Add brokers during operation
|
||||
- Remove brokers gracefully
|
||||
- Verify partition reassignment
|
||||
|
||||
- **Test**: `TestLoadBalancing`
|
||||
- Verify even load distribution across brokers
|
||||
- Test with varying message sizes and rates
|
||||
- Monitor broker resource utilization
|
||||
|
||||
### 3. Failover and Reliability Tests
|
||||
|
||||
#### 3.1 Broker Failover
|
||||
- **Test**: `TestBrokerFailover`
|
||||
- Kill leader broker during publishing
|
||||
- Verify seamless failover to follower
|
||||
- Test data consistency after failover
|
||||
|
||||
- **Test**: `TestBrokerRecovery`
|
||||
- Broker restart scenarios
|
||||
- State recovery from storage
|
||||
- Partition reassignment after recovery
|
||||
|
||||
#### 3.2 Data Durability
|
||||
- **Test**: `TestMessagePersistence`
|
||||
- Publish messages and restart cluster
|
||||
- Verify all messages are recovered
|
||||
- Test with different replication settings
|
||||
|
||||
- **Test**: `TestFollowerReplication`
|
||||
- Leader-follower message replication
|
||||
- Verify consistency between replicas
|
||||
- Test follower promotion scenarios
|
||||
|
||||
### 4. Agent Functionality Tests
|
||||
|
||||
#### 4.1 Session Management
|
||||
- **Test**: `TestPublishSessions`
|
||||
- Create/close publish sessions
|
||||
- Concurrent session management
|
||||
- Session cleanup after failures
|
||||
|
||||
- **Test**: `TestSubscribeSessions`
|
||||
- Subscribe session lifecycle
|
||||
- Consumer group management
|
||||
- Offset tracking and acknowledgments
|
||||
|
||||
#### 4.2 Error Handling
|
||||
- **Test**: `TestConnectionFailures`
|
||||
- Network partitions between agent and broker
|
||||
- Automatic reconnection logic
|
||||
- Message buffering during outages
|
||||
|
||||
### 5. Performance and Load Tests
|
||||
|
||||
#### 5.1 Throughput Tests
|
||||
- **Test**: `TestHighThroughputPublish`
|
||||
- Publish 100K+ messages/second
|
||||
- Monitor system resources
|
||||
- Verify no message loss
|
||||
|
||||
- **Test**: `TestHighThroughputSubscribe`
|
||||
- Multiple consumers processing high volume
|
||||
- Monitor processing latency
|
||||
- Test backpressure handling
|
||||
|
||||
#### 5.2 Spike Traffic Tests
|
||||
- **Test**: `TestTrafficSpikes`
|
||||
- Sudden increase in message volume
|
||||
- Auto-scaling behavior verification
|
||||
- Resource utilization patterns
|
||||
|
||||
- **Test**: `TestLargeMessages`
|
||||
- Messages with large payloads (MB size)
|
||||
- Memory usage monitoring
|
||||
- Storage efficiency testing
|
||||
|
||||
### 6. End-to-End Scenarios
|
||||
|
||||
#### 6.1 Complete Workflow Tests
|
||||
- **Test**: `TestProducerConsumerWorkflow`
|
||||
- Multi-stage data processing pipeline
|
||||
- Producer → Topic → Multiple Consumers
|
||||
- Data transformation and aggregation
|
||||
|
||||
- **Test**: `TestMultiTopicOperations`
|
||||
- Multiple topics with different schemas
|
||||
- Cross-topic message routing
|
||||
- Topic management operations
|
||||
|
||||
## Test Infrastructure
|
||||
|
||||
### Environment Setup
|
||||
|
||||
#### Docker Compose Configuration
|
||||
```yaml
|
||||
# test-environment.yml
|
||||
version: '3.9'
|
||||
services:
|
||||
master-cluster:
|
||||
# 3 master nodes for HA
|
||||
volume-cluster:
|
||||
# 3 volume servers for data storage
|
||||
filer-cluster:
|
||||
# 2 filers for metadata
|
||||
broker-cluster:
|
||||
# 3 brokers for message processing
|
||||
test-runner:
|
||||
# Container to run integration tests
|
||||
```
|
||||
|
||||
#### Test Data Management
|
||||
- Pre-defined test schemas
|
||||
- Sample message datasets
|
||||
- Performance benchmarking data
|
||||
|
||||
### Test Framework Structure
|
||||
|
||||
```go
|
||||
// Base test framework
|
||||
type IntegrationTestSuite struct {
|
||||
masters []string
|
||||
brokers []string
|
||||
filers []string
|
||||
testClient *TestClient
|
||||
cleanup []func()
|
||||
}
|
||||
|
||||
// Test utilities
|
||||
type TestClient struct {
|
||||
publishers map[string]*pub_client.TopicPublisher
|
||||
subscribers map[string]*sub_client.TopicSubscriber
|
||||
agents []*agent.MessageQueueAgent
|
||||
}
|
||||
```
|
||||
|
||||
### Monitoring and Metrics
|
||||
|
||||
#### Health Checks
|
||||
- Broker connectivity status
|
||||
- Master cluster health
|
||||
- Storage system availability
|
||||
- Network connectivity between components
|
||||
|
||||
#### Performance Metrics
|
||||
- Message throughput (msgs/sec)
|
||||
- End-to-end latency
|
||||
- Resource utilization (CPU, Memory, Disk)
|
||||
- Network bandwidth usage
|
||||
|
||||
## Test Execution Strategy
|
||||
|
||||
### Parallel Test Execution
|
||||
- Categorize tests by resource requirements
|
||||
- Run independent tests in parallel
|
||||
- Serialize tests that modify cluster state
|
||||
|
||||
### Continuous Integration
|
||||
- Automated test runs on PR submissions
|
||||
- Performance regression detection
|
||||
- Multi-platform testing (Linux, macOS, Windows)
|
||||
|
||||
### Test Environment Management
|
||||
- Docker-based isolated environments
|
||||
- Automatic cleanup after test completion
|
||||
- Resource monitoring and alerts
|
||||
|
||||
## Success Criteria
|
||||
|
||||
### Functional Requirements
|
||||
- ✅ All messages published are received by subscribers
|
||||
- ✅ Message ordering preserved within partitions
|
||||
- ✅ Schema validation works correctly
|
||||
- ✅ Auto-scaling triggers at expected thresholds
|
||||
- ✅ Failover completes within 30 seconds
|
||||
- ✅ No data loss during normal operations
|
||||
|
||||
### Performance Requirements
|
||||
- ✅ Throughput: 50K+ messages/second/broker
|
||||
- ✅ Latency: P95 < 100ms end-to-end
|
||||
- ✅ Memory usage: < 1GB per broker under normal load
|
||||
- ✅ Storage efficiency: < 20% overhead vs raw message size
|
||||
|
||||
### Reliability Requirements
|
||||
- ✅ 99.9% uptime during normal operations
|
||||
- ✅ Automatic recovery from single component failures
|
||||
- ✅ Data consistency maintained across all scenarios
|
||||
- ✅ Graceful degradation under resource constraints
|
||||
|
||||
## Implementation Timeline
|
||||
|
||||
### Phase 1: Core Functionality (Week 1-2)
|
||||
- Basic pub/sub tests
|
||||
- Schema validation tests
|
||||
- Simple failover scenarios
|
||||
|
||||
### Phase 2: Advanced Features (Week 3-4)
|
||||
- Auto-scaling tests
|
||||
- Complex failover scenarios
|
||||
- Agent functionality tests
|
||||
|
||||
### Phase 3: Performance & Load (Week 5-6)
|
||||
- Throughput and latency tests
|
||||
- Spike traffic handling
|
||||
- Resource utilization monitoring
|
||||
|
||||
### Phase 4: End-to-End (Week 7-8)
|
||||
- Complete workflow tests
|
||||
- Multi-component integration
|
||||
- Performance regression testing
|
||||
|
||||
## Maintenance and Updates
|
||||
|
||||
### Regular Updates
|
||||
- Add tests for new features
|
||||
- Update performance baselines
|
||||
- Enhance error scenarios coverage
|
||||
|
||||
### Test Data Refresh
|
||||
- Generate new test datasets quarterly
|
||||
- Update schema examples
|
||||
- Refresh performance benchmarks
|
||||
|
||||
This comprehensive test design ensures SeaweedMQ's reliability, performance, and functionality across all critical use cases and failure scenarios.
|
||||
@@ -0,0 +1,54 @@
|
||||
global:
|
||||
scrape_interval: 15s
|
||||
evaluation_interval: 15s
|
||||
|
||||
rule_files:
|
||||
# - "first_rules.yml"
|
||||
# - "second_rules.yml"
|
||||
|
||||
scrape_configs:
|
||||
# SeaweedFS Masters
|
||||
- job_name: 'seaweedfs-master'
|
||||
static_configs:
|
||||
- targets:
|
||||
- 'master0:9333'
|
||||
- 'master1:9334'
|
||||
- 'master2:9335'
|
||||
metrics_path: '/metrics'
|
||||
scrape_interval: 10s
|
||||
|
||||
# SeaweedFS Volume Servers
|
||||
- job_name: 'seaweedfs-volume'
|
||||
static_configs:
|
||||
- targets:
|
||||
- 'volume1:8080'
|
||||
- 'volume2:8081'
|
||||
- 'volume3:8082'
|
||||
metrics_path: '/metrics'
|
||||
scrape_interval: 10s
|
||||
|
||||
# SeaweedFS Filers
|
||||
- job_name: 'seaweedfs-filer'
|
||||
static_configs:
|
||||
- targets:
|
||||
- 'filer1:8888'
|
||||
- 'filer2:8889'
|
||||
metrics_path: '/metrics'
|
||||
scrape_interval: 10s
|
||||
|
||||
# SeaweedMQ Brokers
|
||||
- job_name: 'seaweedmq-broker'
|
||||
static_configs:
|
||||
- targets:
|
||||
- 'broker1:17777'
|
||||
- 'broker2:17778'
|
||||
- 'broker3:17779'
|
||||
metrics_path: '/metrics'
|
||||
scrape_interval: 5s
|
||||
|
||||
# Docker containers
|
||||
- job_name: 'docker'
|
||||
static_configs:
|
||||
- targets: ['localhost:9323']
|
||||
metrics_path: '/metrics'
|
||||
scrape_interval: 30s
|
||||
@@ -0,0 +1,56 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"log"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/mq/client/pub_client"
|
||||
"github.com/seaweedfs/seaweedfs/weed/mq/topic"
|
||||
)
|
||||
|
||||
func main() {
|
||||
log.Println("Starting SeaweedMQ logging test...")
|
||||
|
||||
// Create publisher configuration
|
||||
config := &pub_client.PublisherConfiguration{
|
||||
Topic: topic.NewTopic("test", "logging-demo"),
|
||||
PartitionCount: 3,
|
||||
Brokers: []string{"127.0.0.1:17777"},
|
||||
PublisherName: "logging-test-client",
|
||||
}
|
||||
|
||||
log.Println("Creating topic publisher...")
|
||||
publisher, err := pub_client.NewTopicPublisher(config)
|
||||
if err != nil {
|
||||
log.Printf("Failed to create publisher: %v", err)
|
||||
return
|
||||
}
|
||||
defer publisher.Shutdown()
|
||||
|
||||
log.Println("Publishing test messages...")
|
||||
|
||||
// Publish some test messages
|
||||
for i := 0; i < 100; i++ {
|
||||
key := fmt.Sprintf("key-%d", i)
|
||||
value := fmt.Sprintf("message-%d-timestamp-%d", i, time.Now().Unix())
|
||||
|
||||
err := publisher.Publish([]byte(key), []byte(value))
|
||||
if err != nil {
|
||||
log.Printf("Failed to publish message %d: %v", i, err)
|
||||
}
|
||||
|
||||
// Small delay to create some connection stress
|
||||
if i%10 == 0 {
|
||||
time.Sleep(10 * time.Millisecond)
|
||||
}
|
||||
}
|
||||
|
||||
log.Println("Finishing publish...")
|
||||
err = publisher.FinishPublish()
|
||||
if err != nil {
|
||||
log.Printf("Failed to finish publish: %v", err)
|
||||
}
|
||||
|
||||
log.Println("Test completed successfully!")
|
||||
}
|
||||
@@ -32,7 +32,7 @@ func init() {
|
||||
|
||||
var cmdMqAgent = &Command{
|
||||
UsageLine: "mq.agent [-port=16777] [-broker=<ip:port>]",
|
||||
Short: "<WIP> start a message queue agent",
|
||||
Short: "start a message queue agent",
|
||||
Long: `start a message queue agent
|
||||
|
||||
The agent runs on local server to accept gRPC calls to write or read messages.
|
||||
|
||||
@@ -43,7 +43,7 @@ func init() {
|
||||
|
||||
var cmdMqBroker = &Command{
|
||||
UsageLine: "mq.broker [-port=17777] [-master=<ip:port>]",
|
||||
Short: "<WIP> start a message queue broker",
|
||||
Short: "start a message queue broker",
|
||||
Long: `start a message queue broker
|
||||
|
||||
The broker can accept gRPC calls to write or read messages. The messages are stored via filer.
|
||||
|
||||
@@ -3,15 +3,19 @@ package broker
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"github.com/seaweedfs/seaweedfs/weed/glog"
|
||||
"github.com/seaweedfs/seaweedfs/weed/mq/topic"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/mq_pb"
|
||||
"google.golang.org/grpc/peer"
|
||||
"io"
|
||||
"math/rand"
|
||||
"net"
|
||||
"strings"
|
||||
"sync/atomic"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/glog"
|
||||
"github.com/seaweedfs/seaweedfs/weed/mq/topic"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/mq_pb"
|
||||
"google.golang.org/grpc/codes"
|
||||
"google.golang.org/grpc/peer"
|
||||
"google.golang.org/grpc/status"
|
||||
)
|
||||
|
||||
// PUB
|
||||
@@ -126,7 +130,12 @@ func (b *MessageQueueBroker) PublishMessage(stream mq_pb.SeaweedMessaging_Publis
|
||||
if err == io.EOF {
|
||||
break
|
||||
}
|
||||
glog.V(0).Infof("topic %v partition %v publish stream from %s error: %v", initMessage.Topic, initMessage.Partition, initMessage.PublisherName, err)
|
||||
// Use appropriate logging level based on error type
|
||||
if isConnectionError(err) {
|
||||
glog.V(1).Infof("topic %v partition %v publish stream from %s connection error (client disconnected): %v", initMessage.Topic, initMessage.Partition, initMessage.PublisherName, err)
|
||||
} else {
|
||||
glog.V(0).Infof("topic %v partition %v publish stream from %s error: %v", initMessage.Topic, initMessage.Partition, initMessage.PublisherName, err)
|
||||
}
|
||||
break
|
||||
}
|
||||
|
||||
@@ -145,7 +154,7 @@ func (b *MessageQueueBroker) PublishMessage(stream mq_pb.SeaweedMessaging_Publis
|
||||
}
|
||||
}
|
||||
|
||||
glog.V(0).Infof("topic %v partition %v publish stream from %s closed.", initMessage.Topic, initMessage.Partition, initMessage.PublisherName)
|
||||
glog.V(1).Infof("topic %v partition %v publish stream from %s closed.", initMessage.Topic, initMessage.Partition, initMessage.PublisherName)
|
||||
|
||||
return nil
|
||||
}
|
||||
@@ -164,3 +173,40 @@ func findClientAddress(ctx context.Context) string {
|
||||
}
|
||||
return pr.Addr.String()
|
||||
}
|
||||
|
||||
// isConnectionError checks if an error is a connection-level error
|
||||
func isConnectionError(err error) bool {
|
||||
if err == nil {
|
||||
return false
|
||||
}
|
||||
|
||||
// Check gRPC status codes for connection errors
|
||||
if grpcStatus, ok := status.FromError(err); ok {
|
||||
switch grpcStatus.Code() {
|
||||
case codes.Unavailable, codes.DeadlineExceeded, codes.Canceled, codes.Unknown:
|
||||
return true
|
||||
}
|
||||
}
|
||||
|
||||
// Check for common connection error patterns
|
||||
errStr := strings.ToLower(err.Error())
|
||||
connectionErrorPatterns := []string{
|
||||
"eof",
|
||||
"error reading server preface",
|
||||
"connection refused",
|
||||
"connection reset",
|
||||
"broken pipe",
|
||||
"no such host",
|
||||
"network is unreachable",
|
||||
"context deadline exceeded",
|
||||
"context canceled",
|
||||
}
|
||||
|
||||
for _, pattern := range connectionErrorPatterns {
|
||||
if strings.Contains(errStr, pattern) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
|
||||
return false
|
||||
}
|
||||
|
||||
@@ -2,13 +2,14 @@ package broker
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"io"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/glog"
|
||||
"github.com/seaweedfs/seaweedfs/weed/mq/topic"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/mq_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util/buffered_queue"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util/log_buffer"
|
||||
"io"
|
||||
"time"
|
||||
)
|
||||
|
||||
type memBuffer struct {
|
||||
@@ -43,7 +44,12 @@ func (b *MessageQueueBroker) PublishFollowMe(stream mq_pb.SeaweedMessaging_Publi
|
||||
err = nil
|
||||
break
|
||||
}
|
||||
glog.V(0).Infof("topic %v partition %v publish stream error: %v", initMessage.Topic, initMessage.Partition, err)
|
||||
// Use appropriate logging level based on error type
|
||||
if isConnectionError(err) {
|
||||
glog.V(1).Infof("topic %v partition %v publish follower stream connection error (leader disconnected): %v", initMessage.Topic, initMessage.Partition, err)
|
||||
} else {
|
||||
glog.V(0).Infof("topic %v partition %v publish stream error: %v", initMessage.Topic, initMessage.Partition, err)
|
||||
}
|
||||
break
|
||||
}
|
||||
|
||||
@@ -62,10 +68,10 @@ func (b *MessageQueueBroker) PublishFollowMe(stream mq_pb.SeaweedMessaging_Publi
|
||||
}
|
||||
// println("ack", string(dataMessage.Key), dataMessage.TsNs)
|
||||
} else if closeMessage := req.GetClose(); closeMessage != nil {
|
||||
glog.V(0).Infof("topic %v partition %v publish stream closed: %v", initMessage.Topic, initMessage.Partition, closeMessage)
|
||||
glog.V(1).Infof("topic %v partition %v publish stream closed: %v", initMessage.Topic, initMessage.Partition, closeMessage)
|
||||
break
|
||||
} else if flushMessage := req.GetFlush(); flushMessage != nil {
|
||||
glog.V(0).Infof("topic %v partition %v publish stream flushed: %v", initMessage.Topic, initMessage.Partition, flushMessage)
|
||||
glog.V(1).Infof("topic %v partition %v publish stream flushed: %v", initMessage.Topic, initMessage.Partition, flushMessage)
|
||||
|
||||
lastFlushTsNs = flushMessage.TsNs
|
||||
|
||||
@@ -124,7 +130,7 @@ func (b *MessageQueueBroker) PublishFollowMe(stream mq_pb.SeaweedMessaging_Publi
|
||||
glog.V(0).Infof("flushed remaining data at %v to %s size %d", mem.stopTime.UnixNano(), targetFile, len(mem.buf))
|
||||
}
|
||||
|
||||
glog.V(0).Infof("shut down follower for %v %v", t, p)
|
||||
glog.V(1).Infof("shut down follower for %v %v", t, p)
|
||||
|
||||
return err
|
||||
}
|
||||
|
||||
@@ -3,6 +3,13 @@ package pub_client
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"log"
|
||||
"sort"
|
||||
"strings"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/glog"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/mq_pb"
|
||||
@@ -11,11 +18,6 @@ import (
|
||||
"google.golang.org/grpc/codes"
|
||||
"google.golang.org/grpc/credentials/insecure"
|
||||
"google.golang.org/grpc/status"
|
||||
"log"
|
||||
"sort"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"time"
|
||||
)
|
||||
|
||||
type EachPartitionError struct {
|
||||
@@ -39,7 +41,7 @@ func (p *TopicPublisher) startSchedulerThread(wg *sync.WaitGroup) error {
|
||||
return fmt.Errorf("configure topic %s: %v", p.config.Topic, err)
|
||||
}
|
||||
|
||||
log.Printf("start scheduler thread for topic %s", p.config.Topic)
|
||||
glog.V(0).Infof("start scheduler thread for topic %s", p.config.Topic)
|
||||
|
||||
generation := 0
|
||||
var errChan chan EachPartitionError
|
||||
@@ -66,7 +68,12 @@ func (p *TopicPublisher) startSchedulerThread(wg *sync.WaitGroup) error {
|
||||
for {
|
||||
select {
|
||||
case eachErr := <-errChan:
|
||||
glog.Errorf("gen %d publish to topic %s partition %v: %v", eachErr.generation, p.config.Topic, eachErr.Partition, eachErr.Err)
|
||||
// Use appropriate logging level based on error type
|
||||
if isConnectionError(eachErr.Err) {
|
||||
glog.V(1).Infof("gen %d connection error for topic %s partition %v (will retry): %v", eachErr.generation, p.config.Topic, eachErr.Partition, eachErr.Err)
|
||||
} else {
|
||||
glog.Errorf("gen %d publish to topic %s partition %v: %v", eachErr.generation, p.config.Topic, eachErr.Partition, eachErr.Err)
|
||||
}
|
||||
if eachErr.generation < generation {
|
||||
continue
|
||||
}
|
||||
@@ -114,7 +121,12 @@ func (p *TopicPublisher) onEachAssignments(generation int, assignments []*mq_pb.
|
||||
go func(job *EachPartitionPublishJob) {
|
||||
defer job.wg.Done()
|
||||
if err := p.doPublishToPartition(job); err != nil {
|
||||
log.Printf("publish to %s partition %v: %v", p.config.Topic, job.Partition, err)
|
||||
// Use appropriate logging level based on error type
|
||||
if isConnectionError(err) {
|
||||
glog.V(1).Infof("connection error publishing to %s partition %v (will retry): %v", p.config.Topic, job.Partition, err)
|
||||
} else {
|
||||
log.Printf("publish to %s partition %v: %v", p.config.Topic, job.Partition, err)
|
||||
}
|
||||
errChan <- EachPartitionError{assignment, err, generation}
|
||||
}
|
||||
}(job)
|
||||
@@ -128,7 +140,7 @@ func (p *TopicPublisher) onEachAssignments(generation int, assignments []*mq_pb.
|
||||
|
||||
func (p *TopicPublisher) doPublishToPartition(job *EachPartitionPublishJob) error {
|
||||
|
||||
log.Printf("connecting to %v for topic partition %+v", job.LeaderBroker, job.Partition)
|
||||
glog.V(1).Infof("connecting to %v for topic partition %+v", job.LeaderBroker, job.Partition)
|
||||
|
||||
grpcConnection, err := grpc.NewClient(job.LeaderBroker, grpc.WithTransportCredentials(insecure.NewCredentials()), p.grpcDialOption)
|
||||
if err != nil {
|
||||
@@ -176,20 +188,25 @@ func (p *TopicPublisher) doPublishToPartition(job *EachPartitionPublishJob) erro
|
||||
if err != nil {
|
||||
e, _ := status.FromError(err)
|
||||
if e.Code() == codes.Unknown && e.Message() == "EOF" {
|
||||
log.Printf("publish to %s EOF", publishClient.Broker)
|
||||
glog.V(1).Infof("publish stream to %s closed (EOF)", publishClient.Broker)
|
||||
return
|
||||
}
|
||||
publishClient.Err = err
|
||||
log.Printf("publish1 to %s error: %v\n", publishClient.Broker, err)
|
||||
// Use appropriate logging level based on error type
|
||||
if isConnectionError(err) {
|
||||
glog.V(1).Infof("connection error receiving from %s (will retry): %v", publishClient.Broker, err)
|
||||
} else {
|
||||
log.Printf("publish receive error from %s: %v", publishClient.Broker, err)
|
||||
}
|
||||
return
|
||||
}
|
||||
if ackResp.Error != "" {
|
||||
publishClient.Err = fmt.Errorf("ack error: %v", ackResp.Error)
|
||||
log.Printf("publish2 to %s error: %v\n", publishClient.Broker, ackResp.Error)
|
||||
log.Printf("publish ack error from %s: %v", publishClient.Broker, ackResp.Error)
|
||||
return
|
||||
}
|
||||
if ackResp.AckSequence > 0 {
|
||||
log.Printf("ack %d published %d hasMoreData:%d", ackResp.AckSequence, atomic.LoadInt64(&publishedTsNs), atomic.LoadInt32(&hasMoreData))
|
||||
glog.V(2).Infof("ack %d published %d hasMoreData:%d", ackResp.AckSequence, atomic.LoadInt64(&publishedTsNs), atomic.LoadInt32(&hasMoreData))
|
||||
}
|
||||
if atomic.LoadInt64(&publishedTsNs) <= ackResp.AckSequence && atomic.LoadInt32(&hasMoreData) == 0 {
|
||||
return
|
||||
@@ -222,7 +239,7 @@ func (p *TopicPublisher) doPublishToPartition(job *EachPartitionPublishJob) erro
|
||||
}
|
||||
}
|
||||
|
||||
log.Printf("published %d messages to %v for topic partition %+v", publishCounter, job.LeaderBroker, job.Partition)
|
||||
glog.V(1).Infof("published %d messages to %v for topic partition %+v", publishCounter, job.LeaderBroker, job.Partition)
|
||||
|
||||
return nil
|
||||
}
|
||||
@@ -296,3 +313,40 @@ func (p *TopicPublisher) doLookupTopicPartitions() (assignments []*mq_pb.BrokerP
|
||||
return nil, fmt.Errorf("lookup topic %s: %v", p.config.Topic, lastErr)
|
||||
|
||||
}
|
||||
|
||||
// isConnectionError checks if an error is a connection-level error that is handled automatically
|
||||
func isConnectionError(err error) bool {
|
||||
if err == nil {
|
||||
return false
|
||||
}
|
||||
|
||||
// Check gRPC status codes for connection errors
|
||||
if grpcStatus, ok := status.FromError(err); ok {
|
||||
switch grpcStatus.Code() {
|
||||
case codes.Unavailable, codes.DeadlineExceeded, codes.Canceled, codes.Unknown:
|
||||
return true
|
||||
}
|
||||
}
|
||||
|
||||
// Check for common connection error patterns
|
||||
errStr := strings.ToLower(err.Error())
|
||||
connectionErrorPatterns := []string{
|
||||
"eof",
|
||||
"error reading server preface",
|
||||
"connection refused",
|
||||
"connection reset",
|
||||
"broken pipe",
|
||||
"no such host",
|
||||
"network is unreachable",
|
||||
"context deadline exceeded",
|
||||
"context canceled",
|
||||
}
|
||||
|
||||
for _, pattern := range connectionErrorPatterns {
|
||||
if strings.Contains(errStr, pattern) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
|
||||
return false
|
||||
}
|
||||
|
||||
@@ -3,6 +3,11 @@ package topic
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"strings"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/glog"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/mq_pb"
|
||||
@@ -10,9 +15,6 @@ import (
|
||||
"google.golang.org/grpc"
|
||||
"google.golang.org/grpc/codes"
|
||||
"google.golang.org/grpc/status"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"time"
|
||||
)
|
||||
|
||||
type LocalPartition struct {
|
||||
@@ -182,7 +184,12 @@ func (p *LocalPartition) MaybeConnectToFollowers(initMessage *mq_pb.PublishMessa
|
||||
glog.V(0).Infof("local partition %v follower %v stopped", p.Partition, p.Follower)
|
||||
return
|
||||
}
|
||||
glog.Errorf("Receiving local partition %v follower %s ack: %v", p.Partition, p.Follower, err)
|
||||
// Use appropriate logging level based on error type
|
||||
if isConnectionError(err) {
|
||||
glog.V(1).Infof("local partition %v follower %s connection error (will retry): %v", p.Partition, p.Follower, err)
|
||||
} else {
|
||||
glog.Errorf("Receiving local partition %v follower %s ack: %v", p.Partition, p.Follower, err)
|
||||
}
|
||||
return
|
||||
}
|
||||
atomic.StoreInt64(&p.AckTsNs, ack.AckTsNs)
|
||||
@@ -242,3 +249,40 @@ func (p *LocalPartition) NotifyLogFlushed(flushTsNs int64) {
|
||||
// println("notifying", p.Follower, "flushed at", flushTsNs)
|
||||
}
|
||||
}
|
||||
|
||||
// isConnectionError checks if an error is a connection-level error
|
||||
func isConnectionError(err error) bool {
|
||||
if err == nil {
|
||||
return false
|
||||
}
|
||||
|
||||
// Check gRPC status codes for connection errors
|
||||
if grpcStatus, ok := status.FromError(err); ok {
|
||||
switch grpcStatus.Code() {
|
||||
case codes.Unavailable, codes.DeadlineExceeded, codes.Canceled, codes.Unknown:
|
||||
return true
|
||||
}
|
||||
}
|
||||
|
||||
// Check for common connection error patterns
|
||||
errStr := strings.ToLower(err.Error())
|
||||
connectionErrorPatterns := []string{
|
||||
"eof",
|
||||
"error reading server preface",
|
||||
"connection refused",
|
||||
"connection reset",
|
||||
"broken pipe",
|
||||
"no such host",
|
||||
"network is unreachable",
|
||||
"context deadline exceeded",
|
||||
"context canceled",
|
||||
}
|
||||
|
||||
for _, pattern := range connectionErrorPatterns {
|
||||
if strings.Contains(errStr, pattern) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
|
||||
return false
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user