Compare commits

...
Author SHA1 Message Date
chrislu e4fe657bd9 perf tests run fine 2025-06-25 17:42:39 -07:00
chrislu 31473fbe10 fix filer starting 2025-06-25 17:41:47 -07:00
chrislu 618235bd72 fix error logging to reduce noise 2025-06-25 17:31:28 -07:00
chrislu 49e1e56aa1 remove WIP 2025-06-25 16:22:31 -07:00
chrislu ad6a4ef9b2 make test-basic 2025-06-24 09:35:08 -07:00
chrislu 46602a229c make test-basic-native 2025-06-23 21:08:32 -07:00
chrislu b0290dc33d build linux arm64 2025-06-23 19:59:36 -07:00
chrislu 914810fd13 fix other files 2025-06-23 19:53:07 -07:00
chrislu 2c6ed03823 fix docker compose 2025-06-23 19:50:59 -07:00
chrislu 6a97079454 running 2025-06-23 11:20:58 -07:00
chrislu 1c879148d8 Merge branch 'master' into adding-message-queue-integration-tests 2025-06-23 10:55:13 -07:00
chrislu 0347212b64 init version 2025-06-23 10:55:02 -07:00
19 changed files with 3739 additions and 32 deletions
+38
View File
@@ -0,0 +1,38 @@
FROM golang:1.24-alpine
# Install necessary tools
RUN apk add --no-cache \
curl \
netcat-openbsd \
bash \
git \
build-base \
ca-certificates
# Set working directory
WORKDIR /workspace
# Copy go mod files first for better caching
COPY go.mod go.sum ./
RUN go mod download
# Copy only the necessary source code for building and testing
COPY weed/ ./weed/
COPY test/mq/integration/ ./test/mq/integration/
# Build the weed binary for testing (optional - tests can use external cluster)
RUN cd weed && CGO_ENABLED=1 GOOS=linux go build -o ../weed .
# Create test results directory
RUN mkdir -p /test-results
# Set up environment
ENV CGO_ENABLED=1
ENV GOOS=linux
ENV GO111MODULE=on
# Default working directory for tests
WORKDIR /workspace
# Entry point for running tests - use explicit bash
ENTRYPOINT ["/bin/bash", "-c"]
+223
View File
@@ -0,0 +1,223 @@
.PHONY: help build test test-basic test-performance test-failover test-agent clean up down logs
# Detect architecture and Docker platform compatibility
ARCH := $(shell uname -m)
OS := $(shell uname -s)
ifeq ($(ARCH),arm64)
ifeq ($(OS),Darwin)
# On Apple Silicon macOS, use native arm64 for better performance
DOCKER_PLATFORM := linux/arm64
else
DOCKER_PLATFORM := linux/arm64
endif
else
DOCKER_PLATFORM := linux/amd64
endif
# Default target
help:
@echo "SeaweedMQ Integration Test Suite"
@echo ""
@echo "Available targets:"
@echo " build - Build SeaweedFS Docker images"
@echo " test - Run all integration tests (in Docker)"
@echo " test-basic - Run basic pub/sub tests (in Docker)"
@echo " test-native - Run all tests natively (no Docker test container)"
@echo " test-basic-native - Run basic tests natively (recommended for Apple Silicon)"
@echo " test-performance - Run performance tests"
@echo " test-failover - Run failover tests"
@echo " test-agent - Run agent tests"
@echo " up - Start test environment (local build)"
@echo " up-prod - Start test environment (production images)"
@echo " up-cluster - Start cluster only (no test runner)"
@echo " down - Stop test environment"
@echo " clean - Clean up test environment and results"
@echo " logs - Show container logs"
# Build SeaweedFS Docker images
build:
@echo "Building SeaweedFS Docker image for $(DOCKER_PLATFORM)..."
cd ../.. && docker build --platform $(DOCKER_PLATFORM) -f docker/Dockerfile.go_build -t chrislusf/seaweedfs:local .
@echo "Building test runner image for $(DOCKER_PLATFORM)..."
cd ../.. && docker build --platform $(DOCKER_PLATFORM) -f test/mq/Dockerfile.test -t seaweedfs-test-runner .
# Start the test environment
up: build
@echo "Starting SeaweedMQ test environment..."
docker-compose -f docker-compose.test.yml up -d master0 master1 master2
@echo "Waiting for masters to be ready..."
sleep 10
docker-compose -f docker-compose.test.yml up -d volume1 volume2 volume3
@echo "Waiting for volumes to be ready..."
sleep 10
docker-compose -f docker-compose.test.yml up -d filer1 filer2
@echo "Waiting for filers to be ready..."
sleep 15
docker-compose -f docker-compose.test.yml up -d broker1 broker2 broker3
@echo "Waiting for brokers to be ready..."
sleep 20
@echo "Test environment is ready!"
# Start the test environment with production images (no build required)
up-prod: build-test-runner
@echo "Starting SeaweedMQ test environment with production images..."
docker-compose -f docker-compose.production.yml up -d master0 master1 master2
@echo "Waiting for masters to be ready..."
sleep 10
docker-compose -f docker-compose.production.yml up -d volume1 volume2 volume3
@echo "Waiting for volumes to be ready..."
sleep 10
docker-compose -f docker-compose.production.yml up -d filer1 filer2
@echo "Waiting for filers to be ready..."
sleep 15
docker-compose -f docker-compose.production.yml up -d broker1 broker2 broker3
@echo "Waiting for brokers to be ready..."
sleep 20
@echo "Test environment is ready!"
# Build only the test runner image (for production setup)
build-test-runner:
@echo "Building test runner image for $(DOCKER_PLATFORM)..."
cd ../.. && docker build --platform $(DOCKER_PLATFORM) -f test/mq/Dockerfile.test -t seaweedfs-test-runner .
# Start cluster only (no test runner, no build required)
up-cluster:
@echo "Starting SeaweedMQ cluster only..."
docker-compose -f docker-compose.cluster.yml up -d master0 master1 master2
@echo "Waiting for masters to be ready..."
sleep 10
docker-compose -f docker-compose.cluster.yml up -d volume1 volume2 volume3
@echo "Waiting for volumes to be ready..."
sleep 10
docker-compose -f docker-compose.cluster.yml up -d filer1 filer2
@echo "Waiting for filers to be ready..."
sleep 15
docker-compose -f docker-compose.cluster.yml up -d broker1 broker2 broker3
@echo "Waiting for brokers to be ready..."
sleep 20
@echo "SeaweedMQ cluster is ready!"
@echo "Masters: http://localhost:19333, http://localhost:19334, http://localhost:19335"
@echo "Filers: http://localhost:18888, http://localhost:18889"
@echo "Brokers: localhost:17777, localhost:17778, localhost:17779"
# Stop the test environment
down:
@echo "Stopping SeaweedMQ test environment..."
docker-compose -f docker-compose.test.yml down
docker-compose -f docker-compose.production.yml down
docker-compose -f docker-compose.cluster.yml down
# Clean up everything
clean:
@echo "Cleaning up test environment..."
docker-compose -f docker-compose.test.yml down -v
docker system prune -f
sudo rm -rf /tmp/test-results/*
# Show container logs
logs:
docker-compose -f docker-compose.test.yml logs -f
# Run all integration tests
test:
@echo "Running all integration tests..."
docker-compose -f docker-compose.test.yml run --rm test-runner \
sh -c "go test -v -timeout=30m ./test/mq/integration/... -args -test.parallel=4"
# Run basic pub/sub tests
test-basic:
@echo "Running basic pub/sub tests natively (no container restart)..."
cd ../.. && SEAWEED_MASTERS="localhost:19333,localhost:19334,localhost:19335" \
SEAWEED_BROKERS="localhost:17777,localhost:17778,localhost:17779" \
SEAWEED_FILERS="localhost:18888,localhost:18889" \
go test -v -timeout=10m ./test/mq/integration/ -run TestBasic
# Run performance tests
test-performance:
@echo "Running performance tests..."
docker-compose -f docker-compose.test.yml run --rm test-runner \
sh -c "go test -v -timeout=20m ./test/mq/integration/ -run TestPerformance"
# Run failover tests
test-failover:
@echo "Running failover tests..."
docker-compose -f docker-compose.test.yml run --rm test-runner \
sh -c "go test -v -timeout=15m ./test/mq/integration/ -run TestFailover"
# Run agent tests
test-agent:
@echo "Running agent tests..."
docker-compose -f docker-compose.test.yml run --rm test-runner \
sh -c "go test -v -timeout=10m ./test/mq/integration/ -run TestAgent"
# Development targets (run tests natively without Docker container)
test-dev:
@echo "Running tests in development mode (using local binaries)..."
SEAWEED_MASTERS="localhost:19333,localhost:19334,localhost:19335" \
SEAWEED_BROKERS="localhost:17777,localhost:17778,localhost:17779" \
SEAWEED_FILERS="localhost:18888,localhost:18889" \
go test -v -timeout=10m ./integration/...
# Native test running (no Docker container for tests)
test-native:
@echo "Running tests natively (without Docker container for tests)..."
cd ../.. && SEAWEED_MASTERS="localhost:19333,localhost:19334,localhost:19335" \
SEAWEED_BROKERS="localhost:17777,localhost:17778,localhost:17779" \
SEAWEED_FILERS="localhost:18888,localhost:18889" \
go test -v -timeout=10m ./test/mq/integration/...
# Basic native tests
test-basic-native:
@echo "Running basic tests natively..."
cd ../.. && SEAWEED_MASTERS="localhost:19333,localhost:19334,localhost:19335" \
SEAWEED_BROKERS="localhost:17777,localhost:17778,localhost:17779" \
SEAWEED_FILERS="localhost:18888,localhost:18889" \
go test -v -timeout=10m ./test/mq/integration/ -run TestBasic
# Quick smoke test
smoke-test:
@echo "Running smoke test..."
docker-compose -f docker-compose.test.yml run --rm test-runner \
sh -c "go test -v -timeout=5m ./test/mq/integration/ -run TestBasicPublishSubscribe"
# Performance benchmarks
benchmark:
@echo "Running performance benchmarks..."
docker-compose -f docker-compose.test.yml run --rm test-runner \
sh -c "go test -v -timeout=30m -bench=. ./test/mq/integration/..."
# Check test environment health
health:
@echo "Checking test environment health..."
@echo "Masters:"
@curl -s http://localhost:19333/cluster/status || echo "Master 0 not accessible"
@curl -s http://localhost:19334/cluster/status || echo "Master 1 not accessible"
@curl -s http://localhost:19335/cluster/status || echo "Master 2 not accessible"
@echo ""
@echo "Filers:"
@curl -s http://localhost:18888/ || echo "Filer 1 not accessible"
@curl -s http://localhost:18889/ || echo "Filer 2 not accessible"
@echo ""
@echo "Brokers:"
@nc -z localhost 17777 && echo "Broker 1 accessible" || echo "Broker 1 not accessible"
@nc -z localhost 17778 && echo "Broker 2 accessible" || echo "Broker 2 not accessible"
@nc -z localhost 17779 && echo "Broker 3 accessible" || echo "Broker 3 not accessible"
# Generate test reports
report:
@echo "Generating test reports..."
docker-compose -f docker-compose.test.yml run --rm test-runner \
sh -c "go test -v -timeout=30m ./test/mq/integration/... -json > /test-results/test-report.json"
# Load testing
load-test:
@echo "Running load tests..."
docker-compose -f docker-compose.test.yml run --rm test-runner \
sh -c "go test -v -timeout=45m ./test/mq/integration/ -run TestLoad"
# View monitoring dashboards
monitoring:
@echo "Starting monitoring stack..."
docker-compose -f docker-compose.test.yml up -d prometheus grafana
@echo "Prometheus: http://localhost:19090"
@echo "Grafana: http://localhost:13000 (admin/admin)"
+414
View File
@@ -0,0 +1,414 @@
# SeaweedMQ Integration Test Suite
This directory contains a comprehensive integration test suite for SeaweedMQ, designed to validate all critical functionalities from basic pub/sub operations to advanced features like auto-scaling, failover, and performance testing.
## Overview
The integration test suite provides:
- **Automated Environment Setup**: Docker Compose based test clusters
- **Comprehensive Test Coverage**: Basic pub/sub, scaling, failover, performance
- **Monitoring & Metrics**: Prometheus and Grafana integration
- **CI/CD Ready**: Configurable for continuous integration pipelines
- **Load Testing**: Performance benchmarks and stress tests
## Architecture
The test environment consists of:
```
┌─────────────────┐ ┌─────────────────┐ ┌─────────────────┐
│ Master Cluster │ │ Volume Servers │ │ Filer Cluster │
│ (3 nodes) │ │ (3 nodes) │ │ (2 nodes) │
└─────────────────┘ └─────────────────┘ └─────────────────┘
│ │ │
└───────────────────────┼───────────────────────┘
│
┌─────────────────┐ ┌─────────────────┐ ┌─────────────────┐
│ Broker Cluster │ │ Test Framework │ │ Monitoring │
│ (3 nodes) │ │ (Go Tests) │ │ (Prometheus + │
└─────────────────┘ └─────────────────┘ │ Grafana) │
└─────────────────┘
```
## Quick Start
### Prerequisites
- Docker and Docker Compose
- Go 1.21+
- Make
- 8GB+ RAM recommended
- 20GB+ disk space for test data
### Basic Usage
Choose one of three approaches:
#### Option 1: Quick Cluster Setup (Fastest)
Just starts a SeaweedMQ cluster - no test runner, no build required.
1. **Start Cluster**:
```bash
cd test/mq
make up-cluster
```
2. **Test manually** or with your own code against:
- Masters: http://localhost:19333, http://localhost:19334, http://localhost:19335
- Brokers: localhost:17777, localhost:17778, localhost:17779
- Filers: http://localhost:18888, http://localhost:18889
#### Option 2: Use Production Images (Recommended for testing)
Uses official SeaweedFS images from Docker Hub with test runner.
1. **Start Test Environment**:
```bash
cd test/mq
make up-prod
```
2. **Run Tests**:
```bash
make test-basic # Basic pub/sub tests
```
#### Option 3: Build from Source (Development)
Builds SeaweedFS from your local source code to test latest changes.
1. **Start Test Environment**:
```bash
cd test/mq
make up # This will build and start
```
2. **Run Tests**:
```bash
make test-basic # Basic pub/sub tests
```
#### Common Commands
3. **Run All Tests**:
```bash
make test
```
4. **Run Specific Test Categories**:
```bash
make test-performance # Performance tests
make test-failover # Failover tests
make test-agent # Agent tests
```
5. **Quick Smoke Test**:
```bash
make smoke-test
```
6. **Check Health**:
```bash
make health
```
7. **Clean Up**:
```bash
make down
```
## Test Categories
### 1. Basic Functionality Tests
**File**: `integration/basic_pubsub_test.go`
- **TestBasicPublishSubscribe**: Basic message publishing and consumption
- **TestMultipleConsumers**: Load balancing across multiple consumers
- **TestMessageOrdering**: FIFO ordering within partitions
- **TestSchemaValidation**: Schema validation and complex nested structures
### 2. Partitioning and Scaling Tests
**File**: `integration/scaling_test.go` (to be implemented)
- **TestPartitionDistribution**: Message distribution across partitions
- **TestAutoSplitMerge**: Automatic partition split/merge based on load
- **TestBrokerScaling**: Adding/removing brokers during operation
- **TestLoadBalancing**: Even load distribution verification
### 3. Failover and Reliability Tests
**File**: `integration/failover_test.go` (to be implemented)
- **TestBrokerFailover**: Leader failover scenarios
- **TestBrokerRecovery**: Recovery from broker failures
- **TestMessagePersistence**: Data durability across restarts
- **TestFollowerReplication**: Leader-follower consistency
### 4. Performance Tests
**File**: `integration/performance_test.go` (to be implemented)
- **TestHighThroughputPublish**: High-volume message publishing
- **TestHighThroughputSubscribe**: High-volume message consumption
- **TestLatencyMeasurement**: End-to-end latency analysis
- **TestResourceUtilization**: CPU, memory, and disk usage
### 5. Agent Tests
**File**: `integration/agent_test.go` (to be implemented)
- **TestAgentPublishSessions**: Session management for publishers
- **TestAgentSubscribeSessions**: Session management for subscribers
- **TestAgentFailover**: Agent reconnection and failover
- **TestAgentConcurrency**: Concurrent session handling
## Configuration
### Environment Variables
The test framework supports configuration via environment variables:
```bash
# Cluster endpoints
SEAWEED_MASTERS="master0:9333,master1:9334,master2:9335"
SEAWEED_BROKERS="broker1:17777,broker2:17778,broker3:17779"
SEAWEED_FILERS="filer1:8888,filer2:8889"
# Test configuration
GO_TEST_TIMEOUT="30m"
TEST_RESULTS_DIR="/test-results"
```
### Docker Compose Override
Create `docker-compose.override.yml` to customize the test environment:
```yaml
version: '3.9'
services:
broker1:
environment:
- CUSTOM_ENV_VAR=value
test-runner:
volumes:
- ./custom-config:/config
```
## Monitoring and Metrics
### Prometheus Metrics
Access Prometheus at: http://localhost:19090
Key metrics to monitor:
- Message throughput: `seaweedmq_messages_published_total`
- Consumer lag: `seaweedmq_consumer_lag_seconds`
- Broker health: `seaweedmq_broker_health`
- Resource usage: `seaweedfs_disk_usage_bytes`
### Grafana Dashboards
Access Grafana at: http://localhost:13000 (admin/admin)
Pre-configured dashboards:
- **SeaweedMQ Overview**: System health and throughput
- **Performance Metrics**: Latency and resource usage
- **Error Analysis**: Error rates and failure patterns
## Development
### Writing New Tests
1. **Create Test File**:
```bash
touch integration/my_new_test.go
```
2. **Use Test Framework**:
```go
func TestMyFeature(t *testing.T) {
suite := NewIntegrationTestSuite(t)
require.NoError(t, suite.Setup())
// Your test logic here
}
```
3. **Run Specific Test**:
```bash
go test -v ./integration/ -run TestMyFeature
```
### Test Framework Components
**IntegrationTestSuite**: Base test framework with cluster management
**MessageCollector**: Utility for collecting and verifying received messages
**TestMessage**: Standard message structure for testing
**Schema Builders**: Helpers for creating test schemas
### Local Development
Run tests against a local SeaweedMQ cluster:
```bash
make test-dev
```
This uses local binaries instead of Docker containers.
## Continuous Integration
### GitHub Actions Example
```yaml
name: Integration Tests
on: [push, pull_request]
jobs:
integration-tests:
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v3
- uses: actions/setup-go@v3
with:
go-version: 1.21
- name: Run Integration Tests
run: |
cd test/mq
make test
```
### Jenkins Pipeline
```groovy
pipeline {
agent any
stages {
stage('Setup') {
steps {
sh 'cd test/mq && make up'
}
}
stage('Test') {
steps {
sh 'cd test/mq && make test'
}
post {
always {
sh 'cd test/mq && make down'
}
}
}
}
}
```
## Troubleshooting
### Common Issues
1. **Port Conflicts**:
```bash
# Check port usage
netstat -tulpn | grep :19333
# Kill conflicting processes
sudo kill -9 $(lsof -t -i:19333)
```
2. **Docker Resource Issues**:
```bash
# Increase Docker memory (8GB+)
# Clean up Docker resources
docker system prune -a
```
3. **Test Timeouts**:
```bash
# Increase timeout
GO_TEST_TIMEOUT=60m make test
```
### Debug Mode
Run tests with verbose logging:
```bash
docker-compose -f docker-compose.test.yml run --rm test-runner \
sh -c "go test -v -race ./test/mq/integration/... -args -test.v"
```
### Container Logs
View real-time logs:
```bash
make logs
# Or specific service
docker-compose -f docker-compose.test.yml logs -f broker1
```
## Performance Benchmarks
### Throughput Benchmarks
```bash
make benchmark
```
Expected performance (on 8-core, 16GB RAM):
- **Publish Throughput**: 50K+ messages/second/broker
- **Subscribe Throughput**: 100K+ messages/second/broker
- **End-to-End Latency**: P95 < 100ms
- **Storage Efficiency**: < 20% overhead
### Load Testing
```bash
make load-test
```
Stress tests with:
- 1M+ messages
- 100+ concurrent producers
- 50+ concurrent consumers
- Multiple topic scenarios
## Contributing
### Test Guidelines
1. **Test Isolation**: Each test should be independent
2. **Resource Cleanup**: Always clean up resources in test teardown
3. **Timeouts**: Set appropriate timeouts for operations
4. **Error Handling**: Test both success and failure scenarios
5. **Documentation**: Document test purpose and expected behavior
### Code Style
- Follow Go testing conventions
- Use testify for assertions
- Include setup/teardown in test functions
- Use descriptive test names
## Future Enhancements
- [ ] Chaos engineering tests (network partitions, node failures)
- [ ] Multi-datacenter deployment testing
- [ ] Schema evolution compatibility tests
- [ ] Security and authentication tests
- [ ] Performance regression detection
- [ ] Automated load pattern generation
## Support
For issues and questions:
- Check existing GitHub issues
- Review SeaweedMQ documentation
- Join SeaweedFS community discussions
---
*This integration test suite ensures SeaweedMQ's reliability, performance, and functionality across all critical use cases and failure scenarios.*
+165
View File
@@ -0,0 +1,165 @@
services:
# Masters
master0:
image: chrislusf/seaweedfs:latest
ports:
- "19333:9333"
volumes:
- /tmp/seaweedfs-test/master0:/data
command: "master -port=9333 -mdir=/data -peers=master0:9333,master1:9334,master2:9335 -ip=master0 -defaultReplication=001"
networks:
- seaweedfs-test
healthcheck:
test: ["CMD", "wget", "-q", "--spider", "http://master0:9333/cluster/status"]
interval: 30s
timeout: 10s
retries: 5
start_period: 30s
master1:
image: chrislusf/seaweedfs:latest
ports:
- "19334:9334"
volumes:
- /tmp/seaweedfs-test/master1:/data
command: "master -port=9334 -mdir=/data -peers=master0:9333,master1:9334,master2:9335 -ip=master1 -defaultReplication=001"
networks:
- seaweedfs-test
healthcheck:
test: ["CMD", "wget", "-q", "--spider", "http://master1:9334/cluster/status"]
interval: 30s
timeout: 10s
retries: 5
start_period: 30s
master2:
image: chrislusf/seaweedfs:latest
ports:
- "19335:9335"
volumes:
- /tmp/seaweedfs-test/master2:/data
command: "master -port=9335 -mdir=/data -peers=master0:9333,master1:9334,master2:9335 -ip=master2 -defaultReplication=001"
networks:
- seaweedfs-test
healthcheck:
test: ["CMD", "wget", "-q", "--spider", "http://master2:9335/cluster/status"]
interval: 30s
timeout: 10s
retries: 5
start_period: 30s
# Volume Servers
volume1:
image: chrislusf/seaweedfs:latest
ports:
- "18080:8080"
volumes:
- /tmp/seaweedfs-test/volume1:/data
command: "volume -port=8080 -mserver=master0:9333,master1:9334,master2:9335 -dir=/data"
depends_on:
- master0
- master1
- master2
networks:
- seaweedfs-test
volume2:
image: chrislusf/seaweedfs:latest
ports:
- "18081:8081"
volumes:
- /tmp/seaweedfs-test/volume2:/data
command: "volume -port=8081 -mserver=master0:9333,master1:9334,master2:9335 -dir=/data"
depends_on:
- master0
- master1
- master2
networks:
- seaweedfs-test
volume3:
image: chrislusf/seaweedfs:latest
ports:
- "18082:8082"
volumes:
- /tmp/seaweedfs-test/volume3:/data
command: "volume -port=8082 -mserver=master0:9333,master1:9334,master2:9335 -dir=/data"
depends_on:
- master0
- master1
- master2
networks:
- seaweedfs-test
# Filers
filer1:
image: chrislusf/seaweedfs:latest
ports:
- "18888:8888"
volumes:
- /tmp/seaweedfs-test/filer1:/data
command: "filer -port=8888 -master=master0:9333,master1:9334,master2:9335"
depends_on:
- volume1
- volume2
- volume3
networks:
- seaweedfs-test
filer2:
image: chrislusf/seaweedfs:latest
ports:
- "18889:8889"
volumes:
- /tmp/seaweedfs-test/filer2:/data
command: "filer -port=8889 -master=master0:9333,master1:9334,master2:9335"
depends_on:
- volume1
- volume2
- volume3
networks:
- seaweedfs-test
# Message Queue Brokers
broker1:
image: chrislusf/seaweedfs:latest
ports:
- "17777:17777"
volumes:
- /tmp/seaweedfs-test/broker1:/data
command: "mq.broker -port=17777 -master=master0:9333,master1:9334,master2:9335"
depends_on:
- filer1
- filer2
networks:
- seaweedfs-test
broker2:
image: chrislusf/seaweedfs:latest
ports:
- "17778:17778"
volumes:
- /tmp/seaweedfs-test/broker2:/data
command: "mq.broker -port=17778 -master=master0:9333,master1:9334,master2:9335"
depends_on:
- filer1
- filer2
networks:
- seaweedfs-test
broker3:
image: chrislusf/seaweedfs:latest
ports:
- "17779:17779"
volumes:
- /tmp/seaweedfs-test/broker3:/data
command: "mq.broker -port=17779 -master=master0:9333,master1:9334,master2:9335"
depends_on:
- filer1
- filer2
networks:
- seaweedfs-test
networks:
seaweedfs-test:
driver: bridge
+260
View File
@@ -0,0 +1,260 @@
services:
# Masters
master0:
image: chrislusf/seaweedfs:latest
ports:
- "19333:19333"
- "9333:9333"
volumes:
- /tmp/seaweedfs-test/master0:/data
command: "master -port=9333 -mdir=/data -peers=master0:9333,master1:9334,master2:9335 -defaultReplication=001"
environment:
WEED_MASTER_VOLUME_GROWTH_COPY_1: 7
WEED_MASTER_VOLUME_GROWTH_COPY_2: 6
WEED_MASTER_VOLUME_GROWTH_COPY_OTHER: 3
networks:
- seaweedfs-test
healthcheck:
test: ["CMD", "wget", "-q", "--spider", "http://master0:9333/cluster/status"]
interval: 10s
timeout: 5s
retries: 3
master1:
image: chrislusf/seaweedfs:latest
ports:
- "19334:9334"
- "9334:9334"
volumes:
- /tmp/seaweedfs-test/master1:/data
command: "master -port=9334 -mdir=/data -peers=master0:9333,master1:9334,master2:9335 -defaultReplication=001"
environment:
WEED_MASTER_VOLUME_GROWTH_COPY_1: 7
WEED_MASTER_VOLUME_GROWTH_COPY_2: 6
WEED_MASTER_VOLUME_GROWTH_COPY_OTHER: 3
networks:
- seaweedfs-test
healthcheck:
test: ["CMD", "wget", "-q", "--spider", "http://master1:9334/cluster/status"]
interval: 10s
timeout: 5s
retries: 3
master2:
image: chrislusf/seaweedfs:latest
ports:
- "19335:9335"
- "9335:9335"
volumes:
- /tmp/seaweedfs-test/master2:/data
command: "master -port=9335 -mdir=/data -peers=master0:9333,master1:9334,master2:9335 -defaultReplication=001"
environment:
WEED_MASTER_VOLUME_GROWTH_COPY_1: 7
WEED_MASTER_VOLUME_GROWTH_COPY_2: 6
WEED_MASTER_VOLUME_GROWTH_COPY_OTHER: 3
networks:
- seaweedfs-test
healthcheck:
test: ["CMD", "wget", "-q", "--spider", "http://master2:9335/cluster/status"]
interval: 10s
timeout: 5s
retries: 3
# Volume Servers
volume1:
image: chrislusf/seaweedfs:latest
ports:
- "18080:8080"
- "8080:8080"
volumes:
- /tmp/seaweedfs-test/volume1:/data
command: "volume -port=8080 -mserver=master0:9333,master1:9334,master2:9335 -dir=/data"
depends_on:
master0:
condition: service_healthy
master1:
condition: service_healthy
master2:
condition: service_healthy
networks:
- seaweedfs-test
environment:
WEED_VOLUME_MAX: 100
volume2:
image: chrislusf/seaweedfs:latest
ports:
- "18081:8081"
- "8081:8081"
volumes:
- /tmp/seaweedfs-test/volume2:/data
command: "volume -port=8081 -mserver=master0:9333,master1:9334,master2:9335 -dir=/data"
depends_on:
master0:
condition: service_healthy
master1:
condition: service_healthy
master2:
condition: service_healthy
networks:
- seaweedfs-test
environment:
WEED_VOLUME_MAX: 100
volume3:
image: chrislusf/seaweedfs:latest
ports:
- "18082:8082"
- "8082:8082"
volumes:
- /tmp/seaweedfs-test/volume3:/data
command: "volume -port=8082 -mserver=master0:9333,master1:9334,master2:9335 -dir=/data"
depends_on:
master0:
condition: service_healthy
master1:
condition: service_healthy
master2:
condition: service_healthy
networks:
- seaweedfs-test
environment:
WEED_VOLUME_MAX: 100
# Filers
filer1:
image: chrislusf/seaweedfs:latest
ports:
- "18888:8888"
- "8888:8888"
volumes:
- /tmp/seaweedfs-test/filer1:/data
command: "filer -port=8888 -master=master0:9333,master1:9334,master2:9335"
depends_on:
- volume1
- volume2
- volume3
networks:
- seaweedfs-test
environment:
WEED_LEVELDB2_DIR: "/data/filerldb2"
filer2:
image: chrislusf/seaweedfs:latest
ports:
- "18889:8889"
- "8889:8889"
volumes:
- /tmp/seaweedfs-test/filer2:/data
command: "filer -port=8889 -master=master0:9333,master1:9334,master2:9335"
depends_on:
- volume1
- volume2
- volume3
networks:
- seaweedfs-test
environment:
WEED_LEVELDB2_DIR: "/data/filerldb2"
# Message Queue Brokers
broker1:
image: chrislusf/seaweedfs:latest
ports:
- "17777:17777"
volumes:
- /tmp/seaweedfs-test/broker1:/data
command: "mq.broker -port=17777 -filer=filer1:8888,filer2:8889"
depends_on:
- filer1
- filer2
networks:
- seaweedfs-test
environment:
WEED_MQ_BROKER_CPU_PERCENT: 80
broker2:
image: chrislusf/seaweedfs:latest
ports:
- "17778:17778"
volumes:
- /tmp/seaweedfs-test/broker2:/data
command: "mq.broker -port=17778 -filer=filer1:8888,filer2:8889"
depends_on:
- filer1
- filer2
networks:
- seaweedfs-test
environment:
WEED_MQ_BROKER_CPU_PERCENT: 80
broker3:
image: chrislusf/seaweedfs:latest
ports:
- "17779:17779"
volumes:
- /tmp/seaweedfs-test/broker3:/data
command: "mq.broker -port=17779 -filer=filer1:8888,filer2:8889"
depends_on:
- filer1
- filer2
networks:
- seaweedfs-test
environment:
WEED_MQ_BROKER_CPU_PERCENT: 80
# Test Runner
test-runner:
image: seaweedfs-test-runner
volumes:
- ../..:/workspace
- /tmp/test-results:/test-results
working_dir: /workspace
networks:
- seaweedfs-test
environment:
- SEAWEED_MASTERS=master0:9333,master1:9334,master2:9335
- SEAWEED_BROKERS=broker1:17777,broker2:17778,broker3:17779
- SEAWEED_FILERS=filer1:8888,filer2:8889
- GO111MODULE=on
- CGO_ENABLED=0
depends_on:
- broker1
- broker2
- broker3
# Monitoring
prometheus:
image: prom/prometheus:latest
ports:
- "19090:9090"
volumes:
- ./prometheus.yml:/etc/prometheus/prometheus.yml
networks:
- seaweedfs-test
depends_on:
- master0
- master1
- master2
- volume1
- volume2
- volume3
- filer1
- filer2
- broker1
- broker2
- broker3
grafana:
image: grafana/grafana:latest
ports:
- "13000:3000"
environment:
- GF_SECURITY_ADMIN_PASSWORD=admin
networks:
- seaweedfs-test
depends_on:
- prometheus
networks:
seaweedfs-test:
driver: bridge
+341
View File
@@ -0,0 +1,341 @@
services:
# Master cluster for coordination and metadata
master0:
image: chrislusf/seaweedfs:local
container_name: test-master0
ports:
- "19333:9333"
- "29333:19333"
command: >
-v=1
master
-volumeSizeLimitMB=100
-resumeState=false
-ip=master0
-port=9333
-peers=master0:9333,master1:9334,master2:9335
-mdir=/tmp/master0
environment:
WEED_MASTER_VOLUME_GROWTH_COPY_1: 1
WEED_MASTER_VOLUME_GROWTH_COPY_2: 2
WEED_MASTER_VOLUME_GROWTH_COPY_OTHER: 1
networks:
- seaweedmq-test
healthcheck:
test: ["CMD", "wget", "-q", "--spider", "http://master0:9333/cluster/status"]
interval: 10s
timeout: 5s
retries: 3
master1:
image: chrislusf/seaweedfs:local
container_name: test-master1
ports:
- "19334:9334"
- "29334:19334"
command: >
-v=1
master
-volumeSizeLimitMB=100
-resumeState=false
-ip=master1
-port=9334
-peers=master0:9333,master1:9334,master2:9335
-mdir=/tmp/master1
environment:
WEED_MASTER_VOLUME_GROWTH_COPY_1: 1
WEED_MASTER_VOLUME_GROWTH_COPY_2: 2
WEED_MASTER_VOLUME_GROWTH_COPY_OTHER: 1
networks:
- seaweedmq-test
depends_on:
- master0
master2:
image: chrislusf/seaweedfs:local
container_name: test-master2
ports:
- "19335:9335"
- "29335:19335"
command: >
-v=1
master
-volumeSizeLimitMB=100
-resumeState=false
-ip=master2
-port=9335
-peers=master0:9333,master1:9334,master2:9335
-mdir=/tmp/master2
environment:
WEED_MASTER_VOLUME_GROWTH_COPY_1: 1
WEED_MASTER_VOLUME_GROWTH_COPY_2: 2
WEED_MASTER_VOLUME_GROWTH_COPY_OTHER: 1
networks:
- seaweedmq-test
depends_on:
- master0
# Volume servers for data storage
volume1:
image: chrislusf/seaweedfs:local
container_name: test-volume1
ports:
- "18080:8080"
- "28080:18080"
volumes:
- volume1-data:/data/volume1
entrypoint: ["/bin/sh", "-c"]
command: >
"mkdir -p /data/volume1 && exec weed -v=1
volume
-dataCenter=dc1
-rack=rack1
-mserver=master0:9333,master1:9334,master2:9335
-port=8080
-ip=volume1
-publicUrl=localhost:18080
-preStopSeconds=1
-dir=/data/volume1"
networks:
- seaweedmq-test
depends_on:
master0:
condition: service_healthy
volume2:
image: chrislusf/seaweedfs:local
container_name: test-volume2
ports:
- "18081:8081"
- "28081:18081"
volumes:
- volume2-data:/data/volume2
entrypoint: ["/bin/sh", "-c"]
command: >
"mkdir -p /data/volume2 && exec weed -v=1
volume
-dataCenter=dc1
-rack=rack2
-mserver=master0:9333,master1:9334,master2:9335
-port=8081
-ip=volume2
-publicUrl=localhost:18081
-preStopSeconds=1
-dir=/data/volume2"
networks:
- seaweedmq-test
depends_on:
master0:
condition: service_healthy
volume3:
image: chrislusf/seaweedfs:local
container_name: test-volume3
ports:
- "18082:8082"
- "28082:18082"
volumes:
- volume3-data:/data/volume3
entrypoint: ["/bin/sh", "-c"]
command: >
"mkdir -p /data/volume3 && exec weed -v=1
volume
-dataCenter=dc2
-rack=rack1
-mserver=master0:9333,master1:9334,master2:9335
-port=8082
-ip=volume3
-publicUrl=localhost:18082
-preStopSeconds=1
-dir=/data/volume3"
networks:
- seaweedmq-test
depends_on:
master0:
condition: service_healthy
# Filer servers for metadata
filer1:
image: chrislusf/seaweedfs:local
container_name: test-filer1
ports:
- "18888:8888"
- "28888:18888"
command: >
-v=1
filer
-defaultReplicaPlacement=100
-master=master0:9333,master1:9334,master2:9335
-port=8888
-ip=filer1
-dataCenter=dc1
networks:
- seaweedmq-test
depends_on:
master0:
condition: service_healthy
healthcheck:
test: ["CMD", "wget", "-q", "--spider", "http://filer1:8888/"]
interval: 10s
timeout: 5s
retries: 3
filer2:
image: chrislusf/seaweedfs:local
container_name: test-filer2
ports:
- "18889:8889"
- "28889:18889"
command: >
-v=1
filer
-defaultReplicaPlacement=100
-master=master0:9333,master1:9334,master2:9335
-port=8889
-ip=filer2
-dataCenter=dc2
networks:
- seaweedmq-test
depends_on:
filer1:
condition: service_healthy
# Message Queue Brokers
broker1:
image: chrislusf/seaweedfs:local
container_name: test-broker1
ports:
- "17777:17777"
command: >
-v=1
mq.broker
-master=master0:9333,master1:9334,master2:9335
-port=17777
-ip=127.0.0.1
-dataCenter=dc1
-rack=rack1
networks:
- seaweedmq-test
depends_on:
filer1:
condition: service_healthy
healthcheck:
test: ["CMD", "nc", "-z", "localhost", "17777"]
interval: 10s
timeout: 5s
retries: 3
broker2:
image: chrislusf/seaweedfs:local
container_name: test-broker2
ports:
- "17778:17778"
command: >
-v=1
mq.broker
-master=master0:9333,master1:9334,master2:9335
-port=17778
-ip=127.0.0.1
-dataCenter=dc1
-rack=rack2
networks:
- seaweedmq-test
depends_on:
broker1:
condition: service_healthy
broker3:
image: chrislusf/seaweedfs:local
container_name: test-broker3
ports:
- "17779:17779"
command: >
-v=1
mq.broker
-master=master0:9333,master1:9334,master2:9335
-port=17779
-ip=127.0.0.1
-dataCenter=dc2
-rack=rack1
networks:
- seaweedmq-test
depends_on:
broker1:
condition: service_healthy
# Test runner container
test-runner:
build:
context: ../../
dockerfile: test/mq/Dockerfile.test
container_name: test-runner
volumes:
- ../../:/app
- /tmp/test-results:/test-results
working_dir: /app
environment:
- SEAWEED_MASTERS=master0:9333,master1:9334,master2:9335
- SEAWEED_BROKERS=broker1:17777,broker2:17778,broker3:17779
- SEAWEED_FILERS=filer1:8888,filer2:8889
- TEST_RESULTS_DIR=/test-results
- GO_TEST_TIMEOUT=30m
networks:
- seaweedmq-test
depends_on:
broker1:
condition: service_healthy
broker2:
condition: service_started
broker3:
condition: service_started
command: >
sh -c "
echo 'Waiting for cluster to be ready...' &&
sleep 30 &&
echo 'Running integration tests...' &&
go test -v -timeout=30m ./test/mq/integration/... -args -test.parallel=4
"
# Monitoring and metrics
prometheus:
image: prom/prometheus:latest
container_name: test-prometheus
ports:
- "19090:9090"
volumes:
- ./prometheus.yml:/etc/prometheus/prometheus.yml
networks:
- seaweedmq-test
command:
- '--config.file=/etc/prometheus/prometheus.yml'
- '--storage.tsdb.path=/prometheus'
- '--web.console.libraries=/etc/prometheus/console_libraries'
- '--web.console.templates=/etc/prometheus/consoles'
- '--web.enable-lifecycle'
grafana:
image: grafana/grafana:latest
container_name: test-grafana
ports:
- "13000:3000"
environment:
- GF_SECURITY_ADMIN_PASSWORD=admin
volumes:
- grafana-storage:/var/lib/grafana
- ./grafana/dashboards:/etc/grafana/provisioning/dashboards
- ./grafana/datasources:/etc/grafana/provisioning/datasources
networks:
- seaweedmq-test
networks:
seaweedmq-test:
driver: bridge
ipam:
config:
- subnet: 172.20.0.0/16
volumes:
grafana-storage:
volume1-data:
volume2-data:
volume3-data:
+339
View File
@@ -0,0 +1,339 @@
package integration
import (
"fmt"
"testing"
"time"
"github.com/seaweedfs/seaweedfs/weed/mq/schema"
"github.com/seaweedfs/seaweedfs/weed/pb/mq_pb"
"github.com/seaweedfs/seaweedfs/weed/pb/schema_pb"
"github.com/stretchr/testify/assert"
"github.com/stretchr/testify/require"
)
func TestBasicPublishSubscribe(t *testing.T) {
suite := NewIntegrationTestSuite(t)
require.NoError(t, suite.Setup())
namespace := "test"
topicName := fmt.Sprintf("basic-pubsub-%d", time.Now().UnixNano()) // Unique topic name per run
testSchema := CreateTestSchema()
messageCount := 10
// Create publisher
pubConfig := &PublisherTestConfig{
Namespace: namespace,
TopicName: topicName,
PartitionCount: 1,
PublisherName: "basic-publisher",
RecordType: testSchema,
}
publisher, err := suite.CreatePublisher(pubConfig)
require.NoError(t, err)
// Create subscriber
subConfig := &SubscriberTestConfig{
Namespace: namespace,
TopicName: topicName,
ConsumerGroup: "test-group",
ConsumerInstanceId: "consumer-1",
MaxPartitionCount: 1,
SlidingWindowSize: 10,
OffsetType: schema_pb.OffsetType_RESET_TO_EARLIEST,
}
subscriber, err := suite.CreateSubscriber(subConfig)
require.NoError(t, err, "Failed to create subscriber")
// Set up message collector
collector := NewMessageCollector(messageCount)
subscriber.SetOnDataMessageFn(func(m *mq_pb.SubscribeMessageResponse_Data) {
t.Logf("[Subscriber] Received message with key: %s, ts: %d", string(m.Data.Key), m.Data.TsNs)
collector.AddMessage(TestMessage{
ID: fmt.Sprintf("msg-%d", len(collector.GetMessages())),
Content: m.Data.Value,
Timestamp: time.Unix(0, m.Data.TsNs),
Key: m.Data.Key,
})
})
// Start subscriber
go func() {
err := subscriber.Subscribe()
if err != nil {
t.Logf("Subscriber error: %v", err)
}
}()
// Wait for subscriber to be ready
t.Logf("[Test] Waiting for subscriber to be ready...")
time.Sleep(2 * time.Second)
// Publish test messages
for i := 0; i < messageCount; i++ {
record := schema.RecordBegin().
SetString("id", fmt.Sprintf("msg-%d", i)).
SetInt64("timestamp", time.Now().UnixNano()).
SetString("content", fmt.Sprintf("Test message %d", i)).
SetInt32("sequence", int32(i)).
RecordEnd()
key := []byte(fmt.Sprintf("key-%d", i))
t.Logf("[Publisher] Publishing message %d with key: %s", i, string(key))
err := publisher.PublishRecord(key, record)
require.NoError(t, err, "Failed to publish message %d", i)
}
t.Logf("[Test] Waiting for messages to be received...")
messages := collector.WaitForMessages(30 * time.Second)
t.Logf("[Test] WaitForMessages returned. Received %d messages.", len(messages))
// Verify all messages were received
assert.Len(t, messages, messageCount, "Expected %d messages, got %d", messageCount, len(messages))
// Verify message content
for i, msg := range messages {
assert.NotEmpty(t, msg.Content, "Message %d should have content", i)
assert.NotEmpty(t, msg.Key, "Message %d should have key", i)
}
t.Logf("[Test] TestBasicPublishSubscribe completed.")
}
func TestMultipleConsumers(t *testing.T) {
suite := NewIntegrationTestSuite(t)
require.NoError(t, suite.Setup())
namespace := "test"
topicName := "multi-consumer"
testSchema := CreateTestSchema()
messageCount := 20
consumerCount := 3
// Create publisher
pubConfig := &PublisherTestConfig{
Namespace: namespace,
TopicName: topicName,
PartitionCount: 3, // Multiple partitions for load distribution
PublisherName: "multi-publisher",
RecordType: testSchema,
}
publisher, err := suite.CreatePublisher(pubConfig)
require.NoError(t, err)
// Create multiple consumers
collectors := make([]*MessageCollector, consumerCount)
for i := 0; i < consumerCount; i++ {
collectors[i] = NewMessageCollector(messageCount / consumerCount) // Expect roughly equal distribution
subConfig := &SubscriberTestConfig{
Namespace: namespace,
TopicName: topicName,
ConsumerGroup: "multi-consumer-group", // Same group for load balancing
ConsumerInstanceId: fmt.Sprintf("consumer-%d", i),
MaxPartitionCount: 1,
SlidingWindowSize: 10,
OffsetType: schema_pb.OffsetType_RESET_TO_EARLIEST,
}
subscriber, err := suite.CreateSubscriber(subConfig)
require.NoError(t, err)
// Set up message collection for this consumer
collectorIndex := i
subscriber.SetOnDataMessageFn(func(m *mq_pb.SubscribeMessageResponse_Data) {
collectors[collectorIndex].AddMessage(TestMessage{
ID: fmt.Sprintf("consumer-%d-msg-%d", collectorIndex, len(collectors[collectorIndex].GetMessages())),
Content: m.Data.Value,
Timestamp: time.Unix(0, m.Data.TsNs),
Key: m.Data.Key,
})
})
// Start subscriber
go func() {
subscriber.Subscribe()
}()
}
// Wait for subscribers to be ready
time.Sleep(3 * time.Second)
// Publish messages with different keys to distribute across partitions
for i := 0; i < messageCount; i++ {
record := schema.RecordBegin().
SetString("id", fmt.Sprintf("multi-msg-%d", i)).
SetInt64("timestamp", time.Now().UnixNano()).
SetString("content", fmt.Sprintf("Multi consumer test message %d", i)).
SetInt32("sequence", int32(i)).
RecordEnd()
key := []byte(fmt.Sprintf("partition-key-%d", i%3)) // Distribute across 3 partitions
err := publisher.PublishRecord(key, record)
require.NoError(t, err)
}
// Wait for all messages to be consumed
time.Sleep(10 * time.Second)
// Verify message distribution
totalReceived := 0
for i, collector := range collectors {
messages := collector.GetMessages()
t.Logf("Consumer %d received %d messages", i, len(messages))
totalReceived += len(messages)
}
// All messages should be consumed across all consumers
assert.Equal(t, messageCount, totalReceived, "Total messages received should equal messages sent")
}
func TestMessageOrdering(t *testing.T) {
suite := NewIntegrationTestSuite(t)
require.NoError(t, suite.Setup())
namespace := "test"
topicName := "ordering-test"
testSchema := CreateTestSchema()
messageCount := 15
// Create publisher
pubConfig := &PublisherTestConfig{
Namespace: namespace,
TopicName: topicName,
PartitionCount: 1, // Single partition to guarantee ordering
PublisherName: "ordering-publisher",
RecordType: testSchema,
}
publisher, err := suite.CreatePublisher(pubConfig)
require.NoError(t, err)
// Create subscriber
subConfig := &SubscriberTestConfig{
Namespace: namespace,
TopicName: topicName,
ConsumerGroup: "ordering-group",
ConsumerInstanceId: "ordering-consumer",
MaxPartitionCount: 1,
SlidingWindowSize: 5,
OffsetType: schema_pb.OffsetType_RESET_TO_EARLIEST,
}
subscriber, err := suite.CreateSubscriber(subConfig)
require.NoError(t, err)
// Set up message collector
collector := NewMessageCollector(messageCount)
subscriber.SetOnDataMessageFn(func(m *mq_pb.SubscribeMessageResponse_Data) {
collector.AddMessage(TestMessage{
ID: fmt.Sprintf("ordered-msg"),
Content: m.Data.Value,
Timestamp: time.Unix(0, m.Data.TsNs),
Key: m.Data.Key,
})
})
// Start subscriber
go func() {
subscriber.Subscribe()
}()
// Wait for consumer to be ready
time.Sleep(2 * time.Second)
// Publish messages with same key to ensure they go to same partition
publishTimes := make([]time.Time, messageCount)
for i := 0; i < messageCount; i++ {
publishTimes[i] = time.Now()
record := schema.RecordBegin().
SetString("id", fmt.Sprintf("ordered-%d", i)).
SetInt64("timestamp", publishTimes[i].UnixNano()).
SetString("content", fmt.Sprintf("Ordered message %d", i)).
SetInt32("sequence", int32(i)).
RecordEnd()
key := []byte("same-partition-key") // Same key ensures same partition
err := publisher.PublishRecord(key, record)
require.NoError(t, err)
// Small delay to ensure different timestamps
time.Sleep(10 * time.Millisecond)
}
// Wait for all messages
messages := collector.WaitForMessages(30 * time.Second)
require.Len(t, messages, messageCount)
// Verify ordering within the partition
suite.AssertMessageOrdering(t, messages)
}
func TestSchemaValidation(t *testing.T) {
suite := NewIntegrationTestSuite(t)
require.NoError(t, suite.Setup())
namespace := "test"
topicName := "schema-validation"
// Test with simple schema
simpleSchema := CreateTestSchema()
pubConfig := &PublisherTestConfig{
Namespace: namespace,
TopicName: topicName,
PartitionCount: 1,
PublisherName: "schema-publisher",
RecordType: simpleSchema,
}
publisher, err := suite.CreatePublisher(pubConfig)
require.NoError(t, err)
// Test valid record
validRecord := schema.RecordBegin().
SetString("id", "valid-msg").
SetInt64("timestamp", time.Now().UnixNano()).
SetString("content", "Valid message").
SetInt32("sequence", 1).
RecordEnd()
err = publisher.PublishRecord([]byte("test-key"), validRecord)
assert.NoError(t, err, "Valid record should be published successfully")
// Test with complex nested schema
complexSchema := CreateComplexTestSchema()
complexPubConfig := &PublisherTestConfig{
Namespace: namespace,
TopicName: topicName + "-complex",
PartitionCount: 1,
PublisherName: "complex-publisher",
RecordType: complexSchema,
}
complexPublisher, err := suite.CreatePublisher(complexPubConfig)
require.NoError(t, err)
// Test complex nested record
complexRecord := schema.RecordBegin().
SetString("user_id", "user123").
SetString("name", "John Doe").
SetInt32("age", 30).
SetStringList("emails", "john@example.com", "john.doe@company.com").
SetRecord("address",
schema.RecordBegin().
SetString("street", "123 Main St").
SetString("city", "New York").
SetString("zipcode", "10001").
RecordEnd()).
SetInt64("created_at", time.Now().UnixNano()).
RecordEnd()
err = complexPublisher.PublishRecord([]byte("complex-key"), complexRecord)
assert.NoError(t, err, "Complex nested record should be published successfully")
}
+379
View File
@@ -0,0 +1,379 @@
package integration
import (
"context"
"fmt"
"os"
"strings"
"sync"
"testing"
"time"
"github.com/seaweedfs/seaweedfs/weed/mq/agent"
"github.com/seaweedfs/seaweedfs/weed/mq/client/pub_client"
"github.com/seaweedfs/seaweedfs/weed/mq/client/sub_client"
"github.com/seaweedfs/seaweedfs/weed/mq/schema"
"github.com/seaweedfs/seaweedfs/weed/mq/topic"
"github.com/seaweedfs/seaweedfs/weed/pb"
"github.com/seaweedfs/seaweedfs/weed/pb/schema_pb"
"github.com/stretchr/testify/require"
"google.golang.org/grpc"
"google.golang.org/grpc/credentials/insecure"
)
// TestEnvironment holds the configuration for the test environment
type TestEnvironment struct {
Masters []string
Brokers []string
Filers []string
TestTimeout time.Duration
CleanupFuncs []func()
mutex sync.Mutex
}
// IntegrationTestSuite provides the base test framework
type IntegrationTestSuite struct {
env *TestEnvironment
agents map[string]*agent.MessageQueueAgent
publishers map[string]*pub_client.TopicPublisher
subscribers map[string]*sub_client.TopicSubscriber
subCancels map[string]context.CancelFunc
cleanupOnce sync.Once
t *testing.T
}
// NewIntegrationTestSuite creates a new test suite instance
func NewIntegrationTestSuite(t *testing.T) *IntegrationTestSuite {
env := &TestEnvironment{
Masters: getEnvList("SEAWEED_MASTERS", []string{"localhost:19333"}),
Brokers: getEnvList("SEAWEED_BROKERS", []string{"localhost:17777"}),
Filers: getEnvList("SEAWEED_FILERS", []string{"localhost:18888"}),
TestTimeout: getEnvDuration("GO_TEST_TIMEOUT", 30*time.Minute),
}
return &IntegrationTestSuite{
env: env,
agents: make(map[string]*agent.MessageQueueAgent),
publishers: make(map[string]*pub_client.TopicPublisher),
subscribers: make(map[string]*sub_client.TopicSubscriber),
subCancels: make(map[string]context.CancelFunc),
t: t,
}
}
// Setup initializes the test environment
func (its *IntegrationTestSuite) Setup() error {
// Wait for cluster to be ready
if err := its.waitForClusterReady(); err != nil {
return fmt.Errorf("cluster not ready: %v", err)
}
// Register cleanup
its.t.Cleanup(its.Cleanup)
return nil
}
// Cleanup performs cleanup operations
func (its *IntegrationTestSuite) Cleanup() {
its.cleanupOnce.Do(func() {
// Close all subscribers first (they use context cancellation)
for name := range its.subscribers {
if cancel, ok := its.subCancels[name]; ok && cancel != nil {
cancel()
its.t.Logf("Cancelled subscriber context: %s", name)
}
its.t.Logf("Cleaned up subscriber: %s", name)
}
// Wait a moment for gRPC connections to close gracefully
time.Sleep(1 * time.Second)
// Close all publishers
for name, publisher := range its.publishers {
if publisher != nil {
// Add timeout to prevent deadlock during shutdown
done := make(chan bool, 1)
go func(p *pub_client.TopicPublisher, n string) {
p.Shutdown()
done <- true
}(publisher, name)
select {
case <-done:
its.t.Logf("Cleaned up publisher: %s", name)
case <-time.After(5 * time.Second):
its.t.Logf("Publisher shutdown timed out: %s", name)
}
}
}
// Execute additional cleanup functions
its.env.mutex.Lock()
for _, cleanup := range its.env.CleanupFuncs {
cleanup()
}
its.env.mutex.Unlock()
})
}
// CreatePublisher creates a new topic publisher
func (its *IntegrationTestSuite) CreatePublisher(config *PublisherTestConfig) (*pub_client.TopicPublisher, error) {
publisherConfig := &pub_client.PublisherConfiguration{
Topic: topic.NewTopic(config.Namespace, config.TopicName),
PartitionCount: config.PartitionCount,
Brokers: its.env.Brokers,
PublisherName: config.PublisherName,
RecordType: config.RecordType,
}
publisher, err := pub_client.NewTopicPublisher(publisherConfig)
if err != nil {
return nil, fmt.Errorf("failed to create publisher: %v", err)
}
its.publishers[config.PublisherName] = publisher
return publisher, nil
}
// CreateSubscriber creates a new topic subscriber
func (its *IntegrationTestSuite) CreateSubscriber(config *SubscriberTestConfig) (*sub_client.TopicSubscriber, error) {
subscriberConfig := &sub_client.SubscriberConfiguration{
ConsumerGroup: config.ConsumerGroup,
ConsumerGroupInstanceId: config.ConsumerInstanceId,
GrpcDialOption: grpc.WithTransportCredentials(insecure.NewCredentials()),
MaxPartitionCount: config.MaxPartitionCount,
SlidingWindowSize: config.SlidingWindowSize,
}
contentConfig := &sub_client.ContentConfiguration{
Topic: topic.NewTopic(config.Namespace, config.TopicName),
Filter: config.Filter,
PartitionOffsets: config.PartitionOffsets,
OffsetType: config.OffsetType,
OffsetTsNs: config.OffsetTsNs,
}
offsetChan := make(chan sub_client.KeyedOffset, 1024)
ctx, cancel := context.WithCancel(context.Background())
subscriber := sub_client.NewTopicSubscriber(
ctx,
its.env.Brokers,
subscriberConfig,
contentConfig,
offsetChan,
)
its.subscribers[config.ConsumerInstanceId] = subscriber
its.subCancels[config.ConsumerInstanceId] = cancel
return subscriber, nil
}
// CreateAgent creates a new message queue agent
func (its *IntegrationTestSuite) CreateAgent(name string) (*agent.MessageQueueAgent, error) {
var brokerAddresses []pb.ServerAddress
for _, broker := range its.env.Brokers {
brokerAddresses = append(brokerAddresses, pb.ServerAddress(broker))
}
agentOptions := &agent.MessageQueueAgentOptions{
SeedBrokers: brokerAddresses,
}
mqAgent := agent.NewMessageQueueAgent(
agentOptions,
grpc.WithTransportCredentials(insecure.NewCredentials()),
)
its.agents[name] = mqAgent
return mqAgent, nil
}
// PublisherTestConfig holds configuration for creating test publishers
type PublisherTestConfig struct {
Namespace string
TopicName string
PartitionCount int32
PublisherName string
RecordType *schema_pb.RecordType
}
// SubscriberTestConfig holds configuration for creating test subscribers
type SubscriberTestConfig struct {
Namespace string
TopicName string
ConsumerGroup string
ConsumerInstanceId string
MaxPartitionCount int32
SlidingWindowSize int32
Filter string
PartitionOffsets []*schema_pb.PartitionOffset
OffsetType schema_pb.OffsetType
OffsetTsNs int64
}
// TestMessage represents a test message with metadata
type TestMessage struct {
ID string
Content []byte
Timestamp time.Time
Key []byte
}
// MessageCollector collects received messages for verification
type MessageCollector struct {
messages []TestMessage
mutex sync.RWMutex
waitCh chan struct{}
expected int
closed bool // protect against closing waitCh multiple times
}
// NewMessageCollector creates a new message collector
func NewMessageCollector(expectedCount int) *MessageCollector {
return &MessageCollector{
messages: make([]TestMessage, 0),
waitCh: make(chan struct{}),
expected: expectedCount,
}
}
// AddMessage adds a received message to the collector
func (mc *MessageCollector) AddMessage(msg TestMessage) {
mc.mutex.Lock()
defer mc.mutex.Unlock()
mc.messages = append(mc.messages, msg)
if len(mc.messages) >= mc.expected && !mc.closed {
close(mc.waitCh)
mc.closed = true
}
}
// WaitForMessages waits for the expected number of messages or timeout
func (mc *MessageCollector) WaitForMessages(timeout time.Duration) []TestMessage {
select {
case <-mc.waitCh:
case <-time.After(timeout):
}
mc.mutex.RLock()
defer mc.mutex.RUnlock()
result := make([]TestMessage, len(mc.messages))
copy(result, mc.messages)
return result
}
// GetMessages returns all collected messages
func (mc *MessageCollector) GetMessages() []TestMessage {
mc.mutex.RLock()
defer mc.mutex.RUnlock()
result := make([]TestMessage, len(mc.messages))
copy(result, mc.messages)
return result
}
// CreateTestSchema creates a simple test schema
func CreateTestSchema() *schema_pb.RecordType {
return schema.RecordTypeBegin().
WithField("id", schema.TypeString).
WithField("timestamp", schema.TypeInt64).
WithField("content", schema.TypeString).
WithField("sequence", schema.TypeInt32).
RecordTypeEnd()
}
// CreateComplexTestSchema creates a complex test schema with nested structures
func CreateComplexTestSchema() *schema_pb.RecordType {
addressType := schema.RecordTypeBegin().
WithField("street", schema.TypeString).
WithField("city", schema.TypeString).
WithField("zipcode", schema.TypeString).
RecordTypeEnd()
return schema.RecordTypeBegin().
WithField("user_id", schema.TypeString).
WithField("name", schema.TypeString).
WithField("age", schema.TypeInt32).
WithField("emails", schema.ListOf(schema.TypeString)).
WithRecordField("address", addressType).
WithField("created_at", schema.TypeInt64).
RecordTypeEnd()
}
// Helper functions
func getEnvList(key string, defaultValue []string) []string {
value := os.Getenv(key)
if value == "" {
return defaultValue
}
return strings.Split(value, ",")
}
func getEnvDuration(key string, defaultValue time.Duration) time.Duration {
value := os.Getenv(key)
if value == "" {
return defaultValue
}
duration, err := time.ParseDuration(value)
if err != nil {
return defaultValue
}
return duration
}
func (its *IntegrationTestSuite) waitForClusterReady() error {
maxRetries := 30
retryInterval := 2 * time.Second
for i := 0; i < maxRetries; i++ {
if its.isClusterReady() {
return nil
}
its.t.Logf("Waiting for cluster to be ready... attempt %d/%d", i+1, maxRetries)
time.Sleep(retryInterval)
}
return fmt.Errorf("cluster not ready after %d attempts", maxRetries)
}
func (its *IntegrationTestSuite) isClusterReady() bool {
// Check if at least one broker is accessible
for _, broker := range its.env.Brokers {
if its.isBrokerReady(broker) {
return true
}
}
return false
}
func (its *IntegrationTestSuite) isBrokerReady(broker string) bool {
// Simple connection test
conn, err := grpc.NewClient(broker, grpc.WithTransportCredentials(insecure.NewCredentials()))
if err != nil {
return false
}
defer conn.Close()
// TODO: Add actual health check call here
return true
}
// AssertMessagesReceived verifies that expected messages were received
func (its *IntegrationTestSuite) AssertMessagesReceived(t *testing.T, collector *MessageCollector, expectedCount int, timeout time.Duration) {
messages := collector.WaitForMessages(timeout)
require.Len(t, messages, expectedCount, "Expected %d messages, got %d", expectedCount, len(messages))
}
// AssertMessageOrdering verifies that messages are received in the expected order
func (its *IntegrationTestSuite) AssertMessageOrdering(t *testing.T, messages []TestMessage) {
for i := 1; i < len(messages); i++ {
require.True(t, messages[i].Timestamp.After(messages[i-1].Timestamp) || messages[i].Timestamp.Equal(messages[i-1].Timestamp),
"Messages not in chronological order: message %d timestamp %v should be >= message %d timestamp %v",
i, messages[i].Timestamp, i-1, messages[i-1].Timestamp)
}
}
+666
View File
@@ -0,0 +1,666 @@
package integration
import (
"fmt"
"sync"
"sync/atomic"
"testing"
"time"
"github.com/seaweedfs/seaweedfs/weed/pb/mq_pb"
"github.com/seaweedfs/seaweedfs/weed/pb/schema_pb"
"github.com/stretchr/testify/assert"
"github.com/stretchr/testify/require"
"google.golang.org/grpc/codes"
"google.golang.org/grpc/status"
)
// PerformanceMetrics holds test metrics
type PerformanceMetrics struct {
MessagesPublished int64
MessagesConsumed int64
PublishLatencies []time.Duration
ConsumeLatencies []time.Duration
StartTime time.Time
EndTime time.Time
ErrorCount int64
mu sync.RWMutex
}
func (m *PerformanceMetrics) AddPublishLatency(d time.Duration) {
m.mu.Lock()
defer m.mu.Unlock()
m.PublishLatencies = append(m.PublishLatencies, d)
}
func (m *PerformanceMetrics) AddConsumeLatency(d time.Duration) {
m.mu.Lock()
defer m.mu.Unlock()
m.ConsumeLatencies = append(m.ConsumeLatencies, d)
}
func (m *PerformanceMetrics) GetThroughput() float64 {
duration := m.EndTime.Sub(m.StartTime).Seconds()
if duration == 0 {
return 0
}
return float64(atomic.LoadInt64(&m.MessagesPublished)) / duration
}
func (m *PerformanceMetrics) GetP95Latency(latencies []time.Duration) time.Duration {
if len(latencies) == 0 {
return 0
}
// Simple P95 calculation - in production use proper percentile library
index := int(float64(len(latencies)) * 0.95)
if index >= len(latencies) {
index = len(latencies) - 1
}
// Sort latencies (simplified)
for i := 0; i < len(latencies)-1; i++ {
for j := 0; j < len(latencies)-i-1; j++ {
if latencies[j] > latencies[j+1] {
latencies[j], latencies[j+1] = latencies[j+1], latencies[j]
}
}
}
return latencies[index]
}
// Enhanced performance metrics with connection error tracking
type EnhancedPerformanceMetrics struct {
MessagesPublished int64
MessagesConsumed int64
PublishLatencies []time.Duration
ConsumeLatencies []time.Duration
StartTime time.Time
EndTime time.Time
ErrorCount int64
ConnectionErrors int64
ApplicationErrors int64
RetryAttempts int64
mu sync.RWMutex
}
func (m *EnhancedPerformanceMetrics) AddPublishLatency(d time.Duration) {
m.mu.Lock()
defer m.mu.Unlock()
m.PublishLatencies = append(m.PublishLatencies, d)
}
func (m *EnhancedPerformanceMetrics) AddConsumeLatency(d time.Duration) {
m.mu.Lock()
defer m.mu.Unlock()
m.ConsumeLatencies = append(m.ConsumeLatencies, d)
}
func (m *EnhancedPerformanceMetrics) GetThroughput() float64 {
duration := m.EndTime.Sub(m.StartTime).Seconds()
if duration == 0 {
return 0
}
return float64(atomic.LoadInt64(&m.MessagesPublished)) / duration
}
func (m *EnhancedPerformanceMetrics) GetP95Latency(latencies []time.Duration) time.Duration {
if len(latencies) == 0 {
return 0
}
// Simple P95 calculation
index := int(float64(len(latencies)) * 0.95)
if index >= len(latencies) {
index = len(latencies) - 1
}
// Sort latencies (simplified bubble sort for small datasets)
sorted := make([]time.Duration, len(latencies))
copy(sorted, latencies)
for i := 0; i < len(sorted)-1; i++ {
for j := 0; j < len(sorted)-i-1; j++ {
if sorted[j] > sorted[j+1] {
sorted[j], sorted[j+1] = sorted[j+1], sorted[j]
}
}
}
return sorted[index]
}
// isConnectionError determines if an error is a connection-level error
func isConnectionError(err error) bool {
if err == nil {
return false
}
errStr := err.Error()
// Check for gRPC status codes
if st, ok := status.FromError(err); ok {
switch st.Code() {
case codes.Unavailable, codes.DeadlineExceeded, codes.Canceled, codes.Unknown:
return true
}
}
// Check for common connection error strings
connectionErrorPatterns := []string{
"EOF",
"error reading server preface",
"connection refused",
"connection reset",
"broken pipe",
"network is unreachable",
"no route to host",
"transport is closing",
"connection error",
"dial tcp",
"context deadline exceeded",
}
for _, pattern := range connectionErrorPatterns {
if containsString(errStr, pattern) {
return true
}
}
return false
}
func containsString(s, substr string) bool {
for i := 0; i <= len(s)-len(substr); i++ {
if s[i:i+len(substr)] == substr {
return true
}
}
return false
}
func TestPerformanceThroughput(t *testing.T) {
suite := NewIntegrationTestSuite(t)
require.NoError(t, suite.Setup())
topicName := "performance-throughput-test"
namespace := "perf-test"
metrics := &PerformanceMetrics{
StartTime: time.Now(),
}
// Test parameters
numMessages := 50000
numPublishers := 5
messageSize := 1024 // 1KB messages
// Create message payload
payload := make([]byte, messageSize)
for i := range payload {
payload[i] = byte(i % 256)
}
t.Logf("Starting throughput test: %d messages, %d publishers, %d bytes per message",
numMessages, numPublishers, messageSize)
// Start publishers
var publishWg sync.WaitGroup
messagesPerPublisher := numMessages / numPublishers
for i := 0; i < numPublishers; i++ {
publishWg.Add(1)
go func(publisherID int) {
defer publishWg.Done()
// Create publisher for this goroutine
pubConfig := &PublisherTestConfig{
Namespace: namespace,
TopicName: topicName,
PartitionCount: 4, // Multiple partitions for better throughput
PublisherName: fmt.Sprintf("perf-publisher-%d", publisherID),
RecordType: nil, // Use raw publish
}
publisher, err := suite.CreatePublisher(pubConfig)
if err != nil {
atomic.AddInt64(&metrics.ErrorCount, 1)
t.Errorf("Failed to create publisher %d: %v", publisherID, err)
return
}
for j := 0; j < messagesPerPublisher; j++ {
messageKey := fmt.Sprintf("publisher-%d-msg-%d", publisherID, j)
start := time.Now()
err := publisher.Publish([]byte(messageKey), payload)
latency := time.Since(start)
if err != nil {
atomic.AddInt64(&metrics.ErrorCount, 1)
continue
}
atomic.AddInt64(&metrics.MessagesPublished, 1)
metrics.AddPublishLatency(latency)
// Small delay to prevent overwhelming the system
if j%1000 == 0 {
time.Sleep(1 * time.Millisecond)
}
}
}(i)
}
// Wait for publishing to complete
publishWg.Wait()
metrics.EndTime = time.Now()
// Verify results
publishedCount := atomic.LoadInt64(&metrics.MessagesPublished)
errorCount := atomic.LoadInt64(&metrics.ErrorCount)
throughput := metrics.GetThroughput()
t.Logf("Performance Results:")
t.Logf(" Messages Published: %d", publishedCount)
t.Logf(" Errors: %d", errorCount)
t.Logf(" Throughput: %.2f messages/second", throughput)
t.Logf(" Duration: %v", metrics.EndTime.Sub(metrics.StartTime))
if len(metrics.PublishLatencies) > 0 {
p95Latency := metrics.GetP95Latency(metrics.PublishLatencies)
t.Logf(" P95 Publish Latency: %v", p95Latency)
// Performance assertions
assert.Less(t, p95Latency, 100*time.Millisecond, "P95 publish latency should be under 100ms")
}
// Throughput requirements
expectedMinThroughput := 5000.0 // 5K messages/sec minimum (relaxed from 10K for initial testing)
assert.Greater(t, throughput, expectedMinThroughput,
"Throughput should exceed %.0f messages/second", expectedMinThroughput)
// Error rate should be low
errorRate := float64(errorCount) / float64(publishedCount+errorCount)
assert.Less(t, errorRate, 0.05, "Error rate should be less than 5%")
}
func TestPerformanceLatency(t *testing.T) {
suite := NewIntegrationTestSuite(t)
require.NoError(t, suite.Setup())
topicName := "performance-latency-test"
namespace := "perf-test"
// Create publisher
pubConfig := &PublisherTestConfig{
Namespace: namespace,
TopicName: topicName,
PartitionCount: 1, // Single partition for latency testing
PublisherName: "latency-publisher",
RecordType: nil,
}
publisher, err := suite.CreatePublisher(pubConfig)
require.NoError(t, err)
// Start consumer first
subConfig := &SubscriberTestConfig{
Namespace: namespace,
TopicName: topicName,
ConsumerGroup: "latency-test-group",
ConsumerInstanceId: "latency-consumer-1",
MaxPartitionCount: 1,
SlidingWindowSize: 10,
OffsetType: schema_pb.OffsetType_RESET_TO_EARLIEST,
}
subscriber, err := suite.CreateSubscriber(subConfig)
require.NoError(t, err)
numMessages := 5000
collector := NewMessageCollector(numMessages)
// Set up message handler
subscriber.SetOnDataMessageFn(func(m *mq_pb.SubscribeMessageResponse_Data) {
collector.AddMessage(TestMessage{
ID: string(m.Data.Key),
Content: m.Data.Value,
Timestamp: time.Unix(0, m.Data.TsNs),
Key: m.Data.Key,
})
})
// Start subscriber
go func() {
err := subscriber.Subscribe()
if err != nil {
t.Logf("Subscriber error: %v", err)
}
}()
// Wait for consumer to be ready
time.Sleep(2 * time.Second)
metrics := &PerformanceMetrics{
StartTime: time.Now(),
}
t.Logf("Starting latency test with %d messages", numMessages)
// Publish messages with controlled timing
for i := 0; i < numMessages; i++ {
messageKey := fmt.Sprintf("latency-msg-%d", i)
payload := fmt.Sprintf("test-payload-data-%d", i)
start := time.Now()
err := publisher.Publish([]byte(messageKey), []byte(payload))
publishLatency := time.Since(start)
require.NoError(t, err)
metrics.AddPublishLatency(publishLatency)
// Controlled rate for latency measurement
if i%100 == 0 {
time.Sleep(10 * time.Millisecond)
}
}
metrics.EndTime = time.Now()
// Wait for messages to be consumed
messages := collector.WaitForMessages(30 * time.Second)
// Analyze latency results
t.Logf("Latency Test Results:")
t.Logf(" Messages Published: %d", numMessages)
t.Logf(" Messages Consumed: %d", len(messages))
if len(metrics.PublishLatencies) > 0 {
p95PublishLatency := metrics.GetP95Latency(metrics.PublishLatencies)
t.Logf(" P95 Publish Latency: %v", p95PublishLatency)
// Latency assertions
assert.Less(t, p95PublishLatency, 50*time.Millisecond,
"P95 publish latency should be under 50ms")
}
// Verify message delivery
deliveryRate := float64(len(messages)) / float64(numMessages)
assert.Greater(t, deliveryRate, 0.80, "Should deliver at least 80% of messages")
}
func TestPerformanceConcurrentConsumers(t *testing.T) {
suite := NewIntegrationTestSuite(t)
require.NoError(t, suite.Setup())
topicName := "performance-concurrent-test"
namespace := "perf-test"
numPartitions := int32(4)
// Create publisher first
pubConfig := &PublisherTestConfig{
Namespace: namespace,
TopicName: topicName,
PartitionCount: numPartitions,
PublisherName: "concurrent-publisher",
RecordType: nil,
}
publisher, err := suite.CreatePublisher(pubConfig)
require.NoError(t, err)
// Test parameters
numConsumers := 8
numMessages := 20000
consumerGroup := "concurrent-perf-group"
t.Logf("Starting concurrent consumer test: %d consumers, %d messages, %d partitions",
numConsumers, numMessages, numPartitions)
// Start multiple consumers
var collectors []*MessageCollector
for i := 0; i < numConsumers; i++ {
collector := NewMessageCollector(numMessages / numConsumers) // Expected per consumer
collectors = append(collectors, collector)
subConfig := &SubscriberTestConfig{
Namespace: namespace,
TopicName: topicName,
ConsumerGroup: consumerGroup,
ConsumerInstanceId: fmt.Sprintf("consumer-%d", i),
MaxPartitionCount: numPartitions,
SlidingWindowSize: 10,
OffsetType: schema_pb.OffsetType_RESET_TO_EARLIEST,
}
subscriber, err := suite.CreateSubscriber(subConfig)
require.NoError(t, err)
// Set up message handler for this consumer
func(consumerID int, c *MessageCollector) {
subscriber.SetOnDataMessageFn(func(m *mq_pb.SubscribeMessageResponse_Data) {
c.AddMessage(TestMessage{
ID: fmt.Sprintf("%d-%s", consumerID, string(m.Data.Key)),
Content: m.Data.Value,
Timestamp: time.Unix(0, m.Data.TsNs),
Key: m.Data.Key,
})
})
}(i, collector)
// Start subscriber
go func(consumerID int, s *SubscriberTestConfig) {
sub, _ := suite.CreateSubscriber(s)
err := sub.Subscribe()
if err != nil {
t.Logf("Consumer %d error: %v", consumerID, err)
}
}(i, subConfig)
}
// Wait for consumers to initialize
time.Sleep(3 * time.Second)
// Start publishing
startTime := time.Now()
var publishWg sync.WaitGroup
numPublishers := 3
messagesPerPublisher := numMessages / numPublishers
for i := 0; i < numPublishers; i++ {
publishWg.Add(1)
go func(publisherID int) {
defer publishWg.Done()
for j := 0; j < messagesPerPublisher; j++ {
messageKey := fmt.Sprintf("concurrent-msg-%d-%d", publisherID, j)
payload := fmt.Sprintf("publisher-%d-message-%d-data", publisherID, j)
err := publisher.Publish([]byte(messageKey), []byte(payload))
if err != nil {
t.Logf("Publish error: %v", err)
}
// Rate limiting
if j%1000 == 0 {
time.Sleep(2 * time.Millisecond)
}
}
}(i)
}
publishWg.Wait()
publishDuration := time.Since(startTime)
// Allow time for message consumption
time.Sleep(10 * time.Second)
// Analyze results
totalConsumed := int64(0)
for i, collector := range collectors {
messages := collector.GetMessages()
consumed := int64(len(messages))
totalConsumed += consumed
t.Logf("Consumer %d consumed %d messages", i, consumed)
}
publishThroughput := float64(numMessages) / publishDuration.Seconds()
consumeThroughput := float64(totalConsumed) / publishDuration.Seconds()
t.Logf("Concurrent Consumer Test Results:")
t.Logf(" Total Published: %d", numMessages)
t.Logf(" Total Consumed: %d", totalConsumed)
t.Logf(" Publish Throughput: %.2f msg/sec", publishThroughput)
t.Logf(" Consume Throughput: %.2f msg/sec", consumeThroughput)
t.Logf(" Test Duration: %v", publishDuration)
// Performance assertions (relaxed for initial testing)
deliveryRate := float64(totalConsumed) / float64(numMessages)
assert.Greater(t, deliveryRate, 0.70, "Should consume at least 70% of messages")
expectedMinThroughput := 2000.0 // 2K messages/sec minimum for concurrent consumption
assert.Greater(t, consumeThroughput, expectedMinThroughput,
"Consume throughput should exceed %.0f messages/second", expectedMinThroughput)
}
func TestPerformanceWithErrorHandling(t *testing.T) {
suite := NewIntegrationTestSuite(t)
require.NoError(t, suite.Setup())
topicName := "performance-error-handling-test"
namespace := "perf-test"
metrics := &EnhancedPerformanceMetrics{
StartTime: time.Now(),
}
// Test parameters
numMessages := 50000
numPublishers := 5
messageSize := 1024 // 1KB messages
// Create message payload
payload := make([]byte, messageSize)
for i := range payload {
payload[i] = byte(i % 256)
}
t.Logf("Starting performance test with enhanced error handling: %d messages, %d publishers, %d bytes per message",
numMessages, numPublishers, messageSize)
// Start publishers
var publishWg sync.WaitGroup
messagesPerPublisher := numMessages / numPublishers
for i := 0; i < numPublishers; i++ {
publishWg.Add(1)
go func(publisherID int) {
defer publishWg.Done()
// Create publisher for this goroutine
pubConfig := &PublisherTestConfig{
Namespace: namespace,
TopicName: topicName,
PartitionCount: 4, // Multiple partitions for better throughput
PublisherName: fmt.Sprintf("error-aware-publisher-%d", publisherID),
RecordType: nil, // Use raw publish
}
publisher, err := suite.CreatePublisher(pubConfig)
if err != nil {
atomic.AddInt64(&metrics.ErrorCount, 1)
t.Errorf("Failed to create publisher %d: %v", publisherID, err)
return
}
for j := 0; j < messagesPerPublisher; j++ {
messageKey := fmt.Sprintf("publisher-%d-msg-%d", publisherID, j)
start := time.Now()
err := publisher.Publish([]byte(messageKey), payload)
latency := time.Since(start)
if err != nil {
atomic.AddInt64(&metrics.ErrorCount, 1)
// Classify the error type
if isConnectionError(err) {
atomic.AddInt64(&metrics.ConnectionErrors, 1)
t.Logf("Connection error (publisher %d, msg %d): %v", publisherID, j, err)
} else {
atomic.AddInt64(&metrics.ApplicationErrors, 1)
t.Logf("Application error (publisher %d, msg %d): %v", publisherID, j, err)
}
continue
}
atomic.AddInt64(&metrics.MessagesPublished, 1)
metrics.AddPublishLatency(latency)
// Small delay to prevent overwhelming the system
if j%1000 == 0 {
time.Sleep(1 * time.Millisecond)
}
}
}(i)
}
// Wait for publishing to complete
publishWg.Wait()
metrics.EndTime = time.Now()
// Analyze results with enhanced error reporting
publishedCount := atomic.LoadInt64(&metrics.MessagesPublished)
totalErrors := atomic.LoadInt64(&metrics.ErrorCount)
connectionErrors := atomic.LoadInt64(&metrics.ConnectionErrors)
applicationErrors := atomic.LoadInt64(&metrics.ApplicationErrors)
throughput := metrics.GetThroughput()
t.Logf("Enhanced Performance Results:")
t.Logf(" Messages Successfully Published: %d", publishedCount)
t.Logf(" Total Errors: %d", totalErrors)
t.Logf(" Connection-Level Errors: %d", connectionErrors)
t.Logf(" Application-Level Errors: %d", applicationErrors)
t.Logf(" Throughput: %.2f messages/second", throughput)
t.Logf(" Duration: %v", metrics.EndTime.Sub(metrics.StartTime))
if len(metrics.PublishLatencies) > 0 {
p95Latency := metrics.GetP95Latency(metrics.PublishLatencies)
t.Logf(" P95 Publish Latency: %v", p95Latency)
// Performance assertions (adjusted for error handling overhead)
assert.Less(t, p95Latency, 100*time.Millisecond, "P95 publish latency should be under 100ms")
}
// Enhanced error analysis
totalAttempts := publishedCount + totalErrors
if totalAttempts > 0 {
successRate := float64(publishedCount) / float64(totalAttempts)
connectionErrorRate := float64(connectionErrors) / float64(totalAttempts)
applicationErrorRate := float64(applicationErrors) / float64(totalAttempts)
t.Logf("Error Analysis:")
t.Logf(" Success Rate: %.2f%%", successRate*100)
t.Logf(" Connection Error Rate: %.2f%%", connectionErrorRate*100)
t.Logf(" Application Error Rate: %.2f%%", applicationErrorRate*100)
// Assertions based on error types
assert.Greater(t, successRate, 0.80, "Success rate should be greater than 80%")
assert.Less(t, applicationErrorRate, 0.01, "Application error rate should be less than 1%")
// Connection errors are expected under high load but should be handled
if connectionErrors > 0 {
t.Logf("Note: %d connection errors detected - this indicates the test is successfully stressing the system", connectionErrors)
t.Logf("Recommendation: Implement retry logic for production applications to handle these connection errors")
}
}
// Throughput requirements (adjusted for error handling)
expectedMinThroughput := 5000.0 // 5K messages/sec minimum
assert.Greater(t, throughput, expectedMinThroughput,
"Throughput should exceed %.0f messages/second", expectedMinThroughput)
}
+336
View File
@@ -0,0 +1,336 @@
package integration
import (
"fmt"
"math"
"sync"
"sync/atomic"
"time"
"github.com/seaweedfs/seaweedfs/weed/glog"
"github.com/seaweedfs/seaweedfs/weed/mq/client/pub_client"
"github.com/seaweedfs/seaweedfs/weed/pb/schema_pb"
"google.golang.org/grpc/codes"
"google.golang.org/grpc/status"
)
// ResilientPublisher wraps TopicPublisher with enhanced error handling
type ResilientPublisher struct {
publisher *pub_client.TopicPublisher
config *PublisherTestConfig
suite *IntegrationTestSuite
// Error tracking
connectionErrors int64
applicationErrors int64
retryAttempts int64
totalPublishes int64
// Retry configuration
maxRetries int
baseDelay time.Duration
maxDelay time.Duration
backoffFactor float64
// Circuit breaker
circuitOpen bool
circuitOpenTime time.Time
circuitTimeout time.Duration
mu sync.RWMutex
}
// RetryConfig holds retry configuration
type RetryConfig struct {
MaxRetries int
BaseDelay time.Duration
MaxDelay time.Duration
BackoffFactor float64
CircuitTimeout time.Duration
}
// DefaultRetryConfig returns sensible defaults for retry configuration
func DefaultRetryConfig() *RetryConfig {
return &RetryConfig{
MaxRetries: 5,
BaseDelay: 10 * time.Millisecond,
MaxDelay: 5 * time.Second,
BackoffFactor: 2.0,
CircuitTimeout: 30 * time.Second,
}
}
// NewResilientPublisher creates a new resilient publisher
func (its *IntegrationTestSuite) CreateResilientPublisher(config *PublisherTestConfig, retryConfig *RetryConfig) (*ResilientPublisher, error) {
if retryConfig == nil {
retryConfig = DefaultRetryConfig()
}
publisher, err := its.CreatePublisher(config)
if err != nil {
return nil, fmt.Errorf("failed to create base publisher: %v", err)
}
return &ResilientPublisher{
publisher: publisher,
config: config,
suite: its,
maxRetries: retryConfig.MaxRetries,
baseDelay: retryConfig.BaseDelay,
maxDelay: retryConfig.MaxDelay,
backoffFactor: retryConfig.BackoffFactor,
circuitTimeout: retryConfig.CircuitTimeout,
}, nil
}
// PublishWithRetry publishes a message with retry logic and error handling
func (rp *ResilientPublisher) PublishWithRetry(key, value []byte) error {
atomic.AddInt64(&rp.totalPublishes, 1)
// Check circuit breaker
if rp.isCircuitOpen() {
atomic.AddInt64(&rp.applicationErrors, 1)
return fmt.Errorf("circuit breaker is open")
}
var lastErr error
for attempt := 0; attempt <= rp.maxRetries; attempt++ {
if attempt > 0 {
atomic.AddInt64(&rp.retryAttempts, 1)
delay := rp.calculateDelay(attempt)
glog.V(1).Infof("Retrying publish after %v (attempt %d/%d)", delay, attempt, rp.maxRetries)
time.Sleep(delay)
}
err := rp.publisher.Publish(key, value)
if err == nil {
// Success - reset circuit breaker if it was open
rp.resetCircuitBreaker()
return nil
}
lastErr = err
// Classify error type
if rp.isConnectionError(err) {
atomic.AddInt64(&rp.connectionErrors, 1)
glog.V(1).Infof("Connection error on attempt %d: %v", attempt+1, err)
// For connection errors, try to recreate the publisher
if attempt < rp.maxRetries {
if recreateErr := rp.recreatePublisher(); recreateErr != nil {
glog.Warningf("Failed to recreate publisher: %v", recreateErr)
}
}
continue
} else {
// Application error - don't retry
atomic.AddInt64(&rp.applicationErrors, 1)
glog.Warningf("Application error (not retrying): %v", err)
break
}
}
// All retries exhausted or non-retryable error
rp.openCircuitBreaker()
return fmt.Errorf("publish failed after %d attempts, last error: %v", rp.maxRetries+1, lastErr)
}
// PublishRecord publishes a record with retry logic
func (rp *ResilientPublisher) PublishRecord(key []byte, record *schema_pb.RecordValue) error {
atomic.AddInt64(&rp.totalPublishes, 1)
if rp.isCircuitOpen() {
atomic.AddInt64(&rp.applicationErrors, 1)
return fmt.Errorf("circuit breaker is open")
}
var lastErr error
for attempt := 0; attempt <= rp.maxRetries; attempt++ {
if attempt > 0 {
atomic.AddInt64(&rp.retryAttempts, 1)
delay := rp.calculateDelay(attempt)
time.Sleep(delay)
}
err := rp.publisher.PublishRecord(key, record)
if err == nil {
rp.resetCircuitBreaker()
return nil
}
lastErr = err
if rp.isConnectionError(err) {
atomic.AddInt64(&rp.connectionErrors, 1)
if attempt < rp.maxRetries {
if recreateErr := rp.recreatePublisher(); recreateErr != nil {
glog.Warningf("Failed to recreate publisher: %v", recreateErr)
}
}
continue
} else {
atomic.AddInt64(&rp.applicationErrors, 1)
break
}
}
rp.openCircuitBreaker()
return fmt.Errorf("publish record failed after %d attempts, last error: %v", rp.maxRetries+1, lastErr)
}
// recreatePublisher attempts to recreate the underlying publisher
func (rp *ResilientPublisher) recreatePublisher() error {
rp.mu.Lock()
defer rp.mu.Unlock()
// Shutdown old publisher
if rp.publisher != nil {
rp.publisher.Shutdown()
}
// Create new publisher
newPublisher, err := rp.suite.CreatePublisher(rp.config)
if err != nil {
return fmt.Errorf("failed to recreate publisher: %v", err)
}
rp.publisher = newPublisher
glog.V(1).Infof("Successfully recreated publisher")
return nil
}
// isConnectionError determines if an error is a connection-level error that should be retried
func (rp *ResilientPublisher) isConnectionError(err error) bool {
if err == nil {
return false
}
errStr := err.Error()
// Check for gRPC status codes
if st, ok := status.FromError(err); ok {
switch st.Code() {
case codes.Unavailable, codes.DeadlineExceeded, codes.Canceled, codes.Unknown:
return true
}
}
// Check for common connection error strings
connectionErrorPatterns := []string{
"EOF",
"error reading server preface",
"connection refused",
"connection reset",
"broken pipe",
"network is unreachable",
"no route to host",
"transport is closing",
"connection error",
"dial tcp",
"context deadline exceeded",
}
for _, pattern := range connectionErrorPatterns {
if contains(errStr, pattern) {
return true
}
}
return false
}
// contains checks if a string contains a substring (case-insensitive)
func contains(s, substr string) bool {
return len(s) >= len(substr) &&
(s == substr ||
len(s) > len(substr) &&
(s[:len(substr)] == substr ||
s[len(s)-len(substr):] == substr ||
containsSubstring(s, substr)))
}
func containsSubstring(s, substr string) bool {
for i := 0; i <= len(s)-len(substr); i++ {
if s[i:i+len(substr)] == substr {
return true
}
}
return false
}
// calculateDelay calculates exponential backoff delay
func (rp *ResilientPublisher) calculateDelay(attempt int) time.Duration {
delay := float64(rp.baseDelay) * math.Pow(rp.backoffFactor, float64(attempt-1))
if delay > float64(rp.maxDelay) {
delay = float64(rp.maxDelay)
}
return time.Duration(delay)
}
// Circuit breaker methods
func (rp *ResilientPublisher) isCircuitOpen() bool {
rp.mu.RLock()
defer rp.mu.RUnlock()
if !rp.circuitOpen {
return false
}
// Check if circuit should be reset
if time.Since(rp.circuitOpenTime) > rp.circuitTimeout {
return false
}
return true
}
func (rp *ResilientPublisher) openCircuitBreaker() {
rp.mu.Lock()
defer rp.mu.Unlock()
rp.circuitOpen = true
rp.circuitOpenTime = time.Now()
glog.Warningf("Circuit breaker opened due to repeated failures")
}
func (rp *ResilientPublisher) resetCircuitBreaker() {
rp.mu.Lock()
defer rp.mu.Unlock()
if rp.circuitOpen {
rp.circuitOpen = false
glog.V(1).Infof("Circuit breaker reset")
}
}
// GetErrorStats returns error statistics
func (rp *ResilientPublisher) GetErrorStats() ErrorStats {
return ErrorStats{
ConnectionErrors: atomic.LoadInt64(&rp.connectionErrors),
ApplicationErrors: atomic.LoadInt64(&rp.applicationErrors),
RetryAttempts: atomic.LoadInt64(&rp.retryAttempts),
TotalPublishes: atomic.LoadInt64(&rp.totalPublishes),
CircuitOpen: rp.isCircuitOpen(),
}
}
// ErrorStats holds error statistics
type ErrorStats struct {
ConnectionErrors int64
ApplicationErrors int64
RetryAttempts int64
TotalPublishes int64
CircuitOpen bool
}
// Shutdown gracefully shuts down the resilient publisher
func (rp *ResilientPublisher) Shutdown() error {
rp.mu.Lock()
defer rp.mu.Unlock()
if rp.publisher != nil {
return rp.publisher.Shutdown()
}
return nil
}
+286
View File
@@ -0,0 +1,286 @@
# SeaweedMQ Integration Test Design
## Overview
This document outlines the comprehensive integration test strategy for SeaweedMQ, covering all critical functionalities from basic pub/sub operations to advanced features like auto-scaling, failover, and performance testing.
## Architecture Under Test
SeaweedMQ consists of:
- **Masters**: Cluster coordination and metadata management
- **Volume Servers**: Storage layer for persistent messages
- **Filers**: File system interface for metadata storage
- **Brokers**: Message processing and routing (stateless)
- **Agents**: Client interface for pub/sub operations
- **Schema System**: Protobuf-based message schema management
## Test Categories
### 1. Basic Functionality Tests
#### 1.1 Basic Pub/Sub Operations
- **Test**: `TestBasicPublishSubscribe`
- Publish messages to a topic
- Subscribe and receive messages
- Verify message content and ordering
- Test with different data types (string, int, bytes, records)
- **Test**: `TestMultipleConsumers`
- Multiple subscribers on same topic
- Verify message distribution
- Test consumer group functionality
- **Test**: `TestMessageOrdering`
- Publish messages in sequence
- Verify FIFO ordering within partitions
- Test with different partition keys
#### 1.2 Schema Management
- **Test**: `TestSchemaValidation`
- Publish with valid schemas
- Reject invalid schema messages
- Test schema evolution scenarios
- **Test**: `TestRecordTypes`
- Nested record structures
- List types and complex schemas
- Schema-to-Parquet conversion
### 2. Partitioning and Scaling Tests
#### 2.1 Partition Management
- **Test**: `TestPartitionDistribution`
- Messages distributed across partitions based on keys
- Verify partition assignment logic
- Test partition rebalancing
- **Test**: `TestAutoSplitMerge`
- Simulate high load to trigger auto-split
- Simulate low load to trigger auto-merge
- Verify data consistency during splits/merges
#### 2.2 Broker Scaling
- **Test**: `TestBrokerAddRemove`
- Add brokers during operation
- Remove brokers gracefully
- Verify partition reassignment
- **Test**: `TestLoadBalancing`
- Verify even load distribution across brokers
- Test with varying message sizes and rates
- Monitor broker resource utilization
### 3. Failover and Reliability Tests
#### 3.1 Broker Failover
- **Test**: `TestBrokerFailover`
- Kill leader broker during publishing
- Verify seamless failover to follower
- Test data consistency after failover
- **Test**: `TestBrokerRecovery`
- Broker restart scenarios
- State recovery from storage
- Partition reassignment after recovery
#### 3.2 Data Durability
- **Test**: `TestMessagePersistence`
- Publish messages and restart cluster
- Verify all messages are recovered
- Test with different replication settings
- **Test**: `TestFollowerReplication`
- Leader-follower message replication
- Verify consistency between replicas
- Test follower promotion scenarios
### 4. Agent Functionality Tests
#### 4.1 Session Management
- **Test**: `TestPublishSessions`
- Create/close publish sessions
- Concurrent session management
- Session cleanup after failures
- **Test**: `TestSubscribeSessions`
- Subscribe session lifecycle
- Consumer group management
- Offset tracking and acknowledgments
#### 4.2 Error Handling
- **Test**: `TestConnectionFailures`
- Network partitions between agent and broker
- Automatic reconnection logic
- Message buffering during outages
### 5. Performance and Load Tests
#### 5.1 Throughput Tests
- **Test**: `TestHighThroughputPublish`
- Publish 100K+ messages/second
- Monitor system resources
- Verify no message loss
- **Test**: `TestHighThroughputSubscribe`
- Multiple consumers processing high volume
- Monitor processing latency
- Test backpressure handling
#### 5.2 Spike Traffic Tests
- **Test**: `TestTrafficSpikes`
- Sudden increase in message volume
- Auto-scaling behavior verification
- Resource utilization patterns
- **Test**: `TestLargeMessages`
- Messages with large payloads (MB size)
- Memory usage monitoring
- Storage efficiency testing
### 6. End-to-End Scenarios
#### 6.1 Complete Workflow Tests
- **Test**: `TestProducerConsumerWorkflow`
- Multi-stage data processing pipeline
- Producer → Topic → Multiple Consumers
- Data transformation and aggregation
- **Test**: `TestMultiTopicOperations`
- Multiple topics with different schemas
- Cross-topic message routing
- Topic management operations
## Test Infrastructure
### Environment Setup
#### Docker Compose Configuration
```yaml
# test-environment.yml
version: '3.9'
services:
master-cluster:
# 3 master nodes for HA
volume-cluster:
# 3 volume servers for data storage
filer-cluster:
# 2 filers for metadata
broker-cluster:
# 3 brokers for message processing
test-runner:
# Container to run integration tests
```
#### Test Data Management
- Pre-defined test schemas
- Sample message datasets
- Performance benchmarking data
### Test Framework Structure
```go
// Base test framework
type IntegrationTestSuite struct {
masters []string
brokers []string
filers []string
testClient *TestClient
cleanup []func()
}
// Test utilities
type TestClient struct {
publishers map[string]*pub_client.TopicPublisher
subscribers map[string]*sub_client.TopicSubscriber
agents []*agent.MessageQueueAgent
}
```
### Monitoring and Metrics
#### Health Checks
- Broker connectivity status
- Master cluster health
- Storage system availability
- Network connectivity between components
#### Performance Metrics
- Message throughput (msgs/sec)
- End-to-end latency
- Resource utilization (CPU, Memory, Disk)
- Network bandwidth usage
## Test Execution Strategy
### Parallel Test Execution
- Categorize tests by resource requirements
- Run independent tests in parallel
- Serialize tests that modify cluster state
### Continuous Integration
- Automated test runs on PR submissions
- Performance regression detection
- Multi-platform testing (Linux, macOS, Windows)
### Test Environment Management
- Docker-based isolated environments
- Automatic cleanup after test completion
- Resource monitoring and alerts
## Success Criteria
### Functional Requirements
- ✅ All messages published are received by subscribers
- ✅ Message ordering preserved within partitions
- ✅ Schema validation works correctly
- ✅ Auto-scaling triggers at expected thresholds
- ✅ Failover completes within 30 seconds
- ✅ No data loss during normal operations
### Performance Requirements
- ✅ Throughput: 50K+ messages/second/broker
- ✅ Latency: P95 < 100ms end-to-end
- ✅ Memory usage: < 1GB per broker under normal load
- ✅ Storage efficiency: < 20% overhead vs raw message size
### Reliability Requirements
- ✅ 99.9% uptime during normal operations
- ✅ Automatic recovery from single component failures
- ✅ Data consistency maintained across all scenarios
- ✅ Graceful degradation under resource constraints
## Implementation Timeline
### Phase 1: Core Functionality (Week 1-2)
- Basic pub/sub tests
- Schema validation tests
- Simple failover scenarios
### Phase 2: Advanced Features (Week 3-4)
- Auto-scaling tests
- Complex failover scenarios
- Agent functionality tests
### Phase 3: Performance & Load (Week 5-6)
- Throughput and latency tests
- Spike traffic handling
- Resource utilization monitoring
### Phase 4: End-to-End (Week 7-8)
- Complete workflow tests
- Multi-component integration
- Performance regression testing
## Maintenance and Updates
### Regular Updates
- Add tests for new features
- Update performance baselines
- Enhance error scenarios coverage
### Test Data Refresh
- Generate new test datasets quarterly
- Update schema examples
- Refresh performance benchmarks
This comprehensive test design ensures SeaweedMQ's reliability, performance, and functionality across all critical use cases and failure scenarios.
+54
View File
@@ -0,0 +1,54 @@
global:
scrape_interval: 15s
evaluation_interval: 15s
rule_files:
# - "first_rules.yml"
# - "second_rules.yml"
scrape_configs:
# SeaweedFS Masters
- job_name: 'seaweedfs-master'
static_configs:
- targets:
- 'master0:9333'
- 'master1:9334'
- 'master2:9335'
metrics_path: '/metrics'
scrape_interval: 10s
# SeaweedFS Volume Servers
- job_name: 'seaweedfs-volume'
static_configs:
- targets:
- 'volume1:8080'
- 'volume2:8081'
- 'volume3:8082'
metrics_path: '/metrics'
scrape_interval: 10s
# SeaweedFS Filers
- job_name: 'seaweedfs-filer'
static_configs:
- targets:
- 'filer1:8888'
- 'filer2:8889'
metrics_path: '/metrics'
scrape_interval: 10s
# SeaweedMQ Brokers
- job_name: 'seaweedmq-broker'
static_configs:
- targets:
- 'broker1:17777'
- 'broker2:17778'
- 'broker3:17779'
metrics_path: '/metrics'
scrape_interval: 5s
# Docker containers
- job_name: 'docker'
static_configs:
- targets: ['localhost:9323']
metrics_path: '/metrics'
scrape_interval: 30s
+56
View File
@@ -0,0 +1,56 @@
package main
import (
"fmt"
"log"
"time"
"github.com/seaweedfs/seaweedfs/weed/mq/client/pub_client"
"github.com/seaweedfs/seaweedfs/weed/mq/topic"
)
func main() {
log.Println("Starting SeaweedMQ logging test...")
// Create publisher configuration
config := &pub_client.PublisherConfiguration{
Topic: topic.NewTopic("test", "logging-demo"),
PartitionCount: 3,
Brokers: []string{"127.0.0.1:17777"},
PublisherName: "logging-test-client",
}
log.Println("Creating topic publisher...")
publisher, err := pub_client.NewTopicPublisher(config)
if err != nil {
log.Printf("Failed to create publisher: %v", err)
return
}
defer publisher.Shutdown()
log.Println("Publishing test messages...")
// Publish some test messages
for i := 0; i < 100; i++ {
key := fmt.Sprintf("key-%d", i)
value := fmt.Sprintf("message-%d-timestamp-%d", i, time.Now().Unix())
err := publisher.Publish([]byte(key), []byte(value))
if err != nil {
log.Printf("Failed to publish message %d: %v", i, err)
}
// Small delay to create some connection stress
if i%10 == 0 {
time.Sleep(10 * time.Millisecond)
}
}
log.Println("Finishing publish...")
err = publisher.FinishPublish()
if err != nil {
log.Printf("Failed to finish publish: %v", err)
}
log.Println("Test completed successfully!")
}
+1 -1
View File
@@ -32,7 +32,7 @@ func init() {
var cmdMqAgent = &Command{
UsageLine: "mq.agent [-port=16777] [-broker=<ip:port>]",
Short: "<WIP> start a message queue agent",
Short: "start a message queue agent",
Long: `start a message queue agent
The agent runs on local server to accept gRPC calls to write or read messages.
+1 -1
View File
@@ -43,7 +43,7 @@ func init() {
var cmdMqBroker = &Command{
UsageLine: "mq.broker [-port=17777] [-master=<ip:port>]",
Short: "<WIP> start a message queue broker",
Short: "start a message queue broker",
Long: `start a message queue broker
The broker can accept gRPC calls to write or read messages. The messages are stored via filer.
+52 -6
View File
@@ -3,15 +3,19 @@ package broker
import (
"context"
"fmt"
"github.com/seaweedfs/seaweedfs/weed/glog"
"github.com/seaweedfs/seaweedfs/weed/mq/topic"
"github.com/seaweedfs/seaweedfs/weed/pb/mq_pb"
"google.golang.org/grpc/peer"
"io"
"math/rand"
"net"
"strings"
"sync/atomic"
"time"
"github.com/seaweedfs/seaweedfs/weed/glog"
"github.com/seaweedfs/seaweedfs/weed/mq/topic"
"github.com/seaweedfs/seaweedfs/weed/pb/mq_pb"
"google.golang.org/grpc/codes"
"google.golang.org/grpc/peer"
"google.golang.org/grpc/status"
)
// PUB
@@ -126,7 +130,12 @@ func (b *MessageQueueBroker) PublishMessage(stream mq_pb.SeaweedMessaging_Publis
if err == io.EOF {
break
}
glog.V(0).Infof("topic %v partition %v publish stream from %s error: %v", initMessage.Topic, initMessage.Partition, initMessage.PublisherName, err)
// Use appropriate logging level based on error type
if isConnectionError(err) {
glog.V(1).Infof("topic %v partition %v publish stream from %s connection error (client disconnected): %v", initMessage.Topic, initMessage.Partition, initMessage.PublisherName, err)
} else {
glog.V(0).Infof("topic %v partition %v publish stream from %s error: %v", initMessage.Topic, initMessage.Partition, initMessage.PublisherName, err)
}
break
}
@@ -145,7 +154,7 @@ func (b *MessageQueueBroker) PublishMessage(stream mq_pb.SeaweedMessaging_Publis
}
}
glog.V(0).Infof("topic %v partition %v publish stream from %s closed.", initMessage.Topic, initMessage.Partition, initMessage.PublisherName)
glog.V(1).Infof("topic %v partition %v publish stream from %s closed.", initMessage.Topic, initMessage.Partition, initMessage.PublisherName)
return nil
}
@@ -164,3 +173,40 @@ func findClientAddress(ctx context.Context) string {
}
return pr.Addr.String()
}
// isConnectionError checks if an error is a connection-level error
func isConnectionError(err error) bool {
if err == nil {
return false
}
// Check gRPC status codes for connection errors
if grpcStatus, ok := status.FromError(err); ok {
switch grpcStatus.Code() {
case codes.Unavailable, codes.DeadlineExceeded, codes.Canceled, codes.Unknown:
return true
}
}
// Check for common connection error patterns
errStr := strings.ToLower(err.Error())
connectionErrorPatterns := []string{
"eof",
"error reading server preface",
"connection refused",
"connection reset",
"broken pipe",
"no such host",
"network is unreachable",
"context deadline exceeded",
"context canceled",
}
for _, pattern := range connectionErrorPatterns {
if strings.Contains(errStr, pattern) {
return true
}
}
return false
}
+12 -6
View File
@@ -2,13 +2,14 @@ package broker
import (
"fmt"
"io"
"time"
"github.com/seaweedfs/seaweedfs/weed/glog"
"github.com/seaweedfs/seaweedfs/weed/mq/topic"
"github.com/seaweedfs/seaweedfs/weed/pb/mq_pb"
"github.com/seaweedfs/seaweedfs/weed/util/buffered_queue"
"github.com/seaweedfs/seaweedfs/weed/util/log_buffer"
"io"
"time"
)
type memBuffer struct {
@@ -43,7 +44,12 @@ func (b *MessageQueueBroker) PublishFollowMe(stream mq_pb.SeaweedMessaging_Publi
err = nil
break
}
glog.V(0).Infof("topic %v partition %v publish stream error: %v", initMessage.Topic, initMessage.Partition, err)
// Use appropriate logging level based on error type
if isConnectionError(err) {
glog.V(1).Infof("topic %v partition %v publish follower stream connection error (leader disconnected): %v", initMessage.Topic, initMessage.Partition, err)
} else {
glog.V(0).Infof("topic %v partition %v publish stream error: %v", initMessage.Topic, initMessage.Partition, err)
}
break
}
@@ -62,10 +68,10 @@ func (b *MessageQueueBroker) PublishFollowMe(stream mq_pb.SeaweedMessaging_Publi
}
// println("ack", string(dataMessage.Key), dataMessage.TsNs)
} else if closeMessage := req.GetClose(); closeMessage != nil {
glog.V(0).Infof("topic %v partition %v publish stream closed: %v", initMessage.Topic, initMessage.Partition, closeMessage)
glog.V(1).Infof("topic %v partition %v publish stream closed: %v", initMessage.Topic, initMessage.Partition, closeMessage)
break
} else if flushMessage := req.GetFlush(); flushMessage != nil {
glog.V(0).Infof("topic %v partition %v publish stream flushed: %v", initMessage.Topic, initMessage.Partition, flushMessage)
glog.V(1).Infof("topic %v partition %v publish stream flushed: %v", initMessage.Topic, initMessage.Partition, flushMessage)
lastFlushTsNs = flushMessage.TsNs
@@ -124,7 +130,7 @@ func (b *MessageQueueBroker) PublishFollowMe(stream mq_pb.SeaweedMessaging_Publi
glog.V(0).Infof("flushed remaining data at %v to %s size %d", mem.stopTime.UnixNano(), targetFile, len(mem.buf))
}
glog.V(0).Infof("shut down follower for %v %v", t, p)
glog.V(1).Infof("shut down follower for %v %v", t, p)
return err
}
+68 -14
View File
@@ -3,6 +3,13 @@ package pub_client
import (
"context"
"fmt"
"log"
"sort"
"strings"
"sync"
"sync/atomic"
"time"
"github.com/seaweedfs/seaweedfs/weed/glog"
"github.com/seaweedfs/seaweedfs/weed/pb"
"github.com/seaweedfs/seaweedfs/weed/pb/mq_pb"
@@ -11,11 +18,6 @@ import (
"google.golang.org/grpc/codes"
"google.golang.org/grpc/credentials/insecure"
"google.golang.org/grpc/status"
"log"
"sort"
"sync"
"sync/atomic"
"time"
)
type EachPartitionError struct {
@@ -39,7 +41,7 @@ func (p *TopicPublisher) startSchedulerThread(wg *sync.WaitGroup) error {
return fmt.Errorf("configure topic %s: %v", p.config.Topic, err)
}
log.Printf("start scheduler thread for topic %s", p.config.Topic)
glog.V(0).Infof("start scheduler thread for topic %s", p.config.Topic)
generation := 0
var errChan chan EachPartitionError
@@ -66,7 +68,12 @@ func (p *TopicPublisher) startSchedulerThread(wg *sync.WaitGroup) error {
for {
select {
case eachErr := <-errChan:
glog.Errorf("gen %d publish to topic %s partition %v: %v", eachErr.generation, p.config.Topic, eachErr.Partition, eachErr.Err)
// Use appropriate logging level based on error type
if isConnectionError(eachErr.Err) {
glog.V(1).Infof("gen %d connection error for topic %s partition %v (will retry): %v", eachErr.generation, p.config.Topic, eachErr.Partition, eachErr.Err)
} else {
glog.Errorf("gen %d publish to topic %s partition %v: %v", eachErr.generation, p.config.Topic, eachErr.Partition, eachErr.Err)
}
if eachErr.generation < generation {
continue
}
@@ -114,7 +121,12 @@ func (p *TopicPublisher) onEachAssignments(generation int, assignments []*mq_pb.
go func(job *EachPartitionPublishJob) {
defer job.wg.Done()
if err := p.doPublishToPartition(job); err != nil {
log.Printf("publish to %s partition %v: %v", p.config.Topic, job.Partition, err)
// Use appropriate logging level based on error type
if isConnectionError(err) {
glog.V(1).Infof("connection error publishing to %s partition %v (will retry): %v", p.config.Topic, job.Partition, err)
} else {
log.Printf("publish to %s partition %v: %v", p.config.Topic, job.Partition, err)
}
errChan <- EachPartitionError{assignment, err, generation}
}
}(job)
@@ -128,7 +140,7 @@ func (p *TopicPublisher) onEachAssignments(generation int, assignments []*mq_pb.
func (p *TopicPublisher) doPublishToPartition(job *EachPartitionPublishJob) error {
log.Printf("connecting to %v for topic partition %+v", job.LeaderBroker, job.Partition)
glog.V(1).Infof("connecting to %v for topic partition %+v", job.LeaderBroker, job.Partition)
grpcConnection, err := grpc.NewClient(job.LeaderBroker, grpc.WithTransportCredentials(insecure.NewCredentials()), p.grpcDialOption)
if err != nil {
@@ -176,20 +188,25 @@ func (p *TopicPublisher) doPublishToPartition(job *EachPartitionPublishJob) erro
if err != nil {
e, _ := status.FromError(err)
if e.Code() == codes.Unknown && e.Message() == "EOF" {
log.Printf("publish to %s EOF", publishClient.Broker)
glog.V(1).Infof("publish stream to %s closed (EOF)", publishClient.Broker)
return
}
publishClient.Err = err
log.Printf("publish1 to %s error: %v\n", publishClient.Broker, err)
// Use appropriate logging level based on error type
if isConnectionError(err) {
glog.V(1).Infof("connection error receiving from %s (will retry): %v", publishClient.Broker, err)
} else {
log.Printf("publish receive error from %s: %v", publishClient.Broker, err)
}
return
}
if ackResp.Error != "" {
publishClient.Err = fmt.Errorf("ack error: %v", ackResp.Error)
log.Printf("publish2 to %s error: %v\n", publishClient.Broker, ackResp.Error)
log.Printf("publish ack error from %s: %v", publishClient.Broker, ackResp.Error)
return
}
if ackResp.AckSequence > 0 {
log.Printf("ack %d published %d hasMoreData:%d", ackResp.AckSequence, atomic.LoadInt64(&publishedTsNs), atomic.LoadInt32(&hasMoreData))
glog.V(2).Infof("ack %d published %d hasMoreData:%d", ackResp.AckSequence, atomic.LoadInt64(&publishedTsNs), atomic.LoadInt32(&hasMoreData))
}
if atomic.LoadInt64(&publishedTsNs) <= ackResp.AckSequence && atomic.LoadInt32(&hasMoreData) == 0 {
return
@@ -222,7 +239,7 @@ func (p *TopicPublisher) doPublishToPartition(job *EachPartitionPublishJob) erro
}
}
log.Printf("published %d messages to %v for topic partition %+v", publishCounter, job.LeaderBroker, job.Partition)
glog.V(1).Infof("published %d messages to %v for topic partition %+v", publishCounter, job.LeaderBroker, job.Partition)
return nil
}
@@ -296,3 +313,40 @@ func (p *TopicPublisher) doLookupTopicPartitions() (assignments []*mq_pb.BrokerP
return nil, fmt.Errorf("lookup topic %s: %v", p.config.Topic, lastErr)
}
// isConnectionError checks if an error is a connection-level error that is handled automatically
func isConnectionError(err error) bool {
if err == nil {
return false
}
// Check gRPC status codes for connection errors
if grpcStatus, ok := status.FromError(err); ok {
switch grpcStatus.Code() {
case codes.Unavailable, codes.DeadlineExceeded, codes.Canceled, codes.Unknown:
return true
}
}
// Check for common connection error patterns
errStr := strings.ToLower(err.Error())
connectionErrorPatterns := []string{
"eof",
"error reading server preface",
"connection refused",
"connection reset",
"broken pipe",
"no such host",
"network is unreachable",
"context deadline exceeded",
"context canceled",
}
for _, pattern := range connectionErrorPatterns {
if strings.Contains(errStr, pattern) {
return true
}
}
return false
}
+48 -4
View File
@@ -3,6 +3,11 @@ package topic
import (
"context"
"fmt"
"strings"
"sync"
"sync/atomic"
"time"
"github.com/seaweedfs/seaweedfs/weed/glog"
"github.com/seaweedfs/seaweedfs/weed/pb"
"github.com/seaweedfs/seaweedfs/weed/pb/mq_pb"
@@ -10,9 +15,6 @@ import (
"google.golang.org/grpc"
"google.golang.org/grpc/codes"
"google.golang.org/grpc/status"
"sync"
"sync/atomic"
"time"
)
type LocalPartition struct {
@@ -182,7 +184,12 @@ func (p *LocalPartition) MaybeConnectToFollowers(initMessage *mq_pb.PublishMessa
glog.V(0).Infof("local partition %v follower %v stopped", p.Partition, p.Follower)
return
}
glog.Errorf("Receiving local partition %v follower %s ack: %v", p.Partition, p.Follower, err)
// Use appropriate logging level based on error type
if isConnectionError(err) {
glog.V(1).Infof("local partition %v follower %s connection error (will retry): %v", p.Partition, p.Follower, err)
} else {
glog.Errorf("Receiving local partition %v follower %s ack: %v", p.Partition, p.Follower, err)
}
return
}
atomic.StoreInt64(&p.AckTsNs, ack.AckTsNs)
@@ -242,3 +249,40 @@ func (p *LocalPartition) NotifyLogFlushed(flushTsNs int64) {
// println("notifying", p.Follower, "flushed at", flushTsNs)
}
}
// isConnectionError checks if an error is a connection-level error
func isConnectionError(err error) bool {
if err == nil {
return false
}
// Check gRPC status codes for connection errors
if grpcStatus, ok := status.FromError(err); ok {
switch grpcStatus.Code() {
case codes.Unavailable, codes.DeadlineExceeded, codes.Canceled, codes.Unknown:
return true
}
}
// Check for common connection error patterns
errStr := strings.ToLower(err.Error())
connectionErrorPatterns := []string{
"eof",
"error reading server preface",
"connection refused",
"connection reset",
"broken pipe",
"no such host",
"network is unreachable",
"context deadline exceeded",
"context canceled",
}
for _, pattern := range connectionErrorPatterns {
if strings.Contains(errStr, pattern) {
return true
}
}
return false
}