mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-08-16 12:16:36 +00:00
* master: stream volume listings A listing of 800k volumes is 36MB on the wire but 305MB as messages, and the master built all of it, then held it while grpc encoded it. Two of those at once is most of a small master's heap, and the maintenance scanner asks every 30 minutes. The topology goes out first, listing nothing, then its volumes in batches, so the master holds a batch rather than a cluster: 341MB of live heap for one listing becomes 4.4MB. It allocates much the same either way -- what changes is how much of it has to be live at once, which is what sets the heap ceiling. Batches are built under their disk's lock and sent outside it, so a slow reader stalls the stream rather than the topology. They therefore do not share one instant, which a single listing did not either: it takes each disk's lock in turn, so a volume moving during either can be seen twice or not at all. The client helper hides which kind of master answered: one too old for the stream is asked the old way and its reply cut into the same batches. Either way the topology handed over lists no volumes, so a caller cannot come to depend on finding them there. * admin: stream the listing the maintenance scan reads It asks for every volume in the cluster every 30 minutes. Reassembling it client-side keeps the scan identical -- ActiveTopology splits disks by the disk ids on the volumes, so it needs them in the topology -- while the master no longer builds the whole reply to send it.
596 lines
17 KiB
Protocol Buffer
596 lines
17 KiB
Protocol Buffer
syntax = "proto3";
|
|
|
|
package master_pb;
|
|
|
|
option go_package = "github.com/seaweedfs/seaweedfs/weed/pb/master_pb";
|
|
|
|
import "volume_server.proto";
|
|
|
|
//////////////////////////////////////////////////
|
|
|
|
service Seaweed {
|
|
rpc SendHeartbeat (stream Heartbeat) returns (stream HeartbeatResponse) {
|
|
}
|
|
rpc KeepConnected (stream KeepConnectedRequest) returns (stream KeepConnectedResponse) {
|
|
}
|
|
rpc LookupVolume (LookupVolumeRequest) returns (LookupVolumeResponse) {
|
|
}
|
|
rpc Assign (AssignRequest) returns (AssignResponse) {
|
|
}
|
|
rpc StreamAssign (stream AssignRequest) returns (stream AssignResponse) {
|
|
}
|
|
rpc Statistics (StatisticsRequest) returns (StatisticsResponse) {
|
|
}
|
|
rpc CollectionList (CollectionListRequest) returns (CollectionListResponse) {
|
|
}
|
|
rpc CollectionDelete (CollectionDeleteRequest) returns (CollectionDeleteResponse) {
|
|
}
|
|
rpc VolumeList (VolumeListRequest) returns (VolumeListResponse) {
|
|
}
|
|
rpc VolumeListStream (VolumeListRequest) returns (stream VolumeListStreamResponse) {
|
|
}
|
|
rpc LookupEcVolume (LookupEcVolumeRequest) returns (LookupEcVolumeResponse) {
|
|
}
|
|
rpc VacuumVolume (VacuumVolumeRequest) returns (VacuumVolumeResponse) {
|
|
}
|
|
rpc DisableVacuum (DisableVacuumRequest) returns (DisableVacuumResponse) {
|
|
}
|
|
rpc EnableVacuum (EnableVacuumRequest) returns (EnableVacuumResponse) {
|
|
}
|
|
rpc VolumeMarkReadonly (VolumeMarkReadonlyRequest) returns (VolumeMarkReadonlyResponse) {
|
|
}
|
|
rpc GetMasterConfiguration (GetMasterConfigurationRequest) returns (GetMasterConfigurationResponse) {
|
|
}
|
|
rpc ListClusterNodes (ListClusterNodesRequest) returns (ListClusterNodesResponse) {
|
|
}
|
|
rpc LeaseAdminToken (LeaseAdminTokenRequest) returns (LeaseAdminTokenResponse) {
|
|
}
|
|
rpc ReleaseAdminToken (ReleaseAdminTokenRequest) returns (ReleaseAdminTokenResponse) {
|
|
}
|
|
rpc GetAdminLockStatus (GetAdminLockStatusRequest) returns (GetAdminLockStatusResponse) {
|
|
}
|
|
rpc Ping (PingRequest) returns (PingResponse) {
|
|
}
|
|
rpc RaftListClusterServers (RaftListClusterServersRequest) returns (RaftListClusterServersResponse) {
|
|
}
|
|
rpc RaftAddServer (RaftAddServerRequest) returns (RaftAddServerResponse) {
|
|
}
|
|
rpc RaftRemoveServer (RaftRemoveServerRequest) returns (RaftRemoveServerResponse) {
|
|
}
|
|
rpc RaftLeadershipTransfer (RaftLeadershipTransferRequest) returns (RaftLeadershipTransferResponse) {
|
|
}
|
|
rpc VolumeGrow (VolumeGrowRequest) returns (VolumeGrowResponse) {
|
|
}
|
|
rpc CollectionStatistics (CollectionStatisticsRequest) returns (CollectionStatisticsResponse) {
|
|
}
|
|
}
|
|
|
|
//////////////////////////////////////////////////
|
|
|
|
message DiskTag {
|
|
uint32 disk_id = 1;
|
|
repeated string tags = 2;
|
|
// Physical disk descriptor, reported for every location including empty ones.
|
|
string type = 3;
|
|
int64 max_volume_count = 4;
|
|
}
|
|
|
|
message Heartbeat {
|
|
string ip = 1;
|
|
uint32 port = 2;
|
|
string public_url = 3;
|
|
uint64 max_file_key = 5;
|
|
string data_center = 6;
|
|
string rack = 7;
|
|
uint32 admin_port = 8;
|
|
repeated VolumeInformationMessage volumes = 9;
|
|
// delta volumes
|
|
repeated VolumeShortInformationMessage new_volumes = 10;
|
|
repeated VolumeShortInformationMessage deleted_volumes = 11;
|
|
bool has_no_volumes = 12;
|
|
|
|
// erasure coding
|
|
repeated VolumeEcShardInformationMessage ec_shards = 16;
|
|
// delta erasure coding shards
|
|
repeated VolumeEcShardInformationMessage new_ec_shards = 17;
|
|
repeated VolumeEcShardInformationMessage deleted_ec_shards = 18;
|
|
bool has_no_ec_shards = 19;
|
|
|
|
map<string, uint32> max_volume_counts = 4;
|
|
uint32 grpc_port = 20;
|
|
repeated string location_uuids = 21;
|
|
string id = 22; // volume server id, independent of ip:port for stable identification
|
|
|
|
// state flags
|
|
volume_server_pb.VolumeServerState state = 23;
|
|
|
|
repeated DiskTag disk_tags = 24;
|
|
|
|
// physical disk capacity per disk type, in bytes, from the underlying filesystem
|
|
map<string, uint64> disk_total_bytes = 25;
|
|
map<string, uint64> disk_free_bytes = 26;
|
|
|
|
// Digest of every volume in this heartbeat's view of the server, letting the
|
|
// master check its copy is current without being sent the whole list. Absent
|
|
// from servers that do not compute it, and distinct from a digest of 0, which
|
|
// is what a server holding no volumes reports.
|
|
optional uint64 volume_digest = 27;
|
|
// Volumes whose reported state changed since the last heartbeat, sent in
|
|
// place of `volumes`. A master that does not understand this never sets
|
|
// volume_digest_supported, so it keeps being sent the whole list.
|
|
repeated VolumeInformationMessage changed_volumes = 28;
|
|
}
|
|
|
|
message HeartbeatResponse {
|
|
uint64 volume_size_limit = 1;
|
|
string leader = 2;
|
|
string metrics_address = 3;
|
|
uint32 metrics_interval_seconds = 4;
|
|
repeated StorageBackend storage_backends = 5;
|
|
repeated string duplicated_uuids = 6;
|
|
bool preallocate = 7;
|
|
// The master's view of this server's volumes disagrees with the reported
|
|
// digest, so it needs the full volume list rather than changes alone.
|
|
bool resend_full_volume_list = 8;
|
|
// The master compares volume digests, so a server that reports one may send
|
|
// changed_volumes in place of its whole list.
|
|
bool volume_digest_supported = 9;
|
|
}
|
|
|
|
message VolumeInformationMessage {
|
|
uint32 id = 1;
|
|
uint64 size = 2;
|
|
string collection = 3;
|
|
uint64 file_count = 4;
|
|
uint64 delete_count = 5;
|
|
uint64 deleted_byte_count = 6;
|
|
bool read_only = 7;
|
|
uint32 replica_placement = 8;
|
|
uint32 version = 9;
|
|
uint32 ttl = 10;
|
|
uint32 compact_revision = 11;
|
|
int64 modified_at_second = 12;
|
|
string remote_storage_name = 13;
|
|
string remote_storage_key = 14;
|
|
string disk_type = 15;
|
|
uint32 disk_id = 16;
|
|
}
|
|
|
|
message VolumeShortInformationMessage {
|
|
uint32 id = 1;
|
|
string collection = 3;
|
|
uint32 replica_placement = 8;
|
|
uint32 version = 9;
|
|
uint32 ttl = 10;
|
|
string disk_type = 15;
|
|
uint32 disk_id = 16;
|
|
}
|
|
|
|
message VolumeEcShardInformationMessage {
|
|
uint32 id = 1;
|
|
string collection = 2;
|
|
uint32 ec_index_bits = 3;
|
|
string disk_type = 4;
|
|
uint64 expire_at_sec = 5; // used to record the destruction time of ec volume
|
|
uint32 disk_id = 6;
|
|
repeated int64 shard_sizes = 7; // optimized: sizes for shards in order of set bits in ec_index_bits
|
|
uint64 file_count = 8; // total needles in the .ecx index (live + tombstoned)
|
|
uint64 delete_count = 9; // node-local tombstones in the .ecj deletion journal
|
|
// encode-run identity (unix nanos) from the .vif EcShardConfig; lets the admin
|
|
// group shards by encode generation. Numbered 14 (not 10) to skip the
|
|
// enterprise fork's reserved 10-13.
|
|
int64 encode_ts_ns = 14;
|
|
// fields 15-19 reserved for future upstream open-source additions.
|
|
// fields 20+ are owned by the enterprise fork (e.g. data_shards/parity_shards)
|
|
// and must not be used here without coordination.
|
|
}
|
|
|
|
message StorageBackend {
|
|
string type = 1;
|
|
string id = 2;
|
|
map<string, string> properties = 3;
|
|
}
|
|
|
|
message Empty {
|
|
}
|
|
|
|
message SuperBlockExtra {
|
|
message ErasureCoding {
|
|
uint32 data = 1;
|
|
uint32 parity = 2;
|
|
repeated uint32 volume_ids = 3;
|
|
}
|
|
ErasureCoding erasure_coding = 1;
|
|
}
|
|
|
|
message KeepConnectedRequest {
|
|
string client_type = 1;
|
|
string client_address = 3;
|
|
string version = 4;
|
|
string filer_group = 5;
|
|
string data_center = 6;
|
|
string rack = 7;
|
|
}
|
|
|
|
message VolumeLocation {
|
|
string url = 1;
|
|
string public_url = 2;
|
|
repeated uint32 new_vids = 3;
|
|
repeated uint32 deleted_vids = 4;
|
|
string leader = 5; // optional when leader is not itself
|
|
string data_center = 6; // optional when DataCenter is in use
|
|
uint32 grpc_port = 7;
|
|
repeated uint32 new_ec_vids = 8;
|
|
repeated uint32 deleted_ec_vids = 9;
|
|
}
|
|
|
|
message ClusterNodeUpdate {
|
|
string node_type = 1;
|
|
string address = 2;
|
|
bool is_add = 4;
|
|
string filer_group = 5;
|
|
int64 created_at_ns = 6;
|
|
}
|
|
|
|
message KeepConnectedResponse {
|
|
VolumeLocation volume_location = 1;
|
|
ClusterNodeUpdate cluster_node_update = 2;
|
|
LockRingUpdate lock_ring_update = 3;
|
|
}
|
|
|
|
// LockRingUpdate is sent by the master to all filers when the lock ring
|
|
// membership changes. The master batches rapid changes (e.g., node drop + join)
|
|
// and sends the complete member list atomically, avoiding intermediate ring
|
|
// states that would cause unnecessary lock churn.
|
|
message LockRingUpdate {
|
|
string filer_group = 1;
|
|
repeated string servers = 2;
|
|
int64 version = 3;
|
|
}
|
|
|
|
message LookupVolumeRequest {
|
|
repeated string volume_or_file_ids = 1;
|
|
string collection = 2; // optional, a bit faster if provided.
|
|
}
|
|
message LookupVolumeResponse {
|
|
message VolumeIdLocation {
|
|
string volume_or_file_id = 1;
|
|
repeated Location locations = 2;
|
|
string error = 3;
|
|
string auth = 4;
|
|
}
|
|
repeated VolumeIdLocation volume_id_locations = 1;
|
|
}
|
|
|
|
message Location {
|
|
string url = 1;
|
|
string public_url = 2;
|
|
uint32 grpc_port = 3;
|
|
string data_center = 4;
|
|
}
|
|
|
|
message AssignRequest {
|
|
uint64 count = 1;
|
|
string replication = 2;
|
|
string collection = 3;
|
|
string ttl = 4;
|
|
string data_center = 5;
|
|
string rack = 6;
|
|
string data_node = 7;
|
|
uint32 memory_map_max_size_mb = 8;
|
|
uint32 writable_volume_count = 9;
|
|
string disk_type = 10;
|
|
uint64 expected_data_size = 11; // hint for size-aware volume selection
|
|
}
|
|
|
|
message VolumeGrowRequest {
|
|
uint32 writable_volume_count = 1;
|
|
string replication = 2;
|
|
string collection = 3;
|
|
string ttl = 4;
|
|
string data_center = 5;
|
|
string rack = 6;
|
|
string data_node = 7;
|
|
uint32 memory_map_max_size_mb = 8;
|
|
string disk_type = 9;
|
|
}
|
|
|
|
message AssignResponse {
|
|
string fid = 1;
|
|
uint64 count = 4;
|
|
string error = 5;
|
|
string auth = 6;
|
|
repeated Location replicas = 7;
|
|
Location location = 8;
|
|
}
|
|
|
|
message StatisticsRequest {
|
|
string replication = 1;
|
|
string collection = 2;
|
|
string ttl = 3;
|
|
string disk_type = 4;
|
|
}
|
|
message StatisticsResponse {
|
|
uint64 total_size = 4;
|
|
uint64 used_size = 5;
|
|
uint64 file_count = 6;
|
|
// sizes counting one copy of the data: a single replica of a regular volume,
|
|
// the data shards of an ec volume. logical_total_size scales the free space
|
|
// by the copies the requested replication makes.
|
|
uint64 logical_total_size = 7;
|
|
uint64 logical_used_size = 8;
|
|
}
|
|
|
|
//
|
|
// collection related
|
|
//
|
|
message Collection {
|
|
string name = 1;
|
|
}
|
|
message CollectionListRequest {
|
|
bool include_normal_volumes = 1;
|
|
bool include_ec_volumes = 2;
|
|
}
|
|
message CollectionListResponse {
|
|
repeated Collection collections = 1;
|
|
}
|
|
|
|
// Summarises what each collection holds, so a caller tracking usage does not
|
|
// have to be sent every volume in the cluster to add it up itself.
|
|
message CollectionStatisticsRequest {
|
|
}
|
|
message CollectionStatisticsResponse {
|
|
repeated CollectionStatistics collections = 1;
|
|
}
|
|
message CollectionStatistics {
|
|
string collection = 1;
|
|
uint64 file_count = 2;
|
|
uint64 delete_count = 3;
|
|
uint64 deleted_byte_count = 4;
|
|
// one copy of the data: a single replica of a regular volume, the data
|
|
// shards of an ec volume
|
|
uint64 size = 5;
|
|
// what is on disk: every replica, and parity shards
|
|
uint64 physical_size = 6;
|
|
uint64 volume_count = 7;
|
|
}
|
|
|
|
message CollectionDeleteRequest {
|
|
string name = 1;
|
|
}
|
|
message CollectionDeleteResponse {
|
|
}
|
|
|
|
//
|
|
// volume related
|
|
//
|
|
message DiskInfo {
|
|
string type = 1;
|
|
int64 volume_count = 2;
|
|
int64 max_volume_count = 3;
|
|
int64 free_volume_count = 4;
|
|
int64 active_volume_count = 5;
|
|
repeated VolumeInformationMessage volume_infos = 6;
|
|
repeated VolumeEcShardInformationMessage ec_shard_infos = 7;
|
|
int64 remote_volume_count = 8;
|
|
// On a per-physical-disk DiskInfo (from SplitByPhysicalDisk) this is the disk's
|
|
// identity; on the type-keyed aggregate it is only a representative fallback
|
|
// (the first volume's disk id).
|
|
uint32 disk_id = 9;
|
|
repeated string tags = 10;
|
|
// Max volume count for every physical disk of this type, keyed by disk id,
|
|
// including disks with no volumes or EC shards; recovers empty disks that
|
|
// carry no per-volume/per-shard records.
|
|
map<uint32, int64> max_volume_count_by_disk = 11;
|
|
// physical disk capacity in bytes, from the underlying filesystem (0 if unknown)
|
|
uint64 disk_total_bytes = 12;
|
|
uint64 disk_free_bytes = 13;
|
|
}
|
|
message DataNodeInfo {
|
|
string id = 1;
|
|
map<string, DiskInfo> diskInfos = 2;
|
|
uint32 grpc_port = 3;
|
|
string address = 4; // ip:port for connecting to the volume server
|
|
}
|
|
message RackInfo {
|
|
string id = 1;
|
|
repeated DataNodeInfo data_node_infos = 2;
|
|
map<string, DiskInfo> diskInfos = 3;
|
|
}
|
|
message DataCenterInfo {
|
|
string id = 1;
|
|
repeated RackInfo rack_infos = 2;
|
|
map<string, DiskInfo> diskInfos = 3;
|
|
}
|
|
message TopologyInfo {
|
|
string id = 1;
|
|
repeated DataCenterInfo data_center_infos = 2;
|
|
map<string, DiskInfo> diskInfos = 3;
|
|
}
|
|
message VolumeListRequest {
|
|
// Empty and zero take everything. Only the volumes and ec shards listed
|
|
// under a disk are selected; the topology and its disk counters are always
|
|
// reported in full.
|
|
string collection = 1;
|
|
uint32 volume_id = 2;
|
|
// The one collection the empty string cannot name. A named collection wins.
|
|
bool default_collection_only = 3;
|
|
}
|
|
message VolumeListResponse {
|
|
TopologyInfo topology_info = 1;
|
|
uint64 volume_size_limit_mb = 2;
|
|
}
|
|
|
|
// VolumeListStream answers the same request as VolumeList without either end
|
|
// holding every volume in the cluster at once. At 800k volumes the reply is
|
|
// 36MB on the wire but 305MB as messages, which the master built in full
|
|
// before sending any of it.
|
|
message VolumeListStreamResponse {
|
|
// Sent once, first, listing no volumes: the topology, its disks and their
|
|
// counters. Every message after carries volumes for one of those disks.
|
|
VolumeListResponse header = 1;
|
|
// Which disk this batch is from. A disk arrives over as many batches as it
|
|
// takes, so append rather than assign.
|
|
string data_center = 2;
|
|
string rack = 3;
|
|
string data_node = 4;
|
|
string disk_type = 5;
|
|
repeated VolumeInformationMessage volume_infos = 6;
|
|
repeated VolumeEcShardInformationMessage ec_shard_infos = 7;
|
|
}
|
|
|
|
message LookupEcVolumeRequest {
|
|
uint32 volume_id = 1;
|
|
}
|
|
message LookupEcVolumeResponse {
|
|
uint32 volume_id = 1;
|
|
message EcShardIdLocation {
|
|
uint32 shard_id = 1;
|
|
repeated Location locations = 2;
|
|
}
|
|
repeated EcShardIdLocation shard_id_locations = 2;
|
|
}
|
|
|
|
message VacuumVolumeRequest {
|
|
float garbage_threshold = 1;
|
|
uint32 volume_id = 2;
|
|
string collection = 3;
|
|
}
|
|
message VacuumVolumeResponse {
|
|
}
|
|
|
|
message DisableVacuumRequest {
|
|
bool by_plugin = 1;
|
|
}
|
|
message DisableVacuumResponse {
|
|
}
|
|
|
|
message EnableVacuumRequest {
|
|
bool by_plugin = 1;
|
|
}
|
|
message EnableVacuumResponse {
|
|
}
|
|
|
|
message VolumeMarkReadonlyRequest {
|
|
string ip = 1;
|
|
uint32 port = 2;
|
|
uint32 volume_id = 4;
|
|
string collection = 5;
|
|
uint32 replica_placement = 6;
|
|
uint32 version = 7;
|
|
uint32 ttl = 8;
|
|
string disk_type = 9;
|
|
bool is_readonly = 10;
|
|
}
|
|
message VolumeMarkReadonlyResponse {
|
|
}
|
|
|
|
message GetMasterConfigurationRequest {
|
|
}
|
|
message GetMasterConfigurationResponse {
|
|
string metrics_address = 1;
|
|
uint32 metrics_interval_seconds = 2;
|
|
repeated StorageBackend storage_backends = 3;
|
|
string default_replication = 4;
|
|
string leader = 5;
|
|
uint32 volume_size_limit_m_b = 6;
|
|
bool volume_preallocate = 7;
|
|
// MIGRATION: fields 8-9 help migrate master.toml [master.maintenance] to admin script plugin. Remove after March 2027.
|
|
string maintenance_scripts = 8;
|
|
uint32 maintenance_sleep_minutes = 9;
|
|
}
|
|
|
|
message ListClusterNodesRequest {
|
|
string client_type = 1;
|
|
string filer_group = 2;
|
|
int32 limit = 4;
|
|
}
|
|
message ListClusterNodesResponse {
|
|
message ClusterNode {
|
|
string address = 1;
|
|
string version = 2;
|
|
int64 created_at_ns = 4;
|
|
string data_center = 5;
|
|
string rack = 6;
|
|
}
|
|
repeated ClusterNode cluster_nodes = 1;
|
|
}
|
|
|
|
message LeaseAdminTokenRequest {
|
|
int64 previous_token = 1;
|
|
int64 previous_lock_time = 2;
|
|
string lock_name = 3;
|
|
string client_name = 4;
|
|
string message = 5;
|
|
}
|
|
message LeaseAdminTokenResponse {
|
|
int64 token = 1;
|
|
int64 lock_ts_ns = 2;
|
|
}
|
|
|
|
message ReleaseAdminTokenRequest {
|
|
int64 previous_token = 1;
|
|
int64 previous_lock_time = 2;
|
|
string lock_name = 3;
|
|
}
|
|
message ReleaseAdminTokenResponse {
|
|
}
|
|
|
|
message GetAdminLockStatusRequest {
|
|
string lock_name = 1;
|
|
}
|
|
message GetAdminLockStatusResponse {
|
|
bool is_locked = 1;
|
|
string client_name = 2;
|
|
string message = 3;
|
|
}
|
|
|
|
message PingRequest {
|
|
string target = 1; // default to ping itself
|
|
string target_type = 2;
|
|
}
|
|
message PingResponse {
|
|
int64 start_time_ns = 1;
|
|
int64 remote_time_ns = 2;
|
|
int64 stop_time_ns = 3;
|
|
}
|
|
|
|
message RaftAddServerRequest {
|
|
string id = 1;
|
|
string address = 2;
|
|
bool voter = 3;
|
|
}
|
|
message RaftAddServerResponse {
|
|
}
|
|
|
|
message RaftRemoveServerRequest {
|
|
string id = 1;
|
|
bool force = 2;
|
|
}
|
|
message RaftRemoveServerResponse {
|
|
}
|
|
|
|
message RaftListClusterServersRequest {
|
|
}
|
|
message RaftListClusterServersResponse {
|
|
message ClusterServers {
|
|
string id = 1;
|
|
string address = 2;
|
|
string suffrage = 3;
|
|
bool isLeader = 4;
|
|
}
|
|
repeated ClusterServers cluster_servers = 1;
|
|
}
|
|
|
|
message RaftLeadershipTransferRequest {
|
|
string target_id = 1; // Optional: target server ID. If empty, transfers to any eligible follower
|
|
string target_address = 2; // Optional: target server address. Required if target_id is specified
|
|
}
|
|
message RaftLeadershipTransferResponse {
|
|
string previous_leader = 1;
|
|
string new_leader = 2;
|
|
}
|
|
|
|
message VolumeGrowResponse {
|
|
}
|