You can not select more than 25 topics Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.

420 lines
14 KiB

3 years ago
3 years ago
6 years ago
5 years ago
6 years ago
6 years ago
3 years ago
3 years ago
3 years ago
3 years ago
4 years ago
8 months ago
6 years ago
  1. package weed_server
  2. import (
  3. "context"
  4. "errors"
  5. "fmt"
  6. "github.com/seaweedfs/seaweedfs/weed/cluster"
  7. "net"
  8. "sort"
  9. "time"
  10. "github.com/seaweedfs/seaweedfs/weed/pb"
  11. "github.com/seaweedfs/seaweedfs/weed/stats"
  12. "github.com/seaweedfs/seaweedfs/weed/storage/backend"
  13. "github.com/seaweedfs/seaweedfs/weed/util"
  14. "github.com/seaweedfs/raft"
  15. "google.golang.org/grpc/peer"
  16. "github.com/seaweedfs/seaweedfs/weed/glog"
  17. "github.com/seaweedfs/seaweedfs/weed/pb/master_pb"
  18. "github.com/seaweedfs/seaweedfs/weed/storage/needle"
  19. "github.com/seaweedfs/seaweedfs/weed/topology"
  20. )
  21. func (ms *MasterServer) RegisterUuids(heartbeat *master_pb.Heartbeat) (duplicated_uuids []string, err error) {
  22. ms.Topo.UuidAccessLock.Lock()
  23. defer ms.Topo.UuidAccessLock.Unlock()
  24. key := fmt.Sprintf("%s:%d", heartbeat.Ip, heartbeat.Port)
  25. if ms.Topo.UuidMap == nil {
  26. ms.Topo.UuidMap = make(map[string][]string)
  27. }
  28. // find whether new uuid exists
  29. for k, v := range ms.Topo.UuidMap {
  30. sort.Strings(v)
  31. for _, id := range heartbeat.LocationUuids {
  32. index := sort.SearchStrings(v, id)
  33. if index < len(v) && v[index] == id {
  34. duplicated_uuids = append(duplicated_uuids, id)
  35. glog.Errorf("directory of %s on %s has been loaded", id, k)
  36. }
  37. }
  38. }
  39. if len(duplicated_uuids) > 0 {
  40. return duplicated_uuids, errors.New("volume: Duplicated volume directories were loaded")
  41. }
  42. ms.Topo.UuidMap[key] = heartbeat.LocationUuids
  43. glog.V(0).Infof("found new uuid:%v %v , %v", key, heartbeat.LocationUuids, ms.Topo.UuidMap)
  44. return nil, nil
  45. }
  46. func (ms *MasterServer) UnRegisterUuids(ip string, port int) {
  47. ms.Topo.UuidAccessLock.Lock()
  48. defer ms.Topo.UuidAccessLock.Unlock()
  49. key := fmt.Sprintf("%s:%d", ip, port)
  50. delete(ms.Topo.UuidMap, key)
  51. glog.V(0).Infof("remove volume server %v, online volume server: %v", key, ms.Topo.UuidMap)
  52. }
  53. func (ms *MasterServer) SendHeartbeat(stream master_pb.Seaweed_SendHeartbeatServer) error {
  54. var dn *topology.DataNode
  55. defer func() {
  56. if dn != nil {
  57. dn.Counter--
  58. if dn.Counter > 0 {
  59. glog.V(0).Infof("disconnect phantom volume server %s:%d remaining %d", dn.Ip, dn.Port, dn.Counter)
  60. return
  61. }
  62. message := &master_pb.VolumeLocation{
  63. DataCenter: dn.GetDataCenterId(),
  64. Url: dn.Url(),
  65. PublicUrl: dn.PublicUrl,
  66. GrpcPort: uint32(dn.GrpcPort),
  67. }
  68. for _, v := range dn.GetVolumes() {
  69. message.DeletedVids = append(message.DeletedVids, uint32(v.Id))
  70. }
  71. for _, s := range dn.GetEcShards() {
  72. message.DeletedVids = append(message.DeletedVids, uint32(s.VolumeId))
  73. }
  74. // if the volume server disconnects and reconnects quickly
  75. // the unregister and register can race with each other
  76. ms.Topo.UnRegisterDataNode(dn)
  77. glog.V(0).Infof("unregister disconnected volume server %s:%d", dn.Ip, dn.Port)
  78. ms.UnRegisterUuids(dn.Ip, dn.Port)
  79. if len(message.DeletedVids) > 0 {
  80. ms.broadcastToClients(&master_pb.KeepConnectedResponse{VolumeLocation: message})
  81. }
  82. }
  83. }()
  84. for {
  85. heartbeat, err := stream.Recv()
  86. if err != nil {
  87. if dn != nil {
  88. glog.Warningf("SendHeartbeat.Recv server %s:%d : %v", dn.Ip, dn.Port, err)
  89. } else {
  90. glog.Warningf("SendHeartbeat.Recv: %v", err)
  91. }
  92. stats.MasterReceivedHeartbeatCounter.WithLabelValues("error").Inc()
  93. return err
  94. }
  95. if !ms.Topo.IsLeader() {
  96. // tell the volume servers about the leader
  97. newLeader, err := ms.Topo.Leader()
  98. if err != nil {
  99. glog.Warningf("SendHeartbeat find leader: %v", err)
  100. return err
  101. }
  102. if err := stream.Send(&master_pb.HeartbeatResponse{
  103. Leader: string(newLeader),
  104. }); err != nil {
  105. if dn != nil {
  106. glog.Warningf("SendHeartbeat.Send response to %s:%d %v", dn.Ip, dn.Port, err)
  107. } else {
  108. glog.Warningf("SendHeartbeat.Send response %v", err)
  109. }
  110. return err
  111. }
  112. continue
  113. }
  114. ms.Topo.Sequence.SetMax(heartbeat.MaxFileKey)
  115. if dn == nil {
  116. // Skip delta heartbeat for volume server versions better than 3.28 https://github.com/seaweedfs/seaweedfs/pull/3630
  117. if heartbeat.Ip == "" {
  118. continue
  119. } // ToDo must be removed after update major version
  120. dcName, rackName := ms.Topo.Configuration.Locate(heartbeat.Ip, heartbeat.DataCenter, heartbeat.Rack)
  121. dc := ms.Topo.GetOrCreateDataCenter(dcName)
  122. rack := dc.GetOrCreateRack(rackName)
  123. dn = rack.GetOrCreateDataNode(heartbeat.Ip, int(heartbeat.Port), int(heartbeat.GrpcPort), heartbeat.PublicUrl, heartbeat.MaxVolumeCounts)
  124. glog.V(0).Infof("added volume server %d: %v:%d %v", dn.Counter, heartbeat.GetIp(), heartbeat.GetPort(), heartbeat.LocationUuids)
  125. uuidlist, err := ms.RegisterUuids(heartbeat)
  126. if err != nil {
  127. if stream_err := stream.Send(&master_pb.HeartbeatResponse{
  128. DuplicatedUuids: uuidlist,
  129. }); stream_err != nil {
  130. glog.Warningf("SendHeartbeat.Send DuplicatedDirectory response to %s:%d %v", dn.Ip, dn.Port, stream_err)
  131. return stream_err
  132. }
  133. return err
  134. }
  135. if err := stream.Send(&master_pb.HeartbeatResponse{
  136. VolumeSizeLimit: uint64(ms.option.VolumeSizeLimitMB) * 1024 * 1024,
  137. }); err != nil {
  138. glog.Warningf("SendHeartbeat.Send volume size to %s:%d %v", dn.Ip, dn.Port, err)
  139. return err
  140. }
  141. stats.MasterReceivedHeartbeatCounter.WithLabelValues("dataNode").Inc()
  142. dn.Counter++
  143. }
  144. dn.AdjustMaxVolumeCounts(heartbeat.MaxVolumeCounts)
  145. glog.V(4).Infof("master received heartbeat %s", heartbeat.String())
  146. stats.MasterReceivedHeartbeatCounter.WithLabelValues("total").Inc()
  147. message := &master_pb.VolumeLocation{
  148. Url: dn.Url(),
  149. PublicUrl: dn.PublicUrl,
  150. DataCenter: dn.GetDataCenterId(),
  151. GrpcPort: uint32(dn.GrpcPort),
  152. }
  153. if len(heartbeat.NewVolumes) > 0 {
  154. stats.MasterReceivedHeartbeatCounter.WithLabelValues("newVolumes").Inc()
  155. }
  156. if len(heartbeat.DeletedVolumes) > 0 {
  157. stats.MasterReceivedHeartbeatCounter.WithLabelValues("deletedVolumes").Inc()
  158. }
  159. if len(heartbeat.NewVolumes) > 0 || len(heartbeat.DeletedVolumes) > 0 {
  160. // process delta volume ids if exists for fast volume id updates
  161. for _, volInfo := range heartbeat.NewVolumes {
  162. message.NewVids = append(message.NewVids, volInfo.Id)
  163. }
  164. for _, volInfo := range heartbeat.DeletedVolumes {
  165. message.DeletedVids = append(message.DeletedVids, volInfo.Id)
  166. }
  167. // update master internal volume layouts
  168. ms.Topo.IncrementalSyncDataNodeRegistration(heartbeat.NewVolumes, heartbeat.DeletedVolumes, dn)
  169. }
  170. if len(heartbeat.Volumes) > 0 || heartbeat.HasNoVolumes {
  171. if heartbeat.Ip != "" {
  172. dcName, rackName := ms.Topo.Configuration.Locate(heartbeat.Ip, heartbeat.DataCenter, heartbeat.Rack)
  173. ms.Topo.DataNodeRegistration(dcName, rackName, dn)
  174. }
  175. // process heartbeat.Volumes
  176. stats.MasterReceivedHeartbeatCounter.WithLabelValues("Volumes").Inc()
  177. newVolumes, deletedVolumes := ms.Topo.SyncDataNodeRegistration(heartbeat.Volumes, dn)
  178. for _, v := range newVolumes {
  179. glog.V(0).Infof("master see new volume %d from %s", uint32(v.Id), dn.Url())
  180. message.NewVids = append(message.NewVids, uint32(v.Id))
  181. }
  182. for _, v := range deletedVolumes {
  183. glog.V(0).Infof("master see deleted volume %d from %s", uint32(v.Id), dn.Url())
  184. message.DeletedVids = append(message.DeletedVids, uint32(v.Id))
  185. }
  186. }
  187. if len(heartbeat.NewEcShards) > 0 || len(heartbeat.DeletedEcShards) > 0 {
  188. stats.MasterReceivedHeartbeatCounter.WithLabelValues("newEcShards").Inc()
  189. // update master internal volume layouts
  190. ms.Topo.IncrementalSyncDataNodeEcShards(heartbeat.NewEcShards, heartbeat.DeletedEcShards, dn)
  191. for _, s := range heartbeat.NewEcShards {
  192. message.NewEcVids = append(message.NewEcVids, s.Id)
  193. }
  194. for _, s := range heartbeat.DeletedEcShards {
  195. if dn.HasEcShards(needle.VolumeId(s.Id)) {
  196. continue
  197. }
  198. message.DeletedEcVids = append(message.DeletedEcVids, s.Id)
  199. }
  200. }
  201. if len(heartbeat.EcShards) > 0 || heartbeat.HasNoEcShards {
  202. stats.MasterReceivedHeartbeatCounter.WithLabelValues("ecShards").Inc()
  203. glog.V(4).Infof("master received ec shards from %s: %+v", dn.Url(), heartbeat.EcShards)
  204. newShards, deletedShards := ms.Topo.SyncDataNodeEcShards(heartbeat.EcShards, dn)
  205. // broadcast the ec vid changes to master clients
  206. for _, s := range newShards {
  207. message.NewEcVids = append(message.NewEcVids, uint32(s.VolumeId))
  208. }
  209. for _, s := range deletedShards {
  210. if dn.HasVolumesById(s.VolumeId) {
  211. continue
  212. }
  213. message.DeletedEcVids = append(message.DeletedEcVids, uint32(s.VolumeId))
  214. }
  215. }
  216. if len(message.NewVids) > 0 || len(message.DeletedVids) > 0 || len(message.NewEcVids) > 0 || len(message.DeletedEcVids) > 0 {
  217. ms.broadcastToClients(&master_pb.KeepConnectedResponse{VolumeLocation: message})
  218. }
  219. }
  220. }
  221. // KeepConnected keep a stream gRPC call to the master. Used by clients to know the master is up.
  222. // And clients gets the up-to-date list of volume locations
  223. func (ms *MasterServer) KeepConnected(stream master_pb.Seaweed_KeepConnectedServer) error {
  224. req, recvErr := stream.Recv()
  225. if recvErr != nil {
  226. return recvErr
  227. }
  228. if !ms.Topo.IsLeader() {
  229. return ms.informNewLeader(stream)
  230. }
  231. peerAddress := pb.ServerAddress(req.ClientAddress)
  232. // buffer by 1 so we don't end up getting stuck writing to stopChan forever
  233. stopChan := make(chan bool, 1)
  234. clientName, messageChan := ms.addClient(req.FilerGroup, req.ClientType, peerAddress)
  235. for _, update := range ms.Cluster.AddClusterNode(req.FilerGroup, req.ClientType, cluster.DataCenter(req.DataCenter), cluster.Rack(req.Rack), peerAddress, req.Version) {
  236. ms.broadcastToClients(update)
  237. }
  238. defer func() {
  239. for _, update := range ms.Cluster.RemoveClusterNode(req.FilerGroup, req.ClientType, peerAddress) {
  240. ms.broadcastToClients(update)
  241. }
  242. ms.deleteClient(clientName)
  243. }()
  244. for i, message := range ms.Topo.ToVolumeLocations() {
  245. if i == 0 {
  246. if leader, err := ms.Topo.Leader(); err == nil {
  247. message.Leader = string(leader)
  248. }
  249. }
  250. if sendErr := stream.Send(&master_pb.KeepConnectedResponse{VolumeLocation: message}); sendErr != nil {
  251. return sendErr
  252. }
  253. }
  254. go func() {
  255. for {
  256. _, err := stream.Recv()
  257. if err != nil {
  258. glog.V(2).Infof("- client %v: %v", clientName, err)
  259. go func() {
  260. // consume message chan to avoid deadlock, go routine exit when message chan is closed
  261. for range messageChan {
  262. // no op
  263. }
  264. }()
  265. close(stopChan)
  266. return
  267. }
  268. }
  269. }()
  270. ticker := time.NewTicker(5 * time.Second)
  271. defer ticker.Stop()
  272. for {
  273. select {
  274. case message := <-messageChan:
  275. if err := stream.Send(message); err != nil {
  276. glog.V(0).Infof("=> client %v: %+v", clientName, message)
  277. return err
  278. }
  279. case <-ticker.C:
  280. if !ms.Topo.IsLeader() {
  281. stats.MasterRaftIsleader.Set(0)
  282. stats.MasterAdminLock.Reset()
  283. stats.MasterReplicaPlacementMismatch.Reset()
  284. return ms.informNewLeader(stream)
  285. } else {
  286. stats.MasterRaftIsleader.Set(1)
  287. }
  288. case <-stopChan:
  289. return nil
  290. }
  291. }
  292. }
  293. func (ms *MasterServer) broadcastToClients(message *master_pb.KeepConnectedResponse) {
  294. ms.clientChansLock.RLock()
  295. for _, ch := range ms.clientChans {
  296. ch <- message
  297. }
  298. ms.clientChansLock.RUnlock()
  299. }
  300. func (ms *MasterServer) informNewLeader(stream master_pb.Seaweed_KeepConnectedServer) error {
  301. leader, err := ms.Topo.Leader()
  302. if err != nil {
  303. glog.Errorf("topo leader: %v", err)
  304. return raft.NotLeaderError
  305. }
  306. if err := stream.Send(&master_pb.KeepConnectedResponse{
  307. VolumeLocation: &master_pb.VolumeLocation{
  308. Leader: string(leader),
  309. },
  310. }); err != nil {
  311. return err
  312. }
  313. return nil
  314. }
  315. func (ms *MasterServer) addClient(filerGroup, clientType string, clientAddress pb.ServerAddress) (clientName string, messageChan chan *master_pb.KeepConnectedResponse) {
  316. clientName = filerGroup + "." + clientType + "@" + string(clientAddress)
  317. glog.V(0).Infof("+ client %v", clientName)
  318. // we buffer this because otherwise we end up in a potential deadlock where
  319. // the KeepConnected loop is no longer listening on this channel but we're
  320. // trying to send to it in SendHeartbeat and so we can't lock the
  321. // clientChansLock to remove the channel and we're stuck writing to it
  322. // 100 is probably overkill
  323. messageChan = make(chan *master_pb.KeepConnectedResponse, 100)
  324. ms.clientChansLock.Lock()
  325. ms.clientChans[clientName] = messageChan
  326. ms.clientChansLock.Unlock()
  327. return
  328. }
  329. func (ms *MasterServer) deleteClient(clientName string) {
  330. glog.V(0).Infof("- client %v", clientName)
  331. ms.clientChansLock.Lock()
  332. // close message chan, so that the KeepConnected go routine can exit
  333. close(ms.clientChans[clientName])
  334. delete(ms.clientChans, clientName)
  335. ms.clientChansLock.Unlock()
  336. }
  337. func findClientAddress(ctx context.Context, grpcPort uint32) string {
  338. // fmt.Printf("FromContext %+v\n", ctx)
  339. pr, ok := peer.FromContext(ctx)
  340. if !ok {
  341. glog.Error("failed to get peer from ctx")
  342. return ""
  343. }
  344. if pr.Addr == net.Addr(nil) {
  345. glog.Error("failed to get peer address")
  346. return ""
  347. }
  348. if grpcPort == 0 {
  349. return pr.Addr.String()
  350. }
  351. if tcpAddr, ok := pr.Addr.(*net.TCPAddr); ok {
  352. externalIP := tcpAddr.IP
  353. return util.JoinHostPort(externalIP.String(), int(grpcPort))
  354. }
  355. return pr.Addr.String()
  356. }
  357. func (ms *MasterServer) GetMasterConfiguration(ctx context.Context, req *master_pb.GetMasterConfigurationRequest) (*master_pb.GetMasterConfigurationResponse, error) {
  358. // tell the volume servers about the leader
  359. leader, _ := ms.Topo.Leader()
  360. resp := &master_pb.GetMasterConfigurationResponse{
  361. MetricsAddress: ms.option.MetricsAddress,
  362. MetricsIntervalSeconds: uint32(ms.option.MetricsIntervalSec),
  363. StorageBackends: backend.ToPbStorageBackends(),
  364. DefaultReplication: ms.option.DefaultReplicaPlacement,
  365. VolumeSizeLimitMB: uint32(ms.option.VolumeSizeLimitMB),
  366. VolumePreallocate: ms.option.VolumePreallocate,
  367. Leader: string(leader),
  368. }
  369. return resp, nil
  370. }