You can not select more than 25 topics Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.

341 lines
11 KiB

7 years ago
3 years ago
3 years ago
4 years ago
6 years ago
5 years ago
6 years ago
6 years ago
6 years ago
3 years ago
6 years ago
4 years ago
6 years ago
6 years ago
6 years ago
6 years ago
4 years ago
6 years ago
6 years ago
3 years ago
3 years ago
4 years ago
5 years ago
  1. package weed_server
  2. import (
  3. "context"
  4. "github.com/chrislusf/seaweedfs/weed/pb"
  5. "github.com/chrislusf/seaweedfs/weed/stats"
  6. "github.com/chrislusf/seaweedfs/weed/storage/backend"
  7. "github.com/chrislusf/seaweedfs/weed/util"
  8. "net"
  9. "time"
  10. "github.com/chrislusf/raft"
  11. "google.golang.org/grpc/peer"
  12. "github.com/chrislusf/seaweedfs/weed/glog"
  13. "github.com/chrislusf/seaweedfs/weed/pb/master_pb"
  14. "github.com/chrislusf/seaweedfs/weed/storage/needle"
  15. "github.com/chrislusf/seaweedfs/weed/topology"
  16. )
  17. func (ms *MasterServer) SendHeartbeat(stream master_pb.Seaweed_SendHeartbeatServer) error {
  18. var dn *topology.DataNode
  19. defer func() {
  20. if dn != nil {
  21. dn.Counter--
  22. if dn.Counter > 0 {
  23. glog.V(0).Infof("disconnect phantom volume server %s:%d remaining %d", dn.Ip, dn.Port, dn.Counter)
  24. return
  25. }
  26. // if the volume server disconnects and reconnects quickly
  27. // the unregister and register can race with each other
  28. ms.Topo.UnRegisterDataNode(dn)
  29. glog.V(0).Infof("unregister disconnected volume server %s:%d", dn.Ip, dn.Port)
  30. message := &master_pb.VolumeLocation{
  31. Url: dn.Url(),
  32. PublicUrl: dn.PublicUrl,
  33. }
  34. for _, v := range dn.GetVolumes() {
  35. message.DeletedVids = append(message.DeletedVids, uint32(v.Id))
  36. }
  37. for _, s := range dn.GetEcShards() {
  38. message.DeletedVids = append(message.DeletedVids, uint32(s.VolumeId))
  39. }
  40. if len(message.DeletedVids) > 0 {
  41. ms.broadcastToClients(&master_pb.KeepConnectedResponse{VolumeLocation: message})
  42. }
  43. }
  44. }()
  45. for {
  46. heartbeat, err := stream.Recv()
  47. if err != nil {
  48. if dn != nil {
  49. glog.Warningf("SendHeartbeat.Recv server %s:%d : %v", dn.Ip, dn.Port, err)
  50. } else {
  51. glog.Warningf("SendHeartbeat.Recv: %v", err)
  52. }
  53. stats.MasterReceivedHeartbeatCounter.WithLabelValues("error").Inc()
  54. return err
  55. }
  56. ms.Topo.Sequence.SetMax(heartbeat.MaxFileKey)
  57. if dn == nil {
  58. dcName, rackName := ms.Topo.Configuration.Locate(heartbeat.Ip, heartbeat.DataCenter, heartbeat.Rack)
  59. dc := ms.Topo.GetOrCreateDataCenter(dcName)
  60. rack := dc.GetOrCreateRack(rackName)
  61. dn = rack.GetOrCreateDataNode(heartbeat.Ip, int(heartbeat.Port), int(heartbeat.GrpcPort), heartbeat.PublicUrl, heartbeat.MaxVolumeCounts)
  62. glog.V(0).Infof("added volume server %d: %v:%d", dn.Counter, heartbeat.GetIp(), heartbeat.GetPort())
  63. if err := stream.Send(&master_pb.HeartbeatResponse{
  64. VolumeSizeLimit: uint64(ms.option.VolumeSizeLimitMB) * 1024 * 1024,
  65. }); err != nil {
  66. glog.Warningf("SendHeartbeat.Send volume size to %s:%d %v", dn.Ip, dn.Port, err)
  67. return err
  68. }
  69. stats.MasterReceivedHeartbeatCounter.WithLabelValues("dataNode").Inc()
  70. dn.Counter++
  71. }
  72. dn.AdjustMaxVolumeCounts(heartbeat.MaxVolumeCounts)
  73. glog.V(4).Infof("master received heartbeat %s", heartbeat.String())
  74. stats.MasterReceivedHeartbeatCounter.WithLabelValues("total").Inc()
  75. var dataCenter string
  76. if dc := dn.GetDataCenter(); dc != nil {
  77. dataCenter = string(dc.Id())
  78. }
  79. message := &master_pb.VolumeLocation{
  80. Url: dn.Url(),
  81. PublicUrl: dn.PublicUrl,
  82. DataCenter: dataCenter,
  83. }
  84. if len(heartbeat.NewVolumes) > 0 {
  85. stats.FilerRequestCounter.WithLabelValues("newVolumes").Inc()
  86. }
  87. if len(heartbeat.DeletedVolumes) > 0 {
  88. stats.FilerRequestCounter.WithLabelValues("deletedVolumes").Inc()
  89. }
  90. if len(heartbeat.NewVolumes) > 0 || len(heartbeat.DeletedVolumes) > 0 {
  91. // process delta volume ids if exists for fast volume id updates
  92. for _, volInfo := range heartbeat.NewVolumes {
  93. message.NewVids = append(message.NewVids, volInfo.Id)
  94. }
  95. for _, volInfo := range heartbeat.DeletedVolumes {
  96. message.DeletedVids = append(message.DeletedVids, volInfo.Id)
  97. }
  98. // update master internal volume layouts
  99. ms.Topo.IncrementalSyncDataNodeRegistration(heartbeat.NewVolumes, heartbeat.DeletedVolumes, dn)
  100. }
  101. if len(heartbeat.Volumes) > 0 || heartbeat.HasNoVolumes {
  102. dcName, rackName := ms.Topo.Configuration.Locate(heartbeat.Ip, heartbeat.DataCenter, heartbeat.Rack)
  103. ms.Topo.DataNodeRegistration(dcName, rackName, dn)
  104. // process heartbeat.Volumes
  105. stats.MasterReceivedHeartbeatCounter.WithLabelValues("Volumes").Inc()
  106. newVolumes, deletedVolumes := ms.Topo.SyncDataNodeRegistration(heartbeat.Volumes, dn)
  107. for _, v := range newVolumes {
  108. glog.V(0).Infof("master see new volume %d from %s", uint32(v.Id), dn.Url())
  109. message.NewVids = append(message.NewVids, uint32(v.Id))
  110. }
  111. for _, v := range deletedVolumes {
  112. glog.V(0).Infof("master see deleted volume %d from %s", uint32(v.Id), dn.Url())
  113. message.DeletedVids = append(message.DeletedVids, uint32(v.Id))
  114. }
  115. }
  116. if len(heartbeat.NewEcShards) > 0 || len(heartbeat.DeletedEcShards) > 0 {
  117. stats.MasterReceivedHeartbeatCounter.WithLabelValues("newEcShards").Inc()
  118. // update master internal volume layouts
  119. ms.Topo.IncrementalSyncDataNodeEcShards(heartbeat.NewEcShards, heartbeat.DeletedEcShards, dn)
  120. for _, s := range heartbeat.NewEcShards {
  121. message.NewEcVids = append(message.NewEcVids, s.Id)
  122. }
  123. for _, s := range heartbeat.DeletedEcShards {
  124. if dn.HasEcShards(needle.VolumeId(s.Id)) {
  125. continue
  126. }
  127. message.DeletedEcVids = append(message.DeletedEcVids, s.Id)
  128. }
  129. }
  130. if len(heartbeat.EcShards) > 0 || heartbeat.HasNoEcShards {
  131. stats.MasterReceivedHeartbeatCounter.WithLabelValues("ecShards").Inc()
  132. glog.V(4).Infof("master received ec shards from %s: %+v", dn.Url(), heartbeat.EcShards)
  133. newShards, deletedShards := ms.Topo.SyncDataNodeEcShards(heartbeat.EcShards, dn)
  134. // broadcast the ec vid changes to master clients
  135. for _, s := range newShards {
  136. message.NewEcVids = append(message.NewEcVids, uint32(s.VolumeId))
  137. }
  138. for _, s := range deletedShards {
  139. if dn.HasVolumesById(s.VolumeId) {
  140. continue
  141. }
  142. message.DeletedEcVids = append(message.DeletedEcVids, uint32(s.VolumeId))
  143. }
  144. }
  145. if len(message.NewVids) > 0 || len(message.DeletedVids) > 0 || len(message.NewEcVids) > 0 || len(message.DeletedEcVids) > 0 {
  146. ms.broadcastToClients(&master_pb.KeepConnectedResponse{VolumeLocation: message})
  147. }
  148. // tell the volume servers about the leader
  149. newLeader, err := ms.Topo.Leader()
  150. if err != nil {
  151. glog.Warningf("SendHeartbeat find leader: %v", err)
  152. return err
  153. }
  154. if err := stream.Send(&master_pb.HeartbeatResponse{
  155. Leader: string(newLeader),
  156. }); err != nil {
  157. glog.Warningf("SendHeartbeat.Send response to to %s:%d %v", dn.Ip, dn.Port, err)
  158. return err
  159. }
  160. }
  161. }
  162. // KeepConnected keep a stream gRPC call to the master. Used by clients to know the master is up.
  163. // And clients gets the up-to-date list of volume locations
  164. func (ms *MasterServer) KeepConnected(stream master_pb.Seaweed_KeepConnectedServer) error {
  165. req, recvErr := stream.Recv()
  166. if recvErr != nil {
  167. return recvErr
  168. }
  169. if !ms.Topo.IsLeader() {
  170. return ms.informNewLeader(stream)
  171. }
  172. peerAddress := pb.ServerAddress(req.ClientAddress)
  173. // buffer by 1 so we don't end up getting stuck writing to stopChan forever
  174. stopChan := make(chan bool, 1)
  175. clientName, messageChan := ms.addClient(req.FilerGroup, req.ClientType, peerAddress)
  176. for _, update := range ms.Cluster.AddClusterNode(req.FilerGroup, req.ClientType, peerAddress, req.Version) {
  177. ms.broadcastToClients(update)
  178. }
  179. defer func() {
  180. for _, update := range ms.Cluster.RemoveClusterNode(req.FilerGroup, req.ClientType, peerAddress) {
  181. ms.broadcastToClients(update)
  182. }
  183. ms.deleteClient(clientName)
  184. }()
  185. for _, message := range ms.Topo.ToVolumeLocations() {
  186. if sendErr := stream.Send(&master_pb.KeepConnectedResponse{VolumeLocation: message}); sendErr != nil {
  187. return sendErr
  188. }
  189. }
  190. go func() {
  191. for {
  192. _, err := stream.Recv()
  193. if err != nil {
  194. glog.V(2).Infof("- client %v: %v", clientName, err)
  195. close(stopChan)
  196. return
  197. }
  198. }
  199. }()
  200. ticker := time.NewTicker(5 * time.Second)
  201. for {
  202. select {
  203. case message := <-messageChan:
  204. if err := stream.Send(message); err != nil {
  205. glog.V(0).Infof("=> client %v: %+v", clientName, message)
  206. return err
  207. }
  208. case <-ticker.C:
  209. if !ms.Topo.IsLeader() {
  210. stats.MasterRaftIsleader.Set(0)
  211. return ms.informNewLeader(stream)
  212. } else {
  213. stats.MasterRaftIsleader.Set(1)
  214. }
  215. case <-stopChan:
  216. return nil
  217. }
  218. }
  219. }
  220. func (ms *MasterServer) broadcastToClients(message *master_pb.KeepConnectedResponse) {
  221. ms.clientChansLock.RLock()
  222. for _, ch := range ms.clientChans {
  223. ch <- message
  224. }
  225. ms.clientChansLock.RUnlock()
  226. }
  227. func (ms *MasterServer) informNewLeader(stream master_pb.Seaweed_KeepConnectedServer) error {
  228. leader, err := ms.Topo.Leader()
  229. if err != nil {
  230. glog.Errorf("topo leader: %v", err)
  231. return raft.NotLeaderError
  232. }
  233. if err := stream.Send(&master_pb.KeepConnectedResponse{
  234. VolumeLocation: &master_pb.VolumeLocation{
  235. Leader: string(leader),
  236. },
  237. }); err != nil {
  238. return err
  239. }
  240. return nil
  241. }
  242. func (ms *MasterServer) addClient(filerGroup, clientType string, clientAddress pb.ServerAddress) (clientName string, messageChan chan *master_pb.KeepConnectedResponse) {
  243. clientName = filerGroup + "." + clientType + "@" + string(clientAddress)
  244. glog.V(0).Infof("+ client %v", clientName)
  245. // we buffer this because otherwise we end up in a potential deadlock where
  246. // the KeepConnected loop is no longer listening on this channel but we're
  247. // trying to send to it in SendHeartbeat and so we can't lock the
  248. // clientChansLock to remove the channel and we're stuck writing to it
  249. // 100 is probably overkill
  250. messageChan = make(chan *master_pb.KeepConnectedResponse, 100)
  251. ms.clientChansLock.Lock()
  252. ms.clientChans[clientName] = messageChan
  253. ms.clientChansLock.Unlock()
  254. return
  255. }
  256. func (ms *MasterServer) deleteClient(clientName string) {
  257. glog.V(0).Infof("- client %v", clientName)
  258. ms.clientChansLock.Lock()
  259. delete(ms.clientChans, clientName)
  260. ms.clientChansLock.Unlock()
  261. }
  262. func findClientAddress(ctx context.Context, grpcPort uint32) string {
  263. // fmt.Printf("FromContext %+v\n", ctx)
  264. pr, ok := peer.FromContext(ctx)
  265. if !ok {
  266. glog.Error("failed to get peer from ctx")
  267. return ""
  268. }
  269. if pr.Addr == net.Addr(nil) {
  270. glog.Error("failed to get peer address")
  271. return ""
  272. }
  273. if grpcPort == 0 {
  274. return pr.Addr.String()
  275. }
  276. if tcpAddr, ok := pr.Addr.(*net.TCPAddr); ok {
  277. externalIP := tcpAddr.IP
  278. return util.JoinHostPort(externalIP.String(), int(grpcPort))
  279. }
  280. return pr.Addr.String()
  281. }
  282. func (ms *MasterServer) GetMasterConfiguration(ctx context.Context, req *master_pb.GetMasterConfigurationRequest) (*master_pb.GetMasterConfigurationResponse, error) {
  283. // tell the volume servers about the leader
  284. leader, _ := ms.Topo.Leader()
  285. resp := &master_pb.GetMasterConfigurationResponse{
  286. MetricsAddress: ms.option.MetricsAddress,
  287. MetricsIntervalSeconds: uint32(ms.option.MetricsIntervalSec),
  288. StorageBackends: backend.ToPbStorageBackends(),
  289. DefaultReplication: ms.option.DefaultReplicaPlacement,
  290. VolumeSizeLimitMB: uint32(ms.option.VolumeSizeLimitMB),
  291. VolumePreallocate: ms.option.VolumePreallocate,
  292. Leader: string(leader),
  293. }
  294. return resp, nil
  295. }