You can not select more than 25 topics Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.

304 lines
8.7 KiB

6 years ago
6 years ago
6 years ago
5 years ago
6 years ago
6 years ago
6 years ago
6 years ago
4 years ago
6 years ago
6 years ago
6 years ago
4 years ago
6 years ago
6 years ago
6 years ago
6 years ago
5 years ago
5 years ago
  1. package weed_server
  2. import (
  3. "context"
  4. "fmt"
  5. "net"
  6. "strings"
  7. "time"
  8. "github.com/chrislusf/raft"
  9. "google.golang.org/grpc/peer"
  10. "github.com/chrislusf/seaweedfs/weed/glog"
  11. "github.com/chrislusf/seaweedfs/weed/pb/master_pb"
  12. "github.com/chrislusf/seaweedfs/weed/storage/needle"
  13. "github.com/chrislusf/seaweedfs/weed/topology"
  14. )
  15. func (ms *MasterServer) SendHeartbeat(stream master_pb.Seaweed_SendHeartbeatServer) error {
  16. var dn *topology.DataNode
  17. defer func() {
  18. if dn != nil {
  19. // if the volume server disconnects and reconnects quickly
  20. // the unregister and register can race with each other
  21. ms.Topo.UnRegisterDataNode(dn)
  22. glog.V(0).Infof("unregister disconnected volume server %s:%d", dn.Ip, dn.Port)
  23. message := &master_pb.VolumeLocation{
  24. Url: dn.Url(),
  25. PublicUrl: dn.PublicUrl,
  26. }
  27. for _, v := range dn.GetVolumes() {
  28. message.DeletedVids = append(message.DeletedVids, uint32(v.Id))
  29. }
  30. for _, s := range dn.GetEcShards() {
  31. message.DeletedVids = append(message.DeletedVids, uint32(s.VolumeId))
  32. }
  33. if len(message.DeletedVids) > 0 {
  34. ms.clientChansLock.RLock()
  35. for _, ch := range ms.clientChans {
  36. ch <- message
  37. }
  38. ms.clientChansLock.RUnlock()
  39. }
  40. }
  41. }()
  42. for {
  43. heartbeat, err := stream.Recv()
  44. if err != nil {
  45. if dn != nil {
  46. glog.Warningf("SendHeartbeat.Recv server %s:%d : %v", dn.Ip, dn.Port, err)
  47. } else {
  48. glog.Warningf("SendHeartbeat.Recv: %v", err)
  49. }
  50. return err
  51. }
  52. ms.Topo.Sequence.SetMax(heartbeat.MaxFileKey)
  53. if dn == nil {
  54. dcName, rackName := ms.Topo.Configuration.Locate(heartbeat.Ip, heartbeat.DataCenter, heartbeat.Rack)
  55. dc := ms.Topo.GetOrCreateDataCenter(dcName)
  56. rack := dc.GetOrCreateRack(rackName)
  57. dn = rack.GetOrCreateDataNode(heartbeat.Ip,
  58. int(heartbeat.Port), heartbeat.PublicUrl,
  59. int64(heartbeat.MaxVolumeCount))
  60. glog.V(0).Infof("added volume server %v:%d", heartbeat.GetIp(), heartbeat.GetPort())
  61. if err := stream.Send(&master_pb.HeartbeatResponse{
  62. VolumeSizeLimit: uint64(ms.option.VolumeSizeLimitMB) * 1024 * 1024,
  63. }); err != nil {
  64. glog.Warningf("SendHeartbeat.Send volume size to %s:%d %v", dn.Ip, dn.Port, err)
  65. return err
  66. }
  67. }
  68. if heartbeat.MaxVolumeCount != 0 && dn.GetMaxVolumeCount() != int64(heartbeat.MaxVolumeCount) {
  69. delta := int64(heartbeat.MaxVolumeCount) - dn.GetMaxVolumeCount()
  70. dn.UpAdjustMaxVolumeCountDelta(delta)
  71. }
  72. glog.V(4).Infof("master received heartbeat %s", heartbeat.String())
  73. message := &master_pb.VolumeLocation{
  74. Url: dn.Url(),
  75. PublicUrl: dn.PublicUrl,
  76. }
  77. if len(heartbeat.NewVolumes) > 0 || len(heartbeat.DeletedVolumes) > 0 {
  78. // process delta volume ids if exists for fast volume id updates
  79. for _, volInfo := range heartbeat.NewVolumes {
  80. message.NewVids = append(message.NewVids, volInfo.Id)
  81. }
  82. for _, volInfo := range heartbeat.DeletedVolumes {
  83. message.DeletedVids = append(message.DeletedVids, volInfo.Id)
  84. }
  85. // update master internal volume layouts
  86. ms.Topo.IncrementalSyncDataNodeRegistration(heartbeat.NewVolumes, heartbeat.DeletedVolumes, dn)
  87. }
  88. if len(heartbeat.Volumes) > 0 || heartbeat.HasNoVolumes {
  89. // process heartbeat.Volumes
  90. newVolumes, deletedVolumes := ms.Topo.SyncDataNodeRegistration(heartbeat.Volumes, dn)
  91. for _, v := range newVolumes {
  92. glog.V(0).Infof("master see new volume %d from %s", uint32(v.Id), dn.Url())
  93. message.NewVids = append(message.NewVids, uint32(v.Id))
  94. }
  95. for _, v := range deletedVolumes {
  96. glog.V(0).Infof("master see deleted volume %d from %s", uint32(v.Id), dn.Url())
  97. message.DeletedVids = append(message.DeletedVids, uint32(v.Id))
  98. }
  99. }
  100. if len(heartbeat.NewEcShards) > 0 || len(heartbeat.DeletedEcShards) > 0 {
  101. // update master internal volume layouts
  102. ms.Topo.IncrementalSyncDataNodeEcShards(heartbeat.NewEcShards, heartbeat.DeletedEcShards, dn)
  103. for _, s := range heartbeat.NewEcShards {
  104. message.NewVids = append(message.NewVids, s.Id)
  105. }
  106. for _, s := range heartbeat.DeletedEcShards {
  107. if dn.HasVolumesById(needle.VolumeId(s.Id)) {
  108. continue
  109. }
  110. message.DeletedVids = append(message.DeletedVids, s.Id)
  111. }
  112. }
  113. if len(heartbeat.EcShards) > 0 || heartbeat.HasNoEcShards {
  114. glog.V(1).Infof("master received ec shards from %s: %+v", dn.Url(), heartbeat.EcShards)
  115. newShards, deletedShards := ms.Topo.SyncDataNodeEcShards(heartbeat.EcShards, dn)
  116. // broadcast the ec vid changes to master clients
  117. for _, s := range newShards {
  118. message.NewVids = append(message.NewVids, uint32(s.VolumeId))
  119. }
  120. for _, s := range deletedShards {
  121. if dn.HasVolumesById(s.VolumeId) {
  122. continue
  123. }
  124. message.DeletedVids = append(message.DeletedVids, uint32(s.VolumeId))
  125. }
  126. }
  127. if len(message.NewVids) > 0 || len(message.DeletedVids) > 0 {
  128. ms.clientChansLock.RLock()
  129. for host, ch := range ms.clientChans {
  130. glog.V(0).Infof("master send to %s: %s", host, message.String())
  131. ch <- message
  132. }
  133. ms.clientChansLock.RUnlock()
  134. }
  135. // tell the volume servers about the leader
  136. newLeader, err := ms.Topo.Leader()
  137. if err != nil {
  138. glog.Warningf("SendHeartbeat find leader: %v", err)
  139. return err
  140. }
  141. if err := stream.Send(&master_pb.HeartbeatResponse{
  142. Leader: newLeader,
  143. }); err != nil {
  144. glog.Warningf("SendHeartbeat.Send response to to %s:%d %v", dn.Ip, dn.Port, err)
  145. return err
  146. }
  147. }
  148. }
  149. // KeepConnected keep a stream gRPC call to the master. Used by clients to know the master is up.
  150. // And clients gets the up-to-date list of volume locations
  151. func (ms *MasterServer) KeepConnected(stream master_pb.Seaweed_KeepConnectedServer) error {
  152. req, err := stream.Recv()
  153. if err != nil {
  154. return err
  155. }
  156. if !ms.Topo.IsLeader() {
  157. return ms.informNewLeader(stream)
  158. }
  159. peerAddress := findClientAddress(stream.Context(), req.GrpcPort)
  160. // buffer by 1 so we don't end up getting stuck writing to stopChan forever
  161. stopChan := make(chan bool, 1)
  162. clientName, messageChan := ms.addClient(req.Name, peerAddress)
  163. defer ms.deleteClient(clientName)
  164. for _, message := range ms.Topo.ToVolumeLocations() {
  165. if err := stream.Send(message); err != nil {
  166. return err
  167. }
  168. }
  169. go func() {
  170. for {
  171. _, err := stream.Recv()
  172. if err != nil {
  173. glog.V(2).Infof("- client %v: %v", clientName, err)
  174. stopChan <- true
  175. break
  176. }
  177. }
  178. }()
  179. ticker := time.NewTicker(5 * time.Second)
  180. for {
  181. select {
  182. case message := <-messageChan:
  183. if err := stream.Send(message); err != nil {
  184. glog.V(0).Infof("=> client %v: %+v", clientName, message)
  185. return err
  186. }
  187. case <-ticker.C:
  188. if !ms.Topo.IsLeader() {
  189. return ms.informNewLeader(stream)
  190. }
  191. case <-stopChan:
  192. return nil
  193. }
  194. }
  195. }
  196. func (ms *MasterServer) informNewLeader(stream master_pb.Seaweed_KeepConnectedServer) error {
  197. leader, err := ms.Topo.Leader()
  198. if err != nil {
  199. glog.Errorf("topo leader: %v", err)
  200. return raft.NotLeaderError
  201. }
  202. if err := stream.Send(&master_pb.VolumeLocation{
  203. Leader: leader,
  204. }); err != nil {
  205. return err
  206. }
  207. return nil
  208. }
  209. func (ms *MasterServer) addClient(clientType string, clientAddress string) (clientName string, messageChan chan *master_pb.VolumeLocation) {
  210. clientName = clientType + "@" + clientAddress
  211. glog.V(0).Infof("+ client %v", clientName)
  212. // we buffer this because otherwise we end up in a potential deadlock where
  213. // the KeepConnected loop is no longer listening on this channel but we're
  214. // trying to send to it in SendHeartbeat and so we can't lock the
  215. // clientChansLock to remove the channel and we're stuck writing to it
  216. // 100 is probably overkill
  217. messageChan = make(chan *master_pb.VolumeLocation, 100)
  218. ms.clientChansLock.Lock()
  219. ms.clientChans[clientName] = messageChan
  220. ms.clientChansLock.Unlock()
  221. return
  222. }
  223. func (ms *MasterServer) deleteClient(clientName string) {
  224. glog.V(0).Infof("- client %v", clientName)
  225. ms.clientChansLock.Lock()
  226. delete(ms.clientChans, clientName)
  227. ms.clientChansLock.Unlock()
  228. }
  229. func findClientAddress(ctx context.Context, grpcPort uint32) string {
  230. // fmt.Printf("FromContext %+v\n", ctx)
  231. pr, ok := peer.FromContext(ctx)
  232. if !ok {
  233. glog.Error("failed to get peer from ctx")
  234. return ""
  235. }
  236. if pr.Addr == net.Addr(nil) {
  237. glog.Error("failed to get peer address")
  238. return ""
  239. }
  240. if grpcPort == 0 {
  241. return pr.Addr.String()
  242. }
  243. if tcpAddr, ok := pr.Addr.(*net.TCPAddr); ok {
  244. externalIP := tcpAddr.IP
  245. return fmt.Sprintf("%s:%d", externalIP, grpcPort)
  246. }
  247. return pr.Addr.String()
  248. }
  249. func (ms *MasterServer) ListMasterClients(ctx context.Context, req *master_pb.ListMasterClientsRequest) (*master_pb.ListMasterClientsResponse, error) {
  250. resp := &master_pb.ListMasterClientsResponse{}
  251. ms.clientChansLock.RLock()
  252. defer ms.clientChansLock.RUnlock()
  253. for k := range ms.clientChans {
  254. if strings.HasPrefix(k, req.ClientType+"@") {
  255. resp.GrpcAddresses = append(resp.GrpcAddresses, k[len(req.ClientType)+1:])
  256. }
  257. }
  258. return resp, nil
  259. }