You can not select more than 25 topics Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.

392 lines
12 KiB

7 years ago
3 years ago
3 years ago
4 years ago
7 years ago
5 years ago
7 years ago
7 years ago
7 years ago
3 years ago
7 years ago
4 years ago
7 years ago
7 years ago
7 years ago
7 years ago
4 years ago
7 years ago
7 years ago
4 years ago
4 years ago
4 years ago
6 years ago
  1. package weed_server
  2. import (
  3. "context"
  4. "errors"
  5. "fmt"
  6. "net"
  7. "sort"
  8. "time"
  9. "github.com/chrislusf/seaweedfs/weed/pb"
  10. "github.com/chrislusf/seaweedfs/weed/stats"
  11. "github.com/chrislusf/seaweedfs/weed/storage/backend"
  12. "github.com/chrislusf/seaweedfs/weed/util"
  13. "github.com/chrislusf/raft"
  14. "google.golang.org/grpc/peer"
  15. "github.com/chrislusf/seaweedfs/weed/glog"
  16. "github.com/chrislusf/seaweedfs/weed/pb/master_pb"
  17. "github.com/chrislusf/seaweedfs/weed/storage/needle"
  18. "github.com/chrislusf/seaweedfs/weed/topology"
  19. )
  20. func (ms *MasterServer) RegisterUuids(heartbeat *master_pb.Heartbeat) (duplicated_uuids []string, err error) {
  21. ms.Topo.UuidAccessLock.Lock()
  22. defer ms.Topo.UuidAccessLock.Unlock()
  23. key := fmt.Sprintf("%s:%d", heartbeat.Ip, heartbeat.Port)
  24. if ms.Topo.UuidMap == nil {
  25. ms.Topo.UuidMap = make(map[string][]string)
  26. }
  27. // find whether new uuid exists
  28. for k, v := range ms.Topo.UuidMap {
  29. sort.Strings(v)
  30. for _, id := range heartbeat.LocationUuids {
  31. index := sort.SearchStrings(v, id)
  32. if index < len(v) && v[index] == id {
  33. duplicated_uuids = append(duplicated_uuids, id)
  34. glog.Errorf("directory of %s on %s has been loaded", id, k)
  35. }
  36. }
  37. }
  38. if len(duplicated_uuids) > 0 {
  39. return duplicated_uuids, errors.New("volume: Duplicated volume directories were loaded")
  40. }
  41. ms.Topo.UuidMap[key] = heartbeat.LocationUuids
  42. glog.V(0).Infof("found new uuid:%v %v , %v", key, heartbeat.LocationUuids, ms.Topo.UuidMap)
  43. return nil, nil
  44. }
  45. func (ms *MasterServer) UnRegisterUuids(ip string, port int) {
  46. ms.Topo.UuidAccessLock.Lock()
  47. defer ms.Topo.UuidAccessLock.Unlock()
  48. key := fmt.Sprintf("%s:%d", ip, port)
  49. delete(ms.Topo.UuidMap, key)
  50. glog.V(0).Infof("remove volume server %v, online volume server: %v", key, ms.Topo.UuidMap)
  51. }
  52. func (ms *MasterServer) SendHeartbeat(stream master_pb.Seaweed_SendHeartbeatServer) error {
  53. var dn *topology.DataNode
  54. defer func() {
  55. if dn != nil {
  56. dn.Counter--
  57. if dn.Counter > 0 {
  58. glog.V(0).Infof("disconnect phantom volume server %s:%d remaining %d", dn.Ip, dn.Port, dn.Counter)
  59. return
  60. }
  61. // if the volume server disconnects and reconnects quickly
  62. // the unregister and register can race with each other
  63. ms.Topo.UnRegisterDataNode(dn)
  64. glog.V(0).Infof("unregister disconnected volume server %s:%d", dn.Ip, dn.Port)
  65. ms.UnRegisterUuids(dn.Ip, dn.Port)
  66. message := &master_pb.VolumeLocation{
  67. Url: dn.Url(),
  68. PublicUrl: dn.PublicUrl,
  69. }
  70. for _, v := range dn.GetVolumes() {
  71. message.DeletedVids = append(message.DeletedVids, uint32(v.Id))
  72. }
  73. for _, s := range dn.GetEcShards() {
  74. message.DeletedVids = append(message.DeletedVids, uint32(s.VolumeId))
  75. }
  76. if len(message.DeletedVids) > 0 {
  77. ms.broadcastToClients(&master_pb.KeepConnectedResponse{VolumeLocation: message})
  78. }
  79. }
  80. }()
  81. for {
  82. heartbeat, err := stream.Recv()
  83. if err != nil {
  84. if dn != nil {
  85. glog.Warningf("SendHeartbeat.Recv server %s:%d : %v", dn.Ip, dn.Port, err)
  86. } else {
  87. glog.Warningf("SendHeartbeat.Recv: %v", err)
  88. }
  89. stats.MasterReceivedHeartbeatCounter.WithLabelValues("error").Inc()
  90. return err
  91. }
  92. ms.Topo.Sequence.SetMax(heartbeat.MaxFileKey)
  93. if dn == nil {
  94. dcName, rackName := ms.Topo.Configuration.Locate(heartbeat.Ip, heartbeat.DataCenter, heartbeat.Rack)
  95. dc := ms.Topo.GetOrCreateDataCenter(dcName)
  96. rack := dc.GetOrCreateRack(rackName)
  97. dn = rack.GetOrCreateDataNode(heartbeat.Ip, int(heartbeat.Port), int(heartbeat.GrpcPort), heartbeat.PublicUrl, heartbeat.MaxVolumeCounts)
  98. glog.V(0).Infof("added volume server %d: %v:%d %v", dn.Counter, heartbeat.GetIp(), heartbeat.GetPort(), heartbeat.LocationUuids)
  99. uuidlist, err := ms.RegisterUuids(heartbeat)
  100. if err != nil {
  101. if stream_err := stream.Send(&master_pb.HeartbeatResponse{
  102. DuplicatedUuids: uuidlist,
  103. }); stream_err != nil {
  104. glog.Warningf("SendHeartbeat.Send DuplicatedDirectory response to %s:%d %v", dn.Ip, dn.Port, stream_err)
  105. return stream_err
  106. }
  107. return err
  108. }
  109. if err := stream.Send(&master_pb.HeartbeatResponse{
  110. VolumeSizeLimit: uint64(ms.option.VolumeSizeLimitMB) * 1024 * 1024,
  111. }); err != nil {
  112. glog.Warningf("SendHeartbeat.Send volume size to %s:%d %v", dn.Ip, dn.Port, err)
  113. return err
  114. }
  115. stats.MasterReceivedHeartbeatCounter.WithLabelValues("dataNode").Inc()
  116. dn.Counter++
  117. }
  118. dn.AdjustMaxVolumeCounts(heartbeat.MaxVolumeCounts)
  119. glog.V(4).Infof("master received heartbeat %s", heartbeat.String())
  120. stats.MasterReceivedHeartbeatCounter.WithLabelValues("total").Inc()
  121. var dataCenter string
  122. if dc := dn.GetDataCenter(); dc != nil {
  123. dataCenter = string(dc.Id())
  124. }
  125. message := &master_pb.VolumeLocation{
  126. Url: dn.Url(),
  127. PublicUrl: dn.PublicUrl,
  128. DataCenter: dataCenter,
  129. }
  130. if len(heartbeat.NewVolumes) > 0 {
  131. stats.FilerRequestCounter.WithLabelValues("newVolumes").Inc()
  132. }
  133. if len(heartbeat.DeletedVolumes) > 0 {
  134. stats.FilerRequestCounter.WithLabelValues("deletedVolumes").Inc()
  135. }
  136. if len(heartbeat.NewVolumes) > 0 || len(heartbeat.DeletedVolumes) > 0 {
  137. // process delta volume ids if exists for fast volume id updates
  138. for _, volInfo := range heartbeat.NewVolumes {
  139. message.NewVids = append(message.NewVids, volInfo.Id)
  140. }
  141. for _, volInfo := range heartbeat.DeletedVolumes {
  142. message.DeletedVids = append(message.DeletedVids, volInfo.Id)
  143. }
  144. // update master internal volume layouts
  145. ms.Topo.IncrementalSyncDataNodeRegistration(heartbeat.NewVolumes, heartbeat.DeletedVolumes, dn)
  146. }
  147. if len(heartbeat.Volumes) > 0 || heartbeat.HasNoVolumes {
  148. dcName, rackName := ms.Topo.Configuration.Locate(heartbeat.Ip, heartbeat.DataCenter, heartbeat.Rack)
  149. ms.Topo.DataNodeRegistration(dcName, rackName, dn)
  150. // process heartbeat.Volumes
  151. stats.MasterReceivedHeartbeatCounter.WithLabelValues("Volumes").Inc()
  152. newVolumes, deletedVolumes := ms.Topo.SyncDataNodeRegistration(heartbeat.Volumes, dn)
  153. for _, v := range newVolumes {
  154. glog.V(0).Infof("master see new volume %d from %s", uint32(v.Id), dn.Url())
  155. message.NewVids = append(message.NewVids, uint32(v.Id))
  156. }
  157. for _, v := range deletedVolumes {
  158. glog.V(0).Infof("master see deleted volume %d from %s", uint32(v.Id), dn.Url())
  159. message.DeletedVids = append(message.DeletedVids, uint32(v.Id))
  160. }
  161. }
  162. if len(heartbeat.NewEcShards) > 0 || len(heartbeat.DeletedEcShards) > 0 {
  163. stats.MasterReceivedHeartbeatCounter.WithLabelValues("newEcShards").Inc()
  164. // update master internal volume layouts
  165. ms.Topo.IncrementalSyncDataNodeEcShards(heartbeat.NewEcShards, heartbeat.DeletedEcShards, dn)
  166. for _, s := range heartbeat.NewEcShards {
  167. message.NewEcVids = append(message.NewEcVids, s.Id)
  168. }
  169. for _, s := range heartbeat.DeletedEcShards {
  170. if dn.HasEcShards(needle.VolumeId(s.Id)) {
  171. continue
  172. }
  173. message.DeletedEcVids = append(message.DeletedEcVids, s.Id)
  174. }
  175. }
  176. if len(heartbeat.EcShards) > 0 || heartbeat.HasNoEcShards {
  177. stats.MasterReceivedHeartbeatCounter.WithLabelValues("ecShards").Inc()
  178. glog.V(4).Infof("master received ec shards from %s: %+v", dn.Url(), heartbeat.EcShards)
  179. newShards, deletedShards := ms.Topo.SyncDataNodeEcShards(heartbeat.EcShards, dn)
  180. // broadcast the ec vid changes to master clients
  181. for _, s := range newShards {
  182. message.NewEcVids = append(message.NewEcVids, uint32(s.VolumeId))
  183. }
  184. for _, s := range deletedShards {
  185. if dn.HasVolumesById(s.VolumeId) {
  186. continue
  187. }
  188. message.DeletedEcVids = append(message.DeletedEcVids, uint32(s.VolumeId))
  189. }
  190. }
  191. if len(message.NewVids) > 0 || len(message.DeletedVids) > 0 || len(message.NewEcVids) > 0 || len(message.DeletedEcVids) > 0 {
  192. ms.broadcastToClients(&master_pb.KeepConnectedResponse{VolumeLocation: message})
  193. }
  194. // tell the volume servers about the leader
  195. newLeader, err := ms.Topo.Leader()
  196. if err != nil {
  197. glog.Warningf("SendHeartbeat find leader: %v", err)
  198. return err
  199. }
  200. if err := stream.Send(&master_pb.HeartbeatResponse{
  201. Leader: string(newLeader),
  202. }); err != nil {
  203. glog.Warningf("SendHeartbeat.Send response to to %s:%d %v", dn.Ip, dn.Port, err)
  204. return err
  205. }
  206. }
  207. }
  208. // KeepConnected keep a stream gRPC call to the master. Used by clients to know the master is up.
  209. // And clients gets the up-to-date list of volume locations
  210. func (ms *MasterServer) KeepConnected(stream master_pb.Seaweed_KeepConnectedServer) error {
  211. req, recvErr := stream.Recv()
  212. if recvErr != nil {
  213. return recvErr
  214. }
  215. if !ms.Topo.IsLeader() {
  216. return ms.informNewLeader(stream)
  217. }
  218. peerAddress := pb.ServerAddress(req.ClientAddress)
  219. // buffer by 1 so we don't end up getting stuck writing to stopChan forever
  220. stopChan := make(chan bool, 1)
  221. clientName, messageChan := ms.addClient(req.FilerGroup, req.ClientType, peerAddress)
  222. for _, update := range ms.Cluster.AddClusterNode(req.FilerGroup, req.ClientType, peerAddress, req.Version) {
  223. ms.broadcastToClients(update)
  224. }
  225. defer func() {
  226. for _, update := range ms.Cluster.RemoveClusterNode(req.FilerGroup, req.ClientType, peerAddress) {
  227. ms.broadcastToClients(update)
  228. }
  229. ms.deleteClient(clientName)
  230. }()
  231. for _, message := range ms.Topo.ToVolumeLocations() {
  232. if sendErr := stream.Send(&master_pb.KeepConnectedResponse{VolumeLocation: message}); sendErr != nil {
  233. return sendErr
  234. }
  235. }
  236. go func() {
  237. for {
  238. _, err := stream.Recv()
  239. if err != nil {
  240. glog.V(2).Infof("- client %v: %v", clientName, err)
  241. close(stopChan)
  242. return
  243. }
  244. }
  245. }()
  246. ticker := time.NewTicker(5 * time.Second)
  247. for {
  248. select {
  249. case message := <-messageChan:
  250. if err := stream.Send(message); err != nil {
  251. glog.V(0).Infof("=> client %v: %+v", clientName, message)
  252. return err
  253. }
  254. case <-ticker.C:
  255. if !ms.Topo.IsLeader() {
  256. stats.MasterRaftIsleader.Set(0)
  257. return ms.informNewLeader(stream)
  258. } else {
  259. stats.MasterRaftIsleader.Set(1)
  260. }
  261. case <-stopChan:
  262. return nil
  263. }
  264. }
  265. }
  266. func (ms *MasterServer) broadcastToClients(message *master_pb.KeepConnectedResponse) {
  267. ms.clientChansLock.RLock()
  268. for _, ch := range ms.clientChans {
  269. ch <- message
  270. }
  271. ms.clientChansLock.RUnlock()
  272. }
  273. func (ms *MasterServer) informNewLeader(stream master_pb.Seaweed_KeepConnectedServer) error {
  274. leader, err := ms.Topo.Leader()
  275. if err != nil {
  276. glog.Errorf("topo leader: %v", err)
  277. return raft.NotLeaderError
  278. }
  279. if err := stream.Send(&master_pb.KeepConnectedResponse{
  280. VolumeLocation: &master_pb.VolumeLocation{
  281. Leader: string(leader),
  282. },
  283. }); err != nil {
  284. return err
  285. }
  286. return nil
  287. }
  288. func (ms *MasterServer) addClient(filerGroup, clientType string, clientAddress pb.ServerAddress) (clientName string, messageChan chan *master_pb.KeepConnectedResponse) {
  289. clientName = filerGroup + "." + clientType + "@" + string(clientAddress)
  290. glog.V(0).Infof("+ client %v", clientName)
  291. // we buffer this because otherwise we end up in a potential deadlock where
  292. // the KeepConnected loop is no longer listening on this channel but we're
  293. // trying to send to it in SendHeartbeat and so we can't lock the
  294. // clientChansLock to remove the channel and we're stuck writing to it
  295. // 100 is probably overkill
  296. messageChan = make(chan *master_pb.KeepConnectedResponse, 100)
  297. ms.clientChansLock.Lock()
  298. ms.clientChans[clientName] = messageChan
  299. ms.clientChansLock.Unlock()
  300. return
  301. }
  302. func (ms *MasterServer) deleteClient(clientName string) {
  303. glog.V(0).Infof("- client %v", clientName)
  304. ms.clientChansLock.Lock()
  305. delete(ms.clientChans, clientName)
  306. ms.clientChansLock.Unlock()
  307. }
  308. func findClientAddress(ctx context.Context, grpcPort uint32) string {
  309. // fmt.Printf("FromContext %+v\n", ctx)
  310. pr, ok := peer.FromContext(ctx)
  311. if !ok {
  312. glog.Error("failed to get peer from ctx")
  313. return ""
  314. }
  315. if pr.Addr == net.Addr(nil) {
  316. glog.Error("failed to get peer address")
  317. return ""
  318. }
  319. if grpcPort == 0 {
  320. return pr.Addr.String()
  321. }
  322. if tcpAddr, ok := pr.Addr.(*net.TCPAddr); ok {
  323. externalIP := tcpAddr.IP
  324. return util.JoinHostPort(externalIP.String(), int(grpcPort))
  325. }
  326. return pr.Addr.String()
  327. }
  328. func (ms *MasterServer) GetMasterConfiguration(ctx context.Context, req *master_pb.GetMasterConfigurationRequest) (*master_pb.GetMasterConfigurationResponse, error) {
  329. // tell the volume servers about the leader
  330. leader, _ := ms.Topo.Leader()
  331. resp := &master_pb.GetMasterConfigurationResponse{
  332. MetricsAddress: ms.option.MetricsAddress,
  333. MetricsIntervalSeconds: uint32(ms.option.MetricsIntervalSec),
  334. StorageBackends: backend.ToPbStorageBackends(),
  335. DefaultReplication: ms.option.DefaultReplicaPlacement,
  336. VolumeSizeLimitMB: uint32(ms.option.VolumeSizeLimitMB),
  337. VolumePreallocate: ms.option.VolumePreallocate,
  338. Leader: string(leader),
  339. }
  340. return resp, nil
  341. }