You can not select more than 25 topics Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.

332 lines
10 KiB

6 years ago
4 years ago
6 years ago
4 years ago
4 years ago
4 years ago
4 years ago
4 years ago
4 years ago
5 years ago
4 years ago
  1. package storage
  2. import (
  3. "bytes"
  4. "errors"
  5. "fmt"
  6. "os"
  7. "time"
  8. "github.com/seaweedfs/seaweedfs/weed/glog"
  9. "github.com/seaweedfs/seaweedfs/weed/storage/backend"
  10. "github.com/seaweedfs/seaweedfs/weed/storage/needle"
  11. . "github.com/seaweedfs/seaweedfs/weed/storage/types"
  12. )
  13. var ErrorNotFound = errors.New("not found")
  14. var ErrorDeleted = errors.New("already deleted")
  15. var ErrorSizeMismatch = errors.New("size mismatch")
  16. func (v *Volume) checkReadWriteError(err error) {
  17. if err == nil {
  18. if v.lastIoError != nil {
  19. v.lastIoError = nil
  20. }
  21. return
  22. }
  23. if err.Error() == "input/output error" {
  24. v.lastIoError = err
  25. }
  26. }
  27. // isFileUnchanged checks whether this needle to write is same as last one.
  28. // It requires serialized access in the same volume.
  29. func (v *Volume) isFileUnchanged(n *needle.Needle) bool {
  30. if v.Ttl.String() != "" {
  31. return false
  32. }
  33. nv, ok := v.nm.Get(n.Id)
  34. if ok && !nv.Offset.IsZero() && nv.Size.IsValid() {
  35. oldNeedle := new(needle.Needle)
  36. err := oldNeedle.ReadData(v.DataBackend, nv.Offset.ToActualOffset(), nv.Size, v.Version())
  37. if err != nil {
  38. glog.V(0).Infof("Failed to check updated file at offset %d size %d: %v", nv.Offset.ToActualOffset(), nv.Size, err)
  39. return false
  40. }
  41. if oldNeedle.Cookie == n.Cookie && oldNeedle.Checksum == n.Checksum && bytes.Equal(oldNeedle.Data, n.Data) {
  42. n.DataSize = oldNeedle.DataSize
  43. return true
  44. }
  45. }
  46. return false
  47. }
  48. // Destroy removes everything related to this volume
  49. func (v *Volume) Destroy() (err error) {
  50. if v.isCompacting || v.isCommitCompacting {
  51. err = fmt.Errorf("volume %d is compacting", v.Id)
  52. return
  53. }
  54. close(v.asyncRequestsChan)
  55. storageName, storageKey := v.RemoteStorageNameKey()
  56. if v.HasRemoteFile() && storageName != "" && storageKey != "" {
  57. if backendStorage, found := backend.BackendStorages[storageName]; found {
  58. backendStorage.DeleteFile(storageKey)
  59. }
  60. }
  61. v.Close()
  62. removeVolumeFiles(v.DataFileName())
  63. removeVolumeFiles(v.IndexFileName())
  64. return
  65. }
  66. func removeVolumeFiles(filename string) {
  67. // basic
  68. os.Remove(filename + ".dat")
  69. os.Remove(filename + ".idx")
  70. os.Remove(filename + ".vif")
  71. // sorted index file
  72. os.Remove(filename + ".sdx")
  73. // compaction
  74. os.Remove(filename + ".cpd")
  75. os.Remove(filename + ".cpx")
  76. // level db indx file
  77. os.RemoveAll(filename + ".ldb")
  78. // marker for damaged or incomplete volume
  79. os.Remove(filename + ".note")
  80. }
  81. func (v *Volume) asyncRequestAppend(request *needle.AsyncRequest) {
  82. v.asyncRequestsChan <- request
  83. }
  84. func (v *Volume) syncWrite(n *needle.Needle, checkCookie bool) (offset uint64, size Size, isUnchanged bool, err error) {
  85. // glog.V(4).Infof("writing needle %s", needle.NewFileIdFromNeedle(v.Id, n).String())
  86. actualSize := needle.GetActualSize(Size(len(n.Data)), v.Version())
  87. v.dataFileAccessLock.Lock()
  88. defer v.dataFileAccessLock.Unlock()
  89. if MaxPossibleVolumeSize < v.nm.ContentSize()+uint64(actualSize) {
  90. err = fmt.Errorf("volume size limit %d exceeded! current size is %d", MaxPossibleVolumeSize, v.nm.ContentSize())
  91. return
  92. }
  93. return v.doWriteRequest(n, checkCookie)
  94. }
  95. func (v *Volume) writeNeedle2(n *needle.Needle, checkCookie bool, fsync bool) (offset uint64, size Size, isUnchanged bool, err error) {
  96. // glog.V(4).Infof("writing needle %s", needle.NewFileIdFromNeedle(v.Id, n).String())
  97. if n.Ttl == needle.EMPTY_TTL && v.Ttl != needle.EMPTY_TTL {
  98. n.SetHasTtl()
  99. n.Ttl = v.Ttl
  100. }
  101. if !fsync {
  102. return v.syncWrite(n, checkCookie)
  103. } else {
  104. asyncRequest := needle.NewAsyncRequest(n, true)
  105. // using len(n.Data) here instead of n.Size before n.Size is populated in n.Append()
  106. asyncRequest.ActualSize = needle.GetActualSize(Size(len(n.Data)), v.Version())
  107. v.asyncRequestAppend(asyncRequest)
  108. offset, _, isUnchanged, err = asyncRequest.WaitComplete()
  109. return
  110. }
  111. }
  112. func (v *Volume) doWriteRequest(n *needle.Needle, checkCookie bool) (offset uint64, size Size, isUnchanged bool, err error) {
  113. // glog.V(4).Infof("writing needle %s", needle.NewFileIdFromNeedle(v.Id, n).String())
  114. if v.isFileUnchanged(n) {
  115. size = Size(n.DataSize)
  116. isUnchanged = true
  117. return
  118. }
  119. // check whether existing needle cookie matches
  120. nv, ok := v.nm.Get(n.Id)
  121. if ok {
  122. existingNeedle, _, _, existingNeedleReadErr := needle.ReadNeedleHeader(v.DataBackend, v.Version(), nv.Offset.ToActualOffset())
  123. if existingNeedleReadErr != nil {
  124. err = fmt.Errorf("reading existing needle: %v", existingNeedleReadErr)
  125. return
  126. }
  127. if n.Cookie == 0 && !checkCookie {
  128. // this is from batch deletion, and read back again when tailing a remote volume
  129. // which only happens when checkCookie == false and fsync == false
  130. n.Cookie = existingNeedle.Cookie
  131. }
  132. if existingNeedle.Cookie != n.Cookie {
  133. glog.V(0).Infof("write cookie mismatch: existing %s, new %s",
  134. needle.NewFileIdFromNeedle(v.Id, existingNeedle), needle.NewFileIdFromNeedle(v.Id, n))
  135. err = fmt.Errorf("mismatching cookie %x", n.Cookie)
  136. return
  137. }
  138. }
  139. // append to dat file
  140. n.AppendAtNs = uint64(time.Now().UnixNano())
  141. offset, size, _, err = n.Append(v.DataBackend, v.Version())
  142. v.checkReadWriteError(err)
  143. if err != nil {
  144. return
  145. }
  146. v.lastAppendAtNs = n.AppendAtNs
  147. // add to needle map
  148. if !ok || uint64(nv.Offset.ToActualOffset()) < offset {
  149. if err = v.nm.Put(n.Id, ToOffset(int64(offset)), n.Size); err != nil {
  150. glog.V(4).Infof("failed to save in needle map %d: %v", n.Id, err)
  151. }
  152. }
  153. if v.lastModifiedTsSeconds < n.LastModified {
  154. v.lastModifiedTsSeconds = n.LastModified
  155. }
  156. return
  157. }
  158. func (v *Volume) syncDelete(n *needle.Needle) (Size, error) {
  159. // glog.V(4).Infof("delete needle %s", needle.NewFileIdFromNeedle(v.Id, n).String())
  160. actualSize := needle.GetActualSize(0, v.Version())
  161. v.dataFileAccessLock.Lock()
  162. defer v.dataFileAccessLock.Unlock()
  163. if MaxPossibleVolumeSize < v.nm.ContentSize()+uint64(actualSize) {
  164. err := fmt.Errorf("volume size limit %d exceeded! current size is %d", MaxPossibleVolumeSize, v.nm.ContentSize())
  165. return 0, err
  166. }
  167. return v.doDeleteRequest(n)
  168. }
  169. func (v *Volume) deleteNeedle2(n *needle.Needle) (Size, error) {
  170. // todo: delete info is always appended no fsync, it may need fsync in future
  171. fsync := false
  172. if !fsync {
  173. return v.syncDelete(n)
  174. } else {
  175. asyncRequest := needle.NewAsyncRequest(n, false)
  176. asyncRequest.ActualSize = needle.GetActualSize(0, v.Version())
  177. v.asyncRequestAppend(asyncRequest)
  178. _, size, _, err := asyncRequest.WaitComplete()
  179. return Size(size), err
  180. }
  181. }
  182. func (v *Volume) doDeleteRequest(n *needle.Needle) (Size, error) {
  183. glog.V(4).Infof("delete needle %s", needle.NewFileIdFromNeedle(v.Id, n).String())
  184. nv, ok := v.nm.Get(n.Id)
  185. // fmt.Println("key", n.Id, "volume offset", nv.Offset, "data_size", n.Size, "cached size", nv.Size)
  186. if ok && nv.Size.IsValid() {
  187. size := nv.Size
  188. n.Data = nil
  189. n.AppendAtNs = uint64(time.Now().UnixNano())
  190. offset, _, _, err := n.Append(v.DataBackend, v.Version())
  191. v.checkReadWriteError(err)
  192. if err != nil {
  193. return size, err
  194. }
  195. v.lastAppendAtNs = n.AppendAtNs
  196. if err = v.nm.Delete(n.Id, ToOffset(int64(offset))); err != nil {
  197. return size, err
  198. }
  199. return size, err
  200. }
  201. return 0, nil
  202. }
  203. func (v *Volume) startWorker() {
  204. go func() {
  205. chanClosed := false
  206. for {
  207. // chan closed. go thread will exit
  208. if chanClosed {
  209. break
  210. }
  211. currentRequests := make([]*needle.AsyncRequest, 0, 128)
  212. currentBytesToWrite := int64(0)
  213. for {
  214. request, ok := <-v.asyncRequestsChan
  215. // volume may be closed
  216. if !ok {
  217. chanClosed = true
  218. break
  219. }
  220. if MaxPossibleVolumeSize < v.ContentSize()+uint64(currentBytesToWrite+request.ActualSize) {
  221. request.Complete(0, 0, false,
  222. fmt.Errorf("volume size limit %d exceeded! current size is %d", MaxPossibleVolumeSize, v.ContentSize()))
  223. break
  224. }
  225. currentRequests = append(currentRequests, request)
  226. currentBytesToWrite += request.ActualSize
  227. // submit at most 4M bytes or 128 requests at one time to decrease request delay.
  228. // it also need to break if there is no data in channel to avoid io hang.
  229. if currentBytesToWrite >= 4*1024*1024 || len(currentRequests) >= 128 || len(v.asyncRequestsChan) == 0 {
  230. break
  231. }
  232. }
  233. if len(currentRequests) == 0 {
  234. continue
  235. }
  236. v.dataFileAccessLock.Lock()
  237. end, _, e := v.DataBackend.GetStat()
  238. if e != nil {
  239. for i := 0; i < len(currentRequests); i++ {
  240. currentRequests[i].Complete(0, 0, false,
  241. fmt.Errorf("cannot read current volume position: %v", e))
  242. }
  243. v.dataFileAccessLock.Unlock()
  244. continue
  245. }
  246. for i := 0; i < len(currentRequests); i++ {
  247. if currentRequests[i].IsWriteRequest {
  248. offset, size, isUnchanged, err := v.doWriteRequest(currentRequests[i].N, true)
  249. currentRequests[i].UpdateResult(offset, uint64(size), isUnchanged, err)
  250. } else {
  251. size, err := v.doDeleteRequest(currentRequests[i].N)
  252. currentRequests[i].UpdateResult(0, uint64(size), false, err)
  253. }
  254. }
  255. // if sync error, data is not reliable, we should mark the completed request as fail and rollback
  256. if err := v.DataBackend.Sync(); err != nil {
  257. // todo: this may generate dirty data or cause data inconsistent, may be weed need to panic?
  258. if te := v.DataBackend.Truncate(end); te != nil {
  259. glog.V(0).Infof("Failed to truncate %s back to %d with error: %v", v.DataBackend.Name(), end, te)
  260. }
  261. for i := 0; i < len(currentRequests); i++ {
  262. if currentRequests[i].IsSucceed() {
  263. currentRequests[i].UpdateResult(0, 0, false, err)
  264. }
  265. }
  266. }
  267. for i := 0; i < len(currentRequests); i++ {
  268. currentRequests[i].Submit()
  269. }
  270. v.dataFileAccessLock.Unlock()
  271. }
  272. }()
  273. }
  274. func (v *Volume) WriteNeedleBlob(needleId NeedleId, needleBlob []byte, size Size) error {
  275. v.dataFileAccessLock.Lock()
  276. defer v.dataFileAccessLock.Unlock()
  277. if MaxPossibleVolumeSize < v.nm.ContentSize()+uint64(len(needleBlob)) {
  278. return fmt.Errorf("volume size limit %d exceeded! current size is %d", MaxPossibleVolumeSize, v.nm.ContentSize())
  279. }
  280. appendAtNs := uint64(time.Now().UnixNano())
  281. offset, err := needle.WriteNeedleBlob(v.DataBackend, needleBlob, size, appendAtNs, v.Version())
  282. v.checkReadWriteError(err)
  283. if err != nil {
  284. return err
  285. }
  286. v.lastAppendAtNs = appendAtNs
  287. // add to needle map
  288. if err = v.nm.Put(needleId, ToOffset(int64(offset)), size); err != nil {
  289. glog.V(4).Infof("failed to put in needle map %d: %v", needleId, err)
  290. }
  291. return err
  292. }