You can not select more than 25 topics Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.

334 lines
10 KiB

6 years ago
5 years ago
6 years ago
4 years ago
4 years ago
4 years ago
4 years ago
4 years ago
5 years ago
5 years ago
  1. package storage
  2. import (
  3. "bytes"
  4. "errors"
  5. "fmt"
  6. "github.com/seaweedfs/seaweedfs/weed/glog"
  7. "github.com/seaweedfs/seaweedfs/weed/storage/backend"
  8. "github.com/seaweedfs/seaweedfs/weed/storage/needle"
  9. . "github.com/seaweedfs/seaweedfs/weed/storage/types"
  10. "os"
  11. )
  12. var ErrorNotFound = errors.New("not found")
  13. var ErrorDeleted = errors.New("already deleted")
  14. var ErrorSizeMismatch = errors.New("size mismatch")
  15. func (v *Volume) checkReadWriteError(err error) {
  16. if err == nil {
  17. if v.lastIoError != nil {
  18. v.lastIoError = nil
  19. }
  20. return
  21. }
  22. if err.Error() == "input/output error" {
  23. v.lastIoError = err
  24. }
  25. }
  26. // isFileUnchanged checks whether this needle to write is same as last one.
  27. // It requires serialized access in the same volume.
  28. func (v *Volume) isFileUnchanged(n *needle.Needle) bool {
  29. if v.Ttl.String() != "" {
  30. return false
  31. }
  32. nv, ok := v.nm.Get(n.Id)
  33. if ok && !nv.Offset.IsZero() && nv.Size.IsValid() {
  34. oldNeedle := new(needle.Needle)
  35. err := oldNeedle.ReadData(v.DataBackend, nv.Offset.ToActualOffset(), nv.Size, v.Version())
  36. if err != nil {
  37. glog.V(0).Infof("Failed to check updated file at offset %d size %d: %v", nv.Offset.ToActualOffset(), nv.Size, err)
  38. return false
  39. }
  40. if oldNeedle.Cookie == n.Cookie && oldNeedle.Checksum == n.Checksum && bytes.Equal(oldNeedle.Data, n.Data) {
  41. n.DataSize = oldNeedle.DataSize
  42. return true
  43. }
  44. }
  45. return false
  46. }
  47. // Destroy removes everything related to this volume
  48. func (v *Volume) Destroy() (err error) {
  49. if v.isCompacting || v.isCommitCompacting {
  50. err = fmt.Errorf("volume %d is compacting", v.Id)
  51. return
  52. }
  53. close(v.asyncRequestsChan)
  54. storageName, storageKey := v.RemoteStorageNameKey()
  55. if v.HasRemoteFile() && storageName != "" && storageKey != "" {
  56. if backendStorage, found := backend.BackendStorages[storageName]; found {
  57. backendStorage.DeleteFile(storageKey)
  58. }
  59. }
  60. v.Close()
  61. removeVolumeFiles(v.DataFileName())
  62. removeVolumeFiles(v.IndexFileName())
  63. return
  64. }
  65. func removeVolumeFiles(filename string) {
  66. // basic
  67. os.Remove(filename + ".dat")
  68. os.Remove(filename + ".idx")
  69. os.Remove(filename + ".vif")
  70. // sorted index file
  71. os.Remove(filename + ".sdx")
  72. // compaction
  73. os.Remove(filename + ".cpd")
  74. os.Remove(filename + ".cpx")
  75. // level db index file
  76. os.RemoveAll(filename + ".ldb")
  77. // marker for damaged or incomplete volume
  78. os.Remove(filename + ".note")
  79. }
  80. func (v *Volume) asyncRequestAppend(request *needle.AsyncRequest) {
  81. v.asyncRequestsChan <- request
  82. }
  83. func (v *Volume) syncWrite(n *needle.Needle, checkCookie bool) (offset uint64, size Size, isUnchanged bool, err error) {
  84. // glog.V(4).Infof("writing needle %s", needle.NewFileIdFromNeedle(v.Id, n).String())
  85. actualSize := needle.GetActualSize(Size(len(n.Data)), v.Version())
  86. v.dataFileAccessLock.Lock()
  87. defer v.dataFileAccessLock.Unlock()
  88. if MaxPossibleVolumeSize < v.nm.ContentSize()+uint64(actualSize) {
  89. err = fmt.Errorf("volume size limit %d exceeded! current size is %d", MaxPossibleVolumeSize, v.nm.ContentSize())
  90. return
  91. }
  92. return v.doWriteRequest(n, checkCookie)
  93. }
  94. func (v *Volume) writeNeedle2(n *needle.Needle, checkCookie bool, fsync bool) (offset uint64, size Size, isUnchanged bool, err error) {
  95. // glog.V(4).Infof("writing needle %s", needle.NewFileIdFromNeedle(v.Id, n).String())
  96. if n.Ttl == needle.EMPTY_TTL && v.Ttl != needle.EMPTY_TTL {
  97. n.SetHasTtl()
  98. n.Ttl = v.Ttl
  99. }
  100. if !fsync {
  101. return v.syncWrite(n, checkCookie)
  102. } else {
  103. asyncRequest := needle.NewAsyncRequest(n, true)
  104. // using len(n.Data) here instead of n.Size before n.Size is populated in n.Append()
  105. asyncRequest.ActualSize = needle.GetActualSize(Size(len(n.Data)), v.Version())
  106. v.asyncRequestAppend(asyncRequest)
  107. offset, _, isUnchanged, err = asyncRequest.WaitComplete()
  108. return
  109. }
  110. }
  111. func (v *Volume) doWriteRequest(n *needle.Needle, checkCookie bool) (offset uint64, size Size, isUnchanged bool, err error) {
  112. // glog.V(4).Infof("writing needle %s", needle.NewFileIdFromNeedle(v.Id, n).String())
  113. if v.isFileUnchanged(n) {
  114. size = Size(n.DataSize)
  115. isUnchanged = true
  116. return
  117. }
  118. // check whether existing needle cookie matches
  119. nv, ok := v.nm.Get(n.Id)
  120. if ok {
  121. existingNeedle, _, _, existingNeedleReadErr := needle.ReadNeedleHeader(v.DataBackend, v.Version(), nv.Offset.ToActualOffset())
  122. if existingNeedleReadErr != nil {
  123. err = fmt.Errorf("reading existing needle: %v", existingNeedleReadErr)
  124. return
  125. }
  126. if n.Cookie == 0 && !checkCookie {
  127. // this is from batch deletion, and read back again when tailing a remote volume
  128. // which only happens when checkCookie == false and fsync == false
  129. n.Cookie = existingNeedle.Cookie
  130. }
  131. if existingNeedle.Cookie != n.Cookie {
  132. glog.V(0).Infof("write cookie mismatch: existing %s, new %s",
  133. needle.NewFileIdFromNeedle(v.Id, existingNeedle), needle.NewFileIdFromNeedle(v.Id, n))
  134. err = fmt.Errorf("mismatching cookie %x", n.Cookie)
  135. return
  136. }
  137. }
  138. // append to dat file
  139. n.UpdateAppendAtNs(v.lastAppendAtNs)
  140. offset, size, _, err = n.Append(v.DataBackend, v.Version())
  141. v.checkReadWriteError(err)
  142. if err != nil {
  143. return
  144. }
  145. v.lastAppendAtNs = n.AppendAtNs
  146. // add to needle map
  147. if !ok || uint64(nv.Offset.ToActualOffset()) < offset {
  148. if err = v.nm.Put(n.Id, ToOffset(int64(offset)), n.Size); err != nil {
  149. glog.V(4).Infof("failed to save in needle map %d: %v", n.Id, err)
  150. }
  151. }
  152. if v.lastModifiedTsSeconds < n.LastModified {
  153. v.lastModifiedTsSeconds = n.LastModified
  154. }
  155. return
  156. }
  157. func (v *Volume) syncDelete(n *needle.Needle) (Size, error) {
  158. // glog.V(4).Infof("delete needle %s", needle.NewFileIdFromNeedle(v.Id, n).String())
  159. actualSize := needle.GetActualSize(0, v.Version())
  160. v.dataFileAccessLock.Lock()
  161. defer v.dataFileAccessLock.Unlock()
  162. if MaxPossibleVolumeSize < v.nm.ContentSize()+uint64(actualSize) {
  163. err := fmt.Errorf("volume size limit %d exceeded! current size is %d", MaxPossibleVolumeSize, v.nm.ContentSize())
  164. return 0, err
  165. }
  166. return v.doDeleteRequest(n)
  167. }
  168. func (v *Volume) deleteNeedle2(n *needle.Needle) (Size, error) {
  169. // todo: delete info is always appended no fsync, it may need fsync in future
  170. fsync := false
  171. if !fsync {
  172. return v.syncDelete(n)
  173. } else {
  174. asyncRequest := needle.NewAsyncRequest(n, false)
  175. asyncRequest.ActualSize = needle.GetActualSize(0, v.Version())
  176. v.asyncRequestAppend(asyncRequest)
  177. _, size, _, err := asyncRequest.WaitComplete()
  178. return Size(size), err
  179. }
  180. }
  181. func (v *Volume) doDeleteRequest(n *needle.Needle) (Size, error) {
  182. glog.V(4).Infof("delete needle %s", needle.NewFileIdFromNeedle(v.Id, n).String())
  183. nv, ok := v.nm.Get(n.Id)
  184. // fmt.Println("key", n.Id, "volume offset", nv.Offset, "data_size", n.Size, "cached size", nv.Size)
  185. if ok && nv.Size.IsValid() {
  186. var offset uint64
  187. var err error
  188. size := nv.Size
  189. if !v.hasRemoteFile {
  190. n.Data = nil
  191. n.UpdateAppendAtNs(v.lastAppendAtNs)
  192. offset, _, _, err = n.Append(v.DataBackend, v.Version())
  193. v.checkReadWriteError(err)
  194. if err != nil {
  195. return size, err
  196. }
  197. }
  198. v.lastAppendAtNs = n.AppendAtNs
  199. if err = v.nm.Delete(n.Id, ToOffset(int64(offset))); err != nil {
  200. return size, err
  201. }
  202. return size, err
  203. }
  204. return 0, nil
  205. }
  206. func (v *Volume) startWorker() {
  207. go func() {
  208. chanClosed := false
  209. for {
  210. // chan closed. go thread will exit
  211. if chanClosed {
  212. break
  213. }
  214. currentRequests := make([]*needle.AsyncRequest, 0, 128)
  215. currentBytesToWrite := int64(0)
  216. for {
  217. request, ok := <-v.asyncRequestsChan
  218. // volume may be closed
  219. if !ok {
  220. chanClosed = true
  221. break
  222. }
  223. if MaxPossibleVolumeSize < v.ContentSize()+uint64(currentBytesToWrite+request.ActualSize) {
  224. request.Complete(0, 0, false,
  225. fmt.Errorf("volume size limit %d exceeded! current size is %d", MaxPossibleVolumeSize, v.ContentSize()))
  226. break
  227. }
  228. currentRequests = append(currentRequests, request)
  229. currentBytesToWrite += request.ActualSize
  230. // submit at most 4M bytes or 128 requests at one time to decrease request delay.
  231. // it also need to break if there is no data in channel to avoid io hang.
  232. if currentBytesToWrite >= 4*1024*1024 || len(currentRequests) >= 128 || len(v.asyncRequestsChan) == 0 {
  233. break
  234. }
  235. }
  236. if len(currentRequests) == 0 {
  237. continue
  238. }
  239. v.dataFileAccessLock.Lock()
  240. end, _, e := v.DataBackend.GetStat()
  241. if e != nil {
  242. for i := 0; i < len(currentRequests); i++ {
  243. currentRequests[i].Complete(0, 0, false,
  244. fmt.Errorf("cannot read current volume position: %v", e))
  245. }
  246. v.dataFileAccessLock.Unlock()
  247. continue
  248. }
  249. for i := 0; i < len(currentRequests); i++ {
  250. if currentRequests[i].IsWriteRequest {
  251. offset, size, isUnchanged, err := v.doWriteRequest(currentRequests[i].N, true)
  252. currentRequests[i].UpdateResult(offset, uint64(size), isUnchanged, err)
  253. } else {
  254. size, err := v.doDeleteRequest(currentRequests[i].N)
  255. currentRequests[i].UpdateResult(0, uint64(size), false, err)
  256. }
  257. }
  258. // if sync error, data is not reliable, we should mark the completed request as fail and rollback
  259. if err := v.DataBackend.Sync(); err != nil {
  260. // todo: this may generate dirty data or cause data inconsistent, may be weed need to panic?
  261. if te := v.DataBackend.Truncate(end); te != nil {
  262. glog.V(0).Infof("Failed to truncate %s back to %d with error: %v", v.DataBackend.Name(), end, te)
  263. }
  264. for i := 0; i < len(currentRequests); i++ {
  265. if currentRequests[i].IsSucceed() {
  266. currentRequests[i].UpdateResult(0, 0, false, err)
  267. }
  268. }
  269. }
  270. for i := 0; i < len(currentRequests); i++ {
  271. currentRequests[i].Submit()
  272. }
  273. v.dataFileAccessLock.Unlock()
  274. }
  275. }()
  276. }
  277. func (v *Volume) WriteNeedleBlob(needleId NeedleId, needleBlob []byte, size Size) error {
  278. v.dataFileAccessLock.Lock()
  279. defer v.dataFileAccessLock.Unlock()
  280. if MaxPossibleVolumeSize < v.nm.ContentSize()+uint64(len(needleBlob)) {
  281. return fmt.Errorf("volume size limit %d exceeded! current size is %d", MaxPossibleVolumeSize, v.nm.ContentSize())
  282. }
  283. appendAtNs := needle.GetAppendAtNs(v.lastAppendAtNs)
  284. offset, err := needle.WriteNeedleBlob(v.DataBackend, needleBlob, size, appendAtNs, v.Version())
  285. v.checkReadWriteError(err)
  286. if err != nil {
  287. return err
  288. }
  289. v.lastAppendAtNs = appendAtNs
  290. // add to needle map
  291. if err = v.nm.Put(needleId, ToOffset(int64(offset)), size); err != nil {
  292. glog.V(4).Infof("failed to put in needle map %d: %v", needleId, err)
  293. }
  294. return err
  295. }