You can not select more than 25 topics Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.

449 lines
14 KiB

6 years ago
5 years ago
6 years ago
4 years ago
4 years ago
4 years ago
4 years ago
4 years ago
4 years ago
5 years ago
5 years ago
5 years ago
6 years ago
5 years ago
  1. package storage
  2. import (
  3. "bytes"
  4. "errors"
  5. "fmt"
  6. "io"
  7. "os"
  8. "time"
  9. "github.com/chrislusf/seaweedfs/weed/glog"
  10. "github.com/chrislusf/seaweedfs/weed/storage/backend"
  11. "github.com/chrislusf/seaweedfs/weed/storage/needle"
  12. "github.com/chrislusf/seaweedfs/weed/storage/super_block"
  13. . "github.com/chrislusf/seaweedfs/weed/storage/types"
  14. )
  15. var ErrorNotFound = errors.New("not found")
  16. var ErrorDeleted = errors.New("already deleted")
  17. var ErrorSizeMismatch = errors.New("size mismatch")
  18. // isFileUnchanged checks whether this needle to write is same as last one.
  19. // It requires serialized access in the same volume.
  20. func (v *Volume) isFileUnchanged(n *needle.Needle) bool {
  21. if v.Ttl.String() != "" {
  22. return false
  23. }
  24. nv, ok := v.nm.Get(n.Id)
  25. if ok && !nv.Offset.IsZero() && nv.Size.IsValid() {
  26. oldNeedle := new(needle.Needle)
  27. err := oldNeedle.ReadData(v.DataBackend, nv.Offset.ToAcutalOffset(), nv.Size, v.Version())
  28. if err != nil {
  29. glog.V(0).Infof("Failed to check updated file at offset %d size %d: %v", nv.Offset.ToAcutalOffset(), nv.Size, err)
  30. return false
  31. }
  32. if oldNeedle.Cookie == n.Cookie && oldNeedle.Checksum == n.Checksum && bytes.Equal(oldNeedle.Data, n.Data) {
  33. n.DataSize = oldNeedle.DataSize
  34. return true
  35. }
  36. }
  37. return false
  38. }
  39. // Destroy removes everything related to this volume
  40. func (v *Volume) Destroy() (err error) {
  41. if v.isCompacting {
  42. err = fmt.Errorf("volume %d is compacting", v.Id)
  43. return
  44. }
  45. close(v.asyncRequestsChan)
  46. storageName, storageKey := v.RemoteStorageNameKey()
  47. if v.HasRemoteFile() && storageName != "" && storageKey != "" {
  48. if backendStorage, found := backend.BackendStorages[storageName]; found {
  49. backendStorage.DeleteFile(storageKey)
  50. }
  51. }
  52. v.Close()
  53. removeVolumeFiles(v.DataFileName())
  54. removeVolumeFiles(v.IndexFileName())
  55. return
  56. }
  57. func removeVolumeFiles(filename string) {
  58. // basic
  59. os.Remove(filename + ".dat")
  60. os.Remove(filename + ".idx")
  61. os.Remove(filename + ".vif")
  62. // sorted index file
  63. os.Remove(filename + ".sdx")
  64. // compaction
  65. os.Remove(filename + ".cpd")
  66. os.Remove(filename + ".cpx")
  67. // level db indx file
  68. os.RemoveAll(filename + ".ldb")
  69. // marker for damaged or incomplete volume
  70. os.Remove(filename + ".note")
  71. }
  72. func (v *Volume) asyncRequestAppend(request *needle.AsyncRequest) {
  73. v.asyncRequestsChan <- request
  74. }
  75. func (v *Volume) syncWrite(n *needle.Needle) (offset uint64, size Size, isUnchanged bool, err error) {
  76. // glog.V(4).Infof("writing needle %s", needle.NewFileIdFromNeedle(v.Id, n).String())
  77. actualSize := needle.GetActualSize(Size(len(n.Data)), v.Version())
  78. v.dataFileAccessLock.Lock()
  79. defer v.dataFileAccessLock.Unlock()
  80. if MaxPossibleVolumeSize < v.nm.ContentSize()+uint64(actualSize) {
  81. err = fmt.Errorf("volume size limit %d exceeded! current size is %d", MaxPossibleVolumeSize, v.nm.ContentSize())
  82. return
  83. }
  84. if v.isFileUnchanged(n) {
  85. size = Size(n.DataSize)
  86. isUnchanged = true
  87. return
  88. }
  89. // check whether existing needle cookie matches
  90. nv, ok := v.nm.Get(n.Id)
  91. if ok {
  92. existingNeedle, _, _, existingNeedleReadErr := needle.ReadNeedleHeader(v.DataBackend, v.Version(), nv.Offset.ToAcutalOffset())
  93. if existingNeedleReadErr != nil {
  94. err = fmt.Errorf("reading existing needle: %v", existingNeedleReadErr)
  95. return
  96. }
  97. if existingNeedle.Cookie != n.Cookie {
  98. glog.V(0).Infof("write cookie mismatch: existing %x, new %x", existingNeedle.Cookie, n.Cookie)
  99. err = fmt.Errorf("mismatching cookie %x", n.Cookie)
  100. return
  101. }
  102. }
  103. // append to dat file
  104. n.AppendAtNs = uint64(time.Now().UnixNano())
  105. if offset, size, _, err = n.Append(v.DataBackend, v.Version()); err != nil {
  106. return
  107. }
  108. v.lastAppendAtNs = n.AppendAtNs
  109. // add to needle map
  110. if !ok || uint64(nv.Offset.ToAcutalOffset()) < offset {
  111. if err = v.nm.Put(n.Id, ToOffset(int64(offset)), n.Size); err != nil {
  112. glog.V(4).Infof("failed to save in needle map %d: %v", n.Id, err)
  113. }
  114. }
  115. if v.lastModifiedTsSeconds < n.LastModified {
  116. v.lastModifiedTsSeconds = n.LastModified
  117. }
  118. return
  119. }
  120. func (v *Volume) writeNeedle2(n *needle.Needle, fsync bool) (offset uint64, size Size, isUnchanged bool, err error) {
  121. // glog.V(4).Infof("writing needle %s", needle.NewFileIdFromNeedle(v.Id, n).String())
  122. if n.Ttl == needle.EMPTY_TTL && v.Ttl != needle.EMPTY_TTL {
  123. n.SetHasTtl()
  124. n.Ttl = v.Ttl
  125. }
  126. if !fsync {
  127. return v.syncWrite(n)
  128. } else {
  129. asyncRequest := needle.NewAsyncRequest(n, true)
  130. // using len(n.Data) here instead of n.Size before n.Size is populated in n.Append()
  131. asyncRequest.ActualSize = needle.GetActualSize(Size(len(n.Data)), v.Version())
  132. v.asyncRequestAppend(asyncRequest)
  133. offset, _, isUnchanged, err = asyncRequest.WaitComplete()
  134. return
  135. }
  136. }
  137. func (v *Volume) doWriteRequest(n *needle.Needle) (offset uint64, size Size, isUnchanged bool, err error) {
  138. // glog.V(4).Infof("writing needle %s", needle.NewFileIdFromNeedle(v.Id, n).String())
  139. if v.isFileUnchanged(n) {
  140. size = Size(n.DataSize)
  141. isUnchanged = true
  142. return
  143. }
  144. // check whether existing needle cookie matches
  145. nv, ok := v.nm.Get(n.Id)
  146. if ok {
  147. existingNeedle, _, _, existingNeedleReadErr := needle.ReadNeedleHeader(v.DataBackend, v.Version(), nv.Offset.ToAcutalOffset())
  148. if existingNeedleReadErr != nil {
  149. err = fmt.Errorf("reading existing needle: %v", existingNeedleReadErr)
  150. return
  151. }
  152. if existingNeedle.Cookie != n.Cookie {
  153. glog.V(0).Infof("write cookie mismatch: existing %x, new %x", existingNeedle.Cookie, n.Cookie)
  154. err = fmt.Errorf("mismatching cookie %x", n.Cookie)
  155. return
  156. }
  157. }
  158. // append to dat file
  159. n.AppendAtNs = uint64(time.Now().UnixNano())
  160. if offset, size, _, err = n.Append(v.DataBackend, v.Version()); err != nil {
  161. return
  162. }
  163. v.lastAppendAtNs = n.AppendAtNs
  164. // add to needle map
  165. if !ok || uint64(nv.Offset.ToAcutalOffset()) < offset {
  166. if err = v.nm.Put(n.Id, ToOffset(int64(offset)), n.Size); err != nil {
  167. glog.V(4).Infof("failed to save in needle map %d: %v", n.Id, err)
  168. }
  169. }
  170. if v.lastModifiedTsSeconds < n.LastModified {
  171. v.lastModifiedTsSeconds = n.LastModified
  172. }
  173. return
  174. }
  175. func (v *Volume) syncDelete(n *needle.Needle) (Size, error) {
  176. // glog.V(4).Infof("delete needle %s", needle.NewFileIdFromNeedle(v.Id, n).String())
  177. actualSize := needle.GetActualSize(0, v.Version())
  178. v.dataFileAccessLock.Lock()
  179. defer v.dataFileAccessLock.Unlock()
  180. if MaxPossibleVolumeSize < v.nm.ContentSize()+uint64(actualSize) {
  181. err := fmt.Errorf("volume size limit %d exceeded! current size is %d", MaxPossibleVolumeSize, v.nm.ContentSize())
  182. return 0, err
  183. }
  184. nv, ok := v.nm.Get(n.Id)
  185. // fmt.Println("key", n.Id, "volume offset", nv.Offset, "data_size", n.Size, "cached size", nv.Size)
  186. if ok && nv.Size.IsValid() {
  187. size := nv.Size
  188. n.Data = nil
  189. n.AppendAtNs = uint64(time.Now().UnixNano())
  190. offset, _, _, err := n.Append(v.DataBackend, v.Version())
  191. if err != nil {
  192. return size, err
  193. }
  194. v.lastAppendAtNs = n.AppendAtNs
  195. if err = v.nm.Delete(n.Id, ToOffset(int64(offset))); err != nil {
  196. return size, err
  197. }
  198. return size, err
  199. }
  200. return 0, nil
  201. }
  202. func (v *Volume) deleteNeedle2(n *needle.Needle) (Size, error) {
  203. // todo: delete info is always appended no fsync, it may need fsync in future
  204. fsync := false
  205. if !fsync {
  206. return v.syncDelete(n)
  207. } else {
  208. asyncRequest := needle.NewAsyncRequest(n, false)
  209. asyncRequest.ActualSize = needle.GetActualSize(0, v.Version())
  210. v.asyncRequestAppend(asyncRequest)
  211. _, size, _, err := asyncRequest.WaitComplete()
  212. return Size(size), err
  213. }
  214. }
  215. func (v *Volume) doDeleteRequest(n *needle.Needle) (Size, error) {
  216. glog.V(4).Infof("delete needle %s", needle.NewFileIdFromNeedle(v.Id, n).String())
  217. nv, ok := v.nm.Get(n.Id)
  218. // fmt.Println("key", n.Id, "volume offset", nv.Offset, "data_size", n.Size, "cached size", nv.Size)
  219. if ok && nv.Size.IsValid() {
  220. size := nv.Size
  221. n.Data = nil
  222. n.AppendAtNs = uint64(time.Now().UnixNano())
  223. offset, _, _, err := n.Append(v.DataBackend, v.Version())
  224. if err != nil {
  225. return size, err
  226. }
  227. v.lastAppendAtNs = n.AppendAtNs
  228. if err = v.nm.Delete(n.Id, ToOffset(int64(offset))); err != nil {
  229. return size, err
  230. }
  231. return size, err
  232. }
  233. return 0, nil
  234. }
  235. // read fills in Needle content by looking up n.Id from NeedleMapper
  236. func (v *Volume) readNeedle(n *needle.Needle, readOption *ReadOption) (int, error) {
  237. v.dataFileAccessLock.RLock()
  238. defer v.dataFileAccessLock.RUnlock()
  239. nv, ok := v.nm.Get(n.Id)
  240. if !ok || nv.Offset.IsZero() {
  241. return -1, ErrorNotFound
  242. }
  243. readSize := nv.Size
  244. if readSize.IsDeleted() {
  245. if readOption != nil && readOption.ReadDeleted && readSize != TombstoneFileSize {
  246. glog.V(3).Infof("reading deleted %s", n.String())
  247. readSize = -readSize
  248. } else {
  249. return -1, ErrorDeleted
  250. }
  251. }
  252. if readSize == 0 {
  253. return 0, nil
  254. }
  255. err := n.ReadData(v.DataBackend, nv.Offset.ToAcutalOffset(), readSize, v.Version())
  256. if err == needle.ErrorSizeMismatch && OffsetSize == 4 {
  257. err = n.ReadData(v.DataBackend, nv.Offset.ToAcutalOffset()+int64(MaxPossibleVolumeSize), readSize, v.Version())
  258. }
  259. if err != nil {
  260. return 0, err
  261. }
  262. bytesRead := len(n.Data)
  263. if !n.HasTtl() {
  264. return bytesRead, nil
  265. }
  266. ttlMinutes := n.Ttl.Minutes()
  267. if ttlMinutes == 0 {
  268. return bytesRead, nil
  269. }
  270. if !n.HasLastModifiedDate() {
  271. return bytesRead, nil
  272. }
  273. if uint64(time.Now().Unix()) < n.LastModified+uint64(ttlMinutes*60) {
  274. return bytesRead, nil
  275. }
  276. return -1, ErrorNotFound
  277. }
  278. func (v *Volume) startWorker() {
  279. go func() {
  280. chanClosed := false
  281. for {
  282. // chan closed. go thread will exit
  283. if chanClosed {
  284. break
  285. }
  286. currentRequests := make([]*needle.AsyncRequest, 0, 128)
  287. currentBytesToWrite := int64(0)
  288. for {
  289. request, ok := <-v.asyncRequestsChan
  290. // volume may be closed
  291. if !ok {
  292. chanClosed = true
  293. break
  294. }
  295. if MaxPossibleVolumeSize < v.ContentSize()+uint64(currentBytesToWrite+request.ActualSize) {
  296. request.Complete(0, 0, false,
  297. fmt.Errorf("volume size limit %d exceeded! current size is %d", MaxPossibleVolumeSize, v.ContentSize()))
  298. break
  299. }
  300. currentRequests = append(currentRequests, request)
  301. currentBytesToWrite += request.ActualSize
  302. // submit at most 4M bytes or 128 requests at one time to decrease request delay.
  303. // it also need to break if there is no data in channel to avoid io hang.
  304. if currentBytesToWrite >= 4*1024*1024 || len(currentRequests) >= 128 || len(v.asyncRequestsChan) == 0 {
  305. break
  306. }
  307. }
  308. if len(currentRequests) == 0 {
  309. continue
  310. }
  311. v.dataFileAccessLock.Lock()
  312. end, _, e := v.DataBackend.GetStat()
  313. if e != nil {
  314. for i := 0; i < len(currentRequests); i++ {
  315. currentRequests[i].Complete(0, 0, false,
  316. fmt.Errorf("cannot read current volume position: %v", e))
  317. }
  318. v.dataFileAccessLock.Unlock()
  319. continue
  320. }
  321. for i := 0; i < len(currentRequests); i++ {
  322. if currentRequests[i].IsWriteRequest {
  323. offset, size, isUnchanged, err := v.doWriteRequest(currentRequests[i].N)
  324. currentRequests[i].UpdateResult(offset, uint64(size), isUnchanged, err)
  325. } else {
  326. size, err := v.doDeleteRequest(currentRequests[i].N)
  327. currentRequests[i].UpdateResult(0, uint64(size), false, err)
  328. }
  329. }
  330. // if sync error, data is not reliable, we should mark the completed request as fail and rollback
  331. if err := v.DataBackend.Sync(); err != nil {
  332. // todo: this may generate dirty data or cause data inconsistent, may be weed need to panic?
  333. if te := v.DataBackend.Truncate(end); te != nil {
  334. glog.V(0).Infof("Failed to truncate %s back to %d with error: %v", v.DataBackend.Name(), end, te)
  335. }
  336. for i := 0; i < len(currentRequests); i++ {
  337. if currentRequests[i].IsSucceed() {
  338. currentRequests[i].UpdateResult(0, 0, false, err)
  339. }
  340. }
  341. }
  342. for i := 0; i < len(currentRequests); i++ {
  343. currentRequests[i].Submit()
  344. }
  345. v.dataFileAccessLock.Unlock()
  346. }
  347. }()
  348. }
  349. type VolumeFileScanner interface {
  350. VisitSuperBlock(super_block.SuperBlock) error
  351. ReadNeedleBody() bool
  352. VisitNeedle(n *needle.Needle, offset int64, needleHeader, needleBody []byte) error
  353. }
  354. func ScanVolumeFile(dirname string, collection string, id needle.VolumeId,
  355. needleMapKind NeedleMapType,
  356. volumeFileScanner VolumeFileScanner) (err error) {
  357. var v *Volume
  358. if v, err = loadVolumeWithoutIndex(dirname, collection, id, needleMapKind); err != nil {
  359. return fmt.Errorf("failed to load volume %d: %v", id, err)
  360. }
  361. if err = volumeFileScanner.VisitSuperBlock(v.SuperBlock); err != nil {
  362. return fmt.Errorf("failed to process volume %d super block: %v", id, err)
  363. }
  364. defer v.Close()
  365. version := v.Version()
  366. offset := int64(v.SuperBlock.BlockSize())
  367. return ScanVolumeFileFrom(version, v.DataBackend, offset, volumeFileScanner)
  368. }
  369. func ScanVolumeFileFrom(version needle.Version, datBackend backend.BackendStorageFile, offset int64, volumeFileScanner VolumeFileScanner) (err error) {
  370. n, nh, rest, e := needle.ReadNeedleHeader(datBackend, version, offset)
  371. if e != nil {
  372. if e == io.EOF {
  373. return nil
  374. }
  375. return fmt.Errorf("cannot read %s at offset %d: %v", datBackend.Name(), offset, e)
  376. }
  377. for n != nil {
  378. var needleBody []byte
  379. if volumeFileScanner.ReadNeedleBody() {
  380. // println("needle", n.Id.String(), "offset", offset, "size", n.Size, "rest", rest)
  381. if needleBody, err = n.ReadNeedleBody(datBackend, version, offset+NeedleHeaderSize, rest); err != nil {
  382. glog.V(0).Infof("cannot read needle head [%d, %d) body [%d, %d) body length %d: %v", offset, offset+NeedleHeaderSize, offset+NeedleHeaderSize, offset+NeedleHeaderSize+rest, rest, err)
  383. // err = fmt.Errorf("cannot read needle body: %v", err)
  384. // return
  385. }
  386. }
  387. err := volumeFileScanner.VisitNeedle(n, offset, nh, needleBody)
  388. if err == io.EOF {
  389. return nil
  390. }
  391. if err != nil {
  392. glog.V(0).Infof("visit needle error: %v", err)
  393. return fmt.Errorf("visit needle error: %v", err)
  394. }
  395. offset += NeedleHeaderSize + rest
  396. glog.V(4).Infof("==> new entry offset %d", offset)
  397. if n, nh, rest, err = needle.ReadNeedleHeader(datBackend, version, offset); err != nil {
  398. if err == io.EOF {
  399. return nil
  400. }
  401. return fmt.Errorf("cannot read needle header at offset %d: %v", offset, err)
  402. }
  403. glog.V(4).Infof("new entry needle size:%d rest:%d", n.Size, rest)
  404. }
  405. return nil
  406. }