You can not select more than 25 topics Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.

434 lines
13 KiB

6 years ago
4 years ago
6 years ago
5 years ago
4 years ago
4 years ago
6 years ago
5 years ago
  1. package storage
  2. import (
  3. "bytes"
  4. "errors"
  5. "fmt"
  6. "io"
  7. "os"
  8. "time"
  9. "github.com/chrislusf/seaweedfs/weed/glog"
  10. "github.com/chrislusf/seaweedfs/weed/storage/backend"
  11. "github.com/chrislusf/seaweedfs/weed/storage/needle"
  12. "github.com/chrislusf/seaweedfs/weed/storage/super_block"
  13. . "github.com/chrislusf/seaweedfs/weed/storage/types"
  14. )
  15. var ErrorNotFound = errors.New("not found")
  16. var ErrorDeleted = errors.New("already deleted")
  17. // isFileUnchanged checks whether this needle to write is same as last one.
  18. // It requires serialized access in the same volume.
  19. func (v *Volume) isFileUnchanged(n *needle.Needle) bool {
  20. if v.Ttl.String() != "" {
  21. return false
  22. }
  23. nv, ok := v.nm.Get(n.Id)
  24. if ok && !nv.Offset.IsZero() && nv.Size.IsValid() {
  25. oldNeedle := new(needle.Needle)
  26. err := oldNeedle.ReadData(v.DataBackend, nv.Offset.ToAcutalOffset(), nv.Size, v.Version())
  27. if err != nil {
  28. glog.V(0).Infof("Failed to check updated file at offset %d size %d: %v", nv.Offset.ToAcutalOffset(), nv.Size, err)
  29. return false
  30. }
  31. if oldNeedle.Cookie == n.Cookie && oldNeedle.Checksum == n.Checksum && bytes.Equal(oldNeedle.Data, n.Data) {
  32. n.DataSize = oldNeedle.DataSize
  33. return true
  34. }
  35. }
  36. return false
  37. }
  38. // Destroy removes everything related to this volume
  39. func (v *Volume) Destroy() (err error) {
  40. if v.isCompacting {
  41. err = fmt.Errorf("volume %d is compacting", v.Id)
  42. return
  43. }
  44. close(v.asyncRequestsChan)
  45. storageName, storageKey := v.RemoteStorageNameKey()
  46. if v.HasRemoteFile() && storageName != "" && storageKey != "" {
  47. if backendStorage, found := backend.BackendStorages[storageName]; found {
  48. backendStorage.DeleteFile(storageKey)
  49. }
  50. }
  51. v.Close()
  52. os.Remove(v.FileName() + ".dat")
  53. os.Remove(v.FileName() + ".idx")
  54. os.Remove(v.FileName() + ".vif")
  55. os.Remove(v.FileName() + ".sdx")
  56. os.Remove(v.FileName() + ".cpd")
  57. os.Remove(v.FileName() + ".cpx")
  58. os.RemoveAll(v.FileName() + ".ldb")
  59. return
  60. }
  61. func (v *Volume) asyncRequestAppend(request *needle.AsyncRequest) {
  62. v.asyncRequestsChan <- request
  63. }
  64. func (v *Volume) syncWrite(n *needle.Needle) (offset uint64, size Size, isUnchanged bool, err error) {
  65. // glog.V(4).Infof("writing needle %s", needle.NewFileIdFromNeedle(v.Id, n).String())
  66. actualSize := needle.GetActualSize(Size(len(n.Data)), v.Version())
  67. v.dataFileAccessLock.Lock()
  68. defer v.dataFileAccessLock.Unlock()
  69. if MaxPossibleVolumeSize < v.nm.ContentSize()+uint64(actualSize) {
  70. err = fmt.Errorf("volume size limit %d exceeded! current size is %d", MaxPossibleVolumeSize, v.nm.ContentSize())
  71. return
  72. }
  73. if v.isFileUnchanged(n) {
  74. size = Size(n.DataSize)
  75. isUnchanged = true
  76. return
  77. }
  78. // check whether existing needle cookie matches
  79. nv, ok := v.nm.Get(n.Id)
  80. if ok {
  81. existingNeedle, _, _, existingNeedleReadErr := needle.ReadNeedleHeader(v.DataBackend, v.Version(), nv.Offset.ToAcutalOffset())
  82. if existingNeedleReadErr != nil {
  83. err = fmt.Errorf("reading existing needle: %v", existingNeedleReadErr)
  84. return
  85. }
  86. if existingNeedle.Cookie != n.Cookie {
  87. glog.V(0).Infof("write cookie mismatch: existing %x, new %x", existingNeedle.Cookie, n.Cookie)
  88. err = fmt.Errorf("mismatching cookie %x", n.Cookie)
  89. return
  90. }
  91. }
  92. // append to dat file
  93. n.AppendAtNs = uint64(time.Now().UnixNano())
  94. if offset, size, _, err = n.Append(v.DataBackend, v.Version()); err != nil {
  95. return
  96. }
  97. v.lastAppendAtNs = n.AppendAtNs
  98. // add to needle map
  99. if !ok || uint64(nv.Offset.ToAcutalOffset()) < offset {
  100. if err = v.nm.Put(n.Id, ToOffset(int64(offset)), n.Size); err != nil {
  101. glog.V(4).Infof("failed to save in needle map %d: %v", n.Id, err)
  102. }
  103. }
  104. if v.lastModifiedTsSeconds < n.LastModified {
  105. v.lastModifiedTsSeconds = n.LastModified
  106. }
  107. return
  108. }
  109. func (v *Volume) writeNeedle2(n *needle.Needle, fsync bool) (offset uint64, size Size, isUnchanged bool, err error) {
  110. // glog.V(4).Infof("writing needle %s", needle.NewFileIdFromNeedle(v.Id, n).String())
  111. if n.Ttl == needle.EMPTY_TTL && v.Ttl != needle.EMPTY_TTL {
  112. n.SetHasTtl()
  113. n.Ttl = v.Ttl
  114. }
  115. if !fsync {
  116. return v.syncWrite(n)
  117. } else {
  118. asyncRequest := needle.NewAsyncRequest(n, true)
  119. // using len(n.Data) here instead of n.Size before n.Size is populated in n.Append()
  120. asyncRequest.ActualSize = needle.GetActualSize(Size(len(n.Data)), v.Version())
  121. v.asyncRequestAppend(asyncRequest)
  122. offset, _, isUnchanged, err = asyncRequest.WaitComplete()
  123. return
  124. }
  125. }
  126. func (v *Volume) doWriteRequest(n *needle.Needle) (offset uint64, size Size, isUnchanged bool, err error) {
  127. // glog.V(4).Infof("writing needle %s", needle.NewFileIdFromNeedle(v.Id, n).String())
  128. if v.isFileUnchanged(n) {
  129. size = Size(n.DataSize)
  130. isUnchanged = true
  131. return
  132. }
  133. // check whether existing needle cookie matches
  134. nv, ok := v.nm.Get(n.Id)
  135. if ok {
  136. existingNeedle, _, _, existingNeedleReadErr := needle.ReadNeedleHeader(v.DataBackend, v.Version(), nv.Offset.ToAcutalOffset())
  137. if existingNeedleReadErr != nil {
  138. err = fmt.Errorf("reading existing needle: %v", existingNeedleReadErr)
  139. return
  140. }
  141. if existingNeedle.Cookie != n.Cookie {
  142. glog.V(0).Infof("write cookie mismatch: existing %x, new %x", existingNeedle.Cookie, n.Cookie)
  143. err = fmt.Errorf("mismatching cookie %x", n.Cookie)
  144. return
  145. }
  146. }
  147. // append to dat file
  148. n.AppendAtNs = uint64(time.Now().UnixNano())
  149. if offset, size, _, err = n.Append(v.DataBackend, v.Version()); err != nil {
  150. return
  151. }
  152. v.lastAppendAtNs = n.AppendAtNs
  153. // add to needle map
  154. if !ok || uint64(nv.Offset.ToAcutalOffset()) < offset {
  155. if err = v.nm.Put(n.Id, ToOffset(int64(offset)), n.Size); err != nil {
  156. glog.V(4).Infof("failed to save in needle map %d: %v", n.Id, err)
  157. }
  158. }
  159. if v.lastModifiedTsSeconds < n.LastModified {
  160. v.lastModifiedTsSeconds = n.LastModified
  161. }
  162. return
  163. }
  164. func (v *Volume) syncDelete(n *needle.Needle) (Size, error) {
  165. // glog.V(4).Infof("delete needle %s", needle.NewFileIdFromNeedle(v.Id, n).String())
  166. actualSize := needle.GetActualSize(0, v.Version())
  167. v.dataFileAccessLock.Lock()
  168. defer v.dataFileAccessLock.Unlock()
  169. if MaxPossibleVolumeSize < v.nm.ContentSize()+uint64(actualSize) {
  170. err := fmt.Errorf("volume size limit %d exceeded! current size is %d", MaxPossibleVolumeSize, v.nm.ContentSize())
  171. return 0, err
  172. }
  173. nv, ok := v.nm.Get(n.Id)
  174. // fmt.Println("key", n.Id, "volume offset", nv.Offset, "data_size", n.Size, "cached size", nv.Size)
  175. if ok && nv.Size.IsValid() {
  176. size := nv.Size
  177. n.Data = nil
  178. n.AppendAtNs = uint64(time.Now().UnixNano())
  179. offset, _, _, err := n.Append(v.DataBackend, v.Version())
  180. if err != nil {
  181. return size, err
  182. }
  183. v.lastAppendAtNs = n.AppendAtNs
  184. if err = v.nm.Delete(n.Id, ToOffset(int64(offset))); err != nil {
  185. return size, err
  186. }
  187. return size, err
  188. }
  189. return 0, nil
  190. }
  191. func (v *Volume) deleteNeedle2(n *needle.Needle) (Size, error) {
  192. // todo: delete info is always appended no fsync, it may need fsync in future
  193. fsync := false
  194. if !fsync {
  195. return v.syncDelete(n)
  196. } else {
  197. asyncRequest := needle.NewAsyncRequest(n, false)
  198. asyncRequest.ActualSize = needle.GetActualSize(0, v.Version())
  199. v.asyncRequestAppend(asyncRequest)
  200. _, size, _, err := asyncRequest.WaitComplete()
  201. return Size(size), err
  202. }
  203. }
  204. func (v *Volume) doDeleteRequest(n *needle.Needle) (Size, error) {
  205. glog.V(4).Infof("delete needle %s", needle.NewFileIdFromNeedle(v.Id, n).String())
  206. nv, ok := v.nm.Get(n.Id)
  207. // fmt.Println("key", n.Id, "volume offset", nv.Offset, "data_size", n.Size, "cached size", nv.Size)
  208. if ok && nv.Size.IsValid() {
  209. size := nv.Size
  210. n.Data = nil
  211. n.AppendAtNs = uint64(time.Now().UnixNano())
  212. offset, _, _, err := n.Append(v.DataBackend, v.Version())
  213. if err != nil {
  214. return size, err
  215. }
  216. v.lastAppendAtNs = n.AppendAtNs
  217. if err = v.nm.Delete(n.Id, ToOffset(int64(offset))); err != nil {
  218. return size, err
  219. }
  220. return size, err
  221. }
  222. return 0, nil
  223. }
  224. // read fills in Needle content by looking up n.Id from NeedleMapper
  225. func (v *Volume) readNeedle(n *needle.Needle, readOption *ReadOption) (int, error) {
  226. v.dataFileAccessLock.RLock()
  227. defer v.dataFileAccessLock.RUnlock()
  228. nv, ok := v.nm.Get(n.Id)
  229. if !ok || nv.Offset.IsZero() {
  230. return -1, ErrorNotFound
  231. }
  232. readSize := nv.Size
  233. if readSize.IsDeleted() {
  234. if readOption != nil && readOption.ReadDeleted && readSize != TombstoneFileSize {
  235. glog.V(3).Infof("reading deleted %s", n.String())
  236. readSize = -readSize
  237. } else {
  238. return -1, ErrorDeleted
  239. }
  240. }
  241. if readSize == 0 {
  242. return 0, nil
  243. }
  244. err := n.ReadData(v.DataBackend, nv.Offset.ToAcutalOffset(), readSize, v.Version())
  245. if err != nil {
  246. return 0, err
  247. }
  248. bytesRead := len(n.Data)
  249. if !n.HasTtl() {
  250. return bytesRead, nil
  251. }
  252. ttlMinutes := n.Ttl.Minutes()
  253. if ttlMinutes == 0 {
  254. return bytesRead, nil
  255. }
  256. if !n.HasLastModifiedDate() {
  257. return bytesRead, nil
  258. }
  259. if uint64(time.Now().Unix()) < n.LastModified+uint64(ttlMinutes*60) {
  260. return bytesRead, nil
  261. }
  262. return -1, ErrorNotFound
  263. }
  264. func (v *Volume) startWorker() {
  265. go func() {
  266. chanClosed := false
  267. for {
  268. // chan closed. go thread will exit
  269. if chanClosed {
  270. break
  271. }
  272. currentRequests := make([]*needle.AsyncRequest, 0, 128)
  273. currentBytesToWrite := int64(0)
  274. for {
  275. request, ok := <-v.asyncRequestsChan
  276. // volume may be closed
  277. if !ok {
  278. chanClosed = true
  279. break
  280. }
  281. if MaxPossibleVolumeSize < v.ContentSize()+uint64(currentBytesToWrite+request.ActualSize) {
  282. request.Complete(0, 0, false,
  283. fmt.Errorf("volume size limit %d exceeded! current size is %d", MaxPossibleVolumeSize, v.ContentSize()))
  284. break
  285. }
  286. currentRequests = append(currentRequests, request)
  287. currentBytesToWrite += request.ActualSize
  288. // submit at most 4M bytes or 128 requests at one time to decrease request delay.
  289. // it also need to break if there is no data in channel to avoid io hang.
  290. if currentBytesToWrite >= 4*1024*1024 || len(currentRequests) >= 128 || len(v.asyncRequestsChan) == 0 {
  291. break
  292. }
  293. }
  294. if len(currentRequests) == 0 {
  295. continue
  296. }
  297. v.dataFileAccessLock.Lock()
  298. end, _, e := v.DataBackend.GetStat()
  299. if e != nil {
  300. for i := 0; i < len(currentRequests); i++ {
  301. currentRequests[i].Complete(0, 0, false,
  302. fmt.Errorf("cannot read current volume position: %v", e))
  303. }
  304. v.dataFileAccessLock.Unlock()
  305. continue
  306. }
  307. for i := 0; i < len(currentRequests); i++ {
  308. if currentRequests[i].IsWriteRequest {
  309. offset, size, isUnchanged, err := v.doWriteRequest(currentRequests[i].N)
  310. currentRequests[i].UpdateResult(offset, uint64(size), isUnchanged, err)
  311. } else {
  312. size, err := v.doDeleteRequest(currentRequests[i].N)
  313. currentRequests[i].UpdateResult(0, uint64(size), false, err)
  314. }
  315. }
  316. // if sync error, data is not reliable, we should mark the completed request as fail and rollback
  317. if err := v.DataBackend.Sync(); err != nil {
  318. // todo: this may generate dirty data or cause data inconsistent, may be weed need to panic?
  319. if te := v.DataBackend.Truncate(end); te != nil {
  320. glog.V(0).Infof("Failed to truncate %s back to %d with error: %v", v.DataBackend.Name(), end, te)
  321. }
  322. for i := 0; i < len(currentRequests); i++ {
  323. if currentRequests[i].IsSucceed() {
  324. currentRequests[i].UpdateResult(0, 0, false, err)
  325. }
  326. }
  327. }
  328. for i := 0; i < len(currentRequests); i++ {
  329. currentRequests[i].Submit()
  330. }
  331. v.dataFileAccessLock.Unlock()
  332. }
  333. }()
  334. }
  335. type VolumeFileScanner interface {
  336. VisitSuperBlock(super_block.SuperBlock) error
  337. ReadNeedleBody() bool
  338. VisitNeedle(n *needle.Needle, offset int64, needleHeader, needleBody []byte) error
  339. }
  340. func ScanVolumeFile(dirname string, collection string, id needle.VolumeId,
  341. needleMapKind NeedleMapType,
  342. volumeFileScanner VolumeFileScanner) (err error) {
  343. var v *Volume
  344. if v, err = loadVolumeWithoutIndex(dirname, collection, id, needleMapKind); err != nil {
  345. return fmt.Errorf("failed to load volume %d: %v", id, err)
  346. }
  347. if err = volumeFileScanner.VisitSuperBlock(v.SuperBlock); err != nil {
  348. return fmt.Errorf("failed to process volume %d super block: %v", id, err)
  349. }
  350. defer v.Close()
  351. version := v.Version()
  352. offset := int64(v.SuperBlock.BlockSize())
  353. return ScanVolumeFileFrom(version, v.DataBackend, offset, volumeFileScanner)
  354. }
  355. func ScanVolumeFileFrom(version needle.Version, datBackend backend.BackendStorageFile, offset int64, volumeFileScanner VolumeFileScanner) (err error) {
  356. n, nh, rest, e := needle.ReadNeedleHeader(datBackend, version, offset)
  357. if e != nil {
  358. if e == io.EOF {
  359. return nil
  360. }
  361. return fmt.Errorf("cannot read %s at offset %d: %v", datBackend.Name(), offset, e)
  362. }
  363. for n != nil {
  364. var needleBody []byte
  365. if volumeFileScanner.ReadNeedleBody() {
  366. // println("needle", n.Id.String(), "offset", offset, "size", n.Size, "rest", rest)
  367. if needleBody, err = n.ReadNeedleBody(datBackend, version, offset+NeedleHeaderSize, rest); err != nil {
  368. glog.V(0).Infof("cannot read needle head [%d, %d) body [%d, %d) body length %d: %v", offset, offset+NeedleHeaderSize, offset+NeedleHeaderSize, offset+NeedleHeaderSize+rest, rest, err)
  369. // err = fmt.Errorf("cannot read needle body: %v", err)
  370. // return
  371. }
  372. }
  373. err := volumeFileScanner.VisitNeedle(n, offset, nh, needleBody)
  374. if err == io.EOF {
  375. return nil
  376. }
  377. if err != nil {
  378. glog.V(0).Infof("visit needle error: %v", err)
  379. return fmt.Errorf("visit needle error: %v", err)
  380. }
  381. offset += NeedleHeaderSize + rest
  382. glog.V(4).Infof("==> new entry offset %d", offset)
  383. if n, nh, rest, err = needle.ReadNeedleHeader(datBackend, version, offset); err != nil {
  384. if err == io.EOF {
  385. return nil
  386. }
  387. return fmt.Errorf("cannot read needle header at offset %d: %v", offset, err)
  388. }
  389. glog.V(4).Infof("new entry needle size:%d rest:%d", n.Size, rest)
  390. }
  391. return nil
  392. }