volume_read_write.go 14 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443
  1. package storage
  2. import (
  3. "bytes"
  4. "errors"
  5. "fmt"
  6. "io"
  7. "os"
  8. "time"
  9. "github.com/chrislusf/seaweedfs/weed/util/log"
  10. "github.com/chrislusf/seaweedfs/weed/storage/backend"
  11. "github.com/chrislusf/seaweedfs/weed/storage/needle"
  12. "github.com/chrislusf/seaweedfs/weed/storage/super_block"
  13. . "github.com/chrislusf/seaweedfs/weed/storage/types"
  14. )
  15. var ErrorNotFound = errors.New("not found")
  16. var ErrorDeleted = errors.New("already deleted")
  17. var ErrorSizeMismatch = errors.New("size mismatch")
  18. // isFileUnchanged checks whether this needle to write is same as last one.
  19. // It requires serialized access in the same volume.
  20. func (v *Volume) isFileUnchanged(n *needle.Needle) bool {
  21. if v.Ttl.String() != "" {
  22. return false
  23. }
  24. nv, ok := v.nm.Get(n.Id)
  25. if ok && !nv.Offset.IsZero() && nv.Size.IsValid() {
  26. oldNeedle := new(needle.Needle)
  27. err := oldNeedle.ReadData(v.DataBackend, nv.Offset.ToAcutalOffset(), nv.Size, v.Version())
  28. if err != nil {
  29. log.Infof("Failed to check updated file at offset %d size %d: %v", nv.Offset.ToAcutalOffset(), nv.Size, err)
  30. return false
  31. }
  32. if oldNeedle.Cookie == n.Cookie && oldNeedle.Checksum == n.Checksum && bytes.Equal(oldNeedle.Data, n.Data) {
  33. n.DataSize = oldNeedle.DataSize
  34. return true
  35. }
  36. }
  37. return false
  38. }
  39. // Destroy removes everything related to this volume
  40. func (v *Volume) Destroy() (err error) {
  41. if v.isCompacting {
  42. err = fmt.Errorf("volume %d is compacting", v.Id)
  43. return
  44. }
  45. close(v.asyncRequestsChan)
  46. storageName, storageKey := v.RemoteStorageNameKey()
  47. if v.HasRemoteFile() && storageName != "" && storageKey != "" {
  48. if backendStorage, found := backend.BackendStorages[storageName]; found {
  49. backendStorage.DeleteFile(storageKey)
  50. }
  51. }
  52. v.Close()
  53. removeVolumeFiles(v.FileName())
  54. return
  55. }
  56. func removeVolumeFiles(filename string) {
  57. os.Remove(filename + ".dat")
  58. os.Remove(filename + ".idx")
  59. os.Remove(filename + ".vif")
  60. os.Remove(filename + ".sdx")
  61. os.Remove(filename + ".cpd")
  62. os.Remove(filename + ".cpx")
  63. os.RemoveAll(filename + ".ldb")
  64. os.Remove(filename + ".note")
  65. }
  66. func (v *Volume) asyncRequestAppend(request *needle.AsyncRequest) {
  67. v.asyncRequestsChan <- request
  68. }
  69. func (v *Volume) syncWrite(n *needle.Needle) (offset uint64, size Size, isUnchanged bool, err error) {
  70. // log.Tracef("writing needle %s", needle.NewFileIdFromNeedle(v.Id, n).String())
  71. actualSize := needle.GetActualSize(Size(len(n.Data)), v.Version())
  72. v.dataFileAccessLock.Lock()
  73. defer v.dataFileAccessLock.Unlock()
  74. if MaxPossibleVolumeSize < v.nm.ContentSize()+uint64(actualSize) {
  75. err = fmt.Errorf("volume size limit %d exceeded! current size is %d", MaxPossibleVolumeSize, v.nm.ContentSize())
  76. return
  77. }
  78. if v.isFileUnchanged(n) {
  79. size = Size(n.DataSize)
  80. isUnchanged = true
  81. return
  82. }
  83. // check whether existing needle cookie matches
  84. nv, ok := v.nm.Get(n.Id)
  85. if ok {
  86. existingNeedle, _, _, existingNeedleReadErr := needle.ReadNeedleHeader(v.DataBackend, v.Version(), nv.Offset.ToAcutalOffset())
  87. if existingNeedleReadErr != nil {
  88. err = fmt.Errorf("reading existing needle: %v", existingNeedleReadErr)
  89. return
  90. }
  91. if existingNeedle.Cookie != n.Cookie {
  92. log.Infof("write cookie mismatch: existing %x, new %x", existingNeedle.Cookie, n.Cookie)
  93. err = fmt.Errorf("mismatching cookie %x", n.Cookie)
  94. return
  95. }
  96. }
  97. // append to dat file
  98. n.AppendAtNs = uint64(time.Now().UnixNano())
  99. if offset, size, _, err = n.Append(v.DataBackend, v.Version()); err != nil {
  100. return
  101. }
  102. v.lastAppendAtNs = n.AppendAtNs
  103. // add to needle map
  104. if !ok || uint64(nv.Offset.ToAcutalOffset()) < offset {
  105. if err = v.nm.Put(n.Id, ToOffset(int64(offset)), n.Size); err != nil {
  106. log.Tracef("failed to save in needle map %d: %v", n.Id, err)
  107. }
  108. }
  109. if v.lastModifiedTsSeconds < n.LastModified {
  110. v.lastModifiedTsSeconds = n.LastModified
  111. }
  112. return
  113. }
  114. func (v *Volume) writeNeedle2(n *needle.Needle, fsync bool) (offset uint64, size Size, isUnchanged bool, err error) {
  115. // log.Tracef("writing needle %s", needle.NewFileIdFromNeedle(v.Id, n).String())
  116. if n.Ttl == needle.EMPTY_TTL && v.Ttl != needle.EMPTY_TTL {
  117. n.SetHasTtl()
  118. n.Ttl = v.Ttl
  119. }
  120. if !fsync {
  121. return v.syncWrite(n)
  122. } else {
  123. asyncRequest := needle.NewAsyncRequest(n, true)
  124. // using len(n.Data) here instead of n.Size before n.Size is populated in n.Append()
  125. asyncRequest.ActualSize = needle.GetActualSize(Size(len(n.Data)), v.Version())
  126. v.asyncRequestAppend(asyncRequest)
  127. offset, _, isUnchanged, err = asyncRequest.WaitComplete()
  128. return
  129. }
  130. }
  131. func (v *Volume) doWriteRequest(n *needle.Needle) (offset uint64, size Size, isUnchanged bool, err error) {
  132. // log.Tracef("writing needle %s", needle.NewFileIdFromNeedle(v.Id, n).String())
  133. if v.isFileUnchanged(n) {
  134. size = Size(n.DataSize)
  135. isUnchanged = true
  136. return
  137. }
  138. // check whether existing needle cookie matches
  139. nv, ok := v.nm.Get(n.Id)
  140. if ok {
  141. existingNeedle, _, _, existingNeedleReadErr := needle.ReadNeedleHeader(v.DataBackend, v.Version(), nv.Offset.ToAcutalOffset())
  142. if existingNeedleReadErr != nil {
  143. err = fmt.Errorf("reading existing needle: %v", existingNeedleReadErr)
  144. return
  145. }
  146. if existingNeedle.Cookie != n.Cookie {
  147. log.Infof("write cookie mismatch: existing %x, new %x", existingNeedle.Cookie, n.Cookie)
  148. err = fmt.Errorf("mismatching cookie %x", n.Cookie)
  149. return
  150. }
  151. }
  152. // append to dat file
  153. n.AppendAtNs = uint64(time.Now().UnixNano())
  154. if offset, size, _, err = n.Append(v.DataBackend, v.Version()); err != nil {
  155. return
  156. }
  157. v.lastAppendAtNs = n.AppendAtNs
  158. // add to needle map
  159. if !ok || uint64(nv.Offset.ToAcutalOffset()) < offset {
  160. if err = v.nm.Put(n.Id, ToOffset(int64(offset)), n.Size); err != nil {
  161. log.Tracef("failed to save in needle map %d: %v", n.Id, err)
  162. }
  163. }
  164. if v.lastModifiedTsSeconds < n.LastModified {
  165. v.lastModifiedTsSeconds = n.LastModified
  166. }
  167. return
  168. }
  169. func (v *Volume) syncDelete(n *needle.Needle) (Size, error) {
  170. // log.Tracef("delete needle %s", needle.NewFileIdFromNeedle(v.Id, n).String())
  171. actualSize := needle.GetActualSize(0, v.Version())
  172. v.dataFileAccessLock.Lock()
  173. defer v.dataFileAccessLock.Unlock()
  174. if MaxPossibleVolumeSize < v.nm.ContentSize()+uint64(actualSize) {
  175. err := fmt.Errorf("volume size limit %d exceeded! current size is %d", MaxPossibleVolumeSize, v.nm.ContentSize())
  176. return 0, err
  177. }
  178. nv, ok := v.nm.Get(n.Id)
  179. // fmt.Println("key", n.Id, "volume offset", nv.Offset, "data_size", n.Size, "cached size", nv.Size)
  180. if ok && nv.Size.IsValid() {
  181. size := nv.Size
  182. n.Data = nil
  183. n.AppendAtNs = uint64(time.Now().UnixNano())
  184. offset, _, _, err := n.Append(v.DataBackend, v.Version())
  185. if err != nil {
  186. return size, err
  187. }
  188. v.lastAppendAtNs = n.AppendAtNs
  189. if err = v.nm.Delete(n.Id, ToOffset(int64(offset))); err != nil {
  190. return size, err
  191. }
  192. return size, err
  193. }
  194. return 0, nil
  195. }
  196. func (v *Volume) deleteNeedle2(n *needle.Needle) (Size, error) {
  197. // todo: delete info is always appended no fsync, it may need fsync in future
  198. fsync := false
  199. if !fsync {
  200. return v.syncDelete(n)
  201. } else {
  202. asyncRequest := needle.NewAsyncRequest(n, false)
  203. asyncRequest.ActualSize = needle.GetActualSize(0, v.Version())
  204. v.asyncRequestAppend(asyncRequest)
  205. _, size, _, err := asyncRequest.WaitComplete()
  206. return Size(size), err
  207. }
  208. }
  209. func (v *Volume) doDeleteRequest(n *needle.Needle) (Size, error) {
  210. log.Tracef("delete needle %s", needle.NewFileIdFromNeedle(v.Id, n).String())
  211. nv, ok := v.nm.Get(n.Id)
  212. // fmt.Println("key", n.Id, "volume offset", nv.Offset, "data_size", n.Size, "cached size", nv.Size)
  213. if ok && nv.Size.IsValid() {
  214. size := nv.Size
  215. n.Data = nil
  216. n.AppendAtNs = uint64(time.Now().UnixNano())
  217. offset, _, _, err := n.Append(v.DataBackend, v.Version())
  218. if err != nil {
  219. return size, err
  220. }
  221. v.lastAppendAtNs = n.AppendAtNs
  222. if err = v.nm.Delete(n.Id, ToOffset(int64(offset))); err != nil {
  223. return size, err
  224. }
  225. return size, err
  226. }
  227. return 0, nil
  228. }
  229. // read fills in Needle content by looking up n.Id from NeedleMapper
  230. func (v *Volume) readNeedle(n *needle.Needle, readOption *ReadOption) (int, error) {
  231. v.dataFileAccessLock.RLock()
  232. defer v.dataFileAccessLock.RUnlock()
  233. nv, ok := v.nm.Get(n.Id)
  234. if !ok || nv.Offset.IsZero() {
  235. return -1, ErrorNotFound
  236. }
  237. readSize := nv.Size
  238. if readSize.IsDeleted() {
  239. if readOption != nil && readOption.ReadDeleted && readSize != TombstoneFileSize {
  240. log.Tracef("reading deleted %s", n.String())
  241. readSize = -readSize
  242. } else {
  243. return -1, ErrorDeleted
  244. }
  245. }
  246. if readSize == 0 {
  247. return 0, nil
  248. }
  249. err := n.ReadData(v.DataBackend, nv.Offset.ToAcutalOffset(), readSize, v.Version())
  250. if err == needle.ErrorSizeMismatch && OffsetSize == 4 {
  251. err = n.ReadData(v.DataBackend, nv.Offset.ToAcutalOffset()+int64(MaxPossibleVolumeSize), readSize, v.Version())
  252. }
  253. if err != nil {
  254. return 0, err
  255. }
  256. bytesRead := len(n.Data)
  257. if !n.HasTtl() {
  258. return bytesRead, nil
  259. }
  260. ttlMinutes := n.Ttl.Minutes()
  261. if ttlMinutes == 0 {
  262. return bytesRead, nil
  263. }
  264. if !n.HasLastModifiedDate() {
  265. return bytesRead, nil
  266. }
  267. if uint64(time.Now().Unix()) < n.LastModified+uint64(ttlMinutes*60) {
  268. return bytesRead, nil
  269. }
  270. return -1, ErrorNotFound
  271. }
  272. func (v *Volume) startWorker() {
  273. go func() {
  274. chanClosed := false
  275. for {
  276. // chan closed. go thread will exit
  277. if chanClosed {
  278. break
  279. }
  280. currentRequests := make([]*needle.AsyncRequest, 0, 128)
  281. currentBytesToWrite := int64(0)
  282. for {
  283. request, ok := <-v.asyncRequestsChan
  284. // volume may be closed
  285. if !ok {
  286. chanClosed = true
  287. break
  288. }
  289. if MaxPossibleVolumeSize < v.ContentSize()+uint64(currentBytesToWrite+request.ActualSize) {
  290. request.Complete(0, 0, false,
  291. fmt.Errorf("volume size limit %d exceeded! current size is %d", MaxPossibleVolumeSize, v.ContentSize()))
  292. break
  293. }
  294. currentRequests = append(currentRequests, request)
  295. currentBytesToWrite += request.ActualSize
  296. // submit at most 4M bytes or 128 requests at one time to decrease request delay.
  297. // it also need to break if there is no data in channel to avoid io hang.
  298. if currentBytesToWrite >= 4*1024*1024 || len(currentRequests) >= 128 || len(v.asyncRequestsChan) == 0 {
  299. break
  300. }
  301. }
  302. if len(currentRequests) == 0 {
  303. continue
  304. }
  305. v.dataFileAccessLock.Lock()
  306. end, _, e := v.DataBackend.GetStat()
  307. if e != nil {
  308. for i := 0; i < len(currentRequests); i++ {
  309. currentRequests[i].Complete(0, 0, false,
  310. fmt.Errorf("cannot read current volume position: %v", e))
  311. }
  312. v.dataFileAccessLock.Unlock()
  313. continue
  314. }
  315. for i := 0; i < len(currentRequests); i++ {
  316. if currentRequests[i].IsWriteRequest {
  317. offset, size, isUnchanged, err := v.doWriteRequest(currentRequests[i].N)
  318. currentRequests[i].UpdateResult(offset, uint64(size), isUnchanged, err)
  319. } else {
  320. size, err := v.doDeleteRequest(currentRequests[i].N)
  321. currentRequests[i].UpdateResult(0, uint64(size), false, err)
  322. }
  323. }
  324. // if sync error, data is not reliable, we should mark the completed request as fail and rollback
  325. if err := v.DataBackend.Sync(); err != nil {
  326. // todo: this may generate dirty data or cause data inconsistent, may be weed need to panic?
  327. if te := v.DataBackend.Truncate(end); te != nil {
  328. log.Infof("Failed to truncate %s back to %d with error: %v", v.DataBackend.Name(), end, te)
  329. }
  330. for i := 0; i < len(currentRequests); i++ {
  331. if currentRequests[i].IsSucceed() {
  332. currentRequests[i].UpdateResult(0, 0, false, err)
  333. }
  334. }
  335. }
  336. for i := 0; i < len(currentRequests); i++ {
  337. currentRequests[i].Submit()
  338. }
  339. v.dataFileAccessLock.Unlock()
  340. }
  341. }()
  342. }
  343. type VolumeFileScanner interface {
  344. VisitSuperBlock(super_block.SuperBlock) error
  345. ReadNeedleBody() bool
  346. VisitNeedle(n *needle.Needle, offset int64, needleHeader, needleBody []byte) error
  347. }
  348. func ScanVolumeFile(dirname string, collection string, id needle.VolumeId,
  349. needleMapKind NeedleMapType,
  350. volumeFileScanner VolumeFileScanner) (err error) {
  351. var v *Volume
  352. if v, err = loadVolumeWithoutIndex(dirname, collection, id, needleMapKind); err != nil {
  353. return fmt.Errorf("failed to load volume %d: %v", id, err)
  354. }
  355. if err = volumeFileScanner.VisitSuperBlock(v.SuperBlock); err != nil {
  356. return fmt.Errorf("failed to process volume %d super block: %v", id, err)
  357. }
  358. defer v.Close()
  359. version := v.Version()
  360. offset := int64(v.SuperBlock.BlockSize())
  361. return ScanVolumeFileFrom(version, v.DataBackend, offset, volumeFileScanner)
  362. }
  363. func ScanVolumeFileFrom(version needle.Version, datBackend backend.BackendStorageFile, offset int64, volumeFileScanner VolumeFileScanner) (err error) {
  364. n, nh, rest, e := needle.ReadNeedleHeader(datBackend, version, offset)
  365. if e != nil {
  366. if e == io.EOF {
  367. return nil
  368. }
  369. return fmt.Errorf("cannot read %s at offset %d: %v", datBackend.Name(), offset, e)
  370. }
  371. for n != nil {
  372. var needleBody []byte
  373. if volumeFileScanner.ReadNeedleBody() {
  374. // println("needle", n.Id.String(), "offset", offset, "size", n.Size, "rest", rest)
  375. if needleBody, err = n.ReadNeedleBody(datBackend, version, offset+NeedleHeaderSize, rest); err != nil {
  376. log.Infof("cannot read needle head [%d, %d) body [%d, %d) body length %d: %v", offset, offset+NeedleHeaderSize, offset+NeedleHeaderSize, offset+NeedleHeaderSize+rest, rest, err)
  377. // err = fmt.Errorf("cannot read needle body: %v", err)
  378. // return
  379. }
  380. }
  381. err := volumeFileScanner.VisitNeedle(n, offset, nh, needleBody)
  382. if err == io.EOF {
  383. return nil
  384. }
  385. if err != nil {
  386. log.Infof("visit needle error: %v", err)
  387. return fmt.Errorf("visit needle error: %v", err)
  388. }
  389. offset += NeedleHeaderSize + rest
  390. log.Tracef("==> new entry offset %d", offset)
  391. if n, nh, rest, err = needle.ReadNeedleHeader(datBackend, version, offset); err != nil {
  392. if err == io.EOF {
  393. return nil
  394. }
  395. return fmt.Errorf("cannot read needle header at offset %d: %v", offset, err)
  396. }
  397. log.Tracef("new entry needle size:%d rest:%d", n.Size, rest)
  398. }
  399. return nil
  400. }