volume_read_write.go 13 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428
  1. package storage
  2. import (
  3. "bytes"
  4. "errors"
  5. "fmt"
  6. "io"
  7. "os"
  8. "time"
  9. "github.com/chrislusf/seaweedfs/weed/glog"
  10. "github.com/chrislusf/seaweedfs/weed/storage/backend"
  11. "github.com/chrislusf/seaweedfs/weed/storage/needle"
  12. "github.com/chrislusf/seaweedfs/weed/storage/super_block"
  13. . "github.com/chrislusf/seaweedfs/weed/storage/types"
  14. )
  15. var ErrorNotFound = errors.New("not found")
  16. // isFileUnchanged checks whether this needle to write is same as last one.
  17. // It requires serialized access in the same volume.
  18. func (v *Volume) isFileUnchanged(n *needle.Needle) bool {
  19. if v.Ttl.String() != "" {
  20. return false
  21. }
  22. nv, ok := v.nm.Get(n.Id)
  23. if ok && !nv.Offset.IsZero() && nv.Size != TombstoneFileSize {
  24. oldNeedle := new(needle.Needle)
  25. err := oldNeedle.ReadData(v.DataBackend, nv.Offset.ToAcutalOffset(), nv.Size, v.Version())
  26. if err != nil {
  27. glog.V(0).Infof("Failed to check updated file at offset %d size %d: %v", nv.Offset.ToAcutalOffset(), nv.Size, err)
  28. return false
  29. }
  30. if oldNeedle.Cookie == n.Cookie && oldNeedle.Checksum == n.Checksum && bytes.Equal(oldNeedle.Data, n.Data) {
  31. n.DataSize = oldNeedle.DataSize
  32. return true
  33. }
  34. }
  35. return false
  36. }
  37. // Destroy removes everything related to this volume
  38. func (v *Volume) Destroy() (err error) {
  39. if v.isCompacting {
  40. err = fmt.Errorf("volume %d is compacting", v.Id)
  41. return
  42. }
  43. close(v.asyncRequestsChan)
  44. storageName, storageKey := v.RemoteStorageNameKey()
  45. if v.HasRemoteFile() && storageName != "" && storageKey != "" {
  46. if backendStorage, found := backend.BackendStorages[storageName]; found {
  47. backendStorage.DeleteFile(storageKey)
  48. }
  49. }
  50. v.Close()
  51. os.Remove(v.FileName() + ".dat")
  52. os.Remove(v.FileName() + ".idx")
  53. os.Remove(v.FileName() + ".vif")
  54. os.Remove(v.FileName() + ".sdx")
  55. os.Remove(v.FileName() + ".cpd")
  56. os.Remove(v.FileName() + ".cpx")
  57. os.RemoveAll(v.FileName() + ".ldb")
  58. return
  59. }
  60. func (v *Volume) asyncRequestAppend(request *needle.AsyncRequest) {
  61. v.asyncRequestsChan <- request
  62. }
  63. func (v *Volume) syncWrite(n *needle.Needle) (offset uint64, size uint32, isUnchanged bool, err error) {
  64. // glog.V(4).Infof("writing needle %s", needle.NewFileIdFromNeedle(v.Id, n).String())
  65. actualSize := needle.GetActualSize(uint32(len(n.Data)), v.Version())
  66. v.dataFileAccessLock.Lock()
  67. defer v.dataFileAccessLock.Unlock()
  68. if MaxPossibleVolumeSize < v.nm.ContentSize()+uint64(actualSize) {
  69. err = fmt.Errorf("volume size limit %d exceeded! current size is %d", MaxPossibleVolumeSize, v.nm.ContentSize())
  70. return
  71. }
  72. if v.isFileUnchanged(n) {
  73. size = n.DataSize
  74. isUnchanged = true
  75. return
  76. }
  77. // check whether existing needle cookie matches
  78. nv, ok := v.nm.Get(n.Id)
  79. if ok {
  80. existingNeedle, _, _, existingNeedleReadErr := needle.ReadNeedleHeader(v.DataBackend, v.Version(), nv.Offset.ToAcutalOffset())
  81. if existingNeedleReadErr != nil {
  82. err = fmt.Errorf("reading existing needle: %v", existingNeedleReadErr)
  83. return
  84. }
  85. if existingNeedle.Cookie != n.Cookie {
  86. glog.V(0).Infof("write cookie mismatch: existing %x, new %x", existingNeedle.Cookie, n.Cookie)
  87. err = fmt.Errorf("mismatching cookie %x", n.Cookie)
  88. return
  89. }
  90. }
  91. // append to dat file
  92. n.AppendAtNs = uint64(time.Now().UnixNano())
  93. if offset, size, _, err = n.Append(v.DataBackend, v.Version()); err != nil {
  94. return
  95. }
  96. v.lastAppendAtNs = n.AppendAtNs
  97. // add to needle map
  98. if !ok || uint64(nv.Offset.ToAcutalOffset()) < offset {
  99. if err = v.nm.Put(n.Id, ToOffset(int64(offset)), n.Size); err != nil {
  100. glog.V(4).Infof("failed to save in needle map %d: %v", n.Id, err)
  101. }
  102. }
  103. if v.lastModifiedTsSeconds < n.LastModified {
  104. v.lastModifiedTsSeconds = n.LastModified
  105. }
  106. return
  107. }
  108. func (v *Volume) writeNeedle2(n *needle.Needle, fsync bool) (offset uint64, size uint32, isUnchanged bool, err error) {
  109. // glog.V(4).Infof("writing needle %s", needle.NewFileIdFromNeedle(v.Id, n).String())
  110. if n.Ttl == needle.EMPTY_TTL && v.Ttl != needle.EMPTY_TTL {
  111. n.SetHasTtl()
  112. n.Ttl = v.Ttl
  113. }
  114. if !fsync {
  115. return v.syncWrite(n)
  116. } else {
  117. asyncRequest := needle.NewAsyncRequest(n, true)
  118. // using len(n.Data) here instead of n.Size before n.Size is populated in n.Append()
  119. asyncRequest.ActualSize = needle.GetActualSize(uint32(len(n.Data)), v.Version())
  120. v.asyncRequestAppend(asyncRequest)
  121. offset, _, isUnchanged, err = asyncRequest.WaitComplete()
  122. return
  123. }
  124. }
  125. func (v *Volume) doWriteRequest(n *needle.Needle) (offset uint64, size uint32, isUnchanged bool, err error) {
  126. // glog.V(4).Infof("writing needle %s", needle.NewFileIdFromNeedle(v.Id, n).String())
  127. if v.isFileUnchanged(n) {
  128. size = n.DataSize
  129. isUnchanged = true
  130. return
  131. }
  132. // check whether existing needle cookie matches
  133. nv, ok := v.nm.Get(n.Id)
  134. if ok {
  135. existingNeedle, _, _, existingNeedleReadErr := needle.ReadNeedleHeader(v.DataBackend, v.Version(), nv.Offset.ToAcutalOffset())
  136. if existingNeedleReadErr != nil {
  137. err = fmt.Errorf("reading existing needle: %v", existingNeedleReadErr)
  138. return
  139. }
  140. if existingNeedle.Cookie != n.Cookie {
  141. glog.V(0).Infof("write cookie mismatch: existing %x, new %x", existingNeedle.Cookie, n.Cookie)
  142. err = fmt.Errorf("mismatching cookie %x", n.Cookie)
  143. return
  144. }
  145. }
  146. // append to dat file
  147. n.AppendAtNs = uint64(time.Now().UnixNano())
  148. if offset, size, _, err = n.Append(v.DataBackend, v.Version()); err != nil {
  149. return
  150. }
  151. v.lastAppendAtNs = n.AppendAtNs
  152. // add to needle map
  153. if !ok || uint64(nv.Offset.ToAcutalOffset()) < offset {
  154. if err = v.nm.Put(n.Id, ToOffset(int64(offset)), n.Size); err != nil {
  155. glog.V(4).Infof("failed to save in needle map %d: %v", n.Id, err)
  156. }
  157. }
  158. if v.lastModifiedTsSeconds < n.LastModified {
  159. v.lastModifiedTsSeconds = n.LastModified
  160. }
  161. return
  162. }
  163. func (v *Volume) syncDelete(n *needle.Needle) (uint32, error) {
  164. glog.V(4).Infof("delete needle %s", needle.NewFileIdFromNeedle(v.Id, n).String())
  165. actualSize := needle.GetActualSize(0, v.Version())
  166. v.dataFileAccessLock.Lock()
  167. defer v.dataFileAccessLock.Unlock()
  168. if MaxPossibleVolumeSize < v.nm.ContentSize()+uint64(actualSize) {
  169. err := fmt.Errorf("volume size limit %d exceeded! current size is %d", MaxPossibleVolumeSize, v.nm.ContentSize())
  170. return 0, err
  171. }
  172. nv, ok := v.nm.Get(n.Id)
  173. //fmt.Println("key", n.Id, "volume offset", nv.Offset, "data_size", n.Size, "cached size", nv.Size)
  174. if ok && nv.Size != TombstoneFileSize {
  175. size := nv.Size
  176. n.Data = nil
  177. n.AppendAtNs = uint64(time.Now().UnixNano())
  178. offset, _, _, err := n.Append(v.DataBackend, v.Version())
  179. if err != nil {
  180. return size, err
  181. }
  182. v.lastAppendAtNs = n.AppendAtNs
  183. if err = v.nm.Delete(n.Id, ToOffset(int64(offset))); err != nil {
  184. return size, err
  185. }
  186. return size, err
  187. }
  188. return 0, nil
  189. }
  190. func (v *Volume) deleteNeedle2(n *needle.Needle) (uint32, error) {
  191. // todo: delete info is always appended no fsync, it may need fsync in future
  192. fsync := false
  193. if !fsync {
  194. return v.syncDelete(n)
  195. } else {
  196. asyncRequest := needle.NewAsyncRequest(n, false)
  197. asyncRequest.ActualSize = needle.GetActualSize(0, v.Version())
  198. v.asyncRequestAppend(asyncRequest)
  199. _, size, _, err := asyncRequest.WaitComplete()
  200. return uint32(size), err
  201. }
  202. }
  203. func (v *Volume) doDeleteRequest(n *needle.Needle) (uint32, error) {
  204. glog.V(4).Infof("delete needle %s", needle.NewFileIdFromNeedle(v.Id, n).String())
  205. nv, ok := v.nm.Get(n.Id)
  206. //fmt.Println("key", n.Id, "volume offset", nv.Offset, "data_size", n.Size, "cached size", nv.Size)
  207. if ok && nv.Size != TombstoneFileSize {
  208. size := nv.Size
  209. n.Data = nil
  210. n.AppendAtNs = uint64(time.Now().UnixNano())
  211. offset, _, _, err := n.Append(v.DataBackend, v.Version())
  212. if err != nil {
  213. return size, err
  214. }
  215. v.lastAppendAtNs = n.AppendAtNs
  216. if err = v.nm.Delete(n.Id, ToOffset(int64(offset))); err != nil {
  217. return size, err
  218. }
  219. return size, err
  220. }
  221. return 0, nil
  222. }
  223. // read fills in Needle content by looking up n.Id from NeedleMapper
  224. func (v *Volume) readNeedle(n *needle.Needle) (int, error) {
  225. v.dataFileAccessLock.RLock()
  226. defer v.dataFileAccessLock.RUnlock()
  227. nv, ok := v.nm.Get(n.Id)
  228. if !ok || nv.Offset.IsZero() {
  229. return -1, ErrorNotFound
  230. }
  231. if nv.Size == TombstoneFileSize {
  232. return -1, errors.New("already deleted")
  233. }
  234. if nv.Size == 0 {
  235. return 0, nil
  236. }
  237. err := n.ReadData(v.DataBackend, nv.Offset.ToAcutalOffset(), nv.Size, v.Version())
  238. if err != nil {
  239. return 0, err
  240. }
  241. bytesRead := len(n.Data)
  242. if !n.HasTtl() {
  243. return bytesRead, nil
  244. }
  245. ttlMinutes := n.Ttl.Minutes()
  246. if ttlMinutes == 0 {
  247. return bytesRead, nil
  248. }
  249. if !n.HasLastModifiedDate() {
  250. return bytesRead, nil
  251. }
  252. if uint64(time.Now().Unix()) < n.LastModified+uint64(ttlMinutes*60) {
  253. return bytesRead, nil
  254. }
  255. return -1, ErrorNotFound
  256. }
  257. func (v *Volume) startWorker() {
  258. go func() {
  259. chanClosed := false
  260. for {
  261. // chan closed. go thread will exit
  262. if chanClosed {
  263. break
  264. }
  265. currentRequests := make([]*needle.AsyncRequest, 0, 128)
  266. currentBytesToWrite := int64(0)
  267. for {
  268. request, ok := <-v.asyncRequestsChan
  269. //volume may be closed
  270. if !ok {
  271. chanClosed = true
  272. break
  273. }
  274. if MaxPossibleVolumeSize < v.ContentSize()+uint64(currentBytesToWrite+request.ActualSize) {
  275. request.Complete(0, 0, false,
  276. fmt.Errorf("volume size limit %d exceeded! current size is %d", MaxPossibleVolumeSize, v.ContentSize()))
  277. break
  278. }
  279. currentRequests = append(currentRequests, request)
  280. currentBytesToWrite += request.ActualSize
  281. // submit at most 4M bytes or 128 requests at one time to decrease request delay.
  282. // it also need to break if there is no data in channel to avoid io hang.
  283. if currentBytesToWrite >= 4*1024*1024 || len(currentRequests) >= 128 || len(v.asyncRequestsChan) == 0 {
  284. break
  285. }
  286. }
  287. if len(currentRequests) == 0 {
  288. continue
  289. }
  290. v.dataFileAccessLock.Lock()
  291. end, _, e := v.DataBackend.GetStat()
  292. if e != nil {
  293. for i := 0; i < len(currentRequests); i++ {
  294. currentRequests[i].Complete(0, 0, false,
  295. fmt.Errorf("cannot read current volume position: %v", e))
  296. }
  297. v.dataFileAccessLock.Unlock()
  298. continue
  299. }
  300. for i := 0; i < len(currentRequests); i++ {
  301. if currentRequests[i].IsWriteRequest {
  302. offset, size, isUnchanged, err := v.doWriteRequest(currentRequests[i].N)
  303. currentRequests[i].UpdateResult(offset, uint64(size), isUnchanged, err)
  304. } else {
  305. size, err := v.doDeleteRequest(currentRequests[i].N)
  306. currentRequests[i].UpdateResult(0, uint64(size), false, err)
  307. }
  308. }
  309. // if sync error, data is not reliable, we should mark the completed request as fail and rollback
  310. if err := v.DataBackend.Sync(); err != nil {
  311. // todo: this may generate dirty data or cause data inconsistent, may be weed need to panic?
  312. if te := v.DataBackend.Truncate(end); te != nil {
  313. glog.V(0).Infof("Failed to truncate %s back to %d with error: %v", v.DataBackend.Name(), end, te)
  314. }
  315. for i := 0; i < len(currentRequests); i++ {
  316. if currentRequests[i].IsSucceed() {
  317. currentRequests[i].UpdateResult(0, 0, false, err)
  318. }
  319. }
  320. }
  321. for i := 0; i < len(currentRequests); i++ {
  322. currentRequests[i].Submit()
  323. }
  324. v.dataFileAccessLock.Unlock()
  325. }
  326. }()
  327. }
  328. type VolumeFileScanner interface {
  329. VisitSuperBlock(super_block.SuperBlock) error
  330. ReadNeedleBody() bool
  331. VisitNeedle(n *needle.Needle, offset int64, needleHeader, needleBody []byte) error
  332. }
  333. func ScanVolumeFile(dirname string, collection string, id needle.VolumeId,
  334. needleMapKind NeedleMapType,
  335. volumeFileScanner VolumeFileScanner) (err error) {
  336. var v *Volume
  337. if v, err = loadVolumeWithoutIndex(dirname, collection, id, needleMapKind); err != nil {
  338. return fmt.Errorf("failed to load volume %d: %v", id, err)
  339. }
  340. if v.volumeInfo.Version == 0 {
  341. if err = volumeFileScanner.VisitSuperBlock(v.SuperBlock); err != nil {
  342. return fmt.Errorf("failed to process volume %d super block: %v", id, err)
  343. }
  344. }
  345. defer v.Close()
  346. version := v.Version()
  347. offset := int64(v.SuperBlock.BlockSize())
  348. return ScanVolumeFileFrom(version, v.DataBackend, offset, volumeFileScanner)
  349. }
  350. func ScanVolumeFileFrom(version needle.Version, datBackend backend.BackendStorageFile, offset int64, volumeFileScanner VolumeFileScanner) (err error) {
  351. n, nh, rest, e := needle.ReadNeedleHeader(datBackend, version, offset)
  352. if e != nil {
  353. if e == io.EOF {
  354. return nil
  355. }
  356. return fmt.Errorf("cannot read %s at offset %d: %v", datBackend.Name(), offset, e)
  357. }
  358. for n != nil {
  359. var needleBody []byte
  360. if volumeFileScanner.ReadNeedleBody() {
  361. if needleBody, err = n.ReadNeedleBody(datBackend, version, offset+NeedleHeaderSize, rest); err != nil {
  362. glog.V(0).Infof("cannot read needle body: %v", err)
  363. //err = fmt.Errorf("cannot read needle body: %v", err)
  364. //return
  365. }
  366. }
  367. err := volumeFileScanner.VisitNeedle(n, offset, nh, needleBody)
  368. if err == io.EOF {
  369. return nil
  370. }
  371. if err != nil {
  372. glog.V(0).Infof("visit needle error: %v", err)
  373. return fmt.Errorf("visit needle error: %v", err)
  374. }
  375. offset += NeedleHeaderSize + rest
  376. glog.V(4).Infof("==> new entry offset %d", offset)
  377. if n, nh, rest, err = needle.ReadNeedleHeader(datBackend, version, offset); err != nil {
  378. if err == io.EOF {
  379. return nil
  380. }
  381. return fmt.Errorf("cannot read needle header at offset %d: %v", offset, err)
  382. }
  383. glog.V(4).Infof("new entry needle size:%d rest:%d", n.Size, rest)
  384. }
  385. return nil
  386. }