123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443 |
- package storage
- import (
- "bytes"
- "errors"
- "fmt"
- "io"
- "os"
- "time"
- "github.com/chrislusf/seaweedfs/weed/util/log"
- "github.com/chrislusf/seaweedfs/weed/storage/backend"
- "github.com/chrislusf/seaweedfs/weed/storage/needle"
- "github.com/chrislusf/seaweedfs/weed/storage/super_block"
- . "github.com/chrislusf/seaweedfs/weed/storage/types"
- )
- var ErrorNotFound = errors.New("not found")
- var ErrorDeleted = errors.New("already deleted")
- var ErrorSizeMismatch = errors.New("size mismatch")
- // isFileUnchanged checks whether this needle to write is same as last one.
- // It requires serialized access in the same volume.
- func (v *Volume) isFileUnchanged(n *needle.Needle) bool {
- if v.Ttl.String() != "" {
- return false
- }
- nv, ok := v.nm.Get(n.Id)
- if ok && !nv.Offset.IsZero() && nv.Size.IsValid() {
- oldNeedle := new(needle.Needle)
- err := oldNeedle.ReadData(v.DataBackend, nv.Offset.ToAcutalOffset(), nv.Size, v.Version())
- if err != nil {
- log.Infof("Failed to check updated file at offset %d size %d: %v", nv.Offset.ToAcutalOffset(), nv.Size, err)
- return false
- }
- if oldNeedle.Cookie == n.Cookie && oldNeedle.Checksum == n.Checksum && bytes.Equal(oldNeedle.Data, n.Data) {
- n.DataSize = oldNeedle.DataSize
- return true
- }
- }
- return false
- }
- // Destroy removes everything related to this volume
- func (v *Volume) Destroy() (err error) {
- if v.isCompacting {
- err = fmt.Errorf("volume %d is compacting", v.Id)
- return
- }
- close(v.asyncRequestsChan)
- storageName, storageKey := v.RemoteStorageNameKey()
- if v.HasRemoteFile() && storageName != "" && storageKey != "" {
- if backendStorage, found := backend.BackendStorages[storageName]; found {
- backendStorage.DeleteFile(storageKey)
- }
- }
- v.Close()
- removeVolumeFiles(v.FileName())
- return
- }
- func removeVolumeFiles(filename string) {
- os.Remove(filename + ".dat")
- os.Remove(filename + ".idx")
- os.Remove(filename + ".vif")
- os.Remove(filename + ".sdx")
- os.Remove(filename + ".cpd")
- os.Remove(filename + ".cpx")
- os.RemoveAll(filename + ".ldb")
- os.Remove(filename + ".note")
- }
- func (v *Volume) asyncRequestAppend(request *needle.AsyncRequest) {
- v.asyncRequestsChan <- request
- }
- func (v *Volume) syncWrite(n *needle.Needle) (offset uint64, size Size, isUnchanged bool, err error) {
- // log.Tracef("writing needle %s", needle.NewFileIdFromNeedle(v.Id, n).String())
- actualSize := needle.GetActualSize(Size(len(n.Data)), v.Version())
- v.dataFileAccessLock.Lock()
- defer v.dataFileAccessLock.Unlock()
- if MaxPossibleVolumeSize < v.nm.ContentSize()+uint64(actualSize) {
- err = fmt.Errorf("volume size limit %d exceeded! current size is %d", MaxPossibleVolumeSize, v.nm.ContentSize())
- return
- }
- if v.isFileUnchanged(n) {
- size = Size(n.DataSize)
- isUnchanged = true
- return
- }
- // check whether existing needle cookie matches
- nv, ok := v.nm.Get(n.Id)
- if ok {
- existingNeedle, _, _, existingNeedleReadErr := needle.ReadNeedleHeader(v.DataBackend, v.Version(), nv.Offset.ToAcutalOffset())
- if existingNeedleReadErr != nil {
- err = fmt.Errorf("reading existing needle: %v", existingNeedleReadErr)
- return
- }
- if existingNeedle.Cookie != n.Cookie {
- log.Infof("write cookie mismatch: existing %x, new %x", existingNeedle.Cookie, n.Cookie)
- err = fmt.Errorf("mismatching cookie %x", n.Cookie)
- return
- }
- }
- // append to dat file
- n.AppendAtNs = uint64(time.Now().UnixNano())
- if offset, size, _, err = n.Append(v.DataBackend, v.Version()); err != nil {
- return
- }
- v.lastAppendAtNs = n.AppendAtNs
- // add to needle map
- if !ok || uint64(nv.Offset.ToAcutalOffset()) < offset {
- if err = v.nm.Put(n.Id, ToOffset(int64(offset)), n.Size); err != nil {
- log.Tracef("failed to save in needle map %d: %v", n.Id, err)
- }
- }
- if v.lastModifiedTsSeconds < n.LastModified {
- v.lastModifiedTsSeconds = n.LastModified
- }
- return
- }
- func (v *Volume) writeNeedle2(n *needle.Needle, fsync bool) (offset uint64, size Size, isUnchanged bool, err error) {
- // log.Tracef("writing needle %s", needle.NewFileIdFromNeedle(v.Id, n).String())
- if n.Ttl == needle.EMPTY_TTL && v.Ttl != needle.EMPTY_TTL {
- n.SetHasTtl()
- n.Ttl = v.Ttl
- }
- if !fsync {
- return v.syncWrite(n)
- } else {
- asyncRequest := needle.NewAsyncRequest(n, true)
- // using len(n.Data) here instead of n.Size before n.Size is populated in n.Append()
- asyncRequest.ActualSize = needle.GetActualSize(Size(len(n.Data)), v.Version())
- v.asyncRequestAppend(asyncRequest)
- offset, _, isUnchanged, err = asyncRequest.WaitComplete()
- return
- }
- }
- func (v *Volume) doWriteRequest(n *needle.Needle) (offset uint64, size Size, isUnchanged bool, err error) {
- // log.Tracef("writing needle %s", needle.NewFileIdFromNeedle(v.Id, n).String())
- if v.isFileUnchanged(n) {
- size = Size(n.DataSize)
- isUnchanged = true
- return
- }
- // check whether existing needle cookie matches
- nv, ok := v.nm.Get(n.Id)
- if ok {
- existingNeedle, _, _, existingNeedleReadErr := needle.ReadNeedleHeader(v.DataBackend, v.Version(), nv.Offset.ToAcutalOffset())
- if existingNeedleReadErr != nil {
- err = fmt.Errorf("reading existing needle: %v", existingNeedleReadErr)
- return
- }
- if existingNeedle.Cookie != n.Cookie {
- log.Infof("write cookie mismatch: existing %x, new %x", existingNeedle.Cookie, n.Cookie)
- err = fmt.Errorf("mismatching cookie %x", n.Cookie)
- return
- }
- }
- // append to dat file
- n.AppendAtNs = uint64(time.Now().UnixNano())
- if offset, size, _, err = n.Append(v.DataBackend, v.Version()); err != nil {
- return
- }
- v.lastAppendAtNs = n.AppendAtNs
- // add to needle map
- if !ok || uint64(nv.Offset.ToAcutalOffset()) < offset {
- if err = v.nm.Put(n.Id, ToOffset(int64(offset)), n.Size); err != nil {
- log.Tracef("failed to save in needle map %d: %v", n.Id, err)
- }
- }
- if v.lastModifiedTsSeconds < n.LastModified {
- v.lastModifiedTsSeconds = n.LastModified
- }
- return
- }
- func (v *Volume) syncDelete(n *needle.Needle) (Size, error) {
- // log.Tracef("delete needle %s", needle.NewFileIdFromNeedle(v.Id, n).String())
- actualSize := needle.GetActualSize(0, v.Version())
- v.dataFileAccessLock.Lock()
- defer v.dataFileAccessLock.Unlock()
- if MaxPossibleVolumeSize < v.nm.ContentSize()+uint64(actualSize) {
- err := fmt.Errorf("volume size limit %d exceeded! current size is %d", MaxPossibleVolumeSize, v.nm.ContentSize())
- return 0, err
- }
- nv, ok := v.nm.Get(n.Id)
- // fmt.Println("key", n.Id, "volume offset", nv.Offset, "data_size", n.Size, "cached size", nv.Size)
- if ok && nv.Size.IsValid() {
- size := nv.Size
- n.Data = nil
- n.AppendAtNs = uint64(time.Now().UnixNano())
- offset, _, _, err := n.Append(v.DataBackend, v.Version())
- if err != nil {
- return size, err
- }
- v.lastAppendAtNs = n.AppendAtNs
- if err = v.nm.Delete(n.Id, ToOffset(int64(offset))); err != nil {
- return size, err
- }
- return size, err
- }
- return 0, nil
- }
- func (v *Volume) deleteNeedle2(n *needle.Needle) (Size, error) {
- // todo: delete info is always appended no fsync, it may need fsync in future
- fsync := false
- if !fsync {
- return v.syncDelete(n)
- } else {
- asyncRequest := needle.NewAsyncRequest(n, false)
- asyncRequest.ActualSize = needle.GetActualSize(0, v.Version())
- v.asyncRequestAppend(asyncRequest)
- _, size, _, err := asyncRequest.WaitComplete()
- return Size(size), err
- }
- }
- func (v *Volume) doDeleteRequest(n *needle.Needle) (Size, error) {
- log.Tracef("delete needle %s", needle.NewFileIdFromNeedle(v.Id, n).String())
- nv, ok := v.nm.Get(n.Id)
- // fmt.Println("key", n.Id, "volume offset", nv.Offset, "data_size", n.Size, "cached size", nv.Size)
- if ok && nv.Size.IsValid() {
- size := nv.Size
- n.Data = nil
- n.AppendAtNs = uint64(time.Now().UnixNano())
- offset, _, _, err := n.Append(v.DataBackend, v.Version())
- if err != nil {
- return size, err
- }
- v.lastAppendAtNs = n.AppendAtNs
- if err = v.nm.Delete(n.Id, ToOffset(int64(offset))); err != nil {
- return size, err
- }
- return size, err
- }
- return 0, nil
- }
- // read fills in Needle content by looking up n.Id from NeedleMapper
- func (v *Volume) readNeedle(n *needle.Needle, readOption *ReadOption) (int, error) {
- v.dataFileAccessLock.RLock()
- defer v.dataFileAccessLock.RUnlock()
- nv, ok := v.nm.Get(n.Id)
- if !ok || nv.Offset.IsZero() {
- return -1, ErrorNotFound
- }
- readSize := nv.Size
- if readSize.IsDeleted() {
- if readOption != nil && readOption.ReadDeleted && readSize != TombstoneFileSize {
- log.Tracef("reading deleted %s", n.String())
- readSize = -readSize
- } else {
- return -1, ErrorDeleted
- }
- }
- if readSize == 0 {
- return 0, nil
- }
- err := n.ReadData(v.DataBackend, nv.Offset.ToAcutalOffset(), readSize, v.Version())
- if err == needle.ErrorSizeMismatch && OffsetSize == 4 {
- err = n.ReadData(v.DataBackend, nv.Offset.ToAcutalOffset()+int64(MaxPossibleVolumeSize), readSize, v.Version())
- }
- if err != nil {
- return 0, err
- }
- bytesRead := len(n.Data)
- if !n.HasTtl() {
- return bytesRead, nil
- }
- ttlMinutes := n.Ttl.Minutes()
- if ttlMinutes == 0 {
- return bytesRead, nil
- }
- if !n.HasLastModifiedDate() {
- return bytesRead, nil
- }
- if uint64(time.Now().Unix()) < n.LastModified+uint64(ttlMinutes*60) {
- return bytesRead, nil
- }
- return -1, ErrorNotFound
- }
- func (v *Volume) startWorker() {
- go func() {
- chanClosed := false
- for {
- // chan closed. go thread will exit
- if chanClosed {
- break
- }
- currentRequests := make([]*needle.AsyncRequest, 0, 128)
- currentBytesToWrite := int64(0)
- for {
- request, ok := <-v.asyncRequestsChan
- // volume may be closed
- if !ok {
- chanClosed = true
- break
- }
- if MaxPossibleVolumeSize < v.ContentSize()+uint64(currentBytesToWrite+request.ActualSize) {
- request.Complete(0, 0, false,
- fmt.Errorf("volume size limit %d exceeded! current size is %d", MaxPossibleVolumeSize, v.ContentSize()))
- break
- }
- currentRequests = append(currentRequests, request)
- currentBytesToWrite += request.ActualSize
- // submit at most 4M bytes or 128 requests at one time to decrease request delay.
- // it also need to break if there is no data in channel to avoid io hang.
- if currentBytesToWrite >= 4*1024*1024 || len(currentRequests) >= 128 || len(v.asyncRequestsChan) == 0 {
- break
- }
- }
- if len(currentRequests) == 0 {
- continue
- }
- v.dataFileAccessLock.Lock()
- end, _, e := v.DataBackend.GetStat()
- if e != nil {
- for i := 0; i < len(currentRequests); i++ {
- currentRequests[i].Complete(0, 0, false,
- fmt.Errorf("cannot read current volume position: %v", e))
- }
- v.dataFileAccessLock.Unlock()
- continue
- }
- for i := 0; i < len(currentRequests); i++ {
- if currentRequests[i].IsWriteRequest {
- offset, size, isUnchanged, err := v.doWriteRequest(currentRequests[i].N)
- currentRequests[i].UpdateResult(offset, uint64(size), isUnchanged, err)
- } else {
- size, err := v.doDeleteRequest(currentRequests[i].N)
- currentRequests[i].UpdateResult(0, uint64(size), false, err)
- }
- }
- // if sync error, data is not reliable, we should mark the completed request as fail and rollback
- if err := v.DataBackend.Sync(); err != nil {
- // todo: this may generate dirty data or cause data inconsistent, may be weed need to panic?
- if te := v.DataBackend.Truncate(end); te != nil {
- log.Infof("Failed to truncate %s back to %d with error: %v", v.DataBackend.Name(), end, te)
- }
- for i := 0; i < len(currentRequests); i++ {
- if currentRequests[i].IsSucceed() {
- currentRequests[i].UpdateResult(0, 0, false, err)
- }
- }
- }
- for i := 0; i < len(currentRequests); i++ {
- currentRequests[i].Submit()
- }
- v.dataFileAccessLock.Unlock()
- }
- }()
- }
- type VolumeFileScanner interface {
- VisitSuperBlock(super_block.SuperBlock) error
- ReadNeedleBody() bool
- VisitNeedle(n *needle.Needle, offset int64, needleHeader, needleBody []byte) error
- }
- func ScanVolumeFile(dirname string, collection string, id needle.VolumeId,
- needleMapKind NeedleMapType,
- volumeFileScanner VolumeFileScanner) (err error) {
- var v *Volume
- if v, err = loadVolumeWithoutIndex(dirname, collection, id, needleMapKind); err != nil {
- return fmt.Errorf("failed to load volume %d: %v", id, err)
- }
- if err = volumeFileScanner.VisitSuperBlock(v.SuperBlock); err != nil {
- return fmt.Errorf("failed to process volume %d super block: %v", id, err)
- }
- defer v.Close()
- version := v.Version()
- offset := int64(v.SuperBlock.BlockSize())
- return ScanVolumeFileFrom(version, v.DataBackend, offset, volumeFileScanner)
- }
- func ScanVolumeFileFrom(version needle.Version, datBackend backend.BackendStorageFile, offset int64, volumeFileScanner VolumeFileScanner) (err error) {
- n, nh, rest, e := needle.ReadNeedleHeader(datBackend, version, offset)
- if e != nil {
- if e == io.EOF {
- return nil
- }
- return fmt.Errorf("cannot read %s at offset %d: %v", datBackend.Name(), offset, e)
- }
- for n != nil {
- var needleBody []byte
- if volumeFileScanner.ReadNeedleBody() {
- // println("needle", n.Id.String(), "offset", offset, "size", n.Size, "rest", rest)
- if needleBody, err = n.ReadNeedleBody(datBackend, version, offset+NeedleHeaderSize, rest); err != nil {
- log.Infof("cannot read needle head [%d, %d) body [%d, %d) body length %d: %v", offset, offset+NeedleHeaderSize, offset+NeedleHeaderSize, offset+NeedleHeaderSize+rest, rest, err)
- // err = fmt.Errorf("cannot read needle body: %v", err)
- // return
- }
- }
- err := volumeFileScanner.VisitNeedle(n, offset, nh, needleBody)
- if err == io.EOF {
- return nil
- }
- if err != nil {
- log.Infof("visit needle error: %v", err)
- return fmt.Errorf("visit needle error: %v", err)
- }
- offset += NeedleHeaderSize + rest
- log.Tracef("==> new entry offset %d", offset)
- if n, nh, rest, err = needle.ReadNeedleHeader(datBackend, version, offset); err != nil {
- if err == io.EOF {
- return nil
- }
- return fmt.Errorf("cannot read needle header at offset %d: %v", offset, err)
- }
- log.Tracef("new entry needle size:%d rest:%d", n.Size, rest)
- }
- return nil
- }
|