stream.go 11 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388
  1. package filer
  2. import (
  3. "bytes"
  4. "fmt"
  5. "io"
  6. "math"
  7. "strings"
  8. "sync"
  9. "time"
  10. "golang.org/x/exp/slices"
  11. "github.com/seaweedfs/seaweedfs/weed/glog"
  12. "github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
  13. "github.com/seaweedfs/seaweedfs/weed/stats"
  14. "github.com/seaweedfs/seaweedfs/weed/util"
  15. "github.com/seaweedfs/seaweedfs/weed/wdclient"
  16. )
  17. var getLookupFileIdBackoffSchedule = []time.Duration{
  18. 150 * time.Millisecond,
  19. 600 * time.Millisecond,
  20. 1800 * time.Millisecond,
  21. }
  22. func HasData(entry *filer_pb.Entry) bool {
  23. if len(entry.Content) > 0 {
  24. return true
  25. }
  26. return len(entry.GetChunks()) > 0
  27. }
  28. func IsSameData(a, b *filer_pb.Entry) bool {
  29. if len(a.Content) > 0 || len(b.Content) > 0 {
  30. return bytes.Equal(a.Content, b.Content)
  31. }
  32. return isSameChunks(a.Chunks, b.Chunks)
  33. }
  34. func isSameChunks(a, b []*filer_pb.FileChunk) bool {
  35. if len(a) != len(b) {
  36. return false
  37. }
  38. slices.SortFunc(a, func(i, j *filer_pb.FileChunk) int {
  39. return strings.Compare(i.ETag, j.ETag)
  40. })
  41. slices.SortFunc(b, func(i, j *filer_pb.FileChunk) int {
  42. return strings.Compare(i.ETag, j.ETag)
  43. })
  44. for i := 0; i < len(a); i++ {
  45. if a[i].ETag != b[i].ETag {
  46. return false
  47. }
  48. }
  49. return true
  50. }
  51. func NewFileReader(filerClient filer_pb.FilerClient, entry *filer_pb.Entry) io.Reader {
  52. if len(entry.Content) > 0 {
  53. return bytes.NewReader(entry.Content)
  54. }
  55. return NewChunkStreamReader(filerClient, entry.GetChunks())
  56. }
  57. type DoStreamContent func(writer io.Writer) error
  58. func PrepareStreamContent(masterClient wdclient.HasLookupFileIdFunction, jwtFunc VolumeServerJwtFunction, chunks []*filer_pb.FileChunk, offset int64, size int64) (DoStreamContent, error) {
  59. return PrepareStreamContentWithThrottler(masterClient, jwtFunc, chunks, offset, size, 0)
  60. }
  61. type VolumeServerJwtFunction func(fileId string) string
  62. func noJwtFunc(string) string {
  63. return ""
  64. }
  65. func PrepareStreamContentWithThrottler(masterClient wdclient.HasLookupFileIdFunction, jwtFunc VolumeServerJwtFunction, chunks []*filer_pb.FileChunk, offset int64, size int64, downloadMaxBytesPs int64) (DoStreamContent, error) {
  66. glog.V(4).Infof("prepare to stream content for chunks: %d", len(chunks))
  67. chunkViews := ViewFromChunks(masterClient.GetLookupFileIdFunction(), chunks, offset, size)
  68. fileId2Url := make(map[string][]string)
  69. for x := chunkViews.Front(); x != nil; x = x.Next {
  70. chunkView := x.Value
  71. var urlStrings []string
  72. var err error
  73. for _, backoff := range getLookupFileIdBackoffSchedule {
  74. urlStrings, err = masterClient.GetLookupFileIdFunction()(chunkView.FileId)
  75. if err == nil && len(urlStrings) > 0 {
  76. break
  77. }
  78. glog.V(4).Infof("waiting for chunk: %s", chunkView.FileId)
  79. time.Sleep(backoff)
  80. }
  81. if err != nil {
  82. glog.V(1).Infof("operation LookupFileId %s failed, err: %v", chunkView.FileId, err)
  83. return nil, err
  84. } else if len(urlStrings) == 0 {
  85. errUrlNotFound := fmt.Errorf("operation LookupFileId %s failed, err: urls not found", chunkView.FileId)
  86. glog.Error(errUrlNotFound)
  87. return nil, errUrlNotFound
  88. }
  89. fileId2Url[chunkView.FileId] = urlStrings
  90. }
  91. return func(writer io.Writer) error {
  92. downloadThrottler := util.NewWriteThrottler(downloadMaxBytesPs)
  93. remaining := size
  94. for x := chunkViews.Front(); x != nil; x = x.Next {
  95. chunkView := x.Value
  96. if offset < chunkView.ViewOffset {
  97. gap := chunkView.ViewOffset - offset
  98. remaining -= gap
  99. glog.V(4).Infof("zero [%d,%d)", offset, chunkView.ViewOffset)
  100. err := writeZero(writer, gap)
  101. if err != nil {
  102. return fmt.Errorf("write zero [%d,%d)", offset, chunkView.ViewOffset)
  103. }
  104. offset = chunkView.ViewOffset
  105. }
  106. urlStrings := fileId2Url[chunkView.FileId]
  107. start := time.Now()
  108. jwt := jwtFunc(chunkView.FileId)
  109. err := retriedStreamFetchChunkData(writer, urlStrings, jwt, chunkView.CipherKey, chunkView.IsGzipped, chunkView.IsFullChunk(), chunkView.OffsetInChunk, int(chunkView.ViewSize))
  110. offset += int64(chunkView.ViewSize)
  111. remaining -= int64(chunkView.ViewSize)
  112. stats.FilerRequestHistogram.WithLabelValues("chunkDownload").Observe(time.Since(start).Seconds())
  113. if err != nil {
  114. stats.FilerHandlerCounter.WithLabelValues("chunkDownloadError").Inc()
  115. return fmt.Errorf("read chunk: %v", err)
  116. }
  117. stats.FilerHandlerCounter.WithLabelValues("chunkDownload").Inc()
  118. downloadThrottler.MaybeSlowdown(int64(chunkView.ViewSize))
  119. }
  120. if remaining > 0 {
  121. glog.V(4).Infof("zero [%d,%d)", offset, offset+remaining)
  122. err := writeZero(writer, remaining)
  123. if err != nil {
  124. return fmt.Errorf("write zero [%d,%d)", offset, offset+remaining)
  125. }
  126. }
  127. return nil
  128. }, nil
  129. }
  130. func StreamContent(masterClient wdclient.HasLookupFileIdFunction, writer io.Writer, chunks []*filer_pb.FileChunk, offset int64, size int64) error {
  131. streamFn, err := PrepareStreamContent(masterClient, noJwtFunc, chunks, offset, size)
  132. if err != nil {
  133. return err
  134. }
  135. return streamFn(writer)
  136. }
  137. // ---------------- ReadAllReader ----------------------------------
  138. func writeZero(w io.Writer, size int64) (err error) {
  139. zeroPadding := make([]byte, 1024)
  140. var written int
  141. for size > 0 {
  142. if size > 1024 {
  143. written, err = w.Write(zeroPadding)
  144. } else {
  145. written, err = w.Write(zeroPadding[:size])
  146. }
  147. size -= int64(written)
  148. if err != nil {
  149. return
  150. }
  151. }
  152. return
  153. }
  154. func ReadAll(buffer []byte, masterClient *wdclient.MasterClient, chunks []*filer_pb.FileChunk) error {
  155. lookupFileIdFn := func(fileId string) (targetUrls []string, err error) {
  156. return masterClient.LookupFileId(fileId)
  157. }
  158. chunkViews := ViewFromChunks(lookupFileIdFn, chunks, 0, int64(len(buffer)))
  159. idx := 0
  160. for x := chunkViews.Front(); x != nil; x = x.Next {
  161. chunkView := x.Value
  162. urlStrings, err := lookupFileIdFn(chunkView.FileId)
  163. if err != nil {
  164. glog.V(1).Infof("operation LookupFileId %s failed, err: %v", chunkView.FileId, err)
  165. return err
  166. }
  167. n, err := util.RetriedFetchChunkData(buffer[idx:idx+int(chunkView.ViewSize)], urlStrings, chunkView.CipherKey, chunkView.IsGzipped, chunkView.IsFullChunk(), chunkView.OffsetInChunk)
  168. if err != nil {
  169. return err
  170. }
  171. idx += n
  172. }
  173. return nil
  174. }
  175. // ---------------- ChunkStreamReader ----------------------------------
  176. type ChunkStreamReader struct {
  177. head *Interval[*ChunkView]
  178. chunkView *Interval[*ChunkView]
  179. totalSize int64
  180. logicOffset int64
  181. buffer []byte
  182. bufferOffset int64
  183. bufferLock sync.Mutex
  184. chunk string
  185. lookupFileId wdclient.LookupFileIdFunctionType
  186. }
  187. var _ = io.ReadSeeker(&ChunkStreamReader{})
  188. var _ = io.ReaderAt(&ChunkStreamReader{})
  189. func doNewChunkStreamReader(lookupFileIdFn wdclient.LookupFileIdFunctionType, chunks []*filer_pb.FileChunk) *ChunkStreamReader {
  190. chunkViews := ViewFromChunks(lookupFileIdFn, chunks, 0, math.MaxInt64)
  191. var totalSize int64
  192. for x := chunkViews.Front(); x != nil; x = x.Next {
  193. chunk := x.Value
  194. totalSize += int64(chunk.ViewSize)
  195. }
  196. return &ChunkStreamReader{
  197. head: chunkViews.Front(),
  198. chunkView: chunkViews.Front(),
  199. lookupFileId: lookupFileIdFn,
  200. totalSize: totalSize,
  201. }
  202. }
  203. func NewChunkStreamReaderFromFiler(masterClient *wdclient.MasterClient, chunks []*filer_pb.FileChunk) *ChunkStreamReader {
  204. lookupFileIdFn := func(fileId string) (targetUrl []string, err error) {
  205. return masterClient.LookupFileId(fileId)
  206. }
  207. return doNewChunkStreamReader(lookupFileIdFn, chunks)
  208. }
  209. func NewChunkStreamReader(filerClient filer_pb.FilerClient, chunks []*filer_pb.FileChunk) *ChunkStreamReader {
  210. lookupFileIdFn := LookupFn(filerClient)
  211. return doNewChunkStreamReader(lookupFileIdFn, chunks)
  212. }
  213. func (c *ChunkStreamReader) ReadAt(p []byte, off int64) (n int, err error) {
  214. c.bufferLock.Lock()
  215. defer c.bufferLock.Unlock()
  216. if err = c.prepareBufferFor(off); err != nil {
  217. return
  218. }
  219. c.logicOffset = off
  220. return c.doRead(p)
  221. }
  222. func (c *ChunkStreamReader) Read(p []byte) (n int, err error) {
  223. c.bufferLock.Lock()
  224. defer c.bufferLock.Unlock()
  225. return c.doRead(p)
  226. }
  227. func (c *ChunkStreamReader) doRead(p []byte) (n int, err error) {
  228. // fmt.Printf("do read [%d,%d) at %s[%d,%d)\n", c.logicOffset, c.logicOffset+int64(len(p)), c.chunk, c.bufferOffset, c.bufferOffset+int64(len(c.buffer)))
  229. for n < len(p) {
  230. // println("read", c.logicOffset)
  231. if err = c.prepareBufferFor(c.logicOffset); err != nil {
  232. return
  233. }
  234. t := copy(p[n:], c.buffer[c.logicOffset-c.bufferOffset:])
  235. n += t
  236. c.logicOffset += int64(t)
  237. }
  238. return
  239. }
  240. func (c *ChunkStreamReader) isBufferEmpty() bool {
  241. return len(c.buffer) <= int(c.logicOffset-c.bufferOffset)
  242. }
  243. func (c *ChunkStreamReader) Seek(offset int64, whence int) (int64, error) {
  244. c.bufferLock.Lock()
  245. defer c.bufferLock.Unlock()
  246. var err error
  247. switch whence {
  248. case io.SeekStart:
  249. case io.SeekCurrent:
  250. offset += c.logicOffset
  251. case io.SeekEnd:
  252. offset = c.totalSize + offset
  253. }
  254. if offset > c.totalSize {
  255. err = io.ErrUnexpectedEOF
  256. } else {
  257. c.logicOffset = offset
  258. }
  259. return offset, err
  260. }
  261. func insideChunk(offset int64, chunk *ChunkView) bool {
  262. return chunk.ViewOffset <= offset && offset < chunk.ViewOffset+int64(chunk.ViewSize)
  263. }
  264. func (c *ChunkStreamReader) prepareBufferFor(offset int64) (err error) {
  265. // stay in the same chunk
  266. if c.bufferOffset <= offset && offset < c.bufferOffset+int64(len(c.buffer)) {
  267. return nil
  268. }
  269. // glog.V(2).Infof("c.chunkView: %v buffer:[%d,%d) offset:%d totalSize:%d", c.chunkView, c.bufferOffset, c.bufferOffset+int64(len(c.buffer)), offset, c.totalSize)
  270. // find a possible chunk view
  271. p := c.chunkView
  272. for p != nil {
  273. chunk := p.Value
  274. // glog.V(2).Infof("prepareBufferFor check chunk:[%d,%d)", chunk.ViewOffset, chunk.ViewOffset+int64(chunk.ViewSize))
  275. if insideChunk(offset, chunk) {
  276. if c.isBufferEmpty() || c.bufferOffset != chunk.ViewOffset {
  277. c.chunkView = p
  278. return c.fetchChunkToBuffer(chunk)
  279. }
  280. }
  281. if offset < c.bufferOffset {
  282. p = p.Prev
  283. } else {
  284. p = p.Next
  285. }
  286. }
  287. return io.EOF
  288. }
  289. func (c *ChunkStreamReader) fetchChunkToBuffer(chunkView *ChunkView) error {
  290. urlStrings, err := c.lookupFileId(chunkView.FileId)
  291. if err != nil {
  292. glog.V(1).Infof("operation LookupFileId %s failed, err: %v", chunkView.FileId, err)
  293. return err
  294. }
  295. var buffer bytes.Buffer
  296. var shouldRetry bool
  297. for _, urlString := range urlStrings {
  298. shouldRetry, err = util.ReadUrlAsStream(urlString+"?readDeleted=true", chunkView.CipherKey, chunkView.IsGzipped, chunkView.IsFullChunk(), chunkView.OffsetInChunk, int(chunkView.ViewSize), func(data []byte) {
  299. buffer.Write(data)
  300. })
  301. if !shouldRetry {
  302. break
  303. }
  304. if err != nil {
  305. glog.V(1).Infof("read %s failed, err: %v", chunkView.FileId, err)
  306. buffer.Reset()
  307. } else {
  308. break
  309. }
  310. }
  311. if err != nil {
  312. return err
  313. }
  314. c.buffer = buffer.Bytes()
  315. c.bufferOffset = chunkView.ViewOffset
  316. c.chunk = chunkView.FileId
  317. // glog.V(0).Infof("fetched %s [%d,%d)", chunkView.FileId, chunkView.ViewOffset, chunkView.ViewOffset+int64(chunkView.ViewSize))
  318. return nil
  319. }
  320. func (c *ChunkStreamReader) Close() {
  321. // TODO try to release and reuse buffer
  322. }
  323. func VolumeId(fileId string) string {
  324. lastCommaIndex := strings.LastIndex(fileId, ",")
  325. if lastCommaIndex > 0 {
  326. return fileId[:lastCommaIndex]
  327. }
  328. return fileId
  329. }