volume_write.go 11 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358
  1. package storage
  2. import (
  3. "bytes"
  4. "errors"
  5. "fmt"
  6. "os"
  7. "syscall"
  8. "github.com/seaweedfs/seaweedfs/weed/glog"
  9. "github.com/seaweedfs/seaweedfs/weed/storage/backend"
  10. "github.com/seaweedfs/seaweedfs/weed/storage/needle"
  11. . "github.com/seaweedfs/seaweedfs/weed/storage/types"
  12. )
  13. var ErrorNotFound = errors.New("not found")
  14. var ErrorDeleted = errors.New("already deleted")
  15. var ErrorSizeMismatch = errors.New("size mismatch")
  16. func (v *Volume) checkReadWriteError(err error) {
  17. if err == nil {
  18. if v.lastIoError != nil {
  19. v.lastIoError = nil
  20. }
  21. return
  22. }
  23. if errors.Is(err, syscall.EIO) {
  24. v.lastIoError = err
  25. }
  26. }
  27. // isFileUnchanged checks whether this needle to write is same as last one.
  28. // It requires serialized access in the same volume.
  29. func (v *Volume) isFileUnchanged(n *needle.Needle) bool {
  30. if v.Ttl.String() != "" {
  31. return false
  32. }
  33. nv, ok := v.nm.Get(n.Id)
  34. if ok && !nv.Offset.IsZero() && nv.Size.IsValid() {
  35. oldNeedle := new(needle.Needle)
  36. err := oldNeedle.ReadData(v.DataBackend, nv.Offset.ToActualOffset(), nv.Size, v.Version())
  37. if err != nil {
  38. glog.V(0).Infof("Failed to check updated file at offset %d size %d: %v", nv.Offset.ToActualOffset(), nv.Size, err)
  39. return false
  40. }
  41. if oldNeedle.Cookie == n.Cookie && oldNeedle.Checksum == n.Checksum && bytes.Equal(oldNeedle.Data, n.Data) {
  42. n.DataSize = oldNeedle.DataSize
  43. return true
  44. }
  45. }
  46. return false
  47. }
  48. var ErrVolumeNotEmpty = fmt.Errorf("volume not empty")
  49. // Destroy removes everything related to this volume
  50. func (v *Volume) Destroy(onlyEmpty bool) (err error) {
  51. v.dataFileAccessLock.Lock()
  52. defer v.dataFileAccessLock.Unlock()
  53. if onlyEmpty {
  54. isEmpty, e := v.doIsEmpty()
  55. if e != nil {
  56. err = fmt.Errorf("failed to read isEmpty %v", e)
  57. return
  58. }
  59. if !isEmpty {
  60. err = ErrVolumeNotEmpty
  61. return
  62. }
  63. }
  64. if v.isCompacting || v.isCommitCompacting {
  65. err = fmt.Errorf("volume %d is compacting", v.Id)
  66. return
  67. }
  68. close(v.asyncRequestsChan)
  69. storageName, storageKey := v.RemoteStorageNameKey()
  70. if v.HasRemoteFile() && storageName != "" && storageKey != "" {
  71. if backendStorage, found := backend.BackendStorages[storageName]; found {
  72. backendStorage.DeleteFile(storageKey)
  73. }
  74. }
  75. v.doClose()
  76. removeVolumeFiles(v.DataFileName())
  77. removeVolumeFiles(v.IndexFileName())
  78. return
  79. }
  80. func removeVolumeFiles(filename string) {
  81. // basic
  82. os.Remove(filename + ".dat")
  83. os.Remove(filename + ".idx")
  84. os.Remove(filename + ".vif")
  85. // sorted index file
  86. os.Remove(filename + ".sdx")
  87. // compaction
  88. os.Remove(filename + ".cpd")
  89. os.Remove(filename + ".cpx")
  90. // level db index file
  91. os.RemoveAll(filename + ".ldb")
  92. // marker for damaged or incomplete volume
  93. os.Remove(filename + ".note")
  94. }
  95. func (v *Volume) asyncRequestAppend(request *needle.AsyncRequest) {
  96. v.asyncRequestsChan <- request
  97. }
  98. func (v *Volume) syncWrite(n *needle.Needle, checkCookie bool) (offset uint64, size Size, isUnchanged bool, err error) {
  99. // glog.V(4).Infof("writing needle %s", needle.NewFileIdFromNeedle(v.Id, n).String())
  100. v.dataFileAccessLock.Lock()
  101. defer v.dataFileAccessLock.Unlock()
  102. return v.doWriteRequest(n, checkCookie)
  103. }
  104. func (v *Volume) writeNeedle2(n *needle.Needle, checkCookie bool, fsync bool) (offset uint64, size Size, isUnchanged bool, err error) {
  105. // glog.V(4).Infof("writing needle %s", needle.NewFileIdFromNeedle(v.Id, n).String())
  106. if n.Ttl == needle.EMPTY_TTL && v.Ttl != needle.EMPTY_TTL {
  107. n.SetHasTtl()
  108. n.Ttl = v.Ttl
  109. }
  110. if !fsync {
  111. return v.syncWrite(n, checkCookie)
  112. } else {
  113. asyncRequest := needle.NewAsyncRequest(n, true)
  114. // using len(n.Data) here instead of n.Size before n.Size is populated in n.Append()
  115. asyncRequest.ActualSize = needle.GetActualSize(Size(len(n.Data)), v.Version())
  116. v.asyncRequestAppend(asyncRequest)
  117. offset, _, isUnchanged, err = asyncRequest.WaitComplete()
  118. return
  119. }
  120. }
  121. func (v *Volume) doWriteRequest(n *needle.Needle, checkCookie bool) (offset uint64, size Size, isUnchanged bool, err error) {
  122. // glog.V(4).Infof("writing needle %s", needle.NewFileIdFromNeedle(v.Id, n).String())
  123. if v.isFileUnchanged(n) {
  124. size = Size(n.DataSize)
  125. isUnchanged = true
  126. return
  127. }
  128. // check whether existing needle cookie matches
  129. nv, ok := v.nm.Get(n.Id)
  130. if ok {
  131. existingNeedle, _, _, existingNeedleReadErr := needle.ReadNeedleHeader(v.DataBackend, v.Version(), nv.Offset.ToActualOffset())
  132. if existingNeedleReadErr != nil {
  133. err = fmt.Errorf("reading existing needle: %w", existingNeedleReadErr)
  134. return
  135. }
  136. if n.Cookie == 0 && !checkCookie {
  137. // this is from batch deletion, and read back again when tailing a remote volume
  138. // which only happens when checkCookie == false and fsync == false
  139. n.Cookie = existingNeedle.Cookie
  140. }
  141. if existingNeedle.Cookie != n.Cookie {
  142. glog.V(0).Infof("write cookie mismatch: existing %s, new %s",
  143. needle.NewFileIdFromNeedle(v.Id, existingNeedle), needle.NewFileIdFromNeedle(v.Id, n))
  144. err = fmt.Errorf("mismatching cookie %x", n.Cookie)
  145. return
  146. }
  147. }
  148. // append to dat file
  149. n.UpdateAppendAtNs(v.lastAppendAtNs)
  150. var actualSize int64
  151. offset, size, actualSize, err = n.Append(v.DataBackend, v.Version())
  152. v.checkReadWriteError(err)
  153. if err != nil {
  154. err = fmt.Errorf("append to volume %d size %d actualSize %d: %v", v.Id, size, actualSize, err)
  155. return
  156. }
  157. v.lastAppendAtNs = n.AppendAtNs
  158. // add to needle map
  159. if !ok || uint64(nv.Offset.ToActualOffset()) < offset {
  160. if err = v.nm.Put(n.Id, ToOffset(int64(offset)), n.Size); err != nil {
  161. glog.V(4).Infof("failed to save in needle map %d: %v", n.Id, err)
  162. }
  163. }
  164. if v.lastModifiedTsSeconds < n.LastModified {
  165. v.lastModifiedTsSeconds = n.LastModified
  166. }
  167. return
  168. }
  169. func (v *Volume) syncDelete(n *needle.Needle) (Size, error) {
  170. // glog.V(4).Infof("delete needle %s", needle.NewFileIdFromNeedle(v.Id, n).String())
  171. v.dataFileAccessLock.Lock()
  172. defer v.dataFileAccessLock.Unlock()
  173. if v.nm == nil {
  174. return 0, nil
  175. }
  176. return v.doDeleteRequest(n)
  177. }
  178. func (v *Volume) deleteNeedle2(n *needle.Needle) (Size, error) {
  179. // todo: delete info is always appended no fsync, it may need fsync in future
  180. fsync := false
  181. if !fsync {
  182. return v.syncDelete(n)
  183. } else {
  184. asyncRequest := needle.NewAsyncRequest(n, false)
  185. asyncRequest.ActualSize = needle.GetActualSize(0, v.Version())
  186. v.asyncRequestAppend(asyncRequest)
  187. _, size, _, err := asyncRequest.WaitComplete()
  188. return Size(size), err
  189. }
  190. }
  191. func (v *Volume) doDeleteRequest(n *needle.Needle) (Size, error) {
  192. glog.V(4).Infof("delete needle %s", needle.NewFileIdFromNeedle(v.Id, n).String())
  193. nv, ok := v.nm.Get(n.Id)
  194. // fmt.Println("key", n.Id, "volume offset", nv.Offset, "data_size", n.Size, "cached size", nv.Size)
  195. if ok && nv.Size.IsValid() {
  196. var offset uint64
  197. var err error
  198. size := nv.Size
  199. if !v.hasRemoteFile {
  200. n.Data = nil
  201. n.UpdateAppendAtNs(v.lastAppendAtNs)
  202. offset, _, _, err = n.Append(v.DataBackend, v.Version())
  203. v.checkReadWriteError(err)
  204. if err != nil {
  205. return size, err
  206. }
  207. }
  208. v.lastAppendAtNs = n.AppendAtNs
  209. if err = v.nm.Delete(n.Id, ToOffset(int64(offset))); err != nil {
  210. return size, err
  211. }
  212. return size, err
  213. }
  214. return 0, nil
  215. }
  216. func (v *Volume) startWorker() {
  217. go func() {
  218. chanClosed := false
  219. for {
  220. // chan closed. go thread will exit
  221. if chanClosed {
  222. break
  223. }
  224. currentRequests := make([]*needle.AsyncRequest, 0, 128)
  225. currentBytesToWrite := int64(0)
  226. for {
  227. request, ok := <-v.asyncRequestsChan
  228. // volume may be closed
  229. if !ok {
  230. chanClosed = true
  231. break
  232. }
  233. if MaxPossibleVolumeSize < v.ContentSize()+uint64(currentBytesToWrite+request.ActualSize) {
  234. request.Complete(0, 0, false,
  235. fmt.Errorf("volume size limit %d exceeded! current size is %d", MaxPossibleVolumeSize, v.ContentSize()))
  236. break
  237. }
  238. currentRequests = append(currentRequests, request)
  239. currentBytesToWrite += request.ActualSize
  240. // submit at most 4M bytes or 128 requests at one time to decrease request delay.
  241. // it also need to break if there is no data in channel to avoid io hang.
  242. if currentBytesToWrite >= 4*1024*1024 || len(currentRequests) >= 128 || len(v.asyncRequestsChan) == 0 {
  243. break
  244. }
  245. }
  246. if len(currentRequests) == 0 {
  247. continue
  248. }
  249. v.dataFileAccessLock.Lock()
  250. end, _, e := v.DataBackend.GetStat()
  251. if e != nil {
  252. for i := 0; i < len(currentRequests); i++ {
  253. currentRequests[i].Complete(0, 0, false,
  254. fmt.Errorf("cannot read current volume position: %v", e))
  255. }
  256. v.dataFileAccessLock.Unlock()
  257. continue
  258. }
  259. for i := 0; i < len(currentRequests); i++ {
  260. if currentRequests[i].IsWriteRequest {
  261. offset, size, isUnchanged, err := v.doWriteRequest(currentRequests[i].N, true)
  262. currentRequests[i].UpdateResult(offset, uint64(size), isUnchanged, err)
  263. } else {
  264. size, err := v.doDeleteRequest(currentRequests[i].N)
  265. currentRequests[i].UpdateResult(0, uint64(size), false, err)
  266. }
  267. }
  268. // if sync error, data is not reliable, we should mark the completed request as fail and rollback
  269. if err := v.DataBackend.Sync(); err != nil {
  270. // todo: this may generate dirty data or cause data inconsistent, may be weed need to panic?
  271. if te := v.DataBackend.Truncate(end); te != nil {
  272. glog.V(0).Infof("Failed to truncate %s back to %d with error: %v", v.DataBackend.Name(), end, te)
  273. }
  274. for i := 0; i < len(currentRequests); i++ {
  275. if currentRequests[i].IsSucceed() {
  276. currentRequests[i].UpdateResult(0, 0, false, err)
  277. }
  278. }
  279. }
  280. for i := 0; i < len(currentRequests); i++ {
  281. currentRequests[i].Submit()
  282. }
  283. v.dataFileAccessLock.Unlock()
  284. }
  285. }()
  286. }
  287. func (v *Volume) WriteNeedleBlob(needleId NeedleId, needleBlob []byte, size Size) error {
  288. v.dataFileAccessLock.Lock()
  289. defer v.dataFileAccessLock.Unlock()
  290. if MaxPossibleVolumeSize < v.nm.ContentSize()+uint64(len(needleBlob)) {
  291. return fmt.Errorf("volume size limit %d exceeded! current size is %d", MaxPossibleVolumeSize, v.nm.ContentSize())
  292. }
  293. nv, ok := v.nm.Get(needleId)
  294. if ok && nv.Size == size {
  295. oldNeedle := new(needle.Needle)
  296. err := oldNeedle.ReadData(v.DataBackend, nv.Offset.ToActualOffset(), nv.Size, v.Version())
  297. if err == nil {
  298. newNeedle := new(needle.Needle)
  299. err = newNeedle.ReadBytes(needleBlob, nv.Offset.ToActualOffset(), size, v.Version())
  300. if err == nil && oldNeedle.Cookie == newNeedle.Cookie && oldNeedle.Checksum == newNeedle.Checksum && bytes.Equal(oldNeedle.Data, newNeedle.Data) {
  301. glog.V(0).Infof("needle %v already exists", needleId)
  302. return nil
  303. }
  304. }
  305. }
  306. appendAtNs := needle.GetAppendAtNs(v.lastAppendAtNs)
  307. offset, err := needle.WriteNeedleBlob(v.DataBackend, needleBlob, size, appendAtNs, v.Version())
  308. v.checkReadWriteError(err)
  309. if err != nil {
  310. return err
  311. }
  312. v.lastAppendAtNs = appendAtNs
  313. // add to needle map
  314. if err = v.nm.Put(needleId, ToOffset(int64(offset)), size); err != nil {
  315. glog.V(4).Infof("failed to put in needle map %d: %v", needleId, err)
  316. }
  317. return err
  318. }