volumes.go 9.1 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318
  1. package storage
  2. /*
  3. Volumes: folders on this node's local drives that hold cluster files.
  4. */
  5. import (
  6. "encoding/json"
  7. "errors"
  8. "fmt"
  9. "os"
  10. "path/filepath"
  11. "strings"
  12. "time"
  13. uuid "github.com/satori/go.uuid"
  14. "imuslab.com/arozos/mod/cluster/capability"
  15. "imuslab.com/arozos/mod/cluster/metadata"
  16. "imuslab.com/arozos/mod/info/logger"
  17. )
  18. const (
  19. //diskFullFraction is the free space below which a volume stops taking
  20. //new files; it starts again above diskRecoverFraction.
  21. diskFullFraction = 0.05
  22. diskRecoverFraction = 0.07
  23. )
  24. // LocalVolumes lists this node's volumes that are not removed.
  25. func (s *Service) LocalVolumes() []metadata.Volume {
  26. out := []metadata.Volume{}
  27. for _, v := range s.meta.Volumes() {
  28. if v.NodeID == s.m.NodeID() && !v.Removed {
  29. out = append(out, v)
  30. }
  31. }
  32. return out
  33. }
  34. // volumeRoot returns the real directory of a local volume.
  35. func (s *Service) volumeRoot(vol *metadata.Volume) (string, error) {
  36. if vol.NodeID != s.m.NodeID() {
  37. return "", ErrNotLocalVolume
  38. }
  39. roots := s.opt.LocalRoots()
  40. root, ok := roots[vol.FshUUID]
  41. if !ok {
  42. return "", errors.New("local drive " + vol.FshUUID + " is not mounted")
  43. }
  44. return filepath.Clean(filepath.Join(root, filepath.FromSlash(vol.Subpath))), nil
  45. }
  46. // realPath maps a logical path onto a local volume, refusing escapes.
  47. func (s *Service) realPath(vol *metadata.Volume, logical string) (string, error) {
  48. root, err := s.volumeRoot(vol)
  49. if err != nil {
  50. return "", err
  51. }
  52. logical = metadata.NormalizePath(logical)
  53. full := filepath.Clean(filepath.Join(root, filepath.FromSlash(logical)))
  54. if full != root && !strings.HasPrefix(full, root+string(filepath.Separator)) {
  55. return "", ErrPathEscape
  56. }
  57. return full, nil
  58. }
  59. func normalizeSubpath(p string) string {
  60. p = strings.ReplaceAll(strings.TrimSpace(p), "\\", "/")
  61. if !strings.HasPrefix(p, "/") {
  62. p = "/" + p
  63. }
  64. p = filepath.ToSlash(filepath.Clean(p))
  65. if p == "." {
  66. p = "/"
  67. }
  68. return p
  69. }
  70. // AddVolume contributes a folder of a local drive to the cluster.
  71. func (s *Service) AddVolume(fshUUID string, subpath string, name string) (*metadata.Volume, error) {
  72. if !s.m.InCluster() {
  73. return nil, metadata.ErrNotInCluster
  74. }
  75. roots := s.opt.LocalRoots()
  76. root, ok := roots[fshUUID]
  77. if !ok {
  78. return nil, errors.New("drive " + fshUUID + " is not a local drive on this node")
  79. }
  80. subpath = normalizeSubpath(subpath)
  81. if strings.Contains(subpath, "..") {
  82. return nil, ErrPathEscape
  83. }
  84. dir := filepath.Clean(filepath.Join(root, filepath.FromSlash(subpath)))
  85. if subpath == "/" && fshUUID == "user" {
  86. return nil, errors.New("the user root itself cannot be a cluster volume; pick a sub folder such as /cluster")
  87. }
  88. for _, v := range s.meta.Volumes() {
  89. if !v.Removed && v.NodeID == s.m.NodeID() && v.FshUUID == fshUUID && v.Subpath == subpath {
  90. return nil, errors.New("this folder is already a cluster volume")
  91. }
  92. }
  93. if err := os.MkdirAll(dir, 0755); err != nil {
  94. return nil, err
  95. }
  96. if strings.TrimSpace(name) == "" {
  97. name = s.m.NodeName(s.m.NodeID()) + " " + fshUUID + ":" + subpath
  98. }
  99. vol := &metadata.Volume{
  100. ID: uuid.NewV4().String(),
  101. NodeID: s.m.NodeID(),
  102. Name: strings.TrimSpace(name),
  103. FshUUID: fshUUID,
  104. Subpath: subpath,
  105. }
  106. if free, total, err := capability.DiskUsage(dir); err == nil {
  107. vol.Free, vol.Capacity = free, total
  108. }
  109. if err := s.meta.Submit(metadata.KindVolume, vol); err != nil {
  110. return nil, err
  111. }
  112. go s.reconcileVolume(*vol)
  113. return vol, nil
  114. }
  115. // RemoveVolume stops contributing a folder. Files stay on disk. It refuses
  116. // when the volume holds the only healthy copy of any file; evacuate first.
  117. func (s *Service) RemoveVolume(id string) error {
  118. vol, ok := s.meta.Volume(id)
  119. if !ok || vol.Removed {
  120. return errors.New("volume not found")
  121. }
  122. if vol.NodeID != s.m.NodeID() {
  123. return ErrNotLocalVolume
  124. }
  125. if sole := s.SoleCopies(id); sole > 0 {
  126. return fmt.Errorf("this volume holds the only copy of %d file(s); evacuate it first", sole)
  127. }
  128. vol.Removed = true
  129. if err := s.meta.Submit(metadata.KindVolume, vol); err != nil {
  130. return err
  131. }
  132. s.forgetVolumeLocations(id)
  133. return nil
  134. }
  135. // forgetVolumeLocations drops a retired volume from every record. The files
  136. // stay on disk; they are simply no longer reachable through the cluster.
  137. func (s *Service) forgetVolumeLocations(volumeID string) {
  138. for _, rec := range s.meta.AllFiles() {
  139. if _, has := rec.Location(volumeID); !has {
  140. continue
  141. }
  142. r := rec.Clone()
  143. r.RemoveLocation(volumeID)
  144. if r.Primary == volumeID {
  145. r.Primary = ""
  146. if h := r.HealthyLocations(); len(h) > 0 {
  147. r.Primary = h[0].VolumeID
  148. }
  149. }
  150. s.meta.Submit(metadata.KindFile, r)
  151. }
  152. }
  153. // SoleCopies counts files whose only healthy copy sits on the volume.
  154. func (s *Service) SoleCopies(volumeID string) int {
  155. n := 0
  156. for _, rec := range s.meta.AllFiles() {
  157. if rec.Removed || rec.IsDir {
  158. continue
  159. }
  160. healthy := rec.HealthyLocations()
  161. if len(healthy) == 1 && healthy[0].VolumeID == volumeID {
  162. n++
  163. }
  164. }
  165. return n
  166. }
  167. // SetVolumeReadOnly toggles placement on a local volume.
  168. func (s *Service) SetVolumeReadOnly(id string, ro bool) error {
  169. vol, ok := s.meta.Volume(id)
  170. if !ok || vol.Removed {
  171. return errors.New("volume not found")
  172. }
  173. if vol.NodeID != s.m.NodeID() {
  174. return ErrNotLocalVolume
  175. }
  176. vol.ReadOnly = ro
  177. return s.meta.Submit(metadata.KindVolume, vol)
  178. }
  179. // refreshVolumes republishes free space when it moved by more than 1 %.
  180. func (s *Service) refreshVolumes() {
  181. auto := s.AutoReadOnly()
  182. for _, v := range s.LocalVolumes() {
  183. root, err := s.volumeRoot(&v)
  184. if err != nil {
  185. continue
  186. }
  187. free, total, err := capability.DiskUsage(root)
  188. if err != nil {
  189. continue
  190. }
  191. delta := v.Free - free
  192. if delta < 0 {
  193. delta = -delta
  194. }
  195. //A volume under the low water mark stops taking new files until it
  196. //recovers, so a node never fills its own disk (unless an admin turned
  197. //that off for the cluster).
  198. wasFull := v.ReadOnly && v.LowSpace
  199. mark, clear := lowSpaceAction(free, total, wasFull, auto)
  200. changed := total != v.Capacity || (total > 0 && delta*100 > total)
  201. if mark {
  202. v.ReadOnly, v.LowSpace, changed = true, true, true
  203. logger.PrintAndLog("Cluster", "Volume "+v.Name+" is nearly full and stops taking new files", nil)
  204. if s.OnDiskFull != nil {
  205. s.OnDiskFull(v)
  206. }
  207. } else if clear {
  208. v.ReadOnly, v.LowSpace, changed = false, false, true
  209. if auto {
  210. logger.PrintAndLog("Cluster", "Volume "+v.Name+" has room again and takes new files", nil)
  211. } else {
  212. logger.PrintAndLog("Cluster", "Volume "+v.Name+" takes new files again: automatic read-only is turned off", nil)
  213. }
  214. }
  215. if changed {
  216. v.Free, v.Capacity = free, total
  217. s.meta.Submit(metadata.KindVolume, &v)
  218. }
  219. }
  220. }
  221. // AutoReadOnlyKey is the replicated setting that turns the nearly-full guard
  222. // on or off for the whole cluster.
  223. const AutoReadOnlyKey = "storage.autoReadOnly"
  224. type autoReadOnlySetting struct {
  225. Enabled bool `json:"enabled"`
  226. }
  227. // AutoReadOnly reports whether volumes that are nearly full are made read
  228. // only automatically. It is on unless an admin turned it off.
  229. func (s *Service) AutoReadOnly() bool {
  230. st, ok := s.meta.Setting(AutoReadOnlyKey)
  231. if !ok {
  232. return true
  233. }
  234. var v autoReadOnlySetting
  235. if json.Unmarshal(st.Value, &v) != nil {
  236. return true
  237. }
  238. return v.Enabled
  239. }
  240. // SetAutoReadOnly turns the nearly-full guard on or off for every node. This
  241. // node applies it at once; the others at their next volume refresh.
  242. func (s *Service) SetAutoReadOnly(enabled bool) error {
  243. js, err := json.Marshal(autoReadOnlySetting{Enabled: enabled})
  244. if err != nil {
  245. return err
  246. }
  247. if err := s.meta.Submit(metadata.KindSetting, &metadata.Setting{Key: AutoReadOnlyKey, Value: js}); err != nil {
  248. return err
  249. }
  250. go s.refreshVolumes()
  251. return nil
  252. }
  253. // lowSpaceAction decides what the guard does to one volume. wasFull says the
  254. // guard itself made the volume read only earlier. With the guard on, a volume
  255. // is marked below the low water mark and released above the recover mark;
  256. // with it off, nothing is marked and anything the guard had marked is
  257. // released, while read-only set by an admin is never touched.
  258. func lowSpaceAction(free int64, total int64, wasFull bool, auto bool) (mark bool, clear bool) {
  259. if !auto {
  260. return false, wasFull
  261. }
  262. full, recovered := spaceState(free, total, wasFull)
  263. return full && !wasFull, recovered
  264. }
  265. // spaceState decides whether a volume is too full to take new files. The two
  266. // thresholds differ so a volume hovering at the limit does not flap: it stops
  267. // below diskFullFraction and only starts again above diskRecoverFraction.
  268. func spaceState(free int64, total int64, wasFull bool) (full bool, recovered bool) {
  269. if total <= 0 {
  270. return false, false
  271. }
  272. frac := float64(free) / float64(total)
  273. if wasFull {
  274. return true, frac > diskRecoverFraction
  275. }
  276. return frac < diskFullFraction, false
  277. }
  278. // VolumeStats counts files held on a local volume (for the UI).
  279. func (s *Service) VolumeStats(id string) (files int, bytes int64) {
  280. for _, rec := range s.meta.AllFiles() {
  281. if rec.Removed || rec.IsDir {
  282. continue
  283. }
  284. if loc, ok := rec.Location(id); ok && loc.Healthy() {
  285. files++
  286. bytes += rec.Size
  287. }
  288. }
  289. return
  290. }
  291. func modTimeOf(fi os.FileInfo) int64 {
  292. if fi == nil {
  293. return time.Now().Unix()
  294. }
  295. return fi.ModTime().Unix()
  296. }