2
0

worker.go 10 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441
  1. package worker
  2. import (
  3. "fmt"
  4. "net/http"
  5. "os"
  6. "sync"
  7. "syscall"
  8. "time"
  9. "github.com/gin-gonic/gin"
  10. . "github.com/tuna/tunasync/internal"
  11. )
  12. var tunasyncWorker *Worker
  13. // A Worker is a instance of tunasync worker
  14. type Worker struct {
  15. L sync.Mutex
  16. cfg *Config
  17. jobs map[string]*mirrorJob
  18. managerChan chan jobMessage
  19. semaphore chan empty
  20. exit chan empty
  21. schedule *scheduleQueue
  22. httpEngine *gin.Engine
  23. httpClient *http.Client
  24. }
  25. // GetTUNASyncWorker returns a singalton worker
  26. func GetTUNASyncWorker(cfg *Config) *Worker {
  27. if tunasyncWorker != nil {
  28. return tunasyncWorker
  29. }
  30. w := &Worker{
  31. cfg: cfg,
  32. jobs: make(map[string]*mirrorJob),
  33. managerChan: make(chan jobMessage, 32),
  34. semaphore: make(chan empty, cfg.Global.Concurrent),
  35. exit: make(chan empty),
  36. schedule: newScheduleQueue(),
  37. }
  38. if cfg.Manager.CACert != "" {
  39. httpClient, err := CreateHTTPClient(cfg.Manager.CACert)
  40. if err != nil {
  41. logger.Errorf("Error initializing HTTP client: %s", err.Error())
  42. return nil
  43. }
  44. w.httpClient = httpClient
  45. }
  46. w.initJobs()
  47. w.makeHTTPServer()
  48. tunasyncWorker = w
  49. return w
  50. }
  51. // Run runs worker forever
  52. func (w *Worker) Run() {
  53. w.registorWorker()
  54. go w.runHTTPServer()
  55. w.runSchedule()
  56. }
  57. // Halt stops all jobs
  58. func (w *Worker) Halt() {
  59. w.L.Lock()
  60. logger.Notice("Stopping all the jobs")
  61. for _, job := range w.jobs {
  62. if job.State() != stateDisabled {
  63. job.ctrlChan <- jobHalt
  64. }
  65. }
  66. jobsDone.Wait()
  67. logger.Notice("All the jobs are stopped")
  68. w.L.Unlock()
  69. close(w.exit)
  70. }
  71. // ReloadMirrorConfig refresh the providers and jobs
  72. // from new mirror configs
  73. // TODO: deleted job should be removed from manager-side mirror list
  74. func (w *Worker) ReloadMirrorConfig(newMirrors []mirrorConfig) {
  75. w.L.Lock()
  76. defer w.L.Unlock()
  77. logger.Info("Reloading mirror configs")
  78. oldMirrors := w.cfg.Mirrors
  79. difference := diffMirrorConfig(oldMirrors, newMirrors)
  80. // first deal with deletion and modifications
  81. for _, op := range difference {
  82. if op.diffOp == diffAdd {
  83. continue
  84. }
  85. name := op.mirCfg.Name
  86. job, ok := w.jobs[name]
  87. if !ok {
  88. logger.Warningf("Job %s not found", name)
  89. continue
  90. }
  91. switch op.diffOp {
  92. case diffDelete:
  93. w.disableJob(job)
  94. delete(w.jobs, name)
  95. logger.Noticef("Deleted job %s", name)
  96. case diffModify:
  97. jobState := job.State()
  98. w.disableJob(job)
  99. // set new provider
  100. provider := newMirrorProvider(op.mirCfg, w.cfg)
  101. if err := job.SetProvider(provider); err != nil {
  102. logger.Errorf("Error setting job provider of %s: %s", name, err.Error())
  103. continue
  104. }
  105. // re-schedule job according to its previous state
  106. if jobState == stateDisabled {
  107. job.SetState(stateDisabled)
  108. } else if jobState == statePaused {
  109. job.SetState(statePaused)
  110. go job.Run(w.managerChan, w.semaphore)
  111. } else {
  112. job.SetState(stateNone)
  113. go job.Run(w.managerChan, w.semaphore)
  114. w.schedule.AddJob(time.Now(), job)
  115. }
  116. logger.Noticef("Reloaded job %s", name)
  117. }
  118. }
  119. // for added new jobs, just start new jobs
  120. for _, op := range difference {
  121. if op.diffOp != diffAdd {
  122. continue
  123. }
  124. provider := newMirrorProvider(op.mirCfg, w.cfg)
  125. job := newMirrorJob(provider)
  126. w.jobs[provider.Name()] = job
  127. job.SetState(stateNone)
  128. go job.Run(w.managerChan, w.semaphore)
  129. w.schedule.AddJob(time.Now(), job)
  130. logger.Noticef("New job %s", job.Name())
  131. }
  132. w.cfg.Mirrors = newMirrors
  133. }
  134. func (w *Worker) initJobs() {
  135. for _, mirror := range w.cfg.Mirrors {
  136. // Create Provider
  137. provider := newMirrorProvider(mirror, w.cfg)
  138. w.jobs[provider.Name()] = newMirrorJob(provider)
  139. }
  140. }
  141. func (w *Worker) disableJob(job *mirrorJob) {
  142. w.schedule.Remove(job.Name())
  143. if job.State() != stateDisabled {
  144. job.ctrlChan <- jobDisable
  145. <-job.disabled
  146. }
  147. }
  148. // Ctrl server receives commands from the manager
  149. func (w *Worker) makeHTTPServer() {
  150. s := gin.New()
  151. s.Use(gin.Recovery())
  152. s.POST("/", func(c *gin.Context) {
  153. w.L.Lock()
  154. defer w.L.Unlock()
  155. var cmd WorkerCmd
  156. if err := c.BindJSON(&cmd); err != nil {
  157. c.JSON(http.StatusBadRequest, gin.H{"msg": "Invalid request"})
  158. return
  159. }
  160. logger.Noticef("Received command: %v", cmd)
  161. if cmd.MirrorID == "" {
  162. // worker-level commands
  163. switch cmd.Cmd {
  164. case CmdReload:
  165. // send myself a SIGHUP
  166. pid := os.Getpid()
  167. syscall.Kill(pid, syscall.SIGHUP)
  168. default:
  169. c.JSON(http.StatusNotAcceptable, gin.H{"msg": "Invalid Command"})
  170. return
  171. }
  172. }
  173. // job level comands
  174. job, ok := w.jobs[cmd.MirrorID]
  175. if !ok {
  176. c.JSON(http.StatusNotFound, gin.H{"msg": fmt.Sprintf("Mirror ``%s'' not found", cmd.MirrorID)})
  177. return
  178. }
  179. // No matter what command, the existing job
  180. // schedule should be flushed
  181. w.schedule.Remove(job.Name())
  182. // if job disabled, start them first
  183. switch cmd.Cmd {
  184. case CmdStart, CmdRestart:
  185. if job.State() == stateDisabled {
  186. go job.Run(w.managerChan, w.semaphore)
  187. }
  188. }
  189. switch cmd.Cmd {
  190. case CmdStart:
  191. if cmd.Options["force"] {
  192. job.ctrlChan <- jobForceStart
  193. } else {
  194. job.ctrlChan <- jobStart
  195. }
  196. case CmdRestart:
  197. job.ctrlChan <- jobRestart
  198. case CmdStop:
  199. // if job is disabled, no goroutine would be there
  200. // receiving this signal
  201. if job.State() != stateDisabled {
  202. job.ctrlChan <- jobStop
  203. }
  204. case CmdDisable:
  205. w.disableJob(job)
  206. case CmdPing:
  207. // empty
  208. default:
  209. c.JSON(http.StatusNotAcceptable, gin.H{"msg": "Invalid Command"})
  210. return
  211. }
  212. c.JSON(http.StatusOK, gin.H{"msg": "OK"})
  213. })
  214. w.httpEngine = s
  215. }
  216. func (w *Worker) runHTTPServer() {
  217. addr := fmt.Sprintf("%s:%d", w.cfg.Server.Addr, w.cfg.Server.Port)
  218. httpServer := &http.Server{
  219. Addr: addr,
  220. Handler: w.httpEngine,
  221. ReadTimeout: 10 * time.Second,
  222. WriteTimeout: 10 * time.Second,
  223. }
  224. if w.cfg.Server.SSLCert == "" && w.cfg.Server.SSLKey == "" {
  225. if err := httpServer.ListenAndServe(); err != nil {
  226. panic(err)
  227. }
  228. } else {
  229. if err := httpServer.ListenAndServeTLS(w.cfg.Server.SSLCert, w.cfg.Server.SSLKey); err != nil {
  230. panic(err)
  231. }
  232. }
  233. }
  234. func (w *Worker) runSchedule() {
  235. w.L.Lock()
  236. mirrorList := w.fetchJobStatus()
  237. unset := make(map[string]bool)
  238. for name := range w.jobs {
  239. unset[name] = true
  240. }
  241. // Fetch mirror list stored in the manager
  242. // put it on the scheduled time
  243. // if it's disabled, ignore it
  244. for _, m := range mirrorList {
  245. if job, ok := w.jobs[m.Name]; ok {
  246. delete(unset, m.Name)
  247. switch m.Status {
  248. case Disabled:
  249. job.SetState(stateDisabled)
  250. continue
  251. case Paused:
  252. job.SetState(statePaused)
  253. go job.Run(w.managerChan, w.semaphore)
  254. continue
  255. default:
  256. job.SetState(stateNone)
  257. go job.Run(w.managerChan, w.semaphore)
  258. stime := m.LastUpdate.Add(job.provider.Interval())
  259. logger.Debugf("Scheduling job %s @%s", job.Name(), stime.Format("2006-01-02 15:04:05"))
  260. w.schedule.AddJob(stime, job)
  261. }
  262. }
  263. }
  264. // some new jobs may be added
  265. // which does not exist in the
  266. // manager's mirror list
  267. for name := range unset {
  268. job := w.jobs[name]
  269. job.SetState(stateNone)
  270. go job.Run(w.managerChan, w.semaphore)
  271. w.schedule.AddJob(time.Now(), job)
  272. }
  273. w.L.Unlock()
  274. tick := time.Tick(5 * time.Second)
  275. for {
  276. select {
  277. case jobMsg := <-w.managerChan:
  278. // got status update from job
  279. w.L.Lock()
  280. job, ok := w.jobs[jobMsg.name]
  281. w.L.Unlock()
  282. if !ok {
  283. logger.Warningf("Job %s not found", jobMsg.name)
  284. continue
  285. }
  286. if (job.State() != stateReady) && (job.State() != stateHalting) {
  287. logger.Infof("Job %s state is not ready, skip adding new schedule", jobMsg.name)
  288. continue
  289. }
  290. // syncing status is only meaningful when job
  291. // is running. If it's paused or disabled
  292. // a sync failure signal would be emitted
  293. // which needs to be ignored
  294. w.updateStatus(job, jobMsg)
  295. // only successful or the final failure msg
  296. // can trigger scheduling
  297. if jobMsg.schedule {
  298. schedTime := time.Now().Add(job.provider.Interval())
  299. logger.Noticef(
  300. "Next scheduled time for %s: %s",
  301. job.Name(),
  302. schedTime.Format("2006-01-02 15:04:05"),
  303. )
  304. w.schedule.AddJob(schedTime, job)
  305. }
  306. case <-tick:
  307. // check schedule every 5 seconds
  308. if job := w.schedule.Pop(); job != nil {
  309. job.ctrlChan <- jobStart
  310. }
  311. case <-w.exit:
  312. // flush status update messages
  313. w.L.Lock()
  314. defer w.L.Unlock()
  315. for {
  316. select {
  317. case jobMsg := <-w.managerChan:
  318. logger.Debugf("status update from %s", jobMsg.name)
  319. job, ok := w.jobs[jobMsg.name]
  320. if !ok {
  321. continue
  322. }
  323. if jobMsg.status == Failed || jobMsg.status == Success {
  324. w.updateStatus(job, jobMsg)
  325. }
  326. default:
  327. return
  328. }
  329. }
  330. }
  331. }
  332. }
  333. // Name returns worker name
  334. func (w *Worker) Name() string {
  335. return w.cfg.Global.Name
  336. }
  337. // URL returns the url to http server of the worker
  338. func (w *Worker) URL() string {
  339. proto := "https"
  340. if w.cfg.Server.SSLCert == "" && w.cfg.Server.SSLKey == "" {
  341. proto = "http"
  342. }
  343. return fmt.Sprintf("%s://%s:%d/", proto, w.cfg.Server.Hostname, w.cfg.Server.Port)
  344. }
  345. func (w *Worker) registorWorker() {
  346. msg := WorkerStatus{
  347. ID: w.Name(),
  348. URL: w.URL(),
  349. }
  350. for _, root := range w.cfg.Manager.APIBaseList() {
  351. url := fmt.Sprintf("%s/workers", root)
  352. logger.Debugf("register on manager url: %s", url)
  353. if _, err := PostJSON(url, msg, w.httpClient); err != nil {
  354. logger.Errorf("Failed to register worker")
  355. }
  356. }
  357. }
  358. func (w *Worker) updateStatus(job *mirrorJob, jobMsg jobMessage) {
  359. p := job.provider
  360. smsg := MirrorStatus{
  361. Name: jobMsg.name,
  362. Worker: w.cfg.Global.Name,
  363. IsMaster: p.IsMaster(),
  364. Status: jobMsg.status,
  365. Upstream: p.Upstream(),
  366. Size: "unknown",
  367. ErrorMsg: jobMsg.msg,
  368. }
  369. for _, root := range w.cfg.Manager.APIBaseList() {
  370. url := fmt.Sprintf(
  371. "%s/workers/%s/jobs/%s", root, w.Name(), jobMsg.name,
  372. )
  373. logger.Debugf("reporting on manager url: %s", url)
  374. if _, err := PostJSON(url, smsg, w.httpClient); err != nil {
  375. logger.Errorf("Failed to update mirror(%s) status: %s", jobMsg.name, err.Error())
  376. }
  377. }
  378. }
  379. func (w *Worker) fetchJobStatus() []MirrorStatus {
  380. var mirrorList []MirrorStatus
  381. apiBase := w.cfg.Manager.APIBaseList()[0]
  382. url := fmt.Sprintf("%s/workers/%s/jobs", apiBase, w.Name())
  383. if _, err := GetJSON(url, &mirrorList, w.httpClient); err != nil {
  384. logger.Errorf("Failed to fetch job status: %s", err.Error())
  385. }
  386. return mirrorList
  387. }