Fix unstoppable kubewatcher (#208)

Fix a bug due to which watches could never be stopped; simplify control flow a bit.
This commit is contained in:
Ta-Ching Chen
2017-07-10 16:01:48 -07:00
committed by Soam Vasani
parent 51c84f3cfc
commit 46ff55ab49
+37 -16
View File
@@ -191,8 +191,7 @@ func (kw *KubeWatcher) removeWatch(w *fission.Watch) error {
fmt.Sprintf("watch doesn't exist: %v", w.Metadata)) fmt.Sprintf("watch doesn't exist: %v", w.Metadata))
} }
delete(kw.watches, w.Metadata.Uid) delete(kw.watches, w.Metadata.Uid)
atomic.StoreInt32(ws.stopped, 1) ws.stop()
ws.kubeWatch.Stop()
return nil return nil
} }
@@ -262,25 +261,46 @@ func getResourceVersion(obj runtime.Object) (string, error) {
func (ws *watchSubscription) eventDispatchLoop() { func (ws *watchSubscription) eventDispatchLoop() {
log.Println("Listening to watch ", ws.Watch.Metadata.Name) log.Println("Listening to watch ", ws.Watch.Metadata.Name)
for { for {
for { // check watchSubscription is stopped or not before waiting for event
ev, more := <-ws.kubeWatch.ResultChan() // comes from the kubeWatch.ResultChan(). This fix the edge case that
if !more { // new kubewatch is created in the restartWatch() while the old kubewatch
log.Println("Watch stopped", ws.Watch.Metadata.Name) // is being used in watchSubscription.stop().
if ws.isStopped() {
break break
} }
ev, more := <-ws.kubeWatch.ResultChan()
if !more {
if ws.isStopped() {
// watch is removed by user.
log.Println("Watch stopped", ws.Watch.Metadata.Name)
return
} else {
// watch closed due to timeout, restart it.
log.Printf("Watch %v timed out, restarting", ws.Watch.Metadata.Name)
err := ws.restartWatch()
if err != nil {
log.Panicf("Failed to restart watch: %v", err)
}
continue
}
}
if ev.Type == watch.Error { if ev.Type == watch.Error {
e := apierrs.FromObject(ev.Object) e := apierrs.FromObject(ev.Object)
log.Println("Watch error, retrying in a second: %v", e) log.Println("Watch error, retrying in a second: %v", e)
// Start from the beginning to get around "too old resource version" // Start from the beginning to get around "too old resource version"
ws.lastResourceVersion = "" ws.lastResourceVersion = ""
time.Sleep(time.Second) time.Sleep(time.Second)
break err := ws.restartWatch()
if err != nil {
log.Panicf("Failed to restart watch: %v", err)
}
continue
} }
rv, err := getResourceVersion(ev.Object) rv, err := getResourceVersion(ev.Object)
if err != nil { if err != nil {
log.Printf("Error getting resourceVersion from object: %v", err) log.Printf("Error getting resourceVersion from object: %v", err)
} else { } else {
log.Printf("rv=%v", rv)
ws.lastResourceVersion = rv ws.lastResourceVersion = rv
} }
@@ -298,14 +318,15 @@ func (ws *watchSubscription) eventDispatchLoop() {
"X-Kubernetes-Event-Type": string(ev.Type), "X-Kubernetes-Event-Type": string(ev.Type),
"X-Kubernetes-Object-Type": reflect.TypeOf(ev.Object).Elem().Name(), "X-Kubernetes-Object-Type": reflect.TypeOf(ev.Object).Elem().Name(),
} }
// Event and object type aren't in the serialized object
ws.publisher.Publish(buf.String(), headers, ws.Watch.Target) ws.publisher.Publish(buf.String(), headers, ws.Watch.Target)
} }
if atomic.LoadInt32(ws.stopped) == 0 { }
err := ws.restartWatch()
if err != nil { func (ws *watchSubscription) stop() {
log.Panicf("Failed to restart watch: %v", err) atomic.StoreInt32(ws.stopped, 1)
} ws.kubeWatch.Stop()
} }
}
func (ws *watchSubscription) isStopped() bool {
return atomic.LoadInt32(ws.stopped) == 1
} }