syncthing/cmd/stdiscosrv/stats.go

// Copyright (C) 2018 The Syncthing Authors.
//
// This Source Code Form is subject to the terms of the Mozilla Public
// License, v. 2.0. If a copy of the MPL was not distributed with this file,
// You can obtain one at https://mozilla.org/MPL/2.0/.

package main

import (
	"os"

	"github.com/prometheus/client_golang/prometheus"
)

var (
	apiRequestsTotal = prometheus.NewCounterVec(
		prometheus.CounterOpts{
			Namespace: "syncthing",
			Subsystem: "discovery",
			Name:      "api_requests_total",
			Help:      "Number of API requests.",
		}, []string{"type", "result"})
	apiRequestsSeconds = prometheus.NewSummaryVec(
		prometheus.SummaryOpts{
			Namespace:  "syncthing",
			Subsystem:  "discovery",
			Name:       "api_requests_seconds",
			Help:       "Latency of API requests.",
			Objectives: map[float64]float64{0.5: 0.05, 0.9: 0.01, 0.99: 0.001},
		}, []string{"type"})

	lookupRequestsTotal = prometheus.NewCounterVec(
		prometheus.CounterOpts{
			Namespace: "syncthing",
			Subsystem: "discovery",
			Name:      "lookup_requests_total",
			Help:      "Number of lookup requests.",
		}, []string{"result"})
	announceRequestsTotal = prometheus.NewCounterVec(
		prometheus.CounterOpts{
			Namespace: "syncthing",
			Subsystem: "discovery",
			Name:      "announcement_requests_total",
			Help:      "Number of announcement requests.",
		}, []string{"result"})

	replicationSendsTotal = prometheus.NewCounterVec(
		prometheus.CounterOpts{
			Namespace: "syncthing",
			Subsystem: "discovery",
			Name:      "replication_sends_total",
			Help:      "Number of replication sends.",
		}, []string{"result"})
	replicationRecvsTotal = prometheus.NewCounterVec(
		prometheus.CounterOpts{
			Namespace: "syncthing",
			Subsystem: "discovery",
			Name:      "replication_recvs_total",
			Help:      "Number of replication receives.",
		}, []string{"result"})

	databaseKeys = prometheus.NewGaugeVec(
		prometheus.GaugeOpts{
			Namespace: "syncthing",
			Subsystem: "discovery",
			Name:      "database_keys",
			Help:      "Number of database keys at last count.",
		}, []string{"category"})
	databaseStatisticsSeconds = prometheus.NewGauge(
		prometheus.GaugeOpts{
			Namespace: "syncthing",
			Subsystem: "discovery",
			Name:      "database_statistics_seconds",
			Help:      "Time spent running the statistics routine.",
		})

	databaseOperations = prometheus.NewCounterVec(
		prometheus.CounterOpts{
			Namespace: "syncthing",
			Subsystem: "discovery",
			Name:      "database_operations_total",
			Help:      "Number of database operations.",
		}, []string{"operation", "result"})
	databaseOperationSeconds = prometheus.NewSummaryVec(
		prometheus.SummaryOpts{
			Namespace:  "syncthing",
			Subsystem:  "discovery",
			Name:       "database_operation_seconds",
			Help:       "Latency of database operations.",
			Objectives: map[float64]float64{0.5: 0.05, 0.9: 0.01, 0.99: 0.001},
		}, []string{"operation"})
)

const (
	dbOpGet             = "get"
	dbOpPut             = "put"
	dbOpMerge           = "merge"
	dbOpDelete          = "delete"
	dbResSuccess        = "success"
	dbResNotFound       = "not_found"
	dbResError          = "error"
	dbResUnmarshalError = "unmarsh_err"
)

func init() {
	prometheus.MustRegister(apiRequestsTotal, apiRequestsSeconds,
		lookupRequestsTotal, announceRequestsTotal,
		replicationSendsTotal, replicationRecvsTotal,
		databaseKeys, databaseStatisticsSeconds,
		databaseOperations, databaseOperationSeconds)

	prometheus.MustRegister(prometheus.NewProcessCollector(os.Getpid(), "syncthing_discovery"))
}
cmd/stdiscosrv: New discovery server (fixes #4618) This is a new revision of the discovery server. Relevant changes and non-changes: - Protocol towards clients is unchanged. - Recommended large scale design is still to be deployed nehind nginx (I tested, and it's still a lot faster at terminating TLS). - Database backend is leveldb again, only. It scales enough, is easy to setup, and we don't need any backend to take care of. - Server supports replication. This is a simple TCP channel - protect it with a firewall when deploying over the internet. (We deploy this within the same datacenter, and with firewall.) Any incoming client announces are sent over the replication channel(s) to other peer discosrvs. Incoming replication changes are applied to the database as if they came from clients, but without the TLS/certificate overhead. - Metrics are exposed using the prometheus library, when enabled. - The database values and replication protocol is protobuf, because JSON was quite CPU intensive when I tried that and benchmarked it. - The "Retry-After" value for failed lookups gets slowly increased from a default of 120 seconds, by 5 seconds for each failed lookup, independently by each discosrv. This lowers the query load over time for clients that are never seen. The Retry-After maxes out at 3600 after a couple of weeks of this increase. The number of failed lookups is stored in the database, now and then (avoiding making each lookup a database put). All in all this means clients can be pointed towards a cluster using just multiple A / AAAA records to gain both load sharing and redundancy (if one is down, clients will talk to the remaining ones). GitHub-Pull-Request: https://github.com/syncthing/syncthing/pull/4648 2018-01-14 08:52:31 +00:00			`// Copyright (C) 2018 The Syncthing Authors.`
			`//`
			`// This Source Code Form is subject to the terms of the Mozilla Public`
			`// License, v. 2.0. If a copy of the MPL was not distributed with this file,`
			`// You can obtain one at https://mozilla.org/MPL/2.0/.`
Rewrite for a PostgreSQL backend 2015-03-25 07:16:52 +00:00
			`package main`

			`import (`
cmd/stdiscosrv: Expose process metrics Like this: $ curl -s http://localhost:9098/metrics \| egrep '^syncthing_discovery_process' syncthing_discovery_process_cpu_seconds_total 12.92 syncthing_discovery_process_max_fds 10240 syncthing_discovery_process_open_fds 51 syncthing_discovery_process_resident_memory_bytes 1.3674496e+08 syncthing_discovery_process_start_time_seconds 1.52034731837e+09 syncthing_discovery_process_virtual_memory_bytes 1.40324864e+08 2018-03-06 14:43:36 +00:00			`"os"`

cmd/stdiscosrv: New discovery server (fixes #4618) This is a new revision of the discovery server. Relevant changes and non-changes: - Protocol towards clients is unchanged. - Recommended large scale design is still to be deployed nehind nginx (I tested, and it's still a lot faster at terminating TLS). - Database backend is leveldb again, only. It scales enough, is easy to setup, and we don't need any backend to take care of. - Server supports replication. This is a simple TCP channel - protect it with a firewall when deploying over the internet. (We deploy this within the same datacenter, and with firewall.) Any incoming client announces are sent over the replication channel(s) to other peer discosrvs. Incoming replication changes are applied to the database as if they came from clients, but without the TLS/certificate overhead. - Metrics are exposed using the prometheus library, when enabled. - The database values and replication protocol is protobuf, because JSON was quite CPU intensive when I tried that and benchmarked it. - The "Retry-After" value for failed lookups gets slowly increased from a default of 120 seconds, by 5 seconds for each failed lookup, independently by each discosrv. This lowers the query load over time for clients that are never seen. The Retry-After maxes out at 3600 after a couple of weeks of this increase. The number of failed lookups is stored in the database, now and then (avoiding making each lookup a database put). All in all this means clients can be pointed towards a cluster using just multiple A / AAAA records to gain both load sharing and redundancy (if one is down, clients will talk to the remaining ones). GitHub-Pull-Request: https://github.com/syncthing/syncthing/pull/4648 2018-01-14 08:52:31 +00:00			`"github.com/prometheus/client_golang/prometheus"`
Rewrite for a PostgreSQL backend 2015-03-25 07:16:52 +00:00			`)`

cmd/stdiscosrv: New discovery server (fixes #4618) This is a new revision of the discovery server. Relevant changes and non-changes: - Protocol towards clients is unchanged. - Recommended large scale design is still to be deployed nehind nginx (I tested, and it's still a lot faster at terminating TLS). - Database backend is leveldb again, only. It scales enough, is easy to setup, and we don't need any backend to take care of. - Server supports replication. This is a simple TCP channel - protect it with a firewall when deploying over the internet. (We deploy this within the same datacenter, and with firewall.) Any incoming client announces are sent over the replication channel(s) to other peer discosrvs. Incoming replication changes are applied to the database as if they came from clients, but without the TLS/certificate overhead. - Metrics are exposed using the prometheus library, when enabled. - The database values and replication protocol is protobuf, because JSON was quite CPU intensive when I tried that and benchmarked it. - The "Retry-After" value for failed lookups gets slowly increased from a default of 120 seconds, by 5 seconds for each failed lookup, independently by each discosrv. This lowers the query load over time for clients that are never seen. The Retry-After maxes out at 3600 after a couple of weeks of this increase. The number of failed lookups is stored in the database, now and then (avoiding making each lookup a database put). All in all this means clients can be pointed towards a cluster using just multiple A / AAAA records to gain both load sharing and redundancy (if one is down, clients will talk to the remaining ones). GitHub-Pull-Request: https://github.com/syncthing/syncthing/pull/4648 2018-01-14 08:52:31 +00:00			`var (`
			`apiRequestsTotal = prometheus.NewCounterVec(`
			`prometheus.CounterOpts{`
			`Namespace: "syncthing",`
			`Subsystem: "discovery",`
			`Name: "api_requests_total",`
			`Help: "Number of API requests.",`
			`}, []string{"type", "result"})`
			`apiRequestsSeconds = prometheus.NewSummaryVec(`
			`prometheus.SummaryOpts{`
			`Namespace: "syncthing",`
			`Subsystem: "discovery",`
			`Name: "api_requests_seconds",`
			`Help: "Latency of API requests.",`
			`Objectives: map[float64]float64{0.5: 0.05, 0.9: 0.01, 0.99: 0.001},`
			`}, []string{"type"})`

			`lookupRequestsTotal = prometheus.NewCounterVec(`
			`prometheus.CounterOpts{`
			`Namespace: "syncthing",`
			`Subsystem: "discovery",`
			`Name: "lookup_requests_total",`
			`Help: "Number of lookup requests.",`
			`}, []string{"result"})`
			`announceRequestsTotal = prometheus.NewCounterVec(`
			`prometheus.CounterOpts{`
			`Namespace: "syncthing",`
			`Subsystem: "discovery",`
			`Name: "announcement_requests_total",`
			`Help: "Number of announcement requests.",`
			`}, []string{"result"})`

			`replicationSendsTotal = prometheus.NewCounterVec(`
			`prometheus.CounterOpts{`
			`Namespace: "syncthing",`
			`Subsystem: "discovery",`
			`Name: "replication_sends_total",`
			`Help: "Number of replication sends.",`
			`}, []string{"result"})`
			`replicationRecvsTotal = prometheus.NewCounterVec(`
			`prometheus.CounterOpts{`
			`Namespace: "syncthing",`
			`Subsystem: "discovery",`
			`Name: "replication_recvs_total",`
			`Help: "Number of replication receives.",`
			`}, []string{"result"})`

			`databaseKeys = prometheus.NewGaugeVec(`
			`prometheus.GaugeOpts{`
			`Namespace: "syncthing",`
			`Subsystem: "discovery",`
			`Name: "database_keys",`
			`Help: "Number of database keys at last count.",`
			`}, []string{"category"})`
			`databaseStatisticsSeconds = prometheus.NewGauge(`
			`prometheus.GaugeOpts{`
			`Namespace: "syncthing",`
			`Subsystem: "discovery",`
			`Name: "database_statistics_seconds",`
			`Help: "Time spent running the statistics routine.",`
			`})`

			`databaseOperations = prometheus.NewCounterVec(`
			`prometheus.CounterOpts{`
			`Namespace: "syncthing",`
			`Subsystem: "discovery",`
			`Name: "database_operations_total",`
			`Help: "Number of database operations.",`
			`}, []string{"operation", "result"})`
			`databaseOperationSeconds = prometheus.NewSummaryVec(`
			`prometheus.SummaryOpts{`
			`Namespace: "syncthing",`
			`Subsystem: "discovery",`
			`Name: "database_operation_seconds",`
			`Help: "Latency of database operations.",`
			`Objectives: map[float64]float64{0.5: 0.05, 0.9: 0.01, 0.99: 0.001},`
			`}, []string{"operation"})`
			`)`
Stats files 2015-05-31 11:31:28 +00:00
cmd/stdiscosrv: New discovery server (fixes #4618) This is a new revision of the discovery server. Relevant changes and non-changes: - Protocol towards clients is unchanged. - Recommended large scale design is still to be deployed nehind nginx (I tested, and it's still a lot faster at terminating TLS). - Database backend is leveldb again, only. It scales enough, is easy to setup, and we don't need any backend to take care of. - Server supports replication. This is a simple TCP channel - protect it with a firewall when deploying over the internet. (We deploy this within the same datacenter, and with firewall.) Any incoming client announces are sent over the replication channel(s) to other peer discosrvs. Incoming replication changes are applied to the database as if they came from clients, but without the TLS/certificate overhead. - Metrics are exposed using the prometheus library, when enabled. - The database values and replication protocol is protobuf, because JSON was quite CPU intensive when I tried that and benchmarked it. - The "Retry-After" value for failed lookups gets slowly increased from a default of 120 seconds, by 5 seconds for each failed lookup, independently by each discosrv. This lowers the query load over time for clients that are never seen. The Retry-After maxes out at 3600 after a couple of weeks of this increase. The number of failed lookups is stored in the database, now and then (avoiding making each lookup a database put). All in all this means clients can be pointed towards a cluster using just multiple A / AAAA records to gain both load sharing and redundancy (if one is down, clients will talk to the remaining ones). GitHub-Pull-Request: https://github.com/syncthing/syncthing/pull/4648 2018-01-14 08:52:31 +00:00			`const (`
			`dbOpGet = "get"`
			`dbOpPut = "put"`
			`dbOpMerge = "merge"`
cmd/stdiscosrv: Delete records for abandoned devices (#4957) Once a device has been missing for a long time, and noone has asked about it for a long time, delete the record. 2018-05-16 07:26:20 +00:00			`dbOpDelete = "delete"`
cmd/stdiscosrv: New discovery server (fixes #4618) This is a new revision of the discovery server. Relevant changes and non-changes: - Protocol towards clients is unchanged. - Recommended large scale design is still to be deployed nehind nginx (I tested, and it's still a lot faster at terminating TLS). - Database backend is leveldb again, only. It scales enough, is easy to setup, and we don't need any backend to take care of. - Server supports replication. This is a simple TCP channel - protect it with a firewall when deploying over the internet. (We deploy this within the same datacenter, and with firewall.) Any incoming client announces are sent over the replication channel(s) to other peer discosrvs. Incoming replication changes are applied to the database as if they came from clients, but without the TLS/certificate overhead. - Metrics are exposed using the prometheus library, when enabled. - The database values and replication protocol is protobuf, because JSON was quite CPU intensive when I tried that and benchmarked it. - The "Retry-After" value for failed lookups gets slowly increased from a default of 120 seconds, by 5 seconds for each failed lookup, independently by each discosrv. This lowers the query load over time for clients that are never seen. The Retry-After maxes out at 3600 after a couple of weeks of this increase. The number of failed lookups is stored in the database, now and then (avoiding making each lookup a database put). All in all this means clients can be pointed towards a cluster using just multiple A / AAAA records to gain both load sharing and redundancy (if one is down, clients will talk to the remaining ones). GitHub-Pull-Request: https://github.com/syncthing/syncthing/pull/4648 2018-01-14 08:52:31 +00:00			`dbResSuccess = "success"`
			`dbResNotFound = "not_found"`
			`dbResError = "error"`
			`dbResUnmarshalError = "unmarsh_err"`
			`)`
Stats files 2015-05-31 11:31:28 +00:00
cmd/stdiscosrv: New discovery server (fixes #4618) This is a new revision of the discovery server. Relevant changes and non-changes: - Protocol towards clients is unchanged. - Recommended large scale design is still to be deployed nehind nginx (I tested, and it's still a lot faster at terminating TLS). - Database backend is leveldb again, only. It scales enough, is easy to setup, and we don't need any backend to take care of. - Server supports replication. This is a simple TCP channel - protect it with a firewall when deploying over the internet. (We deploy this within the same datacenter, and with firewall.) Any incoming client announces are sent over the replication channel(s) to other peer discosrvs. Incoming replication changes are applied to the database as if they came from clients, but without the TLS/certificate overhead. - Metrics are exposed using the prometheus library, when enabled. - The database values and replication protocol is protobuf, because JSON was quite CPU intensive when I tried that and benchmarked it. - The "Retry-After" value for failed lookups gets slowly increased from a default of 120 seconds, by 5 seconds for each failed lookup, independently by each discosrv. This lowers the query load over time for clients that are never seen. The Retry-After maxes out at 3600 after a couple of weeks of this increase. The number of failed lookups is stored in the database, now and then (avoiding making each lookup a database put). All in all this means clients can be pointed towards a cluster using just multiple A / AAAA records to gain both load sharing and redundancy (if one is down, clients will talk to the remaining ones). GitHub-Pull-Request: https://github.com/syncthing/syncthing/pull/4648 2018-01-14 08:52:31 +00:00			`func init() {`
			`prometheus.MustRegister(apiRequestsTotal, apiRequestsSeconds,`
			`lookupRequestsTotal, announceRequestsTotal,`
			`replicationSendsTotal, replicationRecvsTotal,`
			`databaseKeys, databaseStatisticsSeconds,`
			`databaseOperations, databaseOperationSeconds)`
cmd/stdiscosrv: Expose process metrics Like this: $ curl -s http://localhost:9098/metrics \| egrep '^syncthing_discovery_process' syncthing_discovery_process_cpu_seconds_total 12.92 syncthing_discovery_process_max_fds 10240 syncthing_discovery_process_open_fds 51 syncthing_discovery_process_resident_memory_bytes 1.3674496e+08 syncthing_discovery_process_start_time_seconds 1.52034731837e+09 syncthing_discovery_process_virtual_memory_bytes 1.40324864e+08 2018-03-06 14:43:36 +00:00
			`prometheus.MustRegister(prometheus.NewProcessCollector(os.Getpid(), "syncthing_discovery"))`
Stats files 2015-05-31 11:31:28 +00:00			`}`