mirror of
https://github.com/SigNoz/signoz.git
synced 2026-09-28 22:30:43 +01:00
Compare commits
16 Commits
test/inter
...
srikanth/q
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
4cc7280f0d | ||
|
|
f27b4798be | ||
|
|
98d6ed18ed | ||
|
|
45aab4e6a2 | ||
|
|
0755563d4e | ||
|
|
aa9893e818 | ||
|
|
a9d6a35ccc | ||
|
|
0d279c1b96 | ||
|
|
082cd85e6a | ||
|
|
32603ff9aa | ||
|
|
baf0afd178 | ||
|
|
a9615badc0 | ||
|
|
7174733b84 | ||
|
|
17b8f6a288 | ||
|
|
12783a35ad | ||
|
|
c205ea99b5 |
1
.github/workflows/integrationci.yaml
vendored
1
.github/workflows/integrationci.yaml
vendored
@@ -56,6 +56,7 @@ jobs:
|
||||
- queriermetrics
|
||||
- querierscalar
|
||||
- queriercommon
|
||||
- queriercache
|
||||
- querierai
|
||||
- rawexportdata
|
||||
- promqlconformance
|
||||
|
||||
@@ -8374,6 +8374,8 @@ components:
|
||||
$ref: '#/components/schemas/Querybuildertypesv5FormatOptions'
|
||||
noCache:
|
||||
type: boolean
|
||||
noStepAlignment:
|
||||
type: boolean
|
||||
requestType:
|
||||
$ref: '#/components/schemas/Querybuildertypesv5RequestType'
|
||||
schemaVersion:
|
||||
|
||||
@@ -90,7 +90,7 @@ func prepareAnomalyQueryParams(req *qbtypes.QueryRangeRequest, seasonality Seaso
|
||||
End: end,
|
||||
RequestType: qbtypes.RequestTypeTimeSeries,
|
||||
CompositeQuery: req.CompositeQuery,
|
||||
NoCache: false,
|
||||
NoCache: req.NoCache,
|
||||
}
|
||||
|
||||
var pastPeriodStart, pastPeriodEnd uint64
|
||||
@@ -115,7 +115,7 @@ func prepareAnomalyQueryParams(req *qbtypes.QueryRangeRequest, seasonality Seaso
|
||||
End: pastPeriodEnd,
|
||||
RequestType: qbtypes.RequestTypeTimeSeries,
|
||||
CompositeQuery: req.CompositeQuery,
|
||||
NoCache: false,
|
||||
NoCache: req.NoCache,
|
||||
}
|
||||
|
||||
// seasonality growth trend
|
||||
@@ -137,7 +137,7 @@ func prepareAnomalyQueryParams(req *qbtypes.QueryRangeRequest, seasonality Seaso
|
||||
End: currentGrowthPeriodEnd,
|
||||
RequestType: qbtypes.RequestTypeTimeSeries,
|
||||
CompositeQuery: req.CompositeQuery,
|
||||
NoCache: false,
|
||||
NoCache: req.NoCache,
|
||||
}
|
||||
|
||||
var pastGrowthPeriodStart, pastGrowthPeriodEnd uint64
|
||||
@@ -158,7 +158,7 @@ func prepareAnomalyQueryParams(req *qbtypes.QueryRangeRequest, seasonality Seaso
|
||||
End: pastGrowthPeriodEnd,
|
||||
RequestType: qbtypes.RequestTypeTimeSeries,
|
||||
CompositeQuery: req.CompositeQuery,
|
||||
NoCache: false,
|
||||
NoCache: req.NoCache,
|
||||
}
|
||||
|
||||
var past2GrowthPeriodStart, past2GrowthPeriodEnd uint64
|
||||
@@ -179,7 +179,7 @@ func prepareAnomalyQueryParams(req *qbtypes.QueryRangeRequest, seasonality Seaso
|
||||
End: past2GrowthPeriodEnd,
|
||||
RequestType: qbtypes.RequestTypeTimeSeries,
|
||||
CompositeQuery: req.CompositeQuery,
|
||||
NoCache: false,
|
||||
NoCache: req.NoCache,
|
||||
}
|
||||
|
||||
var past3GrowthPeriodStart, past3GrowthPeriodEnd uint64
|
||||
@@ -200,7 +200,7 @@ func prepareAnomalyQueryParams(req *qbtypes.QueryRangeRequest, seasonality Seaso
|
||||
End: past3GrowthPeriodEnd,
|
||||
RequestType: qbtypes.RequestTypeTimeSeries,
|
||||
CompositeQuery: req.CompositeQuery,
|
||||
NoCache: false,
|
||||
NoCache: req.NoCache,
|
||||
}
|
||||
|
||||
return &anomalyQueryParams{
|
||||
|
||||
@@ -121,7 +121,8 @@ func (r *AnomalyRule) prepareQueryRange(ctx context.Context, ts time.Time) *qbty
|
||||
CompositeQuery: qbtypes.CompositeQuery{
|
||||
Queries: make([]qbtypes.QueryEnvelope, 0),
|
||||
},
|
||||
NoCache: true,
|
||||
NoCache: true,
|
||||
NoStepAlignment: true,
|
||||
}
|
||||
req.CompositeQuery.Queries = make([]qbtypes.QueryEnvelope, len(r.Condition().CompositeQuery.Queries))
|
||||
copy(req.CompositeQuery.Queries, r.Condition().CompositeQuery.Queries)
|
||||
|
||||
@@ -9794,6 +9794,10 @@ export interface Querybuildertypesv5QueryRangeRequestDTO {
|
||||
* @type boolean
|
||||
*/
|
||||
noCache?: boolean;
|
||||
/**
|
||||
* @type boolean
|
||||
*/
|
||||
noStepAlignment?: boolean;
|
||||
requestType?: Querybuildertypesv5RequestTypeDTO;
|
||||
/**
|
||||
* @type string
|
||||
|
||||
35
pkg/cache/memorycache/budget_test.go
vendored
Normal file
35
pkg/cache/memorycache/budget_test.go
vendored
Normal file
@@ -0,0 +1,35 @@
|
||||
package memorycache
|
||||
|
||||
import (
|
||||
"context"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/SigNoz/signoz/pkg/cache"
|
||||
"github.com/SigNoz/signoz/pkg/instrumentation/instrumentationtest"
|
||||
qbtypes "github.com/SigNoz/signoz/pkg/types/querybuildertypes/querybuildertypesv5"
|
||||
"github.com/SigNoz/signoz/pkg/valuer"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
func cachedDataOfSize(n int) *qbtypes.CachedData {
|
||||
return &qbtypes.CachedData{Buckets: []*qbtypes.CachedBucket{{EndMs: 1, Type: qbtypes.RequestTypeTimeSeries, Value: make([]byte, n)}}}
|
||||
}
|
||||
|
||||
// ristretto admits an update of an existing key without checking the budget,
|
||||
// so an entry that grows in place could take the process past MaxCost.
|
||||
func TestSet_GrowingEntryStaysWithinBudget(t *testing.T) {
|
||||
const budget = 1 << 20
|
||||
p, err := New(context.Background(), instrumentationtest.New().ToProviderSettings(), cache.Config{Provider: "memory", Memory: cache.Memory{NumCounters: 1000, MaxCost: budget}})
|
||||
require.NoError(t, err)
|
||||
prov := p.(*provider)
|
||||
orgID := valuer.GenerateUUID()
|
||||
ctx := context.Background()
|
||||
|
||||
require.NoError(t, prov.Set(ctx, orgID, "entry", cachedDataOfSize(512<<10), time.Hour))
|
||||
require.NoError(t, prov.Set(ctx, orgID, "entry", cachedDataOfSize(2<<20), time.Hour))
|
||||
|
||||
used := int64(prov.cc.Metrics.CostAdded()) - int64(prov.cc.Metrics.CostEvicted())
|
||||
assert.LessOrEqual(t, used, int64(budget), "the cache holds %d bytes against a budget of %d", used, budget)
|
||||
}
|
||||
11
pkg/cache/memorycache/provider.go
vendored
11
pkg/cache/memorycache/provider.go
vendored
@@ -120,7 +120,12 @@ func (provider *provider) Set(ctx context.Context, orgID valuer.UUID, cacheKey s
|
||||
span.SetAttributes(attribute.Bool("memory.cloneable", true))
|
||||
span.SetAttributes(attribute.Int64("memory.cost", cost))
|
||||
toCache := cloneable.Clone()
|
||||
if ok := provider.cc.SetWithTTL(strings.Join([]string{orgID.StringValue(), cacheKey}, "::"), toCache, cost, ttl); !ok {
|
||||
// ristretto updates an existing key in place without admission, so an
|
||||
// entry that grows would take the cache past MaxCost; delete first so
|
||||
// the new cost is admitted like a new key.
|
||||
key := strings.Join([]string{orgID.StringValue(), cacheKey}, "::")
|
||||
provider.cc.Del(key)
|
||||
if ok := provider.cc.SetWithTTL(key, toCache, cost, ttl); !ok {
|
||||
return errors.New(errors.TypeInternal, errors.CodeInternal, "error writing to cache")
|
||||
}
|
||||
|
||||
@@ -137,7 +142,9 @@ func (provider *provider) Set(ctx context.Context, orgID valuer.UUID, cacheKey s
|
||||
span.SetAttributes(attribute.Bool("memory.cloneable", false))
|
||||
span.SetAttributes(attribute.Int64("memory.cost", cost))
|
||||
|
||||
if ok := provider.cc.SetWithTTL(strings.Join([]string{orgID.StringValue(), cacheKey}, "::"), toCache, cost, ttl); !ok {
|
||||
key := strings.Join([]string{orgID.StringValue(), cacheKey}, "::")
|
||||
provider.cc.Del(key)
|
||||
if ok := provider.cc.SetWithTTL(key, toCache, cost, ttl); !ok {
|
||||
return errors.New(errors.TypeInternal, errors.CodeInternal, "error writing to cache")
|
||||
}
|
||||
|
||||
|
||||
@@ -7,6 +7,7 @@ import (
|
||||
"fmt"
|
||||
"net/http"
|
||||
"strconv"
|
||||
"strings"
|
||||
|
||||
"github.com/SigNoz/signoz/pkg/analytics"
|
||||
"github.com/SigNoz/signoz/pkg/errors"
|
||||
@@ -63,6 +64,11 @@ func (handler *handler) QueryRange(rw http.ResponseWriter, req *http.Request) {
|
||||
render.Error(rw, err)
|
||||
return
|
||||
}
|
||||
// The standard way for a client to ask for a fresh answer; the body flag
|
||||
// stays for callers that build the request themselves.
|
||||
if strings.Contains(strings.ToLower(req.Header.Get("Cache-Control")), "no-cache") {
|
||||
queryRangeRequest.NoCache = true
|
||||
}
|
||||
|
||||
orgID, err := valuer.NewUUID(claims.OrgID)
|
||||
if err != nil {
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -16,59 +16,38 @@ import (
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
// BenchmarkBucketCache_GetMissRanges benchmarks the GetMissRanges operation.
|
||||
const benchStepMs = uint64(1000)
|
||||
|
||||
func benchStep() qbtypes.Step { return qbtypes.Step{Duration: time.Second} }
|
||||
|
||||
func benchRequest(fingerprint string, startMs, endMs uint64) CacheRequest {
|
||||
return CacheRequest{Key: CacheKey(fingerprint), Window: qbtypes.TimeRange{From: startMs, To: endMs}, Step: benchStep(), Kind: qbtypes.RequestTypeTimeSeries}
|
||||
}
|
||||
|
||||
func BenchmarkBucketCache_GetMissRanges(b *testing.B) {
|
||||
bc := createBenchmarkBucketCache(b)
|
||||
ctx := context.Background()
|
||||
orgID := valuer.UUID{}
|
||||
|
||||
// Pre-populate cache with some data
|
||||
for i := 0; i < 10; i++ {
|
||||
query := &mockQuery{
|
||||
fingerprint: fmt.Sprintf("bench-query-%d", i),
|
||||
startMs: uint64(i * 10000),
|
||||
endMs: uint64((i + 1) * 10000),
|
||||
}
|
||||
result := createBenchmarkResult(query.startMs, query.endMs, 1000)
|
||||
bc.Put(ctx, orgID, query, qbtypes.Step{Duration: 1000 * time.Millisecond}, result)
|
||||
req := benchRequest(fmt.Sprintf("bench-query-%d", i), uint64(i*10000), uint64((i+1)*10000))
|
||||
bc.Put(ctx, orgID, req, req.Window, createBenchmarkResult(req.Window.From, req.Window.To))
|
||||
}
|
||||
|
||||
// Create test queries with varying cache hit patterns
|
||||
queries := []struct {
|
||||
name string
|
||||
query *mockQuery
|
||||
requests := []struct {
|
||||
name string
|
||||
req CacheRequest
|
||||
}{
|
||||
{
|
||||
name: "full_cache_hit",
|
||||
query: &mockQuery{
|
||||
fingerprint: "bench-query-5",
|
||||
startMs: 50000,
|
||||
endMs: 60000,
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "full_cache_miss",
|
||||
query: &mockQuery{
|
||||
fingerprint: "bench-query-new",
|
||||
startMs: 100000,
|
||||
endMs: 110000,
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "partial_cache_hit",
|
||||
query: &mockQuery{
|
||||
fingerprint: "bench-query-5",
|
||||
startMs: 45000,
|
||||
endMs: 65000,
|
||||
},
|
||||
},
|
||||
{name: "full_cache_hit", req: benchRequest("bench-query-5", 50000, 60000)},
|
||||
{name: "full_cache_miss", req: benchRequest("bench-query-new", 100000, 110000)},
|
||||
{name: "partial_cache_hit", req: benchRequest("bench-query-5", 45000, 65000)},
|
||||
}
|
||||
|
||||
for _, tc := range queries {
|
||||
for _, tc := range requests {
|
||||
b.Run(tc.name, func(b *testing.B) {
|
||||
b.ResetTimer()
|
||||
for i := 0; i < b.N; i++ {
|
||||
cached, missing := bc.GetMissRanges(ctx, orgID, tc.query, qbtypes.Step{Duration: 1000 * time.Millisecond})
|
||||
cached, missing := bc.GetMissRanges(ctx, orgID, tc.req)
|
||||
_ = cached
|
||||
_ = missing
|
||||
}
|
||||
@@ -76,222 +55,165 @@ func BenchmarkBucketCache_GetMissRanges(b *testing.B) {
|
||||
}
|
||||
}
|
||||
|
||||
// BenchmarkBucketCache_Put benchmarks the Put operation.
|
||||
func BenchmarkBucketCache_Put(b *testing.B) {
|
||||
bc := createBenchmarkBucketCache(b)
|
||||
ctx := context.Background()
|
||||
orgID := valuer.UUID{}
|
||||
|
||||
testCases := []struct {
|
||||
name string
|
||||
numSeries int
|
||||
numValues int
|
||||
numQueries int
|
||||
name string
|
||||
numSeries int
|
||||
numValues int
|
||||
}{
|
||||
{"small_result_1_series_100_values", 1, 100, 1},
|
||||
{"medium_result_10_series_100_values", 10, 100, 1},
|
||||
{"large_result_100_series_100_values", 100, 100, 1},
|
||||
{"huge_result_1000_series_100_values", 1000, 100, 1},
|
||||
{"many_values_10_series_1000_values", 10, 1000, 1},
|
||||
{"small_result_1_series_100_values", 1, 100},
|
||||
{"medium_result_10_series_100_values", 10, 100},
|
||||
{"large_result_100_series_100_values", 100, 100},
|
||||
{"huge_result_1000_series_100_values", 1000, 100},
|
||||
{"many_values_10_series_1000_values", 10, 1000},
|
||||
}
|
||||
|
||||
for _, tc := range testCases {
|
||||
b.Run(tc.name, func(b *testing.B) {
|
||||
// Create test data
|
||||
queries := make([]*mockQuery, tc.numQueries)
|
||||
results := make([]*qbtypes.Result, tc.numQueries)
|
||||
|
||||
for i := 0; i < tc.numQueries; i++ {
|
||||
queries[i] = &mockQuery{
|
||||
fingerprint: fmt.Sprintf("bench-put-query-%d", i),
|
||||
startMs: uint64(i * 100000),
|
||||
endMs: uint64((i + 1) * 100000),
|
||||
}
|
||||
results[i] = createBenchmarkResultWithSeries(
|
||||
queries[i].startMs,
|
||||
queries[i].endMs,
|
||||
1000,
|
||||
tc.numSeries,
|
||||
tc.numValues,
|
||||
)
|
||||
}
|
||||
req := benchRequest("bench-put-"+tc.name, 0, uint64(tc.numValues)*benchStepMs)
|
||||
result := createBenchmarkResultWithSeries(req.Window.From, req.Window.To, tc.numSeries, tc.numValues)
|
||||
|
||||
b.ResetTimer()
|
||||
b.ReportAllocs()
|
||||
|
||||
for i := 0; i < b.N; i++ {
|
||||
for j := 0; j < tc.numQueries; j++ {
|
||||
bc.Put(ctx, orgID, queries[j], qbtypes.Step{Duration: 1000 * time.Millisecond}, results[j])
|
||||
}
|
||||
bc.Put(ctx, orgID, req, req.Window, result)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// BenchmarkBucketCache_MergeTimeSeriesValues benchmarks merging of time series data.
|
||||
func BenchmarkBucketCache_MergeTimeSeriesValues(b *testing.B) {
|
||||
bc := createBenchmarkBucketCache(b).(*bucketCache)
|
||||
// BenchmarkBucketCache_SlidingRefresh is the dashboard pattern: every refresh
|
||||
// moves the window one step and writes the new step back.
|
||||
func BenchmarkBucketCache_SlidingRefresh(b *testing.B) {
|
||||
bc := createBenchmarkBucketCache(b)
|
||||
ctx := context.Background()
|
||||
orgID := valuer.UUID{}
|
||||
window := uint64(3600) * benchStepMs
|
||||
|
||||
testCases := []struct {
|
||||
name string
|
||||
numBuckets int
|
||||
numSeries int
|
||||
numValues int
|
||||
}{
|
||||
{"small_2_buckets_10_series", 2, 10, 100},
|
||||
{"medium_5_buckets_50_series", 5, 50, 100},
|
||||
{"large_10_buckets_100_series", 10, 100, 100},
|
||||
{"many_buckets_20_buckets_50_series", 20, 50, 100},
|
||||
}
|
||||
|
||||
for _, tc := range testCases {
|
||||
b.Run(tc.name, func(b *testing.B) {
|
||||
// Create test buckets
|
||||
buckets := make([]*qbtypes.CachedBucket, tc.numBuckets)
|
||||
for i := 0; i < tc.numBuckets; i++ {
|
||||
startMs := uint64(i * 10000)
|
||||
endMs := uint64((i + 1) * 10000)
|
||||
result := createBenchmarkResultWithSeries(startMs, endMs, 1000, tc.numSeries, tc.numValues)
|
||||
|
||||
valueBytes, _ := json.Marshal(result.Value)
|
||||
buckets[i] = &qbtypes.CachedBucket{
|
||||
StartMs: startMs,
|
||||
EndMs: endMs,
|
||||
Type: qbtypes.RequestTypeTimeSeries,
|
||||
Value: valueBytes,
|
||||
Stats: result.Stats,
|
||||
}
|
||||
}
|
||||
|
||||
b.ResetTimer()
|
||||
b.ReportAllocs()
|
||||
|
||||
for i := 0; i < b.N; i++ {
|
||||
result := bc.mergeTimeSeriesValues(context.Background(), buckets)
|
||||
_ = result
|
||||
}
|
||||
})
|
||||
b.ReportAllocs()
|
||||
for i := 0; i < b.N; i++ {
|
||||
start := uint64(i) * benchStepMs
|
||||
req := benchRequest("bench-sliding", start, start+window)
|
||||
cached, missing := bc.GetMissRanges(ctx, orgID, req)
|
||||
_ = cached
|
||||
for _, gap := range missing {
|
||||
bc.Put(ctx, orgID, req, gap, createBenchmarkResult(gap.From, gap.To))
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// BenchmarkBucketCache_FindMissingRangesWithStep benchmarks finding missing ranges.
|
||||
func BenchmarkBucketCache_FindMissingRangesWithStep(b *testing.B) {
|
||||
bc := createBenchmarkBucketCache(b).(*bucketCache)
|
||||
|
||||
testCases := []struct {
|
||||
name string
|
||||
numBuckets int
|
||||
gapPattern string // "none", "uniform", "random"
|
||||
}{
|
||||
{"no_gaps_10_buckets", 10, "none"},
|
||||
{"uniform_gaps_10_buckets", 10, "uniform"},
|
||||
{"random_gaps_20_buckets", 20, "random"},
|
||||
{"many_buckets_100", 100, "uniform"},
|
||||
}
|
||||
|
||||
for _, tc := range testCases {
|
||||
b.Run(tc.name, func(b *testing.B) {
|
||||
// Create test buckets based on pattern
|
||||
buckets := createBucketsWithPattern(tc.numBuckets, tc.gapPattern)
|
||||
startMs := uint64(0)
|
||||
endMs := uint64(tc.numBuckets * 20000)
|
||||
stepMs := uint64(1000)
|
||||
|
||||
b.ResetTimer()
|
||||
b.ReportAllocs()
|
||||
|
||||
for i := 0; i < b.N; i++ {
|
||||
missing := bc.findMissingRangesWithStep(buckets, startMs, endMs, stepMs)
|
||||
_ = missing
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// BenchmarkGetUniqueSeriesKey benchmarks the series key generation.
|
||||
func BenchmarkGetUniqueSeriesKey(b *testing.B) {
|
||||
func BenchmarkMergeTimeSeriesData(b *testing.B) {
|
||||
testCases := []struct {
|
||||
name string
|
||||
numLabels int
|
||||
numParts int
|
||||
numSeries int
|
||||
numValues int
|
||||
}{
|
||||
{"1_label", 1},
|
||||
{"5_labels", 5},
|
||||
{"10_labels", 10},
|
||||
{"20_labels", 20},
|
||||
{"50_labels", 50},
|
||||
{"small_2_parts_10_series", 2, 10, 100},
|
||||
{"medium_5_parts_50_series", 5, 50, 100},
|
||||
{"large_10_parts_100_series", 10, 100, 100},
|
||||
{"many_parts_20_parts_50_series", 20, 50, 100},
|
||||
}
|
||||
|
||||
for _, tc := range testCases {
|
||||
b.Run(tc.name, func(b *testing.B) {
|
||||
labels := make([]*qbtypes.Label, tc.numLabels)
|
||||
for i := 0; i < tc.numLabels; i++ {
|
||||
parts := make([]*qbtypes.TimeSeriesData, tc.numParts)
|
||||
for i := range parts {
|
||||
startMs := uint64(i) * uint64(tc.numValues) * benchStepMs
|
||||
parts[i] = createBenchmarkResultWithSeries(startMs, startMs+uint64(tc.numValues)*benchStepMs, tc.numSeries, tc.numValues).Value.(*qbtypes.TimeSeriesData)
|
||||
}
|
||||
|
||||
b.ResetTimer()
|
||||
b.ReportAllocs()
|
||||
for i := 0; i < b.N; i++ {
|
||||
_ = mergeTimeSeriesData(parts)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func BenchmarkBucketCache_Decode(b *testing.B) {
|
||||
bc := createBenchmarkBucketCache(b).(*bucketCache)
|
||||
testCases := []struct {
|
||||
name string
|
||||
numBuckets int
|
||||
numSeries int
|
||||
}{
|
||||
{"1_bucket_10_series", 1, 10},
|
||||
{"5_buckets_50_series", 5, 50},
|
||||
{"20_buckets_100_series", 20, 100},
|
||||
}
|
||||
|
||||
for _, tc := range testCases {
|
||||
b.Run(tc.name, func(b *testing.B) {
|
||||
buckets := make([]*qbtypes.CachedBucket, tc.numBuckets)
|
||||
for i := range buckets {
|
||||
startMs := uint64(i * 10000)
|
||||
result := createBenchmarkResultWithSeries(startMs, startMs+10000, tc.numSeries, 10)
|
||||
value, err := json.Marshal(result.Value)
|
||||
require.NoError(b, err)
|
||||
buckets[i] = &qbtypes.CachedBucket{StartMs: startMs, EndMs: startMs + 10000, WrittenAtMs: time.Now().UnixMilli(), Type: qbtypes.RequestTypeTimeSeries, Value: value}
|
||||
}
|
||||
|
||||
b.ResetTimer()
|
||||
b.ReportAllocs()
|
||||
for i := 0; i < b.N; i++ {
|
||||
_ = bc.decode(context.Background(), buckets, qbtypes.TimeRange{To: ^uint64(0)})
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func BenchmarkGetUniqueSeriesKey(b *testing.B) {
|
||||
for _, numLabels := range []int{1, 5, 10, 20, 50} {
|
||||
b.Run(fmt.Sprintf("%d_labels", numLabels), func(b *testing.B) {
|
||||
labels := make([]*qbtypes.Label, numLabels)
|
||||
for i := range labels {
|
||||
labels[i] = &qbtypes.Label{
|
||||
Key: telemetrytypes.TelemetryFieldKey{
|
||||
Name: fmt.Sprintf("label_%d", i),
|
||||
FieldDataType: telemetrytypes.FieldDataTypeString,
|
||||
},
|
||||
Key: telemetrytypes.TelemetryFieldKey{Name: fmt.Sprintf("label_%d", i), FieldDataType: telemetrytypes.FieldDataTypeString},
|
||||
Value: fmt.Sprintf("value_%d", i),
|
||||
}
|
||||
}
|
||||
|
||||
b.ResetTimer()
|
||||
b.ReportAllocs()
|
||||
|
||||
for i := 0; i < b.N; i++ {
|
||||
key := qbtypes.GetUniqueSeriesKey(labels)
|
||||
_ = key
|
||||
_ = qbtypes.GetUniqueSeriesKey(labels)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// BenchmarkBucketCache_ConcurrentOperations benchmarks concurrent cache operations.
|
||||
func BenchmarkBucketCache_ConcurrentOperations(b *testing.B) {
|
||||
bc := createBenchmarkBucketCache(b)
|
||||
ctx := context.Background()
|
||||
orgID := valuer.UUID{}
|
||||
|
||||
// Pre-populate cache
|
||||
for i := 0; i < 100; i++ {
|
||||
query := &mockQuery{
|
||||
fingerprint: fmt.Sprintf("concurrent-query-%d", i),
|
||||
startMs: uint64(i * 10000),
|
||||
endMs: uint64((i + 1) * 10000),
|
||||
}
|
||||
result := createBenchmarkResult(query.startMs, query.endMs, 1000)
|
||||
bc.Put(ctx, orgID, query, qbtypes.Step{Duration: 1000 * time.Millisecond}, result)
|
||||
req := benchRequest(fmt.Sprintf("concurrent-query-%d", i), uint64(i*10000), uint64((i+1)*10000))
|
||||
bc.Put(ctx, orgID, req, req.Window, createBenchmarkResult(req.Window.From, req.Window.To))
|
||||
}
|
||||
|
||||
b.ResetTimer()
|
||||
b.RunParallel(func(pb *testing.PB) {
|
||||
i := 0
|
||||
for pb.Next() {
|
||||
// Mix of operations
|
||||
switch i % 3 {
|
||||
case 0: // Read
|
||||
query := &mockQuery{
|
||||
fingerprint: fmt.Sprintf("concurrent-query-%d", i%100),
|
||||
startMs: uint64((i % 100) * 10000),
|
||||
endMs: uint64(((i % 100) + 1) * 10000),
|
||||
}
|
||||
cached, missing := bc.GetMissRanges(ctx, orgID, query, qbtypes.Step{Duration: 1000 * time.Millisecond})
|
||||
case 0:
|
||||
req := benchRequest(fmt.Sprintf("concurrent-query-%d", i%100), uint64((i%100)*10000), uint64(((i%100)+1)*10000))
|
||||
cached, missing := bc.GetMissRanges(ctx, orgID, req)
|
||||
_ = cached
|
||||
_ = missing
|
||||
case 1: // Write
|
||||
query := &mockQuery{
|
||||
fingerprint: fmt.Sprintf("concurrent-query-new-%d", i),
|
||||
startMs: uint64(i * 10000),
|
||||
endMs: uint64((i + 1) * 10000),
|
||||
}
|
||||
result := createBenchmarkResult(query.startMs, query.endMs, 1000)
|
||||
bc.Put(ctx, orgID, query, qbtypes.Step{Duration: 1000 * time.Millisecond}, result)
|
||||
case 2: // Partial read
|
||||
query := &mockQuery{
|
||||
fingerprint: fmt.Sprintf("concurrent-query-%d", i%100),
|
||||
startMs: uint64((i%100)*10000 - 5000),
|
||||
endMs: uint64(((i%100)+1)*10000 + 5000),
|
||||
}
|
||||
cached, missing := bc.GetMissRanges(ctx, orgID, query, qbtypes.Step{Duration: 1000 * time.Millisecond})
|
||||
case 1:
|
||||
req := benchRequest(fmt.Sprintf("concurrent-query-new-%d", i), uint64(i*10000), uint64((i+1)*10000))
|
||||
bc.Put(ctx, orgID, req, req.Window, createBenchmarkResult(req.Window.From, req.Window.To))
|
||||
case 2:
|
||||
req := benchRequest(fmt.Sprintf("concurrent-query-%d", i%100), uint64((i%100)*10000+5000), uint64(((i%100)+1)*10000+5000))
|
||||
cached, missing := bc.GetMissRanges(ctx, orgID, req)
|
||||
_ = cached
|
||||
_ = missing
|
||||
}
|
||||
@@ -300,41 +222,6 @@ func BenchmarkBucketCache_ConcurrentOperations(b *testing.B) {
|
||||
})
|
||||
}
|
||||
|
||||
// BenchmarkBucketCache_FilterResultToTimeRange benchmarks filtering results to time range.
|
||||
func BenchmarkBucketCache_FilterResultToTimeRange(b *testing.B) {
|
||||
bc := createBenchmarkBucketCache(b).(*bucketCache)
|
||||
|
||||
testCases := []struct {
|
||||
name string
|
||||
numSeries int
|
||||
numValues int
|
||||
}{
|
||||
{"small_10_series_100_values", 10, 100},
|
||||
{"medium_50_series_500_values", 50, 500},
|
||||
{"large_100_series_1000_values", 100, 1000},
|
||||
}
|
||||
|
||||
for _, tc := range testCases {
|
||||
b.Run(tc.name, func(b *testing.B) {
|
||||
// Create a large result
|
||||
result := createBenchmarkResultWithSeries(0, 100000, 1000, tc.numSeries, tc.numValues)
|
||||
|
||||
// Filter to middle 50%
|
||||
startMs := uint64(25000)
|
||||
endMs := uint64(75000)
|
||||
|
||||
b.ResetTimer()
|
||||
b.ReportAllocs()
|
||||
|
||||
for i := 0; i < b.N; i++ {
|
||||
filtered := bc.filterResultToTimeRange(result, startMs, endMs)
|
||||
_ = filtered
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// Helper function to create benchmark bucket cache.
|
||||
func createBenchmarkBucketCache(tb testing.TB) BucketCache {
|
||||
config := cache.Config{
|
||||
Provider: "memory",
|
||||
@@ -348,65 +235,40 @@ func createBenchmarkBucketCache(tb testing.TB) BucketCache {
|
||||
return NewBucketCache(instrumentationtest.New().ToProviderSettings(), memCache, time.Hour, 5*time.Minute)
|
||||
}
|
||||
|
||||
// Helper function to create benchmark result.
|
||||
func createBenchmarkResult(startMs, endMs uint64, step uint64) *qbtypes.Result {
|
||||
return createBenchmarkResultWithSeries(startMs, endMs, step, 10, 100)
|
||||
func createBenchmarkResult(startMs, endMs uint64) *qbtypes.Result {
|
||||
return createBenchmarkResultWithSeries(startMs, endMs, 10, int((endMs-startMs)/benchStepMs))
|
||||
}
|
||||
|
||||
// Helper function to create benchmark result with specific series and values.
|
||||
func createBenchmarkResultWithSeries(startMs, endMs uint64, _ uint64, numSeries, numValuesPerSeries int) *qbtypes.Result {
|
||||
// createBenchmarkResultWithSeries spreads numValuesPerSeries points over
|
||||
// [startMs, endMs) on the step grid.
|
||||
func createBenchmarkResultWithSeries(startMs, endMs uint64, numSeries, numValuesPerSeries int) *qbtypes.Result {
|
||||
series := make([]*qbtypes.TimeSeries, numSeries)
|
||||
valueStep := max((endMs-startMs)/uint64(max(numValuesPerSeries, 1)), benchStepMs)
|
||||
valueStep -= valueStep % benchStepMs
|
||||
|
||||
for i := 0; i < numSeries; i++ {
|
||||
ts := &qbtypes.TimeSeries{
|
||||
Labels: []*qbtypes.Label{
|
||||
{
|
||||
Key: telemetrytypes.TelemetryFieldKey{
|
||||
Name: "host",
|
||||
FieldDataType: telemetrytypes.FieldDataTypeString,
|
||||
},
|
||||
Value: fmt.Sprintf("server-%d", i),
|
||||
},
|
||||
{
|
||||
Key: telemetrytypes.TelemetryFieldKey{
|
||||
Name: "service",
|
||||
FieldDataType: telemetrytypes.FieldDataTypeString,
|
||||
},
|
||||
Value: fmt.Sprintf("service-%d", i%5),
|
||||
},
|
||||
{Key: telemetrytypes.TelemetryFieldKey{Name: "host", FieldDataType: telemetrytypes.FieldDataTypeString}, Value: fmt.Sprintf("server-%d", i)},
|
||||
{Key: telemetrytypes.TelemetryFieldKey{Name: "service", FieldDataType: telemetrytypes.FieldDataTypeString}, Value: fmt.Sprintf("service-%d", i%5)},
|
||||
},
|
||||
Values: make([]*qbtypes.TimeSeriesValue, 0, numValuesPerSeries),
|
||||
}
|
||||
|
||||
// Generate values
|
||||
valueStep := (endMs - startMs) / uint64(numValuesPerSeries)
|
||||
if valueStep == 0 {
|
||||
valueStep = 1
|
||||
}
|
||||
|
||||
for j := 0; j < numValuesPerSeries; j++ {
|
||||
timestamp := int64(startMs + uint64(j)*valueStep)
|
||||
if timestamp < int64(endMs) {
|
||||
ts.Values = append(ts.Values, &qbtypes.TimeSeriesValue{
|
||||
Timestamp: timestamp,
|
||||
Value: float64(i*100 + j),
|
||||
})
|
||||
timestamp := startMs + uint64(j)*valueStep
|
||||
if timestamp+benchStepMs > endMs {
|
||||
break
|
||||
}
|
||||
ts.Values = append(ts.Values, &qbtypes.TimeSeriesValue{Timestamp: int64(timestamp), Value: float64(i*100 + j)})
|
||||
}
|
||||
|
||||
series[i] = ts
|
||||
}
|
||||
|
||||
return &qbtypes.Result{
|
||||
Type: qbtypes.RequestTypeTimeSeries,
|
||||
Value: &qbtypes.TimeSeriesData{
|
||||
QueryName: "benchmark_query",
|
||||
Aggregations: []*qbtypes.AggregationBucket{
|
||||
{
|
||||
Index: 0,
|
||||
Series: series,
|
||||
},
|
||||
},
|
||||
QueryName: "benchmark_query",
|
||||
Aggregations: []*qbtypes.AggregationBucket{{Index: 0, Series: series}},
|
||||
},
|
||||
Stats: qbtypes.ExecStats{
|
||||
RowsScanned: uint64(numSeries * numValuesPerSeries),
|
||||
@@ -415,31 +277,3 @@ func createBenchmarkResultWithSeries(startMs, endMs uint64, _ uint64, numSeries,
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// Helper function to create buckets with specific gap patterns.
|
||||
func createBucketsWithPattern(numBuckets int, pattern string) []*qbtypes.CachedBucket {
|
||||
buckets := make([]*qbtypes.CachedBucket, 0, numBuckets)
|
||||
|
||||
for i := 0; i < numBuckets; i++ {
|
||||
// Skip some buckets based on pattern
|
||||
if pattern == "uniform" && i%3 == 0 {
|
||||
continue // Create gaps every 3rd bucket
|
||||
}
|
||||
if pattern == "random" && i%7 < 2 {
|
||||
continue // Create random gaps
|
||||
}
|
||||
|
||||
startMs := uint64(i * 10000)
|
||||
endMs := uint64((i + 1) * 10000)
|
||||
|
||||
buckets = append(buckets, &qbtypes.CachedBucket{
|
||||
StartMs: startMs,
|
||||
EndMs: endMs,
|
||||
Type: qbtypes.RequestTypeTimeSeries,
|
||||
Value: json.RawMessage(`{}`),
|
||||
Stats: qbtypes.ExecStats{},
|
||||
})
|
||||
}
|
||||
|
||||
return buckets
|
||||
}
|
||||
|
||||
93
pkg/querier/bucket_cache_codec.go
Normal file
93
pkg/querier/bucket_cache_codec.go
Normal file
@@ -0,0 +1,93 @@
|
||||
package querier
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
|
||||
"github.com/SigNoz/signoz/pkg/errors"
|
||||
qbtypes "github.com/SigNoz/signoz/pkg/types/querybuildertypes/querybuildertypesv5"
|
||||
)
|
||||
|
||||
// The cache keeps points at full precision. The response encoder of
|
||||
// TimeSeriesValue rounds values for readers, and a rounded value fed back
|
||||
// into post-processing gives another answer than the uncached query.
|
||||
|
||||
type cachedPoint struct {
|
||||
Timestamp int64 `json:"t"`
|
||||
Value float64 `json:"v"`
|
||||
Values []float64 `json:"vs,omitempty"`
|
||||
Partial bool `json:"p,omitempty"`
|
||||
}
|
||||
|
||||
type cachedSeries struct {
|
||||
Labels []*qbtypes.Label `json:"labels,omitempty"`
|
||||
Points []*cachedPoint `json:"points"`
|
||||
}
|
||||
|
||||
type cachedAggregation struct {
|
||||
Index int `json:"index"`
|
||||
Alias string `json:"alias,omitempty"`
|
||||
Meta qbtypes.AggregationMeta `json:"meta,omitempty"`
|
||||
Series []*cachedSeries `json:"series"`
|
||||
}
|
||||
|
||||
type cachedValue struct {
|
||||
QueryName string `json:"queryName,omitempty"`
|
||||
Aggregations []*cachedAggregation `json:"aggregations"`
|
||||
}
|
||||
|
||||
func encodeBucketValue(data *qbtypes.TimeSeriesData) ([]byte, error) {
|
||||
value := cachedValue{QueryName: data.QueryName, Aggregations: make([]*cachedAggregation, 0, len(data.Aggregations))}
|
||||
for _, agg := range data.Aggregations {
|
||||
if agg == nil {
|
||||
continue
|
||||
}
|
||||
encoded := &cachedAggregation{Index: agg.Index, Alias: agg.Alias, Meta: agg.Meta, Series: make([]*cachedSeries, 0, len(agg.Series))}
|
||||
for _, s := range agg.Series {
|
||||
if s == nil {
|
||||
continue
|
||||
}
|
||||
series := &cachedSeries{Labels: s.Labels, Points: make([]*cachedPoint, 0, len(s.Values))}
|
||||
for _, v := range s.Values {
|
||||
if v == nil {
|
||||
continue
|
||||
}
|
||||
series.Points = append(series.Points, &cachedPoint{Timestamp: v.Timestamp, Value: v.Value, Values: v.Values, Partial: v.Partial})
|
||||
}
|
||||
encoded.Series = append(encoded.Series, series)
|
||||
}
|
||||
value.Aggregations = append(value.Aggregations, encoded)
|
||||
}
|
||||
return json.Marshal(value)
|
||||
}
|
||||
|
||||
// decodeBucketValue rejects a payload with null elements: a bucket is either
|
||||
// whole or not usable, since a reader cannot tell a dropped point from an
|
||||
// absent one.
|
||||
func decodeBucketValue(raw []byte) (*qbtypes.TimeSeriesData, error) {
|
||||
var value cachedValue
|
||||
if err := json.Unmarshal(raw, &value); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
data := &qbtypes.TimeSeriesData{QueryName: value.QueryName, Aggregations: make([]*qbtypes.AggregationBucket, 0, len(value.Aggregations))}
|
||||
for _, agg := range value.Aggregations {
|
||||
if agg == nil {
|
||||
return nil, errors.NewInternalf(errors.CodeInternal, "cached bucket has a null aggregation")
|
||||
}
|
||||
decoded := &qbtypes.AggregationBucket{Index: agg.Index, Alias: agg.Alias, Meta: agg.Meta, Series: make([]*qbtypes.TimeSeries, 0, len(agg.Series))}
|
||||
for _, s := range agg.Series {
|
||||
if s == nil {
|
||||
return nil, errors.NewInternalf(errors.CodeInternal, "cached bucket has a null series")
|
||||
}
|
||||
series := &qbtypes.TimeSeries{Labels: s.Labels, Values: make([]*qbtypes.TimeSeriesValue, 0, len(s.Points))}
|
||||
for _, p := range s.Points {
|
||||
if p == nil {
|
||||
return nil, errors.NewInternalf(errors.CodeInternal, "cached bucket has a null point")
|
||||
}
|
||||
series.Values = append(series.Values, &qbtypes.TimeSeriesValue{Timestamp: p.Timestamp, Value: p.Value, Values: p.Values, Partial: p.Partial})
|
||||
}
|
||||
decoded.Series = append(decoded.Series, series)
|
||||
}
|
||||
data.Aggregations = append(data.Aggregations, decoded)
|
||||
}
|
||||
return data, nil
|
||||
}
|
||||
@@ -1,117 +0,0 @@
|
||||
package querier
|
||||
|
||||
import (
|
||||
"context"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/SigNoz/signoz/pkg/instrumentation/instrumentationtest"
|
||||
qbtypes "github.com/SigNoz/signoz/pkg/types/querybuildertypes/querybuildertypesv5"
|
||||
"github.com/SigNoz/signoz/pkg/types/telemetrytypes"
|
||||
"github.com/SigNoz/signoz/pkg/valuer"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
func TestBucketCacheStepAlignment(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
orgID := valuer.UUID{}
|
||||
cache := createTestCache(t)
|
||||
bc := NewBucketCache(instrumentationtest.New().ToProviderSettings(), cache, time.Hour, 5*time.Minute)
|
||||
|
||||
// Test with 5-minute step
|
||||
step := qbtypes.Step{Duration: 5 * time.Minute}
|
||||
|
||||
// Query from 12:02 to 12:58 (both unaligned)
|
||||
// Complete intervals: 12:05 to 12:55
|
||||
query := &mockQuery{
|
||||
fingerprint: "test-step-alignment",
|
||||
startMs: 1672563720000, // 12:02
|
||||
endMs: 1672567080000, // 12:58
|
||||
}
|
||||
|
||||
result := &qbtypes.Result{
|
||||
Type: qbtypes.RequestTypeTimeSeries,
|
||||
Value: &qbtypes.TimeSeriesData{
|
||||
QueryName: "test",
|
||||
Aggregations: []*qbtypes.AggregationBucket{
|
||||
{
|
||||
Index: 0,
|
||||
Series: []*qbtypes.TimeSeries{
|
||||
{
|
||||
Labels: []*qbtypes.Label{
|
||||
{Key: telemetrytypes.TelemetryFieldKey{Name: "service"}, Value: "test"},
|
||||
},
|
||||
Values: []*qbtypes.TimeSeriesValue{
|
||||
{Timestamp: 1672563720000, Value: 1, Partial: true}, // 12:02
|
||||
{Timestamp: 1672563900000, Value: 2}, // 12:05
|
||||
{Timestamp: 1672564200000, Value: 2.5}, // 12:10
|
||||
{Timestamp: 1672564500000, Value: 2.6}, // 12:15
|
||||
{Timestamp: 1672566600000, Value: 2.9}, // 12:50
|
||||
{Timestamp: 1672566900000, Value: 3}, // 12:55
|
||||
{Timestamp: 1672567080000, Value: 4, Partial: true}, // 12:58
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
// Put result in cache
|
||||
bc.Put(ctx, orgID, query, step, result)
|
||||
|
||||
// Get cached data
|
||||
cached, missing := bc.GetMissRanges(ctx, orgID, query, step)
|
||||
|
||||
// Should have cached data
|
||||
require.NotNil(t, cached)
|
||||
|
||||
// Log the missing ranges to debug
|
||||
t.Logf("Missing ranges: %v", missing)
|
||||
for i, r := range missing {
|
||||
t.Logf("Missing range %d: From=%d, To=%d", i, r.From, r.To)
|
||||
}
|
||||
|
||||
// Should have 2 missing ranges for partial intervals
|
||||
require.Len(t, missing, 2)
|
||||
|
||||
// First partial: 12:02 to 12:05
|
||||
assert.Equal(t, uint64(1672563720000), missing[0].From)
|
||||
assert.Equal(t, uint64(1672563900000), missing[0].To)
|
||||
|
||||
// Second partial: 12:55 to 12:58
|
||||
assert.Equal(t, uint64(1672566900000), missing[1].From, "Second missing range From")
|
||||
assert.Equal(t, uint64(1672567080000), missing[1].To, "Second missing range To")
|
||||
}
|
||||
|
||||
func TestBucketCacheNoStepInterval(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
orgID := valuer.UUID{}
|
||||
cache := createTestCache(t)
|
||||
bc := NewBucketCache(instrumentationtest.New().ToProviderSettings(), cache, time.Hour, 5*time.Minute)
|
||||
|
||||
// Test with no step (stepMs = 0)
|
||||
step := qbtypes.Step{Duration: 0}
|
||||
|
||||
query := &mockQuery{
|
||||
fingerprint: "test-no-step",
|
||||
startMs: 1672563720000,
|
||||
endMs: 1672567080000,
|
||||
}
|
||||
|
||||
result := &qbtypes.Result{
|
||||
Type: qbtypes.RequestTypeTimeSeries,
|
||||
Value: &qbtypes.TimeSeriesData{
|
||||
QueryName: "test",
|
||||
Aggregations: []*qbtypes.AggregationBucket{{Index: 0, Series: []*qbtypes.TimeSeries{}}},
|
||||
},
|
||||
}
|
||||
|
||||
// Should cache the entire range when step is 0
|
||||
bc.Put(ctx, orgID, query, step, result)
|
||||
|
||||
cached, missing := bc.GetMissRanges(ctx, orgID, query, step)
|
||||
assert.NotNil(t, cached)
|
||||
assert.Len(t, missing, 0)
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -3,8 +3,10 @@ package querier
|
||||
import (
|
||||
"context"
|
||||
"encoding/base64"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"log/slog"
|
||||
"sort"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
@@ -147,11 +149,22 @@ func (q *builderQuery[T]) Fingerprint() string {
|
||||
if q.spec.Filter != nil && q.spec.Filter.Expression != "" {
|
||||
parts = append(parts, fmt.Sprintf("filter=%s", q.spec.Filter.Expression))
|
||||
|
||||
for name, item := range q.variables {
|
||||
// Sorted so the key is the same on every call, and JSON so that
|
||||
// ["a b"] and ["a", "b"], or 1 and "1", get different keys.
|
||||
names := make([]string, 0, len(q.variables))
|
||||
for name := range q.variables {
|
||||
if strings.Contains(q.spec.Filter.Expression, "$"+name) {
|
||||
parts = append(parts, fmt.Sprintf("%s=%s", name, fmt.Sprint(item.Value)))
|
||||
names = append(names, name)
|
||||
}
|
||||
}
|
||||
sort.Strings(names)
|
||||
for _, name := range names {
|
||||
value, err := json.Marshal(q.variables[name].Value)
|
||||
if err != nil {
|
||||
value = []byte(fmt.Sprint(q.variables[name].Value))
|
||||
}
|
||||
parts = append(parts, fmt.Sprintf("%s=%s", name, value))
|
||||
}
|
||||
}
|
||||
|
||||
// Add group by keys
|
||||
@@ -189,9 +202,37 @@ func (q *builderQuery[T]) Fingerprint() string {
|
||||
parts = append(parts, fmt.Sprintf("shiftby=%d", q.spec.ShiftBy))
|
||||
}
|
||||
|
||||
// A top-N is ranked over the statement window, so its result serves only
|
||||
// the identical window.
|
||||
if q.wholeWindowOnly() {
|
||||
parts = append(parts, fmt.Sprintf("window=%d-%d", q.fromMS, q.toMS))
|
||||
}
|
||||
|
||||
return strings.Join(parts, "&")
|
||||
}
|
||||
|
||||
// wholeWindowOnly reports whether the statement ranks or limits groups over
|
||||
// its window (the top-N CTE of logs and traces), which pieces of the window
|
||||
// cannot reproduce. Metrics apply their limit after the statement.
|
||||
func (q *builderQuery[T]) wholeWindowOnly() bool {
|
||||
if q.spec.Limit <= 0 || len(q.spec.GroupBy) == 0 {
|
||||
return false
|
||||
}
|
||||
return q.spec.Signal == telemetrytypes.SignalLogs || q.spec.Signal == telemetrytypes.SignalTraces
|
||||
}
|
||||
|
||||
// lookbackSteps is how many steps before the window the result must carry.
|
||||
// runningDiff drops its first point, so the metrics builder fetches one step
|
||||
// before the window to give the first interval a difference.
|
||||
func (q *builderQuery[T]) lookbackSteps() int {
|
||||
for _, fn := range q.spec.Functions {
|
||||
if fn.Name == qbtypes.FunctionNameRunningDiff {
|
||||
return 1
|
||||
}
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// fingerprintHeatmapBucketing captures only what changes the rows ClickHouse
|
||||
// returns, which is why LogBucketsSpec.Scale is absent: coarsening it happens in
|
||||
// postprocessing, so every scale reads one cache entry.
|
||||
|
||||
658
pkg/querier/cache_differential_test.go
Normal file
658
pkg/querier/cache_differential_test.go
Normal file
@@ -0,0 +1,658 @@
|
||||
package querier
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"math/rand"
|
||||
"sort"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/ClickHouse/clickhouse-go/v2"
|
||||
"github.com/ClickHouse/clickhouse-go/v2/lib/driver"
|
||||
"github.com/DATA-DOG/go-sqlmock"
|
||||
cmock "github.com/SigNoz/clickhouse-go-mock"
|
||||
"github.com/stretchr/testify/require"
|
||||
|
||||
"github.com/SigNoz/signoz/pkg/flagger/flaggertest"
|
||||
"github.com/SigNoz/signoz/pkg/instrumentation/instrumentationtest"
|
||||
"github.com/SigNoz/signoz/pkg/querybuilder"
|
||||
"github.com/SigNoz/signoz/pkg/telemetrystore"
|
||||
"github.com/SigNoz/signoz/pkg/telemetrystore/telemetrystoretest"
|
||||
"github.com/SigNoz/signoz/pkg/types/metrictypes"
|
||||
qbtypes "github.com/SigNoz/signoz/pkg/types/querybuildertypes/querybuildertypesv5"
|
||||
"github.com/SigNoz/signoz/pkg/types/telemetrytypes"
|
||||
"github.com/SigNoz/signoz/pkg/valuer"
|
||||
)
|
||||
|
||||
// Differential check of the bucket cache: random request sequences over a
|
||||
// random dataset are answered twice, through the cache and with NoCache, and
|
||||
// the two answers must be identical. ClickHouse is replaced by an in-process
|
||||
// fake that evaluates the statement the fake builders render (window, step,
|
||||
// limit) against the dataset with the semantics of the real statements:
|
||||
// rows filtered to [start, end), bucketed to the step grid, grouped by
|
||||
// service, the top-N of a limited query chosen over the statement window,
|
||||
// and rate computed against the previous bucket within the lookback.
|
||||
// consume, executeWithCache, the bucket cache and post-processing are the
|
||||
// real code. Mismatches are reported by request shape and symptom.
|
||||
|
||||
const (
|
||||
fuzzDatasetStartMs = epochMs
|
||||
fuzzDatasetMinutes = 6 * 60
|
||||
fuzzServices = 6
|
||||
)
|
||||
|
||||
// fuzzDataset holds, per service and per minute, how many log rows exist (one
|
||||
// second apart from the minute start) and the gauge value reported at the
|
||||
// minute start.
|
||||
type fuzzDataset struct {
|
||||
counts [fuzzServices][fuzzDatasetMinutes]int
|
||||
gauges [fuzzServices][fuzzDatasetMinutes]float64
|
||||
// counters are cumulative samples at the minute start; a negative entry
|
||||
// means no sample that minute (a gap), which is where window functions
|
||||
// see a different predecessor depending on the statement window.
|
||||
counters [fuzzServices][fuzzDatasetMinutes]float64
|
||||
}
|
||||
|
||||
func newFuzzDataset(rng *rand.Rand) *fuzzDataset {
|
||||
d := &fuzzDataset{}
|
||||
for s := 0; s < fuzzServices; s++ {
|
||||
// Every service is active for one or two stretches so that windows
|
||||
// exist where a service has no rows at all.
|
||||
activeFrom := rng.Intn(fuzzDatasetMinutes / 2)
|
||||
activeTo := activeFrom + 30 + rng.Intn(fuzzDatasetMinutes/2)
|
||||
for m := 0; m < fuzzDatasetMinutes; m++ {
|
||||
if m >= activeFrom && m < activeTo {
|
||||
d.counts[s][m] = rng.Intn(6)
|
||||
}
|
||||
d.gauges[s][m] = float64(100 + 10*s + rng.Intn(50))
|
||||
}
|
||||
gapFrom := rng.Intn(fuzzDatasetMinutes - 20)
|
||||
gapTo := gapFrom + 3 + rng.Intn(12)
|
||||
total := float64(1000 * (s + 1))
|
||||
for m := 0; m < fuzzDatasetMinutes; m++ {
|
||||
total += float64(rng.Intn(20))
|
||||
if (m >= gapFrom && m < gapTo) || rng.Intn(25) == 0 {
|
||||
d.counters[s][m] = -1
|
||||
continue
|
||||
}
|
||||
d.counters[s][m] = total
|
||||
}
|
||||
}
|
||||
return d
|
||||
}
|
||||
|
||||
func fuzzServiceName(s int) string { return fmt.Sprintf("svc-%c", 'a'+s) }
|
||||
|
||||
// logRows evaluates the logs time series statement: count of rows in
|
||||
// [startMs, endMs) per (bucket, service); with limit > 0 only the limit
|
||||
// services with the highest count over the window are kept, ties broken by
|
||||
// name as ClickHouse would break them deterministically for one plan.
|
||||
func (d *fuzzDataset) logRows(startMs, endMs, stepMs uint64, limit int) [][]any {
|
||||
type key struct {
|
||||
ts uint64
|
||||
service int
|
||||
}
|
||||
perBucket := map[key]float64{}
|
||||
total := make([]float64, fuzzServices)
|
||||
for s := 0; s < fuzzServices; s++ {
|
||||
for m := 0; m < fuzzDatasetMinutes; m++ {
|
||||
minuteMs := fuzzDatasetStartMs + uint64(m)*60_000
|
||||
for i := 0; i < d.counts[s][m]; i++ {
|
||||
ts := minuteMs + uint64(i+1)*1000
|
||||
if ts < startMs || ts >= endMs {
|
||||
continue
|
||||
}
|
||||
perBucket[key{ts - ts%stepMs, s}]++
|
||||
total[s]++
|
||||
}
|
||||
}
|
||||
}
|
||||
keep := map[int]bool{}
|
||||
if limit > 0 {
|
||||
order := make([]int, 0, fuzzServices)
|
||||
for s := 0; s < fuzzServices; s++ {
|
||||
if total[s] > 0 {
|
||||
order = append(order, s)
|
||||
}
|
||||
}
|
||||
sort.Slice(order, func(i, j int) bool {
|
||||
if total[order[i]] != total[order[j]] {
|
||||
return total[order[i]] > total[order[j]]
|
||||
}
|
||||
return order[i] < order[j]
|
||||
})
|
||||
for i, s := range order {
|
||||
if i < limit {
|
||||
keep[s] = true
|
||||
}
|
||||
}
|
||||
}
|
||||
var rows [][]any
|
||||
for k, count := range perBucket {
|
||||
if limit > 0 && !keep[k.service] {
|
||||
continue
|
||||
}
|
||||
rows = append(rows, []any{time.UnixMilli(int64(k.ts)), fuzzServiceName(k.service), count})
|
||||
}
|
||||
sort.Slice(rows, func(i, j int) bool {
|
||||
ti, tj := rows[i][0].(time.Time), rows[j][0].(time.Time)
|
||||
if !ti.Equal(tj) {
|
||||
return ti.Before(tj)
|
||||
}
|
||||
return rows[i][1].(string) < rows[j][1].(string)
|
||||
})
|
||||
return rows
|
||||
}
|
||||
|
||||
// gaugeRows evaluates the metrics statement for avg over the gauge: one row
|
||||
// per (bucket, service) with the mean of the samples in [startMs, endMs).
|
||||
func (d *fuzzDataset) gaugeRows(startMs, endMs, stepMs uint64) [][]any {
|
||||
type key struct {
|
||||
ts uint64
|
||||
service int
|
||||
}
|
||||
sum := map[key]float64{}
|
||||
n := map[key]float64{}
|
||||
for s := 0; s < fuzzServices; s++ {
|
||||
for m := 0; m < fuzzDatasetMinutes; m++ {
|
||||
ts := fuzzDatasetStartMs + uint64(m)*60_000
|
||||
if ts < startMs || ts >= endMs {
|
||||
continue
|
||||
}
|
||||
k := key{ts - ts%stepMs, s}
|
||||
sum[k] += d.gauges[s][m]
|
||||
n[k]++
|
||||
}
|
||||
}
|
||||
var rows [][]any
|
||||
for k := range sum {
|
||||
rows = append(rows, []any{time.UnixMilli(int64(k.ts)), fuzzServiceName(k.service), sum[k] / n[k]})
|
||||
}
|
||||
sort.Slice(rows, func(i, j int) bool {
|
||||
ti, tj := rows[i][0].(time.Time), rows[j][0].(time.Time)
|
||||
if !ti.Equal(tj) {
|
||||
return ti.Before(tj)
|
||||
}
|
||||
return rows[i][1].(string) < rows[j][1].(string)
|
||||
})
|
||||
return rows
|
||||
}
|
||||
|
||||
// rateRows evaluates the metrics statement for rate over the cumulative
|
||||
// counter: per (bucket, service) the last sample of the bucket, then for
|
||||
// each bucket the difference to the previous present bucket divided by the
|
||||
// seconds between them (resets fall back to value / dt). A bucket without a
|
||||
// predecessor within the lookback is nan, which consume drops, and the
|
||||
// lookback buckets before the window are not part of the answer.
|
||||
func (d *fuzzDataset) rateRows(startMs, endMs, stepMs uint64) [][]any {
|
||||
lookbackMs := querybuilder.RateLookbackMs(stepMs)
|
||||
var rows [][]any
|
||||
for s := 0; s < fuzzServices; s++ {
|
||||
type bucket struct {
|
||||
ts uint64
|
||||
value float64
|
||||
}
|
||||
var buckets []bucket
|
||||
for m := 0; m < fuzzDatasetMinutes; m++ {
|
||||
ts := fuzzDatasetStartMs + uint64(m)*60_000
|
||||
if ts < startMs || ts >= endMs || d.counters[s][m] < 0 {
|
||||
continue
|
||||
}
|
||||
b := ts - ts%stepMs
|
||||
if len(buckets) > 0 && buckets[len(buckets)-1].ts == b {
|
||||
buckets[len(buckets)-1].value = d.counters[s][m]
|
||||
continue
|
||||
}
|
||||
buckets = append(buckets, bucket{ts: b, value: d.counters[s][m]})
|
||||
}
|
||||
for i := 1; i < len(buckets); i++ {
|
||||
if buckets[i].ts-buckets[i-1].ts > lookbackMs || buckets[i].ts < startMs+lookbackMs {
|
||||
continue
|
||||
}
|
||||
dt := float64(buckets[i].ts-buckets[i-1].ts) / 1000
|
||||
rate := (buckets[i].value - buckets[i-1].value) / dt
|
||||
if buckets[i].value < buckets[i-1].value {
|
||||
rate = buckets[i].value / dt
|
||||
}
|
||||
rows = append(rows, []any{time.UnixMilli(int64(buckets[i].ts)), fuzzServiceName(s), rate})
|
||||
}
|
||||
}
|
||||
sort.Slice(rows, func(i, j int) bool {
|
||||
ti, tj := rows[i][0].(time.Time), rows[j][0].(time.Time)
|
||||
if !ti.Equal(tj) {
|
||||
return ti.Before(tj)
|
||||
}
|
||||
return rows[i][1].(string) < rows[j][1].(string)
|
||||
})
|
||||
return rows
|
||||
}
|
||||
|
||||
// fuzzConn answers the statements the fuzz builders render from the dataset.
|
||||
type fuzzConn struct {
|
||||
clickhouse.Conn
|
||||
data *fuzzDataset
|
||||
}
|
||||
|
||||
func (c *fuzzConn) Query(_ context.Context, query string, _ ...any) (driver.Rows, error) {
|
||||
var kind string
|
||||
var start, end, step uint64
|
||||
var limit int
|
||||
if _, err := fmt.Sscanf(query, "FUZZ %s %d %d %d %d", &kind, &start, &end, &step, &limit); err != nil {
|
||||
return nil, fmt.Errorf("fuzz conn: cannot parse %q: %w", query, err)
|
||||
}
|
||||
switch kind {
|
||||
case "logs":
|
||||
return cmock.NewRows(windowColumns, c.data.logRows(start, end, step, limit)), nil
|
||||
case "gauge":
|
||||
return cmock.NewRows(windowColumns, c.data.gaugeRows(start, end, step)), nil
|
||||
case "rate":
|
||||
return cmock.NewRows(windowColumns, c.data.rateRows(start, end, step)), nil
|
||||
}
|
||||
return nil, fmt.Errorf("fuzz conn: unknown kind %q", kind)
|
||||
}
|
||||
|
||||
type fuzzStore struct {
|
||||
*telemetrystoretest.Provider
|
||||
conn clickhouse.Conn
|
||||
}
|
||||
|
||||
func (s *fuzzStore) ClickhouseDB() clickhouse.Conn { return s.conn }
|
||||
|
||||
type fuzzLogStmtBuilder struct{}
|
||||
|
||||
func (fuzzLogStmtBuilder) Build(_ context.Context, _ valuer.UUID, start, end uint64, _ qbtypes.RequestType, query qbtypes.QueryBuilderQuery[qbtypes.LogAggregation], _ map[string]qbtypes.VariableItem) (*qbtypes.Statement, error) {
|
||||
return &qbtypes.Statement{Query: fmt.Sprintf("FUZZ logs %d %d %d %d", start, end, uint64(query.StepInterval.Milliseconds()), query.Limit)}, nil
|
||||
}
|
||||
|
||||
type fuzzMetricStmtBuilder struct{}
|
||||
|
||||
func (fuzzMetricStmtBuilder) Build(_ context.Context, _ valuer.UUID, start, end uint64, _ qbtypes.RequestType, query qbtypes.QueryBuilderQuery[qbtypes.MetricAggregation], _ map[string]qbtypes.VariableItem) (*qbtypes.Statement, error) {
|
||||
start, end = querybuilder.AdjustedMetricTimeRange(start, end, uint64(query.StepInterval.Seconds()), query)
|
||||
kind := "gauge"
|
||||
if query.Aggregations[0].TimeAggregation == metrictypes.TimeAggregationRate {
|
||||
kind = "rate"
|
||||
}
|
||||
return &qbtypes.Statement{Query: fmt.Sprintf("FUZZ %s %d %d %d 0", kind, start, end, uint64(query.StepInterval.Milliseconds()))}, nil
|
||||
}
|
||||
|
||||
// fuzzShape is one query shape a session keeps for all its requests.
|
||||
type fuzzShape struct {
|
||||
metrics bool
|
||||
rate bool
|
||||
stepMs uint64
|
||||
limit int
|
||||
shiftSec int64
|
||||
runningDiff bool
|
||||
}
|
||||
|
||||
func (s fuzzShape) String() string {
|
||||
parts := []string{fmt.Sprintf("step=%ds", s.stepMs/1000)}
|
||||
if s.rate {
|
||||
parts = append(parts, "metrics/rate")
|
||||
} else if s.metrics {
|
||||
parts = append(parts, "metrics/avg")
|
||||
} else {
|
||||
parts = append(parts, "logs/count")
|
||||
}
|
||||
if s.limit > 0 {
|
||||
parts = append(parts, fmt.Sprintf("limit=%d", s.limit))
|
||||
}
|
||||
if s.shiftSec > 0 {
|
||||
parts = append(parts, fmt.Sprintf("timeShift=%d", s.shiftSec))
|
||||
}
|
||||
if s.runningDiff {
|
||||
parts = append(parts, "runningDiff")
|
||||
}
|
||||
return strings.Join(parts, " ")
|
||||
}
|
||||
|
||||
func (s fuzzShape) envelope() (qbtypes.QueryEnvelope, qbtypes.Step) {
|
||||
step := qbtypes.Step{Duration: time.Duration(s.stepMs) * time.Millisecond}
|
||||
var functions []qbtypes.Function
|
||||
if s.shiftSec > 0 {
|
||||
functions = append(functions, qbtypes.Function{Name: qbtypes.FunctionNameTimeShift, Args: []qbtypes.FunctionArg{{Value: float64(s.shiftSec)}}})
|
||||
}
|
||||
if s.runningDiff {
|
||||
functions = append(functions, qbtypes.Function{Name: qbtypes.FunctionNameRunningDiff})
|
||||
}
|
||||
groupBy := []qbtypes.GroupByKey{{TelemetryFieldKey: telemetrytypes.TelemetryFieldKey{Name: "service.name", FieldDataType: telemetrytypes.FieldDataTypeString, FieldContext: telemetrytypes.FieldContextResource}}}
|
||||
if s.rate {
|
||||
return qbtypes.QueryEnvelope{Type: qbtypes.QueryTypeBuilder, Spec: qbtypes.QueryBuilderQuery[qbtypes.MetricAggregation]{
|
||||
Name: "A", Signal: telemetrytypes.SignalMetrics, StepInterval: step, GroupBy: groupBy, Functions: functions,
|
||||
Aggregations: []qbtypes.MetricAggregation{{MetricName: "fuzz_counter", Type: metrictypes.SumType, Temporality: metrictypes.Cumulative, TimeAggregation: metrictypes.TimeAggregationRate, SpaceAggregation: metrictypes.SpaceAggregationSum}},
|
||||
}}, step
|
||||
}
|
||||
if s.metrics {
|
||||
return qbtypes.QueryEnvelope{Type: qbtypes.QueryTypeBuilder, Spec: qbtypes.QueryBuilderQuery[qbtypes.MetricAggregation]{
|
||||
Name: "A", Signal: telemetrytypes.SignalMetrics, StepInterval: step, GroupBy: groupBy, Functions: functions,
|
||||
Aggregations: []qbtypes.MetricAggregation{{MetricName: "fuzz_gauge", Type: metrictypes.GaugeType, TimeAggregation: metrictypes.TimeAggregationAvg, SpaceAggregation: metrictypes.SpaceAggregationAvg}},
|
||||
}}, step
|
||||
}
|
||||
spec := qbtypes.QueryBuilderQuery[qbtypes.LogAggregation]{
|
||||
Name: "A", Signal: telemetrytypes.SignalLogs, StepInterval: step, GroupBy: groupBy, Functions: functions,
|
||||
Aggregations: []qbtypes.LogAggregation{{Expression: "count()"}},
|
||||
}
|
||||
if s.limit > 0 {
|
||||
spec.Limit = s.limit
|
||||
spec.Order = []qbtypes.OrderBy{{Key: qbtypes.OrderByKey{TelemetryFieldKey: telemetrytypes.TelemetryFieldKey{Name: "count()"}}, Direction: qbtypes.OrderDirectionDesc}}
|
||||
}
|
||||
return qbtypes.QueryEnvelope{Type: qbtypes.QueryTypeBuilder, Spec: spec}, step
|
||||
}
|
||||
|
||||
// fuzzPoint is the comparable projection of one response value.
|
||||
type fuzzPoint struct {
|
||||
ts int64
|
||||
value float64
|
||||
partial bool
|
||||
}
|
||||
|
||||
type fuzzResponse map[string][]fuzzPoint
|
||||
|
||||
func runFuzzRequest(t *testing.T, q *querier, orgID valuer.UUID, shape fuzzShape, window qbtypes.TimeRange, noCache bool) fuzzResponse {
|
||||
t.Helper()
|
||||
envelope, step := shape.envelope()
|
||||
req := &qbtypes.QueryRangeRequest{Start: window.From, End: window.To, RequestType: qbtypes.RequestTypeTimeSeries, NoCache: noCache, CompositeQuery: qbtypes.CompositeQuery{Queries: []qbtypes.QueryEnvelope{envelope}}}
|
||||
// The same steps QueryRange takes before run: shift extraction and window adjustment.
|
||||
var query qbtypes.Query
|
||||
switch spec := envelope.Spec.(type) {
|
||||
case qbtypes.QueryBuilderQuery[qbtypes.LogAggregation]:
|
||||
spec.ShiftBy = extractShiftFromBuilderQuery(spec)
|
||||
query = newBuilderQuery(q.logger, q.telemetryStore, orgID, q.logStmtBuilder, qbtypes.QueryTypeBuilder, spec, adjustTimeRangeForShift(spec, window, req.RequestType), req.RequestType, nil, builderConfig{})
|
||||
case qbtypes.QueryBuilderQuery[qbtypes.MetricAggregation]:
|
||||
spec.ShiftBy = extractShiftFromBuilderQuery(spec)
|
||||
query = newBuilderQuery(q.logger, q.telemetryStore, orgID, q.metricStmtBuilder, qbtypes.QueryTypeBuilder, spec, adjustTimeRangeForShift(spec, window, req.RequestType), req.RequestType, nil, builderConfig{})
|
||||
}
|
||||
resp, err := q.run(context.Background(), orgID, map[string]qbtypes.Query{"A": query}, req, map[string]qbtypes.Step{"A": step}, &qbtypes.QBEvent{}, nil)
|
||||
require.NoError(t, err)
|
||||
out := fuzzResponse{}
|
||||
for _, result := range resp.Data.Results {
|
||||
tsData, ok := result.(*qbtypes.TimeSeriesData)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
for _, agg := range tsData.Aggregations {
|
||||
for _, s := range agg.Series {
|
||||
name := ""
|
||||
if len(s.Labels) > 0 {
|
||||
name = fmt.Sprint(s.Labels[0].Value)
|
||||
}
|
||||
points := make([]fuzzPoint, 0, len(s.Values))
|
||||
for _, v := range s.Values {
|
||||
points = append(points, fuzzPoint{ts: v.Timestamp, value: v.Value, partial: v.Partial})
|
||||
}
|
||||
sort.Slice(points, func(i, j int) bool { return points[i].ts < points[j].ts })
|
||||
out[name] = points
|
||||
}
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// fuzzWindow draws a request window. Ends are on the step grid half of the
|
||||
// time and a random number of seconds off it otherwise, like dashboards.
|
||||
func fuzzWindow(rng *rand.Rand, stepMs uint64, previous *qbtypes.TimeRange, alignedOnly bool) qbtypes.TimeRange {
|
||||
datasetEnd := fuzzDatasetStartMs + uint64(fuzzDatasetMinutes)*60_000
|
||||
lengths := []uint64{30_000, 3 * 60_000, 17 * 60_000, 60 * 60_000, 2 * 60 * 60_000}
|
||||
length := lengths[rng.Intn(len(lengths))]
|
||||
var start uint64
|
||||
if previous != nil && rng.Intn(3) > 0 {
|
||||
// Related to the previous window: slide, grow, shrink, or nest.
|
||||
delta := int64(rng.Intn(31)-15) * 60_000
|
||||
start = uint64(int64(previous.From) + delta)
|
||||
if rng.Intn(2) == 0 {
|
||||
length = previous.To - previous.From
|
||||
}
|
||||
} else {
|
||||
start = fuzzDatasetStartMs + uint64(rng.Intn(fuzzDatasetMinutes-10))*60_000
|
||||
}
|
||||
if start < fuzzDatasetStartMs {
|
||||
start = fuzzDatasetStartMs
|
||||
}
|
||||
if !alignedOnly && rng.Intn(2) == 0 {
|
||||
start += uint64(rng.Intn(60)) * 1000
|
||||
}
|
||||
end := start + length
|
||||
if !alignedOnly && rng.Intn(2) == 0 {
|
||||
end += uint64(rng.Intn(60)) * 1000
|
||||
}
|
||||
if alignedOnly {
|
||||
start -= start % stepMs
|
||||
end -= end % stepMs
|
||||
}
|
||||
if end > datasetEnd {
|
||||
end = datasetEnd
|
||||
}
|
||||
if end <= start {
|
||||
end = start + stepMs
|
||||
}
|
||||
return qbtypes.TimeRange{From: start, To: end}
|
||||
}
|
||||
|
||||
type fuzzMismatch struct {
|
||||
shape fuzzShape
|
||||
window qbtypes.TimeRange
|
||||
relation string
|
||||
symptom string
|
||||
detail string
|
||||
}
|
||||
|
||||
func describeWindow(w qbtypes.TimeRange, stepMs uint64, history []qbtypes.TimeRange) string {
|
||||
var parts []string
|
||||
if w.From%stepMs != 0 {
|
||||
parts = append(parts, "start-unaligned")
|
||||
}
|
||||
if w.To%stepMs != 0 {
|
||||
parts = append(parts, "end-unaligned")
|
||||
}
|
||||
if w.To-w.From < stepMs {
|
||||
parts = append(parts, "sub-step")
|
||||
}
|
||||
relation := "first"
|
||||
if len(history) > 0 {
|
||||
relation = "disjoint"
|
||||
for _, h := range history {
|
||||
switch {
|
||||
case h.From == w.From && h.To == w.To:
|
||||
relation = "repeat"
|
||||
case w.From >= h.From && w.To <= h.To:
|
||||
relation = "inside-cached"
|
||||
case w.From <= h.From && w.To >= h.To:
|
||||
relation = "covers-cached"
|
||||
case w.From < h.To && w.To > h.From:
|
||||
relation = "overlaps-cached"
|
||||
}
|
||||
if relation != "disjoint" {
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
parts = append(parts, relation)
|
||||
return strings.Join(parts, ",")
|
||||
}
|
||||
|
||||
func compareFuzz(cached, fresh fuzzResponse) (symptom, detail string) {
|
||||
var symptoms []string
|
||||
var details []string
|
||||
for name := range fresh {
|
||||
if _, ok := cached[name]; !ok {
|
||||
symptoms = append(symptoms, "series-missing")
|
||||
details = append(details, fmt.Sprintf("%s missing", name))
|
||||
}
|
||||
}
|
||||
for name := range cached {
|
||||
if _, ok := fresh[name]; !ok {
|
||||
symptoms = append(symptoms, "series-extra")
|
||||
details = append(details, fmt.Sprintf("%s extra (%d points)", name, len(cached[name])))
|
||||
}
|
||||
}
|
||||
names := make([]string, 0, len(fresh))
|
||||
for name := range fresh {
|
||||
if _, ok := cached[name]; ok {
|
||||
names = append(names, name)
|
||||
}
|
||||
}
|
||||
sort.Strings(names)
|
||||
for _, name := range names {
|
||||
want, got := fresh[name], cached[name]
|
||||
wantByTs := map[int64]fuzzPoint{}
|
||||
for _, p := range want {
|
||||
wantByTs[p.ts] = p
|
||||
}
|
||||
gotByTs := map[int64]fuzzPoint{}
|
||||
for _, p := range got {
|
||||
gotByTs[p.ts] = p
|
||||
}
|
||||
var tss []int64
|
||||
for ts := range wantByTs {
|
||||
tss = append(tss, ts)
|
||||
}
|
||||
for ts := range gotByTs {
|
||||
if _, ok := wantByTs[ts]; !ok {
|
||||
tss = append(tss, ts)
|
||||
}
|
||||
}
|
||||
sort.Slice(tss, func(i, j int) bool { return tss[i] < tss[j] })
|
||||
for _, ts := range tss {
|
||||
g, gok := gotByTs[ts]
|
||||
w, wok := wantByTs[ts]
|
||||
switch {
|
||||
case !gok:
|
||||
symptoms = append(symptoms, "points-missing")
|
||||
details = append(details, fmt.Sprintf("%s@%s missing (fresh %g%s)", name, fuzzClock(ts), w.value, fuzzFlag(w.partial)))
|
||||
case !wok:
|
||||
symptoms = append(symptoms, "points-extra")
|
||||
details = append(details, fmt.Sprintf("%s@%s extra (cached %g%s)", name, fuzzClock(ts), g.value, fuzzFlag(g.partial)))
|
||||
case g.value != w.value:
|
||||
symptoms = append(symptoms, "value-differs")
|
||||
details = append(details, fmt.Sprintf("%s@%s cached %g%s fresh %g%s", name, fuzzClock(ts), g.value, fuzzFlag(g.partial), w.value, fuzzFlag(w.partial)))
|
||||
case g.partial != w.partial:
|
||||
symptoms = append(symptoms, "partial-flag-differs")
|
||||
details = append(details, fmt.Sprintf("%s@%s cached %g%s fresh %g%s", name, fuzzClock(ts), g.value, fuzzFlag(g.partial), w.value, fuzzFlag(w.partial)))
|
||||
}
|
||||
}
|
||||
}
|
||||
sort.Strings(symptoms)
|
||||
symptoms = uniqueStrings(symptoms)
|
||||
if len(details) > 6 {
|
||||
details = append(details[:6], fmt.Sprintf("... %d more", len(details)-6))
|
||||
}
|
||||
return strings.Join(symptoms, "+"), strings.Join(details, "; ")
|
||||
}
|
||||
|
||||
func fuzzClock(ms int64) string {
|
||||
return time.UnixMilli(ms).UTC().Format("15:04:05")
|
||||
}
|
||||
|
||||
func fuzzFlag(partial bool) string {
|
||||
if partial {
|
||||
return "(partial)"
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
func uniqueStrings(in []string) []string {
|
||||
out := in[:0]
|
||||
for i, s := range in {
|
||||
if i == 0 || s != in[i-1] {
|
||||
out = append(out, s)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// TestCacheDifferential_CachedMatchesUncached runs random request sequences
|
||||
// and requires every cached answer to equal the uncached one. The report
|
||||
// groups mismatches by shape, window relation and symptom.
|
||||
func TestCacheDifferential_CachedMatchesUncached(t *testing.T) {
|
||||
for _, seed := range []int64{20260910, 1, 2, 3} {
|
||||
t.Run(fmt.Sprintf("seed_%d", seed), func(t *testing.T) { runCacheDifferential(t, seed, 600, false) })
|
||||
}
|
||||
}
|
||||
|
||||
// TestCacheDifferential_AlignedWindowsMatchUncached keeps every window on the
|
||||
// step grid, the shape an aligned client sends.
|
||||
func TestCacheDifferential_AlignedWindowsMatchUncached(t *testing.T) {
|
||||
for _, seed := range []int64{20260910, 1} {
|
||||
t.Run(fmt.Sprintf("seed_%d", seed), func(t *testing.T) { runCacheDifferential(t, seed, 600, true) })
|
||||
}
|
||||
}
|
||||
|
||||
func runCacheDifferential(t *testing.T, seed int64, sessions int, alignedOnly bool) {
|
||||
rng := rand.New(rand.NewSource(seed))
|
||||
data := newFuzzDataset(rng)
|
||||
store := &fuzzStore{Provider: telemetrystoretest.New(telemetrystore.Config{}, sqlmock.QueryMatcherRegexp), conn: &fuzzConn{data: data}}
|
||||
|
||||
var mismatches []fuzzMismatch
|
||||
requests := 0
|
||||
for session := 0; session < sessions; session++ {
|
||||
shape := fuzzShape{stepMs: []uint64{60_000, 300_000}[rng.Intn(2)]}
|
||||
switch rng.Intn(7) {
|
||||
case 0:
|
||||
shape.limit = 1 + rng.Intn(2)
|
||||
case 1:
|
||||
shape.shiftSec = 3600
|
||||
case 2:
|
||||
shape.metrics = true
|
||||
case 3:
|
||||
shape.metrics = true
|
||||
shape.runningDiff = true
|
||||
case 4:
|
||||
shape.metrics = true
|
||||
shape.rate = true
|
||||
}
|
||||
q := &querier{
|
||||
logger: instrumentationtest.New().Logger(),
|
||||
fl: flaggertest.New(t),
|
||||
telemetryStore: store,
|
||||
logStmtBuilder: fuzzLogStmtBuilder{},
|
||||
metricStmtBuilder: fuzzMetricStmtBuilder{},
|
||||
bucketCache: createTestBucketCache(t),
|
||||
maxConcurrentQueries: DefaultMaxConcurrentQueries,
|
||||
}
|
||||
orgID := valuer.GenerateUUID()
|
||||
var history []qbtypes.TimeRange
|
||||
steps := 2 + rng.Intn(5)
|
||||
for i := 0; i < steps; i++ {
|
||||
var previous *qbtypes.TimeRange
|
||||
if len(history) > 0 {
|
||||
previous = &history[len(history)-1]
|
||||
}
|
||||
window := fuzzWindow(rng, shape.stepMs, previous, alignedOnly)
|
||||
cached := runFuzzRequest(t, q, orgID, shape, window, false)
|
||||
fresh := runFuzzRequest(t, q, orgID, shape, window, true)
|
||||
requests++
|
||||
if symptom, detail := compareFuzz(cached, fresh); symptom != "" {
|
||||
mismatches = append(mismatches, fuzzMismatch{shape: shape, window: window, relation: describeWindow(window, shape.stepMs, history), symptom: symptom, detail: detail})
|
||||
}
|
||||
history = append(history, window)
|
||||
}
|
||||
}
|
||||
|
||||
if len(mismatches) == 0 {
|
||||
return
|
||||
}
|
||||
type class struct{ shape, geometry, symptom string }
|
||||
counts := map[class]int{}
|
||||
example := map[class]fuzzMismatch{}
|
||||
for _, m := range mismatches {
|
||||
c := class{m.shape.String(), m.relation, m.symptom}
|
||||
counts[c]++
|
||||
if _, ok := example[c]; !ok {
|
||||
example[c] = m
|
||||
}
|
||||
}
|
||||
classes := make([]class, 0, len(counts))
|
||||
for c := range counts {
|
||||
classes = append(classes, c)
|
||||
}
|
||||
sort.Slice(classes, func(i, j int) bool { return counts[classes[i]] > counts[classes[j]] })
|
||||
var report strings.Builder
|
||||
fmt.Fprintf(&report, "%d of %d requests differ from the uncached answer (seed %d); %d classes\n", len(mismatches), requests, seed, len(classes))
|
||||
for _, c := range classes {
|
||||
e := example[c]
|
||||
fmt.Fprintf(&report, " %4d [%s] %s -> %s\n e.g. %s-%s: %s\n", counts[c], c.shape, c.geometry, c.symptom, fuzzClock(int64(e.window.From)), fuzzClock(int64(e.window.To)), e.detail)
|
||||
}
|
||||
t.Fatal(report.String())
|
||||
}
|
||||
@@ -3,14 +3,13 @@ package querier
|
||||
import (
|
||||
"testing"
|
||||
|
||||
"github.com/SigNoz/signoz/pkg/instrumentation/instrumentationtest"
|
||||
qbtypes "github.com/SigNoz/signoz/pkg/types/querybuildertypes/querybuildertypesv5"
|
||||
"github.com/SigNoz/signoz/pkg/types/telemetrytypes"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
func TestMergeTimeSeriesResultsUnionsHeatmapAxes(t *testing.T) {
|
||||
func TestMergeTimeSeriesDataUnionsHeatmapAxes(t *testing.T) {
|
||||
// a log axis holds whichever bands the data reached, so a wide cached range
|
||||
// and a narrow fresh one routinely disagree on which bands exist
|
||||
cached := &qbtypes.TimeSeriesData{
|
||||
@@ -24,21 +23,19 @@ func TestMergeTimeSeriesResultsUnionsHeatmapAxes(t *testing.T) {
|
||||
}},
|
||||
}},
|
||||
}
|
||||
fresh := []*qbtypes.Result{{
|
||||
Value: &qbtypes.TimeSeriesData{
|
||||
QueryName: "A",
|
||||
Aggregations: []*qbtypes.AggregationBucket{{
|
||||
Index: 0,
|
||||
Meta: qbtypes.AggregationMeta{Buckets: []float64{2, 4}},
|
||||
Series: []*qbtypes.TimeSeries{{
|
||||
Labels: []*qbtypes.Label{{Key: telemetrytypes.TelemetryFieldKey{Name: "host.name"}, Value: "node-1"}},
|
||||
Values: []*qbtypes.TimeSeriesValue{{Timestamp: 1710000060000, Values: []float64{5, 6, 7}}},
|
||||
}},
|
||||
fresh := &qbtypes.TimeSeriesData{
|
||||
QueryName: "A",
|
||||
Aggregations: []*qbtypes.AggregationBucket{{
|
||||
Index: 0,
|
||||
Meta: qbtypes.AggregationMeta{Buckets: []float64{2, 4}},
|
||||
Series: []*qbtypes.TimeSeries{{
|
||||
Labels: []*qbtypes.Label{{Key: telemetrytypes.TelemetryFieldKey{Name: "host.name"}, Value: "node-1"}},
|
||||
Values: []*qbtypes.TimeSeriesValue{{Timestamp: 1710000060000, Values: []float64{5, 6, 7}}},
|
||||
}},
|
||||
},
|
||||
}}
|
||||
}},
|
||||
}
|
||||
|
||||
merged := (&querier{}).mergeTimeSeriesResults(cached, fresh)
|
||||
merged := mergeTimeSeriesData([]*qbtypes.TimeSeriesData{cached, fresh})
|
||||
|
||||
require.Len(t, merged.Aggregations, 1)
|
||||
aggBucket := merged.Aggregations[0]
|
||||
@@ -50,40 +47,10 @@ func TestMergeTimeSeriesResultsUnionsHeatmapAxes(t *testing.T) {
|
||||
assert.Equal(t, []float64{1, 0, 2, 3, 4}, aggBucket.Series[0].Values[0].Values)
|
||||
// and the fresh 2 band survives even though the cached range never had it
|
||||
assert.Equal(t, []float64{0, 5, 6, 0, 7}, aggBucket.Series[0].Values[1].Values)
|
||||
}
|
||||
|
||||
func TestTrimResultToFluxBoundaryKeepsTheHeatmapAxis(t *testing.T) {
|
||||
cache := &bucketCache{logger: instrumentationtest.New().Logger()}
|
||||
|
||||
result := &qbtypes.Result{
|
||||
Type: qbtypes.RequestTypeHeatmap,
|
||||
Value: &qbtypes.TimeSeriesData{
|
||||
Aggregations: []*qbtypes.AggregationBucket{{
|
||||
Index: 0,
|
||||
Alias: "__result_0",
|
||||
Meta: qbtypes.AggregationMeta{Unit: "By", Buckets: []float64{1, 2, 4}},
|
||||
Series: []*qbtypes.TimeSeries{{
|
||||
Values: []*qbtypes.TimeSeriesValue{
|
||||
{Timestamp: 1710000000000, Values: []float64{1, 2, 3, 4}},
|
||||
},
|
||||
}},
|
||||
}},
|
||||
},
|
||||
}
|
||||
|
||||
trimmed := cache.trimResultToFluxBoundary(result, 1710000060000)
|
||||
|
||||
tsData, ok := trimmed.Value.(*qbtypes.TimeSeriesData)
|
||||
require.True(t, ok)
|
||||
require.Len(t, tsData.Aggregations, 1)
|
||||
|
||||
// the counts are positional against the axis, so a cached bucket that lost
|
||||
// Meta.Buckets would be realigned from an empty axis and collapse into the
|
||||
// overflow slot on the way back out
|
||||
aggBucket := tsData.Aggregations[0]
|
||||
assert.Equal(t, []float64{1, 2, 4}, aggBucket.Meta.Buckets)
|
||||
assert.Equal(t, "By", aggBucket.Meta.Unit)
|
||||
assert.Equal(t, "__result_0", aggBucket.Alias)
|
||||
// the parts are left as they were, since the fresh one is written to the cache afterwards
|
||||
assert.Equal(t, []float64{2, 4}, fresh.Aggregations[0].Meta.Buckets)
|
||||
assert.Equal(t, []float64{5, 6, 7}, fresh.Aggregations[0].Series[0].Values[0].Values)
|
||||
}
|
||||
|
||||
func TestRealignFromAnEmptyAxisCollapsesIntoTheOverflow(t *testing.T) {
|
||||
|
||||
@@ -22,7 +22,7 @@ import (
|
||||
const promHistogramBucketLabel = "le"
|
||||
|
||||
// cumulativeColumn maps a bucket's upper bound to the cumulative count at it.
|
||||
// Differencing turns it into the per-band counts a heatmapColumn holds.
|
||||
// Differencing turns it into the per-bucket counts a heatmapColumn holds.
|
||||
type cumulativeColumn map[float64]float64
|
||||
|
||||
// promHeatmapGroup assembles one group across the several matrix series its `le`
|
||||
@@ -34,8 +34,8 @@ type promHeatmapGroup struct {
|
||||
}
|
||||
|
||||
// foldMatrixAsHeatmap folds a matrix of one cumulative series per (group, `le`)
|
||||
// into one series per group whose points hold a count per band.
|
||||
func foldMatrixAsHeatmap(matrix promql.Matrix, queryWindow *qbv5.TimeRange, stepMs uint64, queryName string) (*qbv5.TimeSeriesData, error) {
|
||||
// into one series per group whose points hold a count per bucket.
|
||||
func foldMatrixAsHeatmap(matrix promql.Matrix, queryName string) (*qbv5.TimeSeriesData, error) {
|
||||
groups, groupOrder := collectCumulativeGroups(matrix)
|
||||
|
||||
// An empty matrix is only ever the window having no data, but series that
|
||||
@@ -53,11 +53,12 @@ func foldMatrixAsHeatmap(matrix promql.Matrix, queryWindow *qbv5.TimeRange, step
|
||||
}
|
||||
}
|
||||
|
||||
return accumulator.foldSeries(queryWindow, stepMs, queryName)
|
||||
// a promql data point can never be partial, hence nil and 0 are sent here
|
||||
return accumulator.foldSeries(nil, 0, queryName)
|
||||
}
|
||||
|
||||
// collectCumulativeGroups reads the matrix into one group per label set. A series
|
||||
// without `le` has no band to sit in, so an expression that dropped the label
|
||||
// without `le` has no bucket to sit in, so an expression that dropped the label
|
||||
// draws nothing.
|
||||
func collectCumulativeGroups(matrix promql.Matrix) (groups map[string]*promHeatmapGroup, groupOrder []string) {
|
||||
groups = map[string]*promHeatmapGroup{}
|
||||
|
||||
@@ -16,7 +16,7 @@ import (
|
||||
|
||||
// The cache key is the fingerprint alone, so two request types over one
|
||||
// expression must not produce the same one — a time series payload served to a
|
||||
// heatmap request has no axis and reads back as a single collapsed band.
|
||||
// heatmap request has no axis and reads back as a single collapsed bucket.
|
||||
func TestFingerprintSeparatesHeatmapFromTimeSeries(t *testing.T) {
|
||||
fingerprintFor := func(requestType qbv5.RequestType) string {
|
||||
q := &promqlQuery{
|
||||
@@ -50,7 +50,7 @@ func TestFoldMatrixAsHeatmapClampsADecreasingCumulativeCount(t *testing.T) {
|
||||
},
|
||||
}
|
||||
|
||||
data, err := foldMatrixAsHeatmap(matrix, &qbv5.TimeRange{From: 1710000000000, To: 1710000060000}, uint64(time.Minute.Milliseconds()), "A")
|
||||
data, err := foldMatrixAsHeatmap(matrix, "A")
|
||||
require.NoError(t, err)
|
||||
require.Len(t, data.Aggregations, 1)
|
||||
|
||||
@@ -76,7 +76,7 @@ func TestFoldMatrixAsHeatmapWidensTheBandOverAMissingUpperBound(t *testing.T) {
|
||||
},
|
||||
}
|
||||
|
||||
data, err := foldMatrixAsHeatmap(matrix, &qbv5.TimeRange{From: 1710000000000, To: 1710000060000}, uint64(time.Minute.Milliseconds()), "A")
|
||||
data, err := foldMatrixAsHeatmap(matrix, "A")
|
||||
require.NoError(t, err)
|
||||
require.Len(t, data.Aggregations, 1)
|
||||
|
||||
|
||||
@@ -95,13 +95,18 @@ func enhancePromQLError(query string, parseErr error) error {
|
||||
}
|
||||
|
||||
type promqlQuery struct {
|
||||
logger *slog.Logger
|
||||
promEngine prometheus.Prometheus
|
||||
parser parser.Parser
|
||||
query qbv5.PromQuery
|
||||
tr qbv5.TimeRange
|
||||
requestType qbv5.RequestType
|
||||
vars map[string]qbv5.VariableItem
|
||||
logger *slog.Logger
|
||||
promEngine prometheus.Prometheus
|
||||
parser parser.Parser
|
||||
query qbv5.PromQuery
|
||||
// tr is the evaluation range: instants tr.From, tr.From+step, ... <= tr.To.
|
||||
tr qbv5.TimeRange
|
||||
// requestWindow is the window of the request this query answers. A
|
||||
// query ranged over a gap of it renders $start_timestamp and friends
|
||||
// from here, not from the gap.
|
||||
requestWindow qbv5.TimeRange
|
||||
requestType qbv5.RequestType
|
||||
vars map[string]qbv5.VariableItem
|
||||
}
|
||||
|
||||
var _ qbv5.Query = (*promqlQuery)(nil)
|
||||
@@ -116,16 +121,29 @@ func newPromqlQuery(
|
||||
variables map[string]qbv5.VariableItem,
|
||||
) *promqlQuery {
|
||||
return &promqlQuery{
|
||||
logger: logger,
|
||||
promEngine: promEngine,
|
||||
parser: prometheus.NewParser(),
|
||||
query: query,
|
||||
tr: tr,
|
||||
requestType: requestType,
|
||||
vars: variables,
|
||||
logger: logger,
|
||||
promEngine: promEngine,
|
||||
parser: prometheus.NewParser(),
|
||||
query: query,
|
||||
tr: tr,
|
||||
requestWindow: tr,
|
||||
requestType: requestType,
|
||||
vars: variables,
|
||||
}
|
||||
}
|
||||
|
||||
// ranged copies the query over a gap [from, to) of its request window as the
|
||||
// cache reports it: to is exclusive on the step grid, so the last instant to
|
||||
// evaluate is one step before it.
|
||||
func (q *promqlQuery) ranged(gap qbv5.TimeRange) *promqlQuery {
|
||||
copied := *q
|
||||
copied.query = q.query.Copy()
|
||||
copied.tr = qbv5.TimeRange{From: gap.From, To: gap.To - uint64(q.query.Step.Milliseconds())}
|
||||
return &copied
|
||||
}
|
||||
|
||||
func (q *promqlQuery) stepMs() uint64 { return uint64(q.query.Step.Milliseconds()) }
|
||||
|
||||
func (q *promqlQuery) Fingerprint() string {
|
||||
switch q.requestType {
|
||||
case qbv5.RequestTypeTimeSeries, qbv5.RequestTypeHeatmap:
|
||||
@@ -133,11 +151,27 @@ func (q *promqlQuery) Fingerprint() string {
|
||||
return ""
|
||||
}
|
||||
|
||||
query, err := q.renderVars(q.query.Query, q.vars, q.tr.From, q.tr.To)
|
||||
// Evaluation instants are start + k*step. Only a start on the step grid
|
||||
// shares instants with other windows of the same query; anything else is
|
||||
// served without the cache rather than mixed with grid points.
|
||||
if stepMs := q.stepMs(); stepMs == 0 || q.tr.From%stepMs != 0 {
|
||||
return ""
|
||||
}
|
||||
|
||||
query, err := q.renderVars(q.query.Query, q.vars, q.requestWindow.From, q.requestWindow.To)
|
||||
if err != nil {
|
||||
q.logger.ErrorContext(context.TODO(), "failed render template variables", slog.String("query", q.query.Query))
|
||||
return ""
|
||||
}
|
||||
// @ start() and @ end() resolve to the window of the evaluation, so a
|
||||
// piece of the window evaluates something else than the whole.
|
||||
exprParser := q.parser
|
||||
if exprParser == nil {
|
||||
exprParser = prometheus.NewParser()
|
||||
}
|
||||
if expr, err := exprParser.ParseExpr(query); err != nil || usesStartOrEnd(expr) {
|
||||
return ""
|
||||
}
|
||||
parts := []string{
|
||||
"promql",
|
||||
// one expression returns a different shape per request type
|
||||
@@ -149,8 +183,30 @@ func (q *promqlQuery) Fingerprint() string {
|
||||
return strings.Join(parts, "&")
|
||||
}
|
||||
|
||||
func usesStartOrEnd(expr parser.Expr) bool {
|
||||
found := false
|
||||
parser.Inspect(expr, func(node parser.Node, _ []parser.Node) error {
|
||||
switch n := node.(type) {
|
||||
case *parser.VectorSelector:
|
||||
found = found || n.StartOrEnd != 0
|
||||
case *parser.SubqueryExpr:
|
||||
found = found || n.StartOrEnd != 0
|
||||
}
|
||||
return nil
|
||||
})
|
||||
return found
|
||||
}
|
||||
|
||||
// Window is the range of instants the query evaluates, half-open on the step
|
||||
// grid: the last instant is tr.To (or the last grid point before it), and the
|
||||
// window ends one step after it.
|
||||
func (q *promqlQuery) Window() (uint64, uint64) {
|
||||
return q.tr.From, q.tr.To
|
||||
stepMs := q.stepMs()
|
||||
if stepMs == 0 || q.tr.To < q.tr.From {
|
||||
return q.tr.From, q.tr.To
|
||||
}
|
||||
last := q.tr.From + (q.tr.To-q.tr.From)/stepMs*stepMs
|
||||
return q.tr.From, last + stepMs
|
||||
}
|
||||
|
||||
// removeAllVarMatchers removes label matchers from a PromQL query that reference variables with __all__ value.
|
||||
@@ -236,7 +292,7 @@ func (q *promqlQuery) renderVars(query string, vars map[string]qbv5.VariableItem
|
||||
// Statement renders the PromQL string (no SQL args) without executing it, for
|
||||
// the preview path.
|
||||
func (q *promqlQuery) Statement(_ context.Context) (*qbv5.Statement, error) {
|
||||
rendered, err := q.renderVars(q.query.Query, q.vars, q.tr.From, q.tr.To)
|
||||
rendered, err := q.renderVars(q.query.Query, q.vars, q.requestWindow.From, q.requestWindow.To)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
@@ -246,7 +302,7 @@ func (q *promqlQuery) Statement(_ context.Context) (*qbv5.Statement, error) {
|
||||
// PreviewStatements returns the ClickHouse statement(s) this PromQL query
|
||||
// would run on the engine path, captured without executing them.
|
||||
func (q *promqlQuery) PreviewStatements(ctx context.Context) ([]prometheus.CapturedStatement, error) {
|
||||
rendered, err := q.renderVars(q.query.Query, q.vars, q.tr.From, q.tr.To)
|
||||
rendered, err := q.renderVars(q.query.Query, q.vars, q.requestWindow.From, q.requestWindow.To)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
@@ -271,7 +327,7 @@ func (q *promqlQuery) Execute(ctx context.Context) (*qbv5.Result, error) {
|
||||
start := int64(querybuilder.ToNanoSecs(q.tr.From))
|
||||
end := int64(querybuilder.ToNanoSecs(q.tr.To))
|
||||
|
||||
query, err := q.renderVars(q.query.Query, q.vars, q.tr.From, q.tr.To)
|
||||
query, err := q.renderVars(q.query.Query, q.vars, q.requestWindow.From, q.requestWindow.To)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
@@ -352,7 +408,7 @@ func (q *promqlQuery) toResult(matrix promql.Matrix, warnings []string, began ti
|
||||
}
|
||||
|
||||
func (q *promqlQuery) toResultForHeatmap(matrix promql.Matrix, warnings []string, began time.Time, statsMu *sync.Mutex, rowsScanned, bytesScanned *uint64) (*qbv5.Result, error) {
|
||||
tsData, err := foldMatrixAsHeatmap(matrix, &q.tr, uint64(q.query.Step.Milliseconds()), q.query.Name)
|
||||
tsData, err := foldMatrixAsHeatmap(matrix, q.query.Name)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
@@ -448,6 +448,38 @@ func TestQuotedMetricOutsideBracesPattern(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// promql reports at the window start and every step after it, so only a
|
||||
// window that starts on the step grid shares instants with other windows of
|
||||
// the same query and is cached.
|
||||
func TestFingerprintCachesOnlyWindowsOnTheStepGrid(t *testing.T) {
|
||||
minuteStep := qbv5.Step{Duration: time.Minute}
|
||||
fingerprint := func(tr qbv5.TimeRange) string {
|
||||
return newPromqlQuery(slog.Default(), nil, qbv5.PromQuery{Query: "up", Step: minuteStep}, tr, qbv5.RequestTypeTimeSeries, nil).Fingerprint()
|
||||
}
|
||||
|
||||
onTheMinute := fingerprint(qbv5.TimeRange{From: 600_000, To: 1_200_000})
|
||||
halfAStepLater := fingerprint(qbv5.TimeRange{From: 630_000, To: 1_230_000})
|
||||
aWholeMinuteLater := fingerprint(qbv5.TimeRange{From: 900_000, To: 1_500_000})
|
||||
|
||||
require.NotEmpty(t, onTheMinute)
|
||||
assert.Empty(t, halfAStepLater, "a window off the grid is not cached")
|
||||
assert.Equal(t, onTheMinute, aWholeMinuteLater, "windows whole steps apart report at the same instants")
|
||||
}
|
||||
|
||||
func TestPromQLWindowIsHalfOpenOnTheStepGrid(t *testing.T) {
|
||||
minuteStep := qbv5.Step{Duration: time.Minute}
|
||||
window := func(tr qbv5.TimeRange) qbv5.TimeRange {
|
||||
from, to := newPromqlQuery(slog.Default(), nil, qbv5.PromQuery{Query: "up", Step: minuteStep}, tr, qbv5.RequestTypeTimeSeries, nil).Window()
|
||||
return qbv5.TimeRange{From: from, To: to}
|
||||
}
|
||||
|
||||
assert.Equal(t, qbv5.TimeRange{From: 600_000, To: 1_260_000}, window(qbv5.TimeRange{From: 600_000, To: 1_200_000}), "the instant at the end is evaluated and lies inside the window")
|
||||
assert.Equal(t, qbv5.TimeRange{From: 600_000, To: 1_260_000}, window(qbv5.TimeRange{From: 600_000, To: 1_230_000}), "the last instant is the last grid point at or before the end")
|
||||
|
||||
ranged := newPromqlQuery(slog.Default(), nil, qbv5.PromQuery{Query: "up", Step: minuteStep}, qbv5.TimeRange{From: 600_000, To: 1_200_000}, qbv5.RequestTypeTimeSeries, nil).ranged(qbv5.TimeRange{From: 900_000, To: 1_260_000})
|
||||
assert.Equal(t, qbv5.TimeRange{From: 900_000, To: 1_200_000}, ranged.tr, "a gap of the window evaluates up to the instant before its end")
|
||||
}
|
||||
|
||||
func TestToResultDropsNonFiniteValues(t *testing.T) {
|
||||
tests := []struct {
|
||||
description string
|
||||
|
||||
@@ -22,6 +22,8 @@ import (
|
||||
"github.com/SigNoz/signoz/pkg/query-service/utils"
|
||||
"github.com/SigNoz/signoz/pkg/querybuilder"
|
||||
"github.com/SigNoz/signoz/pkg/statsreporter"
|
||||
"github.com/SigNoz/signoz/pkg/telemetryschema/metertelemetryschema"
|
||||
"github.com/SigNoz/signoz/pkg/telemetryschema/metricstelemetryschema"
|
||||
"github.com/SigNoz/signoz/pkg/telemetrystore"
|
||||
"github.com/SigNoz/signoz/pkg/types/ctxtypes"
|
||||
"github.com/SigNoz/signoz/pkg/types/instrumentationtypes"
|
||||
@@ -218,7 +220,11 @@ func (q *querier) buildQueries(
|
||||
if !ok {
|
||||
return nil, nil, errors.NewInvalidInputf(errors.CodeInvalidInput, "invalid promql query spec %T", query.Spec)
|
||||
}
|
||||
promqlQuery := newPromqlQuery(q.logger, q.promEngine, promQuery, qbtypes.TimeRange{From: req.Start, To: req.End}, req.RequestType, tmplVars)
|
||||
timeRange := qbtypes.TimeRange{From: req.Start, To: req.End}
|
||||
if !req.NoStepAlignment {
|
||||
timeRange = alignWindowToStep(timeRange, promQuery.Step)
|
||||
}
|
||||
promqlQuery := newPromqlQuery(q.logger, q.promEngine, promQuery, timeRange, req.RequestType, tmplVars)
|
||||
queries[promQuery.Name] = promqlQuery
|
||||
steps[promQuery.Name] = promQuery.Step
|
||||
case qbtypes.QueryTypeClickHouseSQL:
|
||||
@@ -671,13 +677,21 @@ func (q *querier) run(
|
||||
eg, egCtx := errgroup.WithContext(ctx)
|
||||
for i, name := range names {
|
||||
query := qs[name]
|
||||
eg.Go(func() error {
|
||||
eg.Go(func() (err error) {
|
||||
// A panic here would end the process: errgroup does not recover
|
||||
// and the HTTP recovery middleware only covers the handler goroutine.
|
||||
defer func() {
|
||||
if r := recover(); r != nil {
|
||||
q.logger.ErrorContext(egCtx, "query execution panicked", slog.String("query", name), slog.Any("panic", r))
|
||||
err = errors.NewInternalf(errors.CodeInternal, "query %s failed", name)
|
||||
}
|
||||
}()
|
||||
// Skip cache if NoCache is set, or if cache is not available
|
||||
if req.NoCache || q.bucketCache == nil || query.Fingerprint() == "" {
|
||||
if req.NoCache {
|
||||
q.logger.DebugContext(egCtx, "NoCache flag set, bypassing cache", slog.String("query", name))
|
||||
} else {
|
||||
q.logger.InfoContext(egCtx, "no bucket cache or fingerprint, executing query", slog.String("fingerprint", query.Fingerprint()))
|
||||
q.logger.DebugContext(egCtx, "no bucket cache or fingerprint, executing query", slog.String("query", name))
|
||||
}
|
||||
sem <- struct{}{}
|
||||
result, err := query.Execute(egCtx)
|
||||
@@ -777,120 +791,107 @@ func (q *querier) run(
|
||||
return resp, nil
|
||||
}
|
||||
|
||||
// executeWithCache executes a query using the bucket cache. sem limits how
|
||||
// many queries run at once for the whole request.
|
||||
// executeWithCache serves a query from the bucket cache: the cached part of
|
||||
// the window plus one statement per missing range, merged and written back
|
||||
// range by range. sem limits how many statements run at once for the whole
|
||||
// request.
|
||||
func (q *querier) executeWithCache(ctx context.Context, orgID valuer.UUID, query qbtypes.Query, step qbtypes.Step, sem chan struct{}) (*qbtypes.Result, error) {
|
||||
// Get cached data and missing ranges
|
||||
cachedResult, missingRanges := q.bucketCache.GetMissRanges(ctx, orgID, query, step)
|
||||
|
||||
// If no missing ranges, return cached result
|
||||
if len(missingRanges) == 0 && cachedResult != nil {
|
||||
return cachedResult, nil
|
||||
from, to := query.Window()
|
||||
stepMs := uint64(step.Milliseconds())
|
||||
// Functions such as runningDiff need the step before the window; the
|
||||
// cache window includes it so a hit carries it too.
|
||||
lookbackMs := uint64(lookbackSteps(query)) * stepMs
|
||||
req := CacheRequest{
|
||||
Key: CacheKey(query.Fingerprint()),
|
||||
Window: qbtypes.TimeRange{From: from - min(lookbackMs, from), To: to},
|
||||
Step: step,
|
||||
Kind: queryKind(query),
|
||||
TrimHeatmapAxis: trimsHeatmapAxis(query),
|
||||
}
|
||||
|
||||
// If entire range is missing, execute normally
|
||||
if cachedResult == nil && len(missingRanges) == 1 {
|
||||
startMs, endMs := query.Window()
|
||||
if missingRanges[0].From == startMs && missingRanges[0].To == endMs {
|
||||
sem <- struct{}{}
|
||||
result, err := query.Execute(ctx)
|
||||
<-sem
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
// Store in cache for future use
|
||||
q.bucketCache.Put(ctx, orgID, query, step, result)
|
||||
return result, nil
|
||||
execute := func(qry qbtypes.Query) (*qbtypes.Result, error) {
|
||||
sem <- struct{}{}
|
||||
defer func() { <-sem }()
|
||||
return qry.Execute(ctx)
|
||||
}
|
||||
|
||||
cached, missing := q.bucketCache.GetMissRanges(ctx, orgID, req)
|
||||
if len(missing) == 0 && cached != nil {
|
||||
flagPartialPoints(query, cached, from, to, stepMs)
|
||||
return cached, nil
|
||||
}
|
||||
// A statement that ranks or limits over its window cannot be assembled
|
||||
// from pieces, and a window that is entirely missing is cheaper as one
|
||||
// statement; both run the original query over its own window.
|
||||
entirelyMissing := cached == nil && len(missing) == 1 && missing[0] == req.Window
|
||||
if entirelyMissing || wholeWindowOnly(query) || len(missing) == 0 {
|
||||
result, err := execute(query)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
q.bucketCache.Put(ctx, orgID, req, req.Window, result)
|
||||
return result, nil
|
||||
}
|
||||
|
||||
// Execute queries for missing ranges with bounded parallelism
|
||||
freshResults := make([]*qbtypes.Result, len(missingRanges))
|
||||
errs := make([]error, len(missingRanges))
|
||||
totalStats := qbtypes.ExecStats{}
|
||||
|
||||
q.logger.DebugContext(ctx, "executing queries for missing ranges",
|
||||
slog.Int("missing_ranges_count", len(missingRanges)),
|
||||
slog.Any("ranges", missingRanges))
|
||||
|
||||
fresh := make([]*qbtypes.Result, len(missing))
|
||||
errs := make([]error, len(missing))
|
||||
var wg sync.WaitGroup
|
||||
|
||||
for i, timeRange := range missingRanges {
|
||||
for i, timeRange := range missing {
|
||||
wg.Add(1)
|
||||
go func(idx int, tr *qbtypes.TimeRange) {
|
||||
go func(i int, timeRange qbtypes.TimeRange) {
|
||||
defer wg.Done()
|
||||
|
||||
sem <- struct{}{}
|
||||
defer func() { <-sem }()
|
||||
|
||||
// Create a new query with the missing time range
|
||||
rangedQuery := q.createRangedQuery(orgID, query, *tr)
|
||||
if rangedQuery == nil {
|
||||
errs[idx] = errors.NewInternalf(errors.CodeInternal, "failed to create ranged query for range %d-%d", tr.From, tr.To)
|
||||
ranged := q.createRangedQuery(query, timeRange)
|
||||
if ranged == nil {
|
||||
errs[i] = errors.NewInternalf(errors.CodeInternal, "cannot range query over %d-%d", timeRange.From, timeRange.To)
|
||||
return
|
||||
}
|
||||
|
||||
// Execute the ranged query
|
||||
result, err := rangedQuery.Execute(ctx)
|
||||
if err != nil {
|
||||
errs[idx] = err
|
||||
return
|
||||
}
|
||||
|
||||
freshResults[idx] = result
|
||||
fresh[i], errs[i] = execute(ranged)
|
||||
}(i, timeRange)
|
||||
}
|
||||
|
||||
// Wait for all queries to complete
|
||||
wg.Wait()
|
||||
|
||||
// Check for errors
|
||||
for _, err := range errs {
|
||||
if err != nil {
|
||||
// If any query failed, fall back to full execution
|
||||
q.logger.ErrorContext(ctx, "parallel query execution failed", errors.Attr(err))
|
||||
sem <- struct{}{}
|
||||
result, err := query.Execute(ctx)
|
||||
<-sem
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
q.bucketCache.Put(ctx, orgID, query, step, result)
|
||||
return result, nil
|
||||
return nil, err
|
||||
}
|
||||
}
|
||||
|
||||
// Calculate total stats and filter out nil results
|
||||
validResults := make([]*qbtypes.Result, 0, len(freshResults))
|
||||
for _, result := range freshResults {
|
||||
if result != nil {
|
||||
validResults = append(validResults, result)
|
||||
totalStats.RowsScanned += result.Stats.RowsScanned
|
||||
totalStats.BytesScanned += result.Stats.BytesScanned
|
||||
totalStats.DurationMS += result.Stats.DurationMS
|
||||
}
|
||||
merged := mergeResults(req, cached, fresh)
|
||||
for i, timeRange := range missing {
|
||||
q.bucketCache.Put(ctx, orgID, req, timeRange, fresh[i])
|
||||
}
|
||||
freshResults = validResults
|
||||
|
||||
// Merge cached and fresh results
|
||||
mergedResult := q.mergeResults(cachedResult, freshResults)
|
||||
mergedResult.Stats.RowsScanned += totalStats.RowsScanned
|
||||
mergedResult.Stats.BytesScanned += totalStats.BytesScanned
|
||||
mergedResult.Stats.DurationMS += totalStats.DurationMS
|
||||
|
||||
// Store merged result in cache
|
||||
q.bucketCache.Put(ctx, orgID, query, step, mergedResult)
|
||||
|
||||
return mergedResult, nil
|
||||
flagPartialPoints(query, merged, from, to, stepMs)
|
||||
return merged, nil
|
||||
}
|
||||
|
||||
// createRangedQuery creates a copy of the query with a different time range.
|
||||
func (q *querier) createRangedQuery(_ valuer.UUID, originalQuery qbtypes.Query, timeRange qbtypes.TimeRange) qbtypes.Query {
|
||||
// this is called in a goroutine, so we create a copy of the query to avoid race conditions
|
||||
switch qt := originalQuery.(type) {
|
||||
// flagPartialPoints marks the points of a served result the way consume
|
||||
// marks them for the query's own window: a bucket holds the flag of the
|
||||
// window that fetched it, and a piece is fetched over a window of its own.
|
||||
// PromQL evaluates instants and has no partial points.
|
||||
func flagPartialPoints(query qbtypes.Query, result *qbtypes.Result, from, to, stepMs uint64) {
|
||||
if _, ok := query.(*promqlQuery); ok || result == nil {
|
||||
return
|
||||
}
|
||||
data, ok := result.Value.(*qbtypes.TimeSeriesData)
|
||||
if !ok || data == nil {
|
||||
return
|
||||
}
|
||||
window := &qbtypes.TimeRange{From: from, To: to}
|
||||
for _, agg := range data.Aggregations {
|
||||
for _, s := range agg.Series {
|
||||
for _, v := range s.Values {
|
||||
v.Partial = isPartialValue(v.Timestamp, window, stepMs)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// createRangedQuery copies a query over another window. The window is in the
|
||||
// query's own clock: a timeShift query already reports a shifted window, so
|
||||
// the copy takes the range as it is.
|
||||
func (q *querier) createRangedQuery(original qbtypes.Query, timeRange qbtypes.TimeRange) qbtypes.Query {
|
||||
switch qt := original.(type) {
|
||||
case *promqlQuery:
|
||||
queryCopy := qt.query.Copy()
|
||||
return newPromqlQuery(q.logger, qt.promEngine, queryCopy, timeRange, qt.requestType, qt.vars)
|
||||
return qt.ranged(timeRange)
|
||||
|
||||
case *chSQLQuery:
|
||||
queryCopy := qt.query.Copy()
|
||||
@@ -899,40 +900,34 @@ func (q *querier) createRangedQuery(_ valuer.UUID, originalQuery qbtypes.Query,
|
||||
return newchSQLQuery(q.logger, q.telemetryStore, queryCopy, argsCopy, timeRange, qt.kind, qt.vars)
|
||||
|
||||
case *builderQuery[qbtypes.TraceAggregation]:
|
||||
specCopy := qt.spec.Copy()
|
||||
specCopy.ShiftBy = extractShiftFromBuilderQuery(specCopy)
|
||||
adjustedTimeRange := adjustTimeRangeForShift(specCopy, timeRange, qt.kind)
|
||||
// reuse the original query's statement builder and type so an AI query
|
||||
// keeps its AI builder and cache key
|
||||
return newBuilderQuery(q.logger, q.telemetryStore, qt.orgID, qt.stmtBuilder, qt.queryType, specCopy, adjustedTimeRange, qt.kind, qt.variables, qt.builderConfig)
|
||||
return newBuilderQuery(q.logger, q.telemetryStore, qt.orgID, qt.stmtBuilder, qt.queryType, qt.spec.Copy(), timeRange, qt.kind, qt.variables, qt.builderConfig)
|
||||
|
||||
case *builderQuery[qbtypes.LogAggregation]:
|
||||
specCopy := qt.spec.Copy()
|
||||
specCopy.ShiftBy = extractShiftFromBuilderQuery(specCopy)
|
||||
adjustedTimeRange := adjustTimeRangeForShift(specCopy, timeRange, qt.kind)
|
||||
shiftStmtBuilder := q.logStmtBuilder
|
||||
if qt.spec.Source == telemetrytypes.SourceAudit {
|
||||
shiftStmtBuilder = q.auditStmtBuilder
|
||||
}
|
||||
return newBuilderQuery(q.logger, q.telemetryStore, qt.orgID, shiftStmtBuilder, qt.queryType, specCopy, adjustedTimeRange, qt.kind, qt.variables, q.builderConfig)
|
||||
return newBuilderQuery(q.logger, q.telemetryStore, qt.orgID, qt.stmtBuilder, qt.queryType, qt.spec.Copy(), timeRange, qt.kind, qt.variables, qt.builderConfig)
|
||||
|
||||
case *builderQuery[qbtypes.MetricAggregation]:
|
||||
specCopy := qt.spec.Copy()
|
||||
specCopy.ShiftBy = extractShiftFromBuilderQuery(specCopy)
|
||||
adjustedTimeRange := adjustTimeRangeForShift(specCopy, timeRange, qt.kind)
|
||||
if qt.spec.Source == telemetrytypes.SourceMeter {
|
||||
return newBuilderQuery(q.logger, q.telemetryStore, qt.orgID, q.meterStmtBuilder, qt.queryType, specCopy, adjustedTimeRange, qt.kind, qt.variables, builderConfig{})
|
||||
// The builder picks its tables from the window it is given; a piece
|
||||
// must read the tables the whole request reads.
|
||||
for i, agg := range specCopy.Aggregations {
|
||||
if specCopy.Source == telemetrytypes.SourceMeter {
|
||||
specCopy.Aggregations[i].TableHints = metertelemetryschema.TableHintsForWindow(qt.fromMS, qt.toMS, agg.Type, agg.TimeAggregation, agg.TableHints)
|
||||
} else {
|
||||
specCopy.Aggregations[i].TableHints = metricstelemetryschema.TableHintsForWindow(qt.fromMS, qt.toMS, agg.Type, agg.TimeAggregation, agg.Reduced, agg.TableHints)
|
||||
}
|
||||
}
|
||||
return newBuilderQuery(q.logger, q.telemetryStore, qt.orgID, q.metricStmtBuilder, qt.queryType, specCopy, adjustedTimeRange, qt.kind, qt.variables, builderConfig{})
|
||||
return newBuilderQuery(q.logger, q.telemetryStore, qt.orgID, qt.stmtBuilder, qt.queryType, specCopy, timeRange, qt.kind, qt.variables, qt.builderConfig)
|
||||
|
||||
case *traceOperatorQuery:
|
||||
specCopy := qt.spec.Copy()
|
||||
return &traceOperatorQuery{
|
||||
telemetryStore: q.telemetryStore,
|
||||
orgID: qt.orgID,
|
||||
stmtBuilder: q.traceOperatorStmtBuilder,
|
||||
spec: specCopy,
|
||||
fromMS: uint64(timeRange.From),
|
||||
toMS: uint64(timeRange.To),
|
||||
spec: qt.spec.Copy(),
|
||||
fromMS: timeRange.From,
|
||||
toMS: timeRange.To,
|
||||
compositeQuery: qt.compositeQuery,
|
||||
kind: qt.kind,
|
||||
}
|
||||
@@ -941,217 +936,117 @@ func (q *querier) createRangedQuery(_ valuer.UUID, originalQuery qbtypes.Query,
|
||||
}
|
||||
}
|
||||
|
||||
// mergeResults merges cached result with fresh results.
|
||||
func (q *querier) mergeResults(cached *qbtypes.Result, fresh []*qbtypes.Result) *qbtypes.Result {
|
||||
if cached == nil {
|
||||
if len(fresh) == 1 {
|
||||
return fresh[0]
|
||||
// mergeResults joins the cached part with the fresh pieces. Fresh points win
|
||||
// over cached points at the same timestamp, and a partial point never wins
|
||||
// over a whole one (a metrics piece returns the step before its range as a
|
||||
// partial point that the cached part already holds whole).
|
||||
func mergeResults(req CacheRequest, cached *qbtypes.Result, fresh []*qbtypes.Result) *qbtypes.Result {
|
||||
merged := &qbtypes.Result{Type: req.Kind}
|
||||
parts := make([]*qbtypes.TimeSeriesData, 0, len(fresh)+1)
|
||||
add := func(result *qbtypes.Result) {
|
||||
if result == nil {
|
||||
return
|
||||
}
|
||||
if len(fresh) == 0 {
|
||||
return nil
|
||||
if data, ok := result.Value.(*qbtypes.TimeSeriesData); ok && data != nil {
|
||||
parts = append(parts, data)
|
||||
}
|
||||
// If cached is nil but we have multiple fresh results, we need to merge them
|
||||
// We need to merge all fresh results properly to avoid duplicates
|
||||
merged := &qbtypes.Result{
|
||||
Type: fresh[0].Type,
|
||||
Stats: fresh[0].Stats,
|
||||
Warnings: fresh[0].Warnings,
|
||||
WarningsDocURL: fresh[0].WarningsDocURL,
|
||||
merged.Stats.RowsScanned += result.Stats.RowsScanned
|
||||
merged.Stats.BytesScanned += result.Stats.BytesScanned
|
||||
merged.Stats.DurationMS += result.Stats.DurationMS
|
||||
merged.Warnings = append(merged.Warnings, result.Warnings...)
|
||||
if merged.WarningsDocURL == "" {
|
||||
merged.WarningsDocURL = result.WarningsDocURL
|
||||
}
|
||||
|
||||
// Merge all fresh results including the first one
|
||||
switch merged.Type {
|
||||
case qbtypes.RequestTypeTimeSeries, qbtypes.RequestTypeHeatmap:
|
||||
// Pass nil as cached value to ensure proper merging of all fresh results
|
||||
merged.Value = q.mergeTimeSeriesResults(nil, fresh)
|
||||
}
|
||||
|
||||
return merged
|
||||
}
|
||||
|
||||
// Start with cached result
|
||||
merged := &qbtypes.Result{
|
||||
Type: cached.Type,
|
||||
Value: cached.Value,
|
||||
Stats: cached.Stats,
|
||||
Warnings: cached.Warnings,
|
||||
WarningsDocURL: cached.WarningsDocURL,
|
||||
add(cached)
|
||||
for _, result := range fresh {
|
||||
add(result)
|
||||
}
|
||||
|
||||
// If no fresh results, return cached
|
||||
if len(fresh) == 0 {
|
||||
return merged
|
||||
}
|
||||
|
||||
switch merged.Type {
|
||||
case qbtypes.RequestTypeTimeSeries, qbtypes.RequestTypeHeatmap:
|
||||
merged.Value = q.mergeTimeSeriesResults(cached.Value.(*qbtypes.TimeSeriesData), fresh)
|
||||
}
|
||||
|
||||
if len(fresh) > 0 {
|
||||
totalWarnings := len(merged.Warnings)
|
||||
for _, result := range fresh {
|
||||
totalWarnings += len(result.Warnings)
|
||||
}
|
||||
|
||||
allWarnings := make([]string, 0, totalWarnings)
|
||||
allWarnings = append(allWarnings, merged.Warnings...)
|
||||
for _, result := range fresh {
|
||||
allWarnings = append(allWarnings, result.Warnings...)
|
||||
}
|
||||
merged.Warnings = allWarnings
|
||||
}
|
||||
|
||||
merged.Warnings = dedupeWarnings(merged.Warnings)
|
||||
stepMs := uint64(req.Step.Milliseconds())
|
||||
windowStart := req.Window.From - req.Window.From%stepMs
|
||||
// A piece widened by the builder reports points before its own range;
|
||||
// only the ones inside the request window (and its partial first step)
|
||||
// belong to the answer, as with one statement over the whole window.
|
||||
merged.Value = selectPoints(mergeTimeSeriesData(parts), func(v *qbtypes.TimeSeriesValue) bool {
|
||||
ts := uint64(v.Timestamp)
|
||||
return ts >= windowStart && ts < req.Window.To
|
||||
})
|
||||
return merged
|
||||
}
|
||||
|
||||
func mergeBucketUpperBounds(cachedValue *qbtypes.TimeSeriesData, freshResults []*qbtypes.Result) map[int][]float64 {
|
||||
upperBoundSources := make([]*qbtypes.TimeSeriesData, 0, len(freshResults)+1)
|
||||
upperBoundSources = append(upperBoundSources, cachedValue)
|
||||
for _, result := range freshResults {
|
||||
freshTS, _ := result.Value.(*qbtypes.TimeSeriesData)
|
||||
upperBoundSources = append(upperBoundSources, freshTS)
|
||||
// alignWindowToStep moves both ends of a window down to the step grid, the
|
||||
// way a query frontend does before a results cache: PromQL evaluates at
|
||||
// start + k*step, so only windows on one grid share instants.
|
||||
func alignWindowToStep(window qbtypes.TimeRange, step qbtypes.Step) qbtypes.TimeRange {
|
||||
stepMs := uint64(step.Milliseconds())
|
||||
if stepMs == 0 {
|
||||
return window
|
||||
}
|
||||
return qbtypes.MergeBucketUpperBounds(upperBoundSources...)
|
||||
return qbtypes.TimeRange{From: window.From - window.From%stepMs, To: window.To - window.To%stepMs}
|
||||
}
|
||||
|
||||
// mergeTimeSeriesResults merges time series data.
|
||||
func (q *querier) mergeTimeSeriesResults(cachedValue *qbtypes.TimeSeriesData, freshResults []*qbtypes.Result) *qbtypes.TimeSeriesData {
|
||||
// queryKind is the request type a query answers with.
|
||||
func queryKind(query qbtypes.Query) qbtypes.RequestType {
|
||||
switch qt := query.(type) {
|
||||
case *promqlQuery:
|
||||
return qt.requestType
|
||||
case *builderQuery[qbtypes.TraceAggregation]:
|
||||
return qt.kind
|
||||
case *builderQuery[qbtypes.LogAggregation]:
|
||||
return qt.kind
|
||||
case *builderQuery[qbtypes.MetricAggregation]:
|
||||
return qt.kind
|
||||
case *chSQLQuery:
|
||||
return qt.kind
|
||||
case *traceOperatorQuery:
|
||||
return qt.kind
|
||||
}
|
||||
return qbtypes.RequestTypeTimeSeries
|
||||
}
|
||||
|
||||
// Map to store merged series by aggregation index and series key
|
||||
seriesMap := make(map[int]map[string]*qbtypes.TimeSeries)
|
||||
// Map to store aggregation bucket metadata
|
||||
bucketMetadata := make(map[int]*qbtypes.AggregationBucket)
|
||||
// wholeWindowOnly reports whether the query's statement depends on the
|
||||
// whole window, so its cached result serves only the identical window.
|
||||
func wholeWindowOnly(query qbtypes.Query) bool {
|
||||
switch qt := query.(type) {
|
||||
case *builderQuery[qbtypes.TraceAggregation]:
|
||||
return qt.wholeWindowOnly()
|
||||
case *builderQuery[qbtypes.LogAggregation]:
|
||||
return qt.wholeWindowOnly()
|
||||
case *builderQuery[qbtypes.MetricAggregation]:
|
||||
return qt.wholeWindowOnly()
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
mergedUpperBounds := mergeBucketUpperBounds(cachedValue, freshResults)
|
||||
// lookbackSteps is how many steps before the window the answer must carry.
|
||||
func lookbackSteps(query qbtypes.Query) int {
|
||||
if qt, ok := query.(*builderQuery[qbtypes.MetricAggregation]); ok {
|
||||
return qt.lookbackSteps()
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// Process cached data if available
|
||||
if cachedValue != nil && cachedValue.Aggregations != nil {
|
||||
for _, aggBucket := range cachedValue.Aggregations {
|
||||
if seriesMap[aggBucket.Index] == nil {
|
||||
seriesMap[aggBucket.Index] = make(map[string]*qbtypes.TimeSeries)
|
||||
}
|
||||
aggBucket.ReindexValuesToNewUpperBounds(mergedUpperBounds[aggBucket.Index])
|
||||
if bucketMetadata[aggBucket.Index] == nil {
|
||||
bucketMetadata[aggBucket.Index] = aggBucket
|
||||
}
|
||||
for _, series := range aggBucket.Series {
|
||||
key := qbtypes.GetUniqueSeriesKey(series.Labels)
|
||||
if existingSeries, ok := seriesMap[aggBucket.Index][key]; ok {
|
||||
// Merge values from duplicate series in cached data, avoiding duplicate timestamps
|
||||
timestampMap := make(map[int64]bool)
|
||||
for _, v := range existingSeries.Values {
|
||||
timestampMap[v.Timestamp] = true
|
||||
}
|
||||
|
||||
// Only add values with new timestamps
|
||||
for _, v := range series.Values {
|
||||
if !timestampMap[v.Timestamp] {
|
||||
existingSeries.Values = append(existingSeries.Values, v)
|
||||
}
|
||||
}
|
||||
} else {
|
||||
// Create a copy to avoid modifying the cached data
|
||||
seriesCopy := &qbtypes.TimeSeries{
|
||||
Labels: series.Labels,
|
||||
Values: make([]*qbtypes.TimeSeriesValue, len(series.Values)),
|
||||
}
|
||||
copy(seriesCopy.Values, series.Values)
|
||||
seriesMap[aggBucket.Index][key] = seriesCopy
|
||||
}
|
||||
// trimsHeatmapAxis reports whether the query computes its heatmap axis from
|
||||
// the served columns. A histogram metric, promql and clickhouse name their
|
||||
// own buckets, and an empty one of theirs still belongs on the axis.
|
||||
func trimsHeatmapAxis(query qbtypes.Query) bool {
|
||||
switch qt := query.(type) {
|
||||
case *builderQuery[qbtypes.TraceAggregation]:
|
||||
return qt.kind == qbtypes.RequestTypeHeatmap
|
||||
case *builderQuery[qbtypes.LogAggregation]:
|
||||
return qt.kind == qbtypes.RequestTypeHeatmap
|
||||
case *builderQuery[qbtypes.MetricAggregation]:
|
||||
if qt.kind != qbtypes.RequestTypeHeatmap {
|
||||
return false
|
||||
}
|
||||
for _, agg := range qt.spec.Aggregations {
|
||||
if agg.HeatmapBucketing != nil {
|
||||
return true
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Add fresh series
|
||||
for _, result := range freshResults {
|
||||
freshTS, ok := result.Value.(*qbtypes.TimeSeriesData)
|
||||
if !ok || freshTS == nil || freshTS.Aggregations == nil {
|
||||
continue
|
||||
}
|
||||
|
||||
for _, aggBucket := range freshTS.Aggregations {
|
||||
if seriesMap[aggBucket.Index] == nil {
|
||||
seriesMap[aggBucket.Index] = make(map[string]*qbtypes.TimeSeries)
|
||||
}
|
||||
// Prefer fresh metadata over cached metadata
|
||||
if aggBucket.Alias != "" || aggBucket.Meta.Unit != "" {
|
||||
bucketMetadata[aggBucket.Index] = aggBucket
|
||||
} else if bucketMetadata[aggBucket.Index] == nil {
|
||||
bucketMetadata[aggBucket.Index] = aggBucket
|
||||
}
|
||||
}
|
||||
|
||||
for _, aggBucket := range freshTS.Aggregations {
|
||||
aggBucket.ReindexValuesToNewUpperBounds(mergedUpperBounds[aggBucket.Index])
|
||||
for _, series := range aggBucket.Series {
|
||||
key := qbtypes.GetUniqueSeriesKey(series.Labels)
|
||||
|
||||
if existingSeries, ok := seriesMap[aggBucket.Index][key]; ok {
|
||||
// Merge values, avoiding duplicate timestamps
|
||||
// Create a map to track existing timestamps
|
||||
timestampMap := make(map[int64]bool)
|
||||
for _, v := range existingSeries.Values {
|
||||
timestampMap[v.Timestamp] = true
|
||||
}
|
||||
|
||||
// Only add values with new timestamps
|
||||
for _, v := range series.Values {
|
||||
if !timestampMap[v.Timestamp] {
|
||||
existingSeries.Values = append(existingSeries.Values, v)
|
||||
}
|
||||
}
|
||||
} else {
|
||||
// New series
|
||||
seriesMap[aggBucket.Index][key] = series
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
result := &qbtypes.TimeSeriesData{
|
||||
Aggregations: []*qbtypes.AggregationBucket{},
|
||||
}
|
||||
|
||||
// Set QueryName from cached or first fresh result
|
||||
if cachedValue != nil {
|
||||
result.QueryName = cachedValue.QueryName
|
||||
} else if len(freshResults) > 0 {
|
||||
if freshTS, ok := freshResults[0].Value.(*qbtypes.TimeSeriesData); ok && freshTS != nil {
|
||||
result.QueryName = freshTS.QueryName
|
||||
}
|
||||
}
|
||||
|
||||
for index, series := range seriesMap {
|
||||
var aggSeries []*qbtypes.TimeSeries
|
||||
for _, s := range series {
|
||||
// Sort values by timestamp
|
||||
slices.SortFunc(s.Values, func(a, b *qbtypes.TimeSeriesValue) int {
|
||||
if a.Timestamp < b.Timestamp {
|
||||
return -1
|
||||
}
|
||||
if a.Timestamp > b.Timestamp {
|
||||
return 1
|
||||
}
|
||||
return 0
|
||||
})
|
||||
aggSeries = append(aggSeries, s)
|
||||
}
|
||||
|
||||
// Preserve bucket metadata from either cached or fresh results
|
||||
bucket := &qbtypes.AggregationBucket{
|
||||
Index: index,
|
||||
Series: aggSeries,
|
||||
}
|
||||
if metadata, ok := bucketMetadata[index]; ok {
|
||||
bucket.Alias = metadata.Alias
|
||||
bucket.Meta = metadata.Meta
|
||||
}
|
||||
|
||||
result.Aggregations = append(result.Aggregations, bucket)
|
||||
}
|
||||
|
||||
return result
|
||||
return false
|
||||
}
|
||||
|
||||
func secondsStep(s uint64) qbtypes.Step {
|
||||
|
||||
551
pkg/querier/querier_cache_test.go
Normal file
551
pkg/querier/querier_cache_test.go
Normal file
@@ -0,0 +1,551 @@
|
||||
package querier
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"regexp"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/DATA-DOG/go-sqlmock"
|
||||
cmock "github.com/SigNoz/clickhouse-go-mock"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
|
||||
"github.com/SigNoz/signoz/pkg/flagger/flaggertest"
|
||||
"github.com/SigNoz/signoz/pkg/instrumentation/instrumentationtest"
|
||||
"github.com/SigNoz/signoz/pkg/prometheus"
|
||||
"github.com/SigNoz/signoz/pkg/prometheus/prometheustest"
|
||||
"github.com/SigNoz/signoz/pkg/querybuilder"
|
||||
"github.com/SigNoz/signoz/pkg/telemetryschema/metertelemetryschema"
|
||||
"github.com/SigNoz/signoz/pkg/telemetryschema/metricstelemetryschema"
|
||||
"github.com/SigNoz/signoz/pkg/telemetrystore"
|
||||
"github.com/SigNoz/signoz/pkg/telemetrystore/telemetrystoretest"
|
||||
"github.com/SigNoz/signoz/pkg/types/metrictypes"
|
||||
qbtypes "github.com/SigNoz/signoz/pkg/types/querybuildertypes/querybuildertypesv5"
|
||||
"github.com/SigNoz/signoz/pkg/types/telemetrytypes"
|
||||
"github.com/SigNoz/signoz/pkg/valuer"
|
||||
)
|
||||
|
||||
// windowLogStmtBuilder stands in for the logs statement builder. It records
|
||||
// every window it is asked to build and renders the window and the limit
|
||||
// into the SQL text so the ClickHouse mock can answer each statement.
|
||||
type windowLogStmtBuilder struct {
|
||||
mu sync.Mutex
|
||||
ranges []qbtypes.TimeRange
|
||||
}
|
||||
|
||||
func (b *windowLogStmtBuilder) Build(_ context.Context, _ valuer.UUID, start, end uint64, _ qbtypes.RequestType, query qbtypes.QueryBuilderQuery[qbtypes.LogAggregation], _ map[string]qbtypes.VariableItem) (*qbtypes.Statement, error) {
|
||||
b.mu.Lock()
|
||||
defer b.mu.Unlock()
|
||||
b.ranges = append(b.ranges, qbtypes.TimeRange{From: start, To: end})
|
||||
return &qbtypes.Statement{Query: windowSQL(start, end, query.Limit)}, nil
|
||||
}
|
||||
|
||||
func (b *windowLogStmtBuilder) built() []qbtypes.TimeRange {
|
||||
b.mu.Lock()
|
||||
defer b.mu.Unlock()
|
||||
return append([]qbtypes.TimeRange(nil), b.ranges...)
|
||||
}
|
||||
|
||||
func windowSQL(start, end uint64, limit int) string {
|
||||
return fmt.Sprintf("SELECT ts, `service.name`, __result_0 FROM logs WHERE range = '%d-%d' AND lim = %d", start, end, limit)
|
||||
}
|
||||
|
||||
var windowColumns = []cmock.ColumnType{
|
||||
{Name: "ts", Type: "DateTime"},
|
||||
{Name: "service.name", Type: "String"},
|
||||
{Name: "__result_0", Type: "Float64"},
|
||||
}
|
||||
|
||||
// windowRows renders one row per minute and service in [start, end).
|
||||
func windowRows(start, end uint64, services []string, valueAt func(ts uint64, service string) float64) *cmock.Rows {
|
||||
var values [][]any
|
||||
for ts := start; ts < end; ts += minuteStepMs {
|
||||
for _, service := range services {
|
||||
values = append(values, []any{time.UnixMilli(int64(ts)), service, valueAt(ts, service)})
|
||||
}
|
||||
}
|
||||
return cmock.NewRows(windowColumns, values)
|
||||
}
|
||||
|
||||
func expectWindowQuery(store *telemetrystoretest.Provider, start, end uint64, limit int, rows *cmock.Rows) {
|
||||
store.Mock().ExpectQuery(regexp.QuoteMeta(windowSQL(start, end, limit))).WillReturnRows(rows)
|
||||
}
|
||||
|
||||
func logSpec(limit int) qbtypes.QueryBuilderQuery[qbtypes.LogAggregation] {
|
||||
spec := qbtypes.QueryBuilderQuery[qbtypes.LogAggregation]{
|
||||
Name: "A",
|
||||
Signal: telemetrytypes.SignalLogs,
|
||||
StepInterval: minuteStep(),
|
||||
Aggregations: []qbtypes.LogAggregation{{Expression: "count()"}},
|
||||
GroupBy: []qbtypes.GroupByKey{{TelemetryFieldKey: telemetrytypes.TelemetryFieldKey{
|
||||
Name: "service.name", FieldDataType: telemetrytypes.FieldDataTypeString, FieldContext: telemetrytypes.FieldContextResource,
|
||||
}}},
|
||||
}
|
||||
if limit > 0 {
|
||||
spec.Limit = limit
|
||||
spec.Order = []qbtypes.OrderBy{{
|
||||
Key: qbtypes.OrderByKey{TelemetryFieldKey: telemetrytypes.TelemetryFieldKey{Name: "count()"}},
|
||||
Direction: qbtypes.OrderDirectionDesc,
|
||||
}}
|
||||
}
|
||||
return spec
|
||||
}
|
||||
|
||||
func newCachingQuerier(t *testing.T, builder *windowLogStmtBuilder, store *telemetrystoretest.Provider) *querier {
|
||||
t.Helper()
|
||||
return &querier{
|
||||
logger: instrumentationtest.New().Logger(),
|
||||
fl: flaggertest.New(t),
|
||||
telemetryStore: store,
|
||||
logStmtBuilder: builder,
|
||||
bucketCache: createTestBucketCache(t),
|
||||
maxConcurrentQueries: DefaultMaxConcurrentQueries,
|
||||
}
|
||||
}
|
||||
|
||||
func logRequest(startMs, endMs uint64, spec qbtypes.QueryBuilderQuery[qbtypes.LogAggregation]) *qbtypes.QueryRangeRequest {
|
||||
return &qbtypes.QueryRangeRequest{
|
||||
Start: startMs,
|
||||
End: endMs,
|
||||
RequestType: qbtypes.RequestTypeTimeSeries,
|
||||
CompositeQuery: qbtypes.CompositeQuery{
|
||||
Queries: []qbtypes.QueryEnvelope{{Type: qbtypes.QueryTypeBuilder, Spec: spec}},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
func runLogRequest(t *testing.T, q *querier, orgID valuer.UUID, req *qbtypes.QueryRangeRequest, spec qbtypes.QueryBuilderQuery[qbtypes.LogAggregation]) *qbtypes.TimeSeriesData {
|
||||
t.Helper()
|
||||
bq := newBuilderQuery(q.logger, q.telemetryStore, orgID, q.logStmtBuilder, qbtypes.QueryTypeBuilder, spec, qbtypes.TimeRange{From: req.Start, To: req.End}, req.RequestType, nil, builderConfig{})
|
||||
resp, err := q.run(context.Background(), orgID, map[string]qbtypes.Query{spec.Name: bq}, req, map[string]qbtypes.Step{spec.Name: spec.StepInterval}, &qbtypes.QBEvent{}, nil)
|
||||
require.NoError(t, err)
|
||||
require.Len(t, resp.Data.Results, 1)
|
||||
tsData, ok := resp.Data.Results[0].(*qbtypes.TimeSeriesData)
|
||||
require.True(t, ok, "result is %T", resp.Data.Results[0])
|
||||
return tsData
|
||||
}
|
||||
|
||||
func pointsByService(data *qbtypes.TimeSeriesData) map[string][]float64 {
|
||||
out := map[string][]float64{}
|
||||
for _, agg := range data.Aggregations {
|
||||
for _, s := range agg.Series {
|
||||
name := fmt.Sprint(s.Labels[0].Value)
|
||||
for _, v := range s.Values {
|
||||
out[name] = append(out[name], v.Value)
|
||||
}
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// A timeShift query already reports its window in the shifted clock, so a
|
||||
// gap of that window is fetched as it is.
|
||||
func TestCreateRangedQuery_TimeShiftIsAppliedOnce(t *testing.T) {
|
||||
q := &querier{logger: instrumentationtest.New().Logger(), logStmtBuilder: &windowLogStmtBuilder{}}
|
||||
spec := logSpec(0)
|
||||
spec.Functions = []qbtypes.Function{{Name: qbtypes.FunctionNameTimeShift, Args: []qbtypes.FunctionArg{{Value: 3600.0}}}}
|
||||
spec.ShiftBy = extractShiftFromBuilderQuery(spec)
|
||||
|
||||
requested := qbtypes.TimeRange{From: epochMs, To: epochMs + 60*minuteStepMs}
|
||||
shifted := adjustTimeRangeForShift(spec, requested, qbtypes.RequestTypeTimeSeries)
|
||||
require.Equal(t, requested.From-3_600_000, shifted.From)
|
||||
bq := newBuilderQuery(q.logger, nil, valuer.GenerateUUID(), q.logStmtBuilder, qbtypes.QueryTypeBuilder, spec, shifted, qbtypes.RequestTypeTimeSeries, nil, builderConfig{})
|
||||
|
||||
missing := qbtypes.TimeRange{From: shifted.From + 30*minuteStepMs, To: shifted.To}
|
||||
ranged := q.createRangedQuery(bq, missing)
|
||||
require.NotNil(t, ranged)
|
||||
|
||||
from, to := ranged.Window()
|
||||
assert.Equal(t, missing, qbtypes.TimeRange{From: from, To: to})
|
||||
}
|
||||
|
||||
func TestMergeResults_FreshPointReplacesCachedPoint(t *testing.T) {
|
||||
ts := int64(epochMs)
|
||||
point := func(value float64, partial bool) *qbtypes.Result {
|
||||
return seriesResult(1, &qbtypes.TimeSeries{
|
||||
Labels: []*qbtypes.Label{{Key: telemetrytypes.TelemetryFieldKey{Name: "service"}, Value: "a"}},
|
||||
Values: []*qbtypes.TimeSeriesValue{{Timestamp: ts, Value: value, Partial: partial}},
|
||||
})
|
||||
}
|
||||
req := CacheRequest{Window: qbtypes.TimeRange{From: epochMs, To: epochMs + minuteStepMs}, Step: minuteStep(), Kind: qbtypes.RequestTypeTimeSeries}
|
||||
|
||||
merged := mergeResults(req, point(3, false), []*qbtypes.Result{point(7, false)})
|
||||
assert.Equal(t, map[string][]float64{"a": {7}}, pointsByService(merged.Value.(*qbtypes.TimeSeriesData)))
|
||||
assert.Equal(t, uint64(2), merged.Stats.RowsScanned)
|
||||
|
||||
merged = mergeResults(req, point(3, false), []*qbtypes.Result{point(7, true)})
|
||||
assert.Equal(t, map[string][]float64{"a": {3}}, pointsByService(merged.Value.(*qbtypes.TimeSeriesData)), "a partial point never replaces a whole one")
|
||||
}
|
||||
|
||||
func TestMergeResults_DropsPointsBeforeTheWindow(t *testing.T) {
|
||||
req := CacheRequest{Window: qbtypes.TimeRange{From: epochMs + 30_000, To: epochMs + 3*minuteStepMs}, Step: minuteStep(), Kind: qbtypes.RequestTypeTimeSeries}
|
||||
fresh := seriesResult(1, minuteSeries("a", epochMs-2*minuteStepMs, epochMs+3*minuteStepMs, 1))
|
||||
|
||||
merged := mergeResults(req, nil, []*qbtypes.Result{fresh})
|
||||
|
||||
assert.Equal(t, map[string][]float64{"a": {1, 1, 1}}, pointsByService(merged.Value.(*qbtypes.TimeSeriesData)), "the partial first step is kept, the widened lookback is not")
|
||||
}
|
||||
|
||||
// A grouped query with a limit is a top-N over the requested window; its
|
||||
// answer is cached for that window only and never assembled from pieces.
|
||||
func TestRun_LimitedGroupByIsNeverAssembledFromPieces(t *testing.T) {
|
||||
head := qbtypes.TimeRange{From: epochMs, To: epochMs + 10*minuteStepMs}
|
||||
tail := qbtypes.TimeRange{From: head.To, To: head.To + 10*minuteStepMs}
|
||||
full := qbtypes.TimeRange{From: head.From, To: tail.To}
|
||||
|
||||
// Top-2 of the head is {a, b}, of the tail {c, d}, of the full window {a, c}.
|
||||
headCounts := map[string]float64{"a": 10, "b": 8, "c": 1, "d": 1}
|
||||
tailCounts := map[string]float64{"a": 1, "b": 1, "c": 10, "d": 8}
|
||||
valueAt := func(ts uint64, service string) float64 {
|
||||
if ts < tail.From {
|
||||
return headCounts[service]
|
||||
}
|
||||
return tailCounts[service]
|
||||
}
|
||||
|
||||
store := telemetrystoretest.New(telemetrystore.Config{}, sqlmock.QueryMatcherRegexp)
|
||||
store.Mock().MatchExpectationsInOrder(false)
|
||||
expectWindowQuery(store, head.From, head.To, 2, windowRows(head.From, head.To, []string{"a", "b"}, valueAt))
|
||||
expectWindowQuery(store, tail.From, tail.To, 2, windowRows(tail.From, tail.To, []string{"c", "d"}, valueAt))
|
||||
expectWindowQuery(store, full.From, full.To, 2, windowRows(full.From, full.To, []string{"a", "c"}, valueAt))
|
||||
|
||||
builder := &windowLogStmtBuilder{}
|
||||
q := newCachingQuerier(t, builder, store)
|
||||
orgID := valuer.GenerateUUID()
|
||||
spec := logSpec(2)
|
||||
|
||||
runLogRequest(t, q, orgID, logRequest(head.From, head.To, spec), spec)
|
||||
got := runLogRequest(t, q, orgID, logRequest(full.From, full.To, spec), spec)
|
||||
assert.Equal(t, map[string][]float64{"a": append(repeat(10, 10), repeat(1, 10)...), "c": append(repeat(1, 10), repeat(10, 10)...)}, pointsByService(got))
|
||||
assert.Equal(t, []qbtypes.TimeRange{head, full}, builder.built())
|
||||
|
||||
got = runLogRequest(t, q, orgID, logRequest(full.From, full.To, spec), spec)
|
||||
assert.Len(t, builder.built(), 2, "the repeated window is a cache hit")
|
||||
assert.Len(t, got.Aggregations[0].Series, 2)
|
||||
}
|
||||
|
||||
func repeat(value float64, n int) []float64 {
|
||||
out := make([]float64, n)
|
||||
for i := range out {
|
||||
out[i] = value
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// A request whose window the cache does not cover at all runs as one
|
||||
// statement, whatever the grid alignment of its ends.
|
||||
func TestExecuteWithCache_EntirelyMissingWindowRunsOneStatement(t *testing.T) {
|
||||
store := telemetrystoretest.New(telemetrystore.Config{}, sqlmock.QueryMatcherRegexp)
|
||||
store.Mock().MatchExpectationsInOrder(false)
|
||||
builder := &windowLogStmtBuilder{}
|
||||
q := newCachingQuerier(t, builder, store)
|
||||
orgID := valuer.GenerateUUID()
|
||||
ctx := context.Background()
|
||||
spec := logSpec(0)
|
||||
one := func(uint64, string) float64 { return 1 }
|
||||
|
||||
older := qbtypes.TimeRange{From: epochMs - 120*minuteStepMs, To: epochMs - 60*minuteStepMs}
|
||||
olderQuery := newBuilderQuery(q.logger, store, orgID, builder, qbtypes.QueryTypeBuilder, spec, older, qbtypes.RequestTypeTimeSeries, nil, builderConfig{})
|
||||
req := CacheRequest{Key: CacheKey(olderQuery.Fingerprint()), Window: older, Step: spec.StepInterval, Kind: qbtypes.RequestTypeTimeSeries}
|
||||
q.bucketCache.Put(ctx, orgID, req, older, seriesResult(1, minuteSeries("a", older.From, older.To, 1)))
|
||||
|
||||
window := qbtypes.TimeRange{From: epochMs + 7_000, To: epochMs + 60*minuteStepMs}
|
||||
expectWindowQuery(store, window.From, window.To, 0, windowRows(window.From, window.To, []string{"a"}, one))
|
||||
|
||||
bq := newBuilderQuery(q.logger, store, orgID, builder, qbtypes.QueryTypeBuilder, spec, window, qbtypes.RequestTypeTimeSeries, nil, builderConfig{})
|
||||
_, err := q.executeWithCache(ctx, orgID, bq, spec.StepInterval, make(chan struct{}, q.maxConcurrentQueries))
|
||||
require.NoError(t, err)
|
||||
|
||||
assert.Equal(t, []qbtypes.TimeRange{window}, builder.built())
|
||||
}
|
||||
|
||||
func TestExecuteWithCache_GapsRunAsSeparateStatementsAndAreWrittenBack(t *testing.T) {
|
||||
store := telemetrystoretest.New(telemetrystore.Config{}, sqlmock.QueryMatcherRegexp)
|
||||
store.Mock().MatchExpectationsInOrder(false)
|
||||
builder := &windowLogStmtBuilder{}
|
||||
q := newCachingQuerier(t, builder, store)
|
||||
orgID := valuer.GenerateUUID()
|
||||
spec := logSpec(0)
|
||||
one := func(uint64, string) float64 { return 1 }
|
||||
|
||||
middle := qbtypes.TimeRange{From: epochMs + 20*minuteStepMs, To: epochMs + 40*minuteStepMs}
|
||||
whole := qbtypes.TimeRange{From: epochMs, To: epochMs + 60*minuteStepMs}
|
||||
before := qbtypes.TimeRange{From: whole.From, To: middle.From}
|
||||
after := qbtypes.TimeRange{From: middle.To, To: whole.To}
|
||||
for _, window := range []qbtypes.TimeRange{middle, before, after} {
|
||||
expectWindowQuery(store, window.From, window.To, 0, windowRows(window.From, window.To, []string{"a"}, one))
|
||||
}
|
||||
|
||||
runLogRequest(t, q, orgID, logRequest(middle.From, middle.To, spec), spec)
|
||||
got := runLogRequest(t, q, orgID, logRequest(whole.From, whole.To, spec), spec)
|
||||
assert.Equal(t, map[string][]float64{"a": repeat(1, 60)}, pointsByService(got))
|
||||
assert.ElementsMatch(t, []qbtypes.TimeRange{middle, before, after}, builder.built())
|
||||
|
||||
runLogRequest(t, q, orgID, logRequest(whole.From, whole.To, spec), spec)
|
||||
assert.Len(t, builder.built(), 3, "the assembled window is a cache hit afterwards")
|
||||
}
|
||||
|
||||
func TestExecuteWithCache_FailedGapFailsTheRequest(t *testing.T) {
|
||||
store := telemetrystoretest.New(telemetrystore.Config{}, sqlmock.QueryMatcherRegexp)
|
||||
store.Mock().MatchExpectationsInOrder(false)
|
||||
builder := &windowLogStmtBuilder{}
|
||||
q := newCachingQuerier(t, builder, store)
|
||||
orgID := valuer.GenerateUUID()
|
||||
spec := logSpec(0)
|
||||
one := func(uint64, string) float64 { return 1 }
|
||||
|
||||
cached := qbtypes.TimeRange{From: epochMs, To: epochMs + 20*minuteStepMs}
|
||||
whole := qbtypes.TimeRange{From: epochMs, To: epochMs + 40*minuteStepMs}
|
||||
gap := qbtypes.TimeRange{From: cached.To, To: whole.To}
|
||||
expectWindowQuery(store, cached.From, cached.To, 0, windowRows(cached.From, cached.To, []string{"a"}, one))
|
||||
store.Mock().ExpectQuery(regexp.QuoteMeta(windowSQL(gap.From, gap.To, 0))).WillReturnError(fmt.Errorf("clickhouse is away"))
|
||||
|
||||
runLogRequest(t, q, orgID, logRequest(cached.From, cached.To, spec), spec)
|
||||
bq := newBuilderQuery(q.logger, store, orgID, builder, qbtypes.QueryTypeBuilder, spec, whole, qbtypes.RequestTypeTimeSeries, nil, builderConfig{})
|
||||
_, err := q.executeWithCache(context.Background(), orgID, bq, spec.StepInterval, make(chan struct{}, q.maxConcurrentQueries))
|
||||
|
||||
require.Error(t, err)
|
||||
assert.Equal(t, []qbtypes.TimeRange{cached, gap}, builder.built(), "the whole window is not run again after a failed gap")
|
||||
}
|
||||
|
||||
func TestRun_FormulaByAliasSurvivesFullCacheHit(t *testing.T) {
|
||||
window := qbtypes.TimeRange{From: epochMs, To: epochMs + 10*minuteStepMs}
|
||||
store := telemetrystoretest.New(telemetrystore.Config{}, sqlmock.QueryMatcherRegexp)
|
||||
three := func(uint64, string) float64 { return 3 }
|
||||
expectWindowQuery(store, window.From, window.To, 0, windowRows(window.From, window.To, []string{"a"}, three))
|
||||
|
||||
q := newCachingQuerier(t, &windowLogStmtBuilder{}, store)
|
||||
orgID := valuer.GenerateUUID()
|
||||
spec := logSpec(0)
|
||||
req := logRequest(window.From, window.To, spec)
|
||||
req.CompositeQuery.Queries = append(req.CompositeQuery.Queries, qbtypes.QueryEnvelope{
|
||||
Type: qbtypes.QueryTypeFormula,
|
||||
Spec: qbtypes.QueryBuilderFormula{Name: "F", Expression: "[A.__result_0] * 2"},
|
||||
})
|
||||
|
||||
formulaValues := func() []float64 {
|
||||
bq := newBuilderQuery(q.logger, q.telemetryStore, orgID, q.logStmtBuilder, qbtypes.QueryTypeBuilder, spec, window, req.RequestType, nil, builderConfig{})
|
||||
resp, err := q.run(context.Background(), orgID, map[string]qbtypes.Query{"A": bq}, req, map[string]qbtypes.Step{"A": spec.StepInterval}, &qbtypes.QBEvent{}, nil)
|
||||
require.NoError(t, err)
|
||||
for _, result := range resp.Data.Results {
|
||||
tsData, ok := result.(*qbtypes.TimeSeriesData)
|
||||
if !ok || tsData.QueryName != "F" {
|
||||
continue
|
||||
}
|
||||
var values []float64
|
||||
for _, agg := range tsData.Aggregations {
|
||||
for _, s := range agg.Series {
|
||||
for _, v := range s.Values {
|
||||
values = append(values, v.Value)
|
||||
}
|
||||
}
|
||||
}
|
||||
return values
|
||||
}
|
||||
t.Fatal("no result for formula F")
|
||||
return nil
|
||||
}
|
||||
|
||||
first := formulaValues()
|
||||
require.Equal(t, repeat(6, 10), first)
|
||||
assert.Equal(t, first, formulaValues())
|
||||
}
|
||||
|
||||
// lookbackMetricStmtBuilder stands in for the metrics statement builder. It
|
||||
// widens the window the way the real builder does, so a runningDiff query
|
||||
// fetches one step before the request.
|
||||
type lookbackMetricStmtBuilder struct{}
|
||||
|
||||
func (b *lookbackMetricStmtBuilder) Build(_ context.Context, _ valuer.UUID, start, end uint64, _ qbtypes.RequestType, query qbtypes.QueryBuilderQuery[qbtypes.MetricAggregation], _ map[string]qbtypes.VariableItem) (*qbtypes.Statement, error) {
|
||||
start, end = querybuilder.AdjustedMetricTimeRange(start, end, uint64(query.StepInterval.Seconds()), query)
|
||||
return &qbtypes.Statement{Query: windowSQL(start, end, 0)}, nil
|
||||
}
|
||||
|
||||
func TestRun_RunningDiffKeepsFirstIntervalOnEveryCachePath(t *testing.T) {
|
||||
window := qbtypes.TimeRange{From: epochMs, To: epochMs + 3*minuteStepMs}
|
||||
lookback := window.From - minuteStepMs
|
||||
gauge := func(ts uint64, _ string) float64 { return 100 + float64((ts-lookback)/minuteStepMs)*10 }
|
||||
|
||||
store := telemetrystoretest.New(telemetrystore.Config{}, sqlmock.QueryMatcherRegexp)
|
||||
store.Mock().MatchExpectationsInOrder(false)
|
||||
expectWindowQuery(store, lookback, window.To, 0, windowRows(lookback, window.To, []string{"a"}, gauge))
|
||||
// The window then grows by one step at the end; only that step is fetched, with its own lookback.
|
||||
grown := qbtypes.TimeRange{From: window.From, To: window.To + minuteStepMs}
|
||||
expectWindowQuery(store, window.To-minuteStepMs, grown.To, 0, windowRows(window.To-minuteStepMs, grown.To, []string{"a"}, gauge))
|
||||
|
||||
q := newCachingQuerier(t, &windowLogStmtBuilder{}, store)
|
||||
q.metricStmtBuilder = &lookbackMetricStmtBuilder{}
|
||||
orgID := valuer.GenerateUUID()
|
||||
spec := qbtypes.QueryBuilderQuery[qbtypes.MetricAggregation]{
|
||||
Name: "A",
|
||||
Signal: telemetrytypes.SignalMetrics,
|
||||
StepInterval: minuteStep(),
|
||||
Aggregations: []qbtypes.MetricAggregation{{MetricName: "gauge", TimeAggregation: metrictypes.TimeAggregationAvg, SpaceAggregation: metrictypes.SpaceAggregationAvg}},
|
||||
Functions: []qbtypes.Function{{Name: qbtypes.FunctionNameRunningDiff}},
|
||||
}
|
||||
diffs := func(window qbtypes.TimeRange) []float64 {
|
||||
req := &qbtypes.QueryRangeRequest{
|
||||
Start: window.From, End: window.To, RequestType: qbtypes.RequestTypeTimeSeries,
|
||||
CompositeQuery: qbtypes.CompositeQuery{Queries: []qbtypes.QueryEnvelope{{Type: qbtypes.QueryTypeBuilder, Spec: spec}}},
|
||||
}
|
||||
bq := newBuilderQuery(q.logger, q.telemetryStore, orgID, q.metricStmtBuilder, qbtypes.QueryTypeBuilder, spec, window, req.RequestType, nil, builderConfig{})
|
||||
resp, err := q.run(context.Background(), orgID, map[string]qbtypes.Query{"A": bq}, req, map[string]qbtypes.Step{"A": spec.StepInterval}, &qbtypes.QBEvent{}, nil)
|
||||
require.NoError(t, err)
|
||||
require.Len(t, resp.Data.Results, 1)
|
||||
return pointsByService(resp.Data.Results[0].(*qbtypes.TimeSeriesData))["a"]
|
||||
}
|
||||
|
||||
require.Equal(t, []float64{10, 10, 10}, diffs(window), "uncached")
|
||||
assert.Equal(t, []float64{10, 10, 10}, diffs(window), "full cache hit")
|
||||
assert.Equal(t, []float64{10, 10, 10, 10}, diffs(grown), "partial cache hit")
|
||||
}
|
||||
|
||||
func TestCreateRangedQuery_PromQLReservedVariablesKeepRequestWindow(t *testing.T) {
|
||||
store := telemetrystoretest.New(telemetrystore.Config{}, sqlmock.QueryMatcherRegexp)
|
||||
engine := prometheustest.New(context.Background(), instrumentationtest.New().ToProviderSettings(), prometheus.Config{Timeout: time.Minute}, store)
|
||||
q := &querier{logger: instrumentationtest.New().Logger()}
|
||||
|
||||
window := qbtypes.TimeRange{From: epochMs, To: epochMs + 4*minuteStepMs}
|
||||
original := newPromqlQuery(q.logger, engine, qbtypes.PromQuery{Name: "A", Query: "vector($start_timestamp)", Step: minuteStep()}, window, qbtypes.RequestTypeTimeSeries, nil)
|
||||
|
||||
gap := qbtypes.TimeRange{From: window.From + 2*minuteStepMs, To: window.To}
|
||||
ranged := q.createRangedQuery(original, gap)
|
||||
require.NotNil(t, ranged)
|
||||
|
||||
stmt, err := ranged.(*promqlQuery).Statement(context.Background())
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, fmt.Sprintf("vector(%d)", window.From/1000), stmt.Query)
|
||||
assert.Equal(t, original.Fingerprint(), ranged.Fingerprint())
|
||||
from, to := ranged.Window()
|
||||
assert.Equal(t, gap, qbtypes.TimeRange{From: from, To: to})
|
||||
}
|
||||
|
||||
func TestBuilderQueryFingerprint_IsStableWithSeveralVariables(t *testing.T) {
|
||||
spec := logSpec(0)
|
||||
spec.Filter = &qbtypes.Filter{Expression: "service.name = $svc AND deployment.environment = $env AND cloud.region = $region"}
|
||||
variables := map[string]qbtypes.VariableItem{
|
||||
"svc": {Value: "checkout"},
|
||||
"env": {Value: "prod"},
|
||||
"region": {Value: "eu-west-1"},
|
||||
}
|
||||
bq := newBuilderQuery(instrumentationtest.New().Logger(), nil, valuer.GenerateUUID(), &windowLogStmtBuilder{}, qbtypes.QueryTypeBuilder, spec, qbtypes.TimeRange{From: epochMs, To: epochMs + minuteStepMs}, qbtypes.RequestTypeTimeSeries, variables, builderConfig{})
|
||||
|
||||
seen := map[string]struct{}{}
|
||||
for range 50 {
|
||||
seen[bq.Fingerprint()] = struct{}{}
|
||||
}
|
||||
assert.Len(t, seen, 1)
|
||||
}
|
||||
|
||||
func TestBuilderQueryFingerprint_DistinguishesVariableValues(t *testing.T) {
|
||||
spec := logSpec(0)
|
||||
spec.Filter = &qbtypes.Filter{Expression: "service.name IN $svc"}
|
||||
key := func(value any) string {
|
||||
bq := newBuilderQuery(instrumentationtest.New().Logger(), nil, valuer.GenerateUUID(), &windowLogStmtBuilder{}, qbtypes.QueryTypeBuilder, spec, qbtypes.TimeRange{From: epochMs, To: epochMs + minuteStepMs}, qbtypes.RequestTypeTimeSeries, map[string]qbtypes.VariableItem{"svc": {Value: value}}, builderConfig{})
|
||||
return bq.Fingerprint()
|
||||
}
|
||||
assert.NotEqual(t, key([]any{"a b"}), key([]any{"a", "b"}))
|
||||
assert.NotEqual(t, key(1.0), key("1"))
|
||||
}
|
||||
|
||||
func TestBuilderQueryFingerprint_LimitedGroupByIncludesTheWindow(t *testing.T) {
|
||||
key := func(limit int, window qbtypes.TimeRange) string {
|
||||
return newBuilderQuery(instrumentationtest.New().Logger(), nil, valuer.GenerateUUID(), &windowLogStmtBuilder{}, qbtypes.QueryTypeBuilder, logSpec(limit), window, qbtypes.RequestTypeTimeSeries, nil, builderConfig{}).Fingerprint()
|
||||
}
|
||||
first := qbtypes.TimeRange{From: epochMs, To: epochMs + 10*minuteStepMs}
|
||||
second := qbtypes.TimeRange{From: epochMs, To: epochMs + 20*minuteStepMs}
|
||||
assert.NotEqual(t, key(2, first), key(2, second))
|
||||
assert.Equal(t, key(0, first), key(0, second))
|
||||
}
|
||||
|
||||
func TestCreateRangedQuery_MetricsPieceReadsTheTablesOfTheRequest(t *testing.T) {
|
||||
q := &querier{logger: instrumentationtest.New().Logger(), metricStmtBuilder: &lookbackMetricStmtBuilder{}}
|
||||
spec := qbtypes.QueryBuilderQuery[qbtypes.MetricAggregation]{
|
||||
Name: "A",
|
||||
Signal: telemetrytypes.SignalMetrics,
|
||||
StepInterval: qbtypes.Step{Duration: 30 * time.Minute},
|
||||
Aggregations: []qbtypes.MetricAggregation{{MetricName: "gauge", Type: metrictypes.GaugeType, TimeAggregation: metrictypes.TimeAggregationAvg, SpaceAggregation: metrictypes.SpaceAggregationAvg}},
|
||||
}
|
||||
request := qbtypes.TimeRange{From: epochMs, To: epochMs + 3*24*60*minuteStepMs}
|
||||
bq := newBuilderQuery(q.logger, nil, valuer.GenerateUUID(), q.metricStmtBuilder, qbtypes.QueryTypeBuilder, spec, request, qbtypes.RequestTypeTimeSeries, nil, builderConfig{})
|
||||
|
||||
gap := qbtypes.TimeRange{From: request.To - 60*minuteStepMs, To: request.To}
|
||||
ranged := q.createRangedQuery(bq, gap)
|
||||
require.NotNil(t, ranged)
|
||||
|
||||
want := metricstelemetryschema.TableHintsForWindow(request.From, request.To, spec.Aggregations[0].Type, spec.Aggregations[0].TimeAggregation, false, nil)
|
||||
require.NotNil(t, want)
|
||||
got := ranged.(*builderQuery[qbtypes.MetricAggregation]).spec.Aggregations[0].TableHints
|
||||
assert.Equal(t, want, got)
|
||||
assert.Nil(t, bq.spec.Aggregations[0].TableHints, "the original query is left as it is")
|
||||
}
|
||||
|
||||
func TestCreateRangedQuery_MeterPieceReadsTheTableOfTheRequest(t *testing.T) {
|
||||
q := &querier{logger: instrumentationtest.New().Logger(), meterStmtBuilder: &lookbackMetricStmtBuilder{}}
|
||||
spec := qbtypes.QueryBuilderQuery[qbtypes.MetricAggregation]{
|
||||
Name: "A",
|
||||
Signal: telemetrytypes.SignalMetrics,
|
||||
Source: telemetrytypes.SourceMeter,
|
||||
StepInterval: qbtypes.Step{Duration: 24 * time.Hour},
|
||||
Aggregations: []qbtypes.MetricAggregation{{MetricName: "meter", Type: metrictypes.SumType, TimeAggregation: metrictypes.TimeAggregationSum, SpaceAggregation: metrictypes.SpaceAggregationSum}},
|
||||
}
|
||||
request := qbtypes.TimeRange{From: epochMs, To: epochMs + 60*24*60*minuteStepMs}
|
||||
bq := newBuilderQuery(q.logger, nil, valuer.GenerateUUID(), q.meterStmtBuilder, qbtypes.QueryTypeBuilder, spec, request, qbtypes.RequestTypeTimeSeries, nil, builderConfig{})
|
||||
|
||||
gap := qbtypes.TimeRange{From: request.To - 24*60*minuteStepMs, To: request.To}
|
||||
ranged := q.createRangedQuery(bq, gap)
|
||||
require.NotNil(t, ranged)
|
||||
|
||||
got := ranged.(*builderQuery[qbtypes.MetricAggregation]).spec.Aggregations[0].TableHints
|
||||
require.NotNil(t, got)
|
||||
assert.Equal(t, metertelemetryschema.SamplesAgg1dTableName, got.SamplesTableName)
|
||||
assert.Equal(t, metertelemetryschema.SamplesTableName, metertelemetryschema.WhichSamplesTableToUse(gap.From, gap.To, spec.Aggregations[0].Type, spec.Aggregations[0].TimeAggregation, nil), "the piece alone would read the raw table")
|
||||
}
|
||||
|
||||
func TestTrimsHeatmapAxis_OnlyForAnAxisComputedFromTheData(t *testing.T) {
|
||||
metricHeatmap := func(bucketing *qbtypes.HeatmapBucketing) qbtypes.Query {
|
||||
spec := qbtypes.QueryBuilderQuery[qbtypes.MetricAggregation]{
|
||||
Name: "A",
|
||||
Signal: telemetrytypes.SignalMetrics,
|
||||
StepInterval: minuteStep(),
|
||||
Aggregations: []qbtypes.MetricAggregation{{MetricName: "m", HeatmapBucketing: bucketing}},
|
||||
}
|
||||
return newBuilderQuery(instrumentationtest.New().Logger(), nil, valuer.GenerateUUID(), &lookbackMetricStmtBuilder{}, qbtypes.QueryTypeBuilder, spec, qbtypes.TimeRange{From: epochMs, To: epochMs + minuteStepMs}, qbtypes.RequestTypeHeatmap, nil, builderConfig{})
|
||||
}
|
||||
assert.True(t, trimsHeatmapAxis(metricHeatmap(&qbtypes.HeatmapBucketing{Kind: qbtypes.BucketsKindLog})), "a gauge heatmap buckets the served values")
|
||||
assert.False(t, trimsHeatmapAxis(metricHeatmap(nil)), "a histogram reports its own le buckets")
|
||||
assert.False(t, trimsHeatmapAxis(newPromqlQuery(instrumentationtest.New().Logger(), nil, qbtypes.PromQuery{Query: "up", Step: minuteStep()}, qbtypes.TimeRange{From: epochMs, To: epochMs + minuteStepMs}, qbtypes.RequestTypeHeatmap, nil)))
|
||||
}
|
||||
|
||||
func TestBuildQueries_PromQLWindowIsMovedToTheStepGridUnlessAskedNotTo(t *testing.T) {
|
||||
q := &querier{logger: instrumentationtest.New().Logger()}
|
||||
req := &qbtypes.QueryRangeRequest{
|
||||
Start: epochMs + 17_000, End: epochMs + 10*minuteStepMs + 45_000, RequestType: qbtypes.RequestTypeTimeSeries,
|
||||
CompositeQuery: qbtypes.CompositeQuery{Queries: []qbtypes.QueryEnvelope{{Type: qbtypes.QueryTypePromQL, Spec: qbtypes.PromQuery{Name: "A", Query: "up", Step: minuteStep()}}}},
|
||||
}
|
||||
|
||||
queries, _, err := q.buildQueries(valuer.GenerateUUID(), req, nil, nil, &qbtypes.QBEvent{})
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, qbtypes.TimeRange{From: epochMs, To: epochMs + 10*minuteStepMs}, queries["A"].(*promqlQuery).tr)
|
||||
assert.NotEmpty(t, queries["A"].Fingerprint())
|
||||
|
||||
req.NoStepAlignment = true
|
||||
queries, _, err = q.buildQueries(valuer.GenerateUUID(), req, nil, nil, &qbtypes.QBEvent{})
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, qbtypes.TimeRange{From: req.Start, To: req.End}, queries["A"].(*promqlQuery).tr)
|
||||
assert.Empty(t, queries["A"].Fingerprint(), "a window kept off the grid is not cached")
|
||||
}
|
||||
|
||||
func TestPromQLFingerprint_StartOrEndModifierIsNotCached(t *testing.T) {
|
||||
fingerprint := func(expr string) string {
|
||||
return newPromqlQuery(instrumentationtest.New().Logger(), nil, qbtypes.PromQuery{Query: expr, Step: minuteStep()}, qbtypes.TimeRange{From: epochMs, To: epochMs + 10*minuteStepMs}, qbtypes.RequestTypeTimeSeries, nil).Fingerprint()
|
||||
}
|
||||
assert.NotEmpty(t, fingerprint("sum(rate(up[5m]))"))
|
||||
assert.NotEmpty(t, fingerprint("up @ 1672531200"), "a fixed instant means the same in every piece")
|
||||
assert.Empty(t, fingerprint("up @ end()"))
|
||||
assert.Empty(t, fingerprint("sum(up @ start())"))
|
||||
assert.Empty(t, fingerprint("max_over_time(up[5m:1m] @ end())"))
|
||||
}
|
||||
@@ -78,7 +78,8 @@ func (r *ThresholdRule) prepareQueryRange(ctx context.Context, ts time.Time) (*q
|
||||
CompositeQuery: qbtypes.CompositeQuery{
|
||||
Queries: make([]qbtypes.QueryEnvelope, 0),
|
||||
},
|
||||
NoCache: true,
|
||||
NoCache: true,
|
||||
NoStepAlignment: true,
|
||||
}
|
||||
req.CompositeQuery.Queries = make([]qbtypes.QueryEnvelope, len(r.Condition().CompositeQuery.Queries))
|
||||
copy(req.CompositeQuery.Queries, r.Condition().CompositeQuery.Queries)
|
||||
|
||||
@@ -176,6 +176,31 @@ func MinAllowedStepIntervalForMetric(start, end uint64) uint64 {
|
||||
return minAllowed
|
||||
}
|
||||
|
||||
// RateLookbackMs is how far before a bucket the previous sample of a
|
||||
// cumulative series may lie for rate and increase to use it. The bound makes
|
||||
// the value of a bucket depend only on the samples within the lookback, not
|
||||
// on where the statement window starts, so a window served in pieces agrees
|
||||
// with one statement over the whole window. Five minutes is the Prometheus
|
||||
// lookback; a step longer than that keeps one step.
|
||||
func RateLookbackMs(stepMs uint64) uint64 {
|
||||
return max(stepMs, uint64((5 * time.Minute).Milliseconds()))
|
||||
}
|
||||
|
||||
// MetricRateLookbackMs is the lookback the statement of mq reads before its
|
||||
// window, or zero when mq computes no rate. A histogram percentile or count
|
||||
// computes a rate or increase over its buckets whatever time aggregation the
|
||||
// query names.
|
||||
func MetricRateLookbackMs(stepMs uint64, mq qbtypes.QueryBuilderQuery[qbtypes.MetricAggregation]) uint64 {
|
||||
agg := mq.Aggregations[0]
|
||||
usesRate := agg.TimeAggregation == metrictypes.TimeAggregationRate ||
|
||||
agg.TimeAggregation == metrictypes.TimeAggregationIncrease ||
|
||||
agg.Type == metrictypes.HistogramType
|
||||
if !usesRate || agg.Temporality == metrictypes.Delta {
|
||||
return 0
|
||||
}
|
||||
return RateLookbackMs(stepMs)
|
||||
}
|
||||
|
||||
func AdjustedMetricTimeRange(start, end, step uint64, mq qbtypes.QueryBuilderQuery[qbtypes.MetricAggregation]) (uint64, uint64) {
|
||||
// align the start to the step interval
|
||||
start = start - (start % (step * 1000))
|
||||
@@ -188,10 +213,7 @@ func AdjustedMetricTimeRange(start, end, step uint64, mq qbtypes.QueryBuilderQue
|
||||
break
|
||||
}
|
||||
}
|
||||
if (mq.Aggregations[0].TimeAggregation == metrictypes.TimeAggregationRate || mq.Aggregations[0].TimeAggregation == metrictypes.TimeAggregationIncrease) &&
|
||||
mq.Aggregations[0].Temporality != metrictypes.Delta {
|
||||
start -= step * 1000
|
||||
}
|
||||
start -= MetricRateLookbackMs(step*1000, mq)
|
||||
if hasRunningDiff {
|
||||
start -= step * 1000
|
||||
}
|
||||
|
||||
@@ -364,7 +364,7 @@ func (b *meterQueryStatementBuilder) buildTemporalAggCumulativeOrUnspecified(
|
||||
for i, g := range query.GroupBy {
|
||||
wrapped.SelectMore(sqlbuilder.Escape(metricsstatementbuilder.GroupByColumnAlias(i, g.Name)))
|
||||
}
|
||||
wrapped.SelectMore(fmt.Sprintf("%s AS per_series_value", metricsstatementbuilder.RateTmpl))
|
||||
wrapped.SelectMore(fmt.Sprintf("%s AS per_series_value", metricsstatementbuilder.RateExpr(querybuilder.RateLookbackMs(uint64(query.StepInterval.Milliseconds()))/1000)))
|
||||
wrapped.From(fmt.Sprintf("(%s) WINDOW rate_window AS (PARTITION BY fingerprint ORDER BY fingerprint, ts)", innerQuery))
|
||||
q, args := wrapped.BuildWithFlavor(sqlbuilder.ClickHouse, innerArgs...)
|
||||
return fmt.Sprintf("__temporal_aggregation_cte AS (%s)", q), args, nil
|
||||
@@ -375,7 +375,7 @@ func (b *meterQueryStatementBuilder) buildTemporalAggCumulativeOrUnspecified(
|
||||
for i, g := range query.GroupBy {
|
||||
wrapped.SelectMore(sqlbuilder.Escape(metricsstatementbuilder.GroupByColumnAlias(i, g.Name)))
|
||||
}
|
||||
wrapped.SelectMore(fmt.Sprintf("%s AS per_series_value", metricsstatementbuilder.IncreaseTmpl))
|
||||
wrapped.SelectMore(fmt.Sprintf("%s AS per_series_value", metricsstatementbuilder.IncreaseExpr(querybuilder.RateLookbackMs(uint64(query.StepInterval.Milliseconds()))/1000)))
|
||||
wrapped.From(fmt.Sprintf("(%s) WINDOW rate_window AS (PARTITION BY fingerprint ORDER BY fingerprint, ts)", innerQuery))
|
||||
q, args := wrapped.BuildWithFlavor(sqlbuilder.ClickHouse, innerArgs...)
|
||||
return fmt.Sprintf("__temporal_aggregation_cte AS (%s)", q), args, nil
|
||||
|
||||
@@ -56,7 +56,7 @@ func TestStatementBuilder(t *testing.T) {
|
||||
},
|
||||
},
|
||||
expected: qbtypes.Statement{
|
||||
Query: "WITH __temporal_aggregation_cte AS (SELECT ts, `__GROUP_BY_KEY_0_service.name`, multiIf(row_number() OVER rate_window = 1, nan, (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) < 0, per_series_value / (ts - lagInFrame(ts, 1) OVER rate_window), (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) / (ts - lagInFrame(ts, 1) OVER rate_window)) AS per_series_value FROM (SELECT fingerprint, toStartOfInterval(toDateTime(intDiv(unix_milli, 1000)), toIntervalSecond(86400)) AS ts, JSONExtractString(labels, 'service.name') AS `__GROUP_BY_KEY_0_service.name`, max(value) AS per_series_value FROM signoz_meter.distributed_samples AS points WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli < ? AND JSONExtractString(labels, 'service.name') = ? AND LOWER(temporality) LIKE LOWER(?) GROUP BY fingerprint, ts, `__GROUP_BY_KEY_0_service.name` ORDER BY fingerprint, ts) WINDOW rate_window AS (PARTITION BY fingerprint ORDER BY fingerprint, ts)), __spatial_aggregation_cte AS (SELECT ts, `__GROUP_BY_KEY_0_service.name`, sum(per_series_value) AS value FROM __temporal_aggregation_cte WHERE isNaN(per_series_value) = ? GROUP BY ts, `__GROUP_BY_KEY_0_service.name`) SELECT * FROM __spatial_aggregation_cte ORDER BY `__GROUP_BY_KEY_0_service.name`, ts",
|
||||
Query: "WITH __temporal_aggregation_cte AS (SELECT ts, `__GROUP_BY_KEY_0_service.name`, multiIf((row_number() OVER rate_window = 1 OR (ts - lagInFrame(ts, 1) OVER rate_window) > 86400), nan, (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) < 0, per_series_value / (ts - lagInFrame(ts, 1) OVER rate_window), (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) / (ts - lagInFrame(ts, 1) OVER rate_window)) AS per_series_value FROM (SELECT fingerprint, toStartOfInterval(toDateTime(intDiv(unix_milli, 1000)), toIntervalSecond(86400)) AS ts, JSONExtractString(labels, 'service.name') AS `__GROUP_BY_KEY_0_service.name`, max(value) AS per_series_value FROM signoz_meter.distributed_samples AS points WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli < ? AND JSONExtractString(labels, 'service.name') = ? AND LOWER(temporality) LIKE LOWER(?) GROUP BY fingerprint, ts, `__GROUP_BY_KEY_0_service.name` ORDER BY fingerprint, ts) WINDOW rate_window AS (PARTITION BY fingerprint ORDER BY fingerprint, ts)), __spatial_aggregation_cte AS (SELECT ts, `__GROUP_BY_KEY_0_service.name`, sum(per_series_value) AS value FROM __temporal_aggregation_cte WHERE isNaN(per_series_value) = ? GROUP BY ts, `__GROUP_BY_KEY_0_service.name`) SELECT * FROM __spatial_aggregation_cte ORDER BY `__GROUP_BY_KEY_0_service.name`, ts",
|
||||
Args: []any{"signoz_calls_total", uint64(1747785600000), uint64(1747983420000), "cartservice", "cumulative", 0},
|
||||
},
|
||||
expectedErr: nil,
|
||||
|
||||
33
pkg/statementbuilder/metricsstatementbuilder/rate.go
Normal file
33
pkg/statementbuilder/metricsstatementbuilder/rate.go
Normal file
@@ -0,0 +1,33 @@
|
||||
package metricsstatementbuilder
|
||||
|
||||
import "fmt"
|
||||
|
||||
// noPredecessor is true for the first bucket of a series in the statement
|
||||
// and for a bucket whose previous bucket lies further back than the lookback.
|
||||
// Both get nan: a value computed against a sample outside the lookback would
|
||||
// depend on where the statement window starts.
|
||||
func noPredecessor(lookbackSec uint64) string {
|
||||
return fmt.Sprintf("(row_number() OVER rate_window = 1 OR (ts - lagInFrame(ts, 1) OVER rate_window) > %d)", lookbackSec)
|
||||
}
|
||||
|
||||
// RateExpr is the per-bucket rate of a cumulative series: the increase since
|
||||
// the previous bucket divided by the seconds between them; a reset counts the
|
||||
// bucket's own value.
|
||||
func RateExpr(lookbackSec uint64) string {
|
||||
return fmt.Sprintf(`multiIf(%s, nan, (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) < 0, per_series_value / (ts - lagInFrame(ts, 1) OVER rate_window), (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) / (ts - lagInFrame(ts, 1) OVER rate_window))`, noPredecessor(lookbackSec))
|
||||
}
|
||||
|
||||
// IncreaseExpr is the per-bucket increase of a cumulative series.
|
||||
func IncreaseExpr(lookbackSec uint64) string {
|
||||
return fmt.Sprintf(`multiIf(%s, nan, (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) < 0, per_series_value, per_series_value - lagInFrame(per_series_value, 1) OVER rate_window)`, noPredecessor(lookbackSec))
|
||||
}
|
||||
|
||||
func rateMultiTemporalityExpr(lookbackSec uint64, delta, cumulative string) string {
|
||||
return fmt.Sprintf(`IF(LOWER(temporality) LIKE LOWER('delta'), %s, multiIf(%s, nan, (%s - lagInFrame(%s, 1) OVER rate_window) < 0, %s / (ts - lagInFrame(ts, 1) OVER rate_window), (%s - lagInFrame(%s, 1) OVER rate_window) / (ts - lagInFrame(ts, 1) OVER rate_window))) AS per_series_value`,
|
||||
delta, noPredecessor(lookbackSec), cumulative, cumulative, cumulative, cumulative, cumulative)
|
||||
}
|
||||
|
||||
func increaseMultiTemporalityExpr(lookbackSec uint64, delta, cumulative string) string {
|
||||
return fmt.Sprintf(`IF(LOWER(temporality) LIKE LOWER('delta'), %s, multiIf(%s, nan, (%s - lagInFrame(%s, 1) OVER rate_window) < 0, %s, (%s - lagInFrame(%s, 1) OVER rate_window))) AS per_series_value`,
|
||||
delta, noPredecessor(lookbackSec), cumulative, cumulative, cumulative, cumulative, cumulative)
|
||||
}
|
||||
@@ -90,23 +90,23 @@ func TestReducedStatementBuilder(t *testing.T) {
|
||||
name: "counter_sum_rate",
|
||||
query: reducedQuery("test.metric.sum", metrictypes.SumType, metrictypes.Cumulative, metrictypes.TimeAggregationRate, metrictypes.SpaceAggregationSum),
|
||||
expected: qbtypes.Statement{
|
||||
Query: "SELECT * FROM (WITH __temporal_aggregation_cte AS (SELECT ts, multiIf(row_number() OVER rate_window = 1, nan, (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) < 0, per_series_value / (ts - lagInFrame(ts, 1) OVER rate_window), (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) / (ts - lagInFrame(ts, 1) OVER rate_window)) AS per_series_value FROM (SELECT fingerprint, toStartOfInterval(toDateTime(intDiv(unix_milli, 1000)), toIntervalSecond(300)) AS ts, max(max) AS per_series_value FROM signoz_metrics.distributed_samples_v4_agg_5m AS points INNER JOIN (SELECT fingerprint FROM signoz_metrics.time_series_v4_1day WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli <= ? AND LOWER(temporality) LIKE LOWER(?) GROUP BY fingerprint) AS filtered_time_series ON points.fingerprint = filtered_time_series.fingerprint WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli < ? GROUP BY fingerprint, ts ORDER BY fingerprint, ts) WINDOW rate_window AS (PARTITION BY fingerprint ORDER BY fingerprint, ts)), __spatial_aggregation_cte AS (SELECT ts, sum(per_series_value) AS value FROM __temporal_aggregation_cte WHERE isNaN(per_series_value) = ? GROUP BY ts) SELECT * FROM __spatial_aggregation_cte ORDER BY ts) UNION ALL SELECT * FROM (WITH __spatial_aggregation_cte AS (SELECT toStartOfInterval(toDateTime(intDiv(unix_milli, 1000)), toIntervalSecond(300)) AS ts, sum(`sum`) / 300 AS value FROM signoz_metrics.distributed_samples_v4_reduced_sum_60s AS points FINAL INNER JOIN (SELECT fingerprint FROM signoz_metrics.time_series_v4_reduced WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli <= ? GROUP BY fingerprint) AS filtered_time_series ON points.reduced_fingerprint = filtered_time_series.fingerprint WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli < ? GROUP BY ts) SELECT * FROM __spatial_aggregation_cte ORDER BY ts) ORDER BY ts SETTINGS do_not_merge_across_partitions_select_final = 1, optimize_move_to_prewhere_if_final = 1",
|
||||
Args: []any{"test.metric.sum", uint64(1746921600000), uint64(1747172760000), "cumulative", "test.metric.sum", uint64(1746999600000), uint64(1747172760000), 0, "test.metric.sum", uint64(1746997200000), uint64(1747172760000), "test.metric.sum", uint64(1746999600000), uint64(1747172760000)},
|
||||
Query: "SELECT * FROM (WITH __temporal_aggregation_cte AS (SELECT * FROM (SELECT ts, multiIf((row_number() OVER rate_window = 1 OR (ts - lagInFrame(ts, 1) OVER rate_window) > 300), nan, (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) < 0, per_series_value / (ts - lagInFrame(ts, 1) OVER rate_window), (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) / (ts - lagInFrame(ts, 1) OVER rate_window)) AS per_series_value FROM (SELECT fingerprint, toStartOfInterval(toDateTime(intDiv(unix_milli, 1000)), toIntervalSecond(300)) AS ts, max(max) AS per_series_value FROM signoz_metrics.distributed_samples_v4_agg_5m AS points INNER JOIN (SELECT fingerprint FROM signoz_metrics.time_series_v4_1day WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli <= ? AND LOWER(temporality) LIKE LOWER(?) GROUP BY fingerprint) AS filtered_time_series ON points.fingerprint = filtered_time_series.fingerprint WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli < ? GROUP BY fingerprint, ts ORDER BY fingerprint, ts) WINDOW rate_window AS (PARTITION BY fingerprint ORDER BY fingerprint, ts)) WHERE ts >= toDateTime(1746999900)), __spatial_aggregation_cte AS (SELECT ts, sum(per_series_value) AS value FROM __temporal_aggregation_cte WHERE isNaN(per_series_value) = ? GROUP BY ts) SELECT * FROM __spatial_aggregation_cte ORDER BY ts) UNION ALL SELECT * FROM (WITH __spatial_aggregation_cte AS (SELECT toStartOfInterval(toDateTime(intDiv(unix_milli, 1000)), toIntervalSecond(300)) AS ts, sum(`sum`) / 300 AS value FROM signoz_metrics.distributed_samples_v4_reduced_sum_60s AS points FINAL INNER JOIN (SELECT fingerprint FROM signoz_metrics.time_series_v4_reduced WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli <= ? GROUP BY fingerprint) AS filtered_time_series ON points.reduced_fingerprint = filtered_time_series.fingerprint WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli < ? GROUP BY ts) SELECT * FROM __spatial_aggregation_cte ORDER BY ts) ORDER BY ts SETTINGS do_not_merge_across_partitions_select_final = 1, optimize_move_to_prewhere_if_final = 1",
|
||||
Args: []any{"test.metric.sum", uint64(1746921600000), uint64(1747172760000), "cumulative", "test.metric.sum", uint64(1746999600000), uint64(1747172760000), 0, "test.metric.sum", uint64(1746997200000), uint64(1747172760000), "test.metric.sum", uint64(1746999900000), uint64(1747172760000)},
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "counter_avg_increase",
|
||||
query: reducedQuery("test.metric", metrictypes.SumType, metrictypes.Cumulative, metrictypes.TimeAggregationIncrease, metrictypes.SpaceAggregationAvg),
|
||||
expected: qbtypes.Statement{
|
||||
Query: "SELECT * FROM (WITH __temporal_aggregation_cte AS (SELECT ts, multiIf(row_number() OVER rate_window = 1, nan, (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) < 0, per_series_value, per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) AS per_series_value FROM (SELECT fingerprint, toStartOfInterval(toDateTime(intDiv(unix_milli, 1000)), toIntervalSecond(300)) AS ts, max(max) AS per_series_value FROM signoz_metrics.distributed_samples_v4_agg_5m AS points INNER JOIN (SELECT fingerprint FROM signoz_metrics.time_series_v4_1day WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli <= ? AND LOWER(temporality) LIKE LOWER(?) GROUP BY fingerprint) AS filtered_time_series ON points.fingerprint = filtered_time_series.fingerprint WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli < ? GROUP BY fingerprint, ts ORDER BY fingerprint, ts) WINDOW rate_window AS (PARTITION BY fingerprint ORDER BY fingerprint, ts)), __spatial_aggregation_cte AS (SELECT ts, avg(per_series_value) AS value FROM __temporal_aggregation_cte WHERE isNaN(per_series_value) = ? GROUP BY ts) SELECT * FROM __spatial_aggregation_cte ORDER BY ts) UNION ALL SELECT * FROM (WITH __temporal_aggregation_cte AS (SELECT points.reduced_fingerprint AS fingerprint, toStartOfInterval(toDateTime(intDiv(unix_milli, 1000)), toIntervalSecond(300)) AS ts, sum(`sum`) AS per_series_value, avg(`count_series`) AS per_series_weight FROM signoz_metrics.distributed_samples_v4_reduced_sum_60s AS points FINAL INNER JOIN (SELECT fingerprint FROM signoz_metrics.time_series_v4_reduced WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli <= ? GROUP BY fingerprint) AS filtered_time_series ON points.reduced_fingerprint = filtered_time_series.fingerprint WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli < ? GROUP BY fingerprint, ts), __spatial_aggregation_cte AS (SELECT ts, sum(per_series_value) / sum(per_series_weight) AS value FROM __temporal_aggregation_cte GROUP BY ts) SELECT * FROM __spatial_aggregation_cte ORDER BY ts) ORDER BY ts SETTINGS do_not_merge_across_partitions_select_final = 1, optimize_move_to_prewhere_if_final = 1",
|
||||
Args: []any{"test.metric", uint64(1746921600000), uint64(1747172760000), "cumulative", "test.metric", uint64(1746999600000), uint64(1747172760000), 0, "test.metric", uint64(1746997200000), uint64(1747172760000), "test.metric", uint64(1746999600000), uint64(1747172760000)},
|
||||
Query: "SELECT * FROM (WITH __temporal_aggregation_cte AS (SELECT * FROM (SELECT ts, multiIf((row_number() OVER rate_window = 1 OR (ts - lagInFrame(ts, 1) OVER rate_window) > 300), nan, (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) < 0, per_series_value, per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) AS per_series_value FROM (SELECT fingerprint, toStartOfInterval(toDateTime(intDiv(unix_milli, 1000)), toIntervalSecond(300)) AS ts, max(max) AS per_series_value FROM signoz_metrics.distributed_samples_v4_agg_5m AS points INNER JOIN (SELECT fingerprint FROM signoz_metrics.time_series_v4_1day WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli <= ? AND LOWER(temporality) LIKE LOWER(?) GROUP BY fingerprint) AS filtered_time_series ON points.fingerprint = filtered_time_series.fingerprint WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli < ? GROUP BY fingerprint, ts ORDER BY fingerprint, ts) WINDOW rate_window AS (PARTITION BY fingerprint ORDER BY fingerprint, ts)) WHERE ts >= toDateTime(1746999900)), __spatial_aggregation_cte AS (SELECT ts, avg(per_series_value) AS value FROM __temporal_aggregation_cte WHERE isNaN(per_series_value) = ? GROUP BY ts) SELECT * FROM __spatial_aggregation_cte ORDER BY ts) UNION ALL SELECT * FROM (WITH __temporal_aggregation_cte AS (SELECT points.reduced_fingerprint AS fingerprint, toStartOfInterval(toDateTime(intDiv(unix_milli, 1000)), toIntervalSecond(300)) AS ts, sum(`sum`) AS per_series_value, avg(`count_series`) AS per_series_weight FROM signoz_metrics.distributed_samples_v4_reduced_sum_60s AS points FINAL INNER JOIN (SELECT fingerprint FROM signoz_metrics.time_series_v4_reduced WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli <= ? GROUP BY fingerprint) AS filtered_time_series ON points.reduced_fingerprint = filtered_time_series.fingerprint WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli < ? GROUP BY fingerprint, ts), __spatial_aggregation_cte AS (SELECT ts, sum(per_series_value) / sum(per_series_weight) AS value FROM __temporal_aggregation_cte GROUP BY ts) SELECT * FROM __spatial_aggregation_cte ORDER BY ts) ORDER BY ts SETTINGS do_not_merge_across_partitions_select_final = 1, optimize_move_to_prewhere_if_final = 1",
|
||||
Args: []any{"test.metric", uint64(1746921600000), uint64(1747172760000), "cumulative", "test.metric", uint64(1746999600000), uint64(1747172760000), 0, "test.metric", uint64(1746997200000), uint64(1747172760000), "test.metric", uint64(1746999900000), uint64(1747172760000)},
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "counter_min_omitted",
|
||||
query: reducedQuery("test.metric", metrictypes.SumType, metrictypes.Cumulative, metrictypes.TimeAggregationRate, metrictypes.SpaceAggregationMin),
|
||||
expected: qbtypes.Statement{
|
||||
Query: "WITH __temporal_aggregation_cte AS (SELECT ts, multiIf(row_number() OVER rate_window = 1, nan, (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) < 0, per_series_value / (ts - lagInFrame(ts, 1) OVER rate_window), (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) / (ts - lagInFrame(ts, 1) OVER rate_window)) AS per_series_value FROM (SELECT fingerprint, toStartOfInterval(toDateTime(intDiv(unix_milli, 1000)), toIntervalSecond(300)) AS ts, max(max) AS per_series_value FROM signoz_metrics.distributed_samples_v4_agg_5m AS points INNER JOIN (SELECT fingerprint FROM signoz_metrics.time_series_v4_1day WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli <= ? AND LOWER(temporality) LIKE LOWER(?) GROUP BY fingerprint) AS filtered_time_series ON points.fingerprint = filtered_time_series.fingerprint WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli < ? GROUP BY fingerprint, ts ORDER BY fingerprint, ts) WINDOW rate_window AS (PARTITION BY fingerprint ORDER BY fingerprint, ts)), __spatial_aggregation_cte AS (SELECT ts, min(per_series_value) AS value FROM __temporal_aggregation_cte WHERE isNaN(per_series_value) = ? GROUP BY ts) SELECT * FROM __spatial_aggregation_cte ORDER BY ts",
|
||||
Query: "WITH __temporal_aggregation_cte AS (SELECT * FROM (SELECT ts, multiIf((row_number() OVER rate_window = 1 OR (ts - lagInFrame(ts, 1) OVER rate_window) > 300), nan, (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) < 0, per_series_value / (ts - lagInFrame(ts, 1) OVER rate_window), (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) / (ts - lagInFrame(ts, 1) OVER rate_window)) AS per_series_value FROM (SELECT fingerprint, toStartOfInterval(toDateTime(intDiv(unix_milli, 1000)), toIntervalSecond(300)) AS ts, max(max) AS per_series_value FROM signoz_metrics.distributed_samples_v4_agg_5m AS points INNER JOIN (SELECT fingerprint FROM signoz_metrics.time_series_v4_1day WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli <= ? AND LOWER(temporality) LIKE LOWER(?) GROUP BY fingerprint) AS filtered_time_series ON points.fingerprint = filtered_time_series.fingerprint WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli < ? GROUP BY fingerprint, ts ORDER BY fingerprint, ts) WINDOW rate_window AS (PARTITION BY fingerprint ORDER BY fingerprint, ts)) WHERE ts >= toDateTime(1746999900)), __spatial_aggregation_cte AS (SELECT ts, min(per_series_value) AS value FROM __temporal_aggregation_cte WHERE isNaN(per_series_value) = ? GROUP BY ts) SELECT * FROM __spatial_aggregation_cte ORDER BY ts",
|
||||
Args: []any{"test.metric", uint64(1746921600000), uint64(1747172760000), "cumulative", "test.metric", uint64(1746999600000), uint64(1747172760000), 0},
|
||||
},
|
||||
},
|
||||
@@ -114,7 +114,7 @@ func TestReducedStatementBuilder(t *testing.T) {
|
||||
name: "counter_max_omitted",
|
||||
query: reducedQuery("test.metric", metrictypes.SumType, metrictypes.Cumulative, metrictypes.TimeAggregationRate, metrictypes.SpaceAggregationMax),
|
||||
expected: qbtypes.Statement{
|
||||
Query: "WITH __temporal_aggregation_cte AS (SELECT ts, multiIf(row_number() OVER rate_window = 1, nan, (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) < 0, per_series_value / (ts - lagInFrame(ts, 1) OVER rate_window), (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) / (ts - lagInFrame(ts, 1) OVER rate_window)) AS per_series_value FROM (SELECT fingerprint, toStartOfInterval(toDateTime(intDiv(unix_milli, 1000)), toIntervalSecond(300)) AS ts, max(max) AS per_series_value FROM signoz_metrics.distributed_samples_v4_agg_5m AS points INNER JOIN (SELECT fingerprint FROM signoz_metrics.time_series_v4_1day WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli <= ? AND LOWER(temporality) LIKE LOWER(?) GROUP BY fingerprint) AS filtered_time_series ON points.fingerprint = filtered_time_series.fingerprint WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli < ? GROUP BY fingerprint, ts ORDER BY fingerprint, ts) WINDOW rate_window AS (PARTITION BY fingerprint ORDER BY fingerprint, ts)), __spatial_aggregation_cte AS (SELECT ts, max(per_series_value) AS value FROM __temporal_aggregation_cte WHERE isNaN(per_series_value) = ? GROUP BY ts) SELECT * FROM __spatial_aggregation_cte ORDER BY ts",
|
||||
Query: "WITH __temporal_aggregation_cte AS (SELECT * FROM (SELECT ts, multiIf((row_number() OVER rate_window = 1 OR (ts - lagInFrame(ts, 1) OVER rate_window) > 300), nan, (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) < 0, per_series_value / (ts - lagInFrame(ts, 1) OVER rate_window), (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) / (ts - lagInFrame(ts, 1) OVER rate_window)) AS per_series_value FROM (SELECT fingerprint, toStartOfInterval(toDateTime(intDiv(unix_milli, 1000)), toIntervalSecond(300)) AS ts, max(max) AS per_series_value FROM signoz_metrics.distributed_samples_v4_agg_5m AS points INNER JOIN (SELECT fingerprint FROM signoz_metrics.time_series_v4_1day WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli <= ? AND LOWER(temporality) LIKE LOWER(?) GROUP BY fingerprint) AS filtered_time_series ON points.fingerprint = filtered_time_series.fingerprint WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli < ? GROUP BY fingerprint, ts ORDER BY fingerprint, ts) WINDOW rate_window AS (PARTITION BY fingerprint ORDER BY fingerprint, ts)) WHERE ts >= toDateTime(1746999900)), __spatial_aggregation_cte AS (SELECT ts, max(per_series_value) AS value FROM __temporal_aggregation_cte WHERE isNaN(per_series_value) = ? GROUP BY ts) SELECT * FROM __spatial_aggregation_cte ORDER BY ts",
|
||||
Args: []any{"test.metric", uint64(1746921600000), uint64(1747172760000), "cumulative", "test.metric", uint64(1746999600000), uint64(1747172760000), 0},
|
||||
},
|
||||
},
|
||||
@@ -122,16 +122,16 @@ func TestReducedStatementBuilder(t *testing.T) {
|
||||
name: "histogram_p99",
|
||||
query: reducedQuery("test.metric.bucket", metrictypes.HistogramType, metrictypes.Cumulative, metrictypes.TimeAggregationUnspecified, metrictypes.SpaceAggregationPercentile99),
|
||||
expected: qbtypes.Statement{
|
||||
Query: "SELECT * FROM (WITH __temporal_aggregation_cte AS (SELECT ts, `le`, multiIf(row_number() OVER rate_window = 1, nan, (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) < 0, per_series_value / (ts - lagInFrame(ts, 1) OVER rate_window), (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) / (ts - lagInFrame(ts, 1) OVER rate_window)) AS per_series_value FROM (SELECT fingerprint, toStartOfInterval(toDateTime(intDiv(unix_milli, 1000)), toIntervalSecond(300)) AS ts, `le`, max(max) AS per_series_value FROM signoz_metrics.distributed_samples_v4_agg_5m AS points INNER JOIN (SELECT fingerprint, JSONExtractString(labels, 'le') AS `le` FROM signoz_metrics.time_series_v4_1day WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli <= ? AND LOWER(temporality) LIKE LOWER(?) GROUP BY fingerprint, `le`) AS filtered_time_series ON points.fingerprint = filtered_time_series.fingerprint WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli < ? GROUP BY fingerprint, ts, `le` ORDER BY fingerprint, ts) WINDOW rate_window AS (PARTITION BY fingerprint ORDER BY fingerprint, ts)), __spatial_aggregation_cte AS (SELECT ts, `le`, sum(per_series_value) AS value FROM __temporal_aggregation_cte WHERE isNaN(per_series_value) = ? GROUP BY ts, `le`) SELECT ts, histogramQuantile(arrayMap(x -> toFloat64(x), groupArray(le)), groupArray(value), 0.990) AS value FROM __spatial_aggregation_cte GROUP BY ts ORDER BY ts) UNION ALL SELECT * FROM (WITH __spatial_aggregation_cte AS (SELECT toStartOfInterval(toDateTime(intDiv(unix_milli, 1000)), toIntervalSecond(300)) AS ts, `le`, sum(`sum`) / 300 AS value FROM signoz_metrics.distributed_samples_v4_reduced_sum_60s AS points FINAL INNER JOIN (SELECT fingerprint, JSONExtractString(labels, 'le') AS `le` FROM signoz_metrics.time_series_v4_reduced WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli <= ? GROUP BY fingerprint, `le`) AS filtered_time_series ON points.reduced_fingerprint = filtered_time_series.fingerprint WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli < ? GROUP BY ts, `le`) SELECT ts, histogramQuantile(arrayMap(x -> toFloat64(x), groupArray(le)), groupArray(value), 0.990) AS value FROM __spatial_aggregation_cte GROUP BY ts ORDER BY ts) ORDER BY ts SETTINGS do_not_merge_across_partitions_select_final = 1, optimize_move_to_prewhere_if_final = 1",
|
||||
Args: []any{"test.metric.bucket", uint64(1746921600000), uint64(1747172760000), "cumulative", "test.metric.bucket", uint64(1746999900000), uint64(1747172760000), 0, "test.metric.bucket", uint64(1746997200000), uint64(1747172760000), "test.metric.bucket", uint64(1746999900000), uint64(1747172760000)},
|
||||
Query: "SELECT * FROM (WITH __temporal_aggregation_cte AS (SELECT * FROM (SELECT ts, `le`, multiIf((row_number() OVER rate_window = 1 OR (ts - lagInFrame(ts, 1) OVER rate_window) > 300), nan, (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) < 0, per_series_value / (ts - lagInFrame(ts, 1) OVER rate_window), (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) / (ts - lagInFrame(ts, 1) OVER rate_window)) AS per_series_value FROM (SELECT fingerprint, toStartOfInterval(toDateTime(intDiv(unix_milli, 1000)), toIntervalSecond(300)) AS ts, `le`, max(max) AS per_series_value FROM signoz_metrics.distributed_samples_v4_agg_5m AS points INNER JOIN (SELECT fingerprint, JSONExtractString(labels, 'le') AS `le` FROM signoz_metrics.time_series_v4_1day WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli <= ? AND LOWER(temporality) LIKE LOWER(?) GROUP BY fingerprint, `le`) AS filtered_time_series ON points.fingerprint = filtered_time_series.fingerprint WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli < ? GROUP BY fingerprint, ts, `le` ORDER BY fingerprint, ts) WINDOW rate_window AS (PARTITION BY fingerprint ORDER BY fingerprint, ts)) WHERE ts >= toDateTime(1746999900)), __spatial_aggregation_cte AS (SELECT ts, `le`, sum(per_series_value) AS value FROM __temporal_aggregation_cte WHERE isNaN(per_series_value) = ? GROUP BY ts, `le`) SELECT ts, histogramQuantile(arrayMap(x -> toFloat64(x), groupArray(le)), groupArray(value), 0.990) AS value FROM __spatial_aggregation_cte GROUP BY ts ORDER BY ts) UNION ALL SELECT * FROM (WITH __spatial_aggregation_cte AS (SELECT toStartOfInterval(toDateTime(intDiv(unix_milli, 1000)), toIntervalSecond(300)) AS ts, `le`, sum(`sum`) / 300 AS value FROM signoz_metrics.distributed_samples_v4_reduced_sum_60s AS points FINAL INNER JOIN (SELECT fingerprint, JSONExtractString(labels, 'le') AS `le` FROM signoz_metrics.time_series_v4_reduced WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli <= ? GROUP BY fingerprint, `le`) AS filtered_time_series ON points.reduced_fingerprint = filtered_time_series.fingerprint WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli < ? GROUP BY ts, `le`) SELECT ts, histogramQuantile(arrayMap(x -> toFloat64(x), groupArray(le)), groupArray(value), 0.990) AS value FROM __spatial_aggregation_cte GROUP BY ts ORDER BY ts) ORDER BY ts SETTINGS do_not_merge_across_partitions_select_final = 1, optimize_move_to_prewhere_if_final = 1",
|
||||
Args: []any{"test.metric.bucket", uint64(1746921600000), uint64(1747172760000), "cumulative", "test.metric.bucket", uint64(1746999600000), uint64(1747172760000), 0, "test.metric.bucket", uint64(1746997200000), uint64(1747172760000), "test.metric.bucket", uint64(1746999900000), uint64(1747172760000)},
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "histogram_p99_group_by",
|
||||
query: withGroupBy(reducedQuery("test.metric.bucket", metrictypes.HistogramType, metrictypes.Cumulative, metrictypes.TimeAggregationUnspecified, metrictypes.SpaceAggregationPercentile99), "service.name"),
|
||||
expected: qbtypes.Statement{
|
||||
Query: "SELECT * FROM (WITH __temporal_aggregation_cte AS (SELECT ts, `__GROUP_BY_KEY_0_service.name`, `le`, multiIf(row_number() OVER rate_window = 1, nan, (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) < 0, per_series_value / (ts - lagInFrame(ts, 1) OVER rate_window), (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) / (ts - lagInFrame(ts, 1) OVER rate_window)) AS per_series_value FROM (SELECT fingerprint, toStartOfInterval(toDateTime(intDiv(unix_milli, 1000)), toIntervalSecond(300)) AS ts, `__GROUP_BY_KEY_0_service.name`, `le`, max(max) AS per_series_value FROM signoz_metrics.distributed_samples_v4_agg_5m AS points INNER JOIN (SELECT fingerprint, JSONExtractString(labels, 'service.name') AS `__GROUP_BY_KEY_0_service.name`, JSONExtractString(labels, 'le') AS `le` FROM signoz_metrics.time_series_v4_1day WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli <= ? AND LOWER(temporality) LIKE LOWER(?) GROUP BY fingerprint, `__GROUP_BY_KEY_0_service.name`, `le`) AS filtered_time_series ON points.fingerprint = filtered_time_series.fingerprint WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli < ? GROUP BY fingerprint, ts, `__GROUP_BY_KEY_0_service.name`, `le` ORDER BY fingerprint, ts) WINDOW rate_window AS (PARTITION BY fingerprint ORDER BY fingerprint, ts)), __spatial_aggregation_cte AS (SELECT ts, `__GROUP_BY_KEY_0_service.name`, `le`, sum(per_series_value) AS value FROM __temporal_aggregation_cte WHERE isNaN(per_series_value) = ? GROUP BY ts, `__GROUP_BY_KEY_0_service.name`, `le`) SELECT ts, `__GROUP_BY_KEY_0_service.name`, histogramQuantile(arrayMap(x -> toFloat64(x), groupArray(le)), groupArray(value), 0.990) AS value FROM __spatial_aggregation_cte GROUP BY `__GROUP_BY_KEY_0_service.name`, ts ORDER BY `__GROUP_BY_KEY_0_service.name`, ts) UNION ALL SELECT * FROM (WITH __spatial_aggregation_cte AS (SELECT toStartOfInterval(toDateTime(intDiv(unix_milli, 1000)), toIntervalSecond(300)) AS ts, `__GROUP_BY_KEY_0_service.name`, `le`, sum(`sum`) / 300 AS value FROM signoz_metrics.distributed_samples_v4_reduced_sum_60s AS points FINAL INNER JOIN (SELECT fingerprint, JSONExtractString(labels, 'service.name') AS `__GROUP_BY_KEY_0_service.name`, JSONExtractString(labels, 'le') AS `le` FROM signoz_metrics.time_series_v4_reduced WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli <= ? GROUP BY fingerprint, `__GROUP_BY_KEY_0_service.name`, `le`) AS filtered_time_series ON points.reduced_fingerprint = filtered_time_series.fingerprint WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli < ? GROUP BY ts, `__GROUP_BY_KEY_0_service.name`, `le`) SELECT ts, `__GROUP_BY_KEY_0_service.name`, histogramQuantile(arrayMap(x -> toFloat64(x), groupArray(le)), groupArray(value), 0.990) AS value FROM __spatial_aggregation_cte GROUP BY `__GROUP_BY_KEY_0_service.name`, ts ORDER BY `__GROUP_BY_KEY_0_service.name`, ts) ORDER BY `__GROUP_BY_KEY_0_service.name`, ts SETTINGS do_not_merge_across_partitions_select_final = 1, optimize_move_to_prewhere_if_final = 1",
|
||||
Args: []any{"test.metric.bucket", uint64(1746921600000), uint64(1747172760000), "cumulative", "test.metric.bucket", uint64(1746999900000), uint64(1747172760000), 0, "test.metric.bucket", uint64(1746997200000), uint64(1747172760000), "test.metric.bucket", uint64(1746999900000), uint64(1747172760000)},
|
||||
Query: "SELECT * FROM (WITH __temporal_aggregation_cte AS (SELECT * FROM (SELECT ts, `__GROUP_BY_KEY_0_service.name`, `le`, multiIf((row_number() OVER rate_window = 1 OR (ts - lagInFrame(ts, 1) OVER rate_window) > 300), nan, (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) < 0, per_series_value / (ts - lagInFrame(ts, 1) OVER rate_window), (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) / (ts - lagInFrame(ts, 1) OVER rate_window)) AS per_series_value FROM (SELECT fingerprint, toStartOfInterval(toDateTime(intDiv(unix_milli, 1000)), toIntervalSecond(300)) AS ts, `__GROUP_BY_KEY_0_service.name`, `le`, max(max) AS per_series_value FROM signoz_metrics.distributed_samples_v4_agg_5m AS points INNER JOIN (SELECT fingerprint, JSONExtractString(labels, 'service.name') AS `__GROUP_BY_KEY_0_service.name`, JSONExtractString(labels, 'le') AS `le` FROM signoz_metrics.time_series_v4_1day WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli <= ? AND LOWER(temporality) LIKE LOWER(?) GROUP BY fingerprint, `__GROUP_BY_KEY_0_service.name`, `le`) AS filtered_time_series ON points.fingerprint = filtered_time_series.fingerprint WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli < ? GROUP BY fingerprint, ts, `__GROUP_BY_KEY_0_service.name`, `le` ORDER BY fingerprint, ts) WINDOW rate_window AS (PARTITION BY fingerprint ORDER BY fingerprint, ts)) WHERE ts >= toDateTime(1746999900)), __spatial_aggregation_cte AS (SELECT ts, `__GROUP_BY_KEY_0_service.name`, `le`, sum(per_series_value) AS value FROM __temporal_aggregation_cte WHERE isNaN(per_series_value) = ? GROUP BY ts, `__GROUP_BY_KEY_0_service.name`, `le`) SELECT ts, `__GROUP_BY_KEY_0_service.name`, histogramQuantile(arrayMap(x -> toFloat64(x), groupArray(le)), groupArray(value), 0.990) AS value FROM __spatial_aggregation_cte GROUP BY `__GROUP_BY_KEY_0_service.name`, ts ORDER BY `__GROUP_BY_KEY_0_service.name`, ts) UNION ALL SELECT * FROM (WITH __spatial_aggregation_cte AS (SELECT toStartOfInterval(toDateTime(intDiv(unix_milli, 1000)), toIntervalSecond(300)) AS ts, `__GROUP_BY_KEY_0_service.name`, `le`, sum(`sum`) / 300 AS value FROM signoz_metrics.distributed_samples_v4_reduced_sum_60s AS points FINAL INNER JOIN (SELECT fingerprint, JSONExtractString(labels, 'service.name') AS `__GROUP_BY_KEY_0_service.name`, JSONExtractString(labels, 'le') AS `le` FROM signoz_metrics.time_series_v4_reduced WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli <= ? GROUP BY fingerprint, `__GROUP_BY_KEY_0_service.name`, `le`) AS filtered_time_series ON points.reduced_fingerprint = filtered_time_series.fingerprint WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli < ? GROUP BY ts, `__GROUP_BY_KEY_0_service.name`, `le`) SELECT ts, `__GROUP_BY_KEY_0_service.name`, histogramQuantile(arrayMap(x -> toFloat64(x), groupArray(le)), groupArray(value), 0.990) AS value FROM __spatial_aggregation_cte GROUP BY `__GROUP_BY_KEY_0_service.name`, ts ORDER BY `__GROUP_BY_KEY_0_service.name`, ts) ORDER BY `__GROUP_BY_KEY_0_service.name`, ts SETTINGS do_not_merge_across_partitions_select_final = 1, optimize_move_to_prewhere_if_final = 1",
|
||||
Args: []any{"test.metric.bucket", uint64(1746921600000), uint64(1747172760000), "cumulative", "test.metric.bucket", uint64(1746999600000), uint64(1747172760000), 0, "test.metric.bucket", uint64(1746997200000), uint64(1747172760000), "test.metric.bucket", uint64(1746999900000), uint64(1747172760000)},
|
||||
},
|
||||
},
|
||||
{
|
||||
|
||||
@@ -8,7 +8,6 @@ import (
|
||||
"slices"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/SigNoz/signoz/pkg/clickhousesql"
|
||||
"github.com/SigNoz/signoz/pkg/errors"
|
||||
@@ -24,17 +23,7 @@ import (
|
||||
"github.com/huandu/go-sqlbuilder"
|
||||
)
|
||||
|
||||
const (
|
||||
RateTmpl = `multiIf(row_number() OVER rate_window = 1, nan, (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) < 0, per_series_value / (ts - lagInFrame(ts, 1) OVER rate_window), (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) / (ts - lagInFrame(ts, 1) OVER rate_window))`
|
||||
|
||||
IncreaseTmpl = `multiIf(row_number() OVER rate_window = 1, nan, (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) < 0, per_series_value, per_series_value - lagInFrame(per_series_value, 1) OVER rate_window)`
|
||||
|
||||
RateMultiTemporalityTmpl = `IF(LOWER(temporality) LIKE LOWER('delta'), %s, multiIf(row_number() OVER rate_window = 1, nan, (%s - lagInFrame(%s, 1) OVER rate_window) < 0, %s / (ts - lagInFrame(ts, 1) OVER rate_window), (%s - lagInFrame(%s, 1) OVER rate_window) / (ts - lagInFrame(ts, 1) OVER rate_window))) AS per_series_value`
|
||||
|
||||
IncreaseMultiTemporality = `IF(LOWER(temporality) LIKE LOWER('delta'), %s, multiIf(row_number() OVER rate_window = 1, nan, (%s - lagInFrame(%s, 1) OVER rate_window) < 0, %s, (%s - lagInFrame(%s, 1) OVER rate_window))) AS per_series_value`
|
||||
|
||||
OthersMultiTemporality = `IF(LOWER(temporality) LIKE LOWER('delta'), %s, %s) AS per_series_value`
|
||||
)
|
||||
const OthersMultiTemporality = `IF(LOWER(temporality) LIKE LOWER('delta'), %s, %s) AS per_series_value`
|
||||
|
||||
type StatementBuilder struct {
|
||||
logger *slog.Logger
|
||||
@@ -154,9 +143,7 @@ func (b *StatementBuilder) buildPipelineStatement(
|
||||
// samples_v4/agg (unioned with the reduced tables) otherwise. The buffer is
|
||||
// shaped exactly like samples_v4 / time_series_v4, so once the table names are
|
||||
// chosen the rest of the pipeline is unchanged.
|
||||
useBuffer := agg.Reduced &&
|
||||
end-start < metricstelemetryschema.OneDayInMilliseconds &&
|
||||
start >= uint64(time.Now().UnixMilli())-metricstelemetryschema.OneDayInMilliseconds
|
||||
useBuffer := metricstelemetryschema.UsesBuffer(start, end, agg.Reduced, agg.TableHints)
|
||||
|
||||
samplesTable, _ := metricstelemetryschema.WhichSamplesTableToUse(start, end, agg.Type, agg.TimeAggregation, useBuffer, agg.TableHints)
|
||||
tsStart, tsEnd, _, tsTable := metricstelemetryschema.WhichTSTableToUse(start, end, useBuffer, agg.TableHints)
|
||||
@@ -199,6 +186,9 @@ func (b *StatementBuilder) buildPipelineStatement(
|
||||
if agg.Reduced && !useBuffer {
|
||||
var tsCTE string
|
||||
var tsArgs []any
|
||||
// The reduced rows hold per-bucket values that need no predecessor,
|
||||
// so this half starts where the answer starts.
|
||||
start := start + querybuilder.MetricRateLookbackMs(uint64(query.StepInterval.Milliseconds()), cteQuery)
|
||||
// time series rows are written on hour boundaries
|
||||
tsStart := start - (start % metricstelemetryschema.OneHourInMilliseconds)
|
||||
if tsCTE, tsArgs, err = b.buildReducedTimeSeriesCTE(ctx, orgID, tsStart, end, cteQuery, keys, variables); err != nil {
|
||||
@@ -652,31 +642,28 @@ func (b *StatementBuilder) buildTemporalAggCumulativeOrUnspecified(
|
||||
|
||||
innerQuery, innerArgs := baseSb.BuildWithFlavor(sqlbuilder.ClickHouse, timeSeriesCTEArgs...)
|
||||
|
||||
lookbackSec := querybuilder.RateLookbackMs(uint64(stepSec)*1000) / 1000
|
||||
var expr string
|
||||
switch query.Aggregations[0].TimeAggregation {
|
||||
case metrictypes.TimeAggregationRate:
|
||||
wrapped := sqlbuilder.NewSelectBuilder()
|
||||
wrapped.Select("ts")
|
||||
for i, g := range query.GroupBy {
|
||||
wrapped.SelectMore(sqlbuilder.Escape(GroupByColumnAlias(i, g.Name)))
|
||||
}
|
||||
wrapped.SelectMore(fmt.Sprintf("%s AS per_series_value", RateTmpl))
|
||||
wrapped.From(fmt.Sprintf("(%s) WINDOW rate_window AS (PARTITION BY fingerprint ORDER BY fingerprint, ts)", sqlbuilder.Escape(innerQuery)))
|
||||
q, args := wrapped.BuildWithFlavor(sqlbuilder.ClickHouse, innerArgs...)
|
||||
return fmt.Sprintf("__temporal_aggregation_cte AS (%s)", q), args, nil
|
||||
|
||||
expr = RateExpr(lookbackSec)
|
||||
case metrictypes.TimeAggregationIncrease:
|
||||
wrapped := sqlbuilder.NewSelectBuilder()
|
||||
wrapped.Select("ts")
|
||||
for i, g := range query.GroupBy {
|
||||
wrapped.SelectMore(sqlbuilder.Escape(GroupByColumnAlias(i, g.Name)))
|
||||
}
|
||||
wrapped.SelectMore(fmt.Sprintf("%s AS per_series_value", IncreaseTmpl))
|
||||
wrapped.From(fmt.Sprintf("(%s) WINDOW rate_window AS (PARTITION BY fingerprint ORDER BY fingerprint, ts)", sqlbuilder.Escape(innerQuery)))
|
||||
q, args := wrapped.BuildWithFlavor(sqlbuilder.ClickHouse, innerArgs...)
|
||||
return fmt.Sprintf("__temporal_aggregation_cte AS (%s)", q), args, nil
|
||||
expr = IncreaseExpr(lookbackSec)
|
||||
default:
|
||||
return fmt.Sprintf("__temporal_aggregation_cte AS (%s)", innerQuery), innerArgs, nil
|
||||
}
|
||||
wrapped := sqlbuilder.NewSelectBuilder()
|
||||
wrapped.Select("ts")
|
||||
for i, g := range query.GroupBy {
|
||||
wrapped.SelectMore(sqlbuilder.Escape(GroupByColumnAlias(i, g.Name)))
|
||||
}
|
||||
wrapped.SelectMore(fmt.Sprintf("%s AS per_series_value", expr))
|
||||
wrapped.From(fmt.Sprintf("(%s) WINDOW rate_window AS (PARTITION BY fingerprint ORDER BY fingerprint, ts)", sqlbuilder.Escape(innerQuery)))
|
||||
q, args := wrapped.BuildWithFlavor(sqlbuilder.ClickHouse, innerArgs...)
|
||||
// The lookback rows exist to give the first buckets a predecessor; they
|
||||
// are not part of the answer.
|
||||
q = fmt.Sprintf("SELECT * FROM (%s) WHERE ts >= toDateTime(%d)", q, (start+querybuilder.RateLookbackMs(uint64(stepSec)*1000))/1000)
|
||||
return fmt.Sprintf("__temporal_aggregation_cte AS (%s)", q), args, nil
|
||||
}
|
||||
|
||||
func (b *StatementBuilder) buildTemporalAggForMultipleTemporalities(
|
||||
@@ -710,21 +697,15 @@ func (b *StatementBuilder) buildTemporalAggForMultipleTemporalities(
|
||||
aggForDeltaTemporality = fmt.Sprintf("%s/%d", aggForDeltaTemporality, stepSec)
|
||||
}
|
||||
|
||||
lookbackSec := querybuilder.RateLookbackMs(uint64(stepSec)*1000) / 1000
|
||||
usesLookback := false
|
||||
switch query.Aggregations[0].TimeAggregation {
|
||||
case metrictypes.TimeAggregationRate:
|
||||
rateExpr := fmt.Sprintf(RateMultiTemporalityTmpl,
|
||||
aggForDeltaTemporality,
|
||||
aggForCumulativeTemporality, aggForCumulativeTemporality, aggForCumulativeTemporality,
|
||||
aggForCumulativeTemporality, aggForCumulativeTemporality,
|
||||
)
|
||||
sb.SelectMore(rateExpr)
|
||||
sb.SelectMore(rateMultiTemporalityExpr(lookbackSec, aggForDeltaTemporality, aggForCumulativeTemporality))
|
||||
usesLookback = true
|
||||
case metrictypes.TimeAggregationIncrease:
|
||||
increaseExpr := fmt.Sprintf(IncreaseMultiTemporality,
|
||||
aggForDeltaTemporality,
|
||||
aggForCumulativeTemporality, aggForCumulativeTemporality, aggForCumulativeTemporality,
|
||||
aggForCumulativeTemporality, aggForCumulativeTemporality,
|
||||
)
|
||||
sb.SelectMore(increaseExpr)
|
||||
sb.SelectMore(increaseMultiTemporalityExpr(lookbackSec, aggForDeltaTemporality, aggForCumulativeTemporality))
|
||||
usesLookback = true
|
||||
default:
|
||||
expr := fmt.Sprintf(OthersMultiTemporality, aggForDeltaTemporality, aggForCumulativeTemporality)
|
||||
sb.SelectMore(expr)
|
||||
@@ -741,6 +722,9 @@ func (b *StatementBuilder) buildTemporalAggForMultipleTemporalities(
|
||||
sb.GroupBy(GroupByAliases(query.GroupBy)...)
|
||||
queryWithoutWindow, args := sb.BuildWithFlavor(sqlbuilder.ClickHouse, timeSeriesCTEArgs...)
|
||||
queryWithWindowAndOrder := queryWithoutWindow + " WINDOW rate_window AS (PARTITION BY fingerprint ORDER BY fingerprint ASC, ts ASC) ORDER BY ts"
|
||||
if usesLookback {
|
||||
queryWithWindowAndOrder = fmt.Sprintf("SELECT * FROM (%s) WHERE ts >= toDateTime(%d)", queryWithWindowAndOrder, (start+querybuilder.RateLookbackMs(uint64(stepSec)*1000))/1000)
|
||||
}
|
||||
return fmt.Sprintf("__temporal_aggregation_cte AS (%s)", queryWithWindowAndOrder), args, nil
|
||||
}
|
||||
|
||||
|
||||
@@ -55,8 +55,8 @@ func TestStatementBuilder(t *testing.T) {
|
||||
},
|
||||
},
|
||||
expected: qbtypes.Statement{
|
||||
Query: "WITH __temporal_aggregation_cte AS (SELECT ts, `__GROUP_BY_KEY_0_service.name`, multiIf(row_number() OVER rate_window = 1, nan, (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) < 0, per_series_value / (ts - lagInFrame(ts, 1) OVER rate_window), (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) / (ts - lagInFrame(ts, 1) OVER rate_window)) AS per_series_value FROM (SELECT fingerprint, toStartOfInterval(toDateTime(intDiv(unix_milli, 1000)), toIntervalSecond(30)) AS ts, `__GROUP_BY_KEY_0_service.name`, max(value) AS per_series_value FROM signoz_metrics.distributed_samples_v4 AS points INNER JOIN (SELECT fingerprint, JSONExtractString(labels, 'service.name') AS `__GROUP_BY_KEY_0_service.name` FROM signoz_metrics.time_series_v4_6hrs WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli <= ? AND LOWER(temporality) LIKE LOWER(?) AND JSONExtractString(labels, 'service.name') = ? GROUP BY fingerprint, `__GROUP_BY_KEY_0_service.name`) AS filtered_time_series ON points.fingerprint = filtered_time_series.fingerprint WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli < ? GROUP BY fingerprint, ts, `__GROUP_BY_KEY_0_service.name` ORDER BY fingerprint, ts) WINDOW rate_window AS (PARTITION BY fingerprint ORDER BY fingerprint, ts)), __spatial_aggregation_cte AS (SELECT ts, `__GROUP_BY_KEY_0_service.name`, sum(per_series_value) AS value FROM __temporal_aggregation_cte WHERE isNaN(per_series_value) = ? GROUP BY ts, `__GROUP_BY_KEY_0_service.name`) SELECT * FROM __spatial_aggregation_cte ORDER BY `__GROUP_BY_KEY_0_service.name`, ts",
|
||||
Args: []any{"signoz_calls_total", uint64(1747936800000), uint64(1747983420000), "cumulative", "cartservice", "signoz_calls_total", uint64(1747947360000), uint64(1747983420000), 0},
|
||||
Query: "WITH __temporal_aggregation_cte AS (SELECT * FROM (SELECT ts, `__GROUP_BY_KEY_0_service.name`, multiIf((row_number() OVER rate_window = 1 OR (ts - lagInFrame(ts, 1) OVER rate_window) > 300), nan, (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) < 0, per_series_value / (ts - lagInFrame(ts, 1) OVER rate_window), (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) / (ts - lagInFrame(ts, 1) OVER rate_window)) AS per_series_value FROM (SELECT fingerprint, toStartOfInterval(toDateTime(intDiv(unix_milli, 1000)), toIntervalSecond(30)) AS ts, `__GROUP_BY_KEY_0_service.name`, max(value) AS per_series_value FROM signoz_metrics.distributed_samples_v4 AS points INNER JOIN (SELECT fingerprint, JSONExtractString(labels, 'service.name') AS `__GROUP_BY_KEY_0_service.name` FROM signoz_metrics.time_series_v4_6hrs WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli <= ? AND LOWER(temporality) LIKE LOWER(?) AND JSONExtractString(labels, 'service.name') = ? GROUP BY fingerprint, `__GROUP_BY_KEY_0_service.name`) AS filtered_time_series ON points.fingerprint = filtered_time_series.fingerprint WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli < ? GROUP BY fingerprint, ts, `__GROUP_BY_KEY_0_service.name` ORDER BY fingerprint, ts) WINDOW rate_window AS (PARTITION BY fingerprint ORDER BY fingerprint, ts)) WHERE ts >= toDateTime(1747947390)), __spatial_aggregation_cte AS (SELECT ts, `__GROUP_BY_KEY_0_service.name`, sum(per_series_value) AS value FROM __temporal_aggregation_cte WHERE isNaN(per_series_value) = ? GROUP BY ts, `__GROUP_BY_KEY_0_service.name`) SELECT * FROM __spatial_aggregation_cte ORDER BY `__GROUP_BY_KEY_0_service.name`, ts",
|
||||
Args: []any{"signoz_calls_total", uint64(1747936800000), uint64(1747983420000), "cumulative", "cartservice", "signoz_calls_total", uint64(1747947090000), uint64(1747983420000), 0},
|
||||
},
|
||||
expectedErr: nil,
|
||||
},
|
||||
@@ -88,8 +88,8 @@ func TestStatementBuilder(t *testing.T) {
|
||||
},
|
||||
},
|
||||
expected: qbtypes.Statement{
|
||||
Query: "WITH __temporal_aggregation_cte AS (SELECT ts, `__GROUP_BY_KEY_0_service.name`, multiIf(row_number() OVER rate_window = 1, nan, (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) < 0, per_series_value / (ts - lagInFrame(ts, 1) OVER rate_window), (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) / (ts - lagInFrame(ts, 1) OVER rate_window)) AS per_series_value FROM (SELECT fingerprint, toStartOfInterval(toDateTime(intDiv(unix_milli, 1000)), toIntervalSecond(30)) AS ts, `__GROUP_BY_KEY_0_service.name`, max(value) AS per_series_value FROM signoz_metrics.distributed_samples_v4 AS points INNER JOIN (SELECT fingerprint, JSONExtractString(labels, 'service.name') AS `__GROUP_BY_KEY_0_service.name` FROM signoz_metrics.time_series_v4_6hrs WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli <= ? AND LOWER(temporality) LIKE LOWER(?) AND (match(JSONExtractString(labels, 'materialized.key.name'), ?) OR JSONExtractString(labels, 'service.name') = ?) GROUP BY fingerprint, `__GROUP_BY_KEY_0_service.name`) AS filtered_time_series ON points.fingerprint = filtered_time_series.fingerprint WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli < ? GROUP BY fingerprint, ts, `__GROUP_BY_KEY_0_service.name` ORDER BY fingerprint, ts) WINDOW rate_window AS (PARTITION BY fingerprint ORDER BY fingerprint, ts)), __spatial_aggregation_cte AS (SELECT ts, `__GROUP_BY_KEY_0_service.name`, sum(per_series_value) AS value FROM __temporal_aggregation_cte WHERE isNaN(per_series_value) = ? GROUP BY ts, `__GROUP_BY_KEY_0_service.name`) SELECT * FROM __spatial_aggregation_cte ORDER BY `__GROUP_BY_KEY_0_service.name`, ts",
|
||||
Args: []any{"signoz_calls_total", uint64(1747936800000), uint64(1747983420000), "cumulative", "cartservice", "cartservice", "signoz_calls_total", uint64(1747947360000), uint64(1747983420000), 0},
|
||||
Query: "WITH __temporal_aggregation_cte AS (SELECT * FROM (SELECT ts, `__GROUP_BY_KEY_0_service.name`, multiIf((row_number() OVER rate_window = 1 OR (ts - lagInFrame(ts, 1) OVER rate_window) > 300), nan, (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) < 0, per_series_value / (ts - lagInFrame(ts, 1) OVER rate_window), (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) / (ts - lagInFrame(ts, 1) OVER rate_window)) AS per_series_value FROM (SELECT fingerprint, toStartOfInterval(toDateTime(intDiv(unix_milli, 1000)), toIntervalSecond(30)) AS ts, `__GROUP_BY_KEY_0_service.name`, max(value) AS per_series_value FROM signoz_metrics.distributed_samples_v4 AS points INNER JOIN (SELECT fingerprint, JSONExtractString(labels, 'service.name') AS `__GROUP_BY_KEY_0_service.name` FROM signoz_metrics.time_series_v4_6hrs WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli <= ? AND LOWER(temporality) LIKE LOWER(?) AND (match(JSONExtractString(labels, 'materialized.key.name'), ?) OR JSONExtractString(labels, 'service.name') = ?) GROUP BY fingerprint, `__GROUP_BY_KEY_0_service.name`) AS filtered_time_series ON points.fingerprint = filtered_time_series.fingerprint WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli < ? GROUP BY fingerprint, ts, `__GROUP_BY_KEY_0_service.name` ORDER BY fingerprint, ts) WINDOW rate_window AS (PARTITION BY fingerprint ORDER BY fingerprint, ts)) WHERE ts >= toDateTime(1747947390)), __spatial_aggregation_cte AS (SELECT ts, `__GROUP_BY_KEY_0_service.name`, sum(per_series_value) AS value FROM __temporal_aggregation_cte WHERE isNaN(per_series_value) = ? GROUP BY ts, `__GROUP_BY_KEY_0_service.name`) SELECT * FROM __spatial_aggregation_cte ORDER BY `__GROUP_BY_KEY_0_service.name`, ts",
|
||||
Args: []any{"signoz_calls_total", uint64(1747936800000), uint64(1747983420000), "cumulative", "cartservice", "cartservice", "signoz_calls_total", uint64(1747947090000), uint64(1747983420000), 0},
|
||||
},
|
||||
expectedErr: nil,
|
||||
},
|
||||
@@ -440,8 +440,8 @@ func TestStatementBuilder(t *testing.T) {
|
||||
},
|
||||
},
|
||||
expected: qbtypes.Statement{
|
||||
Query: "WITH __temporal_aggregation_cte AS (SELECT ts, `__GROUP_BY_KEY_0_service.name`, `le`, multiIf(row_number() OVER rate_window = 1, nan, (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) < 0, per_series_value, per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) AS per_series_value FROM (SELECT fingerprint, toStartOfInterval(toDateTime(intDiv(unix_milli, 1000)), toIntervalSecond(60)) AS ts, `__GROUP_BY_KEY_0_service.name`, `le`, max(value) AS per_series_value FROM signoz_metrics.distributed_samples_v4 AS points INNER JOIN (SELECT fingerprint, JSONExtractString(labels, 'service.name') AS `__GROUP_BY_KEY_0_service.name`, JSONExtractString(labels, 'le') AS `le` FROM signoz_metrics.time_series_v4_6hrs WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli <= ? AND LOWER(temporality) LIKE LOWER(?) GROUP BY fingerprint, `__GROUP_BY_KEY_0_service.name`, `le`) AS filtered_time_series ON points.fingerprint = filtered_time_series.fingerprint WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli < ? GROUP BY fingerprint, ts, `__GROUP_BY_KEY_0_service.name`, `le` ORDER BY fingerprint, ts) WINDOW rate_window AS (PARTITION BY fingerprint ORDER BY fingerprint, ts)), __spatial_aggregation_cte AS (SELECT ts, `__GROUP_BY_KEY_0_service.name`, `le`, sum(per_series_value) AS value FROM __temporal_aggregation_cte WHERE isNaN(per_series_value) = ? GROUP BY ts, `__GROUP_BY_KEY_0_service.name`, `le`) SELECT ts, `__GROUP_BY_KEY_0_service.name`, lagInFrame(toFloat64(le), 1, toFloat64('-Inf')) OVER __heatmap_window AS __bucket_min, toFloat64(le) AS __bucket_max, greatest(value - lagInFrame(value, 1, 0) OVER __heatmap_window, 0) AS __result_0 FROM __spatial_aggregation_cte WINDOW __heatmap_window AS (PARTITION BY `__GROUP_BY_KEY_0_service.name`, ts ORDER BY toFloat64(le)) ORDER BY `__GROUP_BY_KEY_0_service.name`, ts, toFloat64(le)",
|
||||
Args: []any{"http_server_duration_bucket", uint64(1747936800000), uint64(1747983420000), "cumulative", "http_server_duration_bucket", uint64(1747947300000), uint64(1747983420000), 0},
|
||||
Query: "WITH __temporal_aggregation_cte AS (SELECT * FROM (SELECT ts, `__GROUP_BY_KEY_0_service.name`, `le`, multiIf((row_number() OVER rate_window = 1 OR (ts - lagInFrame(ts, 1) OVER rate_window) > 300), nan, (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) < 0, per_series_value, per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) AS per_series_value FROM (SELECT fingerprint, toStartOfInterval(toDateTime(intDiv(unix_milli, 1000)), toIntervalSecond(60)) AS ts, `__GROUP_BY_KEY_0_service.name`, `le`, max(value) AS per_series_value FROM signoz_metrics.distributed_samples_v4 AS points INNER JOIN (SELECT fingerprint, JSONExtractString(labels, 'service.name') AS `__GROUP_BY_KEY_0_service.name`, JSONExtractString(labels, 'le') AS `le` FROM signoz_metrics.time_series_v4_6hrs WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli <= ? AND LOWER(temporality) LIKE LOWER(?) GROUP BY fingerprint, `__GROUP_BY_KEY_0_service.name`, `le`) AS filtered_time_series ON points.fingerprint = filtered_time_series.fingerprint WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli < ? GROUP BY fingerprint, ts, `__GROUP_BY_KEY_0_service.name`, `le` ORDER BY fingerprint, ts) WINDOW rate_window AS (PARTITION BY fingerprint ORDER BY fingerprint, ts)) WHERE ts >= toDateTime(1747947360)), __spatial_aggregation_cte AS (SELECT ts, `__GROUP_BY_KEY_0_service.name`, `le`, sum(per_series_value) AS value FROM __temporal_aggregation_cte WHERE isNaN(per_series_value) = ? GROUP BY ts, `__GROUP_BY_KEY_0_service.name`, `le`) SELECT ts, `__GROUP_BY_KEY_0_service.name`, lagInFrame(toFloat64(le), 1, toFloat64('-Inf')) OVER __heatmap_window AS __bucket_min, toFloat64(le) AS __bucket_max, greatest(value - lagInFrame(value, 1, 0) OVER __heatmap_window, 0) AS __result_0 FROM __spatial_aggregation_cte WINDOW __heatmap_window AS (PARTITION BY `__GROUP_BY_KEY_0_service.name`, ts ORDER BY toFloat64(le)) ORDER BY `__GROUP_BY_KEY_0_service.name`, ts, toFloat64(le)",
|
||||
Args: []any{"http_server_duration_bucket", uint64(1747936800000), uint64(1747983420000), "cumulative", "http_server_duration_bucket", uint64(1747947060000), uint64(1747983420000), 0},
|
||||
},
|
||||
expectedErr: nil,
|
||||
},
|
||||
@@ -474,8 +474,8 @@ func TestStatementBuilder(t *testing.T) {
|
||||
},
|
||||
},
|
||||
expected: qbtypes.Statement{
|
||||
Query: "WITH __temporal_aggregation_cte AS (SELECT ts, `__GROUP_BY_KEY_0_service.name`, multiIf(row_number() OVER rate_window = 1, nan, (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) < 0, per_series_value, per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) AS per_series_value FROM (SELECT fingerprint, toStartOfInterval(toDateTime(intDiv(unix_milli, 1000)), toIntervalSecond(60)) AS ts, `__GROUP_BY_KEY_0_service.name`, max(value) AS per_series_value FROM signoz_metrics.distributed_samples_v4 AS points INNER JOIN (SELECT fingerprint, JSONExtractString(labels, 'service.name') AS `__GROUP_BY_KEY_0_service.name` FROM signoz_metrics.time_series_v4_6hrs WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli <= ? AND LOWER(temporality) LIKE LOWER(?) GROUP BY fingerprint, `__GROUP_BY_KEY_0_service.name`) AS filtered_time_series ON points.fingerprint = filtered_time_series.fingerprint WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli < ? GROUP BY fingerprint, ts, `__GROUP_BY_KEY_0_service.name` ORDER BY fingerprint, ts) WINDOW rate_window AS (PARTITION BY fingerprint ORDER BY fingerprint, ts)), __spatial_aggregation_cte AS (SELECT ts, `__GROUP_BY_KEY_0_service.name`, sum(per_series_value) AS value FROM __temporal_aggregation_cte WHERE isNaN(per_series_value) = ? GROUP BY ts, `__GROUP_BY_KEY_0_service.name`) SELECT ts, `__GROUP_BY_KEY_0_service.name`, multiIf(value <= 0, toFloat64('-Inf'), value <= 2.3283064365386963e-10, toFloat64(0), value > 1.8446744073709552e+19, toFloat64(1.8446744073709552e+19), pow(2, (ceil(log2(value) * 16) - 1) / 16)) AS __bucket_min, multiIf(value <= 0, toFloat64(0), value <= 2.3283064365386963e-10, 2.3283064365386963e-10, value > 1.8446744073709552e+19, toFloat64('+Inf'), pow(2, ceil(log2(value) * 16) / 16)) AS __bucket_max, toFloat64(1) AS __result_0 FROM __spatial_aggregation_cte ORDER BY `__GROUP_BY_KEY_0_service.name`, ts, __bucket_max",
|
||||
Args: []any{"signoz_calls_total", uint64(1747936800000), uint64(1747983420000), "cumulative", "signoz_calls_total", uint64(1747947300000), uint64(1747983420000), 0},
|
||||
Query: "WITH __temporal_aggregation_cte AS (SELECT * FROM (SELECT ts, `__GROUP_BY_KEY_0_service.name`, multiIf((row_number() OVER rate_window = 1 OR (ts - lagInFrame(ts, 1) OVER rate_window) > 300), nan, (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) < 0, per_series_value, per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) AS per_series_value FROM (SELECT fingerprint, toStartOfInterval(toDateTime(intDiv(unix_milli, 1000)), toIntervalSecond(60)) AS ts, `__GROUP_BY_KEY_0_service.name`, max(value) AS per_series_value FROM signoz_metrics.distributed_samples_v4 AS points INNER JOIN (SELECT fingerprint, JSONExtractString(labels, 'service.name') AS `__GROUP_BY_KEY_0_service.name` FROM signoz_metrics.time_series_v4_6hrs WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli <= ? AND LOWER(temporality) LIKE LOWER(?) GROUP BY fingerprint, `__GROUP_BY_KEY_0_service.name`) AS filtered_time_series ON points.fingerprint = filtered_time_series.fingerprint WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli < ? GROUP BY fingerprint, ts, `__GROUP_BY_KEY_0_service.name` ORDER BY fingerprint, ts) WINDOW rate_window AS (PARTITION BY fingerprint ORDER BY fingerprint, ts)) WHERE ts >= toDateTime(1747947360)), __spatial_aggregation_cte AS (SELECT ts, `__GROUP_BY_KEY_0_service.name`, sum(per_series_value) AS value FROM __temporal_aggregation_cte WHERE isNaN(per_series_value) = ? GROUP BY ts, `__GROUP_BY_KEY_0_service.name`) SELECT ts, `__GROUP_BY_KEY_0_service.name`, multiIf(value <= 0, toFloat64('-Inf'), value <= 2.3283064365386963e-10, toFloat64(0), value > 1.8446744073709552e+19, toFloat64(1.8446744073709552e+19), pow(2, (ceil(log2(value) * 16) - 1) / 16)) AS __bucket_min, multiIf(value <= 0, toFloat64(0), value <= 2.3283064365386963e-10, 2.3283064365386963e-10, value > 1.8446744073709552e+19, toFloat64('+Inf'), pow(2, ceil(log2(value) * 16) / 16)) AS __bucket_max, toFloat64(1) AS __result_0 FROM __spatial_aggregation_cte ORDER BY `__GROUP_BY_KEY_0_service.name`, ts, __bucket_max",
|
||||
Args: []any{"signoz_calls_total", uint64(1747936800000), uint64(1747983420000), "cumulative", "signoz_calls_total", uint64(1747947060000), uint64(1747983420000), 0},
|
||||
},
|
||||
expectedErr: nil,
|
||||
},
|
||||
@@ -537,8 +537,8 @@ func TestStatementBuilder(t *testing.T) {
|
||||
},
|
||||
},
|
||||
expected: qbtypes.Statement{
|
||||
Query: "WITH __temporal_aggregation_cte AS (SELECT ts, `__GROUP_BY_KEY_0_service.name`, `le`, multiIf(row_number() OVER rate_window = 1, nan, (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) < 0, per_series_value / (ts - lagInFrame(ts, 1) OVER rate_window), (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) / (ts - lagInFrame(ts, 1) OVER rate_window)) AS per_series_value FROM (SELECT fingerprint, toStartOfInterval(toDateTime(intDiv(unix_milli, 1000)), toIntervalSecond(30)) AS ts, `__GROUP_BY_KEY_0_service.name`, `le`, max(value) AS per_series_value FROM signoz_metrics.distributed_samples_v4 AS points INNER JOIN (SELECT fingerprint, JSONExtractString(labels, 'service.name') AS `__GROUP_BY_KEY_0_service.name`, JSONExtractString(labels, 'le') AS `le` FROM signoz_metrics.time_series_v4_6hrs WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli <= ? AND LOWER(temporality) LIKE LOWER(?) GROUP BY fingerprint, `__GROUP_BY_KEY_0_service.name`, `le`) AS filtered_time_series ON points.fingerprint = filtered_time_series.fingerprint WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli < ? GROUP BY fingerprint, ts, `__GROUP_BY_KEY_0_service.name`, `le` ORDER BY fingerprint, ts) WINDOW rate_window AS (PARTITION BY fingerprint ORDER BY fingerprint, ts)), __spatial_aggregation_cte AS (SELECT ts, `__GROUP_BY_KEY_0_service.name`, `le`, sum(per_series_value) AS value FROM __temporal_aggregation_cte WHERE isNaN(per_series_value) = ? GROUP BY ts, `__GROUP_BY_KEY_0_service.name`, `le`) SELECT ts, `__GROUP_BY_KEY_0_service.name`, histogramQuantile(arrayMap(x -> toFloat64(x), groupArray(le)), groupArray(value), 0.950) AS value FROM __spatial_aggregation_cte GROUP BY `__GROUP_BY_KEY_0_service.name`, ts ORDER BY `__GROUP_BY_KEY_0_service.name`, ts",
|
||||
Args: []any{"http_server_duration_bucket", uint64(1747936800000), uint64(1747983420000), "cumulative", "http_server_duration_bucket", uint64(1747947360000), uint64(1747983420000), 0},
|
||||
Query: "WITH __temporal_aggregation_cte AS (SELECT * FROM (SELECT ts, `__GROUP_BY_KEY_0_service.name`, `le`, multiIf((row_number() OVER rate_window = 1 OR (ts - lagInFrame(ts, 1) OVER rate_window) > 300), nan, (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) < 0, per_series_value / (ts - lagInFrame(ts, 1) OVER rate_window), (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) / (ts - lagInFrame(ts, 1) OVER rate_window)) AS per_series_value FROM (SELECT fingerprint, toStartOfInterval(toDateTime(intDiv(unix_milli, 1000)), toIntervalSecond(30)) AS ts, `__GROUP_BY_KEY_0_service.name`, `le`, max(value) AS per_series_value FROM signoz_metrics.distributed_samples_v4 AS points INNER JOIN (SELECT fingerprint, JSONExtractString(labels, 'service.name') AS `__GROUP_BY_KEY_0_service.name`, JSONExtractString(labels, 'le') AS `le` FROM signoz_metrics.time_series_v4_6hrs WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli <= ? AND LOWER(temporality) LIKE LOWER(?) GROUP BY fingerprint, `__GROUP_BY_KEY_0_service.name`, `le`) AS filtered_time_series ON points.fingerprint = filtered_time_series.fingerprint WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli < ? GROUP BY fingerprint, ts, `__GROUP_BY_KEY_0_service.name`, `le` ORDER BY fingerprint, ts) WINDOW rate_window AS (PARTITION BY fingerprint ORDER BY fingerprint, ts)) WHERE ts >= toDateTime(1747947390)), __spatial_aggregation_cte AS (SELECT ts, `__GROUP_BY_KEY_0_service.name`, `le`, sum(per_series_value) AS value FROM __temporal_aggregation_cte WHERE isNaN(per_series_value) = ? GROUP BY ts, `__GROUP_BY_KEY_0_service.name`, `le`) SELECT ts, `__GROUP_BY_KEY_0_service.name`, histogramQuantile(arrayMap(x -> toFloat64(x), groupArray(le)), groupArray(value), 0.950) AS value FROM __spatial_aggregation_cte GROUP BY `__GROUP_BY_KEY_0_service.name`, ts ORDER BY `__GROUP_BY_KEY_0_service.name`, ts",
|
||||
Args: []any{"http_server_duration_bucket", uint64(1747936800000), uint64(1747983420000), "cumulative", "http_server_duration_bucket", uint64(1747947090000), uint64(1747983420000), 0},
|
||||
},
|
||||
expectedErr: nil,
|
||||
},
|
||||
@@ -569,8 +569,8 @@ func TestStatementBuilder(t *testing.T) {
|
||||
},
|
||||
},
|
||||
expected: qbtypes.Statement{
|
||||
Query: "WITH __temporal_aggregation_cte AS (SELECT ts, `__GROUP_BY_KEY_0_k8s.statefulset.name`, multiIf(row_number() OVER rate_window = 1, nan, (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) < 0, per_series_value / (ts - lagInFrame(ts, 1) OVER rate_window), (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) / (ts - lagInFrame(ts, 1) OVER rate_window)) AS per_series_value FROM (SELECT fingerprint, toStartOfInterval(toDateTime(intDiv(unix_milli, 1000)), toIntervalSecond(30)) AS ts, `__GROUP_BY_KEY_0_k8s.statefulset.name`, max(value) AS per_series_value FROM signoz_metrics.distributed_samples_v4 AS points INNER JOIN (SELECT fingerprint, JSONExtractString(labels, 'k8s.statefulset.name') AS `__GROUP_BY_KEY_0_k8s.statefulset.name` FROM signoz_metrics.time_series_v4_6hrs WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli <= ? AND LOWER(temporality) LIKE LOWER(?) AND JSONExtractString(labels, 'k8s.statefulset.name') = ? GROUP BY fingerprint, `__GROUP_BY_KEY_0_k8s.statefulset.name`) AS filtered_time_series ON points.fingerprint = filtered_time_series.fingerprint WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli < ? GROUP BY fingerprint, ts, `__GROUP_BY_KEY_0_k8s.statefulset.name` ORDER BY fingerprint, ts) WINDOW rate_window AS (PARTITION BY fingerprint ORDER BY fingerprint, ts)), __spatial_aggregation_cte AS (SELECT ts, `__GROUP_BY_KEY_0_k8s.statefulset.name`, sum(per_series_value) AS value FROM __temporal_aggregation_cte WHERE isNaN(per_series_value) = ? GROUP BY ts, `__GROUP_BY_KEY_0_k8s.statefulset.name`) SELECT * FROM __spatial_aggregation_cte ORDER BY `__GROUP_BY_KEY_0_k8s.statefulset.name`, ts",
|
||||
Args: []any{"signoz_calls_total", uint64(1747936800000), uint64(1747983420000), "cumulative", "my-statefulset", "signoz_calls_total", uint64(1747947360000), uint64(1747983420000), 0},
|
||||
Query: "WITH __temporal_aggregation_cte AS (SELECT * FROM (SELECT ts, `__GROUP_BY_KEY_0_k8s.statefulset.name`, multiIf((row_number() OVER rate_window = 1 OR (ts - lagInFrame(ts, 1) OVER rate_window) > 300), nan, (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) < 0, per_series_value / (ts - lagInFrame(ts, 1) OVER rate_window), (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) / (ts - lagInFrame(ts, 1) OVER rate_window)) AS per_series_value FROM (SELECT fingerprint, toStartOfInterval(toDateTime(intDiv(unix_milli, 1000)), toIntervalSecond(30)) AS ts, `__GROUP_BY_KEY_0_k8s.statefulset.name`, max(value) AS per_series_value FROM signoz_metrics.distributed_samples_v4 AS points INNER JOIN (SELECT fingerprint, JSONExtractString(labels, 'k8s.statefulset.name') AS `__GROUP_BY_KEY_0_k8s.statefulset.name` FROM signoz_metrics.time_series_v4_6hrs WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli <= ? AND LOWER(temporality) LIKE LOWER(?) AND JSONExtractString(labels, 'k8s.statefulset.name') = ? GROUP BY fingerprint, `__GROUP_BY_KEY_0_k8s.statefulset.name`) AS filtered_time_series ON points.fingerprint = filtered_time_series.fingerprint WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli < ? GROUP BY fingerprint, ts, `__GROUP_BY_KEY_0_k8s.statefulset.name` ORDER BY fingerprint, ts) WINDOW rate_window AS (PARTITION BY fingerprint ORDER BY fingerprint, ts)) WHERE ts >= toDateTime(1747947390)), __spatial_aggregation_cte AS (SELECT ts, `__GROUP_BY_KEY_0_k8s.statefulset.name`, sum(per_series_value) AS value FROM __temporal_aggregation_cte WHERE isNaN(per_series_value) = ? GROUP BY ts, `__GROUP_BY_KEY_0_k8s.statefulset.name`) SELECT * FROM __spatial_aggregation_cte ORDER BY `__GROUP_BY_KEY_0_k8s.statefulset.name`, ts",
|
||||
Args: []any{"signoz_calls_total", uint64(1747936800000), uint64(1747983420000), "cumulative", "my-statefulset", "signoz_calls_total", uint64(1747947090000), uint64(1747983420000), 0},
|
||||
Warnings: []string{"key `k8s.statefulset.name` not found in metadata; querying the underlying data directly. If this is unexpected, check the key name for typos."},
|
||||
},
|
||||
expectedErr: nil,
|
||||
@@ -602,8 +602,8 @@ func TestStatementBuilder(t *testing.T) {
|
||||
},
|
||||
},
|
||||
expected: qbtypes.Statement{
|
||||
Query: "WITH __temporal_aggregation_cte AS (SELECT ts, `__GROUP_BY_KEY_0_service.name`, multiIf(row_number() OVER rate_window = 1, nan, (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) < 0, per_series_value / (ts - lagInFrame(ts, 1) OVER rate_window), (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) / (ts - lagInFrame(ts, 1) OVER rate_window)) AS per_series_value FROM (SELECT fingerprint, toStartOfInterval(toDateTime(intDiv(unix_milli, 1000)), toIntervalSecond(30)) AS ts, `__GROUP_BY_KEY_0_service.name`, max(value) AS per_series_value FROM signoz_metrics.distributed_samples_v4 AS points INNER JOIN (SELECT fingerprint, JSONExtractString(labels, 'service.name') AS `__GROUP_BY_KEY_0_service.name` FROM signoz_metrics.time_series_v4_6hrs WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli <= ? AND LOWER(temporality) LIKE LOWER(?) AND accurateCastOrNull(JSONExtractString(labels, 'success'), 'Bool') = ? GROUP BY fingerprint, `__GROUP_BY_KEY_0_service.name`) AS filtered_time_series ON points.fingerprint = filtered_time_series.fingerprint WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli < ? GROUP BY fingerprint, ts, `__GROUP_BY_KEY_0_service.name` ORDER BY fingerprint, ts) WINDOW rate_window AS (PARTITION BY fingerprint ORDER BY fingerprint, ts)), __spatial_aggregation_cte AS (SELECT ts, `__GROUP_BY_KEY_0_service.name`, sum(per_series_value) AS value FROM __temporal_aggregation_cte WHERE isNaN(per_series_value) = ? GROUP BY ts, `__GROUP_BY_KEY_0_service.name`) SELECT * FROM __spatial_aggregation_cte ORDER BY `__GROUP_BY_KEY_0_service.name`, ts",
|
||||
Args: []any{"signoz_calls_total", uint64(1747936800000), uint64(1747983420000), "cumulative", true, "signoz_calls_total", uint64(1747947360000), uint64(1747983420000), 0},
|
||||
Query: "WITH __temporal_aggregation_cte AS (SELECT * FROM (SELECT ts, `__GROUP_BY_KEY_0_service.name`, multiIf((row_number() OVER rate_window = 1 OR (ts - lagInFrame(ts, 1) OVER rate_window) > 300), nan, (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) < 0, per_series_value / (ts - lagInFrame(ts, 1) OVER rate_window), (per_series_value - lagInFrame(per_series_value, 1) OVER rate_window) / (ts - lagInFrame(ts, 1) OVER rate_window)) AS per_series_value FROM (SELECT fingerprint, toStartOfInterval(toDateTime(intDiv(unix_milli, 1000)), toIntervalSecond(30)) AS ts, `__GROUP_BY_KEY_0_service.name`, max(value) AS per_series_value FROM signoz_metrics.distributed_samples_v4 AS points INNER JOIN (SELECT fingerprint, JSONExtractString(labels, 'service.name') AS `__GROUP_BY_KEY_0_service.name` FROM signoz_metrics.time_series_v4_6hrs WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli <= ? AND LOWER(temporality) LIKE LOWER(?) AND accurateCastOrNull(JSONExtractString(labels, 'success'), 'Bool') = ? GROUP BY fingerprint, `__GROUP_BY_KEY_0_service.name`) AS filtered_time_series ON points.fingerprint = filtered_time_series.fingerprint WHERE metric_name IN (?) AND unix_milli >= ? AND unix_milli < ? GROUP BY fingerprint, ts, `__GROUP_BY_KEY_0_service.name` ORDER BY fingerprint, ts) WINDOW rate_window AS (PARTITION BY fingerprint ORDER BY fingerprint, ts)) WHERE ts >= toDateTime(1747947390)), __spatial_aggregation_cte AS (SELECT ts, `__GROUP_BY_KEY_0_service.name`, sum(per_series_value) AS value FROM __temporal_aggregation_cte WHERE isNaN(per_series_value) = ? GROUP BY ts, `__GROUP_BY_KEY_0_service.name`) SELECT * FROM __spatial_aggregation_cte ORDER BY `__GROUP_BY_KEY_0_service.name`, ts",
|
||||
Args: []any{"signoz_calls_total", uint64(1747936800000), uint64(1747983420000), "cumulative", true, "signoz_calls_total", uint64(1747947090000), uint64(1747983420000), 0},
|
||||
},
|
||||
expectedErr: nil,
|
||||
},
|
||||
|
||||
13
pkg/telemetryschema/metertelemetryschema/table_hints.go
Normal file
13
pkg/telemetryschema/metertelemetryschema/table_hints.go
Normal file
@@ -0,0 +1,13 @@
|
||||
package metertelemetryschema
|
||||
|
||||
import "github.com/SigNoz/signoz/pkg/types/metrictypes"
|
||||
|
||||
// TableHintsForWindow pins the samples table the builder picks for
|
||||
// [start, end), so a statement over a piece of that window reads the same
|
||||
// table.
|
||||
func TableHintsForWindow(start, end uint64, metricType metrictypes.Type, timeAggregation metrictypes.TimeAggregation, tableHints *metrictypes.MetricTableHints) *metrictypes.MetricTableHints {
|
||||
if tableHints != nil {
|
||||
return tableHints
|
||||
}
|
||||
return &metrictypes.MetricTableHints{SamplesTableName: WhichSamplesTableToUse(start, end, metricType, timeAggregation, nil)}
|
||||
}
|
||||
31
pkg/telemetryschema/metricstelemetryschema/table_hints.go
Normal file
31
pkg/telemetryschema/metricstelemetryschema/table_hints.go
Normal file
@@ -0,0 +1,31 @@
|
||||
package metricstelemetryschema
|
||||
|
||||
import (
|
||||
"time"
|
||||
|
||||
"github.com/SigNoz/signoz/pkg/types/metrictypes"
|
||||
)
|
||||
|
||||
// UsesBuffer reports whether a reduced metric reads the raw buffer, which
|
||||
// holds the recent short window. Table hints pin the tables instead, so a
|
||||
// hinted aggregation never switches to the buffer.
|
||||
func UsesBuffer(start, end uint64, reduced bool, tableHints *metrictypes.MetricTableHints) bool {
|
||||
return tableHints == nil && reduced &&
|
||||
end-start < OneDayInMilliseconds &&
|
||||
start >= uint64(time.Now().UnixMilli())-OneDayInMilliseconds
|
||||
}
|
||||
|
||||
// TableHintsForWindow pins the tables the builder picks for [start, end), so
|
||||
// a statement over a piece of that window reads the same tables. Nil when
|
||||
// the window reads the buffer: every piece of such a window reads it too.
|
||||
func TableHintsForWindow(start, end uint64, metricType metrictypes.Type, timeAggregation metrictypes.TimeAggregation, reduced bool, tableHints *metrictypes.MetricTableHints) *metrictypes.MetricTableHints {
|
||||
if tableHints != nil {
|
||||
return tableHints
|
||||
}
|
||||
if UsesBuffer(start, end, reduced, nil) {
|
||||
return nil
|
||||
}
|
||||
samplesTable, _ := WhichSamplesTableToUse(start, end, metricType, timeAggregation, false, nil)
|
||||
_, _, _, tsLocalTable := WhichTSTableToUse(start, end, false, nil)
|
||||
return &metrictypes.MetricTableHints{SamplesTableName: samplesTable, TimeSeriesTableName: tsLocalTable}
|
||||
}
|
||||
@@ -4,39 +4,61 @@ import (
|
||||
"bytes"
|
||||
"encoding/json"
|
||||
"maps"
|
||||
"slices"
|
||||
|
||||
"github.com/SigNoz/signoz/pkg/types/cachetypes"
|
||||
)
|
||||
|
||||
var _ cachetypes.Cacheable = (*CachedData)(nil)
|
||||
|
||||
// CachedBucketEdge says which part of a request window a bucket holds. A body
|
||||
// bucket holds whole steps and is valid for any window that contains its
|
||||
// range. An edge bucket holds the partial aggregate of one window end and is
|
||||
// valid only for a window with exactly that end.
|
||||
type CachedBucketEdge string
|
||||
|
||||
const (
|
||||
CachedBucketBody CachedBucketEdge = ""
|
||||
CachedBucketHead CachedBucketEdge = "head"
|
||||
CachedBucketTail CachedBucketEdge = "tail"
|
||||
CachedBucketWhole CachedBucketEdge = "whole"
|
||||
)
|
||||
|
||||
// CachedBucket holds the points of one query for [StartMs, EndMs) on the step
|
||||
// grid, and nothing outside it.
|
||||
type CachedBucket struct {
|
||||
StartMs uint64 `json:"startMs"`
|
||||
EndMs uint64 `json:"endMs"`
|
||||
Type RequestType `json:"type"`
|
||||
Value json.RawMessage `json:"value"`
|
||||
Stats ExecStats `json:"stats"`
|
||||
StartMs uint64 `json:"startMs"`
|
||||
EndMs uint64 `json:"endMs"`
|
||||
Edge CachedBucketEdge `json:"edge,omitempty"`
|
||||
// WrittenAtMs is when the oldest points of the bucket were fetched; a
|
||||
// bucket expires on its own clock, since the entry's TTL restarts on
|
||||
// every write.
|
||||
WrittenAtMs int64 `json:"writtenAtMs"`
|
||||
Type RequestType `json:"type"`
|
||||
Value json.RawMessage `json:"value"`
|
||||
Stats ExecStats `json:"stats"`
|
||||
Warnings []string `json:"warnings,omitempty"`
|
||||
WarningsDocURL string `json:"warningsDocURL,omitempty"`
|
||||
}
|
||||
|
||||
func (c *CachedBucket) Clone() *CachedBucket {
|
||||
return &CachedBucket{
|
||||
StartMs: c.StartMs,
|
||||
EndMs: c.EndMs,
|
||||
Type: c.Type,
|
||||
Value: bytes.Clone(c.Value),
|
||||
Stats: ExecStats{
|
||||
RowsScanned: c.Stats.RowsScanned,
|
||||
BytesScanned: c.Stats.BytesScanned,
|
||||
DurationMS: c.Stats.DurationMS,
|
||||
StepIntervals: maps.Clone(c.Stats.StepIntervals),
|
||||
},
|
||||
StartMs: c.StartMs,
|
||||
EndMs: c.EndMs,
|
||||
Edge: c.Edge,
|
||||
WrittenAtMs: c.WrittenAtMs,
|
||||
Type: c.Type,
|
||||
Value: bytes.Clone(c.Value),
|
||||
Stats: c.Stats.Clone(),
|
||||
Warnings: slices.Clone(c.Warnings),
|
||||
WarningsDocURL: c.WarningsDocURL,
|
||||
}
|
||||
}
|
||||
|
||||
// CachedData represents the full cached data for a query.
|
||||
// CachedData is the cache entry of one query: body buckets are disjoint and
|
||||
// sorted by start, edge buckets follow.
|
||||
type CachedData struct {
|
||||
Buckets []*CachedBucket `json:"buckets"`
|
||||
Warnings []string `json:"warnings"`
|
||||
Buckets []*CachedBucket `json:"buckets"`
|
||||
}
|
||||
|
||||
func (c *CachedData) UnmarshalBinary(data []byte) error {
|
||||
@@ -48,16 +70,14 @@ func (c *CachedData) MarshalBinary() ([]byte, error) {
|
||||
}
|
||||
|
||||
func (c *CachedData) Clone() cachetypes.Cacheable {
|
||||
clonedCachedData := new(CachedData)
|
||||
clonedCachedData.Buckets = make([]*CachedBucket, len(c.Buckets))
|
||||
for i := range c.Buckets {
|
||||
clonedCachedData.Buckets[i] = c.Buckets[i].Clone()
|
||||
cloned := &CachedData{Buckets: make([]*CachedBucket, 0, len(c.Buckets))}
|
||||
for _, bucket := range c.Buckets {
|
||||
if bucket == nil {
|
||||
continue
|
||||
}
|
||||
cloned.Buckets = append(cloned.Buckets, bucket.Clone())
|
||||
}
|
||||
|
||||
clonedCachedData.Warnings = make([]string, len(c.Warnings))
|
||||
copy(clonedCachedData.Warnings, c.Warnings)
|
||||
|
||||
return clonedCachedData
|
||||
return cloned
|
||||
}
|
||||
|
||||
// Cost approximates the retained bytes of this CachedData for use as the
|
||||
@@ -69,11 +89,17 @@ func (c *CachedData) Cost() int64 {
|
||||
if b == nil {
|
||||
continue
|
||||
}
|
||||
// Value is the bulk of the payload
|
||||
size += int64(len(b.Value))
|
||||
}
|
||||
for _, w := range c.Warnings {
|
||||
size += int64(len(w))
|
||||
for _, w := range b.Warnings {
|
||||
size += int64(len(w))
|
||||
}
|
||||
}
|
||||
return size
|
||||
}
|
||||
|
||||
// Clone returns a deep copy; StepIntervals is the only reference field.
|
||||
func (e ExecStats) Clone() ExecStats {
|
||||
cloned := e
|
||||
cloned.StepIntervals = maps.Clone(e.StepIntervals)
|
||||
return cloned
|
||||
}
|
||||
|
||||
@@ -388,6 +388,12 @@ type QueryRangeRequest struct {
|
||||
// NoCache is a flag to disable caching for the request.
|
||||
NoCache bool `json:"noCache,omitempty"`
|
||||
|
||||
// NoStepAlignment evaluates PromQL queries at the request's own start and
|
||||
// end instead of moving both down to the step grid. Every client gets the
|
||||
// grid by default so that windows a fraction of a step apart evaluate the
|
||||
// same instants; a query kept off the grid is not cached.
|
||||
NoStepAlignment bool `json:"noStepAlignment,omitempty"`
|
||||
|
||||
FormatOptions *FormatOptions `json:"formatOptions,omitempty"`
|
||||
}
|
||||
|
||||
|
||||
@@ -189,6 +189,50 @@ func (a *AggregationBucket) ReindexValuesToNewUpperBounds(onto []float64) {
|
||||
a.Meta.Buckets = onto
|
||||
}
|
||||
|
||||
// TrimAxisToCountedBuckets drops the buckets at either end of Meta.Buckets that hold
|
||||
// no counts, since an axis runs from the lowest value in the window to the highest.
|
||||
// Not for a query that chose its own buckets: an empty `le` is still one it reported.
|
||||
func (a *AggregationBucket) TrimAxisToCountedBuckets() {
|
||||
if a == nil || len(a.Meta.Buckets) == 0 {
|
||||
return
|
||||
}
|
||||
|
||||
lowestCounted, highestCounted := len(a.Meta.Buckets), -1
|
||||
for _, series := range a.Series {
|
||||
for _, point := range series.Values {
|
||||
for slot := 0; slot < len(a.Meta.Buckets) && slot < len(point.Values); slot++ {
|
||||
if point.Values[slot] != 0 {
|
||||
lowestCounted = min(lowestCounted, slot)
|
||||
highestCounted = max(highestCounted, slot)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
if highestCounted < 0 {
|
||||
return
|
||||
}
|
||||
|
||||
if lowestCounted == 0 && highestCounted == len(a.Meta.Buckets)-1 {
|
||||
return
|
||||
}
|
||||
|
||||
trimmed := a.Meta.Buckets[lowestCounted : highestCounted+1]
|
||||
for _, series := range a.Series {
|
||||
for _, point := range series.Values {
|
||||
if len(point.Values) == 0 {
|
||||
continue
|
||||
}
|
||||
counts := make([]float64, len(trimmed)+1)
|
||||
for slot, count := range point.Values {
|
||||
counts[min(max(slot-lowestCounted, 0), len(trimmed))] += count
|
||||
}
|
||||
point.Values = counts
|
||||
}
|
||||
}
|
||||
|
||||
a.Meta.Buckets = trimmed
|
||||
}
|
||||
|
||||
type AggregationMeta struct {
|
||||
Unit string `json:"unit,omitempty"`
|
||||
// Buckets holds ascending upper bounds shared by every series in the
|
||||
@@ -247,8 +291,10 @@ func GetUniqueSeriesKey(labels []*Label) string {
|
||||
if len(labels) == 0 {
|
||||
return ""
|
||||
}
|
||||
// Values are quoted so that a value holding "=" or "," cannot read as
|
||||
// another label set.
|
||||
if len(labels) == 1 {
|
||||
return fmt.Sprintf("%s=%v,", labels[0].Key.Name, labels[0].Value)
|
||||
return labels[0].Key.Name + "=" + strconv.Quote(labelValueString(labels[0].Value)) + ","
|
||||
}
|
||||
|
||||
// Use a map to collect labels for consistent ordering without copying
|
||||
@@ -260,13 +306,9 @@ func GetUniqueSeriesKey(labels []*Label) string {
|
||||
for _, label := range labels {
|
||||
if _, exists := labelMap[label.Key.Name]; !exists {
|
||||
keys = append(keys, label.Key.Name)
|
||||
estimatedSize += len(label.Key.Name) + 2 // key + '=' + ','
|
||||
}
|
||||
// get the value as string
|
||||
value, ok := label.Value.(string)
|
||||
if !ok {
|
||||
value = fmt.Sprintf("%v", label.Value)
|
||||
estimatedSize += len(label.Key.Name) + 4 // key + '=' + quotes + ','
|
||||
}
|
||||
value := labelValueString(label.Value)
|
||||
estimatedSize += len(value)
|
||||
|
||||
labelMap[label.Key.Name] = value
|
||||
@@ -282,13 +324,21 @@ func GetUniqueSeriesKey(labels []*Label) string {
|
||||
for _, k := range keys {
|
||||
key.WriteString(k)
|
||||
key.WriteByte('=')
|
||||
key.WriteString(labelMap[k])
|
||||
key.WriteString(strconv.Quote(labelMap[k]))
|
||||
key.WriteByte(',')
|
||||
}
|
||||
|
||||
return key.String()
|
||||
}
|
||||
|
||||
// labelValueString is the value as a string, as the response prints it.
|
||||
func labelValueString(value any) string {
|
||||
if s, ok := value.(string); ok {
|
||||
return s
|
||||
}
|
||||
return fmt.Sprintf("%v", value)
|
||||
}
|
||||
|
||||
type TimeSeriesValue struct {
|
||||
Timestamp int64 `json:"timestamp"`
|
||||
Value float64 `json:"value"`
|
||||
|
||||
@@ -111,6 +111,13 @@ def pytest_addoption(parser: pytest.Parser):
|
||||
default="25.12.5",
|
||||
help="clickhouse version",
|
||||
)
|
||||
parser.addoption(
|
||||
"--cache-fuzz-seed",
|
||||
action="store",
|
||||
type=int,
|
||||
default=20260910,
|
||||
help="seed for the randomised cache differential test (integration/tests/queriercache/02_differential.py)",
|
||||
)
|
||||
parser.addoption(
|
||||
"--schema-migrator-version",
|
||||
action="store",
|
||||
|
||||
24
tests/fixtures/logs.py
vendored
24
tests/fixtures/logs.py
vendored
@@ -714,3 +714,27 @@ def remove_logs_ttl_settings(signoz: types.SigNoz):
|
||||
signoz.telemetrystore.conn.query(alter_query)
|
||||
except Exception as e: # pylint: disable=broad-exception-caught
|
||||
print(f"Error removing TTL from table {table}: {e}")
|
||||
|
||||
|
||||
def minutely_logs(
|
||||
start: datetime.datetime,
|
||||
end: datetime.datetime,
|
||||
per_minute: dict[str, int],
|
||||
attributes: dict[str, Any] | None = None,
|
||||
) -> list[Logs]:
|
||||
"""per_minute[service] logs with resource service.name=service in every whole minute of [start, end), one second apart from the minute start."""
|
||||
logs = []
|
||||
minute = start
|
||||
while minute < end:
|
||||
for service, count in per_minute.items():
|
||||
for i in range(count):
|
||||
logs.append(
|
||||
Logs(
|
||||
timestamp=minute + datetime.timedelta(seconds=1 + i),
|
||||
resources={"service.name": service},
|
||||
attributes=attributes or {},
|
||||
body=f"{service} {minute.isoformat()} {i}",
|
||||
)
|
||||
)
|
||||
minute += datetime.timedelta(minutes=1)
|
||||
return logs
|
||||
|
||||
38
tests/fixtures/querier.py
vendored
38
tests/fixtures/querier.py
vendored
@@ -175,6 +175,7 @@ def make_query_request(
|
||||
format_options: dict | None = None,
|
||||
variables: dict | None = None,
|
||||
no_cache: bool = True,
|
||||
no_step_alignment: bool = False,
|
||||
timeout: int = QUERY_TIMEOUT,
|
||||
) -> requests.Response:
|
||||
if format_options is None:
|
||||
@@ -191,6 +192,8 @@ def make_query_request(
|
||||
}
|
||||
if variables:
|
||||
payload["variables"] = variables
|
||||
if no_step_alignment:
|
||||
payload["noStepAlignment"] = True
|
||||
|
||||
return requests.post(
|
||||
signoz.self.host_configs["8080"].get("/api/v5/query_range"),
|
||||
@@ -1215,3 +1218,38 @@ def run_query_case(signoz: types.SigNoz, token: str, now: datetime, case: dict[s
|
||||
)
|
||||
assert response.status_code == 200, f"HTTP {response.status_code} for case '{case['name']}': {response.text}"
|
||||
assert case["validate"](response), f"Validation failed for case '{case['name']}': {response.json()}"
|
||||
|
||||
|
||||
def series_points_by_label(response_json: dict, query_name: str, label: str = "service.name") -> dict[str, dict[int, float]]:
|
||||
"""Points of every series of the named result keyed by the series' value for `label`, as {timestamp_ms: value}."""
|
||||
by_label = index_series_by_label(get_all_series(response_json, query_name), label)
|
||||
return {value: {point["timestamp"]: point["value"] for point in series["values"]} for value, series in by_label.items()}
|
||||
|
||||
|
||||
def assert_series_points_equal(
|
||||
response_json: dict,
|
||||
expected_json: dict,
|
||||
query_name: str,
|
||||
context: str,
|
||||
label: str = "service.name",
|
||||
) -> None:
|
||||
"""assert_all_series_equal with a failure message that names the series and points that differ."""
|
||||
got = series_points_by_label(response_json, query_name, label)
|
||||
want = series_points_by_label(expected_json, query_name, label)
|
||||
problems = []
|
||||
if set(got) != set(want):
|
||||
problems.append(f"series: got={sorted(got)} expected={sorted(want)}")
|
||||
for value in sorted(set(got) & set(want)):
|
||||
got_points, want_points = got[value], want[value]
|
||||
missing = sorted(set(want_points) - set(got_points))
|
||||
extra = sorted(set(got_points) - set(want_points))
|
||||
changed = sorted(ts for ts in set(got_points) & set(want_points) if not compare_values(got_points[ts], want_points[ts]))
|
||||
if missing:
|
||||
problems.append(f"{value}: {len(missing)} of {len(want_points)} expected points missing, first at {datetime.fromtimestamp(missing[0] / 1000, tz=UTC).isoformat()}")
|
||||
if extra:
|
||||
problems.append(f"{value}: {len(extra)} unexpected points, first at {datetime.fromtimestamp(extra[0] / 1000, tz=UTC).isoformat()}")
|
||||
if changed:
|
||||
first = changed[0]
|
||||
problems.append(f"{value}: {len(changed)} values differ, first at {datetime.fromtimestamp(first / 1000, tz=UTC).isoformat()}: got={got_points[first]} expected={want_points[first]}")
|
||||
assert not problems, f"{context}: response differs from the expected response:\n " + "\n ".join(problems)
|
||||
assert_all_series_equal(response_json, expected_json, query_name, context)
|
||||
|
||||
7
tests/fixtures/time.py
vendored
7
tests/fixtures/time.py
vendored
@@ -1,4 +1,5 @@
|
||||
import datetime
|
||||
import time
|
||||
from typing import Any
|
||||
|
||||
import isodate
|
||||
@@ -19,3 +20,9 @@ def parse_duration(duration: Any) -> datetime.timedelta:
|
||||
if isinstance(duration, datetime.timedelta):
|
||||
return duration
|
||||
return datetime.timedelta(seconds=duration)
|
||||
|
||||
|
||||
def wait_until_second_of_minute(low: int, high: int) -> None:
|
||||
"""Block until the wall-clock second is within [low, high]."""
|
||||
while not low <= datetime.datetime.now(tz=datetime.UTC).second <= high:
|
||||
time.sleep(1)
|
||||
|
||||
@@ -55,7 +55,8 @@ def test_upstream_promqltest_corpus(
|
||||
}
|
||||
case_id = f"{case['source']}[{case['variant']}]"
|
||||
|
||||
response = make_query_request(signoz, token, req_start_ms, end_ms, [query])
|
||||
# the expected values were computed at the case's own instants
|
||||
response = make_query_request(signoz, token, req_start_ms, end_ms, [query], no_step_alignment=True)
|
||||
if response.status_code != HTTPStatus.OK:
|
||||
failures.append(f"{case_id}: HTTP {response.status_code} for {case['expr']!r}: {response.text[:200]}")
|
||||
continue
|
||||
|
||||
@@ -58,8 +58,7 @@ def test_promql_ratio_with_zero_denominator_is_dropped_and_cached(
|
||||
assert set(first["active_job"].values()) == {25.0}, sorted(set(first["active_job"].values()))
|
||||
assert len(first["active_job"]) == expected_points, f"expected {expected_points} points, got {len(first['active_job'])}"
|
||||
|
||||
# The cached read excludes end_ms, the one legitimate difference.
|
||||
# Both reads must agree exactly, including the point promql reports at end_ms.
|
||||
assert set(second) == set(first), sorted(second)
|
||||
for job_name, points in first.items():
|
||||
expected = {ts: value for ts, value in points.items() if ts < end_ms}
|
||||
assert second[job_name] == expected, f"{job_name}: got {len(second[job_name])} of {len(expected)} points"
|
||||
assert second[job_name] == points, f"{job_name}: got {len(second[job_name])} of {len(points)} points"
|
||||
|
||||
675
tests/integration/tests/queriercache/01_cache_consistency.py
Normal file
675
tests/integration/tests/queriercache/01_cache_consistency.py
Normal file
@@ -0,0 +1,675 @@
|
||||
from collections.abc import Callable
|
||||
from datetime import UTC, datetime, timedelta
|
||||
from http import HTTPStatus
|
||||
from uuid import uuid4
|
||||
|
||||
from fixtures import types
|
||||
from fixtures.auth import USER_ADMIN_EMAIL, USER_ADMIN_PASSWORD
|
||||
from fixtures.logs import Logs, minutely_logs
|
||||
from fixtures.metrics import Metrics
|
||||
from fixtures.querier import (
|
||||
assert_all_series_equal,
|
||||
assert_series_points_equal,
|
||||
build_aggregation,
|
||||
build_builder_query,
|
||||
build_formula_query,
|
||||
build_function,
|
||||
build_group_by_field,
|
||||
build_order_by,
|
||||
build_scalar_query,
|
||||
find_named_result,
|
||||
get_all_series,
|
||||
make_query_request,
|
||||
series_points_by_label,
|
||||
)
|
||||
from fixtures.time import wait_until_second_of_minute
|
||||
|
||||
STEP = 60
|
||||
MINUTE = timedelta(minutes=1)
|
||||
|
||||
|
||||
def test_full_hit_matches_no_cache(
|
||||
signoz: types.SigNoz,
|
||||
create_user_admin: None, # pylint: disable=unused-argument
|
||||
get_token: Callable[[str, str], str],
|
||||
insert_logs: Callable[[list[Logs]], None],
|
||||
) -> None:
|
||||
"""
|
||||
Setup:
|
||||
Two services with steady per-minute counts in a window that ended more
|
||||
than the flux interval (5m) ago, so the first request fills the cache.
|
||||
|
||||
Tests:
|
||||
The same window requested twice through the cache (miss, then full hit)
|
||||
equals the request with noCache. Baseline for the other tests.
|
||||
"""
|
||||
now = datetime.now(tz=UTC).replace(second=0, microsecond=0)
|
||||
run = uuid4().hex[:8]
|
||||
start, end = now - 40 * MINUTE, now - 20 * MINUTE
|
||||
insert_logs(minutely_logs(start, end, {"svc-a": 3, "svc-b": 1}, attributes={"run": run}))
|
||||
token = get_token(USER_ADMIN_EMAIL, USER_ADMIN_PASSWORD)
|
||||
query = build_scalar_query(
|
||||
name="A",
|
||||
signal="logs",
|
||||
aggregations=[build_aggregation("count()")],
|
||||
group_by=[build_group_by_field("service.name", "string", "resource")],
|
||||
filter_expression=f"run = '{run}'",
|
||||
step_interval=STEP,
|
||||
)
|
||||
start_ms, end_ms = int(start.timestamp() * 1000), int(end.timestamp() * 1000)
|
||||
|
||||
warm = make_query_request(signoz, token, start_ms, end_ms, [query], no_cache=False)
|
||||
assert warm.status_code == HTTPStatus.OK, warm.text
|
||||
cached = make_query_request(signoz, token, start_ms, end_ms, [query], no_cache=False)
|
||||
assert cached.status_code == HTTPStatus.OK, cached.text
|
||||
fresh = make_query_request(signoz, token, start_ms, end_ms, [query], no_cache=True)
|
||||
assert fresh.status_code == HTTPStatus.OK, fresh.text
|
||||
|
||||
assert len(get_all_series(fresh.json(), "A")) == 2, fresh.text
|
||||
assert_series_points_equal(cached.json(), fresh.json(), "A", "full cache hit")
|
||||
|
||||
|
||||
def test_full_hit_keeps_aggregation_alias(
|
||||
signoz: types.SigNoz,
|
||||
create_user_admin: None, # pylint: disable=unused-argument
|
||||
get_token: Callable[[str, str], str],
|
||||
insert_logs: Callable[[list[Logs]], None],
|
||||
) -> None:
|
||||
"""
|
||||
Setup:
|
||||
One service with logs in a fully cacheable window.
|
||||
|
||||
Tests:
|
||||
A response served entirely from cache carries the same aggregation alias
|
||||
and index as the response that filled the cache.
|
||||
"""
|
||||
now = datetime.now(tz=UTC).replace(second=0, microsecond=0)
|
||||
run = uuid4().hex[:8]
|
||||
start, end = now - 30 * MINUTE, now - 20 * MINUTE
|
||||
insert_logs(minutely_logs(start, end, {"svc-a": 2}, attributes={"run": run}))
|
||||
token = get_token(USER_ADMIN_EMAIL, USER_ADMIN_PASSWORD)
|
||||
query = build_scalar_query(
|
||||
name="A",
|
||||
signal="logs",
|
||||
aggregations=[build_aggregation("count()", "total")],
|
||||
group_by=[build_group_by_field("service.name", "string", "resource")],
|
||||
filter_expression=f"run = '{run}'",
|
||||
step_interval=STEP,
|
||||
)
|
||||
start_ms, end_ms = int(start.timestamp() * 1000), int(end.timestamp() * 1000)
|
||||
|
||||
first = make_query_request(signoz, token, start_ms, end_ms, [query], no_cache=False)
|
||||
assert first.status_code == HTTPStatus.OK, first.text
|
||||
second = make_query_request(signoz, token, start_ms, end_ms, [query], no_cache=False)
|
||||
assert second.status_code == HTTPStatus.OK, second.text
|
||||
|
||||
first_agg = find_named_result(first.json()["data"]["data"]["results"], "A")["aggregations"][0]
|
||||
second_agg = find_named_result(second.json()["data"]["data"]["results"], "A")["aggregations"][0]
|
||||
assert first_agg.get("alias"), first_agg
|
||||
assert second_agg.get("alias") == first_agg.get("alias"), f"cache hit changed the alias: first={first_agg.get('alias')!r} second={second_agg.get('alias')!r}"
|
||||
assert second_agg.get("index") == first_agg.get("index")
|
||||
|
||||
|
||||
def test_limited_group_by_partial_hit_matches_no_cache(
|
||||
signoz: types.SigNoz,
|
||||
create_user_admin: None, # pylint: disable=unused-argument
|
||||
get_token: Callable[[str, str], str],
|
||||
insert_logs: Callable[[list[Logs]], None],
|
||||
) -> None:
|
||||
"""
|
||||
Setup:
|
||||
Four services. svc-a and svc-b dominate the first 20 minutes, svc-c and
|
||||
svc-d the next 20. Over the full 40 minutes the top two are svc-a and
|
||||
svc-c.
|
||||
|
||||
Tests:
|
||||
A top-2 time series (limit=2, order by count() desc) whose first half is
|
||||
cached returns the same two series with points for the whole window as
|
||||
the request with noCache. The limit is a top-N over the requested
|
||||
window, not over each cached piece.
|
||||
"""
|
||||
now = datetime.now(tz=UTC).replace(second=0, microsecond=0)
|
||||
run = uuid4().hex[:8]
|
||||
head_start, head_end, tail_end = now - 60 * MINUTE, now - 40 * MINUTE, now - 20 * MINUTE
|
||||
insert_logs(minutely_logs(head_start, head_end, {"svc-a": 10, "svc-b": 8, "svc-c": 1, "svc-d": 1}, attributes={"run": run}))
|
||||
insert_logs(minutely_logs(head_end, tail_end, {"svc-a": 1, "svc-b": 1, "svc-c": 10, "svc-d": 8}, attributes={"run": run}))
|
||||
token = get_token(USER_ADMIN_EMAIL, USER_ADMIN_PASSWORD)
|
||||
query = build_scalar_query(
|
||||
name="A",
|
||||
signal="logs",
|
||||
aggregations=[build_aggregation("count()")],
|
||||
group_by=[build_group_by_field("service.name", "string", "resource")],
|
||||
order=[build_order_by("count()", "desc")],
|
||||
limit=2,
|
||||
filter_expression=f"run = '{run}'",
|
||||
step_interval=STEP,
|
||||
)
|
||||
head_start_ms, head_end_ms, tail_end_ms = (int(t.timestamp() * 1000) for t in (head_start, head_end, tail_end))
|
||||
|
||||
warm = make_query_request(signoz, token, head_start_ms, head_end_ms, [query], no_cache=False)
|
||||
assert warm.status_code == HTTPStatus.OK, warm.text
|
||||
cached = make_query_request(signoz, token, head_start_ms, tail_end_ms, [query], no_cache=False)
|
||||
assert cached.status_code == HTTPStatus.OK, cached.text
|
||||
fresh = make_query_request(signoz, token, head_start_ms, tail_end_ms, [query], no_cache=True)
|
||||
assert fresh.status_code == HTTPStatus.OK, fresh.text
|
||||
|
||||
fresh_points = series_points_by_label(fresh.json(), "A")
|
||||
assert set(fresh_points) == {"svc-a", "svc-c"}, fresh_points
|
||||
assert all(len(points) == 40 for points in fresh_points.values()), {s: len(p) for s, p in fresh_points.items()}
|
||||
assert_series_points_equal(cached.json(), fresh.json(), "A", "top-N after a partial cache hit")
|
||||
|
||||
|
||||
def test_time_shift_partial_hit_matches_no_cache(
|
||||
signoz: types.SigNoz,
|
||||
create_user_admin: None, # pylint: disable=unused-argument
|
||||
get_token: Callable[[str, str], str],
|
||||
insert_logs: Callable[[list[Logs]], None],
|
||||
) -> None:
|
||||
"""
|
||||
Setup:
|
||||
Logs one hour before the requested window, so a timeShift(3600) query
|
||||
reads them.
|
||||
|
||||
Tests:
|
||||
A timeShift query whose first 30 minutes are cached returns the same
|
||||
points for the next 20 minutes as the request with noCache. The missing
|
||||
range is already in the shifted clock; the ranged sub-query must not
|
||||
shift it again.
|
||||
"""
|
||||
now = datetime.now(tz=UTC).replace(second=0, microsecond=0)
|
||||
run = uuid4().hex[:8]
|
||||
insert_logs(minutely_logs(now - 120 * MINUTE, now - 60 * MINUTE, {"svc-a": 4, "svc-b": 2}, attributes={"run": run}))
|
||||
token = get_token(USER_ADMIN_EMAIL, USER_ADMIN_PASSWORD)
|
||||
query = build_scalar_query(
|
||||
name="A",
|
||||
signal="logs",
|
||||
aggregations=[build_aggregation("count()")],
|
||||
group_by=[build_group_by_field("service.name", "string", "resource")],
|
||||
filter_expression=f"run = '{run}'",
|
||||
step_interval=STEP,
|
||||
functions=[build_function("timeShift", 3600)],
|
||||
)
|
||||
start_ms = int((now - 60 * MINUTE).timestamp() * 1000)
|
||||
first_end_ms = int((now - 30 * MINUTE).timestamp() * 1000)
|
||||
second_end_ms = int((now - 10 * MINUTE).timestamp() * 1000)
|
||||
|
||||
warm = make_query_request(signoz, token, start_ms, first_end_ms, [query], no_cache=False)
|
||||
assert warm.status_code == HTTPStatus.OK, warm.text
|
||||
cached = make_query_request(signoz, token, start_ms, second_end_ms, [query], no_cache=False)
|
||||
assert cached.status_code == HTTPStatus.OK, cached.text
|
||||
fresh = make_query_request(signoz, token, start_ms, second_end_ms, [query], no_cache=True)
|
||||
assert fresh.status_code == HTTPStatus.OK, fresh.text
|
||||
|
||||
fresh_points = series_points_by_label(fresh.json(), "A")
|
||||
assert set(fresh_points) == {"svc-a", "svc-b"}, fresh_points
|
||||
assert all(len(points) == 50 for points in fresh_points.values()), {s: len(p) for s, p in fresh_points.items()}
|
||||
assert_series_points_equal(cached.json(), fresh.json(), "A", "timeShift after a partial cache hit")
|
||||
|
||||
|
||||
def test_promql_unaligned_start_partial_hit_matches_no_cache(
|
||||
signoz: types.SigNoz,
|
||||
create_user_admin: None, # pylint: disable=unused-argument
|
||||
get_token: Callable[[str, str], str],
|
||||
insert_metrics: Callable[[list[Metrics]], None],
|
||||
) -> None:
|
||||
"""
|
||||
Setup:
|
||||
A gauge with one sample per minute for two services.
|
||||
|
||||
Tests:
|
||||
Two PromQL requests whose starts sit at different offsets inside the
|
||||
step (17s, then 43s), as relative dashboard windows do. Both are moved
|
||||
onto the step grid, so the second request is a partial hit on the
|
||||
first one's entry, and its response equals the request with noCache.
|
||||
"""
|
||||
now = datetime.now(tz=UTC).replace(second=0, microsecond=0)
|
||||
metric = f"cache_gauge_{uuid4().hex[:8]}"
|
||||
insert_metrics(
|
||||
[
|
||||
Metrics(
|
||||
metric_name=metric,
|
||||
labels={"service": service},
|
||||
timestamp=now - minute * MINUTE,
|
||||
value=value,
|
||||
temporality="Unspecified",
|
||||
type_="Gauge",
|
||||
is_monotonic=False,
|
||||
)
|
||||
for minute in range(46, 8, -1)
|
||||
for service, value in (("svc-a", 10.0), ("svc-b", 20.0))
|
||||
]
|
||||
)
|
||||
token = get_token(USER_ADMIN_EMAIL, USER_ADMIN_PASSWORD)
|
||||
query = {"type": "promql", "spec": {"name": "A", "query": f"sum by (service) ({metric})", "step": STEP}}
|
||||
|
||||
first_start_ms = int((now - 40 * MINUTE + timedelta(seconds=17)).timestamp() * 1000)
|
||||
first_end_ms = int((now - 11 * MINUTE + timedelta(seconds=17)).timestamp() * 1000)
|
||||
second_start_ms = int((now - 40 * MINUTE + timedelta(seconds=43)).timestamp() * 1000)
|
||||
second_end_ms = int((now - 10 * MINUTE + timedelta(seconds=43)).timestamp() * 1000)
|
||||
|
||||
warm = make_query_request(signoz, token, first_start_ms, first_end_ms, [query], no_cache=False)
|
||||
assert warm.status_code == HTTPStatus.OK, warm.text
|
||||
cached = make_query_request(signoz, token, second_start_ms, second_end_ms, [query], no_cache=False)
|
||||
assert cached.status_code == HTTPStatus.OK, cached.text
|
||||
fresh = make_query_request(signoz, token, second_start_ms, second_end_ms, [query], no_cache=True)
|
||||
assert fresh.status_code == HTTPStatus.OK, fresh.text
|
||||
|
||||
fresh_points = series_points_by_label(fresh.json(), "A", label="service")
|
||||
assert set(fresh_points) == {"svc-a", "svc-b"}, fresh_points
|
||||
phases = {ts % (STEP * 1000) for points in fresh_points.values() for ts in points}
|
||||
assert phases == {0}, phases
|
||||
assert_series_points_equal(cached.json(), fresh.json(), "A", "PromQL after a partial cache hit with an unaligned start", label="service")
|
||||
|
||||
|
||||
def test_flux_boundary_interval_is_refreshed_with_late_data(
|
||||
signoz: types.SigNoz,
|
||||
create_user_admin: None, # pylint: disable=unused-argument
|
||||
get_token: Callable[[str, str], str],
|
||||
insert_logs: Callable[[list[Logs]], None],
|
||||
) -> None:
|
||||
"""
|
||||
Setup:
|
||||
Logs up to and into the minute T that contains the flux boundary
|
||||
(now - 5m). After the first request, more logs arrive inside T after the
|
||||
boundary, as an export that lags by a few minutes does.
|
||||
|
||||
Tests:
|
||||
T is still filling, so it must not be served from cache. The second
|
||||
request reports the new count for T, as the request with noCache does.
|
||||
"""
|
||||
wait_until_second_of_minute(12, 40)
|
||||
now = datetime.now(tz=UTC)
|
||||
run = uuid4().hex[:8]
|
||||
interval_start = (now - 5 * MINUTE).replace(second=0, microsecond=0)
|
||||
window_start = interval_start - 10 * MINUTE
|
||||
insert_logs(minutely_logs(window_start, interval_start, {"svc-a": 2}, attributes={"run": run}))
|
||||
insert_logs([Logs(timestamp=interval_start + timedelta(seconds=2 + i), resources={"service.name": "svc-a"}, attributes={"run": run}, body=f"early {i}") for i in range(3)])
|
||||
token = get_token(USER_ADMIN_EMAIL, USER_ADMIN_PASSWORD)
|
||||
query = build_scalar_query(
|
||||
name="A",
|
||||
signal="logs",
|
||||
aggregations=[build_aggregation("count()")],
|
||||
group_by=[build_group_by_field("service.name", "string", "resource")],
|
||||
filter_expression=f"run = '{run}'",
|
||||
step_interval=STEP,
|
||||
)
|
||||
window_start_ms = int(window_start.timestamp() * 1000)
|
||||
|
||||
warm = make_query_request(signoz, token, window_start_ms, int(datetime.now(tz=UTC).timestamp() * 1000), [query], no_cache=False)
|
||||
assert warm.status_code == HTTPStatus.OK, warm.text
|
||||
|
||||
insert_logs([Logs(timestamp=interval_start + timedelta(seconds=50, milliseconds=i), resources={"service.name": "svc-a"}, attributes={"run": run}, body=f"late {i}") for i in range(4)])
|
||||
|
||||
later_ms = int(datetime.now(tz=UTC).timestamp() * 1000)
|
||||
cached = make_query_request(signoz, token, window_start_ms, later_ms, [query], no_cache=False)
|
||||
assert cached.status_code == HTTPStatus.OK, cached.text
|
||||
fresh = make_query_request(signoz, token, window_start_ms, later_ms, [query], no_cache=True)
|
||||
assert fresh.status_code == HTTPStatus.OK, fresh.text
|
||||
|
||||
interval_ms = int(interval_start.timestamp() * 1000)
|
||||
fresh_points = series_points_by_label(fresh.json(), "A")["svc-a"]
|
||||
assert fresh_points[interval_ms] == 7, fresh_points
|
||||
cached_points = series_points_by_label(cached.json(), "A")["svc-a"]
|
||||
assert cached_points[interval_ms] == fresh_points[interval_ms], f"cache served the stale count {cached_points[interval_ms]} for the interval that contains the flux boundary"
|
||||
|
||||
|
||||
def test_full_hit_after_sliding_refreshes_matches_no_cache(
|
||||
signoz: types.SigNoz,
|
||||
create_user_admin: None, # pylint: disable=unused-argument
|
||||
get_token: Callable[[str, str], str],
|
||||
insert_logs: Callable[[list[Logs]], None],
|
||||
) -> None:
|
||||
"""
|
||||
Setup:
|
||||
One service with logs over an hour, all older than the flux interval.
|
||||
|
||||
Tests:
|
||||
Five refreshes of a 30 minute window that slides by one minute (the
|
||||
auto-refresh pattern), then a full cache hit of the last window. The
|
||||
full hit equals the request with noCache, and the rowsScanned it reports
|
||||
does not exceed the rows the refreshes scanned in total. Every refresh
|
||||
stores the merged window as one more overlapping bucket whose stats
|
||||
already include the previous buckets, and a read sums all of them.
|
||||
"""
|
||||
now = datetime.now(tz=UTC).replace(second=0, microsecond=0)
|
||||
run = uuid4().hex[:8]
|
||||
insert_logs(minutely_logs(now - 70 * MINUTE, now - 10 * MINUTE, {"svc-a": 3}, attributes={"run": run}))
|
||||
token = get_token(USER_ADMIN_EMAIL, USER_ADMIN_PASSWORD)
|
||||
query = build_scalar_query(
|
||||
name="A",
|
||||
signal="logs",
|
||||
aggregations=[build_aggregation("count()")],
|
||||
group_by=[build_group_by_field("service.name", "string", "resource")],
|
||||
filter_expression=f"run = '{run}'",
|
||||
step_interval=STEP,
|
||||
)
|
||||
base = now - 60 * MINUTE
|
||||
|
||||
refreshes = 5
|
||||
for i in range(refreshes):
|
||||
refresh = make_query_request(signoz, token, int((base + i * MINUTE).timestamp() * 1000), int((base + 30 * MINUTE + i * MINUTE).timestamp() * 1000), [query], no_cache=False)
|
||||
assert refresh.status_code == HTTPStatus.OK, refresh.text
|
||||
|
||||
last_start_ms = int((base + (refreshes - 1) * MINUTE).timestamp() * 1000)
|
||||
last_end_ms = int((base + 30 * MINUTE + (refreshes - 1) * MINUTE).timestamp() * 1000)
|
||||
cached = make_query_request(signoz, token, last_start_ms, last_end_ms, [query], no_cache=False)
|
||||
assert cached.status_code == HTTPStatus.OK, cached.text
|
||||
fresh = make_query_request(signoz, token, last_start_ms, last_end_ms, [query], no_cache=True)
|
||||
assert fresh.status_code == HTTPStatus.OK, fresh.text
|
||||
|
||||
fresh_rows = fresh.json()["data"]["meta"]["rowsScanned"]
|
||||
cached_rows = cached.json()["data"]["meta"]["rowsScanned"]
|
||||
assert fresh_rows > 0, fresh.json()["data"]["meta"]
|
||||
assert_series_points_equal(cached.json(), fresh.json(), "A", f"full hit after {refreshes} sliding refreshes")
|
||||
assert cached_rows <= refreshes * fresh_rows, f"full cache hit reports {cached_rows} rows scanned; {refreshes} refreshes of a query that scans {fresh_rows} rows cannot have scanned more than {refreshes * fresh_rows}"
|
||||
|
||||
|
||||
def test_sub_step_window_matches_no_cache(
|
||||
signoz: types.SigNoz,
|
||||
create_user_admin: None, # pylint: disable=unused-argument
|
||||
get_token: Callable[[str, str], str],
|
||||
insert_logs: Callable[[list[Logs]], None],
|
||||
) -> None:
|
||||
"""
|
||||
Setup:
|
||||
One log per minute for ten minutes, cached with a 5 minute step.
|
||||
|
||||
Tests:
|
||||
A request for the first three minutes with the same step equals the
|
||||
request with noCache, which aggregates the partial interval (3 logs).
|
||||
The cache holds the whole interval (5 logs) and serves it.
|
||||
"""
|
||||
now = datetime.now(tz=UTC).replace(second=0, microsecond=0)
|
||||
start = now.replace(minute=now.minute - now.minute % 5) - 30 * MINUTE
|
||||
end = start + 10 * MINUTE
|
||||
run = uuid4().hex[:8]
|
||||
insert_logs(minutely_logs(start, end, {"svc-a": 1}, attributes={"run": run}))
|
||||
token = get_token(USER_ADMIN_EMAIL, USER_ADMIN_PASSWORD)
|
||||
query = build_scalar_query(
|
||||
name="A",
|
||||
signal="logs",
|
||||
aggregations=[build_aggregation("count()")],
|
||||
group_by=[build_group_by_field("service.name", "string", "resource")],
|
||||
filter_expression=f"run = '{run}'",
|
||||
step_interval=300,
|
||||
)
|
||||
start_ms, end_ms = int(start.timestamp() * 1000), int(end.timestamp() * 1000)
|
||||
short_end_ms = int((start + 3 * MINUTE).timestamp() * 1000)
|
||||
|
||||
warm = make_query_request(signoz, token, start_ms, end_ms, [query], no_cache=False)
|
||||
assert warm.status_code == HTTPStatus.OK, warm.text
|
||||
cached = make_query_request(signoz, token, start_ms, short_end_ms, [query], no_cache=False)
|
||||
assert cached.status_code == HTTPStatus.OK, cached.text
|
||||
fresh = make_query_request(signoz, token, start_ms, short_end_ms, [query], no_cache=True)
|
||||
assert fresh.status_code == HTTPStatus.OK, fresh.text
|
||||
|
||||
assert series_points_by_label(fresh.json(), "A") == {"svc-a": {start_ms: 3}}, fresh.text
|
||||
assert_series_points_equal(cached.json(), fresh.json(), "A", "window shorter than the step")
|
||||
|
||||
|
||||
def test_running_diff_full_hit_matches_no_cache(
|
||||
signoz: types.SigNoz,
|
||||
create_user_admin: None, # pylint: disable=unused-argument
|
||||
get_token: Callable[[str, str], str],
|
||||
insert_metrics: Callable[[list[Metrics]], None],
|
||||
) -> None:
|
||||
"""
|
||||
Setup:
|
||||
A gauge that grows by 10 every minute over four minutes.
|
||||
|
||||
Tests:
|
||||
runningDiff over the last three minutes, requested twice. The metrics
|
||||
builder fetches one lookback point before the window so the first
|
||||
interval has a difference; the cache does not keep that point, so the
|
||||
full hit loses the first difference.
|
||||
"""
|
||||
now = datetime.now(tz=UTC).replace(second=0, microsecond=0)
|
||||
metric = f"cache_diff_{uuid4().hex[:8]}"
|
||||
insert_metrics(
|
||||
[
|
||||
Metrics(
|
||||
metric_name=metric,
|
||||
labels={"service": "svc-a"},
|
||||
timestamp=now - (14 - i) * MINUTE,
|
||||
value=100.0 + 10 * i,
|
||||
temporality="Unspecified",
|
||||
type_="Gauge",
|
||||
is_monotonic=False,
|
||||
)
|
||||
for i in range(4)
|
||||
]
|
||||
)
|
||||
token = get_token(USER_ADMIN_EMAIL, USER_ADMIN_PASSWORD)
|
||||
query = build_builder_query("A", metric, "avg", "avg", group_by=["service"], functions=[build_function("runningDiff")])
|
||||
start_ms, end_ms = int((now - 13 * MINUTE).timestamp() * 1000), int((now - 10 * MINUTE).timestamp() * 1000)
|
||||
|
||||
warm = make_query_request(signoz, token, start_ms, end_ms, [query], no_cache=False)
|
||||
assert warm.status_code == HTTPStatus.OK, warm.text
|
||||
cached = make_query_request(signoz, token, start_ms, end_ms, [query], no_cache=False)
|
||||
assert cached.status_code == HTTPStatus.OK, cached.text
|
||||
fresh = make_query_request(signoz, token, start_ms, end_ms, [query], no_cache=True)
|
||||
assert fresh.status_code == HTTPStatus.OK, fresh.text
|
||||
|
||||
fresh_points = series_points_by_label(fresh.json(), "A", label="service")
|
||||
assert sorted(fresh_points["svc-a"].values()) == [10, 10, 10], fresh_points
|
||||
assert_series_points_equal(cached.json(), fresh.json(), "A", "runningDiff on a full cache hit", label="service")
|
||||
|
||||
|
||||
def test_limited_group_by_shrunk_window_matches_no_cache(
|
||||
signoz: types.SigNoz,
|
||||
create_user_admin: None, # pylint: disable=unused-argument
|
||||
get_token: Callable[[str, str], str],
|
||||
insert_logs: Callable[[list[Logs]], None],
|
||||
) -> None:
|
||||
"""
|
||||
Setup:
|
||||
svc-a dominates the first five minutes (20/min against 1/min), svc-b the
|
||||
next five (10/min against 1/min). Over ten minutes svc-a wins.
|
||||
|
||||
Tests:
|
||||
After the whole window is cached with limit=1, a request for the second
|
||||
half equals the request with noCache: svc-b. The cache serves the winner
|
||||
of the window that filled it.
|
||||
"""
|
||||
now = datetime.now(tz=UTC).replace(second=0, microsecond=0)
|
||||
run = uuid4().hex[:8]
|
||||
start, middle, end = now - 40 * MINUTE, now - 35 * MINUTE, now - 30 * MINUTE
|
||||
insert_logs(minutely_logs(start, middle, {"svc-a": 20, "svc-b": 1}, attributes={"run": run}))
|
||||
insert_logs(minutely_logs(middle, end, {"svc-a": 1, "svc-b": 10}, attributes={"run": run}))
|
||||
token = get_token(USER_ADMIN_EMAIL, USER_ADMIN_PASSWORD)
|
||||
query = build_scalar_query(
|
||||
name="A",
|
||||
signal="logs",
|
||||
aggregations=[build_aggregation("count()")],
|
||||
group_by=[build_group_by_field("service.name", "string", "resource")],
|
||||
order=[build_order_by("count()", "desc")],
|
||||
limit=1,
|
||||
filter_expression=f"run = '{run}'",
|
||||
step_interval=STEP,
|
||||
)
|
||||
start_ms, middle_ms, end_ms = (int(t.timestamp() * 1000) for t in (start, middle, end))
|
||||
|
||||
warm = make_query_request(signoz, token, start_ms, end_ms, [query], no_cache=False)
|
||||
assert warm.status_code == HTTPStatus.OK, warm.text
|
||||
cached = make_query_request(signoz, token, middle_ms, end_ms, [query], no_cache=False)
|
||||
assert cached.status_code == HTTPStatus.OK, cached.text
|
||||
fresh = make_query_request(signoz, token, middle_ms, end_ms, [query], no_cache=True)
|
||||
assert fresh.status_code == HTTPStatus.OK, fresh.text
|
||||
|
||||
fresh_points = series_points_by_label(fresh.json(), "A")
|
||||
assert set(fresh_points) == {"svc-b"} and sorted(fresh_points["svc-b"].values()) == [10] * 5, fresh_points
|
||||
assert_series_points_equal(cached.json(), fresh.json(), "A", "top-1 of a window smaller than the cached window")
|
||||
|
||||
|
||||
def test_formula_by_alias_full_hit_matches_no_cache(
|
||||
signoz: types.SigNoz,
|
||||
create_user_admin: None, # pylint: disable=unused-argument
|
||||
get_token: Callable[[str, str], str],
|
||||
insert_logs: Callable[[list[Logs]], None],
|
||||
) -> None:
|
||||
"""
|
||||
Setup:
|
||||
One log per minute for one service, fully cacheable window.
|
||||
|
||||
Tests:
|
||||
A formula that references the aggregation by alias, `[A.__result_0] * 2`,
|
||||
requested twice. The full hit returns the aggregation without its alias,
|
||||
so the reference resolves to nothing and the formula is empty.
|
||||
"""
|
||||
now = datetime.now(tz=UTC).replace(second=0, microsecond=0)
|
||||
run = uuid4().hex[:8]
|
||||
start, end = now - 30 * MINUTE, now - 20 * MINUTE
|
||||
insert_logs(minutely_logs(start, end, {"svc-a": 1}, attributes={"run": run}))
|
||||
token = get_token(USER_ADMIN_EMAIL, USER_ADMIN_PASSWORD)
|
||||
queries = [
|
||||
build_scalar_query(
|
||||
name="A",
|
||||
signal="logs",
|
||||
aggregations=[build_aggregation("count()")],
|
||||
group_by=[build_group_by_field("service.name", "string", "resource")],
|
||||
filter_expression=f"run = '{run}'",
|
||||
step_interval=STEP,
|
||||
),
|
||||
build_formula_query("F", "[A.__result_0] * 2"),
|
||||
]
|
||||
start_ms, end_ms = int(start.timestamp() * 1000), int(end.timestamp() * 1000)
|
||||
|
||||
warm = make_query_request(signoz, token, start_ms, end_ms, queries, no_cache=False)
|
||||
assert warm.status_code == HTTPStatus.OK, warm.text
|
||||
cached = make_query_request(signoz, token, start_ms, end_ms, queries, no_cache=False)
|
||||
assert cached.status_code == HTTPStatus.OK, cached.text
|
||||
fresh = make_query_request(signoz, token, start_ms, end_ms, queries, no_cache=True)
|
||||
assert fresh.status_code == HTTPStatus.OK, fresh.text
|
||||
|
||||
fresh_points = series_points_by_label(fresh.json(), "F")
|
||||
assert set(fresh_points["svc-a"].values()) == {2}, fresh_points
|
||||
assert_series_points_equal(cached.json(), fresh.json(), "F", "formula by alias on a full cache hit")
|
||||
|
||||
|
||||
def test_promql_reserved_variable_partial_hit_matches_no_cache(
|
||||
signoz: types.SigNoz,
|
||||
create_user_admin: None, # pylint: disable=unused-argument
|
||||
get_token: Callable[[str, str], str],
|
||||
) -> None:
|
||||
"""
|
||||
Setup:
|
||||
No data needed: vector($start_timestamp) returns the request start.
|
||||
|
||||
Tests:
|
||||
A two minute request fills the cache, then the same start with a four
|
||||
minute window. Every point of the second response equals the request
|
||||
start, as with noCache. The gap sub-query renders the variable from its
|
||||
own fragment start.
|
||||
"""
|
||||
now = datetime.now(tz=UTC).replace(second=0, microsecond=0)
|
||||
start = now - 20 * MINUTE
|
||||
token = get_token(USER_ADMIN_EMAIL, USER_ADMIN_PASSWORD)
|
||||
query = {"type": "promql", "spec": {"name": "A", "query": "vector($start_timestamp)", "step": STEP}}
|
||||
start_ms = int(start.timestamp() * 1000)
|
||||
|
||||
warm = make_query_request(signoz, token, start_ms, int((start + 2 * MINUTE).timestamp() * 1000), [query], no_cache=False)
|
||||
assert warm.status_code == HTTPStatus.OK, warm.text
|
||||
cached = make_query_request(signoz, token, start_ms, int((start + 4 * MINUTE).timestamp() * 1000), [query], no_cache=False)
|
||||
assert cached.status_code == HTTPStatus.OK, cached.text
|
||||
fresh = make_query_request(signoz, token, start_ms, int((start + 4 * MINUTE).timestamp() * 1000), [query], no_cache=True)
|
||||
assert fresh.status_code == HTTPStatus.OK, fresh.text
|
||||
|
||||
fresh_values = {v["value"] for s in get_all_series(fresh.json(), "A") for v in s["values"]}
|
||||
assert fresh_values == {start_ms / 1000}, fresh_values
|
||||
cached_values = {v["value"] for s in get_all_series(cached.json(), "A") for v in s["values"]}
|
||||
assert cached_values == fresh_values, f"gap sub-query rendered $start_timestamp from its fragment: {sorted(cached_values)}"
|
||||
assert_all_series_equal(cached.json(), fresh.json(), "A", "PromQL reserved variable after a partial cache hit")
|
||||
|
||||
|
||||
def test_promql_sub_step_window_at_flux_boundary_matches_no_cache(
|
||||
signoz: types.SigNoz,
|
||||
create_user_admin: None, # pylint: disable=unused-argument
|
||||
get_token: Callable[[str, str], str],
|
||||
insert_metrics: Callable[[list[Metrics]], None],
|
||||
) -> None:
|
||||
"""
|
||||
Setup:
|
||||
A gauge with one sample per minute for the last 30 minutes, and a cache
|
||||
entry for the query from an older window.
|
||||
|
||||
Tests:
|
||||
A one minute window centred on the flux boundary (now - 5m) with a
|
||||
5 minute step evaluates once, at the grid instant its start is moved
|
||||
to, with the cache as with noCache.
|
||||
"""
|
||||
now = datetime.now(tz=UTC).replace(second=0, microsecond=0)
|
||||
metric = f"cache_flux_{uuid4().hex[:8]}"
|
||||
insert_metrics(
|
||||
[
|
||||
Metrics(
|
||||
metric_name=metric,
|
||||
labels={"service": "svc-a"},
|
||||
timestamp=now - minute * MINUTE,
|
||||
value=10.0,
|
||||
temporality="Unspecified",
|
||||
type_="Gauge",
|
||||
is_monotonic=False,
|
||||
)
|
||||
for minute in range(30, -1, -1)
|
||||
]
|
||||
)
|
||||
token = get_token(USER_ADMIN_EMAIL, USER_ADMIN_PASSWORD)
|
||||
query = {"type": "promql", "spec": {"name": "A", "query": f"sum by (service) ({metric})", "step": 300}}
|
||||
|
||||
warm = make_query_request(signoz, token, int((now - 30 * MINUTE).timestamp() * 1000), int((now - 20 * MINUTE).timestamp() * 1000), [query], no_cache=False)
|
||||
assert warm.status_code == HTTPStatus.OK, warm.text
|
||||
|
||||
boundary = datetime.now(tz=UTC) - 5 * MINUTE
|
||||
start_ms = int((boundary - timedelta(seconds=30)).timestamp() * 1000)
|
||||
end_ms = int((boundary + timedelta(seconds=30)).timestamp() * 1000)
|
||||
cached = make_query_request(signoz, token, start_ms, end_ms, [query], no_cache=False)
|
||||
assert cached.status_code == HTTPStatus.OK, cached.text
|
||||
fresh = make_query_request(signoz, token, start_ms, end_ms, [query], no_cache=True)
|
||||
assert fresh.status_code == HTTPStatus.OK, fresh.text
|
||||
|
||||
fresh_points = series_points_by_label(fresh.json(), "A", label="service")
|
||||
assert fresh_points == {"svc-a": {start_ms - start_ms % (300 * 1000): 10}}, fresh_points
|
||||
assert_series_points_equal(cached.json(), fresh.json(), "A", "PromQL window shorter than the step at the flux boundary", label="service")
|
||||
|
||||
|
||||
def test_narrower_window_does_not_serve_series_without_points(
|
||||
signoz: types.SigNoz,
|
||||
create_user_admin: None, # pylint: disable=unused-argument
|
||||
get_token: Callable[[str, str], str],
|
||||
insert_logs: Callable[[list[Logs]], None],
|
||||
) -> None:
|
||||
"""
|
||||
Setup:
|
||||
svc-a logs for ten minutes, svc-b logs for the first five only.
|
||||
|
||||
Tests:
|
||||
After the ten minutes are cached, a request for the last five minutes
|
||||
returns only svc-a, as the request with noCache does. The cache keeps
|
||||
every series of the bucket and only drops their points, so svc-b comes
|
||||
back with no values.
|
||||
"""
|
||||
now = datetime.now(tz=UTC).replace(second=0, microsecond=0)
|
||||
run = uuid4().hex[:8]
|
||||
start, middle, end = now - 40 * MINUTE, now - 35 * MINUTE, now - 30 * MINUTE
|
||||
insert_logs(minutely_logs(start, end, {"svc-a": 1}, attributes={"run": run}))
|
||||
insert_logs(minutely_logs(start, middle, {"svc-b": 1}, attributes={"run": run}))
|
||||
token = get_token(USER_ADMIN_EMAIL, USER_ADMIN_PASSWORD)
|
||||
query = build_scalar_query(
|
||||
name="A",
|
||||
signal="logs",
|
||||
aggregations=[build_aggregation("count()")],
|
||||
group_by=[build_group_by_field("service.name", "string", "resource")],
|
||||
filter_expression=f"run = '{run}'",
|
||||
step_interval=STEP,
|
||||
)
|
||||
start_ms, middle_ms, end_ms = (int(t.timestamp() * 1000) for t in (start, middle, end))
|
||||
|
||||
warm = make_query_request(signoz, token, start_ms, end_ms, [query], no_cache=False)
|
||||
assert warm.status_code == HTTPStatus.OK, warm.text
|
||||
cached = make_query_request(signoz, token, middle_ms, end_ms, [query], no_cache=False)
|
||||
assert cached.status_code == HTTPStatus.OK, cached.text
|
||||
fresh = make_query_request(signoz, token, middle_ms, end_ms, [query], no_cache=True)
|
||||
assert fresh.status_code == HTTPStatus.OK, fresh.text
|
||||
|
||||
assert set(series_points_by_label(fresh.json(), "A")) == {"svc-a"}, fresh.text
|
||||
assert_series_points_equal(cached.json(), fresh.json(), "A", "series set of a window narrower than the cached window")
|
||||
194
tests/integration/tests/queriercache/02_differential.py
Normal file
194
tests/integration/tests/queriercache/02_differential.py
Normal file
@@ -0,0 +1,194 @@
|
||||
import random
|
||||
from collections.abc import Callable
|
||||
from datetime import UTC, datetime, timedelta
|
||||
from http import HTTPStatus
|
||||
from uuid import uuid4
|
||||
|
||||
import pytest
|
||||
|
||||
from fixtures import types
|
||||
from fixtures.auth import USER_ADMIN_EMAIL, USER_ADMIN_PASSWORD
|
||||
from fixtures.logs import Logs
|
||||
from fixtures.metrics import Metrics
|
||||
from fixtures.querier import (
|
||||
build_aggregation,
|
||||
build_builder_query,
|
||||
build_function,
|
||||
build_group_by_field,
|
||||
build_order_by,
|
||||
build_scalar_query,
|
||||
make_query_request,
|
||||
series_points_by_label,
|
||||
)
|
||||
|
||||
STEP = 60
|
||||
MINUTE = timedelta(minutes=1)
|
||||
SERVICES = ["svc-a", "svc-b", "svc-c", "svc-d"]
|
||||
DATASET_MINUTES = 150
|
||||
|
||||
|
||||
def test_random_request_sequences_match_no_cache(
|
||||
signoz: types.SigNoz,
|
||||
create_user_admin: None, # pylint: disable=unused-argument
|
||||
get_token: Callable[[str, str], str],
|
||||
insert_logs: Callable[[list[Logs]], None],
|
||||
insert_metrics: Callable[[list[Metrics]], None],
|
||||
pytestconfig: pytest.Config,
|
||||
) -> None:
|
||||
"""
|
||||
Setup:
|
||||
Four services with random per-minute log counts and a gauge sample per
|
||||
minute over the last 150 minutes, all older than the flux interval.
|
||||
|
||||
Tests:
|
||||
Random sessions, each with one query shape (logs count with or without a
|
||||
limit or a timeShift, a metrics avg with or without runningDiff, or a
|
||||
PromQL sum) and a random sequence of windows that slide, grow, shrink,
|
||||
nest or repeat, with ends on and off the step grid. Every request is sent
|
||||
through the cache and with noCache, and the two answers must agree. The
|
||||
failure message groups the differences by shape, window relation and
|
||||
symptom. Reproduce a run with --cache-fuzz-seed.
|
||||
"""
|
||||
seed = pytestconfig.getoption("--cache-fuzz-seed")
|
||||
rng = random.Random(seed)
|
||||
now = datetime.now(tz=UTC).replace(second=0, microsecond=0)
|
||||
dataset_end = now - 10 * MINUTE
|
||||
dataset_start = dataset_end - DATASET_MINUTES * MINUTE
|
||||
run = uuid4().hex[:8]
|
||||
|
||||
counts = {service: [rng.randint(0, 5) if rng.random() < 0.8 else 0 for _ in range(DATASET_MINUTES)] for service in SERVICES}
|
||||
insert_logs(
|
||||
[
|
||||
Logs(
|
||||
timestamp=dataset_start + minute * MINUTE + timedelta(seconds=1 + i),
|
||||
resources={"service.name": service},
|
||||
attributes={"run": run},
|
||||
body=f"{service} {minute} {i}",
|
||||
)
|
||||
for service in SERVICES
|
||||
for minute in range(DATASET_MINUTES)
|
||||
for i in range(counts[service][minute])
|
||||
]
|
||||
)
|
||||
metric = f"cache_fuzz_gauge_{run}"
|
||||
insert_metrics(
|
||||
[
|
||||
Metrics(
|
||||
metric_name=metric,
|
||||
labels={"service": service},
|
||||
timestamp=dataset_start + minute * MINUTE,
|
||||
value=float(100 + 10 * SERVICES.index(service) + rng.randint(0, 50)),
|
||||
temporality="Unspecified",
|
||||
type_="Gauge",
|
||||
is_monotonic=False,
|
||||
)
|
||||
for service in SERVICES[:2]
|
||||
for minute in range(DATASET_MINUTES)
|
||||
]
|
||||
)
|
||||
token = get_token(USER_ADMIN_EMAIL, USER_ADMIN_PASSWORD)
|
||||
|
||||
shapes = []
|
||||
for session in range(12):
|
||||
kind = session % 6
|
||||
session_token = f"{run}-{session}"
|
||||
if kind == 4:
|
||||
shapes.append(("promql", f"sum by (service) ({metric}) + {session}", "service", None))
|
||||
elif kind == 5:
|
||||
shapes.append(("metrics/avg runningDiff", build_builder_query("A", metric, "avg", "avg", group_by=["service"], functions=[build_function("runningDiff")], filter_expression=f"service != 'none-{session_token}'"), "service", None))
|
||||
else:
|
||||
limit = 2 if kind == 1 else None
|
||||
functions = [build_function("timeShift", 3600)] if kind == 2 else None
|
||||
label = "logs/count" + (" limit=2" if limit else "") + (" timeShift=3600" if functions else "")
|
||||
shapes.append(
|
||||
(
|
||||
label,
|
||||
build_scalar_query(
|
||||
name="A",
|
||||
signal="logs",
|
||||
aggregations=[build_aggregation("count()")],
|
||||
group_by=[build_group_by_field("service.name", "string", "resource")],
|
||||
order=[build_order_by("count()", "desc")] if limit else None,
|
||||
limit=limit,
|
||||
filter_expression=f"run = '{run}' AND run != 'none-{session_token}'",
|
||||
step_interval=STEP,
|
||||
functions=functions,
|
||||
),
|
||||
"service.name",
|
||||
None,
|
||||
)
|
||||
)
|
||||
|
||||
mismatches = []
|
||||
requests = 0
|
||||
for label, shape, series_label, _ in shapes:
|
||||
query = {"type": "promql", "spec": {"name": "A", "query": shape, "step": STEP}} if label == "promql" else shape
|
||||
history = []
|
||||
for _ in range(4):
|
||||
length = rng.choice([30, 3 * 60, 17 * 60, 60 * 60])
|
||||
if history and rng.random() < 0.66:
|
||||
start = history[-1][0] + rng.randint(-15, 15) * 60
|
||||
if rng.random() < 0.5:
|
||||
length = history[-1][1] - history[-1][0]
|
||||
else:
|
||||
start = int(dataset_start.timestamp()) + rng.randint(65, DATASET_MINUTES - 10) * 60
|
||||
start = max(start, int(dataset_start.timestamp()) + 61 * 60)
|
||||
if rng.random() < 0.5:
|
||||
start += rng.randint(0, 59)
|
||||
end = start + length + (rng.randint(0, 59) if rng.random() < 0.5 else 0)
|
||||
end = min(end, int(dataset_end.timestamp()))
|
||||
if end <= start:
|
||||
end = start + STEP
|
||||
|
||||
cached = make_query_request(signoz, token, start * 1000, end * 1000, [query], no_cache=False)
|
||||
assert cached.status_code == HTTPStatus.OK, cached.text
|
||||
fresh = make_query_request(signoz, token, start * 1000, end * 1000, [query], no_cache=True)
|
||||
assert fresh.status_code == HTTPStatus.OK, fresh.text
|
||||
requests += 1
|
||||
|
||||
got = {name: points for name, points in series_points_by_label(cached.json(), "A", series_label).items()}
|
||||
want = {name: points for name, points in series_points_by_label(fresh.json(), "A", series_label).items()}
|
||||
symptoms = set()
|
||||
details = []
|
||||
if set(got) != set(want):
|
||||
symptoms.add("series-set")
|
||||
details.append(f"series got={sorted(got)} expected={sorted(want)}")
|
||||
for name in sorted(set(got) & set(want)):
|
||||
missing = set(want[name]) - set(got[name])
|
||||
extra = set(got[name]) - set(want[name])
|
||||
changed = [ts for ts in set(got[name]) & set(want[name]) if abs(got[name][ts] - want[name][ts]) > 1e-9]
|
||||
if missing:
|
||||
symptoms.add("points-missing")
|
||||
details.append(f"{name}: {len(missing)} of {len(want[name])} missing")
|
||||
if extra:
|
||||
symptoms.add("points-extra")
|
||||
details.append(f"{name}: {len(extra)} extra")
|
||||
if changed:
|
||||
symptoms.add("value-differs")
|
||||
details.append(f"{name}: {len(changed)} values differ")
|
||||
|
||||
relation = "first"
|
||||
for prev_start, prev_end in history:
|
||||
if (start, end) == (prev_start, prev_end):
|
||||
relation = "repeat"
|
||||
elif prev_start <= start and end <= prev_end:
|
||||
relation = "inside-cached"
|
||||
elif start <= prev_start and prev_end <= end:
|
||||
relation = "covers-cached"
|
||||
elif start < prev_end and end > prev_start:
|
||||
relation = "overlaps-cached"
|
||||
else:
|
||||
continue
|
||||
break
|
||||
else:
|
||||
relation = "disjoint" if history else "first"
|
||||
geometry = ",".join(part for part, present in (("start-unaligned", start % STEP != 0), ("end-unaligned", end % STEP != 0), ("sub-step", end - start < STEP), (relation, True)) if present)
|
||||
if symptoms:
|
||||
mismatches.append((label, geometry, "+".join(sorted(symptoms)), f"{datetime.fromtimestamp(start, tz=UTC):%H:%M:%S}-{datetime.fromtimestamp(end, tz=UTC):%H:%M:%S}", "; ".join(details)))
|
||||
history.append((start, end))
|
||||
|
||||
classes: dict[tuple[str, str, str], list] = {}
|
||||
for label, geometry, symptom, window, detail in mismatches:
|
||||
classes.setdefault((label, geometry, symptom), []).append((window, detail))
|
||||
report = "\n".join(f" {len(items):3d} [{label}] {geometry} -> {symptom} e.g. {items[0][0]}: {items[0][1]}" for (label, geometry, symptom), items in sorted(classes.items(), key=lambda kv: -len(kv[1])))
|
||||
assert not mismatches, f"{len(mismatches)} of {requests} random requests differ from the uncached answer (seed {seed}), {len(classes)} classes:\n{report}"
|
||||
101
tests/integration/tests/queriercache/03_overlap_shapes.py
Normal file
101
tests/integration/tests/queriercache/03_overlap_shapes.py
Normal file
@@ -0,0 +1,101 @@
|
||||
import os
|
||||
from collections.abc import Callable
|
||||
from datetime import UTC, datetime, timedelta
|
||||
from http import HTTPStatus
|
||||
from uuid import uuid4
|
||||
|
||||
import pytest
|
||||
|
||||
from fixtures import types
|
||||
from fixtures.auth import USER_ADMIN_EMAIL, USER_ADMIN_PASSWORD
|
||||
from fixtures.metrics import Metrics
|
||||
from fixtures.querier import (
|
||||
assert_series_points_equal,
|
||||
build_builder_query,
|
||||
make_query_request,
|
||||
)
|
||||
|
||||
TESTDATA_DIR = os.path.join(os.path.dirname(__file__), "..", "..", "testdata")
|
||||
CUMULATIVE_COUNTERS_FILE = os.path.join(TESTDATA_DIR, "cumulative_counters_1h.jsonl")
|
||||
MINUTE = timedelta(minutes=1)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"cached_windows,query_window",
|
||||
[
|
||||
pytest.param([(60, 40)], (70, 50), id="right_overlap"),
|
||||
pytest.param([(70, 50)], (60, 40), id="left_overlap"),
|
||||
pytest.param([(80, 40)], (70, 50), id="subset"),
|
||||
pytest.param([(65, 55)], (80, 40), id="superset"),
|
||||
pytest.param([(50, 40), (70, 60)], (70, 40), id="gap_in_middle"),
|
||||
pytest.param([(45, 40), (60, 55), (75, 70)], (80, 40), id="multiple_gaps"),
|
||||
],
|
||||
)
|
||||
def test_cumulative_rate_overlap_shapes_match_no_cache(
|
||||
signoz: types.SigNoz,
|
||||
create_user_admin: None, # pylint: disable=unused-argument
|
||||
get_token: Callable[[str, str], str],
|
||||
insert_metrics: Callable[[list[Metrics]], None],
|
||||
cached_windows: list[tuple[int, int]],
|
||||
query_window: tuple[int, int],
|
||||
) -> None:
|
||||
"""
|
||||
Setup:
|
||||
The cumulative counter fixture (one hour of samples, five endpoints) placed
|
||||
90 minutes back, so every window is older than the flux interval.
|
||||
|
||||
Tests:
|
||||
The overlap shapes of PR 9977 on a rate over a cumulative counter, whose
|
||||
first point needs the sample before the window. Windows are minutes ago
|
||||
as (start, end). The cached windows are requested first, then the query
|
||||
window through the cache and with noCache, and both must agree.
|
||||
"""
|
||||
now = datetime.now(tz=UTC).replace(second=0, microsecond=0)
|
||||
metric = f"cache_shape_{uuid4().hex[:8]}"
|
||||
insert_metrics(Metrics.load_from_file(CUMULATIVE_COUNTERS_FILE, base_time=now - 90 * MINUTE, metric_name_override=metric))
|
||||
token = get_token(USER_ADMIN_EMAIL, USER_ADMIN_PASSWORD)
|
||||
query = build_builder_query("A", metric, "rate", "sum", temporality="cumulative", group_by=["endpoint"])
|
||||
|
||||
for start_ago, end_ago in cached_windows:
|
||||
warm = make_query_request(signoz, token, int((now - start_ago * MINUTE).timestamp() * 1000), int((now - end_ago * MINUTE).timestamp() * 1000), [query], no_cache=False)
|
||||
assert warm.status_code == HTTPStatus.OK, warm.text
|
||||
|
||||
start_ms = int((now - query_window[0] * MINUTE).timestamp() * 1000)
|
||||
end_ms = int((now - query_window[1] * MINUTE).timestamp() * 1000)
|
||||
fresh = make_query_request(signoz, token, start_ms, end_ms, [query], no_cache=True)
|
||||
assert fresh.status_code == HTTPStatus.OK, fresh.text
|
||||
cached = make_query_request(signoz, token, start_ms, end_ms, [query], no_cache=False)
|
||||
assert cached.status_code == HTTPStatus.OK, cached.text
|
||||
|
||||
assert_series_points_equal(cached.json(), fresh.json(), "A", f"cumulative rate, cached {cached_windows}, query {query_window}", label="endpoint")
|
||||
|
||||
|
||||
def test_cumulative_rate_sliding_refreshes_match_no_cache(
|
||||
signoz: types.SigNoz,
|
||||
create_user_admin: None, # pylint: disable=unused-argument
|
||||
get_token: Callable[[str, str], str],
|
||||
insert_metrics: Callable[[list[Metrics]], None],
|
||||
) -> None:
|
||||
"""
|
||||
Setup:
|
||||
The cumulative counter fixture placed 90 minutes back.
|
||||
|
||||
Tests:
|
||||
A 30 minute window over the rate that slides by one minute for ten
|
||||
refreshes (PR 9977's continuous fetching case); every refresh equals the
|
||||
request with noCache.
|
||||
"""
|
||||
now = datetime.now(tz=UTC).replace(second=0, microsecond=0)
|
||||
metric = f"cache_slide_{uuid4().hex[:8]}"
|
||||
insert_metrics(Metrics.load_from_file(CUMULATIVE_COUNTERS_FILE, base_time=now - 90 * MINUTE, metric_name_override=metric))
|
||||
token = get_token(USER_ADMIN_EMAIL, USER_ADMIN_PASSWORD)
|
||||
query = build_builder_query("A", metric, "rate", "sum", temporality="cumulative", group_by=["endpoint"])
|
||||
|
||||
for refresh in range(10):
|
||||
start_ms = int((now - (80 - refresh) * MINUTE).timestamp() * 1000)
|
||||
end_ms = int((now - (50 - refresh) * MINUTE).timestamp() * 1000)
|
||||
cached = make_query_request(signoz, token, start_ms, end_ms, [query], no_cache=False)
|
||||
assert cached.status_code == HTTPStatus.OK, cached.text
|
||||
fresh = make_query_request(signoz, token, start_ms, end_ms, [query], no_cache=True)
|
||||
assert fresh.status_code == HTTPStatus.OK, fresh.text
|
||||
assert_series_points_equal(cached.json(), fresh.json(), "A", f"cumulative rate, refresh {refresh}", label="endpoint")
|
||||
@@ -151,8 +151,9 @@ def test_group_by_endpoint(
|
||||
assert v["value"] == stable_health_value, f"Expected /health rate {stable_health_value}, got {v['value']}"
|
||||
|
||||
# /products: 51 data points with 10-minute gap (t20-t29 missing), steady +20/min
|
||||
# the bucket after the gap has no sample within the rate lookback and gets no value
|
||||
products_values = endpoint_values["/products"]
|
||||
assert len(products_values) >= 49, f"Expected >= 49 values for /products, got {len(products_values)}"
|
||||
assert len(products_values) >= 48, f"Expected >= 48 values for /products, got {len(products_values)}"
|
||||
count_steady_products = sum(1 for v in products_values if v["value"] == stable_products_value)
|
||||
|
||||
# most values should be stable, some boundary values differ due to 10-min gap
|
||||
|
||||
@@ -125,9 +125,10 @@ def test_rate_group_by_endpoint(
|
||||
assert v["value"] == 0.167, f"Expected /health rate 0.167, got {v['value']}"
|
||||
|
||||
# /products: 51 data points with 10-minute gap (t20-t29 missing), steady +20/min
|
||||
# rate = 20/60 = 0.333, gap causes lower averaged rate at boundary
|
||||
# rate = 20/60 = 0.333; the bucket after the gap has no sample within the
|
||||
# rate lookback and gets no value
|
||||
products_values = endpoint_values["/products"]
|
||||
assert len(products_values) >= 49, f"Expected >= 49 values for /products, got {len(products_values)}"
|
||||
assert len(products_values) >= 48, f"Expected >= 48 values for /products, got {len(products_values)}"
|
||||
count_steady_products = sum(1 for v in products_values if v["value"] == 0.333)
|
||||
|
||||
# most values should be 0.333, some boundary values differ due to 10-min gap
|
||||
|
||||
325
tests/integration/tests/queriermetrics/17_cache.py
Normal file
325
tests/integration/tests/queriermetrics/17_cache.py
Normal file
@@ -0,0 +1,325 @@
|
||||
from collections.abc import Callable
|
||||
from datetime import UTC, datetime, timedelta
|
||||
from http import HTTPStatus
|
||||
from uuid import uuid4
|
||||
|
||||
from fixtures import types
|
||||
from fixtures.auth import USER_ADMIN_EMAIL, USER_ADMIN_PASSWORD
|
||||
from fixtures.metrics import Metrics
|
||||
from fixtures.querier import (
|
||||
assert_results_equal,
|
||||
build_builder_query,
|
||||
get_series_values,
|
||||
make_query_request,
|
||||
)
|
||||
|
||||
MINUTE_MS = 60_000
|
||||
|
||||
|
||||
def test_builder_shortening_the_time_range_at_the_end(
|
||||
signoz: types.SigNoz,
|
||||
create_user_admin: None, # pylint: disable=unused-argument
|
||||
get_token: Callable[[str, str], str],
|
||||
insert_metrics: Callable[[list[Metrics]], None],
|
||||
) -> None:
|
||||
# the cache outlives the run, so a fixed name would serve the previous run's
|
||||
# points back to this one
|
||||
metric_name = f"cache_end_shortened_{uuid4().hex[:8]}"
|
||||
|
||||
# 40 minutes back clears the flux interval, which holds recent data out of
|
||||
# the cache. Flooring to a multiple of the 5m step makes the base query span
|
||||
# two whole steps, so both its points are complete
|
||||
start_time = datetime.fromtimestamp(int((datetime.now(tz=UTC) - timedelta(minutes=40)).timestamp()) // 300 * 300, tz=UTC)
|
||||
start_time_ms = int(start_time.timestamp() * 1000)
|
||||
end_time_ms_base_query = start_time_ms + 10 * MINUTE_MS
|
||||
end_time_ms_shortened_query = start_time_ms + 7 * MINUTE_MS
|
||||
|
||||
query = [build_builder_query("A", metric_name, "max", "max", step_interval=300)]
|
||||
|
||||
# the 5m step splits the ten minutes into two points, each the max over its
|
||||
# own step: minutes 0-4 and minutes 5-9. The second changes partway through,
|
||||
# 256 until minute 7 and then 4096, so ending the range at minute 7 has to
|
||||
# reach a different value than ending it at minute 10
|
||||
insert_metrics(
|
||||
[
|
||||
Metrics(
|
||||
metric_name=metric_name,
|
||||
labels={"service": "api"},
|
||||
timestamp=start_time + timedelta(minutes=minute),
|
||||
value=(16, 16, 16, 16, 16, 256, 256, 4096, 4096, 4096)[minute],
|
||||
type_="Gauge",
|
||||
is_monotonic=False,
|
||||
)
|
||||
for minute in range(10)
|
||||
]
|
||||
)
|
||||
|
||||
token = get_token(USER_ADMIN_EMAIL, USER_ADMIN_PASSWORD)
|
||||
|
||||
base_query = make_query_request(signoz, token, start_time_ms, end_time_ms_base_query, query, no_cache=False)
|
||||
assert base_query.status_code == HTTPStatus.OK, base_query.text
|
||||
points = sorted(get_series_values(base_query.json(), "A"), key=lambda point: point["timestamp"])
|
||||
returned_points = [(point["value"], point.get("partial", False)) for point in points]
|
||||
assert returned_points == [(16, False), (4096, False)]
|
||||
|
||||
from_cache = make_query_request(signoz, token, start_time_ms, end_time_ms_shortened_query, query, no_cache=False)
|
||||
assert from_cache.status_code == HTTPStatus.OK, from_cache.text
|
||||
|
||||
uncached = make_query_request(signoz, token, start_time_ms, end_time_ms_shortened_query, query, no_cache=True)
|
||||
assert uncached.status_code == HTTPStatus.OK, uncached.text
|
||||
|
||||
assert_results_equal(from_cache.json(), uncached.json(), "A", "shortened end")
|
||||
|
||||
# the shortened end reaches only minutes 5-6 of the second point, so it comes
|
||||
# back as 256 and partial, where the cached one spans all five minutes
|
||||
for label, response in (("from cache", from_cache), ("uncached", uncached)):
|
||||
points = sorted(get_series_values(response.json(), "A"), key=lambda point: point["timestamp"])
|
||||
returned_points = [(point["value"], point.get("partial", False)) for point in points]
|
||||
assert returned_points == [(16, False), (256, True)], label
|
||||
|
||||
|
||||
def test_builder_shortening_the_time_range_at_the_start(
|
||||
signoz: types.SigNoz,
|
||||
create_user_admin: None, # pylint: disable=unused-argument
|
||||
get_token: Callable[[str, str], str],
|
||||
insert_metrics: Callable[[list[Metrics]], None],
|
||||
) -> None:
|
||||
metric_name = f"cache_start_shortened_{uuid4().hex[:8]}"
|
||||
|
||||
# 40 minutes back clears the flux interval, which holds recent data out of
|
||||
# the cache. Flooring to a multiple of the 5m step makes the base query span
|
||||
# two whole steps, so both its points are complete
|
||||
start_time = datetime.fromtimestamp(int((datetime.now(tz=UTC) - timedelta(minutes=40)).timestamp()) // 300 * 300, tz=UTC)
|
||||
start_time_ms_base_query = int(start_time.timestamp() * 1000)
|
||||
start_time_ms_shortened_query = start_time_ms_base_query + 3 * MINUTE_MS
|
||||
end_time_ms = start_time_ms_base_query + 10 * MINUTE_MS
|
||||
|
||||
query = [build_builder_query("A", metric_name, "max", "max", step_interval=300)]
|
||||
|
||||
# the 5m step splits the ten minutes into two points, each the max over its
|
||||
# own step: minutes 0-4 and minutes 5-9. Only minute 0 holds 65536, so a first
|
||||
# point reaching it says the whole step was read even though the shortened
|
||||
# range opens at minute 3
|
||||
insert_metrics(
|
||||
[
|
||||
Metrics(
|
||||
metric_name=metric_name,
|
||||
labels={"service": "api"},
|
||||
timestamp=start_time + timedelta(minutes=minute),
|
||||
value=(65536, 16, 16, 16, 16, 4096, 4096, 4096, 4096, 4096)[minute],
|
||||
type_="Gauge",
|
||||
is_monotonic=False,
|
||||
)
|
||||
for minute in range(10)
|
||||
]
|
||||
)
|
||||
|
||||
token = get_token(USER_ADMIN_EMAIL, USER_ADMIN_PASSWORD)
|
||||
|
||||
base_query = make_query_request(signoz, token, start_time_ms_base_query, end_time_ms, query, no_cache=False)
|
||||
assert base_query.status_code == HTTPStatus.OK, base_query.text
|
||||
points = sorted(get_series_values(base_query.json(), "A"), key=lambda point: point["timestamp"])
|
||||
returned_points = [(point["value"], point.get("partial", False)) for point in points]
|
||||
assert returned_points == [(65536, False), (4096, False)]
|
||||
|
||||
from_cache = make_query_request(signoz, token, start_time_ms_shortened_query, end_time_ms, query, no_cache=False)
|
||||
assert from_cache.status_code == HTTPStatus.OK, from_cache.text
|
||||
|
||||
uncached = make_query_request(signoz, token, start_time_ms_shortened_query, end_time_ms, query, no_cache=True)
|
||||
assert uncached.status_code == HTTPStatus.OK, uncached.text
|
||||
|
||||
assert_results_equal(from_cache.json(), uncached.json(), "A", "shortened start")
|
||||
|
||||
# starting inside the first point's step flags that point partial without
|
||||
# clipping its value, which still covers the whole step and so reaches the
|
||||
# 65536 at minute 0
|
||||
for label, response in (("from cache", from_cache), ("uncached", uncached)):
|
||||
points = sorted(get_series_values(response.json(), "A"), key=lambda point: point["timestamp"])
|
||||
returned_points = [(point["value"], point.get("partial", False)) for point in points]
|
||||
assert returned_points == [(65536, True), (4096, False)], label
|
||||
|
||||
|
||||
def test_promql_running_the_same_query_twice(
|
||||
signoz: types.SigNoz,
|
||||
create_user_admin: None, # pylint: disable=unused-argument
|
||||
get_token: Callable[[str, str], str],
|
||||
insert_metrics: Callable[[list[Metrics]], None],
|
||||
) -> None:
|
||||
metric_name = f"cache_repeat_total_{uuid4().hex[:8]}"
|
||||
|
||||
# 40 minutes back clears the flux interval, which holds recent data out of
|
||||
# the cache
|
||||
start_time = datetime.fromtimestamp(int((datetime.now(tz=UTC) - timedelta(minutes=40)).timestamp()) // 60 * 60, tz=UTC)
|
||||
start_time_ms = int(start_time.timestamp() * 1000)
|
||||
end_time_ms = start_time_ms + 2 * MINUTE_MS
|
||||
|
||||
query = [{"type": "promql", "spec": {"name": "A", "query": f"sum(increase({metric_name}[2m]))", "step": 60}}]
|
||||
|
||||
# the counter opens a minute before the query so its first point has something
|
||||
# to increase over, and starts far above its own rise across the range, below
|
||||
# which increase clips its back-extrapolation at the counter's zero point. It
|
||||
# rises by a different amount each minute, so every point is its own number
|
||||
insert_metrics(
|
||||
[
|
||||
Metrics(
|
||||
metric_name=metric_name,
|
||||
labels={"service": "api"},
|
||||
timestamp=start_time + timedelta(minutes=minute),
|
||||
value=(1000, 1010, 1030, 1060, 1100)[minute + 1],
|
||||
temporality="Cumulative",
|
||||
type_="Sum",
|
||||
is_monotonic=True,
|
||||
)
|
||||
for minute in range(-1, 4)
|
||||
]
|
||||
)
|
||||
|
||||
token = get_token(USER_ADMIN_EMAIL, USER_ADMIN_PASSWORD)
|
||||
|
||||
first = make_query_request(signoz, token, start_time_ms, end_time_ms, query, no_cache=False)
|
||||
assert first.status_code == HTTPStatus.OK, first.text
|
||||
|
||||
second = make_query_request(signoz, token, start_time_ms, end_time_ms, query, no_cache=False)
|
||||
assert second.status_code == HTTPStatus.OK, second.text
|
||||
|
||||
assert_results_equal(first.json(), second.json(), "A", "the same query twice")
|
||||
|
||||
# promql reports a point at the instant the range closes, and the second run,
|
||||
# answered out of what the first one cached, has to keep it
|
||||
for run, response in (("first", first), ("second", second)):
|
||||
points = sorted(get_series_values(response.json(), "A"), key=lambda point: point["timestamp"])
|
||||
returned_points = [(point["timestamp"], point["value"]) for point in points]
|
||||
## at each timestamp t, promql looks at points in (t-2minutes, t].
|
||||
assert returned_points == [
|
||||
(start_time_ms, 20), # t = 0, points taken 1000, 1010. hence diff over 1m is 10, extrapolated to 20.
|
||||
(start_time_ms + MINUTE_MS, 40), # t = 1m, points taken 1010, 1030. hence diff over 1m is 20, extrapolated to 40.
|
||||
(end_time_ms, 60), # t = 2m, points taken 1030, 1060. hence diff over 1m is 30, extrapolated to 60.
|
||||
], f"{run} run"
|
||||
|
||||
|
||||
def test_promql_shifting_the_time_range(
|
||||
signoz: types.SigNoz,
|
||||
create_user_admin: None, # pylint: disable=unused-argument
|
||||
get_token: Callable[[str, str], str],
|
||||
insert_metrics: Callable[[list[Metrics]], None],
|
||||
) -> None:
|
||||
metric_name = f"cache_shift_gauge_{uuid4().hex[:8]}"
|
||||
|
||||
# 40 minutes back clears the flux interval, which holds recent data out of
|
||||
# the cache. Flooring to a whole minute is what makes the first query aligned
|
||||
# to its 1m step, and the unaligned one half a step off it
|
||||
start_time = datetime.fromtimestamp(int((datetime.now(tz=UTC) - timedelta(minutes=40)).timestamp()) // 60 * 60, tz=UTC)
|
||||
aligned_start_time_ms = int(start_time.timestamp() * 1000)
|
||||
aligned_end_time_ms = aligned_start_time_ms + 3 * MINUTE_MS
|
||||
unaligned_start_time_ms = aligned_start_time_ms + MINUTE_MS // 2
|
||||
unaligned_end_time_ms = aligned_end_time_ms + MINUTE_MS // 2
|
||||
|
||||
query = [{"type": "promql", "spec": {"name": "A", "query": f"max_over_time({metric_name}[2m])", "step": 60}}]
|
||||
|
||||
# a sample every 30s, rising by 100 each time, so the instants on the grid
|
||||
# and the instants half a step off it land on different samples
|
||||
insert_metrics(
|
||||
[
|
||||
Metrics(
|
||||
metric_name=metric_name,
|
||||
labels={"service": "api"},
|
||||
timestamp=start_time + timedelta(seconds=30 * half_minute),
|
||||
value=100 * (half_minute + 4),
|
||||
type_="Gauge",
|
||||
is_monotonic=False,
|
||||
)
|
||||
for half_minute in range(-3, 8)
|
||||
]
|
||||
)
|
||||
|
||||
token = get_token(USER_ADMIN_EMAIL, USER_ADMIN_PASSWORD)
|
||||
|
||||
aligned_and_cached = make_query_request(signoz, token, aligned_start_time_ms, aligned_end_time_ms, query, no_cache=False)
|
||||
assert aligned_and_cached.status_code == HTTPStatus.OK, aligned_and_cached.text
|
||||
|
||||
## at each timestamp t, promql takes the highest sample in (t-2minutes, t],
|
||||
## which is the one at t itself since the gauge only rises.
|
||||
on_the_grid = [
|
||||
(aligned_start_time_ms, 400), # t = 0
|
||||
(aligned_start_time_ms + MINUTE_MS, 600), # t = 1m
|
||||
(aligned_start_time_ms + 2 * MINUTE_MS, 800), # t = 2m
|
||||
(aligned_end_time_ms, 1000), # t = 3m
|
||||
]
|
||||
points = sorted(get_series_values(aligned_and_cached.json(), "A"), key=lambda point: point["timestamp"])
|
||||
assert [(point["timestamp"], point["value"]) for point in points] == on_the_grid
|
||||
|
||||
# a window half a step off the grid is moved onto it, cached or not, so
|
||||
# every client evaluates the same instants and shares the cache entry
|
||||
for no_cache in (True, False):
|
||||
shifted = make_query_request(signoz, token, unaligned_start_time_ms, unaligned_end_time_ms, query, no_cache=no_cache)
|
||||
assert shifted.status_code == HTTPStatus.OK, shifted.text
|
||||
points = sorted(get_series_values(shifted.json(), "A"), key=lambda point: point["timestamp"])
|
||||
assert [(point["timestamp"], point["value"]) for point in points] == on_the_grid, f"shifted window, no_cache={no_cache}"
|
||||
|
||||
# a client that asks for its own instants gets them, and they are not
|
||||
# served from the grid entry: every point falls on a sample the grid run
|
||||
# never reported
|
||||
off_the_grid = [
|
||||
(unaligned_start_time_ms, 500), # t = 30s
|
||||
(unaligned_start_time_ms + MINUTE_MS, 700), # t = 1m30s
|
||||
(unaligned_start_time_ms + 2 * MINUTE_MS, 900), # t = 2m30s
|
||||
(unaligned_end_time_ms, 1100), # t = 3m30s
|
||||
]
|
||||
for no_cache in (True, False):
|
||||
exact = make_query_request(signoz, token, unaligned_start_time_ms, unaligned_end_time_ms, query, no_cache=no_cache, no_step_alignment=True)
|
||||
assert exact.status_code == HTTPStatus.OK, exact.text
|
||||
points = sorted(get_series_values(exact.json(), "A"), key=lambda point: point["timestamp"])
|
||||
assert [(point["timestamp"], point["value"]) for point in points] == off_the_grid, f"exact window, no_cache={no_cache}"
|
||||
|
||||
|
||||
def test_builder_refreshing_a_sliding_time_range(
|
||||
signoz: types.SigNoz,
|
||||
create_user_admin: None, # pylint: disable=unused-argument
|
||||
get_token: Callable[[str, str], str],
|
||||
insert_metrics: Callable[[list[Metrics]], None],
|
||||
) -> None:
|
||||
metric_name = f"cache_sliding_{uuid4().hex[:8]}"
|
||||
|
||||
# 90 minutes back so even the twentieth refresh closes clear of the flux
|
||||
# interval, which holds recent data out of the cache
|
||||
start_time = datetime.fromtimestamp(int((datetime.now(tz=UTC) - timedelta(minutes=90)).timestamp()) // 60 * 60, tz=UTC)
|
||||
start_time_ms = int(start_time.timestamp() * 1000)
|
||||
|
||||
query = [build_builder_query("A", metric_name, "max", "max")]
|
||||
|
||||
# the 1m step gives one point per seeded minute, and a value no other minute
|
||||
# carries, so a point stitched in from the wrong range reads as the wrong minute
|
||||
insert_metrics(
|
||||
[
|
||||
Metrics(
|
||||
metric_name=metric_name,
|
||||
labels={"service": "api"},
|
||||
timestamp=start_time + timedelta(minutes=minute),
|
||||
value=1000 + minute,
|
||||
type_="Gauge",
|
||||
is_monotonic=False,
|
||||
)
|
||||
for minute in range(80)
|
||||
]
|
||||
)
|
||||
|
||||
token = get_token(USER_ADMIN_EMAIL, USER_ADMIN_PASSWORD)
|
||||
|
||||
# a dashboard left open on a one hour range, re-running a minute later each time
|
||||
for refresh in range(20):
|
||||
refresh_start_ms = start_time_ms + refresh * MINUTE_MS
|
||||
from_cache = make_query_request(signoz, token, refresh_start_ms, refresh_start_ms + 60 * MINUTE_MS, query, no_cache=False)
|
||||
assert from_cache.status_code == HTTPStatus.OK, from_cache.text
|
||||
|
||||
# each refresh is stitched out of overlapping cached ranges, so this catches
|
||||
# a point served twice, dropped, or carried over from an earlier refresh
|
||||
points = sorted(get_series_values(from_cache.json(), "A"), key=lambda point: point["timestamp"])
|
||||
returned_points = [(point["timestamp"], point["value"], point.get("partial", False)) for point in points]
|
||||
expected_points = [(start_time_ms + minute * MINUTE_MS, 1000 + minute, False) for minute in range(refresh, refresh + 60)]
|
||||
assert returned_points == expected_points, f"refresh {refresh} did not return the minutes it covers"
|
||||
|
||||
last_refresh_start_ms = start_time_ms + 19 * MINUTE_MS
|
||||
uncached = make_query_request(signoz, token, last_refresh_start_ms, last_refresh_start_ms + 60 * MINUTE_MS, query, no_cache=True)
|
||||
assert uncached.status_code == HTTPStatus.OK, uncached.text
|
||||
|
||||
assert_results_equal(from_cache.json(), uncached.json(), "A", "the twentieth refresh")
|
||||
543
tests/integration/tests/queriermetrics/18_heatmap_cache.py
Normal file
543
tests/integration/tests/queriermetrics/18_heatmap_cache.py
Normal file
@@ -0,0 +1,543 @@
|
||||
from collections.abc import Callable
|
||||
from datetime import UTC, datetime, timedelta
|
||||
from http import HTTPStatus
|
||||
from uuid import uuid4
|
||||
|
||||
import pytest
|
||||
|
||||
from fixtures import types
|
||||
from fixtures.auth import USER_ADMIN_EMAIL, USER_ADMIN_PASSWORD
|
||||
from fixtures.metrics import Metrics
|
||||
from fixtures.querier import (
|
||||
RequestType,
|
||||
assert_identical_query_response,
|
||||
build_builder_query,
|
||||
build_linear_bucket_options,
|
||||
get_heatmap_buckets,
|
||||
get_heatmap_columns,
|
||||
make_query_request,
|
||||
)
|
||||
|
||||
MINUTE_MS = 60_000
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"first_minute, expected_buckets",
|
||||
[
|
||||
pytest.param(0, [100, 200], id="the_lower_half"),
|
||||
pytest.param(5, [800, 900], id="the_upper_half"),
|
||||
],
|
||||
)
|
||||
def test_builder_narrowing_to_half_the_range(
|
||||
signoz: types.SigNoz,
|
||||
create_user_admin: None, # pylint: disable=unused-argument
|
||||
get_token: Callable[[str, str], str],
|
||||
insert_metrics: Callable[[list[Metrics]], None],
|
||||
first_minute: int,
|
||||
expected_buckets: list[int],
|
||||
) -> None:
|
||||
metric_name = f"heatmap_cache_narrowed_{uuid4().hex[:8]}"
|
||||
|
||||
start_time = datetime.fromtimestamp(int((datetime.now(tz=UTC) - timedelta(minutes=40)).timestamp()) // 60 * 60, tz=UTC)
|
||||
start_time_ms = int(start_time.timestamp() * 1000)
|
||||
end_time_ms = start_time_ms + 10 * MINUTE_MS
|
||||
|
||||
# 100 wide buckets, and the first five minutes sit seven buckets under the
|
||||
# last five, so the axis over all ten covers a stretch neither half reaches
|
||||
insert_metrics(
|
||||
[
|
||||
Metrics(
|
||||
metric_name=metric_name,
|
||||
labels={"service": "api"},
|
||||
timestamp=start_time + timedelta(minutes=minute),
|
||||
value=150 if minute < 5 else 850,
|
||||
type_="Gauge",
|
||||
is_monotonic=False,
|
||||
)
|
||||
for minute in range(10)
|
||||
]
|
||||
)
|
||||
|
||||
token = get_token(USER_ADMIN_EMAIL, USER_ADMIN_PASSWORD)
|
||||
|
||||
query = [build_builder_query("A", metric_name, "max", "max", bucket_options=build_linear_bucket_options(1000, 10))]
|
||||
|
||||
# the whole range first, which is what puts its axis in the cache
|
||||
whole_range = make_query_request(signoz, token, start_time_ms, end_time_ms, query, request_type=RequestType.HEATMAP, no_cache=False)
|
||||
assert whole_range.status_code == HTTPStatus.OK, whole_range.text
|
||||
assert get_heatmap_buckets(whole_range.json(), "A") == pytest.approx([100, 200, 300, 400, 500, 600, 700, 800, 900])
|
||||
assert [column["values"] for column in get_heatmap_columns(whole_range.json(), "A")] == [
|
||||
[0, 1, 0, 0, 0, 0, 0, 0, 0, 0],
|
||||
[0, 1, 0, 0, 0, 0, 0, 0, 0, 0],
|
||||
[0, 1, 0, 0, 0, 0, 0, 0, 0, 0],
|
||||
[0, 1, 0, 0, 0, 0, 0, 0, 0, 0],
|
||||
[0, 1, 0, 0, 0, 0, 0, 0, 0, 0],
|
||||
[0, 0, 0, 0, 0, 0, 0, 0, 1, 0],
|
||||
[0, 0, 0, 0, 0, 0, 0, 0, 1, 0],
|
||||
[0, 0, 0, 0, 0, 0, 0, 0, 1, 0],
|
||||
[0, 0, 0, 0, 0, 0, 0, 0, 1, 0],
|
||||
[0, 0, 0, 0, 0, 0, 0, 0, 1, 0],
|
||||
]
|
||||
|
||||
half_start_ms = start_time_ms + first_minute * MINUTE_MS
|
||||
half_end_ms = half_start_ms + 5 * MINUTE_MS
|
||||
from_cache = make_query_request(signoz, token, half_start_ms, half_end_ms, query, request_type=RequestType.HEATMAP, no_cache=False)
|
||||
assert from_cache.status_code == HTTPStatus.OK, from_cache.text
|
||||
|
||||
uncached = make_query_request(signoz, token, half_start_ms, half_end_ms, query, request_type=RequestType.HEATMAP, no_cache=True)
|
||||
assert uncached.status_code == HTTPStatus.OK, uncached.text
|
||||
|
||||
for source, response in (("uncached", uncached), ("from cache", from_cache)):
|
||||
assert get_heatmap_buckets(response.json(), "A") == pytest.approx(expected_buckets), source
|
||||
assert [column["values"] for column in get_heatmap_columns(response.json(), "A")] == [
|
||||
[0, 1, 0],
|
||||
[0, 1, 0],
|
||||
[0, 1, 0],
|
||||
[0, 1, 0],
|
||||
[0, 1, 0],
|
||||
], source
|
||||
|
||||
assert_identical_query_response(from_cache, uncached)
|
||||
|
||||
|
||||
def test_builder_narrowing_a_histogram(
|
||||
signoz: types.SigNoz,
|
||||
create_user_admin: None, # pylint: disable=unused-argument
|
||||
get_token: Callable[[str, str], str],
|
||||
insert_metrics: Callable[[list[Metrics]], None],
|
||||
) -> None:
|
||||
metric_name = f"heatmap_cache_histogram_{uuid4().hex[:8]}_bucket"
|
||||
|
||||
start_time = datetime.fromtimestamp(int((datetime.now(tz=UTC) - timedelta(minutes=40)).timestamp()) // 60 * 60, tz=UTC)
|
||||
start_time_ms = int(start_time.timestamp() * 1000)
|
||||
end_time_ms = start_time_ms + 10 * MINUTE_MS
|
||||
|
||||
# the count each `le` reports every minute, cumulative across `le` as a
|
||||
# histogram is. For the first five minutes the ten arrivals are all at or
|
||||
# below 1, for the last five they are all between 4 and 8, and the buckets
|
||||
# holding none of them report a count of 0 rather than going unreported
|
||||
le_to_counts = {
|
||||
"1": [10, 10, 10, 10, 10, 0, 0, 0, 0, 0],
|
||||
"2": [10, 10, 10, 10, 10, 0, 0, 0, 0, 0],
|
||||
"4": [10, 10, 10, 10, 10, 0, 0, 0, 0, 0],
|
||||
"8": [10, 10, 10, 10, 10, 10, 10, 10, 10, 10],
|
||||
"+Inf": [10, 10, 10, 10, 10, 10, 10, 10, 10, 10],
|
||||
}
|
||||
insert_metrics(
|
||||
[
|
||||
Metrics(
|
||||
metric_name=metric_name,
|
||||
labels={"le": le},
|
||||
timestamp=start_time + timedelta(minutes=minute),
|
||||
value=count,
|
||||
temporality="Delta",
|
||||
type_="Histogram",
|
||||
)
|
||||
for le, counts in le_to_counts.items()
|
||||
for minute, count in enumerate(counts)
|
||||
]
|
||||
)
|
||||
|
||||
token = get_token(USER_ADMIN_EMAIL, USER_ADMIN_PASSWORD)
|
||||
|
||||
query = [build_builder_query("A", metric_name, "increase", "p50", temporality="delta", group_by=["le"])]
|
||||
|
||||
# the whole range first, which is what puts its axis in the cache
|
||||
whole_range = make_query_request(signoz, token, start_time_ms, end_time_ms, query, request_type=RequestType.HEATMAP, no_cache=False)
|
||||
assert whole_range.status_code == HTTPStatus.OK, whole_range.text
|
||||
assert get_heatmap_buckets(whole_range.json(), "A") == [1, 2, 4, 8]
|
||||
assert [column["values"] for column in get_heatmap_columns(whole_range.json(), "A")] == [
|
||||
[10, 0, 0, 0, 0],
|
||||
[10, 0, 0, 0, 0],
|
||||
[10, 0, 0, 0, 0],
|
||||
[10, 0, 0, 0, 0],
|
||||
[10, 0, 0, 0, 0],
|
||||
[0, 0, 0, 10, 0],
|
||||
[0, 0, 0, 10, 0],
|
||||
[0, 0, 0, 10, 0],
|
||||
[0, 0, 0, 10, 0],
|
||||
[0, 0, 0, 10, 0],
|
||||
]
|
||||
|
||||
# even though this shortened time range has no data below 4, all histogram
|
||||
# buckets are still returned back
|
||||
half_start_ms = start_time_ms + 5 * MINUTE_MS
|
||||
from_cache = make_query_request(signoz, token, half_start_ms, end_time_ms, query, request_type=RequestType.HEATMAP, no_cache=False)
|
||||
assert from_cache.status_code == HTTPStatus.OK, from_cache.text
|
||||
|
||||
uncached = make_query_request(signoz, token, half_start_ms, end_time_ms, query, request_type=RequestType.HEATMAP, no_cache=True)
|
||||
assert uncached.status_code == HTTPStatus.OK, uncached.text
|
||||
|
||||
for source, response in (("uncached", uncached), ("from cache", from_cache)):
|
||||
assert get_heatmap_buckets(response.json(), "A") == [1, 2, 4, 8], source
|
||||
assert [column["values"] for column in get_heatmap_columns(response.json(), "A")] == [
|
||||
[0, 0, 0, 10, 0],
|
||||
[0, 0, 0, 10, 0],
|
||||
[0, 0, 0, 10, 0],
|
||||
[0, 0, 0, 10, 0],
|
||||
[0, 0, 0, 10, 0],
|
||||
], source
|
||||
|
||||
assert_identical_query_response(from_cache, uncached)
|
||||
|
||||
|
||||
def test_builder_shortening_the_time_range_at_the_end(
|
||||
signoz: types.SigNoz,
|
||||
create_user_admin: None, # pylint: disable=unused-argument
|
||||
get_token: Callable[[str, str], str],
|
||||
insert_metrics: Callable[[list[Metrics]], None],
|
||||
) -> None:
|
||||
metric_name = f"heatmap_cache_end_shortened_{uuid4().hex[:8]}"
|
||||
|
||||
start_time = datetime.fromtimestamp(int((datetime.now(tz=UTC) - timedelta(minutes=40)).timestamp()) // 300 * 300, tz=UTC)
|
||||
start_time_ms = int(start_time.timestamp() * 1000)
|
||||
end_time_ms_base_query = start_time_ms + 10 * MINUTE_MS
|
||||
end_time_ms_shortened_query = start_time_ms + 7 * MINUTE_MS
|
||||
|
||||
query = [build_builder_query("A", metric_name, "max", "max", step_interval=300, bucket_options=build_linear_bucket_options(1000, 10))]
|
||||
|
||||
# the 5m step splits the ten minutes into two columns, each the max over its
|
||||
# own step: minutes 0-4 and minutes 5-9. The second changes partway through,
|
||||
# 250 until minute 7 and then 850, which fall six buckets apart, so ending
|
||||
# the range at minute 7 has to reach a different bucket than ending it at
|
||||
# minute 10 and an axis that stops well below it
|
||||
insert_metrics(
|
||||
[
|
||||
Metrics(
|
||||
metric_name=metric_name,
|
||||
labels={"service": "api"},
|
||||
timestamp=start_time + timedelta(minutes=minute),
|
||||
value=(150, 150, 150, 150, 150, 250, 250, 850, 850, 850)[minute],
|
||||
type_="Gauge",
|
||||
is_monotonic=False,
|
||||
)
|
||||
for minute in range(10)
|
||||
]
|
||||
)
|
||||
|
||||
token = get_token(USER_ADMIN_EMAIL, USER_ADMIN_PASSWORD)
|
||||
|
||||
base_query = make_query_request(signoz, token, start_time_ms, end_time_ms_base_query, query, request_type=RequestType.HEATMAP, no_cache=False)
|
||||
assert base_query.status_code == HTTPStatus.OK, base_query.text
|
||||
|
||||
# 100 wide buckets, and the two maxes are 150 and 850, so the axis runs from
|
||||
# the bottom of (100, 200] to the top of (800, 900]
|
||||
assert get_heatmap_buckets(base_query.json(), "A") == pytest.approx([100, 200, 300, 400, 500, 600, 700, 800, 900])
|
||||
base_columns = get_heatmap_columns(base_query.json(), "A")
|
||||
assert [column["values"] for column in base_columns] == [
|
||||
[0, 1, 0, 0, 0, 0, 0, 0, 0, 0],
|
||||
[0, 0, 0, 0, 0, 0, 0, 0, 1, 0],
|
||||
]
|
||||
assert [column.get("partial", False) for column in base_columns] == [False, False]
|
||||
|
||||
from_cache = make_query_request(signoz, token, start_time_ms, end_time_ms_shortened_query, query, request_type=RequestType.HEATMAP, no_cache=False)
|
||||
assert from_cache.status_code == HTTPStatus.OK, from_cache.text
|
||||
|
||||
uncached = make_query_request(signoz, token, start_time_ms, end_time_ms_shortened_query, query, request_type=RequestType.HEATMAP, no_cache=True)
|
||||
assert uncached.status_code == HTTPStatus.OK, uncached.text
|
||||
|
||||
# the shortened end reaches only minutes 5-6 of the second column, whose max
|
||||
# is 250 and which comes back partial. Nothing in this window passes 300, so
|
||||
# the axis stops there rather than carrying the buckets above it
|
||||
for label, response in (("uncached", uncached), ("from cache", from_cache)):
|
||||
assert get_heatmap_buckets(response.json(), "A") == pytest.approx([100, 200, 300]), label
|
||||
columns = get_heatmap_columns(response.json(), "A")
|
||||
assert [column["values"] for column in columns] == [
|
||||
[0, 1, 0, 0],
|
||||
[0, 0, 1, 0],
|
||||
], label
|
||||
assert [column.get("partial", False) for column in columns] == [False, True], label
|
||||
|
||||
assert_identical_query_response(from_cache, uncached)
|
||||
|
||||
|
||||
def test_builder_shortening_the_time_range_at_the_start(
|
||||
signoz: types.SigNoz,
|
||||
create_user_admin: None, # pylint: disable=unused-argument
|
||||
get_token: Callable[[str, str], str],
|
||||
insert_metrics: Callable[[list[Metrics]], None],
|
||||
) -> None:
|
||||
metric_name = f"heatmap_cache_start_shortened_{uuid4().hex[:8]}"
|
||||
|
||||
# 40 minutes back clears the flux interval, which holds recent data out of
|
||||
# the cache. Flooring to a multiple of the 5m step makes the base query span
|
||||
# two whole steps, so both its columns are complete
|
||||
start_time = datetime.fromtimestamp(int((datetime.now(tz=UTC) - timedelta(minutes=40)).timestamp()) // 300 * 300, tz=UTC)
|
||||
start_time_ms_base_query = int(start_time.timestamp() * 1000)
|
||||
start_time_ms_shortened_query = start_time_ms_base_query + 3 * MINUTE_MS
|
||||
end_time_ms = start_time_ms_base_query + 10 * MINUTE_MS
|
||||
|
||||
query = [build_builder_query("A", metric_name, "max", "max", step_interval=300, bucket_options=build_linear_bucket_options(1000, 10))]
|
||||
|
||||
# the 5m step splits the ten minutes into two columns, each the max over its
|
||||
# own step: minutes 0-4 and minutes 5-9. Only minute 0 reaches 950, so a
|
||||
# first column counted in (900, 1000] says the whole step was read even
|
||||
# though the shortened range opens at minute 3
|
||||
insert_metrics(
|
||||
[
|
||||
Metrics(
|
||||
metric_name=metric_name,
|
||||
labels={"service": "api"},
|
||||
timestamp=start_time + timedelta(minutes=minute),
|
||||
value=(950, 150, 150, 150, 150, 350, 350, 350, 350, 350)[minute],
|
||||
type_="Gauge",
|
||||
is_monotonic=False,
|
||||
)
|
||||
for minute in range(10)
|
||||
]
|
||||
)
|
||||
|
||||
token = get_token(USER_ADMIN_EMAIL, USER_ADMIN_PASSWORD)
|
||||
|
||||
base_query = make_query_request(signoz, token, start_time_ms_base_query, end_time_ms, query, request_type=RequestType.HEATMAP, no_cache=False)
|
||||
assert base_query.status_code == HTTPStatus.OK, base_query.text
|
||||
|
||||
# 100 wide buckets, and the two maxes are 950 and 350, so the axis runs from
|
||||
# the bottom of (300, 400] to the top of (900, 1000]
|
||||
assert get_heatmap_buckets(base_query.json(), "A") == pytest.approx([300, 400, 500, 600, 700, 800, 900, 1000])
|
||||
base_columns = get_heatmap_columns(base_query.json(), "A")
|
||||
assert [column["values"] for column in base_columns] == [
|
||||
[0, 0, 0, 0, 0, 0, 0, 1, 0],
|
||||
[0, 1, 0, 0, 0, 0, 0, 0, 0],
|
||||
]
|
||||
assert [column.get("partial", False) for column in base_columns] == [False, False]
|
||||
|
||||
from_cache = make_query_request(signoz, token, start_time_ms_shortened_query, end_time_ms, query, request_type=RequestType.HEATMAP, no_cache=False)
|
||||
assert from_cache.status_code == HTTPStatus.OK, from_cache.text
|
||||
|
||||
uncached = make_query_request(signoz, token, start_time_ms_shortened_query, end_time_ms, query, request_type=RequestType.HEATMAP, no_cache=True)
|
||||
assert uncached.status_code == HTTPStatus.OK, uncached.text
|
||||
|
||||
# starting inside the first column's step flags that column partial without
|
||||
# clipping its counts, which still cover the whole step and so reach the 950
|
||||
# at minute 0, leaving the axis where the base query drew it
|
||||
for label, response in (("uncached", uncached), ("from cache", from_cache)):
|
||||
assert get_heatmap_buckets(response.json(), "A") == pytest.approx([300, 400, 500, 600, 700, 800, 900, 1000]), label
|
||||
columns = get_heatmap_columns(response.json(), "A")
|
||||
assert [column["values"] for column in columns] == [
|
||||
[0, 0, 0, 0, 0, 0, 0, 1, 0],
|
||||
[0, 1, 0, 0, 0, 0, 0, 0, 0],
|
||||
], label
|
||||
assert [column.get("partial", False) for column in columns] == [True, False], label
|
||||
|
||||
assert_identical_query_response(from_cache, uncached)
|
||||
|
||||
|
||||
def test_builder_refreshing_a_sliding_time_range(
|
||||
signoz: types.SigNoz,
|
||||
create_user_admin: None, # pylint: disable=unused-argument
|
||||
get_token: Callable[[str, str], str],
|
||||
insert_metrics: Callable[[list[Metrics]], None],
|
||||
) -> None:
|
||||
metric_name = f"heatmap_cache_sliding_{uuid4().hex[:8]}"
|
||||
|
||||
# 40 minutes back clears the flux interval, which holds recent data out of
|
||||
# the cache
|
||||
start_time = datetime.fromtimestamp(int((datetime.now(tz=UTC) - timedelta(minutes=40)).timestamp()) // 60 * 60, tz=UTC)
|
||||
start_time_ms = int(start_time.timestamp() * 1000)
|
||||
|
||||
query = [build_builder_query("A", metric_name, "max", "max", bucket_options=build_linear_bucket_options(1000, 10))]
|
||||
|
||||
# the 1m step gives one column per seeded minute, and 100 wide buckets give
|
||||
# every minute a bucket no other minute reaches, so a column stitched in from
|
||||
# the wrong range is counted in the wrong bucket
|
||||
insert_metrics(
|
||||
[
|
||||
Metrics(
|
||||
metric_name=metric_name,
|
||||
labels={"service": "api"},
|
||||
timestamp=start_time + timedelta(minutes=minute),
|
||||
value=100 * minute + 50,
|
||||
type_="Gauge",
|
||||
is_monotonic=False,
|
||||
)
|
||||
for minute in range(7)
|
||||
]
|
||||
)
|
||||
|
||||
token = get_token(USER_ADMIN_EMAIL, USER_ADMIN_PASSWORD)
|
||||
|
||||
# the window slides onto a bucket a minute higher each refresh, so an axis
|
||||
# carried over from an earlier one is off by as many buckets
|
||||
expected_buckets_by_refresh = [
|
||||
[0, 100, 200, 300, 400],
|
||||
[100, 200, 300, 400, 500],
|
||||
[200, 300, 400, 500, 600],
|
||||
[300, 400, 500, 600, 700],
|
||||
]
|
||||
|
||||
# whichever four minutes a refresh reads, each is in a bucket of its own and
|
||||
# they arrive in order, so the counts run down the diagonal
|
||||
expected_columns = [
|
||||
[0, 1, 0, 0, 0, 0],
|
||||
[0, 0, 1, 0, 0, 0],
|
||||
[0, 0, 0, 1, 0, 0],
|
||||
[0, 0, 0, 0, 1, 0],
|
||||
]
|
||||
|
||||
# a dashboard left open on a four minute range, re-running a minute later each
|
||||
# time, so every refresh is stitched out of the ranges the ones before it cached
|
||||
for refresh, expected_buckets in enumerate(expected_buckets_by_refresh):
|
||||
refresh_start_ms = start_time_ms + refresh * MINUTE_MS
|
||||
from_cache = make_query_request(signoz, token, refresh_start_ms, refresh_start_ms + 4 * MINUTE_MS, query, request_type=RequestType.HEATMAP, no_cache=False)
|
||||
assert from_cache.status_code == HTTPStatus.OK, from_cache.text
|
||||
|
||||
assert get_heatmap_buckets(from_cache.json(), "A") == pytest.approx(expected_buckets), f"refresh {refresh}"
|
||||
|
||||
# a column served twice, dropped, or carried over from an earlier refresh
|
||||
# breaks the diagonal or the run of timestamps
|
||||
columns = get_heatmap_columns(from_cache.json(), "A")
|
||||
assert [column["timestamp"] for column in columns] == [
|
||||
refresh_start_ms,
|
||||
refresh_start_ms + MINUTE_MS,
|
||||
refresh_start_ms + 2 * MINUTE_MS,
|
||||
refresh_start_ms + 3 * MINUTE_MS,
|
||||
], f"refresh {refresh}"
|
||||
assert [column["values"] for column in columns] == expected_columns, f"refresh {refresh}"
|
||||
|
||||
uncached = make_query_request(signoz, token, refresh_start_ms, refresh_start_ms + 4 * MINUTE_MS, query, request_type=RequestType.HEATMAP, no_cache=True)
|
||||
assert uncached.status_code == HTTPStatus.OK, uncached.text
|
||||
|
||||
assert_identical_query_response(from_cache, uncached)
|
||||
|
||||
|
||||
def test_promql_running_the_same_query_twice(
|
||||
signoz: types.SigNoz,
|
||||
create_user_admin: None, # pylint: disable=unused-argument
|
||||
get_token: Callable[[str, str], str],
|
||||
insert_metrics: Callable[[list[Metrics]], None],
|
||||
) -> None:
|
||||
metric_name = f"heatmap_cache_repeat_{uuid4().hex[:8]}_bucket"
|
||||
|
||||
# 40 minutes back clears the flux interval, which holds recent data out of
|
||||
# the cache
|
||||
start_time = datetime.fromtimestamp(int((datetime.now(tz=UTC) - timedelta(minutes=40)).timestamp()) // 60 * 60, tz=UTC)
|
||||
start_time_ms = int(start_time.timestamp() * 1000)
|
||||
end_time_ms = start_time_ms + 2 * MINUTE_MS
|
||||
|
||||
query = [{"type": "promql", "spec": {"name": "A", "query": f"sum by (le) (increase({metric_name}[2m]))", "step": 60}}]
|
||||
|
||||
# the cumulative count of each `le`, one entry per minute. The counters open
|
||||
# a minute before the query so its first column has something to increase
|
||||
# over, and start far above their own rise across the range, below which
|
||||
# increase clips its back-extrapolation at a counter's zero point
|
||||
le_to_counts = {
|
||||
"1": [1000, 1005, 1010, 1020],
|
||||
"2": [2000, 2010, 2025, 2040],
|
||||
"4": [3000, 3015, 3040, 3070],
|
||||
"+Inf": [4000, 4022, 4050, 4090],
|
||||
}
|
||||
insert_metrics(
|
||||
[
|
||||
Metrics(
|
||||
metric_name=metric_name,
|
||||
labels={"__temporality__": "Cumulative", "service": "api", "le": le},
|
||||
timestamp=start_time + timedelta(minutes=minute),
|
||||
value=count,
|
||||
temporality="Cumulative",
|
||||
type_="Histogram",
|
||||
)
|
||||
for le, counts in le_to_counts.items()
|
||||
for minute, count in enumerate(counts, start=-1)
|
||||
]
|
||||
)
|
||||
|
||||
token = get_token(USER_ADMIN_EMAIL, USER_ADMIN_PASSWORD)
|
||||
|
||||
first = make_query_request(signoz, token, start_time_ms, end_time_ms, query, request_type=RequestType.HEATMAP, no_cache=False)
|
||||
assert first.status_code == HTTPStatus.OK, first.text
|
||||
|
||||
second = make_query_request(signoz, token, start_time_ms, end_time_ms, query, request_type=RequestType.HEATMAP, no_cache=False)
|
||||
assert second.status_code == HTTPStatus.OK, second.text
|
||||
|
||||
# promql reports a column at the instant the range closes, and the second
|
||||
# run, answered out of what the first one cached, has to keep it
|
||||
for run, response in (("first", first), ("second", second)):
|
||||
assert get_heatmap_buckets(response.json(), "A") == [1, 2, 4], run
|
||||
## what the query returns per `le` is cumulative across `le`, so each
|
||||
## count is its own minus the one below it, and `le=+Inf` has no finite
|
||||
## bound to sit on and lands in the trailing slot. increase over a 2m
|
||||
## window of minutely samples extrapolates one minute's rise to two.
|
||||
assert [(column["timestamp"], column["values"]) for column in get_heatmap_columns(response.json(), "A")] == [
|
||||
(start_time_ms, [10, 10, 10, 14]), # t = 0, the minute brings 5, 10, 15 and 22 arrivals at or below each `le`
|
||||
(start_time_ms + MINUTE_MS, [10, 20, 20, 6]), # t = 1m, 5, 15, 25 and 28
|
||||
(end_time_ms, [20, 10, 30, 20]), # t = 2m, 10, 15, 30 and 40
|
||||
], f"{run} run"
|
||||
|
||||
assert_identical_query_response(first, second)
|
||||
|
||||
|
||||
def test_promql_shifting_the_time_range(
|
||||
signoz: types.SigNoz,
|
||||
create_user_admin: None, # pylint: disable=unused-argument
|
||||
get_token: Callable[[str, str], str],
|
||||
insert_metrics: Callable[[list[Metrics]], None],
|
||||
) -> None:
|
||||
metric_name = f"heatmap_cache_shift_{uuid4().hex[:8]}_bucket"
|
||||
|
||||
# 40 minutes back clears the flux interval, which holds recent data out of
|
||||
# the cache. Flooring to a whole minute is what makes the first query aligned
|
||||
# to its 1m step, and the unaligned one half a step off it
|
||||
start_time = datetime.fromtimestamp(int((datetime.now(tz=UTC) - timedelta(minutes=40)).timestamp()) // 60 * 60, tz=UTC)
|
||||
aligned_start_time_ms = int(start_time.timestamp() * 1000)
|
||||
aligned_end_time_ms = aligned_start_time_ms + 3 * MINUTE_MS
|
||||
unaligned_start_time_ms = aligned_start_time_ms + MINUTE_MS // 2
|
||||
unaligned_end_time_ms = aligned_end_time_ms + MINUTE_MS // 2
|
||||
|
||||
query = [{"type": "promql", "spec": {"name": "A", "query": f"sum by (le) (max_over_time({metric_name}[2m]))", "step": 60}}]
|
||||
|
||||
# a sample every 30s, each `le` counting up by its own fixed amount every
|
||||
# time, so the instants on the grid and the instants half a step off it
|
||||
# land on different samples
|
||||
le_to_arrivals_per_sample = {"1": 100, "2": 300, "4": 600, "+Inf": 1000}
|
||||
insert_metrics(
|
||||
[
|
||||
Metrics(
|
||||
metric_name=metric_name,
|
||||
labels={"__temporality__": "Cumulative", "service": "api", "le": le},
|
||||
timestamp=start_time + timedelta(seconds=30 * half_minute),
|
||||
value=arrivals_per_sample * (half_minute + 4),
|
||||
temporality="Cumulative",
|
||||
type_="Histogram",
|
||||
)
|
||||
for le, arrivals_per_sample in le_to_arrivals_per_sample.items()
|
||||
for half_minute in range(-3, 8)
|
||||
]
|
||||
)
|
||||
|
||||
token = get_token(USER_ADMIN_EMAIL, USER_ADMIN_PASSWORD)
|
||||
|
||||
aligned_and_cached = make_query_request(signoz, token, aligned_start_time_ms, aligned_end_time_ms, query, request_type=RequestType.HEATMAP, no_cache=False)
|
||||
assert aligned_and_cached.status_code == HTTPStatus.OK, aligned_and_cached.text
|
||||
|
||||
## each column reads the counters at their latest sample at or before its
|
||||
## timestamp, and a bucket holds its own `le`'s count less the one below it.
|
||||
on_the_grid = [
|
||||
(aligned_start_time_ms, [400, 800, 1200, 1600]), # t = 0, the fourth sample
|
||||
(aligned_start_time_ms + MINUTE_MS, [600, 1200, 1800, 2400]), # t = 1m, the sixth
|
||||
(aligned_start_time_ms + 2 * MINUTE_MS, [800, 1600, 2400, 3200]), # t = 2m, the eighth
|
||||
(aligned_end_time_ms, [1000, 2000, 3000, 4000]), # t = 3m, the tenth
|
||||
]
|
||||
assert get_heatmap_buckets(aligned_and_cached.json(), "A") == [1, 2, 4]
|
||||
assert [(column["timestamp"], column["values"]) for column in get_heatmap_columns(aligned_and_cached.json(), "A")] == on_the_grid
|
||||
|
||||
# a window half a step off the grid is moved onto it, cached or not
|
||||
for no_cache in (True, False):
|
||||
shifted = make_query_request(signoz, token, unaligned_start_time_ms, unaligned_end_time_ms, query, request_type=RequestType.HEATMAP, no_cache=no_cache)
|
||||
assert shifted.status_code == HTTPStatus.OK, shifted.text
|
||||
assert get_heatmap_buckets(shifted.json(), "A") == [1, 2, 4], f"shifted window, no_cache={no_cache}"
|
||||
assert [(column["timestamp"], column["values"]) for column in get_heatmap_columns(shifted.json(), "A")] == on_the_grid, f"shifted window, no_cache={no_cache}"
|
||||
|
||||
## every column of the exact window falls on a sample the grid run never
|
||||
## reported, so being served the grid entry would show in the counts
|
||||
off_the_grid = [
|
||||
(unaligned_start_time_ms, [500, 1000, 1500, 2000]), # t = 30s, the fifth sample
|
||||
(unaligned_start_time_ms + MINUTE_MS, [700, 1400, 2100, 2800]), # t = 1m30s, the seventh
|
||||
(unaligned_start_time_ms + 2 * MINUTE_MS, [900, 1800, 2700, 3600]), # t = 2m30s, the ninth
|
||||
(unaligned_end_time_ms, [1100, 2200, 3300, 4400]), # t = 3m30s, the eleventh
|
||||
]
|
||||
for no_cache in (True, False):
|
||||
exact = make_query_request(signoz, token, unaligned_start_time_ms, unaligned_end_time_ms, query, request_type=RequestType.HEATMAP, no_cache=no_cache, no_step_alignment=True)
|
||||
assert exact.status_code == HTTPStatus.OK, exact.text
|
||||
assert get_heatmap_buckets(exact.json(), "A") == [1, 2, 4], f"exact window, no_cache={no_cache}"
|
||||
assert [(column["timestamp"], column["values"]) for column in get_heatmap_columns(exact.json(), "A")] == off_the_grid, f"exact window, no_cache={no_cache}"
|
||||
Reference in New Issue
Block a user