fix(executor): compute p95 latency as a real 95th percentile

ExecutorMetrics.get_stats picked the p95 element with
int(n * 0.95), which collapses to the last index (the maximum)
for any batch of 20 or fewer requests, so the reported p95 was
actually the worst-case latency for typical scan batches. Use
ceil(n * 0.95) - 1, the standard nearest-rank index.
This commit is contained in:
fei
2026-08-16 18:39:10 +08:00
parent c8458d73c5
commit cf3498264f
2 changed files with 19 additions and 6 deletions
+6 -6
View File
@@ -1,6 +1,7 @@
"""Concurrent executor with rate limiting and circuit breaking."""
import asyncio
import math
import time
from typing import Any
@@ -60,12 +61,11 @@ class ExecutorMetrics:
# Calculate p95 latency
if self.latencies:
sorted_latencies = sorted(self.latencies)
p95_index = int(len(sorted_latencies) * 0.95)
p95_latency_ms = (
sorted_latencies[p95_index] * 1000
if p95_index < len(sorted_latencies)
else 0.0
)
# ceil(n * 0.95) - 1 picks the 95th-percentile element; the
# previous int(n * 0.95) collapsed to the last index (the max)
# for any batch of 20 or fewer requests.
p95_index = max(0, math.ceil(len(sorted_latencies) * 0.95) - 1)
p95_latency_ms = sorted_latencies[p95_index] * 1000
else:
p95_latency_ms = 0.0
+13
View File
@@ -83,6 +83,19 @@ class TestExecutorMetrics:
assert stats["p95_latency_ms"] >= 90.0
assert stats["p95_latency_ms"] <= 100.0
def test_get_stats_p95_latency_small_batch(self):
"""Test that p95 is the 95th percentile, not the maximum, for small batches."""
metrics = ExecutorMetrics()
# 20 requests with latencies 0ms..19ms
for i in range(20):
metrics.record_success(i * 0.001)
stats = metrics.get_stats()
# The 95th percentile of 20 samples is the 19th (18ms), not the max (19ms)
assert stats["p95_latency_ms"] == pytest.approx(18.0)
class TestConcurrentExecutor:
"""Test ConcurrentExecutor functionality."""