mirror of
https://github.com/ruvnet/RuView
synced 2026-08-07 20:01:43 +00:00
fix(api): resolve event loop blocking in health and metrics endpoints
This commit is contained in:
@@ -2,6 +2,7 @@
|
||||
Health check API endpoints
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
import logging
|
||||
import psutil
|
||||
from typing import Dict, Any, Optional
|
||||
@@ -168,7 +169,7 @@ async def health_check(request: Request):
|
||||
overall_status = "degraded"
|
||||
|
||||
# Get system metrics
|
||||
system_metrics = get_system_metrics()
|
||||
system_metrics = await asyncio.to_thread(get_system_metrics)
|
||||
|
||||
uptime_seconds = (datetime.now() - _APP_START_TIME).total_seconds()
|
||||
|
||||
@@ -263,11 +264,12 @@ async def get_health_metrics(
|
||||
):
|
||||
"""Get detailed system metrics."""
|
||||
try:
|
||||
metrics = get_system_metrics()
|
||||
metrics = await asyncio.to_thread(get_system_metrics)
|
||||
|
||||
# Add additional metrics if authenticated
|
||||
if current_user:
|
||||
metrics.update(get_detailed_metrics())
|
||||
detailed = await asyncio.to_thread(get_detailed_metrics)
|
||||
metrics.update(detailed)
|
||||
|
||||
return {
|
||||
"timestamp": datetime.utcnow().isoformat(),
|
||||
@@ -300,7 +302,7 @@ def get_system_metrics() -> Dict[str, Any]:
|
||||
"""Get basic system metrics."""
|
||||
try:
|
||||
# CPU metrics
|
||||
cpu_percent = psutil.cpu_percent(interval=1)
|
||||
cpu_percent = psutil.cpu_percent(interval=None)
|
||||
cpu_count = psutil.cpu_count()
|
||||
|
||||
# Memory metrics
|
||||
|
||||
@@ -180,21 +180,24 @@ class MetricsService:
|
||||
async def _collect_system_metrics(self):
|
||||
"""Collect system-level metrics."""
|
||||
try:
|
||||
# CPU usage
|
||||
cpu_percent = psutil.cpu_percent(interval=1)
|
||||
# Query OS metrics in a background thread to prevent blocking the event loop
|
||||
def gather_metrics():
|
||||
return (
|
||||
psutil.cpu_percent(interval=None),
|
||||
psutil.virtual_memory().percent,
|
||||
psutil.disk_usage('/'),
|
||||
psutil.net_io_counters()
|
||||
)
|
||||
|
||||
cpu_percent, mem_percent, disk, network = await asyncio.to_thread(gather_metrics)
|
||||
|
||||
# Record metrics on the main loop
|
||||
self._metrics["system_cpu_usage"].add_point(cpu_percent)
|
||||
self._metrics["system_memory_usage"].add_point(mem_percent)
|
||||
|
||||
# Memory usage
|
||||
memory = psutil.virtual_memory()
|
||||
self._metrics["system_memory_usage"].add_point(memory.percent)
|
||||
|
||||
# Disk usage
|
||||
disk = psutil.disk_usage('/')
|
||||
disk_percent = (disk.used / disk.total) * 100
|
||||
self._metrics["system_disk_usage"].add_point(disk_percent)
|
||||
|
||||
# Network I/O
|
||||
network = psutil.net_io_counters()
|
||||
self._metrics["system_network_bytes_sent"].add_point(network.bytes_sent)
|
||||
self._metrics["system_network_bytes_recv"].add_point(network.bytes_recv)
|
||||
|
||||
|
||||
@@ -0,0 +1,58 @@
|
||||
import asyncio
|
||||
import time
|
||||
import os
|
||||
import sys
|
||||
|
||||
# Add project root and archive/v1 to sys.path so we can import src modules
|
||||
sys.path.append(os.path.abspath(os.path.join(os.path.dirname(__file__), "../../../../")))
|
||||
sys.path.append(os.path.abspath(os.path.join(os.path.dirname(__file__), "../../")))
|
||||
|
||||
from archive.v1.src.api.routers.health import get_system_metrics
|
||||
|
||||
async def ticker():
|
||||
"""Asynchronous background ticker to measure event loop latency/freezes."""
|
||||
ticks = []
|
||||
for _ in range(15):
|
||||
ticks.append(time.time())
|
||||
await asyncio.sleep(0.1)
|
||||
return ticks
|
||||
|
||||
async def run_test():
|
||||
print("Starting concurrency verification test...")
|
||||
|
||||
# Start the ticker background task
|
||||
ticker_task = asyncio.create_task(ticker())
|
||||
|
||||
# Let ticker run for a few ticks
|
||||
await asyncio.sleep(0.3)
|
||||
|
||||
print("Calling get_system_metrics offloaded to background thread...")
|
||||
start_time = time.time()
|
||||
|
||||
# Query system metrics using to_thread (simulating FastAPI request)
|
||||
metrics = await asyncio.to_thread(get_system_metrics)
|
||||
|
||||
duration = time.time() - start_time
|
||||
print(f"get_system_metrics took: {duration:.4f}s")
|
||||
|
||||
# Wait for the ticker to complete
|
||||
ticks = await ticker_task
|
||||
|
||||
# Calculate gaps between consecutive ticks to check for event loop freezes
|
||||
gaps = [ticks[i+1] - ticks[i] for i in range(len(ticks)-1)]
|
||||
max_gap = max(gaps)
|
||||
|
||||
print(f"All tick gaps: {[round(g, 3) for g in gaps]}")
|
||||
print(f"Max event loop freeze: {max_gap:.4f}s")
|
||||
|
||||
# In pre-fix code, psutil.cpu_percent(interval=1) blocks for 1.0s,
|
||||
# causing a gap of >1.0s. With our fix, it should be close to 0.1s.
|
||||
if max_gap >= 0.35:
|
||||
print("FAIL: Event loop was frozen/blocked!")
|
||||
sys.exit(1)
|
||||
else:
|
||||
print("SUCCESS: Event loop remained fully responsive during metrics query!")
|
||||
sys.exit(0)
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(run_test())
|
||||
Reference in New Issue
Block a user