feat(frontend): update sessions page for enhanced health monitoring
- Add new status badges: starting, probing, unhealthy - Show tunnel error only when tunnel_status is unreachable - Show app error badge with status code for error_response - Add collapsible probe output section for diagnostics - Update health polling to check all active instances - Only show Recreate Tunnel button for unreachable tunnels
This commit is contained in:
@@ -30,11 +30,14 @@ from src.services.docker import (
|
||||
execute_compose_command,
|
||||
find_free_port,
|
||||
get_container_id,
|
||||
get_container_logs,
|
||||
get_container_name,
|
||||
get_container_status,
|
||||
recreate_tunnel,
|
||||
render_compose_template,
|
||||
start_cloudflared_tunnel,
|
||||
stop_cloudflared_tunnel,
|
||||
wait_for_container_running,
|
||||
write_compose_file,
|
||||
write_config_files,
|
||||
write_env_file,
|
||||
@@ -552,20 +555,67 @@ async def start_instance(
|
||||
else:
|
||||
logger.warning("Failed to connect %s to backend network", container_name)
|
||||
|
||||
instance.status = "starting"
|
||||
instance.last_started_at = datetime.now()
|
||||
await session.commit()
|
||||
logger.info("Instance %s container is running, checking readiness", instance.id)
|
||||
|
||||
# Verify container reached running state
|
||||
if instance.container_id:
|
||||
instance.status = "starting"
|
||||
instance.last_started_at = datetime.now()
|
||||
await session.commit()
|
||||
logger.info("Instance %s: verifying container startup...", instance.id)
|
||||
|
||||
startup_result = wait_for_container_running(instance.container_id, timeout=30, interval=2.0)
|
||||
|
||||
if not startup_result["success"]:
|
||||
# Container failed to start
|
||||
error_msg = f"Container failed to start: status={startup_result['status']}"
|
||||
if startup_result["exit_code"] is not None:
|
||||
error_msg += f", exit_code={startup_result['exit_code']}"
|
||||
|
||||
# Get logs for debugging
|
||||
logs = get_container_logs(instance.container_id, tail=50)
|
||||
|
||||
instance.status = "error"
|
||||
await session.commit()
|
||||
logger.error(
|
||||
"Instance %s container startup failed after %.1fs: %s\nLogs:\n%s",
|
||||
instance.id,
|
||||
startup_result["waited_seconds"],
|
||||
error_msg,
|
||||
logs,
|
||||
)
|
||||
return {
|
||||
"status": "error",
|
||||
"error": error_msg,
|
||||
"logs": logs,
|
||||
}
|
||||
|
||||
logger.info(
|
||||
"Instance %s container started successfully after %.1fs",
|
||||
instance.id,
|
||||
startup_result["waited_seconds"],
|
||||
)
|
||||
|
||||
# Execute readiness probe if configured
|
||||
tool_type = await session.get(ToolType, instance.tool_type_id)
|
||||
if tool_type and tool_type.readiness_probe:
|
||||
probe_config = tool_type.readiness_probe
|
||||
probe_command = probe_config.get("command", "")
|
||||
probe_timeout = probe_config.get("timeout", 30)
|
||||
probe_interval = probe_config.get("interval", 2)
|
||||
if tool_type and instance.container_id:
|
||||
# Determine probe command
|
||||
probe_command = None
|
||||
probe_timeout = 30
|
||||
probe_interval = 2
|
||||
|
||||
if probe_command and instance.container_id:
|
||||
if tool_type.readiness_probe:
|
||||
probe_config = tool_type.readiness_probe
|
||||
probe_command = probe_config.get("command", "")
|
||||
probe_timeout = probe_config.get("timeout", 30)
|
||||
probe_interval = probe_config.get("interval", 2)
|
||||
elif "web" in (tool_type.interfaces or []):
|
||||
# Default probe for web tools
|
||||
probe_command = f"curl -f http://localhost:{tool_type.default_port or 8080}"
|
||||
probe_timeout = 30
|
||||
probe_interval = 2
|
||||
|
||||
if probe_command:
|
||||
instance.status = "probing"
|
||||
await session.commit()
|
||||
logger.info(
|
||||
"Executing readiness probe for instance %s: command='%s', timeout=%d, interval=%d",
|
||||
instance.id, probe_command, probe_timeout, probe_interval
|
||||
@@ -578,14 +628,25 @@ async def start_instance(
|
||||
interval=probe_interval,
|
||||
)
|
||||
|
||||
# Store probe result
|
||||
instance.probe_result = {
|
||||
"success": success,
|
||||
"command": probe_command,
|
||||
"logs": probe_logs,
|
||||
"timestamp": datetime.now().isoformat(),
|
||||
}
|
||||
|
||||
if not success:
|
||||
instance.status = "failed"
|
||||
instance.url = None
|
||||
instance.public_url = None
|
||||
instance.status = "unhealthy"
|
||||
await session.commit()
|
||||
logger.error("Readiness probe failed for instance %s: %s", instance.id, "\n".join(probe_logs))
|
||||
logger.error(
|
||||
"Readiness probe failed for instance %s after %ds: %s",
|
||||
instance.id,
|
||||
probe_timeout,
|
||||
"\n".join(probe_logs),
|
||||
)
|
||||
return {
|
||||
"status": "failed",
|
||||
"status": "unhealthy",
|
||||
"error": f"Readiness probe failed after {probe_timeout}s",
|
||||
"probe_logs": probe_logs,
|
||||
}
|
||||
@@ -956,6 +1017,17 @@ async def recreate_tunnel_endpoint(
|
||||
detail="instance must be running to recreate tunnel",
|
||||
)
|
||||
|
||||
# Validate tunnel is actually broken before recreating
|
||||
if instance.url:
|
||||
tunnel_health = check_tunnel_health(instance.url)
|
||||
if tunnel_health["tunnel_status"] == "error_response":
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_400_BAD_REQUEST,
|
||||
detail=f"Tunnel is working but application returned HTTP {tunnel_health.get('status_code')}. Recreating the tunnel will not fix this issue.",
|
||||
)
|
||||
elif tunnel_health["tunnel_status"] == "healthy":
|
||||
return {"status": "healthy", "url": instance.url, "message": "Tunnel is already healthy"}
|
||||
|
||||
# Get tool type for default port
|
||||
tool_type = await session.get(ToolType, instance.tool_type_id)
|
||||
instance_port = tool_type.default_port if tool_type and tool_type.default_port else 8080
|
||||
@@ -987,8 +1059,8 @@ async def recreate_tunnel_endpoint(
|
||||
|
||||
@router.get(
|
||||
"/{project_id}/repositories/{repo_id}/instances/{instance_id}/health",
|
||||
summary="Check tunnel health",
|
||||
description="Check if the temporary Cloudflare tunnel for an instance is healthy.",
|
||||
summary="Check instance health",
|
||||
description="Check container and tunnel health for an instance.",
|
||||
)
|
||||
async def check_instance_tunnel_health(
|
||||
project_id: uuid.UUID,
|
||||
@@ -997,7 +1069,7 @@ async def check_instance_tunnel_health(
|
||||
user_id: uuid.UUID = Depends(get_current_user_id),
|
||||
session: AsyncSession = Depends(get_db_session),
|
||||
) -> dict:
|
||||
"""Check tunnel health for an instance.
|
||||
"""Check health for an instance (container + tunnel).
|
||||
|
||||
Args:
|
||||
project_id: UUID of the project.
|
||||
@@ -1007,7 +1079,7 @@ async def check_instance_tunnel_health(
|
||||
session: Database session.
|
||||
|
||||
Returns:
|
||||
Dictionary with health status.
|
||||
Dictionary with container_status, tunnel_status, probe_status, and overall healthy flag.
|
||||
"""
|
||||
_user = await _get_user(session, user_id)
|
||||
_project = await _get_owned_project(project_id, user_id, session)
|
||||
@@ -1018,11 +1090,50 @@ async def check_instance_tunnel_health(
|
||||
status_code=status.HTTP_404_NOT_FOUND, detail="instance not found"
|
||||
)
|
||||
|
||||
if not instance.url or instance.status != "running":
|
||||
return {"healthy": False, "status_code": None, "error": "instance not running"}
|
||||
# Check container status
|
||||
container_info = {"status": "not_found", "exit_code": None, "health": None}
|
||||
if instance.container_id:
|
||||
container_info = get_container_status(instance.container_id)
|
||||
|
||||
health = check_tunnel_health(instance.url)
|
||||
return health
|
||||
# Build response
|
||||
response = {
|
||||
"healthy": False,
|
||||
"container_status": container_info["status"],
|
||||
"container_health": container_info["health"],
|
||||
"tunnel_status": "not_applicable",
|
||||
"tunnel_status_code": None,
|
||||
"probe_status": "not_applicable",
|
||||
"last_probe_output": None,
|
||||
"error": None,
|
||||
}
|
||||
|
||||
# Determine probe status
|
||||
if instance.status == "probing":
|
||||
response["probe_status"] = "pending"
|
||||
elif instance.probe_result:
|
||||
response["probe_status"] = "success" if instance.probe_result.get("success") else "failed"
|
||||
response["last_probe_output"] = "\n".join(instance.probe_result.get("logs", []))[:500]
|
||||
|
||||
# Check tunnel health if instance has a URL and is web-enabled
|
||||
if instance.url and instance.status in ("running", "unhealthy"):
|
||||
tunnel_health = check_tunnel_health(instance.url)
|
||||
response["tunnel_status"] = tunnel_health["tunnel_status"]
|
||||
response["tunnel_status_code"] = tunnel_health.get("status_code")
|
||||
if tunnel_health.get("error"):
|
||||
response["error"] = tunnel_health["error"]
|
||||
|
||||
# Overall healthy only if container is running AND tunnel is healthy
|
||||
container_healthy = container_info["status"] == "running"
|
||||
tunnel_healthy = response["tunnel_status"] == "healthy"
|
||||
response["healthy"] = container_healthy and tunnel_healthy
|
||||
|
||||
# If container is not running, override error message
|
||||
if not container_healthy:
|
||||
response["error"] = f"Container is {container_info['status']}"
|
||||
if container_info["exit_code"] is not None:
|
||||
response["error"] += f" (exit code: {container_info['exit_code']})"
|
||||
|
||||
return response
|
||||
|
||||
|
||||
@router.get(
|
||||
|
||||
@@ -2,7 +2,7 @@ import uuid
|
||||
from datetime import datetime
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
from sqlalchemy import DateTime, ForeignKey, Integer, String
|
||||
from sqlalchemy import DateTime, ForeignKey, Integer, JSON, String
|
||||
from sqlalchemy import Uuid as UUID
|
||||
from sqlalchemy.orm import Mapped, mapped_column, relationship
|
||||
|
||||
@@ -62,6 +62,9 @@ class ToolInstance(UUIDPrimaryKeyMixin, TimestampMixin, Base):
|
||||
last_stopped_at: Mapped[datetime | None] = mapped_column(
|
||||
DateTime(timezone=True), nullable=True
|
||||
)
|
||||
probe_result: Mapped[dict | None] = mapped_column(
|
||||
JSON, nullable=True
|
||||
)
|
||||
|
||||
tool_type: Mapped["ToolType"] = relationship()
|
||||
repository: Mapped["GitRepository"] = relationship()
|
||||
|
||||
+121
-14
@@ -244,24 +244,94 @@ def connect_container_to_network(container_name: str, network_name: str = "backe
|
||||
return result.returncode == 0
|
||||
|
||||
|
||||
def get_container_status(container_id: str) -> str:
|
||||
def get_container_status(container_id: str) -> dict[str, Any]:
|
||||
"""Get the status of a Docker container.
|
||||
|
||||
Args:
|
||||
container_id: Docker container ID
|
||||
|
||||
Returns:
|
||||
Container status string (running, exited, etc.)
|
||||
Dict with 'status' (running, exited, restarting, not_found),
|
||||
'exit_code' (int or None), and 'health' (health status or None)
|
||||
"""
|
||||
result = subprocess.run(
|
||||
["docker", "inspect", "-f", "{{.State.Status}}", container_id],
|
||||
[
|
||||
"docker", "inspect", "-f",
|
||||
"{{.State.Status}}|{{.State.ExitCode}}|{{if .State.Health}}{{.State.Health.Status}}{{else}}none{{end}}",
|
||||
container_id,
|
||||
],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
)
|
||||
|
||||
if result.returncode == 0:
|
||||
return result.stdout.strip()
|
||||
return "unknown"
|
||||
if result.returncode != 0:
|
||||
return {"status": "not_found", "exit_code": None, "health": None}
|
||||
|
||||
parts = result.stdout.strip().split("|")
|
||||
status = parts[0] if parts else "unknown"
|
||||
exit_code = int(parts[1]) if len(parts) > 1 and parts[1].isdigit() else None
|
||||
health = parts[2] if len(parts) > 2 and parts[2] != "none" else None
|
||||
|
||||
return {"status": status, "exit_code": exit_code, "health": health}
|
||||
|
||||
|
||||
def wait_for_container_running(
|
||||
container_id: str, timeout: int = 30, interval: float = 2.0
|
||||
) -> dict[str, Any]:
|
||||
"""Wait for a container to reach the running state.
|
||||
|
||||
Polls docker inspect until the container status is "running" or timeout.
|
||||
|
||||
Args:
|
||||
container_id: Docker container ID
|
||||
timeout: Maximum seconds to wait
|
||||
interval: Seconds between polls
|
||||
|
||||
Returns:
|
||||
Dict with 'success' (bool), 'status' (str), 'exit_code' (int or None),
|
||||
and 'waited_seconds' (float)
|
||||
"""
|
||||
import time
|
||||
|
||||
start_time = time.time()
|
||||
|
||||
while time.time() - start_time < timeout:
|
||||
info = get_container_status(container_id)
|
||||
|
||||
if info["status"] == "running":
|
||||
return {
|
||||
"success": True,
|
||||
"status": "running",
|
||||
"exit_code": None,
|
||||
"waited_seconds": time.time() - start_time,
|
||||
}
|
||||
|
||||
if info["status"] == "exited":
|
||||
return {
|
||||
"success": False,
|
||||
"status": "exited",
|
||||
"exit_code": info["exit_code"],
|
||||
"waited_seconds": time.time() - start_time,
|
||||
}
|
||||
|
||||
if info["status"] == "not_found":
|
||||
return {
|
||||
"success": False,
|
||||
"status": "not_found",
|
||||
"exit_code": None,
|
||||
"waited_seconds": time.time() - start_time,
|
||||
}
|
||||
|
||||
time.sleep(interval)
|
||||
|
||||
# Timeout reached
|
||||
info = get_container_status(container_id)
|
||||
return {
|
||||
"success": False,
|
||||
"status": info["status"],
|
||||
"exit_code": info["exit_code"],
|
||||
"waited_seconds": time.time() - start_time,
|
||||
}
|
||||
|
||||
|
||||
def get_container_logs(container_id: str, tail: int = 100) -> str:
|
||||
@@ -424,14 +494,15 @@ def recreate_tunnel(
|
||||
|
||||
|
||||
def check_tunnel_health(url: str, timeout: int = 10) -> dict[str, Any]:
|
||||
"""Check if a tunnel URL is healthy.
|
||||
"""Check if a tunnel URL is healthy with smart error classification.
|
||||
|
||||
Args:
|
||||
url: The tunnel URL to check
|
||||
timeout: Request timeout in seconds
|
||||
|
||||
Returns:
|
||||
Dict with 'healthy' (bool) and 'status_code' (int or None)
|
||||
Dict with 'tunnel_status' (healthy, unreachable, error_response, not_applicable),
|
||||
'status_code' (int or None), 'healthy' (bool), and 'error' (str or None)
|
||||
"""
|
||||
import subprocess
|
||||
|
||||
@@ -444,13 +515,49 @@ def check_tunnel_health(url: str, timeout: int = 10) -> dict[str, Any]:
|
||||
timeout=timeout + 5,
|
||||
)
|
||||
status_code = int(result.stdout.strip())
|
||||
|
||||
if 200 <= status_code < 400:
|
||||
return {
|
||||
"tunnel_status": "healthy",
|
||||
"status_code": status_code,
|
||||
"healthy": True,
|
||||
"error": None,
|
||||
}
|
||||
elif status_code in (502, 503, 504):
|
||||
# Application error, not tunnel error
|
||||
return {
|
||||
"tunnel_status": "error_response",
|
||||
"status_code": status_code,
|
||||
"healthy": False,
|
||||
"error": f"Application returned HTTP {status_code}",
|
||||
}
|
||||
else:
|
||||
return {
|
||||
"tunnel_status": "error_response",
|
||||
"status_code": status_code,
|
||||
"healthy": False,
|
||||
"error": f"HTTP {status_code}",
|
||||
}
|
||||
except subprocess.TimeoutExpired:
|
||||
return {
|
||||
"healthy": 200 <= status_code < 400,
|
||||
"status_code": status_code,
|
||||
}
|
||||
except (ValueError, subprocess.TimeoutExpired, Exception) as e:
|
||||
return {
|
||||
"healthy": False,
|
||||
"tunnel_status": "unreachable",
|
||||
"status_code": None,
|
||||
"healthy": False,
|
||||
"error": "Tunnel request timed out",
|
||||
}
|
||||
except (ValueError, Exception) as e:
|
||||
error_str = str(e).lower()
|
||||
# Classify connection errors
|
||||
if any(err in error_str for err in ["connection refused", "econnrefused", "could not resolve", "nodename"]):
|
||||
return {
|
||||
"tunnel_status": "unreachable",
|
||||
"status_code": None,
|
||||
"healthy": False,
|
||||
"error": f"Tunnel unreachable: {e}",
|
||||
}
|
||||
return {
|
||||
"tunnel_status": "unreachable",
|
||||
"status_code": None,
|
||||
"healthy": False,
|
||||
"error": str(e),
|
||||
}
|
||||
|
||||
@@ -25,6 +25,8 @@ export interface Session {
|
||||
project_id: string;
|
||||
status: string;
|
||||
url: string | null;
|
||||
container_status?: string;
|
||||
probe_status?: string;
|
||||
}
|
||||
|
||||
export async function listInstances(
|
||||
@@ -101,11 +103,23 @@ export async function getUserSessions(): Promise<Session[]> {
|
||||
return response.data.sessions;
|
||||
}
|
||||
|
||||
export interface InstanceHealth {
|
||||
healthy: boolean;
|
||||
container_status: string;
|
||||
container_health?: string;
|
||||
container_exit_code?: number | null;
|
||||
tunnel_status: string;
|
||||
tunnel_status_code: number | null;
|
||||
probe_status: string;
|
||||
last_probe_output?: string;
|
||||
error?: string;
|
||||
}
|
||||
|
||||
export async function checkInstanceHealth(
|
||||
projectId: string,
|
||||
repoId: string,
|
||||
instanceId: string
|
||||
): Promise<{ healthy: boolean; status_code: number | null; error?: string }> {
|
||||
): Promise<InstanceHealth> {
|
||||
const response = await apiClient.get(
|
||||
`/projects/${projectId}/repositories/${repoId}/instances/${instanceId}/health`
|
||||
);
|
||||
|
||||
@@ -40,8 +40,18 @@ export const SessionsPage = () => {
|
||||
|
||||
const [deleteConfirmId, setDeleteConfirmId] = useState<string | null>(null);
|
||||
const [stopConfirmId, setStopConfirmId] = useState<string | null>(null);
|
||||
const [tunnelHealth, setTunnelHealth] = useState<Record<string, { healthy: boolean; status_code: number | null; error?: string }>>({});
|
||||
const [tunnelHealth, setTunnelHealth] = useState<Record<string, {
|
||||
healthy: boolean;
|
||||
container_status: string;
|
||||
container_health: string | null;
|
||||
tunnel_status: string;
|
||||
tunnel_status_code: number | null;
|
||||
probe_status: string;
|
||||
last_probe_output: string | null;
|
||||
error: string | null;
|
||||
}>>({});
|
||||
const [recreatingId, setRecreatingId] = useState<string | null>(null);
|
||||
const [expandedProbeId, setExpandedProbeId] = useState<string | null>(null);
|
||||
|
||||
const loadSessions = useCallback(async () => {
|
||||
setStatus("loading");
|
||||
@@ -86,13 +96,13 @@ export const SessionsPage = () => {
|
||||
void loadToolTypes();
|
||||
}, []);
|
||||
|
||||
// Poll tunnel health every 30 seconds for running instances
|
||||
// Poll health every 30 seconds for active instances
|
||||
useEffect(() => {
|
||||
const checkHealth = async () => {
|
||||
const runningSessions = sessions.filter(
|
||||
(s) => s.status === "running" && s.url
|
||||
const activeSessions = sessions.filter(
|
||||
(s) => ["running", "starting", "unhealthy", "probing"].includes(s.status)
|
||||
);
|
||||
for (const session of runningSessions) {
|
||||
for (const session of activeSessions) {
|
||||
try {
|
||||
const health = await checkInstanceHealth(
|
||||
session.project_id,
|
||||
@@ -106,7 +116,14 @@ export const SessionsPage = () => {
|
||||
} catch {
|
||||
setTunnelHealth((prev) => ({
|
||||
...prev,
|
||||
[session.id]: { healthy: false, status_code: null, error: "check failed" },
|
||||
[session.id]: {
|
||||
healthy: false,
|
||||
container_status: "unknown",
|
||||
tunnel_status: "unreachable",
|
||||
tunnel_status_code: null,
|
||||
probe_status: "unknown",
|
||||
error: "check failed",
|
||||
},
|
||||
}));
|
||||
}
|
||||
}
|
||||
@@ -135,7 +152,7 @@ export const SessionsPage = () => {
|
||||
}, [selectedProject]);
|
||||
|
||||
const activeSessions = useMemo(
|
||||
() => sessions.filter((s) => ["running", "building", "pending"].includes(s.status)),
|
||||
() => sessions.filter((s) => ["running", "building", "pending", "starting", "probing", "unhealthy"].includes(s.status)),
|
||||
[sessions]
|
||||
);
|
||||
|
||||
@@ -328,9 +345,35 @@ export const SessionsPage = () => {
|
||||
</p>
|
||||
)}
|
||||
<span className={`status-badge ${session.status}`}>{session.status}</span>
|
||||
{tunnelHealth[session.id] && !tunnelHealth[session.id].healthy && (
|
||||
{session.status === "starting" && (
|
||||
<span className="status-badge starting">starting...</span>
|
||||
)}
|
||||
{session.status === "probing" && (
|
||||
<span className="status-badge probing">checking...</span>
|
||||
)}
|
||||
{tunnelHealth[session.id] && tunnelHealth[session.id].tunnel_status === "unreachable" && (
|
||||
<span className="status-badge error">tunnel error</span>
|
||||
)}
|
||||
{tunnelHealth[session.id] && tunnelHealth[session.id].tunnel_status === "error_response" && (
|
||||
<span className="status-badge warning">app error ({tunnelHealth[session.id].tunnel_status_code})</span>
|
||||
)}
|
||||
{tunnelHealth[session.id]?.last_probe_output && (
|
||||
<div className="probe-output-section">
|
||||
<button
|
||||
className="probe-toggle"
|
||||
onClick={() => setExpandedProbeId(expandedProbeId === session.id ? null : session.id)}
|
||||
type="button"
|
||||
>
|
||||
<Icon name="info" size="sm" />
|
||||
{expandedProbeId === session.id ? "Hide probe output" : "Show probe output"}
|
||||
</button>
|
||||
{expandedProbeId === session.id && (
|
||||
<pre className="probe-output">
|
||||
{tunnelHealth[session.id].last_probe_output}
|
||||
</pre>
|
||||
)}
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
<div className="session-actions">
|
||||
{session.url ? (
|
||||
@@ -353,7 +396,7 @@ export const SessionsPage = () => {
|
||||
Open
|
||||
</button>
|
||||
)}
|
||||
{tunnelHealth[session.id] && !tunnelHealth[session.id].healthy && (
|
||||
{tunnelHealth[session.id] && tunnelHealth[session.id].tunnel_status === "unreachable" && (
|
||||
<button
|
||||
className="secondary-button small"
|
||||
onClick={() => void handleRecreateTunnel(session)}
|
||||
|
||||
Reference in New Issue
Block a user