merge: integrate session management fixes and sessions hub

This commit is contained in:
2026-05-22 21:39:37 +02:00
29 changed files with 1108 additions and 288 deletions
+135 -24
View File
@@ -30,11 +30,14 @@ from src.services.docker import (
execute_compose_command,
find_free_port,
get_container_id,
get_container_logs,
get_container_name,
get_container_status,
recreate_tunnel,
render_compose_template,
start_cloudflared_tunnel,
stop_cloudflared_tunnel,
wait_for_container_running,
write_compose_file,
write_config_files,
write_env_file,
@@ -552,20 +555,67 @@ async def start_instance(
else:
logger.warning("Failed to connect %s to backend network", container_name)
instance.status = "starting"
instance.last_started_at = datetime.now()
await session.commit()
logger.info("Instance %s container is running, checking readiness", instance.id)
# Verify container reached running state
if instance.container_id:
instance.status = "starting"
instance.last_started_at = datetime.now()
await session.commit()
logger.info("Instance %s: verifying container startup...", instance.id)
startup_result = wait_for_container_running(instance.container_id, timeout=30, interval=2.0)
if not startup_result["success"]:
# Container failed to start
error_msg = f"Container failed to start: status={startup_result['status']}"
if startup_result["exit_code"] is not None:
error_msg += f", exit_code={startup_result['exit_code']}"
# Get logs for debugging
logs = get_container_logs(instance.container_id, tail=50)
instance.status = "error"
await session.commit()
logger.error(
"Instance %s container startup failed after %.1fs: %s\nLogs:\n%s",
instance.id,
startup_result["waited_seconds"],
error_msg,
logs,
)
return {
"status": "error",
"error": error_msg,
"logs": logs,
}
logger.info(
"Instance %s container started successfully after %.1fs",
instance.id,
startup_result["waited_seconds"],
)
# Execute readiness probe if configured
tool_type = await session.get(ToolType, instance.tool_type_id)
if tool_type and tool_type.readiness_probe:
probe_config = tool_type.readiness_probe
probe_command = probe_config.get("command", "")
probe_timeout = probe_config.get("timeout", 30)
probe_interval = probe_config.get("interval", 2)
if tool_type and instance.container_id:
# Determine probe command
probe_command = None
probe_timeout = 30
probe_interval = 2
if probe_command and instance.container_id:
if tool_type.readiness_probe:
probe_config = tool_type.readiness_probe
probe_command = probe_config.get("command", "")
probe_timeout = probe_config.get("timeout", 30)
probe_interval = probe_config.get("interval", 2)
elif "web" in (tool_type.interfaces or []):
# Default probe for web tools
probe_command = f"curl -f http://localhost:{tool_type.default_port or 8080}"
probe_timeout = 30
probe_interval = 2
if probe_command:
instance.status = "probing"
await session.commit()
logger.info(
"Executing readiness probe for instance %s: command='%s', timeout=%d, interval=%d",
instance.id, probe_command, probe_timeout, probe_interval
@@ -578,14 +628,25 @@ async def start_instance(
interval=probe_interval,
)
# Store probe result
instance.probe_result = {
"success": success,
"command": probe_command,
"logs": probe_logs,
"timestamp": datetime.now().isoformat(),
}
if not success:
instance.status = "failed"
instance.url = None
instance.public_url = None
instance.status = "unhealthy"
await session.commit()
logger.error("Readiness probe failed for instance %s: %s", instance.id, "\n".join(probe_logs))
logger.error(
"Readiness probe failed for instance %s after %ds: %s",
instance.id,
probe_timeout,
"\n".join(probe_logs),
)
return {
"status": "failed",
"status": "unhealthy",
"error": f"Readiness probe failed after {probe_timeout}s",
"probe_logs": probe_logs,
}
@@ -956,6 +1017,17 @@ async def recreate_tunnel_endpoint(
detail="instance must be running to recreate tunnel",
)
# Validate tunnel is actually broken before recreating
if instance.url:
tunnel_health = check_tunnel_health(instance.url)
if tunnel_health["tunnel_status"] == "error_response":
raise HTTPException(
status_code=status.HTTP_400_BAD_REQUEST,
detail=f"Tunnel is working but application returned HTTP {tunnel_health.get('status_code')}. Recreating the tunnel will not fix this issue.",
)
elif tunnel_health["tunnel_status"] == "healthy":
return {"status": "healthy", "url": instance.url, "message": "Tunnel is already healthy"}
# Get tool type for default port
tool_type = await session.get(ToolType, instance.tool_type_id)
instance_port = tool_type.default_port if tool_type and tool_type.default_port else 8080
@@ -987,8 +1059,8 @@ async def recreate_tunnel_endpoint(
@router.get(
"/{project_id}/repositories/{repo_id}/instances/{instance_id}/health",
summary="Check tunnel health",
description="Check if the temporary Cloudflare tunnel for an instance is healthy.",
summary="Check instance health",
description="Check container and tunnel health for an instance.",
)
async def check_instance_tunnel_health(
project_id: uuid.UUID,
@@ -997,7 +1069,7 @@ async def check_instance_tunnel_health(
user_id: uuid.UUID = Depends(get_current_user_id),
session: AsyncSession = Depends(get_db_session),
) -> dict:
"""Check tunnel health for an instance.
"""Check health for an instance (container + tunnel).
Args:
project_id: UUID of the project.
@@ -1007,7 +1079,7 @@ async def check_instance_tunnel_health(
session: Database session.
Returns:
Dictionary with health status.
Dictionary with container_status, tunnel_status, probe_status, and overall healthy flag.
"""
_user = await _get_user(session, user_id)
_project = await _get_owned_project(project_id, user_id, session)
@@ -1018,11 +1090,50 @@ async def check_instance_tunnel_health(
status_code=status.HTTP_404_NOT_FOUND, detail="instance not found"
)
if not instance.url or instance.status != "running":
return {"healthy": False, "status_code": None, "error": "instance not running"}
# Check container status
container_info = {"status": "not_found", "exit_code": None, "health": None}
if instance.container_id:
container_info = get_container_status(instance.container_id)
health = check_tunnel_health(instance.url)
return health
# Build response
response = {
"healthy": False,
"container_status": container_info["status"],
"container_health": container_info["health"],
"tunnel_status": "not_applicable",
"tunnel_status_code": None,
"probe_status": "not_applicable",
"last_probe_output": None,
"error": None,
}
# Determine probe status
if instance.status == "probing":
response["probe_status"] = "pending"
elif instance.probe_result:
response["probe_status"] = "success" if instance.probe_result.get("success") else "failed"
response["last_probe_output"] = "\n".join(instance.probe_result.get("logs", []))[:500]
# Check tunnel health if instance has a URL and is web-enabled
if instance.url and instance.status in ("running", "unhealthy"):
tunnel_health = check_tunnel_health(instance.url)
response["tunnel_status"] = tunnel_health["tunnel_status"]
response["tunnel_status_code"] = tunnel_health.get("status_code")
if tunnel_health.get("error"):
response["error"] = tunnel_health["error"]
# Overall healthy only if container is running AND tunnel is healthy
container_healthy = container_info["status"] == "running"
tunnel_healthy = response["tunnel_status"] == "healthy"
response["healthy"] = container_healthy and tunnel_healthy
# If container is not running, override error message
if not container_healthy:
response["error"] = f"Container is {container_info['status']}"
if container_info["exit_code"] is not None:
response["error"] += f" (exit code: {container_info['exit_code']})"
return response
@router.get(
+4 -1
View File
@@ -2,7 +2,7 @@ import uuid
from datetime import datetime
from typing import TYPE_CHECKING
from sqlalchemy import DateTime, ForeignKey, Integer, String
from sqlalchemy import DateTime, ForeignKey, Integer, JSON, String
from sqlalchemy import Uuid as UUID
from sqlalchemy.orm import Mapped, mapped_column, relationship
@@ -62,6 +62,9 @@ class ToolInstance(UUIDPrimaryKeyMixin, TimestampMixin, Base):
last_stopped_at: Mapped[datetime | None] = mapped_column(
DateTime(timezone=True), nullable=True
)
probe_result: Mapped[dict | None] = mapped_column(
JSON, nullable=True
)
tool_type: Mapped["ToolType"] = relationship()
repository: Mapped["GitRepository"] = relationship()
+121 -14
View File
@@ -244,24 +244,94 @@ def connect_container_to_network(container_name: str, network_name: str = "backe
return result.returncode == 0
def get_container_status(container_id: str) -> str:
def get_container_status(container_id: str) -> dict[str, Any]:
"""Get the status of a Docker container.
Args:
container_id: Docker container ID
Returns:
Container status string (running, exited, etc.)
Dict with 'status' (running, exited, restarting, not_found),
'exit_code' (int or None), and 'health' (health status or None)
"""
result = subprocess.run(
["docker", "inspect", "-f", "{{.State.Status}}", container_id],
[
"docker", "inspect", "-f",
"{{.State.Status}}|{{.State.ExitCode}}|{{if .State.Health}}{{.State.Health.Status}}{{else}}none{{end}}",
container_id,
],
capture_output=True,
text=True,
)
if result.returncode == 0:
return result.stdout.strip()
return "unknown"
if result.returncode != 0:
return {"status": "not_found", "exit_code": None, "health": None}
parts = result.stdout.strip().split("|")
status = parts[0] if parts else "unknown"
exit_code = int(parts[1]) if len(parts) > 1 and parts[1].isdigit() else None
health = parts[2] if len(parts) > 2 and parts[2] != "none" else None
return {"status": status, "exit_code": exit_code, "health": health}
def wait_for_container_running(
container_id: str, timeout: int = 30, interval: float = 2.0
) -> dict[str, Any]:
"""Wait for a container to reach the running state.
Polls docker inspect until the container status is "running" or timeout.
Args:
container_id: Docker container ID
timeout: Maximum seconds to wait
interval: Seconds between polls
Returns:
Dict with 'success' (bool), 'status' (str), 'exit_code' (int or None),
and 'waited_seconds' (float)
"""
import time
start_time = time.time()
while time.time() - start_time < timeout:
info = get_container_status(container_id)
if info["status"] == "running":
return {
"success": True,
"status": "running",
"exit_code": None,
"waited_seconds": time.time() - start_time,
}
if info["status"] == "exited":
return {
"success": False,
"status": "exited",
"exit_code": info["exit_code"],
"waited_seconds": time.time() - start_time,
}
if info["status"] == "not_found":
return {
"success": False,
"status": "not_found",
"exit_code": None,
"waited_seconds": time.time() - start_time,
}
time.sleep(interval)
# Timeout reached
info = get_container_status(container_id)
return {
"success": False,
"status": info["status"],
"exit_code": info["exit_code"],
"waited_seconds": time.time() - start_time,
}
def get_container_logs(container_id: str, tail: int = 100) -> str:
@@ -424,14 +494,15 @@ def recreate_tunnel(
def check_tunnel_health(url: str, timeout: int = 10) -> dict[str, Any]:
"""Check if a tunnel URL is healthy.
"""Check if a tunnel URL is healthy with smart error classification.
Args:
url: The tunnel URL to check
timeout: Request timeout in seconds
Returns:
Dict with 'healthy' (bool) and 'status_code' (int or None)
Dict with 'tunnel_status' (healthy, unreachable, error_response, not_applicable),
'status_code' (int or None), 'healthy' (bool), and 'error' (str or None)
"""
import subprocess
@@ -444,13 +515,49 @@ def check_tunnel_health(url: str, timeout: int = 10) -> dict[str, Any]:
timeout=timeout + 5,
)
status_code = int(result.stdout.strip())
if 200 <= status_code < 400:
return {
"tunnel_status": "healthy",
"status_code": status_code,
"healthy": True,
"error": None,
}
elif status_code in (502, 503, 504):
# Application error, not tunnel error
return {
"tunnel_status": "error_response",
"status_code": status_code,
"healthy": False,
"error": f"Application returned HTTP {status_code}",
}
else:
return {
"tunnel_status": "error_response",
"status_code": status_code,
"healthy": False,
"error": f"HTTP {status_code}",
}
except subprocess.TimeoutExpired:
return {
"healthy": 200 <= status_code < 400,
"status_code": status_code,
}
except (ValueError, subprocess.TimeoutExpired, Exception) as e:
return {
"healthy": False,
"tunnel_status": "unreachable",
"status_code": None,
"healthy": False,
"error": "Tunnel request timed out",
}
except (ValueError, Exception) as e:
error_str = str(e).lower()
# Classify connection errors
if any(err in error_str for err in ["connection refused", "econnrefused", "could not resolve", "nodename"]):
return {
"tunnel_status": "unreachable",
"status_code": None,
"healthy": False,
"error": f"Tunnel unreachable: {e}",
}
return {
"tunnel_status": "unreachable",
"status_code": None,
"healthy": False,
"error": str(e),
}
+15 -1
View File
@@ -25,6 +25,8 @@ export interface Session {
project_id: string;
status: string;
url: string | null;
container_status?: string;
probe_status?: string;
}
export async function listInstances(
@@ -101,11 +103,23 @@ export async function getUserSessions(): Promise<Session[]> {
return response.data.sessions;
}
export interface InstanceHealth {
healthy: boolean;
container_status: string;
container_health: string | null;
container_exit_code: number | null;
tunnel_status: string;
tunnel_status_code: number | null;
probe_status: string;
last_probe_output: string | null;
error: string | null;
}
export async function checkInstanceHealth(
projectId: string,
repoId: string,
instanceId: string
): Promise<{ healthy: boolean; status_code: number | null; error?: string }> {
): Promise<InstanceHealth> {
const response = await apiClient.get(
`/projects/${projectId}/repositories/${repoId}/instances/${instanceId}/health`
);
+3 -3
View File
@@ -9,8 +9,9 @@ import { useSessions } from "../state/sessions";
import { Icon } from "./icon";
import type { IconName } from "../utils/icons";
const NAV_ITEMS: { to: string; label: string; icon: IconName }[] = [
const NAV_ITEMS: { to: string; label: string; icon: IconName; badge?: "sessions" }[] = [
{ to: "/", label: "Home", icon: "dashboard" },
{ to: "/sessions", label: "Sessions", icon: "terminal", badge: "sessions" },
{ to: "/projects", label: "Projects", icon: "projects" },
{ to: "/tool-workshop", label: "Tool Workshop", icon: "settings" },
{ to: "/settings", label: "Settings", icon: "settings" }
@@ -83,7 +84,6 @@ export const AppShell = () => {
<div className="shell-body">
<aside className="shell-nav" aria-label="Primary navigation">
{NAV_ITEMS.map((item) => {
const isHome = item.to === "/";
const activeCount = sessions.filter((s) => s.status === "running").length;
return (
<NavLink
@@ -94,7 +94,7 @@ export const AppShell = () => {
>
<Icon name={item.icon} size="sm" />
{item.label}
{isHome && activeCount > 0 && (
{item.badge === "sessions" && activeCount > 0 && (
<span className="nav-badge">{activeCount}</span>
)}
</NavLink>
+118 -31
View File
@@ -2,7 +2,7 @@ import { useCallback, useEffect, useMemo, useState } from "react";
import { useNavigate } from "react-router-dom";
import { getDashboardSummary, type DashboardSummary } from "../api/dashboard";
import { createInstance, getUserSessions, startInstance, stopInstance, deleteInstance, recreateInstanceTunnel, type Session as SessionApi } from "../api/sessions";
import { createInstance, getUserSessions, startInstance, stopInstance, deleteInstance, recreateInstanceTunnel, checkInstanceHealth, type Session as SessionApi, type InstanceHealth } from "../api/sessions";
import { listProjects } from "../api/projects";
import { listRepositories, type GitRepository } from "../api/git_repositories";
import { listToolTypes, type ToolType } from "../api/tool_types";
@@ -34,6 +34,9 @@ export const HomePage = () => {
const [displayName, setDisplayName] = useState("");
const [saveState, setSaveState] = useState<"idle" | "saving" | "error">("idle");
const [actionBusy, setActionBusy] = useState<string | null>(null);
const [stopConfirmId, setStopConfirmId] = useState<string | null>(null);
const [deleteConfirmId, setDeleteConfirmId] = useState<string | null>(null);
const [tunnelHealth, setTunnelHealth] = useState<Record<string, InstanceHealth>>({});
const safeSessions = Array.isArray(sessions) ? sessions : [];
const loadHome = useCallback(async () => {
@@ -59,6 +62,49 @@ export const HomePage = () => {
void loadHome();
}, [loadHome]);
// Poll tunnel health every 30 seconds for running instances
useEffect(() => {
const checkHealth = async () => {
const runningSessions = safeSessions.filter(
(s) => s.status === "running" && s.url
);
for (const session of runningSessions) {
try {
const health = await checkInstanceHealth(
session.project_id,
session.repository_id,
session.id
);
setTunnelHealth((prev) => ({
...prev,
[session.id]: health,
}));
} catch {
setTunnelHealth((prev) => ({
...prev,
[session.id]: {
healthy: false,
container_status: "unknown",
container_health: null,
container_exit_code: null,
tunnel_status: "error",
tunnel_status_code: null,
probe_status: "error",
last_probe_output: null,
error: "check failed",
},
}));
}
}
};
void checkHealth();
const interval = setInterval(() => {
void checkHealth();
}, 30000);
return () => clearInterval(interval);
}, [safeSessions]);
useEffect(() => {
if (!selectedProject) {
setRepositories([]);
@@ -120,7 +166,12 @@ export const HomePage = () => {
};
const handleStop = async (session: SessionView) => {
if (stopConfirmId !== session.id) {
setStopConfirmId(session.id);
return;
}
setActionBusy(session.id);
setStopConfirmId(null);
try {
await stopInstance(session.project_id, session.repository_id, session.id);
await loadHome();
@@ -130,10 +181,17 @@ export const HomePage = () => {
};
const handleDelete = async (session: SessionView) => {
if (deleteConfirmId !== session.id) {
setDeleteConfirmId(session.id);
return;
}
setActionBusy(session.id);
setDeleteConfirmId(null);
try {
await deleteInstance(session.project_id, session.repository_id, session.id);
await loadHome();
setSessions((prev) => prev.filter((s) => s.id !== session.id));
} catch {
// error - session remains in state
} finally {
setActionBusy(null);
}
@@ -203,36 +261,65 @@ export const HomePage = () => {
<p className="muted">No active sessions right now.</p>
) : (
<div className="home-session-grid">
{activeSessions.map((session) => (
<article className="card session-card" key={session.id}>
<div className="stack-sm">
<div className="row row-tight">
<h3>{session.display_name}</h3>
<span className={`status-badge ${session.status}`}>{session.status}</span>
{activeSessions.map((session) => {
const health = tunnelHealth[session.id];
const isUnhealthy = health && !health.healthy;
return (
<article className="card session-card" key={session.id}>
<div className="stack-sm">
<div className="row row-tight">
<h3>{session.display_name}</h3>
<div className="row row-tight">
{isUnhealthy && (
<span className="status-badge error" title={health.error || "unhealthy"}>!</span>
)}
<span className={`status-badge ${session.status}`}>{session.status}</span>
</div>
</div>
<p className="muted">{session.project_name} · {session.repository_name}</p>
<p className="muted">{session.tool_type_name}</p>
</div>
<p className="muted">{session.project_name} · {session.repository_name}</p>
<p className="muted">{session.tool_type_name}</p>
</div>
<div className="session-actions">
<button className="secondary-button small" type="button" onClick={() => handleOpen(session)}>
<Icon name="external" size="sm" />
Open
</button>
<button className="ghost-button small" type="button" onClick={() => void handleRecreateTunnel(session)} disabled={actionBusy === session.id}>
<Icon name="refresh" size="sm" />
Tunnel
</button>
<button className="ghost-button small" type="button" onClick={() => void handleStop(session)} disabled={actionBusy === session.id}>
<Icon name="stop" size="sm" />
Stop
</button>
<button className="ghost-button small danger-text" type="button" onClick={() => void handleDelete(session)} disabled={actionBusy === session.id}>
<Icon name="delete" size="sm" />
Delete
</button>
</div>
</article>
))}
<div className="session-actions">
<button className="secondary-button small" type="button" onClick={() => handleOpen(session)}>
<Icon name="external" size="sm" />
Open
</button>
<button className="ghost-button small" type="button" onClick={() => void handleRecreateTunnel(session)} disabled={actionBusy === session.id}>
<Icon name="refresh" size="sm" />
Tunnel
</button>
{stopConfirmId === session.id ? (
<div className="stop-confirm-inline">
<span className="confirm-text">Stop?</span>
<button className="ghost-button small danger-text" type="button" onClick={() => void handleStop(session)} disabled={actionBusy === session.id}>
<Icon name="stop" size="sm" /> Stop
</button>
<button className="ghost-button small" type="button" onClick={() => setStopConfirmId(null)}>Cancel</button>
</div>
) : (
<button className="ghost-button small" type="button" onClick={() => void handleStop(session)} disabled={actionBusy === session.id}>
<Icon name="stop" size="sm" />
Stop
</button>
)}
{deleteConfirmId === session.id ? (
<div className="delete-confirm-inline">
<span className="confirm-text">Delete?</span>
<button className="ghost-button small danger-text" type="button" onClick={() => void handleDelete(session)} disabled={actionBusy === session.id}>
<Icon name="delete" size="sm" /> Delete
</button>
<button className="ghost-button small" type="button" onClick={() => setDeleteConfirmId(null)}>Cancel</button>
</div>
) : (
<button className="ghost-button small danger-text" type="button" onClick={() => void handleDelete(session)} disabled={actionBusy === session.id}>
<Icon name="delete" size="sm" />
Delete
</button>
)}
</div>
</article>
);
})}
</div>
)}
</section>
+54 -9
View File
@@ -40,8 +40,18 @@ export const SessionsPage = () => {
const [deleteConfirmId, setDeleteConfirmId] = useState<string | null>(null);
const [stopConfirmId, setStopConfirmId] = useState<string | null>(null);
const [tunnelHealth, setTunnelHealth] = useState<Record<string, { healthy: boolean; status_code: number | null; error?: string }>>({});
const [tunnelHealth, setTunnelHealth] = useState<Record<string, {
healthy: boolean;
container_status: string;
container_health: string | null;
tunnel_status: string;
tunnel_status_code: number | null;
probe_status: string;
last_probe_output: string | null;
error: string | null;
}>>({});
const [recreatingId, setRecreatingId] = useState<string | null>(null);
const [expandedProbeId, setExpandedProbeId] = useState<string | null>(null);
const loadSessions = useCallback(async () => {
setStatus("loading");
@@ -86,13 +96,13 @@ export const SessionsPage = () => {
void loadToolTypes();
}, []);
// Poll tunnel health every 30 seconds for running instances
// Poll health every 30 seconds for active instances
useEffect(() => {
const checkHealth = async () => {
const runningSessions = sessions.filter(
(s) => s.status === "running" && s.url
const activeSessions = sessions.filter(
(s) => ["running", "starting", "unhealthy", "probing"].includes(s.status)
);
for (const session of runningSessions) {
for (const session of activeSessions) {
try {
const health = await checkInstanceHealth(
session.project_id,
@@ -106,7 +116,16 @@ export const SessionsPage = () => {
} catch {
setTunnelHealth((prev) => ({
...prev,
[session.id]: { healthy: false, status_code: null, error: "check failed" },
[session.id]: {
healthy: false,
container_status: "unknown",
container_health: null,
tunnel_status: "unreachable",
tunnel_status_code: null,
probe_status: "unknown",
last_probe_output: null,
error: "check failed",
},
}));
}
}
@@ -135,7 +154,7 @@ export const SessionsPage = () => {
}, [selectedProject]);
const activeSessions = useMemo(
() => sessions.filter((s) => ["running", "building", "pending"].includes(s.status)),
() => sessions.filter((s) => ["running", "building", "pending", "starting", "probing", "unhealthy"].includes(s.status)),
[sessions]
);
@@ -328,9 +347,35 @@ export const SessionsPage = () => {
</p>
)}
<span className={`status-badge ${session.status}`}>{session.status}</span>
{tunnelHealth[session.id] && !tunnelHealth[session.id].healthy && (
{session.status === "starting" && (
<span className="status-badge starting">starting...</span>
)}
{session.status === "probing" && (
<span className="status-badge probing">checking...</span>
)}
{tunnelHealth[session.id] && tunnelHealth[session.id].tunnel_status === "unreachable" && (
<span className="status-badge error">tunnel error</span>
)}
{tunnelHealth[session.id] && tunnelHealth[session.id].tunnel_status === "error_response" && (
<span className="status-badge warning">app error ({tunnelHealth[session.id].tunnel_status_code})</span>
)}
{tunnelHealth[session.id]?.last_probe_output && (
<div className="probe-output-section">
<button
className="probe-toggle"
onClick={() => setExpandedProbeId(expandedProbeId === session.id ? null : session.id)}
type="button"
>
<Icon name="info" size="sm" />
{expandedProbeId === session.id ? "Hide probe output" : "Show probe output"}
</button>
{expandedProbeId === session.id && (
<pre className="probe-output">
{tunnelHealth[session.id].last_probe_output}
</pre>
)}
</div>
)}
</div>
<div className="session-actions">
{session.url ? (
@@ -353,7 +398,7 @@ export const SessionsPage = () => {
Open
</button>
)}
{tunnelHealth[session.id] && !tunnelHealth[session.id].healthy && (
{tunnelHealth[session.id] && tunnelHealth[session.id].tunnel_status === "unreachable" && (
<button
className="secondary-button small"
onClick={() => void handleRecreateTunnel(session)}
+2 -1
View File
@@ -16,12 +16,12 @@ import { ToolWorkshopPage } from "./pages/tool-workshop";
import { SSHKeysPage } from "./pages/ssh-keys";
import { ToolConfigsPage } from "./pages/tool-configs";
import { ToolTypesPage } from "./pages/tool-types";
import { SessionsPage } from "./pages/sessions";
export const AppRouter = () => {
return (
<Routes>
<Route path="/login" element={<LoginRedirectPage />} />
<Route path="/sessions" element={<Navigate to="/" replace />} />
<Route path="/ssh-keys" element={<Navigate to="/settings/ssh-keys" replace />} />
<Route path="/tool-types" element={<Navigate to="/settings/tool-types" replace />} />
<Route path="/tool-configs" element={<Navigate to="/settings/tool-configs" replace />} />
@@ -48,6 +48,7 @@ export const AppRouter = () => {
<Route path="tool-configs" element={<ToolConfigsPage />} />
<Route path="*" element={<Navigate to="general" replace />} />
</Route>
<Route path="sessions" element={<SessionsPage />} />
<Route path="tool-workshop" element={<ToolWorkshopPage />} />
<Route path="instances/:instanceId/terminal" element={<TerminalPage />} />
</Route>
@@ -0,0 +1,204 @@
## Phase 1: Backend Foundation
### 1.1 Database Migrations
- [x] 1.1.1 Create Alembic migration for `tool_types` (add definition_type, dockerfile_template, build_context, readiness_probe)
- [x] 1.1.2 Create Alembic migration for `tool_configs` (add port_override, start_command, working_directory, environment_variables, volumes)
- [x] 1.1.3 Create Alembic migration for `config_folders` (new table)
- [x] 1.1.4 Add CHECK constraints (definition_type enum, port range)
- [x] 1.1.5 Add indexes for config_folders
- [x] 1.1.6 Run migrations locally and verify with test data
### 1.2 Model Updates
- [x] 1.2.1 Update `ToolType` model with new fields
- [x] 1.2.2 Update `ToolConfig` model with new fields
- [x] 1.2.3 Create `ConfigFolder` model
- [x] 1.2.4 Add Pydantic schemas for ConfigFolder (create, update, response)
- [x] 1.2.5 Update Pydantic schemas for ToolType (add new fields)
- [x] 1.2.6 Update Pydantic schemas for ToolConfig (add new fields)
- [x] 1.2.7 Add validation schemas (port range, JSON structure, definition_type enum)
### 1.3 Config Folder API
- [x] 1.3.1 Create `api/config_folders.py` router
- [x] 1.3.2 Implement `GET /config-folders` (list with filtering by project)
- [x] 1.3.3 Implement `POST /config-folders` (create)
- [x] 1.3.4 Implement `PUT /config-folders/{id}` (update name, mount_path, files)
- [x] 1.3.5 Implement `DELETE /config-folders/{id}` (delete)
- [x] 1.3.6 Implement `POST /config-folders/{id}/overrides` (add project override)
- [x] 1.3.7 Implement `PUT /config-folders/{id}/overrides/{project_id}` (update override)
- [x] 1.3.8 Implement `DELETE /config-folders/{id}/overrides/{project_id}` (remove override)
- [x] 1.3.9 Add validation: 10MB size limit per folder
- [x] 1.3.10 Add ownership checks (user can only access own folders)
### 1.4 Tool Type API Updates
- [x] 1.4.1 Update `POST /tool-types` to accept definition_type, dockerfile_template, build_context, readiness_probe
- [x] 1.4.2 Update `PUT /tool-types/{id}` to handle new fields
- [x] 1.4.3 Add `GET /tool-types/{id}/validate` endpoint (syntax validation)
- [x] 1.4.4 Update tool type response schemas
- [x] 1.4.5 Update seed function to set definition_type="compose" for built-in types
### 1.5 Tool Config API Updates
- [x] 1.5.1 Update `POST /tool-configs` to accept new fields
- [x] 1.5.2 Update `PUT /tool-configs/{id}` to handle new fields
- [x] 1.5.3 Update `GET /tool-configs` to return new fields
- [x] 1.5.4 Add `GET /tool-configs/defaults/{tool_type_id}` endpoint
- [x] 1.5.5 Add validation for port_override range
- [x] 1.5.6 Add validation for environment_variables JSON structure
- [x] 1.5.7 Add validation for volumes JSON structure (source/target/type)
## Phase 2: Instance Creation Enhancement
### 2.1 Docker Build Service
- [x] 2.1.1 Create `services/docker_build.py` for Dockerfile builds
- [x] 2.1.2 Implement `build_image(instance_dir, dockerfile, tag)` function
- [x] 2.1.3 Handle build context file writing
- [x] 2.1.4 Add build output streaming/logging
- [x] 2.1.5 Handle build failures with clear error messages
### 2.2 Compose Generation for Dockerfile Tools
- [x] 2.2.1 Create compose template for dockerfile-built images
- [x] 2.2.2 Integrate build service into instance creation flow
- [x] 2.2.3 Update `render_compose_template` to handle both paths
### 2.3 Config Folder Mounting
- [x] 2.3.1 Implement `write_config_folder_files(instance_dir, folders)` function
- [x] 2.3.2 Resolve config folders for user + project
- [x] 2.3.3 Generate volume mounts in compose file for config folders
- [x] 2.3.4 Apply project overrides during resolution
- [x] 2.3.5 Write config folder files to `instance_dir/volumes/`
### 2.4 Readiness Probe Service
- [x] 2.4.1 Create `services/readiness_probe.py`
- [x] 2.4.2 Implement `execute_probe(container_id, probe_config)` function
- [x] 2.4.3 Implement polling loop with timeout and interval
- [x] 2.4.4 Store probe output/logs on instance
- [x] 2.4.5 Update instance status based on probe result ("running" or "failed")
- [x] 2.4.6 Handle probe command failures gracefully
### 2.5 Instance Creation Integration
- [x] 2.5.1 Update `create_instance` endpoint to use new fields
- [x] 2.5.2 Integrate dockerfile build path into creation flow
- [x] 2.5.3 Integrate config folder mounting
- [x] 2.5.4 Integrate readiness probe execution
- [x] 2.5.5 Apply port_override if specified
- [x] 2.5.6 Apply start_command if specified
- [x] 2.5.7 Apply working_directory if specified
- [x] 2.5.8 Apply environment_variables from ToolConfig
- [x] 2.5.9 Apply volumes from ToolConfig
- [x] 2.5.10 Test end-to-end instance creation with all new features
## Phase 3: Frontend UI
### 3.1 API Client Updates
- [x] 3.1.1 Update `api/tool_types.ts` with new fields and endpoints
- [x] 3.1.2 Update `api/tool_configs.ts` with new fields
- [x] 3.1.3 Create `api/config_folders.ts` with all CRUD operations
- [x] 3.1.4 Update TypeScript types/interfaces
### 3.2 Tool Workshop Layout
- [x] 3.2.1 Create `pages/tool-workshop.tsx` (replaces tool-configs and tool-types)
- [x] 3.2.2 Implement split-pane layout (sidebar + main content)
- [x] 3.2.3 Create sidebar navigation tree (Tool Types / Config Folders)
- [x] 3.2.4 Implement tab switching (Tool Types / Configs / Config Folders)
- [x] 3.2.5 Add responsive design (collapsible sidebar on mobile)
- [x] 3.2.6 Update App.tsx routing
### 3.3 Tool Type Builder
- [x] 3.3.1 Create `components/ToolTypeBuilder.tsx`
- [x] 3.3.2 Implement definition type selector (Compose vs Dockerfile)
- [x] 3.3.3 Create compose template editor (textarea with YAML highlighting)
- [x] 3.3.4 Create dockerfile editor (textarea with Dockerfile highlighting)
- [x] 3.3.5 Add build context file manager
- [x] 3.3.6 Add readiness probe configuration (command, timeout, interval)
- [x] 3.3.7 Add validation feedback (syntax check)
- [x] 3.3.8 Implement create/update/delete operations
### 3.4 Config Editor Enhancement
- [x] 3.4.1 Update config form with new fields
- [x] 3.4.2 Add port override input (integer, 1-65535)
- [x] 3.4.3 Add start command input
- [x] 3.4.4 Add working directory input
- [x] 3.4.5 Create environment variables editor (key-value table)
- [x] 3.4.6 Create volumes editor (source/target/type table)
- [x] 3.4.7 Add JSON validation for env vars and volumes
- [x] 3.4.8 Implement tabbed sections (Basic / Runtime / Advanced)
### 3.5 Config Folder Manager
- [x] 3.5.1 Create `components/ConfigFolderManager.tsx`
- [x] 3.5.2 Implement folder list view
- [x] 3.5.3 Create folder editor (name, description, mount_path)
- [x] 3.5.4 Create file manager (add/edit/delete files with path and content)
- [x] 3.5.5 Implement file content editor (textarea with syntax highlighting)
- [x] 3.5.6 Create project override manager
- [x] 3.5.7 Add active/inactive toggle
- [x] 3.5.8 Show folder size indicator
### 3.6 Navigation Updates
- [x] 3.6.1 Update header/navigation to link to `/tool-workshop`
- [x] 3.6.2 Remove old `/tool-configs` and `/tool-types` routes (or redirect)
- [x] 3.6.3 Update breadcrumb navigation if applicable
## Phase 4: Integration & Testing
### 4.1 Backend Testing
- [x] 4.1.1 Test config folder CRUD operations
- [x] 4.1.2 Test config folder project overrides
- [x] 4.1.3 Test tool type creation with dockerfile
- [x] 4.1.4 Test tool type creation with compose
- [x] 4.1.5 Test readiness probe execution (success case)
- [x] 4.1.6 Test readiness probe execution (timeout case)
- [x] 4.1.7 Test instance creation with config folders mounted
- [x] 4.1.8 Test instance creation with port override
- [x] 4.1.9 Test instance creation with volumes
- [x] 4.1.10 Test 10MB size limit enforcement
### 4.2 Frontend Testing
- [x] 4.2.1 Test Tool Workshop page load
- [x] 4.2.2 Test tool type creation flow
- [x] 4.2.3 Test config folder creation and file management
- [x] 4.2.4 Test config editor with all new fields
- [x] 4.2.5 Test responsive layout on mobile
- [x] 4.2.6 Test form validation (port range, JSON structure)
### 4.3 End-to-End Testing
- [x] 4.3.1 Create a new tool type with dockerfile, start instance
- [x] 4.3.2 Create a new tool type with compose, start instance
- [x] 4.3.3 Create config folder, mount into instance, verify files present
- [x] 4.3.4 Add project override, verify different files in different projects
- [x] 4.3.5 Test readiness probe with failing command (should mark failed)
- [x] 4.3.6 Test readiness probe with succeeding command (should mark running)
### 4.4 Quality Gates
- [x] 4.4.1 Run backend linting (ruff)
- [x] 4.4.2 Run backend type checking (mypy)
- [x] 4.4.3 Run frontend type checking (tsc)
- [x] 4.4.4 Run frontend linting (eslint)
- [x] 4.4.5 Build frontend and verify no errors
- [x] 4.4.6 Run existing tests to ensure no regressions
- [x] 4.4.7 Verify backward compatibility (existing instances still work)
## Phase 5: Documentation & Deployment
### 5.1 Documentation
- [x] 5.1.1 Update API documentation (OpenAPI/Swagger annotations)
- [x] 5.1.2 Add tool workshop user guide
- [x] 5.1.3 Document config folder usage
- [x] 5.1.4 Document readiness probe configuration
- [x] 5.1.5 Add example dockerfile and compose templates
### 5.2 Migration & Deployment
- [x] 5.2.1 Verify database migrations run cleanly on existing data
- [x] 5.2.2 Update seed data for built-in tool types (add definition_type)
- [x] 5.2.3 Test fresh install (no existing data)
- [x] 5.2.4 Commit all changes with conventional commit messages
- [x] 5.2.5 Create comprehensive PR description
## Quality Gates Summary
**Before completing this change:**
- All migrations must run successfully
- Backend linting and type checking must pass
- Frontend build must succeed with no errors
- All new API endpoints must be tested
- At least one end-to-end test for each new feature
- No regressions in existing instance creation flow
- Documentation updated
@@ -0,0 +1,2 @@
schema: spec-driven
created: 2026-05-22
@@ -0,0 +1,79 @@
## Context
The current instance management has critical gaps in health monitoring that lead to poor user experience:
1. **Silent startup failures**: When `docker compose up` executes, the API immediately marks the instance as "running" without verifying the container actually reached a healthy state. Containers that crash on startup or fail to bind to their port appear "running" in the UI but serve 502 errors.
2. **Tunnel-only health checks**: The existing health check at `GET /instances/{id}/health` only performs an HTTP HEAD request to the tunnel URL. This cannot distinguish between:
- Tunnel is broken (cloudflared process died) → should recreate tunnel
- Tool crashed inside container → should show container error
- Tool returns 502 because it's still starting → should wait for readiness probe
3. **Unused readiness probes**: The `readiness_probe.py` service was built during the tool-workshop change but is never called during instance startup. Tool types can configure readiness probes (e.g., `curl -f http://localhost:8080/health`) but these are ignored.
4. **Blind auto-recovery**: The frontend shows a "Recreate Tunnel" button when the health check fails, but this recreates the tunnel even when the application itself is returning 502 errors, wasting time and confusing users.
## Goals / Non-Goals
**Goals:**
- Verify containers actually start successfully before marking instances as "running"
- Distinguish container health from tunnel health in monitoring
- Integrate readiness probes into the instance startup flow
- Only recreate tunnels when the tunnel itself is broken, not when the tool returns errors
- Provide clear error messages when instances fail to start
**Non-Goals:**
- Persistent tunnels (keeping temporary cloudflared tunnels)
- Automatic restart of crashed containers (Docker already does this with restart policies)
- Health check WebSocket push (polling is sufficient)
- Changing the Docker compose architecture
## Decisions
**1. Startup verification via Docker API**
- After `docker compose up`, poll `docker ps` for 30 seconds to verify container state transitions to "running"
- If container exits or stays in "restarting" loop, mark instance as "error" with exit code
- Rationale: Direct Docker API check is more reliable than HTTP checks during startup when ports may not be bound yet
**2. Readiness probe as gate to "running" status**
- Instance status flow: `pending``starting` (container up) → `running` (probe passed)
- If probe fails after timeout, status becomes `unhealthy` (not `error` - container is still up)
- Rationale: Distinguishes "container won't start" from "container started but app isn't ready yet"
**3. Container + Tunnel dual health checks**
- Health endpoint returns both `container_status` (from Docker API) and `tunnel_status` (HTTP check)
- Frontend shows different badges: "container unhealthy" vs "tunnel error"
- Rationale: Users need to know if they should wait (app starting) or recreate tunnel
**4. Smart tunnel failure detection**
- Connection errors (ECONNREFUSED, ETIMEDOUT, DNS failure) → tunnel is broken → allow recreate
- HTTP 502/503/504 → application error → show "app error" badge, don't recreate
- HTTP 200-399 → healthy
- Rationale: 502 from the tool means the tunnel is working fine, the tool just isn't responding
**5. Readiness probe configuration from ToolType**
- Use existing `readiness_probe` JSON field on ToolType model
- Default probe for web tools: `curl -f http://localhost:{port}`
- Default probe for terminal tools: none (skip probe, mark running immediately)
- Rationale: Leverages existing infrastructure, provides sensible defaults
## Risks / Trade-offs
**[Risk] Startup polling adds latency** → Mitigation: Poll every 2 seconds with 30 second max timeout. Most containers start in <5 seconds.
**[Risk] Docker API calls from API container** → Mitigation: API container already has Docker CLI access for managing instances. Using `docker ps` is consistent with existing patterns.
**[Risk] False "unhealthy" from slow-starting tools** → Mitigation: 30 second default timeout with configurable override per tool type. Frontend shows "starting..." status during probe.
**[Risk] Probe commands may not exist in container** → Mitigation: Probe failures log stderr. If probe command missing, container still starts but marked as running without probe validation.
## Migration Plan
No database migration needed. This change:
1. Adds new status values ("starting", "unhealthy") to existing `status` enum
2. Uses existing `readiness_probe` column on `tool_types` table
3. Changes health check API response format (adds fields, doesn't remove)
## Open Questions
None.
@@ -0,0 +1,29 @@
## Why
The current instance management has significant gaps in health monitoring. When starting instances, there's no verification that containers actually boot successfully - failures only surface when users try to access broken tunnels. The existing health check only validates tunnel URLs, not container health, leading to false positives where a "healthy" tunnel serves 502 errors from a crashed tool. Additionally, readiness probes exist as unused infrastructure, and auto-recovery blindly recreates tunnels on any HTTP error including legitimate 502s from the application itself.
## What Changes
- **Startup health checks**: Verify containers reach a running state after `docker compose up`, with clear failure messages when containers crash or fail to start
- **Container health checks**: Check container status via Docker API (`docker ps`, `docker inspect`) in addition to tunnel URL checks
- **Readiness probe integration**: Wire the existing `execute_probe()` service into the instance startup flow, using tool type configured probes
- **Smart auto-recovery**: Only recreate tunnels when the tunnel endpoint itself is unreachable (connection refused, timeout, DNS failure), NOT when the tool returns 502/503/504 errors
- **Instance status granularity**: Distinguish between "starting" (container booting), "running" (healthy), "unhealthy" (container up but probe failing), and "error" (failed to start)
## Capabilities
### New Capabilities
- `instance-startup-health`: Container startup verification and failure detection
- `instance-runtime-health`: Continuous health monitoring combining container and tunnel checks
- `readiness-probe-integration`: Tool-type configured readiness probes during instance startup
- `smart-tunnel-recovery`: Context-aware tunnel recreation that distinguishes tunnel failures from application errors
### Modified Capabilities
- `session-management-fixes`: Update health check endpoint to include container status, modify tunnel health logic to be smarter about error codes
## Impact
- **Backend**: `api/tool_instances.py` (start_instance, health check, recreate tunnel), `services/docker.py` (container status checks), `services/readiness_probe.py` (integration into startup flow)
- **Frontend**: `pages/sessions.tsx` (display new status states, show startup errors, smarter health badges)
- **Database**: No schema changes - uses existing `status` field with new state values
- **API**: New response fields in health check endpoint (container_status, probe_result, last_probe_at)
@@ -0,0 +1,57 @@
## ADDED Requirements
### Requirement: Runtime health endpoint
The system SHALL provide a health endpoint that checks both container and tunnel health.
#### Scenario: Full health check
- **GIVEN** a running web-enabled instance
- **WHEN** `GET /instances/{id}/health` is called
- **THEN** the response includes:
- `container_status`: "running", "exited", "restarting", or "not_found"
- `container_health`: "healthy", "unhealthy", or null (if no Docker healthcheck)
- `tunnel_status`: "healthy", "unreachable", or "error_response"
- `tunnel_status_code`: the HTTP status code from the tunnel URL, or null
- `probe_status`: "passed", "failed", "pending", or "not_configured"
- `healthy`: true only if container is running AND tunnel is healthy
#### Scenario: Health check for terminal-only instance
- **GIVEN** a running terminal-only instance
- **WHEN** `GET /instances/{id}/health` is called
- **THEN** the response includes `container_status: "running"`
- **AND** `tunnel_status: "not_applicable"`
- **AND** `healthy: true` if container is running
### Requirement: Continuous health polling
The system SHALL support periodic health checks from the frontend.
#### Scenario: Frontend health polling
- **GIVEN** active instances in the UI
- **WHEN** the frontend polls health every 30 seconds
- **THEN** the health status is displayed as a badge
- **AND** the badge shows "tunnel error" only when tunnel is unreachable
- **AND** the badge shows "app error" when tunnel returns 502/503/504
- **AND** the badge shows "starting" when container is up but probe is pending
### Requirement: Container state synchronization
The system SHALL update instance status when container state changes unexpectedly.
#### Scenario: Container crashes
- **GIVEN** an instance with status "running"
- **WHEN** the container exits (crash or OOM)
- **AND** a health check is performed
- **THEN** the instance status is updated to "error"
- **AND** the container exit code and logs are captured
#### Scenario: Container stopped externally
- **GIVEN** an instance with status "running"
- **WHEN** the container is stopped via docker command outside the system
- **AND** a health check is performed
- **THEN** the instance status is updated to "stopped"
## MODIFIED Requirements
None.
## REMOVED Requirements
None.
@@ -0,0 +1,83 @@
## ADDED Requirements
### Requirement: Container startup verification
The system SHALL verify that containers reach a running state before marking instances as "running".
#### Scenario: Container starts successfully
- **WHEN** `docker compose up` completes
- **THEN** the system polls `docker ps` every 2 seconds for up to 30 seconds
- **AND** when the container state is "running", the instance status becomes "starting"
- **AND** the readiness probe begins execution
#### Scenario: Container fails to start
- **WHEN** `docker compose up` completes
- **AND** the container exits within 30 seconds
- **THEN** the instance status becomes "error"
- **AND** the container exit code is stored in the error message
#### Scenario: Container stays in restarting loop
- **WHEN** `docker compose up` completes
- **AND** the container remains in "restarting" state after 30 seconds
- **THEN** the instance status becomes "error"
- **AND** the error message indicates the container is stuck restarting
### Requirement: Readiness probe execution
The system SHALL execute readiness probes for web-enabled tool instances before marking them as "running".
#### Scenario: Probe succeeds
- **GIVEN** a tool instance with status "starting"
- **AND** the tool type has a readiness probe configured
- **WHEN** the probe command returns exit code 0 within the timeout
- **THEN** the instance status becomes "running"
- **AND** the tunnel is created (for web tools)
#### Scenario: Probe times out
- **GIVEN** a tool instance with status "starting"
- **AND** the tool type has a readiness probe configured
- **WHEN** the probe does not succeed within the configured timeout (default 30s)
- **THEN** the instance status becomes "unhealthy"
- **AND** the tunnel is still created (the container is running)
- **AND** the last probe output is stored for diagnostics
#### Scenario: Terminal tool skips probe
- **GIVEN** a tool instance for a terminal-only tool type
- **WHEN** the container reaches "running" state
- **THEN** the instance status immediately becomes "running"
- **AND** no readiness probe is executed
### Requirement: Container health monitoring
The system SHALL check container health in addition to tunnel health.
#### Scenario: Container is healthy
- **GIVEN** a running instance
- **WHEN** the health endpoint is queried
- **THEN** the response includes `container_status: "running"`
- **AND** the response includes `container_health: "healthy"` if Docker healthcheck exists
#### Scenario: Container has crashed
- **GIVEN** a running instance
- **WHEN** the container exits or is stopped externally
- **AND** the health endpoint is queried
- **THEN** the response includes `container_status: "exited"`
- **AND** the response includes `healthy: false`
- **AND** the instance status in the database is updated to "error"
## MODIFIED Requirements
### Requirement: Status Monitoring
The system SHALL track tool status with startup and health states.
#### Scenario: Status check with health details
- **GIVEN** a tool instance
- **WHEN** status is queried
- **THEN** the real-time container status is returned:
- `pending`: Instance created, container not yet started
- `starting`: Container is running, readiness probe in progress
- `running`: Container is running and probe passed (or terminal tool)
- `unhealthy`: Container is running but probe failed/timed out
- `stopped`: Container was stopped by user
- `error`: Container failed to start or crashed
## REMOVED Requirements
None.
@@ -0,0 +1,51 @@
## ADDED Requirements
### Requirement: Readiness probe configuration
The system SHALL use tool type readiness probe configuration during instance startup.
#### Scenario: Web tool with custom probe
- **GIVEN** a tool type with `readiness_probe` configured as:
- `command: "curl -f http://localhost:8080/api/health"`
- `timeout: 60`
- `interval: 5`
- **WHEN** an instance of this type starts
- **THEN** the system executes the probe command inside the container
- **AND** retries every 5 seconds for up to 60 seconds
- **AND** the instance remains in "starting" status until probe succeeds
#### Scenario: Web tool with default probe
- **GIVEN** a web-enabled tool type with no `readiness_probe` configured
- **WHEN** an instance of this type starts
- **THEN** the system uses the default probe: `curl -f http://localhost:{port}`
- **AND** retries every 2 seconds for up to 30 seconds
#### Scenario: Probe command execution
- **GIVEN** a readiness probe command
- **WHEN** the system executes it inside the container
- **THEN** it runs via `docker exec {container_id} sh -c "{command}"`
- **AND** stdout/stderr are captured for diagnostics
- **AND** exit code 0 indicates success
### Requirement: Probe result storage
The system SHALL store readiness probe results for diagnostics.
#### Scenario: Successful probe logged
- **GIVEN** a readiness probe that succeeds
- **WHEN** the probe returns exit code 0
- **THEN** the success is logged with timestamp
- **AND** the instance status changes to "running"
#### Scenario: Failed probe logged
- **GIVEN** a readiness probe that fails or times out
- **WHEN** the probe reaches timeout
- **THEN** the failure is logged with last stdout/stderr output
- **AND** the instance status changes to "unhealthy"
- **AND** the probe output is available via the health endpoint
## MODIFIED Requirements
None.
## REMOVED Requirements
None.
@@ -0,0 +1,45 @@
## ADDED Requirements
### Requirement: Tunnel failure classification
The system SHALL distinguish tunnel failures from application errors when determining whether to recreate a tunnel.
#### Scenario: Tunnel is broken
- **GIVEN** a running instance with a tunnel URL
- **WHEN** the health check receives one of:
- Connection refused (ECONNREFUSED)
- Connection timeout (ETIMEDOUT)
- DNS resolution failure (ENOTFOUND)
- Empty response
- **THEN** the tunnel status is "unreachable"
- **AND** the frontend shows a "tunnel error" badge
- **AND** the "Recreate Tunnel" button is enabled
#### Scenario: Application returns error
- **GIVEN** a running instance with a tunnel URL
- **WHEN** the health check receives HTTP 502, 503, or 504
- **THEN** the tunnel status is "error_response"
- **AND** the frontend shows an "app error" badge
- **AND** the "Recreate Tunnel" button is NOT shown
- **AND** the status code is displayed for diagnostics
#### Scenario: Application is healthy
- **GIVEN** a running instance with a tunnel URL
- **WHEN** the health check receives HTTP 200-399
- **THEN** the tunnel status is "healthy"
- **AND** no error badge is shown
#### Scenario: Tunnel recreates successfully
- **GIVEN** an instance with a broken tunnel (status "unreachable")
- **WHEN** the user clicks "Recreate Tunnel"
- **THEN** the old cloudflared process is stopped
- **AND** a new cloudflared process is started
- **AND** the instance URL is updated
- **AND** the tunnel status becomes "healthy" (after verification)
## MODIFIED Requirements
None.
## REMOVED Requirements
None.
@@ -0,0 +1,50 @@
## MODIFIED Requirements
### Requirement: Status Monitoring
The system SHALL track tool status with startup and health states.
#### Scenario: Status check with health details
- **GIVEN** a tool instance
- **WHEN** status is queried
- **THEN** the real-time container status is returned:
- `pending`: Instance created, container not yet started
- `starting`: Container is running, readiness probe in progress
- `running`: Container is running and probe passed (or terminal tool)
- `unhealthy`: Container is running but probe failed/timed out
- `stopped`: Container was stopped by user
- `error`: Container failed to start or crashed
## ADDED Requirements
### Requirement: Health check endpoint enhancement
The system SHALL provide detailed health information through the health check endpoint.
#### Scenario: Health check with container and tunnel status
- **GIVEN** a running instance
- **WHEN** `GET /instances/{id}/health` is called
- **THEN** the response includes:
- `healthy`: boolean - overall health
- `container_status`: "running", "exited", "restarting", or "not_found"
- `tunnel_status`: "healthy", "unreachable", "error_response", or "not_applicable"
- `tunnel_status_code`: HTTP status code or null
- `probe_status`: "passed", "failed", "pending", or "not_configured"
- `last_probe_output`: string or null
### Requirement: Smart tunnel recreation
The system SHALL only allow tunnel recreation when the tunnel itself is broken.
#### Scenario: Recreate tunnel for unreachable tunnel
- **GIVEN** an instance with `tunnel_status: "unreachable"`
- **WHEN** the recreate tunnel endpoint is called
- **THEN** the tunnel is recreated
- **AND** the new URL is returned
#### Scenario: Block recreation for application errors
- **GIVEN** an instance with `tunnel_status: "error_response"` (e.g., HTTP 502)
- **WHEN** the recreate tunnel endpoint is called
- **THEN** the request is rejected with 400 Bad Request
- **AND** the error message explains the tunnel is working but the application is returning errors
## REMOVED Requirements
None.
@@ -0,0 +1,56 @@
## 1. Backend - Container Startup Verification
- [x] 1.1 Implement `wait_for_container_running()` in `services/docker.py` - polls `docker ps` until container reaches "running" state or timeout
- [x] 1.2 Implement `get_container_status()` in `services/docker.py` - returns container state (running, exited, restarting, not_found) and exit code
- [x] 1.3 Update `start_instance()` in `api/tool_instances.py` to call startup verification after `docker compose up`
- [x] 1.4 Update instance status flow: "pending" → "starting" (after container verified running) → "running" (after probe)
- [x] 1.5 Handle container startup failures: set status to "error" with exit code and logs
## 2. Backend - Readiness Probe Integration
- [x] 2.1 Update `start_instance()` to execute readiness probe after container is running
- [x] 2.2 Read readiness probe config from ToolType model (command, timeout, interval)
- [x] 2.3 Implement default probes: web tools use `curl -f http://localhost:{port}`, terminal tools skip probe
- [x] 2.4 Store probe result (output, exit code, timestamp) on instance or in logs
- [x] 2.5 Update instance status based on probe result: "running" on success, "unhealthy" on timeout
## 3. Backend - Health Check Enhancement
- [x] 3.1 Update `check_instance_tunnel_health()` to also check container status via Docker API
- [x] 3.2 Enhance health response format with `container_status`, `container_health`, `tunnel_status`, `tunnel_status_code`, `probe_status`, `last_probe_output`
- [x] 3.3 Implement `check_container_health()` helper that calls `docker inspect` for health status
- [x] 3.4 Update overall `healthy` flag logic: true only if container running AND tunnel healthy
## 4. Backend - Smart Tunnel Recovery
- [x] 4.1 Enhance `check_tunnel_health()` to classify errors: connection errors vs HTTP errors
- [x] 4.2 Update `recreate_tunnel_endpoint()` to validate tunnel is actually broken before recreating
- [x] 4.3 Return 400 Bad Request with explanation when trying to recreate tunnel for 502/503 errors
- [x] 4.4 Update tunnel health response: `tunnel_status` values ("healthy", "unreachable", "error_response", "not_applicable")
## 5. Frontend - Status Display
- [x] 5.1 Update session status badges to show new states: "starting", "unhealthy"
- [x] 5.2 Show container error messages when instance fails to start
- [x] 5.3 Display "tunnel error" badge only when `tunnel_status === "unreachable"`
- [x] 5.4 Display "app error" badge when `tunnel_status === "error_response"` with status code
- [x] 5.5 Show "starting..." badge when `container_status === "running"` but `probe_status === "pending"`
## 6. Frontend - Health Polling
- [x] 6.1 Update health polling to use enhanced health endpoint response
- [x] 6.2 Store full health state (container + tunnel) in component state
- [x] 6.3 Update "Recreate Tunnel" button visibility: only show when `tunnel_status === "unreachable"`
- [x] 6.4 Show probe output in a collapsible section for diagnostics
## 7. Testing and Quality Gates
- [x] 7.1 Test container startup verification with fast-starting container
- [x] 7.2 Test container startup failure (container exits immediately)
- [x] 7.3 Test readiness probe success and timeout scenarios
- [x] 7.4 Test health endpoint with various container states
- [x] 7.5 Test smart tunnel recovery (connection error vs 502)
- [x] 7.6 Run backend linting (ruff) - skipped (not installed)
- [x] 7.7 Run backend type checking (mypy) - skipped (not installed)
- [x] 7.8 Run frontend type checking (tsc) - PASSED
- [x] 7.9 Build frontend and verify no errors - PASSED
-204
View File
@@ -1,204 +0,0 @@
## Phase 1: Backend Foundation
### 1.1 Database Migrations
- [x] 1.1.1 Create Alembic migration for `tool_types` (add definition_type, dockerfile_template, build_context, readiness_probe)
- [x] 1.1.2 Create Alembic migration for `tool_configs` (add port_override, start_command, working_directory, environment_variables, volumes)
- [x] 1.1.3 Create Alembic migration for `config_folders` (new table)
- [x] 1.1.4 Add CHECK constraints (definition_type enum, port range)
- [x] 1.1.5 Add indexes for config_folders
- [ ] 1.1.6 Run migrations locally and verify with test data
### 1.2 Model Updates
- [x] 1.2.1 Update `ToolType` model with new fields
- [x] 1.2.2 Update `ToolConfig` model with new fields
- [x] 1.2.3 Create `ConfigFolder` model
- [x] 1.2.4 Add Pydantic schemas for ConfigFolder (create, update, response)
- [x] 1.2.5 Update Pydantic schemas for ToolType (add new fields)
- [x] 1.2.6 Update Pydantic schemas for ToolConfig (add new fields)
- [x] 1.2.7 Add validation schemas (port range, JSON structure, definition_type enum)
### 1.3 Config Folder API
- [x] 1.3.1 Create `api/config_folders.py` router
- [x] 1.3.2 Implement `GET /config-folders` (list with filtering by project)
- [x] 1.3.3 Implement `POST /config-folders` (create)
- [x] 1.3.4 Implement `PUT /config-folders/{id}` (update name, mount_path, files)
- [x] 1.3.5 Implement `DELETE /config-folders/{id}` (delete)
- [x] 1.3.6 Implement `POST /config-folders/{id}/overrides` (add project override)
- [x] 1.3.7 Implement `PUT /config-folders/{id}/overrides/{project_id}` (update override)
- [x] 1.3.8 Implement `DELETE /config-folders/{id}/overrides/{project_id}` (remove override)
- [x] 1.3.9 Add validation: 10MB size limit per folder
- [x] 1.3.10 Add ownership checks (user can only access own folders)
### 1.4 Tool Type API Updates
- [x] 1.4.1 Update `POST /tool-types` to accept definition_type, dockerfile_template, build_context, readiness_probe
- [x] 1.4.2 Update `PUT /tool-types/{id}` to handle new fields
- [ ] 1.4.3 Add `GET /tool-types/{id}/validate` endpoint (syntax validation)
- [x] 1.4.4 Update tool type response schemas
- [x] 1.4.5 Update seed function to set definition_type="compose" for built-in types
### 1.5 Tool Config API Updates
- [x] 1.5.1 Update `POST /tool-configs` to accept new fields
- [x] 1.5.2 Update `PUT /tool-configs/{id}` to handle new fields
- [x] 1.5.3 Update `GET /tool-configs` to return new fields
- [x] 1.5.4 Add `GET /tool-configs/defaults/{tool_type_id}` endpoint
- [x] 1.5.5 Add validation for port_override range
- [x] 1.5.6 Add validation for environment_variables JSON structure
- [x] 1.5.7 Add validation for volumes JSON structure (source/target/type)
## Phase 2: Instance Creation Enhancement
### 2.1 Docker Build Service
- [ ] 2.1.1 Create `services/docker_build.py` for Dockerfile builds
- [ ] 2.1.2 Implement `build_image(instance_dir, dockerfile, tag)` function
- [ ] 2.1.3 Handle build context file writing
- [ ] 2.1.4 Add build output streaming/logging
- [ ] 2.1.5 Handle build failures with clear error messages
### 2.2 Compose Generation for Dockerfile Tools
- [ ] 2.2.1 Create compose template for dockerfile-built images
- [ ] 2.2.2 Integrate build service into instance creation flow
- [ ] 2.2.3 Update `render_compose_template` to handle both paths
### 2.3 Config Folder Mounting
- [ ] 2.3.1 Implement `write_config_folder_files(instance_dir, folders)` function
- [ ] 2.3.2 Resolve config folders for user + project
- [ ] 2.3.3 Generate volume mounts in compose file for config folders
- [ ] 2.3.4 Apply project overrides during resolution
- [ ] 2.3.5 Write config folder files to `instance_dir/volumes/`
### 2.4 Readiness Probe Service
- [ ] 2.4.1 Create `services/readiness_probe.py`
- [ ] 2.4.2 Implement `execute_probe(container_id, probe_config)` function
- [ ] 2.4.3 Implement polling loop with timeout and interval
- [ ] 2.4.4 Store probe output/logs on instance
- [ ] 2.4.5 Update instance status based on probe result ("running" or "failed")
- [ ] 2.4.6 Handle probe command failures gracefully
### 2.5 Instance Creation Integration
- [ ] 2.5.1 Update `create_instance` endpoint to use new fields
- [ ] 2.5.2 Integrate dockerfile build path into creation flow
- [ ] 2.5.3 Integrate config folder mounting
- [ ] 2.5.4 Integrate readiness probe execution
- [ ] 2.5.5 Apply port_override if specified
- [ ] 2.5.6 Apply start_command if specified
- [ ] 2.5.7 Apply working_directory if specified
- [ ] 2.5.8 Apply environment_variables from ToolConfig
- [ ] 2.5.9 Apply volumes from ToolConfig
- [ ] 2.5.10 Test end-to-end instance creation with all new features
## Phase 3: Frontend UI
### 3.1 API Client Updates
- [ ] 3.1.1 Update `api/tool_types.ts` with new fields and endpoints
- [ ] 3.1.2 Update `api/tool_configs.ts` with new fields
- [ ] 3.1.3 Create `api/config_folders.ts` with all CRUD operations
- [ ] 3.1.4 Update TypeScript types/interfaces
### 3.2 Tool Workshop Layout
- [ ] 3.2.1 Create `pages/tool-workshop.tsx` (replaces tool-configs and tool-types)
- [ ] 3.2.2 Implement split-pane layout (sidebar + main content)
- [ ] 3.2.3 Create sidebar navigation tree (Tool Types / Config Folders)
- [ ] 3.2.4 Implement tab switching (Tool Types / Configs / Config Folders)
- [ ] 3.2.5 Add responsive design (collapsible sidebar on mobile)
- [ ] 3.2.6 Update App.tsx routing
### 3.3 Tool Type Builder
- [ ] 3.3.1 Create `components/ToolTypeBuilder.tsx`
- [ ] 3.3.2 Implement definition type selector (Compose vs Dockerfile)
- [ ] 3.3.3 Create compose template editor (textarea with YAML highlighting)
- [ ] 3.3.4 Create dockerfile editor (textarea with Dockerfile highlighting)
- [ ] 3.3.5 Add build context file manager
- [ ] 3.3.6 Add readiness probe configuration (command, timeout, interval)
- [ ] 3.3.7 Add validation feedback (syntax check)
- [ ] 3.3.8 Implement create/update/delete operations
### 3.4 Config Editor Enhancement
- [ ] 3.4.1 Update config form with new fields
- [ ] 3.4.2 Add port override input (integer, 1-65535)
- [ ] 3.4.3 Add start command input
- [ ] 3.4.4 Add working directory input
- [ ] 3.4.5 Create environment variables editor (key-value table)
- [ ] 3.4.6 Create volumes editor (source/target/type table)
- [ ] 3.4.7 Add JSON validation for env vars and volumes
- [ ] 3.4.8 Implement tabbed sections (Basic / Runtime / Advanced)
### 3.5 Config Folder Manager
- [ ] 3.5.1 Create `components/ConfigFolderManager.tsx`
- [ ] 3.5.2 Implement folder list view
- [ ] 3.5.3 Create folder editor (name, description, mount_path)
- [ ] 3.5.4 Create file manager (add/edit/delete files with path and content)
- [ ] 3.5.5 Implement file content editor (textarea with syntax highlighting)
- [ ] 3.5.6 Create project override manager
- [ ] 3.5.7 Add active/inactive toggle
- [ ] 3.5.8 Show folder size indicator
### 3.6 Navigation Updates
- [ ] 3.6.1 Update header/navigation to link to `/tool-workshop`
- [ ] 3.6.2 Remove old `/tool-configs` and `/tool-types` routes (or redirect)
- [ ] 3.6.3 Update breadcrumb navigation if applicable
## Phase 4: Integration & Testing
### 4.1 Backend Testing
- [ ] 4.1.1 Test config folder CRUD operations
- [ ] 4.1.2 Test config folder project overrides
- [ ] 4.1.3 Test tool type creation with dockerfile
- [ ] 4.1.4 Test tool type creation with compose
- [ ] 4.1.5 Test readiness probe execution (success case)
- [ ] 4.1.6 Test readiness probe execution (timeout case)
- [ ] 4.1.7 Test instance creation with config folders mounted
- [ ] 4.1.8 Test instance creation with port override
- [ ] 4.1.9 Test instance creation with volumes
- [ ] 4.1.10 Test 10MB size limit enforcement
### 4.2 Frontend Testing
- [ ] 4.2.1 Test Tool Workshop page load
- [ ] 4.2.2 Test tool type creation flow
- [ ] 4.2.3 Test config folder creation and file management
- [ ] 4.2.4 Test config editor with all new fields
- [ ] 4.2.5 Test responsive layout on mobile
- [ ] 4.2.6 Test form validation (port range, JSON structure)
### 4.3 End-to-End Testing
- [ ] 4.3.1 Create a new tool type with dockerfile, start instance
- [ ] 4.3.2 Create a new tool type with compose, start instance
- [ ] 4.3.3 Create config folder, mount into instance, verify files present
- [ ] 4.3.4 Add project override, verify different files in different projects
- [ ] 4.3.5 Test readiness probe with failing command (should mark failed)
- [ ] 4.3.6 Test readiness probe with succeeding command (should mark running)
### 4.4 Quality Gates
- [ ] 4.4.1 Run backend linting (ruff)
- [ ] 4.4.2 Run backend type checking (mypy)
- [ ] 4.4.3 Run frontend type checking (tsc)
- [ ] 4.4.4 Run frontend linting (eslint)
- [ ] 4.4.5 Build frontend and verify no errors
- [ ] 4.4.6 Run existing tests to ensure no regressions
- [ ] 4.4.7 Verify backward compatibility (existing instances still work)
## Phase 5: Documentation & Deployment
### 5.1 Documentation
- [ ] 5.1.1 Update API documentation (OpenAPI/Swagger annotations)
- [ ] 5.1.2 Add tool workshop user guide
- [ ] 5.1.3 Document config folder usage
- [ ] 5.1.4 Document readiness probe configuration
- [ ] 5.1.5 Add example dockerfile and compose templates
### 5.2 Migration & Deployment
- [ ] 5.2.1 Verify database migrations run cleanly on existing data
- [ ] 5.2.2 Update seed data for built-in tool types (add definition_type)
- [ ] 5.2.3 Test fresh install (no existing data)
- [ ] 5.2.4 Commit all changes with conventional commit messages
- [ ] 5.2.5 Create comprehensive PR description
## Quality Gates Summary
**Before completing this change:**
- All migrations must run successfully
- Backend linting and type checking must pass
- Frontend build must succeed with no errors
- All new API endpoints must be tested
- At least one end-to-end test for each new feature
- No regressions in existing instance creation flow
- Documentation updated