Enhance provider logging and error handling

- Added functionality to create log files for provider output, improving debugging on Windows.
- Updated error handling to read from log files instead of using subprocess output directly.
- Enhanced logging messages to include log file locations for easier troubleshooting.
This commit is contained in:
Jamie Pine
2026-02-02 02:19:11 -08:00
parent f090759d8f
commit 53b1e8868c
+31 -19
View File
@@ -71,14 +71,22 @@ class ProviderManager:
logger.info(f"Provider binary: {provider_path}") logger.info(f"Provider binary: {provider_path}")
logger.info(f"Data directory: {get_data_dir()}") logger.info(f"Data directory: {get_data_dir()}")
# Create log files for provider output (easier debugging on Windows)
logs_dir = get_data_dir() / "logs"
logs_dir.mkdir(exist_ok=True)
stdout_log = logs_dir / f"{provider_type}-stdout.log"
stderr_log = logs_dir / f"{provider_type}-stderr.log"
logger.info(f"Provider logs will be written to: {logs_dir}")
process = subprocess.Popen( process = subprocess.Popen(
[ [
str(provider_path), str(provider_path),
"--port", str(port), "--port", str(port),
"--data-dir", str(get_data_dir()), "--data-dir", str(get_data_dir()),
], ],
stdout=subprocess.PIPE, stdout=open(stdout_log, 'w'),
stderr=subprocess.PIPE, stderr=open(stderr_log, 'w'),
text=True, text=True,
bufsize=1, bufsize=1,
) )
@@ -88,23 +96,24 @@ class ProviderManager:
try: try:
await self._wait_for_provider_health(base_url, timeout=30) await self._wait_for_provider_health(base_url, timeout=30)
except TimeoutError as e: except TimeoutError as e:
# Capture subprocess output for debugging # Read log files for debugging (works on all platforms unlike select)
stdout_lines = [] stdout_content = ""
stderr_lines = [] stderr_content = ""
# Try to read available output
import select
try: try:
if process.stdout and select.select([process.stdout], [], [], 0)[0]: if stdout_log.exists():
stdout_lines = process.stdout.readlines() stdout_content = stdout_log.read_text()
if process.stderr and select.select([process.stderr], [], [], 0)[0]: if stderr_log.exists():
stderr_lines = process.stderr.readlines() stderr_content = stderr_log.read_text()
except Exception: except Exception as read_err:
# select might not work on all platforms logger.error(f"Failed to read provider logs: {read_err}")
pass
logger.error(f"Provider failed to start. Stdout: {stdout_lines}") logger.error(f"Provider failed to start within 30 seconds")
logger.error(f"Provider failed to start. Stderr: {stderr_lines}") logger.error(f"Check logs at: {logs_dir}")
if stdout_content:
logger.error(f"Stdout: {stdout_content[-2000:]}") # Last 2000 chars
if stderr_content:
logger.error(f"Stderr: {stderr_content[-2000:]}") # Last 2000 chars
# Terminate the process # Terminate the process
process.terminate() process.terminate()
@@ -113,15 +122,18 @@ class ProviderManager:
except subprocess.TimeoutExpired: except subprocess.TimeoutExpired:
process.kill() process.kill()
raise # Raise with log file location for user
raise TimeoutError(
f"Provider {provider_type} failed to start. Check logs at: {logs_dir}"
)
# Create LocalProvider instance # Create LocalProvider instance
self.active_provider = LocalProvider(base_url) self.active_provider = LocalProvider(base_url)
self._provider_process = process self._provider_process = process
self._provider_port = port self._provider_port = port
# Start background task to log subprocess output # Logs are written directly to files (stdout_log, stderr_log)
asyncio.create_task(self._log_subprocess_output(process)) # No need for background task - users can check {logs_dir} for debugging
else: else:
# No external binary, use bundled provider (if available) # No external binary, use bundled provider (if available)
if provider_type == "pytorch-cpu": if provider_type == "pytorch-cpu":