import 'dart:convert'; import 'dart:io'; import 'package:flutter/foundation.dart'; import 'package:http/http.dart' as http; import 'package:path/path.dart' as p; import 'package:path_provider/path_provider.dart'; import 'package:trainhub_flutter/core/constants/ai_constants.dart'; enum AiServerStatus { offline, starting, ready, error } /// Manages the two llama.cpp server processes that provide AI features. /// /// Both processes are kept alive for the lifetime of the app and must be /// killed on shutdown to prevent zombie processes from consuming RAM. /// /// - Qwen3 4B → port 8080 (chat / completions) /// - Nomic Embed → port 8081 (embeddings) /// /// Startup strategy: first attempt offloads all layers to the GPU (Vulkan). /// If the chat server does not become healthy in time — first-run pipeline /// compilation can be very slow, and some GPUs hang outright — everything is /// restarted in CPU-only mode, which loads via mmap and is reliably fast. class AiProcessManager extends ChangeNotifier { Process? _qwenProcess; Process? _nomicProcess; AiServerStatus _status = AiServerStatus.offline; String? _lastError; String? _statusDetail; // Last stderr lines per server — llama.cpp prints its actual failure // reason (bad model file, out of memory, …) there, so keep a short tail // to show the user instead of a generic "crashed" message. final List _qwenStderrTail = []; final List _nomicStderrTail = []; static const int _stderrTailLines = 8; // ------------------------------------------------------------------------- // Public API // ------------------------------------------------------------------------- AiServerStatus get status => _status; bool get isRunning => _status == AiServerStatus.ready; String? get errorMessage => _lastError; /// Live progress line while [status] is [AiServerStatus.starting], /// e.g. "Loading AI model on GPU… 45s". String? get statusDetail => _statusDetail; /// Starts both inference servers. No-ops if already running or starting. Future startServers() async { if (_status == AiServerStatus.starting || _status == AiServerStatus.ready) { return; } _updateStatus(AiServerStatus.starting); _lastError = null; final dir = await getApplicationDocumentsDirectory(); final base = dir.path; final serverBin = p.join(base, AiConstants.serverBinaryName); if (!File(serverBin).existsSync()) { _lastError = 'llama-server executable not found.'; _updateStatus(AiServerStatus.error); return; } try { // Attempt 1: full GPU offload. final gpuReady = await _startAttempt( serverBin, base, gpuLayers: AiConstants.gpuLayerOffload, timeout: AiConstants.serverStartupTimeout, detailLabel: 'GPU', ); if (gpuReady) { _updateStatus(AiServerStatus.ready); return; } // Attempt 2: CPU only. Covers GPUs where Vulkan hangs, crashes // (e.g. out of VRAM) or pipeline compilation never finishes. await _killProcesses(); _lastError = null; _qwenStderrTail.clear(); _nomicStderrTail.clear(); final cpuReady = await _startAttempt( serverBin, base, gpuLayers: 0, timeout: AiConstants.serverStartupTimeoutCpu, detailLabel: 'CPU', ); if (cpuReady) { _updateStatus(AiServerStatus.ready); return; } throw Exception( _lastError ?? 'The AI server did not become ready (tried GPU, then CPU).' '${_diagnostics()}', ); } catch (e) { // Clean up any partially-started processes before returning error. await _killProcesses(); _lastError = e.toString(); _updateStatus(AiServerStatus.error); } finally { _statusDetail = null; } } /// Kills both processes and resets the running flag. /// Safe to call even if servers were never started. Future stopServers() async { await _killProcesses(); if (Platform.isWindows) { try { await Process.run('taskkill', [ '/F', '/IM', AiConstants.serverBinaryName, ]); } catch (_) {} } else if (Platform.isMacOS || Platform.isLinux) { try { await Process.run('pkill', ['-f', 'llama-server']); } catch (_) {} } _updateStatus(AiServerStatus.offline); } // ------------------------------------------------------------------------- // Internals // ------------------------------------------------------------------------- /// Spawns both servers with the given GPU offload and waits for health. /// Returns false on timeout; rethrows nothing — a crashed process is /// reported through [_lastError] by the exit watcher and detected here. Future _startAttempt( String serverBin, String base, { required int gpuLayers, required Duration timeout, required String detailLabel, }) async { _qwenProcess = await _spawn( serverBin, [ '-m', p.join(base, AiConstants.qwenModelFile), '--port', '${AiConstants.chatServerPort}', '--ctx-size', '${AiConstants.qwenContextSize}', '-ngl', '$gpuLayers', // Use the chat template embedded in the GGUF (required for // correct Qwen3 formatting). '--jinja', ], label: 'QWEN', stderrTail: _qwenStderrTail, crashMessage: 'Qwen Chat Server crashed', ); _nomicProcess = await _spawn( serverBin, [ '-m', p.join(base, AiConstants.nomicModelFile), '--port', '${AiConstants.embeddingServerPort}', '--ctx-size', '${AiConstants.nomicContextSize}', '--embedding', ], label: 'NOMIC', stderrTail: _nomicStderrTail, crashMessage: 'Nomic Embedding Server crashed', ); // llama-server binds its port almost immediately but returns 503 until // the model is fully loaded, so poll /health rather than the TCP port. final started = DateTime.now(); final deadline = started.add(timeout); var qwenReady = false; var nomicReady = false; while (DateTime.now().isBefore(deadline) && (!qwenReady || !nomicReady)) { // A crashed process will never become healthy — give up on this // attempt right away (the caller may retry on CPU). if (_lastError != null) return false; await Future.delayed(AiConstants.healthPollInterval); if (!qwenReady) { qwenReady = await _isHealthy(AiConstants.chatHealthUrl); } if (!nomicReady) { nomicReady = await _isHealthy(AiConstants.embeddingHealthUrl); } final elapsed = DateTime.now().difference(started).inSeconds; _statusDetail = 'Loading AI model on $detailLabel… ${elapsed}s ' '(first run can take a few minutes)'; notifyListeners(); } return qwenReady && nomicReady; } Future _killProcesses() async { // Detach before killing so the exit watchers recognise an intentional // shutdown and don't flag it as a crash. final qwen = _qwenProcess; final nomic = _nomicProcess; _qwenProcess = null; _nomicProcess = null; qwen?.kill(); nomic?.kill(); } Future _spawn( String binary, List args, { required String label, required List stderrTail, required String crashMessage, }) async { final process = await Process.start(binary, args, runInShell: false); process.stdout.transform(utf8.decoder).listen((text) { if (kDebugMode) debugPrint('[$label STDOUT] $text'); }); process.stderr.transform(utf8.decoder).listen((text) { for (final line in const LineSplitter().convert(text)) { if (line.trim().isEmpty) continue; stderrTail.add(line); if (stderrTail.length > _stderrTailLines) stderrTail.removeAt(0); } if (kDebugMode) debugPrint('[$label STDERR] $text'); }); // Monitor for unexpected crash. Intentional kills (retry, shutdown) // null out the process reference first, so they are ignored here. process.exitCode.then((code) { final isCurrent = identical(process, _qwenProcess) || identical(process, _nomicProcess); if (!isCurrent) return; if (_status == AiServerStatus.ready || _status == AiServerStatus.starting) { _lastError = '$crashMessage (exit code $code).${_diagnostics()}'; if (_status == AiServerStatus.ready) { _updateStatus(AiServerStatus.error); } } }); return process; } Future _isHealthy(String url) async { try { final response = await http .get(Uri.parse(url)) .timeout(const Duration(seconds: 2)); return response.statusCode == 200; } catch (_) { return false; } } String _diagnostics() { final buffer = StringBuffer(); if (_qwenStderrTail.isNotEmpty) { buffer.write('\n\nChat server log:\n${_qwenStderrTail.join('\n')}'); } if (_nomicStderrTail.isNotEmpty) { buffer.write('\n\nEmbedding server log:\n${_nomicStderrTail.join('\n')}'); } return buffer.toString(); } void _updateStatus(AiServerStatus newStatus) { if (_status != newStatus) { _status = newStatus; notifyListeners(); } } }