feat(client): file logging + llama.cpp fallback cooldown and parallel options
Build & Deploy / build (push) Successful in 2m8s

This commit is contained in:
2026-08-02 01:24:09 +02:00
parent 44b09ffa86
commit b849cfd4b7
7 changed files with 457 additions and 121 deletions
+11 -2
View File
@@ -133,11 +133,11 @@ sudo bash deploy/install-client.sh --server http://your-server:5050 --token "cha
``` ```
or using ollama models with llama.cpp backend using ROCm: or using ollama models with llama.cpp backend using ROCm:
```bash ```bash
sudo bash deploy/install-client.sh --server https://ai.domain.tld --token "change-me" --use-llama-cpp-via-docker --use-ollama-models-path /usr/share/ollama/.ollama/models --llama-cpp-docker-image ghcr.io/ggml-org/llama.cpp:server-rocm --llama-cpp-base-port 8081 sudo bash deploy/install-client.sh --server https://ai.domain.tld --token "change-me" --use-llama-cpp-via-docker --use-ollama-models-path /usr/share/ollama/.ollama/models --llama-cpp-docker-image ghcr.io/ggml-org/llama.cpp:server-rocm --llama-cpp-base-port 8081 --llama-cpp-parallel 128
``` ```
or using ollama models with llama.cpp backend using CUDA: or using ollama models with llama.cpp backend using CUDA:
```bash ```bash
sudo bash deploy/install-client.sh --server https://ai.domain.tld --token "change-me" --use-llama-cpp-via-docker --use-ollama-models-path /usr/share/ollama/.ollama/models --llama-cpp-docker-image ghcr.io/ggml-org/llama.cpp:server-cuda --llama-cpp-base-port 8081 sudo bash deploy/install-client.sh --server https://ai.domain.tld --token "change-me" --use-llama-cpp-via-docker --use-ollama-models-path /usr/share/ollama/.ollama/models --llama-cpp-docker-image ghcr.io/ggml-org/llama.cpp:server-cuda --llama-cpp-base-port 8081 --llama-cpp-parallel 128
``` ```
Options: `--server`, `--token` (required); `--client-id`, `--upstream`, `--install-dir`, `--service-name`, `--no-ollama` (optional). Missing required values are prompted interactively. Options: `--server`, `--token` (required); `--client-id`, `--upstream`, `--install-dir`, `--service-name`, `--no-ollama` (optional). Missing required values are prompted interactively.
@@ -150,6 +150,15 @@ llama.cpp via Docker options (replaces Ollama for inferencing):
| `--use-ollama-models-path <dir>` | Path to Ollama models directory (`manifests/blobs`); required with the flag above | | `--use-ollama-models-path <dir>` | Path to Ollama models directory (`manifests/blobs`); required with the flag above |
| `--llama-cpp-docker-image <img>` | Docker image; defaults to auto-detected (rocm/cuda/cpu) | | `--llama-cpp-docker-image <img>` | Docker image; defaults to auto-detected (rocm/cuda/cpu) |
| `--llama-cpp-base-port <num>` | Base port for containers; defaults to `8081` | | `--llama-cpp-base-port <num>` | Base port for containers; defaults to `8081` |
| `--llama-cpp-parallel <num>` | llama.cpp parallel slots per container; if unset, llama.cpp's own default is used (which is `1`) |
| `--llama-cpp-fallback-cooldown <sec>` | Seconds before llama.cpp is retried after a failed container start; defaults to `180` |
| `--log-dir <dir>` | Directory for log files; defaults to `<app dir>/Logs` |
### llama.cpp fallback to Ollama
Models are served via llama.cpp Docker containers. If a container cannot be started for a model (for example, the model's GGUF blob is incompatible with the llama.cpp build), the client falls back to the Ollama upstream for that model. Transient start failures are remembered for `--llama-cpp-fallback-cooldown` seconds (default 3 minutes) and then retried; a container that starts but exits before becoming ready marks the model as falling back until it is unloaded. `load`/`unload` model commands and on-demand request routing are all covered; a failed container start is detected quickly by watching the container state, and the container log tail is written to the client log to aid debugging.
Note: some hybrid SSM/attention models (e.g. `qwen3.5-coder-next`) are converted by Ollama into a GGUF tensor layout that stock llama.cpp cannot load (`missing tensor 'blk.0.ssm_dt.bias'` and similar). Such models are served via the Ollama fallback above. If you want them to run on llama.cpp instead, use a Hugging Face-converted GGUF (e.g. `unsloth/Qwen3-Coder-Next-GGUF`) rather than the Ollama blob.
The script ensures .NET 10 and Ollama are installed, builds the client self-contained, installs it to `/opt/Ngino-client`, and creates a systemd service (`Ngino-client`). Logs: `journalctl -u Ngino-client -f`. The script ensures .NET 10 and Ollama are installed, builds the client self-contained, installs it to `/opt/Ngino-client`, and creates a systemd service (`Ngino-client`). Logs: `journalctl -u Ngino-client -f`.
+22 -1
View File
@@ -20,6 +20,9 @@ USE_LLAMA_CPP_VIA_DOCKER=false
USE_OLLAMA_MODELS_PATH="" USE_OLLAMA_MODELS_PATH=""
LLAMA_CPP_DOCKER_IMAGE="" LLAMA_CPP_DOCKER_IMAGE=""
LLAMA_CPP_BASE_PORT="" LLAMA_CPP_BASE_PORT=""
LLAMA_CPP_PARALLEL=""
LLAMA_CPP_FALLBACK_COOLDOWN=""
LOG_DIR=""
# ── Colors ──────────────────────────────────────────────────────────────────── # ── Colors ────────────────────────────────────────────────────────────────────
RED='\033[0;31m' RED='\033[0;31m'
@@ -57,13 +60,19 @@ Optional:
llama.cpp Docker image; defaults to auto-detected (rocm/cuda/cpu) llama.cpp Docker image; defaults to auto-detected (rocm/cuda/cpu)
--llama-cpp-base-port <num> --llama-cpp-base-port <num>
Base port for llama.cpp containers; defaults to 8081 Base port for llama.cpp containers; defaults to 8081
--llama-cpp-parallel <num>
llama.cpp parallel slots per container; if unset, llama.cpp's own default is used
--llama-cpp-fallback-cooldown <sec>
Seconds before llama.cpp is retried after a failed container start; defaults to 180
--log-dir <dir> Directory for log files; defaults to <install-dir>/Logs
-h, --help Show this help message -h, --help Show this help message
Examples: Examples:
$0 --server http://gpu-server:5050 --token "my-secret" $0 --server http://gpu-server:5050 --token "my-secret"
$0 --server http://gpu-server:5050 --token "my-secret" --no-ollama $0 --server http://gpu-server:5050 --token "my-secret" --no-ollama
$0 --server http://gpu-server:5050 --token "my-secret" \\ $0 --server http://gpu-server:5050 --token "my-secret" \\
--use-llama-cpp-via-docker --use-ollama-models-path /usr/share/ollama/.ollama/models --use-llama-cpp-via-docker --use-ollama-models-path /usr/share/ollama/.ollama/models \\
--llama-cpp-parallel 128
EOF EOF
exit 0 exit 0
} }
@@ -82,6 +91,9 @@ while [[ $# -gt 0 ]]; do
--use-ollama-models-path) USE_OLLAMA_MODELS_PATH="$2"; shift 2 ;; --use-ollama-models-path) USE_OLLAMA_MODELS_PATH="$2"; shift 2 ;;
--llama-cpp-docker-image) LLAMA_CPP_DOCKER_IMAGE="$2"; shift 2 ;; --llama-cpp-docker-image) LLAMA_CPP_DOCKER_IMAGE="$2"; shift 2 ;;
--llama-cpp-base-port) LLAMA_CPP_BASE_PORT="$2"; shift 2 ;; --llama-cpp-base-port) LLAMA_CPP_BASE_PORT="$2"; shift 2 ;;
--llama-cpp-parallel) LLAMA_CPP_PARALLEL="$2"; shift 2 ;;
--llama-cpp-fallback-cooldown) LLAMA_CPP_FALLBACK_COOLDOWN="$2"; shift 2 ;;
--log-dir) LOG_DIR="$2"; shift 2 ;;
-h|--help) usage ;; -h|--help) usage ;;
*) die "Unknown option: $1" ;; *) die "Unknown option: $1" ;;
esac esac
@@ -311,6 +323,15 @@ mkdir -p "$ENV_DIR"
if [[ -n "$LLAMA_CPP_BASE_PORT" ]]; then if [[ -n "$LLAMA_CPP_BASE_PORT" ]]; then
printf 'NGINO_LLAMA_CPP_BASE_PORT=%s\n' "$LLAMA_CPP_BASE_PORT" printf 'NGINO_LLAMA_CPP_BASE_PORT=%s\n' "$LLAMA_CPP_BASE_PORT"
fi fi
if [[ -n "$LLAMA_CPP_PARALLEL" ]]; then
printf 'NGINO_LLAMA_CPP_PARALLEL=%s\n' "$LLAMA_CPP_PARALLEL"
fi
if [[ -n "$LLAMA_CPP_FALLBACK_COOLDOWN" ]]; then
printf 'NGINO_LLAMA_CPP_FALLBACK_COOLDOWN_SECONDS=%s\n' "$LLAMA_CPP_FALLBACK_COOLDOWN"
fi
fi
if [[ -n "$LOG_DIR" ]]; then
printf 'NGINO_LOG_DIR=%s\n' "$LOG_DIR"
fi fi
} > "$ENV_DIR/env" } > "$ENV_DIR/env"
chmod 600 "$ENV_DIR/env" chmod 600 "$ENV_DIR/env"
+20 -1
View File
@@ -28,6 +28,12 @@ internal sealed class ClientOptions
public int LlamaCppBasePort { get; init; } = 8081; public int LlamaCppBasePort { get; init; } = 8081;
public int? LlamaCppParallel { get; init; }
public TimeSpan LlamaCppFallbackCooldown { get; init; } = TimeSpan.FromMinutes(3);
public string? LogDirectory { get; init; }
public Uri TunnelUri public Uri TunnelUri
{ {
get get
@@ -68,7 +74,10 @@ internal sealed class ClientOptions
UseLlamaCppViaDocker = ReadBool(values, false, "use-llama-cpp-via-docker", "NGINO_USE_LLAMA_CPP_VIA_DOCKER"), UseLlamaCppViaDocker = ReadBool(values, false, "use-llama-cpp-via-docker", "NGINO_USE_LLAMA_CPP_VIA_DOCKER"),
UseOllamaModelsPath = NormalizeDirectoryPath(Read(values, "use-ollama-models-path", "NGINO_USE_OLLAMA_MODELS_PATH")), UseOllamaModelsPath = NormalizeDirectoryPath(Read(values, "use-ollama-models-path", "NGINO_USE_OLLAMA_MODELS_PATH")),
LlamaCppDockerImage = Read(values, "llama-cpp-docker-image", "NGINO_LLAMA_CPP_DOCKER_IMAGE"), LlamaCppDockerImage = Read(values, "llama-cpp-docker-image", "NGINO_LLAMA_CPP_DOCKER_IMAGE"),
LlamaCppBasePort = ReadInt(values, 8081, "llama-cpp-base-port", "NGINO_LLAMA_CPP_BASE_PORT") LlamaCppBasePort = ReadInt(values, 8081, "llama-cpp-base-port", "NGINO_LLAMA_CPP_BASE_PORT"),
LlamaCppParallel = ReadOptionalInt(values, "llama-cpp-parallel", "NGINO_LLAMA_CPP_PARALLEL"),
LlamaCppFallbackCooldown = TimeSpan.FromSeconds(ReadInt(values, 180, "llama-cpp-fallback-cooldown", "NGINO_LLAMA_CPP_FALLBACK_COOLDOWN_SECONDS")),
LogDirectory = NormalizeDirectoryPath(Read(values, "log-dir", "NGINO_LOG_DIR"))
}; };
} }
@@ -87,6 +96,10 @@ internal sealed class ClientOptions
--use-ollama-models-path <dir> Path to Ollama models directory (manifests/blobs), required with --use-llama-cpp-via-docker --use-ollama-models-path <dir> Path to Ollama models directory (manifests/blobs), required with --use-llama-cpp-via-docker
--llama-cpp-docker-image <img> llama.cpp Docker image; defaults to auto-detected (rocm/cuda/cpu) --llama-cpp-docker-image <img> llama.cpp Docker image; defaults to auto-detected (rocm/cuda/cpu)
--llama-cpp-base-port <num> Base port for llama.cpp containers; defaults to 8081 --llama-cpp-base-port <num> Base port for llama.cpp containers; defaults to 8081
--llama-cpp-parallel <num> llama.cpp parallel slots per container; if unset, llama.cpp's own default is used
--llama-cpp-fallback-cooldown <sec>
Seconds before llama.cpp is retried after a failed container start; defaults to 180
--log-dir <dir> Directory for log files; defaults to <app dir>/Logs
"""; """;
private static Dictionary<string, string> ParseArgs(string[] args) private static Dictionary<string, string> ParseArgs(string[] args)
@@ -160,6 +173,12 @@ internal sealed class ClientOptions
return int.TryParse(value, out var parsed) && parsed > 0 ? parsed : fallback; return int.TryParse(value, out var parsed) && parsed > 0 ? parsed : fallback;
} }
private static int? ReadOptionalInt(Dictionary<string, string> values, params string[] keys)
{
var value = Read(values, keys);
return int.TryParse(value, out var parsed) && parsed > 0 ? parsed : null;
}
private static bool ReadBool(Dictionary<string, string> values, bool fallback, params string[] keys) private static bool ReadBool(Dictionary<string, string> values, bool fallback, params string[] keys)
{ {
var value = Read(values, keys); var value = Read(values, keys);
+113
View File
@@ -0,0 +1,113 @@
using System.Text;
using Microsoft.Extensions.Logging;
namespace Ngino.Client;
internal sealed class FileLoggerProvider : ILoggerProvider
{
private const long DefaultMaxFileSizeBytes = 5L * 1024 * 1024;
private const string LogFileName = "ngino-client.log";
private const string RotatedLogFileName = "ngino-client.log.1";
private readonly string _directory;
private readonly long _maxFileSizeBytes;
private readonly object _lock = new();
private StreamWriter _writer = null!;
private string _currentFile = null!;
public FileLoggerProvider(string directory, long maxFileSizeBytes = DefaultMaxFileSizeBytes)
{
_directory = directory;
_maxFileSizeBytes = maxFileSizeBytes;
Directory.CreateDirectory(directory);
OpenFile();
}
public string LogDirectory => _directory;
public ILogger CreateLogger(string categoryName) => new FileLogger(this, categoryName);
public void WriteLog(DateTime timestamp, LogLevel level, string category, string message)
{
lock (_lock)
{
var line = $"{timestamp:yyyy-MM-dd HH:mm:ss.fff} [{level}] {category}: {message}";
if (_writer.BaseStream.Length + line.Length + 2 > _maxFileSizeBytes)
{
RotateFile();
}
_writer.WriteLine(line);
}
}
public void Dispose()
{
lock (_lock)
{
_writer.Dispose();
}
}
private void OpenFile()
{
_currentFile = Path.Combine(_directory, LogFileName);
_writer = new StreamWriter(
new FileStream(_currentFile, FileMode.Append, FileAccess.Write, FileShare.ReadWrite),
new UTF8Encoding(encoderShouldEmitUTF8Identifier: false))
{
AutoFlush = true
};
}
private void RotateFile()
{
_writer.Dispose();
var rotatedFile = Path.Combine(_directory, RotatedLogFileName);
try
{
File.Delete(rotatedFile);
if (File.Exists(_currentFile))
{
File.Move(_currentFile, rotatedFile);
}
}
catch (IOException)
{
}
catch (UnauthorizedAccessException)
{
}
OpenFile();
}
}
internal sealed class FileLogger(FileLoggerProvider provider, string category) : ILogger
{
public IDisposable? BeginScope<TState>(TState state) where TState : notnull => null;
public bool IsEnabled(LogLevel logLevel) => logLevel >= LogLevel.Trace;
public void Log<TState>(
LogLevel logLevel,
EventId eventId,
TState state,
Exception? exception,
Func<TState, Exception?, string> formatter)
{
if (!IsEnabled(logLevel))
{
return;
}
var message = formatter(state, exception);
if (exception is not null)
{
message += Environment.NewLine + exception;
}
provider.WriteLog(DateTime.Now, logLevel, category, message);
}
}
+217 -27
View File
@@ -16,24 +16,35 @@ internal sealed partial class LlamaCppManager : IAsyncDisposable
private const string NginoContainerLabel = "ngino-llamacpp"; private const string NginoContainerLabel = "ngino-llamacpp";
private static readonly TimeSpan DockerTimeout = TimeSpan.FromSeconds(60); private static readonly TimeSpan DockerTimeout = TimeSpan.FromSeconds(60);
private static readonly TimeSpan ContainerStartTimeout = TimeSpan.FromMinutes(5); private static readonly TimeSpan ContainerStartTimeout = TimeSpan.FromMinutes(5);
private static readonly TimeSpan DefaultFallbackCooldown = TimeSpan.FromMinutes(3);
private readonly string _blobsPath; private readonly string _blobsPath;
private readonly string _manifestsPath; private readonly string _manifestsPath;
private readonly string _dockerImage; private readonly string _dockerImage;
private readonly int _basePort; private readonly int _basePort;
private readonly int? _parallel;
private readonly TimeSpan _fallbackCooldown;
private readonly ILogger _logger; private readonly ILogger _logger;
private readonly ConcurrentDictionary<string, int> _modelPorts = new(StringComparer.OrdinalIgnoreCase); private readonly ConcurrentDictionary<string, int> _modelPorts = new(StringComparer.OrdinalIgnoreCase);
private readonly ConcurrentDictionary<int, byte> _reservedPorts = new();
private readonly ConcurrentDictionary<string, DateTime> _fallbackModels = new(StringComparer.OrdinalIgnoreCase);
private readonly ConcurrentDictionary<string, SemaphoreSlim> _modelStartLocks = new(StringComparer.OrdinalIgnoreCase);
private readonly object _portAllocationLock = new();
public LlamaCppManager( public LlamaCppManager(
string ollamaModelsPath, string ollamaModelsPath,
string? dockerImage, string? dockerImage,
int? basePort, int? basePort,
ILogger? logger = null) ILogger? logger = null,
TimeSpan fallbackCooldown = default,
int? parallel = null)
{ {
_manifestsPath = Path.Combine(ollamaModelsPath, "manifests"); _manifestsPath = Path.Combine(ollamaModelsPath, "manifests");
_blobsPath = Path.Combine(ollamaModelsPath, "blobs"); _blobsPath = Path.Combine(ollamaModelsPath, "blobs");
_dockerImage = dockerImage ?? GetDefaultDockerImage(); _dockerImage = dockerImage ?? GetDefaultDockerImage();
_basePort = basePort ?? DefaultBasePort; _basePort = basePort ?? DefaultBasePort;
_fallbackCooldown = fallbackCooldown > TimeSpan.Zero ? fallbackCooldown : DefaultFallbackCooldown;
_parallel = parallel is > 0 ? parallel : null;
_logger = logger ?? NullLogger<LlamaCppManager>.Instance; _logger = logger ?? NullLogger<LlamaCppManager>.Instance;
} }
@@ -78,6 +89,45 @@ internal sealed partial class LlamaCppManager : IAsyncDisposable
return _modelPorts.ContainsKey(ollamaModelName); return _modelPorts.ContainsKey(ollamaModelName);
} }
public void MarkModelAsFallback(string ollamaModelName)
{
_fallbackModels[ollamaModelName] = DateTime.UtcNow.Add(_fallbackCooldown);
_logger.LogWarning(
"Model {Model} will fall back to the Ollama upstream for {Cooldown} before llama.cpp is retried.",
ollamaModelName, _fallbackCooldown);
}
public void MarkModelAsPermanentFallback(string ollamaModelName)
{
_fallbackModels[ollamaModelName] = DateTime.MaxValue;
_logger.LogError(
"Model {Model} exited its llama.cpp container before becoming ready. It will fall back to the Ollama upstream until it is unloaded.",
ollamaModelName);
}
public bool IsModelOnFallback(string ollamaModelName)
{
if (_fallbackModels.TryGetValue(ollamaModelName, out var expiresAt))
{
if (expiresAt > DateTime.UtcNow)
{
return true;
}
_fallbackModels.TryRemove(ollamaModelName, out _);
}
return false;
}
public void ClearModelFallback(string ollamaModelName)
{
if (_fallbackModels.TryRemove(ollamaModelName, out _))
{
_logger.LogInformation("Cleared llama.cpp fallback marker for model {Model}.", ollamaModelName);
}
}
public async Task<bool> IsContainerRunningAsync(string ollamaModelName) public async Task<bool> IsContainerRunningAsync(string ollamaModelName)
{ {
if (!_modelPorts.ContainsKey(ollamaModelName)) if (!_modelPorts.ContainsKey(ollamaModelName))
@@ -96,7 +146,7 @@ internal sealed partial class LlamaCppManager : IAsyncDisposable
_logger.LogWarning( _logger.LogWarning(
"llama.cpp container {ContainerName} is no longer running. Invalidating cached port for {Model}.", "llama.cpp container {ContainerName} is no longer running. Invalidating cached port for {Model}.",
containerName, ollamaModelName); containerName, ollamaModelName);
_modelPorts.TryRemove(ollamaModelName, out _); RemoveModelPort(ollamaModelName);
} }
return running; return running;
@@ -104,11 +154,11 @@ internal sealed partial class LlamaCppManager : IAsyncDisposable
public bool RemoveModelMapping(string ollamaModelName) public bool RemoveModelMapping(string ollamaModelName)
{ {
if (_modelPorts.TryRemove(ollamaModelName, out var port)) if (RemoveModelPort(ollamaModelName))
{ {
_logger.LogWarning( _logger.LogWarning(
"Removed stale llama.cpp port mapping for {Model} (port {Port}).", "Removed stale llama.cpp port mapping for {Model}.",
ollamaModelName, port); ollamaModelName);
return true; return true;
} }
@@ -134,21 +184,52 @@ internal sealed partial class LlamaCppManager : IAsyncDisposable
return false; return false;
} }
var startLock = _modelStartLocks.GetOrAdd(ollamaName, static _ => new SemaphoreSlim(1, 1));
await startLock.WaitAsync(cancellationToken);
try
{
return await StartModelContainerCoreAsync(model, cancellationToken);
}
finally
{
startLock.Release();
}
}
private async Task<bool> StartModelContainerCoreAsync(LlamaCppModel model, CancellationToken cancellationToken)
{
var ollamaName = model.OllamaName;
if (string.IsNullOrWhiteSpace(ollamaName))
{
return false;
}
if (_modelPorts.ContainsKey(ollamaName)) if (_modelPorts.ContainsKey(ollamaName))
{ {
_logger.LogInformation("Model {Model} already has a running container", ollamaName); _logger.LogInformation("Model {Model} already has a running container", ollamaName);
return true; return true;
} }
if (IsModelOnFallback(ollamaName))
{
_logger.LogInformation(
"Model {Model} previously failed to load via llama.cpp. Skipping container start.",
ollamaName);
return false;
}
if (!File.Exists(model.BlobPath)) if (!File.Exists(model.BlobPath))
{ {
_logger.LogError("Model blob not found: {BlobPath}", model.BlobPath); _logger.LogError("Model blob not found: {BlobPath}", model.BlobPath);
MarkModelAsFallback(ollamaName);
return false; return false;
} }
var port = FindAvailablePort(); var port = FindAvailablePort();
var containerName = SanitizeContainerName($"ngino-llamacpp-{ollamaName}"); var containerName = SanitizeContainerName($"ngino-llamacpp-{ollamaName}");
try
{
var existingPort = await FindExistingContainerPortAsync(containerName); var existingPort = await FindExistingContainerPortAsync(containerName);
if (existingPort.HasValue) if (existingPort.HasValue)
{ {
@@ -165,14 +246,13 @@ internal sealed partial class LlamaCppManager : IAsyncDisposable
"Starting llama.cpp container for {Model} on port {Port}: docker {Args}", "Starting llama.cpp container for {Model} on port {Port}: docker {Args}",
ollamaName, port, string.Join(" ", args)); ollamaName, port, string.Join(" ", args));
try
{
var (exitCode, output) = await RunDockerWithOutputAsync(args, cancellationToken); var (exitCode, output) = await RunDockerWithOutputAsync(args, cancellationToken);
if (exitCode != 0) if (exitCode != 0)
{ {
_logger.LogError( _logger.LogError(
"Failed to start llama.cpp container for {Model}, exit code: {ExitCode}, output: {Output}", "Failed to start llama.cpp container for {Model}, exit code: {ExitCode}, output: {Output}",
ollamaName, exitCode, output); ollamaName, exitCode, output);
MarkModelAsFallback(ollamaName);
return false; return false;
} }
@@ -180,12 +260,21 @@ internal sealed partial class LlamaCppManager : IAsyncDisposable
"llama.cpp container for {Model} started on port {Port}. Waiting for it to become ready...", "llama.cpp container for {Model} started on port {Port}. Waiting for it to become ready...",
ollamaName, port); ollamaName, port);
var ready = await WaitForServerReadyAsync("localhost", port, cancellationToken); var result = await WaitForServerReadyAsync("localhost", port, containerName, cancellationToken);
if (!ready) if (result != ContainerStartResult.Ready)
{ {
if (result == ContainerStartResult.ContainerExited)
{
MarkModelAsPermanentFallback(ollamaName);
}
else
{
MarkModelAsFallback(ollamaName);
}
_logger.LogError( _logger.LogError(
"llama.cpp container for {Model} did not become ready on port {Port} within {Timeout}. Stopping it.", "llama.cpp container for {Model} did not become ready on port {Port} within {Timeout} ({Result}). Stopping it.",
ollamaName, port, ContainerStartTimeout); ollamaName, port, ContainerStartTimeout, result);
try try
{ {
@@ -207,13 +296,18 @@ internal sealed partial class LlamaCppManager : IAsyncDisposable
catch (Exception ex) catch (Exception ex)
{ {
_logger.LogError(ex, "Failed to start llama.cpp container for {Model}", ollamaName); _logger.LogError(ex, "Failed to start llama.cpp container for {Model}", ollamaName);
MarkModelAsFallback(ollamaName);
return false; return false;
} }
finally
{
ReleaseReservedPort(port);
}
} }
public async Task<bool> StopModelContainerAsync(string ollamaModelName, CancellationToken cancellationToken) public async Task<bool> StopModelContainerAsync(string ollamaModelName, CancellationToken cancellationToken)
{ {
if (!_modelPorts.TryRemove(ollamaModelName, out _)) if (!RemoveModelPort(ollamaModelName))
{ {
_logger.LogWarning("No running container found for model {Model}", ollamaModelName); _logger.LogWarning("No running container found for model {Model}", ollamaModelName);
return false; return false;
@@ -227,6 +321,7 @@ internal sealed partial class LlamaCppManager : IAsyncDisposable
{ {
await RunDockerAsync(["stop", "--time", "10", containerName], cancellationToken); await RunDockerAsync(["stop", "--time", "10", containerName], cancellationToken);
await RunDockerAsync(["rm", "-f", containerName], cancellationToken); await RunDockerAsync(["rm", "-f", containerName], cancellationToken);
ClearModelFallback(ollamaModelName);
return true; return true;
} }
catch (Exception ex) catch (Exception ex)
@@ -262,6 +357,8 @@ internal sealed partial class LlamaCppManager : IAsyncDisposable
} }
_modelPorts.Clear(); _modelPorts.Clear();
_reservedPorts.Clear();
_fallbackModels.Clear();
} }
public async Task<bool> TestDockerAsync() public async Task<bool> TestDockerAsync()
@@ -389,7 +486,6 @@ internal sealed partial class LlamaCppManager : IAsyncDisposable
{ {
"run", "run",
"-d", "-d",
"--rm",
"--label", $"{NginoContainerLabel}=true", "--label", $"{NginoContainerLabel}=true",
"--name", containerName, "--name", containerName,
"-p", $"{port}:{port}", "-p", $"{port}:{port}",
@@ -413,9 +509,12 @@ internal sealed partial class LlamaCppManager : IAsyncDisposable
args.Add("-m"); args.Add("-m");
args.Add($"/models/blobs/{blobFile}"); args.Add($"/models/blobs/{blobFile}");
args.Add("-ngl"); args.Add("-ngl");
args.Add("999"); args.Add("auto");
if (_parallel.HasValue)
{
args.Add("--parallel"); args.Add("--parallel");
args.Add("4"); args.Add(_parallel.Value.ToString());
}
args.Add("--host"); args.Add("--host");
args.Add("0.0.0.0"); args.Add("0.0.0.0");
args.Add("--port"); args.Add("--port");
@@ -475,8 +574,15 @@ internal sealed partial class LlamaCppManager : IAsyncDisposable
} }
private int FindAvailablePort() private int FindAvailablePort()
{
lock (_portAllocationLock)
{ {
var usedPorts = new HashSet<int>(_modelPorts.Values); var usedPorts = new HashSet<int>(_modelPorts.Values);
foreach (var reservedPort in _reservedPorts.Keys)
{
usedPorts.Add(reservedPort);
}
var port = _basePort; var port = _basePort;
while (usedPorts.Contains(port)) while (usedPorts.Contains(port))
@@ -484,17 +590,36 @@ internal sealed partial class LlamaCppManager : IAsyncDisposable
port++; port++;
} }
_reservedPorts[port] = 0;
return port; return port;
} }
}
private static async Task<bool> WaitForServerReadyAsync( private void ReleaseReservedPort(int port)
string host, int port, CancellationToken cancellationToken) {
_reservedPorts.TryRemove(port, out _);
}
private bool RemoveModelPort(string ollamaModelName)
{
if (_modelPorts.TryRemove(ollamaModelName, out var port))
{
ReleaseReservedPort(port);
return true;
}
return false;
}
private async Task<ContainerStartResult> WaitForServerReadyAsync(
string host, int port, string containerName, CancellationToken cancellationToken)
{ {
var deadline = DateTime.UtcNow + ContainerStartTimeout; var deadline = DateTime.UtcNow + ContainerStartTimeout;
if (!await WaitForTcpPortAsync(host, port, deadline, cancellationToken)) var tcpResult = await WaitForTcpPortAsync(host, port, containerName, deadline, cancellationToken);
if (tcpResult != ContainerStartResult.Ready)
{ {
return false; return tcpResult;
} }
using var handler = new SocketsHttpHandler using var handler = new SocketsHttpHandler
@@ -508,41 +633,57 @@ internal sealed partial class LlamaCppManager : IAsyncDisposable
{ {
cancellationToken.ThrowIfCancellationRequested(); cancellationToken.ThrowIfCancellationRequested();
if (!await IsDockerContainerRunningAsync(containerName))
{
await LogContainerOutputAsync(containerName);
return ContainerStartResult.ContainerExited;
}
try try
{ {
using var response = await httpClient.GetAsync( using var response = await httpClient.GetAsync(
$"http://{host}:{port}/health", cancellationToken); $"http://{host}:{port}/health", cancellationToken);
if (response.IsSuccessStatusCode) if (response.IsSuccessStatusCode)
{ {
return true; return ContainerStartResult.Ready;
} }
} }
catch (OperationCanceledException) when (!cancellationToken.IsCancellationRequested) catch (OperationCanceledException) when (!cancellationToken.IsCancellationRequested)
{ {
} }
catch (Exception) catch (HttpRequestException ex)
{ {
return false; _logger.LogDebug(ex, "Health probe of llama.cpp container {ContainerName} failed; retrying.", containerName);
}
catch (IOException ex)
{
_logger.LogDebug(ex, "Health probe of llama.cpp container {ContainerName} failed; retrying.", containerName);
} }
await Task.Delay(TimeSpan.FromSeconds(2), cancellationToken); await Task.Delay(TimeSpan.FromSeconds(2), cancellationToken);
} }
return false; return ContainerStartResult.TimedOut;
} }
private static async Task<bool> WaitForTcpPortAsync( private async Task<ContainerStartResult> WaitForTcpPortAsync(
string host, int port, DateTime deadline, CancellationToken cancellationToken) string host, int port, string containerName, DateTime deadline, CancellationToken cancellationToken)
{ {
while (DateTime.UtcNow < deadline) while (DateTime.UtcNow < deadline)
{ {
cancellationToken.ThrowIfCancellationRequested(); cancellationToken.ThrowIfCancellationRequested();
if (!await IsDockerContainerRunningAsync(containerName))
{
await LogContainerOutputAsync(containerName);
return ContainerStartResult.ContainerExited;
}
try try
{ {
using var client = new TcpClient(); using var client = new TcpClient();
await client.ConnectAsync(host, port, cancellationToken); await client.ConnectAsync(host, port, cancellationToken);
return true; return ContainerStartResult.Ready;
} }
catch (OperationCanceledException) catch (OperationCanceledException)
{ {
@@ -555,7 +696,49 @@ internal sealed partial class LlamaCppManager : IAsyncDisposable
await Task.Delay(TimeSpan.FromSeconds(2), cancellationToken); await Task.Delay(TimeSpan.FromSeconds(2), cancellationToken);
} }
return false; return ContainerStartResult.TimedOut;
}
private async Task<bool> IsDockerContainerRunningAsync(string containerName)
{
try
{
var (exitCode, output) = await RunDockerWithOutputAsync(
["inspect", "-f", "{{.State.Running}}", containerName],
CancellationToken.None);
return exitCode == 0 && string.Equals(output.Trim(), "true", StringComparison.OrdinalIgnoreCase);
}
catch
{
return true;
}
}
private async Task LogContainerOutputAsync(string containerName)
{
try
{
var (_, output) = await RunDockerWithOutputAsync(
["logs", "--tail", "100", containerName],
CancellationToken.None);
if (!string.IsNullOrWhiteSpace(output))
{
_logger.LogError(
"llama.cpp container {ContainerName} exited before becoming ready. Last output:\n{Output}",
containerName, output);
}
else
{
_logger.LogError(
"llama.cpp container {ContainerName} exited before becoming ready, but produced no output.",
containerName);
}
}
catch (Exception ex)
{
_logger.LogWarning(ex, "Failed to read logs of container {ContainerName}", containerName);
}
} }
private static string SanitizeContainerName(string name) private static string SanitizeContainerName(string name)
@@ -655,6 +838,13 @@ internal sealed partial class LlamaCppManager : IAsyncDisposable
[GeneratedRegex(@"[^a-zA-Z0-9_.-]")] [GeneratedRegex(@"[^a-zA-Z0-9_.-]")]
private static partial Regex InvalidContainerNameChars(); private static partial Regex InvalidContainerNameChars();
private enum ContainerStartResult
{
Ready,
ContainerExited,
TimedOut
}
} }
internal sealed record LlamaCppModel internal sealed record LlamaCppModel
+8
View File
@@ -22,8 +22,15 @@ try
Console.WriteLine($" ollama models path: {options.UseOllamaModelsPath ?? "(not set)"}"); Console.WriteLine($" ollama models path: {options.UseOllamaModelsPath ?? "(not set)"}");
Console.WriteLine($" llama.cpp docker image: {options.LlamaCppDockerImage ?? "(auto)"}"); Console.WriteLine($" llama.cpp docker image: {options.LlamaCppDockerImage ?? "(auto)"}");
Console.WriteLine($" llama.cpp base port: {options.LlamaCppBasePort}"); Console.WriteLine($" llama.cpp base port: {options.LlamaCppBasePort}");
Console.WriteLine(
options.LlamaCppParallel.HasValue
? $" llama.cpp parallel slots: {options.LlamaCppParallel.Value}"
: " llama.cpp parallel slots: (llama.cpp default)");
} }
var logDirectory = options.LogDirectory ?? Path.Combine(AppContext.BaseDirectory, "Logs");
Console.WriteLine($" log directory: {logDirectory}");
// Args are parsed by ClientOptions; keep them away from the host configuration. // Args are parsed by ClientOptions; keep them away from the host configuration.
var builder = Host.CreateApplicationBuilder(new HostApplicationBuilderSettings { Args = [] }); var builder = Host.CreateApplicationBuilder(new HostApplicationBuilderSettings { Args = [] });
builder.Services.AddSingleton(options); builder.Services.AddSingleton(options);
@@ -32,6 +39,7 @@ try
builder.Services.AddWindowsService(service => service.ServiceName = "NginoClient"); builder.Services.AddWindowsService(service => service.ServiceName = "NginoClient");
// The EventLog provider defaults to Warning; connection state is worth seeing there. // The EventLog provider defaults to Warning; connection state is worth seeing there.
builder.Logging.AddFilter<Microsoft.Extensions.Logging.EventLog.EventLogLoggerProvider>("Ngino.Client", LogLevel.Information); builder.Logging.AddFilter<Microsoft.Extensions.Logging.EventLog.EventLogLoggerProvider>("Ngino.Client", LogLevel.Information);
builder.Logging.AddProvider(new FileLoggerProvider(logDirectory));
await builder.Build().RunAsync(); await builder.Build().RunAsync();
return 0; return 0;
+37 -61
View File
@@ -47,7 +47,9 @@ internal sealed class TunnelClient
_options.UseOllamaModelsPath, _options.UseOllamaModelsPath,
_options.LlamaCppDockerImage, _options.LlamaCppDockerImage,
_options.LlamaCppBasePort, _options.LlamaCppBasePort,
_logger); _logger,
_options.LlamaCppFallbackCooldown,
_options.LlamaCppParallel);
} }
} }
@@ -477,13 +479,16 @@ internal sealed class TunnelClient
var started = await _llamaCppManager.StartModelContainerAsync(model, cancellationToken); var started = await _llamaCppManager.StartModelContainerAsync(model, cancellationToken);
if (!started) if (!started)
{ {
return new TunnelMessage _logger.LogWarning(
{ "Unable to load model '{Model}' via llama.cpp. Falling back to Ollama upstream.",
Type = TunnelMessageTypes.ModelCommandResult, modelName);
RequestId = message.RequestId,
StatusCode = 500, using var fallbackRequest = BuildModelCommandRequest(_options.Upstream, "load", modelName);
Error = $"Unable to load model '{modelName}'." using var fallbackResponse = await _httpClient.SendAsync(
}; fallbackRequest, HttpCompletionOption.ResponseHeadersRead, cancellationToken);
var fallbackBody = await fallbackResponse.Content.ReadAsByteArrayAsync(cancellationToken);
return BuildModelCommandResult(message.RequestId, fallbackResponse, fallbackBody);
} }
return BuildModelCommandResult(message.RequestId, 200, "OK", []); return BuildModelCommandResult(message.RequestId, 200, "OK", []);
@@ -494,13 +499,18 @@ internal sealed class TunnelClient
var stopped = await _llamaCppManager!.StopModelContainerAsync(modelName!, cancellationToken); var stopped = await _llamaCppManager!.StopModelContainerAsync(modelName!, cancellationToken);
if (!stopped) if (!stopped)
{ {
return new TunnelMessage _logger.LogWarning(
{ "No running llama.cpp container for model '{Model}'. Falling back to Ollama upstream to unload it.",
Type = TunnelMessageTypes.ModelCommandResult, modelName);
RequestId = message.RequestId,
StatusCode = 404, _llamaCppManager.ClearModelFallback(modelName!);
Error = $"No running llama.cpp container for model '{modelName}'."
}; using var fallbackRequest = BuildModelCommandRequest(_options.Upstream, "unload", modelName);
using var fallbackResponse = await _httpClient.SendAsync(
fallbackRequest, HttpCompletionOption.ResponseHeadersRead, cancellationToken);
var fallbackBody = await fallbackResponse.Content.ReadAsByteArrayAsync(cancellationToken);
return BuildModelCommandResult(message.RequestId, fallbackResponse, fallbackBody);
} }
return BuildModelCommandResult(message.RequestId, 200, "OK", []); return BuildModelCommandResult(message.RequestId, 200, "OK", []);
@@ -712,6 +722,14 @@ internal sealed class TunnelClient
.FirstOrDefault(m => string.Equals(m.OllamaName, modelName, StringComparison.OrdinalIgnoreCase)); .FirstOrDefault(m => string.Equals(m.OllamaName, modelName, StringComparison.OrdinalIgnoreCase));
if (model is not null) if (model is not null)
{
if (_llamaCppManager.IsModelOnFallback(modelName))
{
_logger.LogInformation(
"Model '{Model}' previously failed to load via llama.cpp. Routing directly to the Ollama upstream.",
modelName);
}
else
{ {
_logger.LogInformation( _logger.LogInformation(
"Request for model '{Model}' but no llama.cpp container is running. Starting one on demand...", "Request for model '{Model}' but no llama.cpp container is running. Starting one on demand...",
@@ -724,8 +742,10 @@ internal sealed class TunnelClient
} }
else else
{ {
await SendModelLoadErrorAsync(socket, message, modelName, cancellationToken); _logger.LogWarning(
return; "Unable to load model '{Model}' via llama.cpp. Falling back to the Ollama upstream.",
modelName);
}
} }
} }
else else
@@ -830,50 +850,6 @@ internal sealed class TunnelClient
} }
} }
private async Task SendModelLoadErrorAsync(
ClientWebSocket socket, TunnelMessage message, string modelName, CancellationToken cancellationToken)
{
_logger.LogWarning(
"Unable to load model '{Model}' via llama.cpp. Notifying caller.", modelName);
var body = JsonSerializer.SerializeToUtf8Bytes(
new { error = $"Unable to load model '{modelName}'" },
JsonOptions);
_pendingRequestBodies.TryRemove(message.RequestId, out _);
await SendAsync(
socket,
new TunnelMessage
{
Type = TunnelMessageTypes.HttpResponseHeaders,
RequestId = message.RequestId,
StatusCode = 500,
ReasonPhrase = "Internal Server Error",
Headers = [new HeaderPair("Content-Type", "application/json")]
},
CancellationToken.None);
await SendAsync(
socket,
new TunnelMessage
{
Type = TunnelMessageTypes.HttpResponseBody,
RequestId = message.RequestId,
Body = body
},
CancellationToken.None);
await SendAsync(
socket,
new TunnelMessage
{
Type = TunnelMessageTypes.HttpResponseComplete,
RequestId = message.RequestId
},
CancellationToken.None);
}
private void CancelAllActiveRequests() private void CancelAllActiveRequests()
{ {
foreach (var pair in _activeRequests.ToArray()) foreach (var pair in _activeRequests.ToArray())