Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
94 changes: 84 additions & 10 deletions src/OpenClaw.Connection/LocalAi/LlamaServerRouterConfiguration.cs
Original file line number Diff line number Diff line change
Expand Up @@ -31,19 +31,27 @@ public static LlamaServerRouterLaunchPlan Build(
LocalAiPaths paths,
LocalAiResolvedInstall install,
int? listenPort = null) =>
BuildCore(paths, install, install.ModelPath, listenPort);
BuildCore(paths, install, install.ModelPath, verifiedDraftModelPath: null, listenPort);

/// <param name="verifiedDraftModelPath">
/// The draft checkpoint's handle-resolved physical path, from the same verification
/// that opened it. Passing the persisted snapshot path instead would let a
/// snapshot-link replacement change the file llama-server finally opens, which is
/// exactly what resolving the primary model through its own handle prevents.
/// </param>
internal static LlamaServerRouterLaunchPlan BuildForVerifiedRuntime(
LocalAiPaths paths,
LocalAiResolvedInstall install,
string verifiedModelPath,
string? verifiedDraftModelPath,
int? listenPort = null) =>
BuildCore(paths, install, verifiedModelPath, listenPort);
BuildCore(paths, install, verifiedModelPath, verifiedDraftModelPath, listenPort);

private static LlamaServerRouterLaunchPlan BuildCore(
LocalAiPaths paths,
LocalAiResolvedInstall install,
string modelPath,
string? verifiedDraftModelPath,
int? listenPort)
{
ArgumentNullException.ThrowIfNull(paths);
Expand All @@ -53,13 +61,16 @@ private static LlamaServerRouterLaunchPlan BuildCore(
LocalAiInstallManifest manifest = install.Manifest;
int port = listenPort ?? manifest.RequestedPort;
LocalAiPortPolicy.Validate(port);
LlamaRuntimeVariant runtime = LlamaRuntimeCatalog.Variants.SingleOrDefault(
candidate => string.Equals(candidate.Id, manifest.RuntimeId, StringComparison.Ordinal))
// FindInstalled, not Variants: an installation recorded before the last
// runtime bump must keep launching until setup upgrades it, instead of being
// stranded the moment the catalog moves to a newer pinned release.
LlamaRuntimeVariant runtime = LlamaRuntimeCatalog.FindInstalled(manifest.RuntimeId)
?? throw new InvalidDataException("The managed llama-server runtime is no longer qualified.");
LocalModelInfo model = LocalModelCatalog.FindInstalled(manifest.ModelCatalogId)
?? throw new InvalidDataException("The managed local AI model is no longer qualified.");

LocalInferenceRunProfile profile = ResolveQualifiedReceipt(manifest, runtime, model);
string? draftModelPath = ResolveDraftModelPath(manifest, model, verifiedDraftModelPath);

string presetPath = paths.ResolveContainedPath(
Path.GetRelativePath(paths.RootDirectory, paths.RouterPresetPath),
Expand All @@ -84,7 +95,7 @@ private static LlamaServerRouterLaunchPlan BuildCore(
.WithComparers(StringComparer.OrdinalIgnoreCase)
.Add("CUDA_VISIBLE_DEVICES", manifest.SelectedGpuId),
presetPath,
BuildPreset(model, profile, modelPath),
BuildPreset(model, profile, modelPath, draftModelPath),
model.Id);
}

Expand Down Expand Up @@ -120,7 +131,7 @@ internal static void ValidateArtifactReceipts(
{
throw new InvalidDataException("The managed local AI architecture and runtime receipt do not match.");
}
if (!string.Equals(manifest.EngineVersion, LlamaRuntimeCatalog.ReleaseTag, StringComparison.Ordinal) ||
if (!string.Equals(manifest.EngineVersion, runtime.ReleaseTag, StringComparison.Ordinal) ||
!string.Equals(manifest.ModelAlias, model.Id, StringComparison.Ordinal))
{
throw new InvalidDataException("The managed local AI model recipe receipt does not match the qualified catalog.");
Expand All @@ -145,17 +156,65 @@ internal static void ValidateArtifactReceipts(
{
throw new InvalidDataException("The managed model artifact receipt does not match the qualified catalog.");
}

ImmutableArray<PinnedArtifact> expectedAdditionalArtifacts = LocalModelCatalog.AdditionalArtifacts(model);
if (manifest.AdditionalModelAssetsOrEmpty.Length != expectedAdditionalArtifacts.Length ||
manifest.AdditionalModelPathsOrEmpty.Length != expectedAdditionalArtifacts.Length)
{
throw new InvalidDataException(
"The managed additional model asset receipts do not match the qualified catalog.");
}
for (int i = 0; i < expectedAdditionalArtifacts.Length; i++)
{
PinnedArtifact artifact = expectedAdditionalArtifacts[i];
LocalAiAssetReceipt receipt = manifest.AdditionalModelAssetsOrEmpty[i];
if (!string.Equals(receipt.FileName, Path.GetFileName(artifact.RelativePath), StringComparison.Ordinal) ||
receipt.SizeBytes != artifact.SizeBytes ||
!string.Equals(receipt.Sha256, artifact.Sha256.Value, StringComparison.Ordinal) ||
!string.Equals(receipt.SourceUrl, artifact.DownloadUri.AbsoluteUri, StringComparison.Ordinal))
{
throw new InvalidDataException(
"The managed additional model asset receipts do not match the qualified catalog.");
}
}
}

/// <summary>
/// The DFlash draft checkpoint's path for the preset. Prefers the handle-resolved
/// physical path supplied by the caller that verified and still holds the file, so a
/// snapshot-link replacement cannot change the identity llama-server opens. Falls
/// back to the persisted receipt path only for callers that do not verify first
/// (<see cref="Build"/>, used for inspection rather than launch). Null for recipes
/// with no separate draft checkpoint. Callers must validate the manifest via
/// <see cref="ValidateArtifactReceipts"/> first, which guarantees
/// <c>AdditionalModelPaths</c> has one entry per catalog-pinned artifact.
/// </summary>
private static string? ResolveDraftModelPath(
LocalAiInstallManifest manifest,
LocalModelInfo model,
string? verifiedDraftModelPath)
{
if (model.Recipe.DraftWeights is null)
return null;
return string.IsNullOrWhiteSpace(verifiedDraftModelPath)
? manifest.AdditionalModelPathsOrEmpty[^1]
: verifiedDraftModelPath;
}

private static string BuildPreset(
LocalModelInfo model,
LocalInferenceRunProfile profile,
string modelPath)
string modelPath,
string? draftModelPath)
{
if (modelPath.IndexOfAny(['\r', '\n']) >= 0)
throw new InvalidDataException("The managed model path cannot be represented safely in a llama-server preset.");
if (draftModelPath is not null && draftModelPath.IndexOfAny(['\r', '\n']) >= 0)
throw new InvalidDataException("The managed draft model path cannot be represented safely in a llama-server preset.");

LocalModelRunRecipe recipe = model.Recipe;
if (recipe.SpeculativeDecoding == SpeculativeDecodingMode.DraftDFlash && draftModelPath is null)
throw new InvalidDataException("Draft-flash decoding requires a resolved draft model path.");
ModelSamplingPreset sampling = recipe.Sampling;
var preset = new StringBuilder();
preset.AppendLine("version = 1");
Expand All @@ -178,9 +237,24 @@ private static string BuildPreset(
preset.AppendLine("main-gpu = 0");
preset.AppendLine("fit = off");
preset.AppendLine("load-mode = dio");
preset.AppendLine("spec-type = draft-mtp");
preset.Append("spec-draft-n-max = ").AppendLine(Invariant(recipe.SpeculativeDraftMaxTokens));
preset.AppendLine("spec-draft-backend-sampling = true");
switch (recipe.SpeculativeDecoding)
{
case SpeculativeDecodingMode.DraftMtp:
preset.AppendLine("spec-type = draft-mtp");
preset.Append("spec-draft-n-max = ").AppendLine(Invariant(recipe.SpeculativeDraftMaxTokens));
preset.AppendLine("spec-draft-backend-sampling = true");
break;
case SpeculativeDecodingMode.DraftDFlash:
preset.AppendLine("spec-type = draft-dflash");
preset.Append("spec-draft-model = ").AppendLine(draftModelPath);
preset.Append("spec-draft-n-max = ").AppendLine(Invariant(recipe.SpeculativeDraftMaxTokens));
preset.AppendLine("spec-draft-backend-sampling = true");
break;
case SpeculativeDecodingMode.None:
break;
default:
throw new ArgumentOutOfRangeException(nameof(recipe.SpeculativeDecoding));
}
preset.Append("temperature = ").AppendLine(Invariant(sampling.Temperature));
preset.Append("top-k = ").AppendLine(Invariant(sampling.TopK));
preset.Append("top-p = ").AppendLine(Invariant(sampling.TopP));
Expand Down
47 changes: 44 additions & 3 deletions src/OpenClaw.Connection/LocalAi/LlamaServerRuntimeService.cs
Original file line number Diff line number Diff line change
Expand Up @@ -118,6 +118,7 @@ public sealed class LlamaServerRuntimeService : ILocalAiRuntime
private LocalAiRuntimeSnapshot _snapshot;
private ILocalAiManagedProcess? _managedProcess;
private LocalAiVerifiedModelLease? _verifiedModel;
private readonly List<LocalAiVerifiedModelLease> _verifiedAdditionalAssets = [];
private string? _runtimeModelPath;
private LocalAiResolvedInstall? _install;
private long _generation;
Expand Down Expand Up @@ -378,6 +379,7 @@ private async Task<LocalAiRuntimeSnapshot> EnsureStartedCoreAsync(CancellationTo
_options.Paths,
install,
GetRuntimeModelPath(install),
GetRuntimeDraftModelPath(),
requestedPort);
await WritePresetAtomicallyAsync(launchPlan, cancellationToken).ConfigureAwait(false);
}
Expand Down Expand Up @@ -862,7 +864,7 @@ private async Task ValidateInstalledFilesAsync(
DisposeVerifiedModelHandle();
ValidateInstalledFilesForStatus(install);

if (install.Manifest.SchemaVersion != LocalAiInstallManifest.HubCacheReceiptSchemaVersion)
if (!install.Manifest.UsesHubCache)
{
_runtimeModelPath = install.ModelPath;
return;
Expand All @@ -883,14 +885,41 @@ await _modelFileVerifier.TryOpenAsync(
"The shared Hugging Face cache model is unsafe or no longer matches its receipt.");
}

// Schema-5 extra assets (a DFlash draft checkpoint) are loaded natively
// by llama-server exactly like the primary weights, and they
// live in the same shared, user-writable hub cache. Rehash them here and hold
// the handles for the process lifetime, so a file swapped after setup cannot
// reach the loader with only a structural path check behind it.
foreach ((LocalAiAssetReceipt receipt, string cachedPath) in
install.Manifest.AdditionalModelAssetsOrEmpty
.Zip(install.Manifest.AdditionalModelPathsOrEmpty))
{
LocalAiVerifiedModelLease? verifiedAsset =
await _modelFileVerifier.TryOpenAsync(
install.Manifest.ModelCacheRoot!,
cachedPath,
receipt.SizeBytes,
new Sha256Digest(receipt.Sha256),
cancellationToken)
.ConfigureAwait(false);
if (verifiedAsset is null)
{
DisposeVerifiedModelHandle();
throw new InvalidDataException(
$"The shared Hugging Face cache asset '{receipt.FileName}' is unsafe or no longer matches its receipt.");
}

_verifiedAdditionalAssets.Add(verifiedAsset);
}

_runtimeModelPath = _verifiedModel.ResolvedPath;
}

private static void ValidateInstalledFilesForStatus(LocalAiResolvedInstall install)
{
if (!File.Exists(install.ExecutablePath))
throw new InvalidDataException("The managed llama-server executable is missing.");
if (install.Manifest.SchemaVersion == LocalAiInstallManifest.HubCacheReceiptSchemaVersion)
if (install.Manifest.UsesHubCache)
{
if (!File.Exists(install.ModelPath))
throw new InvalidDataException("The managed GGUF model is missing.");
Expand Down Expand Up @@ -1666,14 +1695,26 @@ private void DisposeVerifiedModelHandle()
{
_verifiedModel?.Dispose();
_verifiedModel = null;
foreach (LocalAiVerifiedModelLease lease in _verifiedAdditionalAssets)
lease.Dispose();
_verifiedAdditionalAssets.Clear();
_runtimeModelPath = null;
}

/// <summary>
/// The handle-resolved path of the draft checkpoint this process verified and still
/// holds open, or null when the recipe has no additional assets. Additional assets are
/// verified in catalog order and the draft checkpoint is always last, matching
/// <see cref="LocalModelCatalog.AdditionalArtifacts"/>.
/// </summary>
private string? GetRuntimeDraftModelPath() =>
_verifiedAdditionalAssets.Count == 0 ? null : _verifiedAdditionalAssets[^1].ResolvedPath;

private string GetRuntimeModelPath(LocalAiResolvedInstall install)
{
if (_runtimeModelPath is not null)
return _runtimeModelPath;
if (install.Manifest.SchemaVersion == LocalAiInstallManifest.HubCacheReceiptSchemaVersion)
if (install.Manifest.UsesHubCache)
throw new InvalidOperationException("The verified shared-cache model identity is unavailable.");
return install.ModelPath;
}
Expand Down
Loading
Loading