mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-09-17 20:31:47 +02:00
* rpc : hash-cache only weights ggml_backend_rpc_buffer_set_tensor and ggml_backend_rpc_set_tensor_async hashed every transfer above HASH_THRESHOLD and let `rpc-server -c` serve it from its file cache. The cache is meant for weights, but the activations ggml_backend_sched copies between backends took the same path: with a two-node split of Qwen3.8-Flash-Next every prefill ubatch above 10 MB was hashed, written to the worker's cache directory (1.4 TB after a day) and later served from there. Use the hash path only for tensors in buffers marked GGML_BACKEND_BUFFER_USAGE_WEIGHTS. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> * rpc : save a cache entry only for the tensor that missed the hash check With the client hashing weights only, the server still wrote every SET_TENSOR above HASH_THRESHOLD to the cache directory, so the compute data the scheduler sends kept filling the disk. Remember the hash of the last SET_TENSOR_HASH that missed and save only the SET_TENSOR that follows it with that hash - the weight the client is re-sending. * rpc : signal the cache decision in the SET_TENSOR payload Replace the server-side `pending_cache` state with a `cache_flag` byte in the SET_TENSOR message: the client sets it when SET_TENSOR_HASH reported a miss, the server saves a cache entry only when it is set. Bump RPC_PROTO_MAJOR_VERSION since the wire format changes. --------- Co-authored-by: Patrick Hoffmann <patrickhoffmann@MacBook-Pro-14-HOP.local> Co-authored-by: Claude Opus 5 <noreply@anthropic.com>
36 lines
1.2 KiB
C
36 lines
1.2 KiB
C
#pragma once
|
|
|
|
#include "ggml-backend.h"
|
|
|
|
#ifdef __cplusplus
|
|
extern "C" {
|
|
#endif
|
|
|
|
#define RPC_PROTO_MAJOR_VERSION 7
|
|
#define RPC_PROTO_MINOR_VERSION 0
|
|
#define RPC_PROTO_PATCH_VERSION 0
|
|
|
|
#ifdef __cplusplus
|
|
static_assert(GGML_OP_COUNT == 101, "GGML_OP_COUNT has changed - update RPC_PROTO_PATCH_VERSION");
|
|
#endif
|
|
|
|
#define GGML_RPC_MAX_SERVERS 16
|
|
|
|
// backend API
|
|
GGML_BACKEND_API ggml_backend_t ggml_backend_rpc_init(const char * endpoint, uint32_t device);
|
|
GGML_BACKEND_API bool ggml_backend_is_rpc(ggml_backend_t backend);
|
|
|
|
GGML_BACKEND_API ggml_backend_buffer_type_t ggml_backend_rpc_buffer_type(const char * endpoint, uint32_t device);
|
|
|
|
GGML_BACKEND_API void ggml_backend_rpc_get_device_memory(const char * endpoint, uint32_t device, size_t * free, size_t * total);
|
|
|
|
GGML_BACKEND_API void ggml_backend_rpc_start_server(const char * endpoint, const char * cache_dir,
|
|
size_t n_threads, size_t n_devices, ggml_backend_dev_t * devices);
|
|
|
|
GGML_BACKEND_API ggml_backend_reg_t ggml_backend_rpc_reg(void);
|
|
GGML_BACKEND_API ggml_backend_reg_t ggml_backend_rpc_add_server(const char * endpoint);
|
|
|
|
#ifdef __cplusplus
|
|
}
|
|
#endif
|