mirror of
https://github.com/qdrant/fastembed.git
synced 2026-10-03 11:27:40 -05:00
docs: Add description for changes
This commit is contained in:
@@ -68,6 +68,9 @@ class OnnxModel(Generic[T]):
|
||||
if device_id is None:
|
||||
onnx_providers = ["CUDAExecutionProvider"]
|
||||
else:
|
||||
# kSameAsRequested: Allocates only the requested memory, avoiding over-allocation.
|
||||
# more precise than 'kNextPowerOfTwo', which grows memory aggressively.
|
||||
# source: https://onnxruntime.ai/docs/get-started/with-c.html#features:~:text=Memory%20arena%20shrinkage:
|
||||
onnx_providers = [
|
||||
(
|
||||
"CUDAExecutionProvider",
|
||||
|
||||
@@ -82,6 +82,9 @@ class OnnxImageModel(OnnxModel[T]):
|
||||
if is_cuda_enabled(cuda, providers):
|
||||
device_id = kwargs.get("device_id", None)
|
||||
device_id = str(device_id if isinstance(device_id, int) else 0)
|
||||
# enables memory arena shrinkage, freeing unused memory after each Run() cycle.
|
||||
# helps prevent excessive memory retention, especially for dynamic workloads.
|
||||
# source: https://onnxruntime.ai/docs/get-started/with-c.html#features:~:text=Memory%20arena%20shrinkage:
|
||||
run_options.add_run_config_entry(
|
||||
"memory.enable_memory_arena_shrinkage", f"gpu:{device_id}"
|
||||
)
|
||||
|
||||
@@ -201,7 +201,6 @@ class Colbert(LateInteractionTextEmbeddingBase, OnnxTextModel[NumpyArray]):
|
||||
current_max_length = self.tokenizer.truncation["max_length"]
|
||||
# ensure not to overflow after adding document-marker
|
||||
self.tokenizer.enable_truncation(max_length=current_max_length - 1)
|
||||
print("ME VERSION")
|
||||
|
||||
def embed(
|
||||
self,
|
||||
|
||||
@@ -111,6 +111,9 @@ class OnnxMultimodalModel(OnnxModel[T]):
|
||||
if is_cuda_enabled(cuda, providers):
|
||||
device_id = kwargs.get("device_id", None)
|
||||
device_id = str(device_id if isinstance(device_id, int) else 0)
|
||||
# enables memory arena shrinkage, freeing unused memory after each Run() cycle.
|
||||
# helps prevent excessive memory retention, especially for dynamic workloads.
|
||||
# source: https://onnxruntime.ai/docs/get-started/with-c.html#features:~:text=Memory%20arena%20shrinkage:
|
||||
run_options.add_run_config_entry(
|
||||
"memory.enable_memory_arena_shrinkage", f"gpu:{device_id}"
|
||||
)
|
||||
@@ -188,6 +191,9 @@ class OnnxMultimodalModel(OnnxModel[T]):
|
||||
if is_cuda_enabled(cuda, providers):
|
||||
device_id = kwargs.get("device_id", None)
|
||||
device_id = str(device_id if isinstance(device_id, int) else 0)
|
||||
# enables memory arena shrinkage, freeing unused memory after each Run() cycle.
|
||||
# helps prevent excessive memory retention, especially for dynamic workloads.
|
||||
# source: https://onnxruntime.ai/docs/get-started/with-c.html#features:~:text=Memory%20arena%20shrinkage:
|
||||
run_options.add_run_config_entry(
|
||||
"memory.enable_memory_arena_shrinkage", f"gpu:{device_id}"
|
||||
)
|
||||
|
||||
@@ -79,6 +79,9 @@ class OnnxCrossEncoderModel(OnnxModel[float]):
|
||||
if is_cuda_enabled(cuda, providers):
|
||||
device_id = kwargs.get("device_id", None)
|
||||
device_id = str(device_id if isinstance(device_id, int) else 0)
|
||||
# Enables memory arena shrinkage, freeing unused memory after each Run() cycle.
|
||||
# Helps prevent excessive memory retention, especially for dynamic workloads.
|
||||
# Source: https://onnxruntime.ai/docs/get-started/with-c.html#features:~:text=Memory%20arena%20shrinkage:
|
||||
run_options.add_run_config_entry(
|
||||
"memory.enable_memory_arena_shrinkage", f"gpu:{device_id}"
|
||||
)
|
||||
|
||||
@@ -89,6 +89,9 @@ class OnnxTextModel(OnnxModel[T]):
|
||||
if is_cuda_enabled(cuda, providers):
|
||||
device_id = kwargs.get("device_id", None)
|
||||
device_id = str(device_id if isinstance(device_id, int) else 0)
|
||||
# enables memory arena shrinkage, freeing unused memory after each Run() cycle.
|
||||
# helps prevent excessive memory retention, especially for dynamic workloads.
|
||||
# source: https://onnxruntime.ai/docs/get-started/with-c.html#features:~:text=Memory%20arena%20shrinkage:
|
||||
run_options.add_run_config_entry(
|
||||
"memory.enable_memory_arena_shrinkage", f"gpu:{device_id}"
|
||||
)
|
||||
|
||||
Reference in New Issue
Block a user