Compare commits

..
5 changed files with 448 additions and 394 deletions
+4 -2
View File
@@ -1,9 +1,10 @@
name: Tests
run-name: Tests (dev)
on:
push:
branches: [ dev ]
branches: [ master, main ]
schedule:
- cron: 0 0 * * *
pull_request:
env:
@@ -11,6 +12,7 @@ env:
jobs:
test:
strategy:
matrix:
python-version:
+1 -1
View File
@@ -2,7 +2,7 @@
FastEmbed is a lightweight, fast, Python library built for embedding generation. We [support popular text models](https://qdrant.github.io/fastembed/examples/Supported_Models/). Please [open a GitHub issue](https://github.com/qdrant/fastembed/issues/new) if you want us to add a new model.
The default text embedding (`TextEmbedding`) model is Flag Embedding, presented in the [MTEB](https://huggingface.co/spaces/mteb/leaderboard) leaderboard. It supports "query" and "passage" prefixes for the input text. Here is an example for [Retrieval Embedding Generation](https://qdrant.github.io/fastembed/qdrant/Retrieval_with_FastEmbed/) and how to use [FastEmbed with Qdrant](https://qdrant.github.io/fastembed/qdrant/Usage_With_Qdrant/).
The default text embedding (`TextEmbedding`) model is Flag Embedding, presented in the [MTEB](https://huggingface.co/spaces/mteb/leaderboard) leaderboard. It supports "query" and "passage" prefixes for the input text. Here is an example for [Retrieval Embedding Generation](https://qdrant.github.io/fastembed/examples/Retrieval_with_FastEmbed/) and how to use [FastEmbed with Qdrant](https://qdrant.github.io/fastembed/examples/Usage_With_Qdrant/).
## 📈 Why FastEmbed?
+24 -2
View File
@@ -1,7 +1,19 @@
import os
from multiprocessing import get_all_start_methods
from pathlib import Path
from typing import Any, Dict, Generic, Iterable, List, Optional, Tuple, Type, TypeVar, Union
from typing import (
Any,
Dict,
Generic,
Iterable,
List,
Optional,
Tuple,
Type,
TypeVar,
Union,
Sequence,
)
import numpy as np
import onnxruntime as ort
@@ -39,11 +51,21 @@ class OnnxModel(Generic[T]):
model_dir: Path,
model_file: str,
threads: Optional[int],
providers: Optional[Sequence[Union[str, Tuple[str, Dict[Any, Any]]]]] = None,
) -> None:
model_path = model_dir / model_file
# List of Execution Providers: https://onnxruntime.ai/docs/execution-providers
onnx_providers = ["CPUExecutionProvider"]
onnx_providers = ["CPUExecutionProvider"] if providers is None else list(providers)
available_providers = ort.get_available_providers()
for provider in onnx_providers:
# check providers available
provider_name = provider if isinstance(provider, str) else provider[0]
if provider_name not in available_providers:
raise ValueError(
f"Provider {provider_name} is not available. Available providers: {available_providers}"
)
so = ort.SessionOptions()
so.graph_optimization_level = ort.GraphOptimizationLevel.ORT_ENABLE_ALL
Generated
+407 -387
View File
File diff suppressed because it is too large Load Diff
+12 -2
View File
@@ -12,8 +12,16 @@ keywords = ["vector", "embedding", "neural", "search", "qdrant", "sentence-trans
[tool.poetry.dependencies]
python = ">=3.8.0,<3.13"
onnx = "^1.15.0"
onnxruntime = "^1.17.0"
onnx = [
{version = "^1.15.0", optional = true, markers = "extra != 'gpu'"}
]
onnxruntime = [
{version = "^1.17.0", optional = true, markers = "extra != 'gpu'"}
]
onnxruntime-gpu = [
{ version = "^1.17.0", optional = true, python = "<3.13", markers = "extra == 'gpu'" }
]
tqdm = "^4.66"
requests = "^2.31"
tokenizers = "^0.15.1"
@@ -37,6 +45,8 @@ pillow = "^10.2.0"
cairosvg = "^2.7.1"
mknotebooks = "^0.8.0"
[tool.poetry.extras]
gpu = ["onnxruntime-gpu"]
[build-system]
requires = ["poetry-core"]