mirror of
https://github.com/qdrant/qdrant.git
synced 2026-09-29 01:17:56 -05:00
* Configurable key prefix for object storage snapshots Snapshot object keys were always derived from the local snapshots path, so every deployment sharing a bucket wrote under the same `snapshots/` root. Each cloud config block gains an optional `prefix`, and objects become `<prefix>/snapshots/...`. The prefix is applied by wrapping the client in `PrefixStore`, so the snapshot operations and the names returned by the API are unchanged. Leading, trailing and repeated slashes are dropped, and an empty prefix is a no-op. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> * Test that snapshot operations cannot escape the storage prefix Hostile targets are handed to the cloud manager directly, past `validate_snapshot_name`: parent references, absolute paths, encoded slashes, backslashes and empty paths. Writes must stay under the prefix, and objects planted outside the prefix must be invisible to list, download, stream and delete. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> * Normalize empty prefix components Refactor prefix handling to remove empty components. Co-authored-by: Copilot Autofix powered by AI <175728472+Copilot@users.noreply.github.com> --------- Co-authored-by: Claude Fable 5.1 <noreply@anthropic.com> Co-authored-by: Tim Visée <tim+github@visee.me> Co-authored-by: Copilot Autofix powered by AI <175728472+Copilot@users.noreply.github.com>
527 lines
20 KiB
YAML
527 lines
20 KiB
YAML
log_level: INFO
|
|
|
|
# Logging configuration
|
|
# Qdrant logs to stdout. You may configure to also write logs to a file on disk.
|
|
# Be aware that this file may grow indefinitely.
|
|
# logger:
|
|
# # Logging format, supports `text` and `json`
|
|
# format: text
|
|
# on_disk:
|
|
# enabled: true
|
|
# log_file: path/to/log/file.log
|
|
# log_level: INFO
|
|
# # Logging format, supports `text` and `json`
|
|
# format: text
|
|
# buffer_size_bytes: 1024
|
|
|
|
storage:
|
|
# Where to store all the data
|
|
storage_path: ./storage
|
|
|
|
# Where to store snapshots
|
|
snapshots_path: ./snapshots
|
|
|
|
snapshots_config:
|
|
# Where to store snapshots: "local", "s3", "gcs" or "azure"
|
|
snapshots_storage: local
|
|
|
|
# Amazon S3 or any S3-compatible storage. Fields left unset are resolved
|
|
# from the AWS_* environment and the default credential chain.
|
|
# s3_config:
|
|
# bucket: ""
|
|
# # Optional key prefix, objects become <prefix>/snapshots/...
|
|
# prefix: ""
|
|
# region: ""
|
|
# access_key: ""
|
|
# secret_key: ""
|
|
# # Custom endpoint, for MinIO and similar
|
|
# endpoint_url: ""
|
|
|
|
# Google Cloud Storage. Without credential fields, Application Default
|
|
# Credentials are used: GOOGLE_* environment, gcloud config or the
|
|
# metadata server on GCE and GKE.
|
|
# gcs_config:
|
|
# bucket: ""
|
|
# # Optional object name prefix, objects become <prefix>/snapshots/...
|
|
# prefix: ""
|
|
# # Use at most one of the following
|
|
# service_account_path: ""
|
|
# service_account_key: ""
|
|
# application_credentials_path: ""
|
|
# # Custom endpoint, for fake-gcs-server and similar
|
|
# endpoint_url: ""
|
|
|
|
# Azure Blob Storage. Without credential fields, the default chain is
|
|
# used: AZURE_* environment, managed identity or the Azure CLI.
|
|
# azure_config:
|
|
# account: ""
|
|
# container: ""
|
|
# # Optional blob name prefix, blobs become <prefix>/snapshots/...
|
|
# prefix: ""
|
|
# # Use one authentication method
|
|
# access_key: ""
|
|
# sas_token: ""
|
|
# client_id: ""
|
|
# client_secret: ""
|
|
# tenant_id: ""
|
|
# # Custom endpoint, for Azurite and similar
|
|
# endpoint_url: ""
|
|
|
|
# Where to store temporary files
|
|
# If null, temporary snapshots are stored in: storage/snapshots_temp/
|
|
temp_path: null
|
|
|
|
# Deprecated: use `payload.memory` instead.
|
|
# If true - point payloads will not be stored in memory.
|
|
# It will be read from the disk every time it is requested.
|
|
# This setting saves RAM by (slightly) increasing the response time.
|
|
# Note: those payload values that are involved in filtering and are indexed - remain in RAM.
|
|
#
|
|
# Default: true
|
|
on_disk_payload: true
|
|
|
|
# Default payload storage configuration for newly created collections.
|
|
# Overrides the deprecated `on_disk_payload` flag if both are set.
|
|
# payload:
|
|
# # Memory placement of the payload storage: cold or cached.
|
|
# memory: cold
|
|
|
|
# Load-time memory mode. Only affects how segments are loaded on startup;
|
|
# does not modify any persisted configuration. Intended as a recovery knob
|
|
# when a node crash-loops on out-of-memory.
|
|
#
|
|
# Options:
|
|
# - disabled (default): load segments as persisted.
|
|
# - no_resident: downgrade components to their on-disk variants where
|
|
# possible — quantization loads as if always_ram=false, payload field
|
|
# indexes as if on_disk=true, payload storage as mmap (not populated).
|
|
# - no_populate: same as no_resident, plus skip mmap prefault on load for
|
|
# vectors, HNSW graph and payload storage.
|
|
#low_memory_mode: disabled
|
|
|
|
# Maximum number of concurrent updates to shard replicas
|
|
# If `null` - maximum concurrency is used.
|
|
update_concurrency: null
|
|
|
|
# Write-ahead-log related configuration
|
|
wal:
|
|
# Size of a single WAL segment
|
|
wal_capacity_mb: 32
|
|
|
|
# Number of WAL segments to create ahead of actual data requirement
|
|
wal_segments_ahead: 0
|
|
|
|
# Number of closed WAL segments to keep
|
|
wal_retain_closed: 1
|
|
|
|
# Normal node - receives all updates and answers all queries
|
|
node_type: "Normal"
|
|
|
|
# Listener node - receives all updates, but does not answer search/read queries
|
|
# Useful for setting up a dedicated backup node
|
|
# node_type: "Listener"
|
|
|
|
performance:
|
|
# Number of parallel threads used for search operations. If 0 - auto selection.
|
|
max_search_threads: 0
|
|
|
|
# CPU budget, how many CPUs (threads) to allocate for an optimization job.
|
|
# If 0 - auto selection, keep 1 or more CPUs unallocated depending on CPU size
|
|
# If negative - subtract this number of CPUs from the available CPUs.
|
|
# If positive - use this exact number of CPUs.
|
|
optimizer_cpu_budget: 0
|
|
|
|
# Prevent DDoS of too many concurrent updates in distributed mode.
|
|
# One external update usually triggers multiple internal updates, which breaks internal
|
|
# timings. For example, the health check timing and consensus timing.
|
|
# If null - auto selection.
|
|
update_rate_limit: null
|
|
|
|
# Limit for number of incoming automatic shard transfers per collection on this node, does not affect user-requested transfers.
|
|
# The same value should be used on all nodes in a cluster.
|
|
# Default is to allow 1 transfer.
|
|
# If null - allow unlimited transfers.
|
|
#incoming_shard_transfers_limit: 1
|
|
|
|
# Limit for number of outgoing automatic shard transfers per collection on this node, does not affect user-requested transfers.
|
|
# The same value should be used on all nodes in a cluster.
|
|
# Default is to allow 1 transfer.
|
|
# If null - allow unlimited transfers.
|
|
#outgoing_shard_transfers_limit: 1
|
|
|
|
# Enable async scorer which uses io_uring when rescoring.
|
|
# Only supported on Linux, must be enabled in your kernel.
|
|
# See: <https://qdrant.tech/articles/io_uring/#and-what-about-qdrant>
|
|
#async_scorer: false
|
|
|
|
# Whether components readable through either a memory mapping or io_uring should use
|
|
# io_uring. Only has an effect on Linux.
|
|
#
|
|
# - unset (default): the immutable vector storages follow `async_scorer`, nothing else
|
|
# uses io_uring.
|
|
# - "disabled": no component uses io_uring.
|
|
# - "auto": use io_uring for components with a `cold` memory placement, where reads hit
|
|
# the disk. Components meant to sit in RAM keep using mmap, which is faster there.
|
|
#io_uring: disabled
|
|
|
|
# Maximum number of collections to load concurrently.
|
|
#max_concurrent_collection_loads: 1
|
|
# Maximum number of local shards to load concurrently when loading a collection.
|
|
#max_concurrent_shard_loads: 1
|
|
# Maximum number of segments to load concurrently when loading a local shard.
|
|
#max_concurrent_segment_loads: 8
|
|
|
|
optimizers:
|
|
# The minimal fraction of deleted vectors in a segment, required to perform segment optimization
|
|
deleted_threshold: 0.2
|
|
|
|
# The minimal number of vectors in a segment, required to perform segment optimization
|
|
vacuum_min_vector_number: 1000
|
|
|
|
# Target amount of segments optimizer will try to keep.
|
|
# Real amount of segments may vary depending on multiple parameters:
|
|
# - Amount of stored points
|
|
# - Current write RPS
|
|
#
|
|
# It is recommended to select default number of segments as a factor of the number of search threads,
|
|
# so that each segment would be handled evenly by one of the threads.
|
|
# If `default_segment_number = 0`, will be automatically selected by the number of available CPUs
|
|
default_segment_number: 0
|
|
|
|
# Do not create segments larger than this size (in KiloBytes).
|
|
# Large segments might require disproportionately long indexation times,
|
|
# therefore it makes sense to limit the size of segments.
|
|
#
|
|
# If indexation speed have more priority for your - make this parameter lower.
|
|
# If search speed is more important - make this parameter higher.
|
|
# Note: 1Kb = 1 vector of size 256
|
|
# If not set, will be automatically selected considering the number of available CPUs.
|
|
max_segment_size_kb: null
|
|
|
|
# Maximum size (in KiloBytes) of vectors allowed for plain index.
|
|
# Default value based on experiments and observations.
|
|
# Note: 1Kb = 1 vector of size 256
|
|
# To explicitly disable vector indexing, set to `0`.
|
|
# If not set, the default value will be used.
|
|
indexing_threshold_kb: 10000
|
|
|
|
# Interval between forced flushes.
|
|
flush_interval_sec: 5
|
|
|
|
# Max number of threads (jobs) for running optimizations per shard.
|
|
# Note: each optimization job will also use `max_indexing_threads` threads by itself for index building.
|
|
# If null - have no limit and choose dynamically to saturate CPU.
|
|
# If 0 - no optimization threads, optimizations will be disabled.
|
|
max_optimization_threads: null
|
|
|
|
# This section has the same options as 'optimizers' above. All values specified here will overwrite the collections
|
|
# optimizers configs regardless of the config above and the options specified at collection creation.
|
|
#optimizers_overwrite:
|
|
# deleted_threshold: 0.2
|
|
# vacuum_min_vector_number: 1000
|
|
# default_segment_number: 0
|
|
# max_segment_size_kb: null
|
|
# indexing_threshold_kb: 10000
|
|
# flush_interval_sec: 5
|
|
# max_optimization_threads: null
|
|
|
|
# Default parameters of HNSW Index. Could be overridden for each collection or named vector individually
|
|
hnsw_index:
|
|
# Number of edges per node in the index graph. Larger the value - more accurate the search, more space required.
|
|
m: 16
|
|
|
|
# Number of neighbours to consider during the index building. Larger the value - more accurate the search, more time required to build index.
|
|
ef_construct: 100
|
|
|
|
# Minimal size threshold (in KiloBytes) below which full-scan is preferred over HNSW search.
|
|
# This measures the total size of vectors being queried against.
|
|
# When the maximum estimated amount of points that a condition satisfies is smaller than
|
|
# `full_scan_threshold_kb`, the query planner will use full-scan search instead of HNSW index
|
|
# traversal for better performance.
|
|
# Note: 1Kb = 1 vector of size 256
|
|
full_scan_threshold_kb: 10000
|
|
|
|
# Number of parallel threads used for background index building.
|
|
# If 0 - automatically select.
|
|
# Best to keep between 8 and 16 to prevent likelihood of building broken/inefficient HNSW graphs.
|
|
# On small CPUs, less threads are used.
|
|
max_indexing_threads: 0
|
|
|
|
# Deprecated: use `memory` instead.
|
|
# Store HNSW index on disk. If set to false, index will be stored in RAM. Default: false
|
|
on_disk: false
|
|
|
|
# Memory placement of the HNSW index: cold, cached or pinned.
|
|
# Overrides the deprecated `on_disk` flag if both are set.
|
|
# memory: cached
|
|
|
|
# Custom M param for hnsw graph built for payload index. If not set, default M will be used.
|
|
payload_m: null
|
|
|
|
# Default shard transfer method to use if none is defined.
|
|
# If null - don't have a shard transfer preference, choose automatically.
|
|
# If stream_records, snapshot or wal_delta - prefer this specific method.
|
|
# More info: https://qdrant.tech/documentation/guides/distributed_deployment/#shard-transfer-method
|
|
shard_transfer_method: null
|
|
|
|
# Default parameters for collections
|
|
collection:
|
|
# Number of replicas of each shard that network tries to maintain
|
|
replication_factor: 1
|
|
|
|
# How many replicas should apply the operation for us to consider it successful
|
|
write_consistency_factor: 1
|
|
|
|
# Default parameters for vectors.
|
|
vectors:
|
|
# Deprecated: use `memory` instead.
|
|
# Whether vectors should be stored in memory or on disk.
|
|
on_disk: null
|
|
|
|
# Memory placement of the vector storage: cold or cached.
|
|
# Overrides the deprecated `on_disk` flag if both are set.
|
|
# memory: null
|
|
|
|
# shard_number_per_node: 1
|
|
|
|
# Default quantization configuration.
|
|
# More info: https://qdrant.tech/documentation/guides/quantization
|
|
quantization: null
|
|
|
|
# Default strict mode parameters for newly created collections.
|
|
#strict_mode:
|
|
# Whether strict mode is enabled for a collection or not.
|
|
#enabled: false
|
|
|
|
# Max allowed `limit` parameter for all APIs that don't have their own max limit.
|
|
#max_query_limit: null
|
|
|
|
# Max allowed `timeout` parameter.
|
|
#max_timeout: null
|
|
|
|
# Allow usage of unindexed fields in retrieval based (eg. search) filters.
|
|
#unindexed_filtering_retrieve: null
|
|
|
|
# Allow usage of unindexed fields in filtered updates (eg. delete by payload).
|
|
#unindexed_filtering_update: null
|
|
|
|
# Max HNSW value allowed in search parameters.
|
|
#search_max_hnsw_ef: null
|
|
|
|
# Whether exact search is allowed or not.
|
|
#search_allow_exact: null
|
|
|
|
# Max oversampling value allowed in search.
|
|
#search_max_oversampling: null
|
|
|
|
# Maximum number of collections allowed to be created
|
|
# If null - no limit.
|
|
max_collections: null
|
|
|
|
# Cluster-wide resource quotas.
|
|
#
|
|
# Memory and disk are node-wide resources, so their limits are configured once for the whole
|
|
# cluster instead of per collection. Quotas reject updates that would consume more of a resource,
|
|
# regardless of whether strict mode is enabled for the collection being written to. The deprecated
|
|
# `max_resident_memory_percent` in a collection's strict mode config can tighten the memory limit
|
|
# for that collection, but never lift it.
|
|
#
|
|
# A limit that is reached has to be cleared by `release_margin_percent` before the node accepts
|
|
# writes again. Without that margin a resource resting on its limit would put the node in and out
|
|
# of service on the noise between two readings, restarting a shard recovery every time.
|
|
#
|
|
# These values only seed the quota manager on the first start. Afterwards the quota config is read
|
|
# from (and written to) `quota.json` in the storage directory, which is kept in sync across the
|
|
# cluster through consensus and can be changed with the `PUT /quotas` API.
|
|
#
|
|
# If this section is absent - no quota is enforced.
|
|
#quotas:
|
|
# Whether the limits below are enforced.
|
|
#enabled: false
|
|
|
|
# Reject memory-consuming updates once process resident memory reaches this percentage of total
|
|
# system memory (or of the cgroup limit, if one applies).
|
|
# If null - resident memory is not capped.
|
|
#max_resident_memory_percent: null
|
|
|
|
# Reject disk-consuming updates once the filesystem hosting the storage directory is filled to
|
|
# this percentage of its capacity.
|
|
# If null - disk usage is not capped.
|
|
#max_disk_usage_percent: null
|
|
|
|
# How many percentage points below its limit a resource has to fall before this node starts
|
|
# accepting work again. Raise it where usage is volatile; 0 releases as soon as usage is back
|
|
# under the limit.
|
|
# If null - the built-in default of 5 applies.
|
|
#release_margin_percent: null
|
|
|
|
service:
|
|
# Maximum size of POST data in a single request in megabytes
|
|
max_request_size_mb: 32
|
|
|
|
# Number of parallel workers used for serving the api.
|
|
# If 0 - available cores minus one, with a minimum of 1.
|
|
# If missing - Same as storage.max_search_threads
|
|
max_workers: 0
|
|
|
|
# Host to bind the service on
|
|
host: 0.0.0.0
|
|
|
|
# HTTP(S) port to bind the service on
|
|
http_port: 6333
|
|
|
|
# Keep-alive timeout for incoming HTTP connections in seconds.
|
|
# Default: 5
|
|
# http_keep_alive_timeout_sec: 5
|
|
|
|
# Timeout for reading HTTP request data from clients in seconds.
|
|
# Default: 5
|
|
# http_client_request_timeout_sec: 5
|
|
|
|
# Timeout for client disconnect handling in seconds.
|
|
# Default: 5
|
|
# http_client_disconnect_timeout_sec: 5
|
|
|
|
# gRPC port to bind the service on.
|
|
# If `null` - gRPC is disabled. Default: 6334
|
|
# Commenting this key in an overlay does not disable gRPC; the base config
|
|
# still supplies 6334. Set `null` to disable.
|
|
grpc_port: 6334
|
|
|
|
# Enable CORS headers in REST API.
|
|
# If enabled, browsers would be allowed to query REST endpoints regardless of query origin.
|
|
# More info: https://developer.mozilla.org/en-US/docs/Web/HTTP/CORS
|
|
# Default: true
|
|
enable_cors: true
|
|
|
|
# Enable HTTPS for the REST and gRPC API
|
|
enable_tls: false
|
|
|
|
# Check user HTTPS client certificate against CA file specified in tls config
|
|
verify_https_client_certificate: false
|
|
|
|
# Set an api-key.
|
|
# If set, all requests must include a header with the api-key.
|
|
# example header: `api-key: <API-KEY>`
|
|
#
|
|
# If you enable this you should also enable TLS.
|
|
# (Either above or via an external service like nginx.)
|
|
# Sending an api-key over an unencrypted channel is insecure.
|
|
#
|
|
# Uncomment to enable.
|
|
# api_key: your_secret_api_key_here
|
|
|
|
# Set an api-key for read-only operations.
|
|
# If set, all requests must include a header with the api-key.
|
|
# example header: `api-key: <API-KEY>`
|
|
#
|
|
# If you enable this you should also enable TLS.
|
|
# (Either above or via an external service like nginx.)
|
|
# Sending an api-key over an unencrypted channel is insecure.
|
|
#
|
|
# Uncomment to enable.
|
|
# read_only_api_key: your_secret_read_only_api_key_here
|
|
|
|
# Uncomment to enable JWT Role Based Access Control (RBAC).
|
|
# If enabled, you can generate JWT tokens with fine-grained rules for access control.
|
|
# Use generated token instead of API key.
|
|
#
|
|
# jwt_rbac: true
|
|
|
|
# Enforce API key / JWT authentication on the internal (p2p) gRPC API.
|
|
# The regular api_key is always forwarded on internal requests, but the
|
|
# receiving side only verifies it when this flag is enabled. Keep disabled
|
|
# during a rolling upgrade so older peers (that do not attach a key yet)
|
|
# can still reach newer peers, then turn it on once every node is upgraded.
|
|
#
|
|
# enforce_internal_auth: true
|
|
|
|
# Hardware reporting adds information to the API responses with a
|
|
# hint on how many resources were used to execute the request.
|
|
#
|
|
# Warning: experimental, this feature is still under development and is not supported yet.
|
|
#
|
|
# Uncomment to enable.
|
|
# hardware_reporting: true
|
|
#
|
|
# Uncomment to enable.
|
|
# Prefix for the names of metrics in the /metrics API.
|
|
# metrics_prefix: qdrant_
|
|
|
|
# Allow snapshot recovery from remote HTTP/HTTPS URLs.
|
|
# If disabled, snapshot recovery will only work with local files and uploads.
|
|
# Disabling this can mitigate SSRF risks in environments where the Qdrant node
|
|
# has access to internal resources that should not be reachable by users.
|
|
# Default: true
|
|
# enable_snapshot_url_recovery: true
|
|
|
|
cluster:
|
|
# Use `enabled: true` to run Qdrant in distributed deployment mode
|
|
enabled: false
|
|
|
|
# Configuration of the inter-cluster communication
|
|
p2p:
|
|
# Port for internal communication between peers
|
|
port: 6335
|
|
|
|
# Use TLS for communication between peers
|
|
enable_tls: false
|
|
|
|
# Configuration related to distributed consensus algorithm
|
|
consensus:
|
|
# How frequently peers should ping each other.
|
|
# Setting this parameter to lower value will allow consensus
|
|
# to detect disconnected nodes earlier, but too frequent
|
|
# tick period may create significant network and CPU overhead.
|
|
# We encourage you NOT to change this parameter unless you know what you are doing.
|
|
tick_period_ms: 100
|
|
|
|
# Compact consensus operations once we have this amount of applied
|
|
# operations. Allows peers to join quickly with a consensus snapshot without
|
|
# replaying a huge amount of operations.
|
|
# If 0 - disable compaction
|
|
compact_wal_entries: 128
|
|
|
|
# Set to true to prevent service from sending usage statistics to the developers.
|
|
# Read more: https://qdrant.tech/documentation/guides/telemetry
|
|
telemetry_disabled: false
|
|
|
|
# TLS configuration.
|
|
# Required if either service.enable_tls or cluster.p2p.enable_tls is true.
|
|
tls:
|
|
# Server certificate chain file
|
|
cert: ./tls/cert.pem
|
|
|
|
# Server private key file
|
|
key: ./tls/key.pem
|
|
|
|
# Certificate authority certificate file.
|
|
# This certificate will be used to validate the certificates
|
|
# presented by other nodes during inter-cluster communication.
|
|
#
|
|
# If verify_https_client_certificate is true, it will verify
|
|
# HTTPS client certificate
|
|
#
|
|
# Required if cluster.p2p.enable_tls is true.
|
|
ca_cert: ./tls/cacert.pem
|
|
|
|
# TTL in seconds to reload certificate from disk, useful for certificate rotations.
|
|
# Only works for HTTPS endpoints. Does not support gRPC (and intra-cluster communication).
|
|
# If `null` - TTL is disabled.
|
|
cert_ttl: 3600
|
|
|
|
# Audit logging configuration.
|
|
# When enabled, Qdrant writes structured JSON audit log entries for every
|
|
# access-checked API request.
|
|
#
|
|
# audit:
|
|
# enabled: false
|
|
# dir: ./storage/audit
|
|
# rotation: daily
|
|
# max_log_files: 7
|
|
# # If true, use X-Forwarded-For header to determine client IP in audit logs.
|
|
# # Only enable this when running behind a trusted reverse proxy or load balancer.
|
|
# # WARNING: Enabling this without a trusted proxy allows clients to spoof their IP.
|
|
# # Default: false
|
|
# trust_forwarded_headers: false
|