Files
qdrant/config/config.yaml
c218fa67c0 Configurable key prefix for object storage snapshots (#10715)
* Configurable key prefix for object storage snapshots

Snapshot object keys were always derived from the local snapshots path,
so every deployment sharing a bucket wrote under the same `snapshots/`
root. Each cloud config block gains an optional `prefix`, and objects
become `<prefix>/snapshots/...`.

The prefix is applied by wrapping the client in `PrefixStore`, so the
snapshot operations and the names returned by the API are unchanged.
Leading, trailing and repeated slashes are dropped, and an empty prefix
is a no-op.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>

* Test that snapshot operations cannot escape the storage prefix

Hostile targets are handed to the cloud manager directly, past
`validate_snapshot_name`: parent references, absolute paths, encoded
slashes, backslashes and empty paths. Writes must stay under the prefix,
and objects planted outside the prefix must be invisible to list,
download, stream and delete.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>

* Normalize empty prefix components

Refactor prefix handling to remove empty components.

Co-authored-by: Copilot Autofix powered by AI <175728472+Copilot@users.noreply.github.com>

---------

Co-authored-by: Claude Fable 5.1 <noreply@anthropic.com>
Co-authored-by: Tim Visée <tim+github@visee.me>
Co-authored-by: Copilot Autofix powered by AI <175728472+Copilot@users.noreply.github.com>
2026-09-22 12:48:26 +02:00

527 lines
20 KiB
YAML

log_level: INFO
# Logging configuration
# Qdrant logs to stdout. You may configure to also write logs to a file on disk.
# Be aware that this file may grow indefinitely.
# logger:
# # Logging format, supports `text` and `json`
# format: text
# on_disk:
# enabled: true
# log_file: path/to/log/file.log
# log_level: INFO
# # Logging format, supports `text` and `json`
# format: text
# buffer_size_bytes: 1024
storage:
# Where to store all the data
storage_path: ./storage
# Where to store snapshots
snapshots_path: ./snapshots
snapshots_config:
# Where to store snapshots: "local", "s3", "gcs" or "azure"
snapshots_storage: local
# Amazon S3 or any S3-compatible storage. Fields left unset are resolved
# from the AWS_* environment and the default credential chain.
# s3_config:
# bucket: ""
# # Optional key prefix, objects become <prefix>/snapshots/...
# prefix: ""
# region: ""
# access_key: ""
# secret_key: ""
# # Custom endpoint, for MinIO and similar
# endpoint_url: ""
# Google Cloud Storage. Without credential fields, Application Default
# Credentials are used: GOOGLE_* environment, gcloud config or the
# metadata server on GCE and GKE.
# gcs_config:
# bucket: ""
# # Optional object name prefix, objects become <prefix>/snapshots/...
# prefix: ""
# # Use at most one of the following
# service_account_path: ""
# service_account_key: ""
# application_credentials_path: ""
# # Custom endpoint, for fake-gcs-server and similar
# endpoint_url: ""
# Azure Blob Storage. Without credential fields, the default chain is
# used: AZURE_* environment, managed identity or the Azure CLI.
# azure_config:
# account: ""
# container: ""
# # Optional blob name prefix, blobs become <prefix>/snapshots/...
# prefix: ""
# # Use one authentication method
# access_key: ""
# sas_token: ""
# client_id: ""
# client_secret: ""
# tenant_id: ""
# # Custom endpoint, for Azurite and similar
# endpoint_url: ""
# Where to store temporary files
# If null, temporary snapshots are stored in: storage/snapshots_temp/
temp_path: null
# Deprecated: use `payload.memory` instead.
# If true - point payloads will not be stored in memory.
# It will be read from the disk every time it is requested.
# This setting saves RAM by (slightly) increasing the response time.
# Note: those payload values that are involved in filtering and are indexed - remain in RAM.
#
# Default: true
on_disk_payload: true
# Default payload storage configuration for newly created collections.
# Overrides the deprecated `on_disk_payload` flag if both are set.
# payload:
# # Memory placement of the payload storage: cold or cached.
# memory: cold
# Load-time memory mode. Only affects how segments are loaded on startup;
# does not modify any persisted configuration. Intended as a recovery knob
# when a node crash-loops on out-of-memory.
#
# Options:
# - disabled (default): load segments as persisted.
# - no_resident: downgrade components to their on-disk variants where
# possible — quantization loads as if always_ram=false, payload field
# indexes as if on_disk=true, payload storage as mmap (not populated).
# - no_populate: same as no_resident, plus skip mmap prefault on load for
# vectors, HNSW graph and payload storage.
#low_memory_mode: disabled
# Maximum number of concurrent updates to shard replicas
# If `null` - maximum concurrency is used.
update_concurrency: null
# Write-ahead-log related configuration
wal:
# Size of a single WAL segment
wal_capacity_mb: 32
# Number of WAL segments to create ahead of actual data requirement
wal_segments_ahead: 0
# Number of closed WAL segments to keep
wal_retain_closed: 1
# Normal node - receives all updates and answers all queries
node_type: "Normal"
# Listener node - receives all updates, but does not answer search/read queries
# Useful for setting up a dedicated backup node
# node_type: "Listener"
performance:
# Number of parallel threads used for search operations. If 0 - auto selection.
max_search_threads: 0
# CPU budget, how many CPUs (threads) to allocate for an optimization job.
# If 0 - auto selection, keep 1 or more CPUs unallocated depending on CPU size
# If negative - subtract this number of CPUs from the available CPUs.
# If positive - use this exact number of CPUs.
optimizer_cpu_budget: 0
# Prevent DDoS of too many concurrent updates in distributed mode.
# One external update usually triggers multiple internal updates, which breaks internal
# timings. For example, the health check timing and consensus timing.
# If null - auto selection.
update_rate_limit: null
# Limit for number of incoming automatic shard transfers per collection on this node, does not affect user-requested transfers.
# The same value should be used on all nodes in a cluster.
# Default is to allow 1 transfer.
# If null - allow unlimited transfers.
#incoming_shard_transfers_limit: 1
# Limit for number of outgoing automatic shard transfers per collection on this node, does not affect user-requested transfers.
# The same value should be used on all nodes in a cluster.
# Default is to allow 1 transfer.
# If null - allow unlimited transfers.
#outgoing_shard_transfers_limit: 1
# Enable async scorer which uses io_uring when rescoring.
# Only supported on Linux, must be enabled in your kernel.
# See: <https://qdrant.tech/articles/io_uring/#and-what-about-qdrant>
#async_scorer: false
# Whether components readable through either a memory mapping or io_uring should use
# io_uring. Only has an effect on Linux.
#
# - unset (default): the immutable vector storages follow `async_scorer`, nothing else
# uses io_uring.
# - "disabled": no component uses io_uring.
# - "auto": use io_uring for components with a `cold` memory placement, where reads hit
# the disk. Components meant to sit in RAM keep using mmap, which is faster there.
#io_uring: disabled
# Maximum number of collections to load concurrently.
#max_concurrent_collection_loads: 1
# Maximum number of local shards to load concurrently when loading a collection.
#max_concurrent_shard_loads: 1
# Maximum number of segments to load concurrently when loading a local shard.
#max_concurrent_segment_loads: 8
optimizers:
# The minimal fraction of deleted vectors in a segment, required to perform segment optimization
deleted_threshold: 0.2
# The minimal number of vectors in a segment, required to perform segment optimization
vacuum_min_vector_number: 1000
# Target amount of segments optimizer will try to keep.
# Real amount of segments may vary depending on multiple parameters:
# - Amount of stored points
# - Current write RPS
#
# It is recommended to select default number of segments as a factor of the number of search threads,
# so that each segment would be handled evenly by one of the threads.
# If `default_segment_number = 0`, will be automatically selected by the number of available CPUs
default_segment_number: 0
# Do not create segments larger than this size (in KiloBytes).
# Large segments might require disproportionately long indexation times,
# therefore it makes sense to limit the size of segments.
#
# If indexation speed have more priority for your - make this parameter lower.
# If search speed is more important - make this parameter higher.
# Note: 1Kb = 1 vector of size 256
# If not set, will be automatically selected considering the number of available CPUs.
max_segment_size_kb: null
# Maximum size (in KiloBytes) of vectors allowed for plain index.
# Default value based on experiments and observations.
# Note: 1Kb = 1 vector of size 256
# To explicitly disable vector indexing, set to `0`.
# If not set, the default value will be used.
indexing_threshold_kb: 10000
# Interval between forced flushes.
flush_interval_sec: 5
# Max number of threads (jobs) for running optimizations per shard.
# Note: each optimization job will also use `max_indexing_threads` threads by itself for index building.
# If null - have no limit and choose dynamically to saturate CPU.
# If 0 - no optimization threads, optimizations will be disabled.
max_optimization_threads: null
# This section has the same options as 'optimizers' above. All values specified here will overwrite the collections
# optimizers configs regardless of the config above and the options specified at collection creation.
#optimizers_overwrite:
# deleted_threshold: 0.2
# vacuum_min_vector_number: 1000
# default_segment_number: 0
# max_segment_size_kb: null
# indexing_threshold_kb: 10000
# flush_interval_sec: 5
# max_optimization_threads: null
# Default parameters of HNSW Index. Could be overridden for each collection or named vector individually
hnsw_index:
# Number of edges per node in the index graph. Larger the value - more accurate the search, more space required.
m: 16
# Number of neighbours to consider during the index building. Larger the value - more accurate the search, more time required to build index.
ef_construct: 100
# Minimal size threshold (in KiloBytes) below which full-scan is preferred over HNSW search.
# This measures the total size of vectors being queried against.
# When the maximum estimated amount of points that a condition satisfies is smaller than
# `full_scan_threshold_kb`, the query planner will use full-scan search instead of HNSW index
# traversal for better performance.
# Note: 1Kb = 1 vector of size 256
full_scan_threshold_kb: 10000
# Number of parallel threads used for background index building.
# If 0 - automatically select.
# Best to keep between 8 and 16 to prevent likelihood of building broken/inefficient HNSW graphs.
# On small CPUs, less threads are used.
max_indexing_threads: 0
# Deprecated: use `memory` instead.
# Store HNSW index on disk. If set to false, index will be stored in RAM. Default: false
on_disk: false
# Memory placement of the HNSW index: cold, cached or pinned.
# Overrides the deprecated `on_disk` flag if both are set.
# memory: cached
# Custom M param for hnsw graph built for payload index. If not set, default M will be used.
payload_m: null
# Default shard transfer method to use if none is defined.
# If null - don't have a shard transfer preference, choose automatically.
# If stream_records, snapshot or wal_delta - prefer this specific method.
# More info: https://qdrant.tech/documentation/guides/distributed_deployment/#shard-transfer-method
shard_transfer_method: null
# Default parameters for collections
collection:
# Number of replicas of each shard that network tries to maintain
replication_factor: 1
# How many replicas should apply the operation for us to consider it successful
write_consistency_factor: 1
# Default parameters for vectors.
vectors:
# Deprecated: use `memory` instead.
# Whether vectors should be stored in memory or on disk.
on_disk: null
# Memory placement of the vector storage: cold or cached.
# Overrides the deprecated `on_disk` flag if both are set.
# memory: null
# shard_number_per_node: 1
# Default quantization configuration.
# More info: https://qdrant.tech/documentation/guides/quantization
quantization: null
# Default strict mode parameters for newly created collections.
#strict_mode:
# Whether strict mode is enabled for a collection or not.
#enabled: false
# Max allowed `limit` parameter for all APIs that don't have their own max limit.
#max_query_limit: null
# Max allowed `timeout` parameter.
#max_timeout: null
# Allow usage of unindexed fields in retrieval based (eg. search) filters.
#unindexed_filtering_retrieve: null
# Allow usage of unindexed fields in filtered updates (eg. delete by payload).
#unindexed_filtering_update: null
# Max HNSW value allowed in search parameters.
#search_max_hnsw_ef: null
# Whether exact search is allowed or not.
#search_allow_exact: null
# Max oversampling value allowed in search.
#search_max_oversampling: null
# Maximum number of collections allowed to be created
# If null - no limit.
max_collections: null
# Cluster-wide resource quotas.
#
# Memory and disk are node-wide resources, so their limits are configured once for the whole
# cluster instead of per collection. Quotas reject updates that would consume more of a resource,
# regardless of whether strict mode is enabled for the collection being written to. The deprecated
# `max_resident_memory_percent` in a collection's strict mode config can tighten the memory limit
# for that collection, but never lift it.
#
# A limit that is reached has to be cleared by `release_margin_percent` before the node accepts
# writes again. Without that margin a resource resting on its limit would put the node in and out
# of service on the noise between two readings, restarting a shard recovery every time.
#
# These values only seed the quota manager on the first start. Afterwards the quota config is read
# from (and written to) `quota.json` in the storage directory, which is kept in sync across the
# cluster through consensus and can be changed with the `PUT /quotas` API.
#
# If this section is absent - no quota is enforced.
#quotas:
# Whether the limits below are enforced.
#enabled: false
# Reject memory-consuming updates once process resident memory reaches this percentage of total
# system memory (or of the cgroup limit, if one applies).
# If null - resident memory is not capped.
#max_resident_memory_percent: null
# Reject disk-consuming updates once the filesystem hosting the storage directory is filled to
# this percentage of its capacity.
# If null - disk usage is not capped.
#max_disk_usage_percent: null
# How many percentage points below its limit a resource has to fall before this node starts
# accepting work again. Raise it where usage is volatile; 0 releases as soon as usage is back
# under the limit.
# If null - the built-in default of 5 applies.
#release_margin_percent: null
service:
# Maximum size of POST data in a single request in megabytes
max_request_size_mb: 32
# Number of parallel workers used for serving the api.
# If 0 - available cores minus one, with a minimum of 1.
# If missing - Same as storage.max_search_threads
max_workers: 0
# Host to bind the service on
host: 0.0.0.0
# HTTP(S) port to bind the service on
http_port: 6333
# Keep-alive timeout for incoming HTTP connections in seconds.
# Default: 5
# http_keep_alive_timeout_sec: 5
# Timeout for reading HTTP request data from clients in seconds.
# Default: 5
# http_client_request_timeout_sec: 5
# Timeout for client disconnect handling in seconds.
# Default: 5
# http_client_disconnect_timeout_sec: 5
# gRPC port to bind the service on.
# If `null` - gRPC is disabled. Default: 6334
# Commenting this key in an overlay does not disable gRPC; the base config
# still supplies 6334. Set `null` to disable.
grpc_port: 6334
# Enable CORS headers in REST API.
# If enabled, browsers would be allowed to query REST endpoints regardless of query origin.
# More info: https://developer.mozilla.org/en-US/docs/Web/HTTP/CORS
# Default: true
enable_cors: true
# Enable HTTPS for the REST and gRPC API
enable_tls: false
# Check user HTTPS client certificate against CA file specified in tls config
verify_https_client_certificate: false
# Set an api-key.
# If set, all requests must include a header with the api-key.
# example header: `api-key: <API-KEY>`
#
# If you enable this you should also enable TLS.
# (Either above or via an external service like nginx.)
# Sending an api-key over an unencrypted channel is insecure.
#
# Uncomment to enable.
# api_key: your_secret_api_key_here
# Set an api-key for read-only operations.
# If set, all requests must include a header with the api-key.
# example header: `api-key: <API-KEY>`
#
# If you enable this you should also enable TLS.
# (Either above or via an external service like nginx.)
# Sending an api-key over an unencrypted channel is insecure.
#
# Uncomment to enable.
# read_only_api_key: your_secret_read_only_api_key_here
# Uncomment to enable JWT Role Based Access Control (RBAC).
# If enabled, you can generate JWT tokens with fine-grained rules for access control.
# Use generated token instead of API key.
#
# jwt_rbac: true
# Enforce API key / JWT authentication on the internal (p2p) gRPC API.
# The regular api_key is always forwarded on internal requests, but the
# receiving side only verifies it when this flag is enabled. Keep disabled
# during a rolling upgrade so older peers (that do not attach a key yet)
# can still reach newer peers, then turn it on once every node is upgraded.
#
# enforce_internal_auth: true
# Hardware reporting adds information to the API responses with a
# hint on how many resources were used to execute the request.
#
# Warning: experimental, this feature is still under development and is not supported yet.
#
# Uncomment to enable.
# hardware_reporting: true
#
# Uncomment to enable.
# Prefix for the names of metrics in the /metrics API.
# metrics_prefix: qdrant_
# Allow snapshot recovery from remote HTTP/HTTPS URLs.
# If disabled, snapshot recovery will only work with local files and uploads.
# Disabling this can mitigate SSRF risks in environments where the Qdrant node
# has access to internal resources that should not be reachable by users.
# Default: true
# enable_snapshot_url_recovery: true
cluster:
# Use `enabled: true` to run Qdrant in distributed deployment mode
enabled: false
# Configuration of the inter-cluster communication
p2p:
# Port for internal communication between peers
port: 6335
# Use TLS for communication between peers
enable_tls: false
# Configuration related to distributed consensus algorithm
consensus:
# How frequently peers should ping each other.
# Setting this parameter to lower value will allow consensus
# to detect disconnected nodes earlier, but too frequent
# tick period may create significant network and CPU overhead.
# We encourage you NOT to change this parameter unless you know what you are doing.
tick_period_ms: 100
# Compact consensus operations once we have this amount of applied
# operations. Allows peers to join quickly with a consensus snapshot without
# replaying a huge amount of operations.
# If 0 - disable compaction
compact_wal_entries: 128
# Set to true to prevent service from sending usage statistics to the developers.
# Read more: https://qdrant.tech/documentation/guides/telemetry
telemetry_disabled: false
# TLS configuration.
# Required if either service.enable_tls or cluster.p2p.enable_tls is true.
tls:
# Server certificate chain file
cert: ./tls/cert.pem
# Server private key file
key: ./tls/key.pem
# Certificate authority certificate file.
# This certificate will be used to validate the certificates
# presented by other nodes during inter-cluster communication.
#
# If verify_https_client_certificate is true, it will verify
# HTTPS client certificate
#
# Required if cluster.p2p.enable_tls is true.
ca_cert: ./tls/cacert.pem
# TTL in seconds to reload certificate from disk, useful for certificate rotations.
# Only works for HTTPS endpoints. Does not support gRPC (and intra-cluster communication).
# If `null` - TTL is disabled.
cert_ttl: 3600
# Audit logging configuration.
# When enabled, Qdrant writes structured JSON audit log entries for every
# access-checked API request.
#
# audit:
# enabled: false
# dir: ./storage/audit
# rotation: daily
# max_log_files: 7
# # If true, use X-Forwarded-For header to determine client IP in audit logs.
# # Only enable this when running behind a trusted reverse proxy or load balancer.
# # WARNING: Enabling this without a trusted proxy allows clients to spoof their IP.
# # Default: false
# trust_forwarded_headers: false