summaryrefslogtreecommitdiffstats
diff options
context:
space:
mode:
-rw-r--r--.github/config/muted_ya.txt4
-rw-r--r--.github/config/muted_ya_asan.txt4
-rw-r--r--.github/workflows/compare_index_performance.yml105
-rw-r--r--.github/workflows/docs_build.yaml36
-rw-r--r--.github/workflows/docs_build_rebuild.yaml29
-rw-r--r--.github/workflows/docs_release.yaml27
-rw-r--r--.github/workflows/ydbdoc-review.yml7
-rw-r--r--.github/workflows/ydbdoc-verify.yml5
-rw-r--r--ydb/core/cms/console/console_tenants_manager.cpp2
-rw-r--r--ydb/core/driver_lib/run/kikimr_services_initializers.cpp2
-rw-r--r--ydb/core/driver_lib/run/ya.make1
-rw-r--r--ydb/core/http_proxy/ut/sqs_topic_ut.cpp14
-rw-r--r--ydb/core/http_proxy/ut/sqs_topic_xml_ut.cpp14
-rw-r--r--ydb/core/http_proxy/ut/ymq_ut.cpp14
-rw-r--r--ydb/core/kqp/common/kqp_tx.cpp3
-rw-r--r--ydb/core/kqp/gateway/behaviour/external_data_source/manager.cpp64
-rw-r--r--ydb/core/kqp/gateway/behaviour/external_data_source/ya.make1
-rw-r--r--ydb/core/kqp/provider/yql_kikimr_exec.cpp62
-rw-r--r--ydb/core/kqp/provider/yql_kikimr_opt_build.cpp25
-rw-r--r--ydb/core/kqp/ut/federated_query/datastreams/streaming_ddl_ut.cpp23
-rw-r--r--ydb/core/kqp/ut/olap/indexes/indexes_ut.cpp33
-rw-r--r--ydb/core/kqp/ut/opt/kqp_concurrent_results_ut.cpp571
-rw-r--r--ydb/core/kqp/ut/opt/ya.make1
-rw-r--r--ydb/core/kqp/ut/scheme/kqp_scheme_ut.cpp37
-rw-r--r--ydb/core/mind/hive/hive.cpp4
-rw-r--r--ydb/core/mind/hive/hive.h2
-rw-r--r--ydb/core/mind/hive/hive_events.h6
-rw-r--r--ydb/core/mind/hive/hive_impl.cpp41
-rw-r--r--ydb/core/mind/hive/hive_impl.h8
-rw-r--r--ydb/core/mind/hive/move_data_actor.cpp (renamed from ydb/core/mind/hive/compact_actor.cpp)35
-rw-r--r--ydb/core/mind/hive/storage_pool_info.cpp17
-rw-r--r--ydb/core/mind/hive/storage_pool_info.h2
-rw-r--r--ydb/core/mind/hive/ya.make2
-rw-r--r--ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_controller.cpp24
-rw-r--r--ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_controller.h19
-rw-r--r--ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_controller_actor.cpp173
-rw-r--r--ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_controller_actor.h72
-rw-r--r--ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_controller_counters.h24
-rw-r--r--ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_controller_database.cpp19
-rw-r--r--ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_controller_database.h26
-rw-r--r--ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_controller_schema.h45
-rw-r--r--ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_controller_tx.h46
-rw-r--r--ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_initschema.cpp47
-rw-r--r--ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_loadstate.cpp49
-rw-r--r--ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/protos/dbs_controller.proto10
-rw-r--r--ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/protos/ya.make14
-rw-r--r--ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/ut/dbs_controller_ut.cpp37
-rw-r--r--ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/ut/ya.make15
-rw-r--r--ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/ya.make29
-rw-r--r--ydb/core/nbs/cloud/blockstore/libs/storage/partition_direct/part_monitoring.cpp21
-rw-r--r--ydb/core/nbs/cloud/blockstore/libs/storage/partition_direct/partition_direct_actor.cpp2
-rw-r--r--ydb/core/nbs/cloud/blockstore/libs/storage/partition_direct/partition_direct_actor.h9
-rw-r--r--ydb/core/nbs/cloud/blockstore/libs/storage/partition_direct/partition_direct_ut.cpp40
-rw-r--r--ydb/core/nbs/cloud/blockstore/libs/storage/partition_direct/protos/ya.make2
-rw-r--r--ydb/core/nbs/cloud/blockstore/libs/storage/ya.make1
-rw-r--r--ydb/core/persqueue/public/constants.h8
-rw-r--r--ydb/core/protos/counters_columnshard.proto2
-rw-r--r--ydb/core/protos/feature_flags.proto1
-rw-r--r--ydb/core/protos/tablet.proto3
-rw-r--r--ydb/core/tx/columnshard/columnshard.cpp4
-rw-r--r--ydb/core/tx/columnshard/counters/portions.cpp2
-rw-r--r--ydb/core/tx/columnshard/counters/portions.h19
-rw-r--r--ydb/core/tx/replication/controller/secret_resolver.cpp3
-rw-r--r--ydb/core/tx/schemeshard/ut_export_reboots_s3/ya.make2
-rw-r--r--ydb/core/ymq/actor/create_topic_tx.cpp8
-rw-r--r--ydb/core/ymq/actor/set_queue_attributes.cpp8
-rw-r--r--ydb/docs/en/core/_assets/resources_weight.drawio6
-rw-r--r--ydb/docs/en/core/_assets/resources_weight.pngbin139033 -> 156898 bytes
-rw-r--r--ydb/docs/en/core/_includes/olap-data-types.md24
-rw-r--r--ydb/docs/en/core/concepts/_includes/secondary_indexes.md90
-rw-r--r--ydb/docs/en/core/concepts/glossary.md444
-rw-r--r--ydb/docs/en/core/concepts/query_execution/json_search.md73
-rw-r--r--ydb/docs/en/core/concepts/query_execution/toc_i.yaml6
-rw-r--r--ydb/docs/en/core/concepts/query_execution/topics.md54
-rw-r--r--ydb/docs/en/core/concepts/toc_i.yaml17
-rw-r--r--ydb/docs/en/core/concepts/topology.md7
-rw-r--r--ydb/docs/en/core/contributor/hive-booting.md6
-rw-r--r--ydb/docs/en/core/dev/json-indexes.md244
-rw-r--r--ydb/docs/en/core/dev/resource-consumption-management.md26
-rw-r--r--ydb/docs/en/core/dev/toc_p.yaml6
-rw-r--r--ydb/docs/en/core/devops/enterprise-manager/ai-assistant.md205
-rw-r--r--ydb/docs/en/core/devops/enterprise-manager/index.md1
-rw-r--r--ydb/docs/en/core/devops/enterprise-manager/toc_p.yaml2
-rw-r--r--ydb/docs/en/core/downloads/ydb-ansible.md1
-rw-r--r--ydb/docs/en/core/recipes/json-search/index.md12
-rw-r--r--ydb/docs/en/core/recipes/json-search/json-index-catalog.md133
-rw-r--r--ydb/docs/en/core/recipes/json-search/json-index-parameters.md139
-rw-r--r--ydb/docs/en/core/recipes/json-search/json-index-quickstart.md110
-rw-r--r--ydb/docs/en/core/recipes/json-search/json-index-typecheck.md94
-rw-r--r--ydb/docs/en/core/recipes/json-search/toc_p.yaml9
-rw-r--r--ydb/docs/en/core/recipes/toc_p.yaml5
-rw-r--r--ydb/docs/en/core/recipes/ydb-sdk/debug-jaeger.md170
-rw-r--r--ydb/docs/en/core/recipes/ydb-sdk/debug-logs-otel.md352
-rw-r--r--ydb/docs/en/core/recipes/ydb-sdk/debug-logs.md411
-rw-r--r--ydb/docs/en/core/recipes/ydb-sdk/debug-otel-metrics.md321
-rw-r--r--ydb/docs/en/core/recipes/ydb-sdk/debug-otel-tracing.md371
-rw-r--r--ydb/docs/en/core/recipes/ydb-sdk/debug-otel.md371
-rw-r--r--ydb/docs/en/core/recipes/ydb-sdk/debug-prometheus.md106
-rw-r--r--ydb/docs/en/core/recipes/ydb-sdk/debug.md12
-rw-r--r--ydb/docs/en/core/recipes/ydb-sdk/index.md5
-rw-r--r--ydb/docs/en/core/recipes/ydb-sdk/toc_i.yaml12
-rw-r--r--ydb/docs/en/core/reference/configuration/feature_flags.md46
-rw-r--r--ydb/docs/en/core/reference/configuration/hive_config.md (renamed from ydb/docs/en/core/reference/configuration/hive.md)0
-rw-r--r--ydb/docs/en/core/reference/configuration/host_configs.md53
-rw-r--r--ydb/docs/en/core/reference/configuration/index.md4
-rw-r--r--ydb/docs/en/core/reference/configuration/kafka_proxy_config.md (renamed from ydb/docs/en/core/reference/configuration/kafka.md)2
-rw-r--r--ydb/docs/en/core/reference/configuration/monitoring_config.md47
-rw-r--r--ydb/docs/en/core/reference/configuration/toc_p.yaml10
-rw-r--r--ydb/docs/en/core/reference/kafka-api/auth.md2
-rw-r--r--ydb/docs/en/core/reference/observability/tracing/external-traces.md2
-rw-r--r--ydb/docs/en/core/yql/reference/syntax/_includes/index_grammar_explanation.md40
-rw-r--r--ydb/docs/en/core/yql/reference/syntax/alter-resource-pool-classifier.md14
-rw-r--r--ydb/docs/en/core/yql/reference/syntax/alter_table/indexes.md141
-rw-r--r--ydb/docs/en/core/yql/reference/syntax/create-resource-pool-classifier.md41
-rw-r--r--ydb/docs/en/core/yql/reference/syntax/create_table/index.md125
-rw-r--r--ydb/docs/en/core/yql/reference/syntax/create_table/json_index.md42
-rw-r--r--ydb/docs/en/core/yql/reference/syntax/create_table/toc_i.yaml3
-rw-r--r--ydb/docs/en/core/yql/reference/syntax/discard.md43
-rw-r--r--ydb/docs/en/core/yql/reference/syntax/index.md4
-rw-r--r--ydb/docs/en/core/yql/reference/syntax/select/index.md238
-rw-r--r--ydb/docs/en/core/yql/reference/syntax/select/json_index.md125
-rw-r--r--ydb/docs/en/core/yql/reference/syntax/select/toc_i.yaml9
-rw-r--r--ydb/docs/en/core/yql/reference/syntax/toc_i.yaml2
-rw-r--r--ydb/docs/en/core/yql/reference/types/primitive.md6
-rw-r--r--ydb/docs/en/core/yql/toc_i.yaml3
-rw-r--r--ydb/docs/en/core/yql/toc_p.yaml4
-rw-r--r--ydb/docs/redirects.yaml30
-rw-r--r--ydb/docs/ru/core/_assets/resources_weight.drawio6
-rw-r--r--ydb/docs/ru/core/_assets/resources_weight.pngbin151477 -> 160131 bytes
-rw-r--r--ydb/docs/ru/core/_includes/olap-data-types.md24
-rw-r--r--ydb/docs/ru/core/concepts/query_execution/topics.md28
-rw-r--r--ydb/docs/ru/core/concepts/streaming-query/watermarks.md6
-rw-r--r--ydb/docs/ru/core/concepts/topology.md7
-rw-r--r--ydb/docs/ru/core/dev/resource-consumption-management.md18
-rw-r--r--ydb/docs/ru/core/dev/streaming-query/watermarks.md8
-rw-r--r--ydb/docs/ru/core/devops/deployment-options/manual/initial-deployment/deployment-configuration-v1.md62
-rw-r--r--ydb/docs/ru/core/devops/enterprise-manager/ai-assistant.md205
-rw-r--r--ydb/docs/ru/core/devops/enterprise-manager/index.md1
-rw-r--r--ydb/docs/ru/core/devops/enterprise-manager/toc_p.yaml2
-rw-r--r--ydb/docs/ru/core/downloads/ydb-ansible.md1
-rw-r--r--ydb/docs/ru/core/recipes/ydb-sdk/debug-jaeger.md172
-rw-r--r--ydb/docs/ru/core/recipes/ydb-sdk/debug-logs-otel.md352
-rw-r--r--ydb/docs/ru/core/recipes/ydb-sdk/debug-prometheus.md106
-rw-r--r--ydb/docs/ru/core/reference/configuration/host_configs.md53
-rw-r--r--ydb/docs/ru/core/reference/configuration/kafka_proxy_config.md2
-rw-r--r--ydb/docs/ru/core/yql/reference/syntax/alter-resource-pool-classifier.md2
-rw-r--r--ydb/docs/ru/core/yql/reference/syntax/create-resource-pool-classifier.md13
-rw-r--r--ydb/docs/ru/core/yql/reference/syntax/discard.md48
-rw-r--r--ydb/docs/ru/core/yql/reference/syntax/index.md4
-rw-r--r--ydb/docs/ru/core/yql/reference/syntax/select/toc_i.yaml2
-rw-r--r--ydb/docs/ru/core/yql/reference/syntax/toc_i.yaml2
-rw-r--r--ydb/docs/ru/core/yql/reference/types/primitive.md7
-rw-r--r--ydb/docs/ru/core/yql/toc_i.yaml5
-rw-r--r--ydb/docs/ru/core/yql/toc_p.yaml4
-rw-r--r--ydb/library/actors/core/actor.cpp10
-rw-r--r--ydb/library/actors/core/actor.h1
-rw-r--r--ydb/library/actors/core/ut/actor_ut.cpp22
-rw-r--r--ydb/library/actors/helpers/actor_liveness_checker.cpp96
-rw-r--r--ydb/library/actors/helpers/actor_liveness_checker.h47
-rw-r--r--ydb/library/actors/helpers/actor_liveness_checker_ut.cpp154
-rw-r--r--ydb/library/actors/helpers/ut/ya.make1
-rw-r--r--ydb/library/actors/helpers/ya.make3
-rw-r--r--ydb/library/actors/interconnect/interconnect_common.h1
-rw-r--r--ydb/library/actors/interconnect/interconnect_tcp_session.cpp15
-rw-r--r--ydb/library/actors/interconnect/interconnect_tcp_session.h5
-rw-r--r--ydb/library/actors/interconnect/interconnect_tcp_session_v2.cpp29
-rw-r--r--ydb/library/actors/interconnect/interconnect_tcp_session_v2.h17
-rw-r--r--ydb/library/actors/interconnect/interconnect_uring_engine.cpp294
-rw-r--r--ydb/library/actors/interconnect/interconnect_uring_event_queue.h44
-rw-r--r--ydb/library/actors/interconnect/subscriber_liveness_checker.cpp77
-rw-r--r--ydb/library/actors/interconnect/subscriber_liveness_checker.h36
-rw-r--r--ydb/library/actors/interconnect/ut/interconnect_ut.cpp150
-rw-r--r--ydb/library/actors/interconnect/ya.make2
-rw-r--r--ydb/library/services/services.proto2
-rw-r--r--ydb/library/workload/vector/vector_command_index.cpp11
-rw-r--r--ydb/library/workload/vector/vector_recall_evaluator.cpp2
-rw-r--r--ydb/library/workload/vector/vector_sampler.cpp8
-rw-r--r--ydb/library/workload/vector/vector_sql.cpp2
-rw-r--r--ydb/library/yql/providers/dq/task_runner/tasks_runner_pipe.cpp8
-rw-r--r--ydb/library/yql/providers/generic/actors/ut/yql_generic_lookup_actor_ut.cpp176
-rw-r--r--ydb/library/yql/providers/generic/actors/yql_generic_lookup_actor.cpp17
-rw-r--r--ydb/library/yql/providers/generic/connector/libcpp/error.cpp27
-rw-r--r--ydb/library/yql/providers/generic/connector/libcpp/ya.make2
-rw-r--r--ydb/library/yql/providers/generic/connector/tests/utils/ya.make1
-rw-r--r--ydb/mvp/meta/support_links/grafana_dashboard_common.cpp23
-rw-r--r--ydb/mvp/meta/support_links/grafana_logging_source.cpp26
-rw-r--r--ydb/mvp/meta/support_links/param_bindings.cpp22
-rw-r--r--ydb/mvp/meta/support_links/param_bindings.h4
-rw-r--r--ydb/public/sdk/cpp/src/client/impl/internal/db_driver_state/state.cpp7
-rw-r--r--ydb/public/sdk/cpp/src/client/topic/ut/basic_usage_ut.cpp35
-rw-r--r--ydb/public/sdk/cpp/src/library/grpc/client/grpc_common.cpp31
-rw-r--r--ydb/public/sdk/cpp/tests/unit/client/driver/driver_ut.cpp41
-rw-r--r--ydb/services/scheme_secret/resolver.h2
-rw-r--r--ydb/services/scheme_secret/service.cpp19
-rw-r--r--ydb/services/sqs_topic/create_queue.cpp6
-rw-r--r--ydb/services/sqs_topic/send_message.cpp61
-rw-r--r--ydb/services/sqs_topic/set_queue_attributes.cpp8
-rw-r--r--ydb/services/sqs_topic/ya.make1
-rw-r--r--ydb/services/workload_manager/metadata_subscription/resource_pool_classifier/manager.cpp3
-rw-r--r--ydb/services/workload_manager/ut/workload_service_ut.cpp4
-rw-r--r--ydb/tests/fq/generic/utils/settings.py9
-rw-r--r--ydb/tests/fq/streaming/generic/conftest.py34
-rw-r--r--ydb/tests/fq/streaming/generic/connector/Dockerfile12
-rwxr-xr-xydb/tests/fq/streaming/generic/connector/entrypoint.sh8
-rw-r--r--ydb/tests/fq/streaming/generic/connector/fq-connector-go.yaml60
-rw-r--r--ydb/tests/fq/streaming/generic/docker-compose.yml10
-rw-r--r--ydb/tests/fq/streaming/generic/test_iam_generic.py244
-rw-r--r--ydb/tests/fq/streaming/generic/ya.make51
-rw-r--r--ydb/tests/fq/streaming/ya.make5
-rw-r--r--ydb/tests/fq/streaming_common/common.py15
-rw-r--r--ydb/tests/fq/streaming_common/iam_grpc_emulator/bin/main.py41
-rw-r--r--ydb/tests/fq/streaming_common/ya.make4
-rw-r--r--ydb/tests/fq/ya.make1
-rw-r--r--ydb/tests/functional/secrets/test_old_secrets_usage.py484
-rw-r--r--ydb/tests/functional/secrets/ya.make1
-rw-r--r--ydb/tests/functional/sqs/messaging/test_generic_messaging.py26
-rw-r--r--ydb/tests/functional/sqs_topic/test_change_message_visibility.py1
-rw-r--r--ydb/tests/functional/sqs_topic/test_change_message_visibility_batch.py1
-rw-r--r--ydb/tests/functional/sqs_topic/test_delete_message.py2
-rw-r--r--ydb/tests/functional/sqs_topic/test_purge_queue.py1
-rw-r--r--ydb/tests/functional/sqs_topic/test_receive_message.py30
-rw-r--r--ydb/tests/functional/sqs_topic/test_send_message.py78
-rw-r--r--ydb/tests/functional/sqs_topic/test_send_message_batch.py1
-rw-r--r--ydb/tests/functional/tenants/test_remove_storage_groups.py12
-rw-r--r--ydb/tests/functional/udf_store/upload_udf/ya.make4
-rw-r--r--ydb/tests/library/sqs/requests_client.py5
-rw-r--r--ydb/tests/stress/compare_index_performance/tests/README.md32
-rw-r--r--ydb/tests/stress/compare_index_performance/tests/test_compare.py293
-rw-r--r--ydb/tests/stress/compare_index_performance/tests/ya.make2
-rw-r--r--ydb/tests/stress/vector_workload/__main__.py22
-rw-r--r--ydb/tests/stress/vector_workload/workload/__init__.py152
231 files changed, 7106 insertions, 4094 deletions
diff --git a/.github/config/muted_ya.txt b/.github/config/muted_ya.txt
index 771bbdb3670..084722be4e3 100644
--- a/.github/config/muted_ya.txt
+++ b/.github/config/muted_ya.txt
@@ -32,7 +32,6 @@ ydb/core/kqp/ut/scheme KqpScheme.CreateDropTableMultipleTime
ydb/core/load_test/ut GroupWriteTest.SimpleRdma
ydb/core/tx/schemeshard/ut_compaction_reboots SchemeshardForcedCompactionTestReboots.ForceCompactWithIndexCreation[TabletRebootsBucket0]
ydb/core/tx/schemeshard/ut_compaction_reboots unittest.[*/*] chunk
-ydb/core/tx/schemeshard/ut_export_reboots_s3 TExportToS3WithRebootsTests.ShouldSucceedOnManyTablesWithParquet[TabletRebootsBucket0]
ydb/library/actors/core/ut TestThreadContextQueueTimestamps.CurrentQueueTimestamps
ydb/library/actors/http/ut Http2Integration.Http2UpgradeFrom101
ydb/library/actors/interconnect/ut InterconnectDirectSession.DoesNotReceiveUnexpectedReplies
@@ -59,8 +58,6 @@ ydb/tests/compatibility/indexes test_bloom_filter_index.py.TestBloomFilterIndex.
ydb/tests/compatibility/indexes test_bloom_filter_index.py.TestBloomFilterIndex.test_bloom_filter_survives_version_change[restart_stable-26-3-1_to_stable-26-3-1]
ydb/tests/compatibility/indexes test_fulltext_index.py.TestFulltextIndex.test_fulltext_index[rolling_stable-26-3-1_to_prestable-26-4]
ydb/tests/compatibility/result_set_format py3test.[test_result_set_arrow.py */*] chunk
-ydb/tests/compatibility/streaming test_streaming.py.TestStreamingRestartToAnotherVersion.test_restart_to_another_version[restart_stable-26-3-1_to_stable-26-2-1-False]
-ydb/tests/compatibility/streaming test_streaming.py.TestStreamingRestartToAnotherVersion.test_restart_to_another_version[restart_stable-26-3-1_to_stable-26-2-1-True]
ydb/tests/compatibility/streaming test_streaming.py.TestStreamingRollingUpgradeAndDowngrade.test_rolling_upgrade[rolling_stable-26-2-1_to_stable-26-3-1-False]
ydb/tests/fq/streaming py3test.[*/*] chunk
ydb/tests/fq/streaming test_watermarks.py.TestWatermarksInYdb.test_empty_partition[shared-False]
@@ -68,6 +65,7 @@ ydb/tests/fq/streaming test_watermarks.py.TestWatermarksInYdb.test_idle_partitio
ydb/tests/functional/blobstorage test_node_warden_cache.py.TestNodeWardenCacheFile.test_cache_file_lifecycle_and_restart
ydb/tests/functional/query_cache/warmup test_warmup.py.TestCompileCacheViewPeerWarnings.test_peer_scan_warnings_increment_on_dead_peer
ydb/tests/functional/query_cache/warmup test_warmup.py.TestWarmupRestart.test_warmup_basic
+ydb/tests/functional/secrets test_old_secrets_usage.py.test_existing_objects_are_ok_if_old_secret_creation_is_disabled
ydb/tests/functional/security/mon/canonical test_mon_endpoints_auth.py.test[require_healthcheck_authentication-with_schema_grants]
ydb/tests/functional/sqs/migration/compatibility/common ydb.tests.functional.sqs.common.test_queue_counters.py.TestSqsGettingCounters.test_sqs_action_counters
ydb/tests/functional/sqs/migration/finished/common ydb.tests.functional.sqs.common.test_queue_counters.py.TestSqsGettingCounters.test_sqs_action_counters
diff --git a/.github/config/muted_ya_asan.txt b/.github/config/muted_ya_asan.txt
index a784de0a52a..5692ab4cc0f 100644
--- a/.github/config/muted_ya_asan.txt
+++ b/.github/config/muted_ya_asan.txt
@@ -461,6 +461,7 @@ ydb/core/kqp/ut/federated_query/datastreams KqpStreamingQueriesDdl.StreamingQuer
ydb/core/kqp/ut/federated_query/datastreams KqpStreamingQueriesDdl.StreamingQueryWithTwoGroupByHopsOnSameKey
ydb/core/kqp/ut/federated_query/datastreams KqpStreamingQueriesDdl.WritingInLocalYdbTablesWithCheckpoints
ydb/core/kqp/ut/federated_query/datastreams KqpStreamingQueriesSysView.ReadSysViewWithRowCountBackPressure
+ydb/core/kqp/ut/indexes/prefixed_vector KqpPrefixedVectorIndexes.PrefixedVectorEmptyIndexedTableInsertWithOverlap-Covered
ydb/core/kqp/ut/indexes/prefixed_vector KqpPrefixedVectorIndexes.PrefixedVectorIndexInsertNewPrefix-Nullable+Covered
ydb/core/kqp/ut/indexes/prefixed_vector KqpPrefixedVectorIndexes.PrefixedVectorIndexInsertNewPrefixWithOverlap+Covered
ydb/core/kqp/ut/indexes/prefixed_vector unittest.[*/*] chunk
@@ -469,6 +470,7 @@ ydb/core/kqp/ut/olap KqpOlapOptimizer.SpecialSliceToOneLayer
ydb/core/kqp/ut/olap KqpOlapOptimizer.TilingPlusPlusTtlDeletion
ydb/core/kqp/ut/olap/indexes KqpOlapIndexes.IndexesActualization-QueryService
ydb/core/kqp/ut/olap/indexes KqpOlapIndexes.MinMaxIndexAppliedToDataAfterCompaction-QueryService-SchemeObjectDisabled
+ydb/core/kqp/ut/olap/types KqpOlapJson.BloomMixIndexesOldSyntaxVariants[false,false,0,0,1000,0.5,CATEGORY_BLOOM_FILTER]
ydb/core/kqp/ut/olap/types KqpOlapJson.CompactionVariants[10,true,1024,1000,100,0.5]
ydb/core/kqp/ut/olap/types KqpOlapJson.SimpleExistsVariants[10,false,1024,1000,0,0.5]
ydb/core/kqp/ut/query KqpLimits.OutOfSpaceYQLUpsertFail
@@ -638,6 +640,7 @@ ydb/core/tx/tx_proxy/ut_ext_tenant TExtSubDomainTest.CreateTableInsideAndAlterDo
ydb/core/viewer/tests test.py.TestViewer.test_viewer_acl_write
ydb/core/viewer/tests test.py.TestViewer.test_viewer_query_long_multipart
ydb/core/viewer/ut Viewer.PutRecordViewer
+ydb/core/viewer/ut Viewer.QueryExecuteScript
ydb/core/ymq/actor/cloud_events/cloud_events_ut unittest.sole chunk
ydb/library/actors/interconnect/ut DynamicProxy.RaceCheck10
ydb/library/actors/interconnect/ut InterconnectDirectSession.DoesNotReceiveUnexpectedReplies
@@ -654,6 +657,7 @@ ydb/library/actors/interconnect/ut_huge_cluster HugeCluster.AllToAll
ydb/library/actors/interconnect/ut_huge_cluster unittest.sole chunk
ydb/library/yql/providers/generic/actors/ut unittest.sole chunk
ydb/public/sdk/cpp/src/client/topic/ut Describe.DescribePartitionPermissions
+ydb/public/sdk/cpp/src/client/topic/ut/slow TxUsage.Write_Only_Big_Messages_In_Wide_Transactions_Table
ydb/public/sdk/cpp/src/client/topic/ut/with_direct_read_ut BasicUsage.Producer_CloseTimeout
ydb/public/sdk/cpp/src/client/topic/ut/with_direct_read_ut BasicUsage.ReadWithRestartsAndLargeData
ydb/public/sdk/cpp/src/client/topic/ut/with_direct_read_ut BasicUsage.ReadWithRestartsAndLargeDataAndShuffle
diff --git a/.github/workflows/compare_index_performance.yml b/.github/workflows/compare_index_performance.yml
index 7430938e174..cf573105efb 100644
--- a/.github/workflows/compare_index_performance.yml
+++ b/.github/workflows/compare_index_performance.yml
@@ -33,10 +33,9 @@ on:
required: false
type: choice
options:
- - all
- vector
- fulltext
- default: all
+ default: vector
rows:
description: 'Number of rows in generated database'
required: false
@@ -74,6 +73,46 @@ on:
required: false
type: boolean
default: true
+ vector_clusters:
+ description: 'Number of clusters in kmeans tree for vector index (empty = server auto-detect)'
+ required: false
+ default: ''
+ vector_levels:
+ description: 'Number of levels in kmeans tree for vector index (empty = server auto-detect)'
+ required: false
+ default: ''
+ dataset_source:
+ description: 'Source of the test dataset: generate (random data) or s3 (import fixed dataset from S3)'
+ required: false
+ type: choice
+ options:
+ - generate
+ - s3
+ default: generate
+ s3_endpoint:
+ description: 'S3 endpoint for dataset import (required when dataset_source=s3)'
+ required: false
+ default: 'https://storage.yandexcloud.net'
+ s3_bucket:
+ description: 'S3 bucket for dataset import (required when dataset_source=s3)'
+ required: false
+ default: 'vector-index'
+ s3_source:
+ description: 'S3 object key prefix for dataset import (required when dataset_source=s3)'
+ required: false
+ default: ''
+ s3_destination:
+ description: 'Database destination path for dataset import (required when dataset_source=s3)'
+ required: false
+ default: '/Root/testdb/wikipedia'
+ s3_query_source:
+ description: 'S3 object key prefix for pre-computed queries table (optional when dataset_source=s3)'
+ required: false
+ default: ''
+ s3_query_destination:
+ description: 'Database destination path for queries table (optional, used with s3_query_source)'
+ required: false
+ default: ''
jobs:
compare:
@@ -125,10 +164,17 @@ jobs:
BASELINE_SHA: ${{ steps.resolve-ref.outputs.baseline_sha }}
INPUT_BUILD_PRESET: ${{ inputs.build_preset || 'release' }}
run: |
+ # Self-hosted runners reuse the workspace, so a ../baseline worktree
+ # (and its .git/worktrees admin entry) may linger from a prior run and
+ # make `git worktree add` fail with "'../baseline' already exists".
+ # Clean any stale copy first so the step is idempotent.
+ git worktree remove --force ../baseline 2>/dev/null || true
+ rm -rf ../baseline
+ git worktree prune
git fetch --depth=1 origin "$BASELINE_SHA"
git worktree add ../baseline "$BASELINE_SHA"
cd ../baseline
- ./ya make --build "$INPUT_BUILD_PRESET" ydb/apps/ydbd
+ ./ya make -DDEBUGINFO_LINES_ONLY --build "$INPUT_BUILD_PRESET" ydb/apps/ydbd
BINARY=$(find . -name ydbd -path '*/ydb/apps/ydbd/ydbd' -not -path '*/ydb/apps/ydbd/ydbd/*' 2>/dev/null | head -1)
BINARY=$(realpath "$BINARY")
echo "binary=$BINARY" >> "$GITHUB_OUTPUT"
@@ -151,14 +197,43 @@ jobs:
INPUT_BASELINE_SHA: ${{ steps.resolve-ref.outputs.baseline_sha }}
INPUT_CURRENT_SHA: ${{ steps.resolve-ref.outputs.checkout_ref }}
INPUT_BASELINE_YDBD: ${{ steps.build-baseline.outputs.binary }}
+ INPUT_VECTOR_CLUSTERS: ${{ inputs.vector_clusters || '' }}
+ INPUT_VECTOR_LEVELS: ${{ inputs.vector_levels || '' }}
+ INPUT_DATASET_SOURCE: ${{ inputs.dataset_source || 'generate' }}
+ INPUT_S3_ENDPOINT: ${{ inputs.s3_endpoint || '' }}
+ INPUT_S3_BUCKET: ${{ inputs.s3_bucket || '' }}
+ INPUT_S3_SOURCE: ${{ inputs.s3_source || '' }}
+ INPUT_S3_DESTINATION: ${{ inputs.s3_destination || '' }}
+ INPUT_S3_QUERY_SOURCE: ${{ inputs.s3_query_source || '' }}
+ INPUT_S3_QUERY_DESTINATION: ${{ inputs.s3_query_destination || '' }}
run: |
# The feature-flag / table_service_config inputs are comma-separated
# strings; the test splits them itself, so they are passed through as-is.
EXTRA_PARAMS=""
if [[ -n "$INPUT_BASELINE_YDBD" ]]; then
- EXTRA_PARAMS="--test-param compare_baseline_ydbd=$INPUT_BASELINE_YDBD"
+ EXTRA_PARAMS="$EXTRA_PARAMS --test-param compare_baseline_ydbd=$INPUT_BASELINE_YDBD"
+ fi
+ if [[ -n "$INPUT_VECTOR_CLUSTERS" ]]; then
+ EXTRA_PARAMS="$EXTRA_PARAMS --test-param compare_vector_clusters=$INPUT_VECTOR_CLUSTERS"
+ fi
+ if [[ -n "$INPUT_VECTOR_LEVELS" ]]; then
+ EXTRA_PARAMS="$EXTRA_PARAMS --test-param compare_vector_levels=$INPUT_VECTOR_LEVELS"
+ fi
+ if [[ "$INPUT_DATASET_SOURCE" == "s3" ]]; then
+ EXTRA_PARAMS="$EXTRA_PARAMS --test-param compare_dataset_source=s3"
+ EXTRA_PARAMS="$EXTRA_PARAMS --test-param compare_s3_endpoint=$INPUT_S3_ENDPOINT"
+ EXTRA_PARAMS="$EXTRA_PARAMS --test-param compare_s3_bucket=$INPUT_S3_BUCKET"
+ EXTRA_PARAMS="$EXTRA_PARAMS --test-param compare_s3_source=$INPUT_S3_SOURCE"
+ EXTRA_PARAMS="$EXTRA_PARAMS --test-param compare_s3_destination=$INPUT_S3_DESTINATION"
+ if [[ -n "$INPUT_S3_QUERY_SOURCE" ]]; then
+ EXTRA_PARAMS="$EXTRA_PARAMS --test-param compare_s3_query_source=$INPUT_S3_QUERY_SOURCE"
+ if [[ -n "$INPUT_S3_QUERY_DESTINATION" ]]; then
+ EXTRA_PARAMS="$EXTRA_PARAMS --test-param compare_s3_query_destination=$INPUT_S3_QUERY_DESTINATION"
+ fi
+ fi
fi
- ./ya make --build "$INPUT_BUILD_PRESET" -tA --test-disable-timeout \
+
+ ./ya make -DDEBUGINFO_LINES_ONLY --build "$INPUT_BUILD_PRESET" -tA --test-disable-timeout \
ydb/tests/stress/compare_index_performance/tests \
--test-param compare_duration="$INPUT_DURATION" \
--test-param compare_build_preset="$INPUT_BUILD_PRESET" \
@@ -175,7 +250,7 @@ jobs:
--test-param compare_current_table_service_config="$INPUT_CURRENT_TABLE_SERVICE_CONFIG" \
--test-param compare_baseline_sha="$INPUT_BASELINE_SHA" \
--test-param compare_current_sha="$INPUT_CURRENT_SHA" \
- $EXTRA_PARAMS
+ $EXTRA_PARAMS 2>&1 | tee ya_make.log; exit "${PIPESTATUS[0]}"
- name: Post results to job summary
if: always()
@@ -191,6 +266,15 @@ jobs:
done
else
echo "No report generated." >> "$GITHUB_STEP_SUMMARY"
+ if [[ -f ya_make.log ]]; then
+ echo "" >> "$GITHUB_STEP_SUMMARY"
+ echo "<details><summary>Build/test log (last 50 lines)</summary>" >> "$GITHUB_STEP_SUMMARY"
+ echo "" >> "$GITHUB_STEP_SUMMARY"
+ echo '```' >> "$GITHUB_STEP_SUMMARY"
+ tail -50 ya_make.log >> "$GITHUB_STEP_SUMMARY"
+ echo '```' >> "$GITHUB_STEP_SUMMARY"
+ echo "</details>" >> "$GITHUB_STEP_SUMMARY"
+ fi
fi
- name: Post results to pull request
@@ -214,6 +298,15 @@ jobs:
done
else
echo "No report generated."
+ if [[ -f ya_make.log ]]; then
+ echo ""
+ echo "<details><summary>Build/test log (last 50 lines)</summary>"
+ echo ""
+ echo '```'
+ tail -50 ya_make.log
+ echo '```'
+ echo "</details>"
+ fi
fi
echo "[Workflow run](${run_url})"
} | .github/scripts/tests/comment-pr.py --rewrite --no-timestamp
diff --git a/.github/workflows/docs_build.yaml b/.github/workflows/docs_build.yaml
index f478f7c836b..bcf6f8b3fcf 100644
--- a/.github/workflows/docs_build.yaml
+++ b/.github/workflows/docs_build.yaml
@@ -40,40 +40,6 @@ jobs:
id: sha
run: echo "value=$(git rev-parse HEAD)" >> $GITHUB_OUTPUT
- # Branches listed below keep the old logic (use-ya-opensource: false).
- # All other branches use the new logic (use-ya-opensource: true).
- - name: Determine ya-opensource flag
- id: ya
- uses: actions/github-script@v8
- with:
- script: |
- const legacyBranches = [
- 'stable-25-1',
- 'stable-25-2',
- 'stable-25-2-1',
- 'stable-25-3',
- 'stable-25-3-1',
- 'stable-25-4',
- 'stable-25-4-1',
- 'stable-26-1',
- 'stable-26-1-1',
- 'stable-26-2',
- 'stable-26-2-1',
- 'stable-26-1-1-enterprise',
- ];
- let baseRef;
- if (context.eventName === 'workflow_dispatch') {
- const { data: pr } = await github.rest.pulls.get({
- owner: context.repo.owner,
- repo: context.repo.repo,
- pull_number: Number(${{ inputs.pull_number }}),
- });
- baseRef = pr.base.ref;
- } else {
- baseRef = context.payload.pull_request.base.ref;
- }
- core.setOutput('use', legacyBranches.includes(baseRef) ? 'false' : 'true');
-
- name: Disable PDF for PR build
run: yq -i '.["docs-viewer"].pdf = false | .singlePage = false' ydb/docs/.yfm
@@ -83,7 +49,7 @@ jobs:
revision: "pr-${{ github.event_name == 'workflow_dispatch' && inputs.pull_number || github.event.pull_request.number }}-${{ steps.sha.outputs.value }}"
src-root: "./ydb/docs"
cli-version: "4.59.12"
- use-ya-opensource: ${{ steps.ya.outputs.use }}
+ use-ya-opensource: true
- name: Fail on documentation build issues
run: |
diff --git a/.github/workflows/docs_build_rebuild.yaml b/.github/workflows/docs_build_rebuild.yaml
index 786a777609a..d327829e867 100644
--- a/.github/workflows/docs_build_rebuild.yaml
+++ b/.github/workflows/docs_build_rebuild.yaml
@@ -64,33 +64,6 @@ jobs:
if: steps.docs-changes.outputs.has_docs == 'true'
run: echo "value=$(git rev-parse HEAD)" >> $GITHUB_OUTPUT
- # Branches listed below keep the old logic (use-ya-opensource: false).
- # All other branches use the new logic (use-ya-opensource: true).
- # This workflow only runs on pull_request_target, so the relevant branch
- # is the pull request base ref.
- - name: Determine ya-opensource flag
- id: ya
- if: steps.docs-changes.outputs.has_docs == 'true'
- uses: actions/github-script@v8
- with:
- script: |
- const legacyBranches = [
- 'stable-25-1',
- 'stable-25-2',
- 'stable-25-2-1',
- 'stable-25-3',
- 'stable-25-3-1',
- 'stable-25-4',
- 'stable-25-4-1',
- 'stable-26-1',
- 'stable-26-1-1',
- 'stable-26-2',
- 'stable-26-2-1',
- ];
- const baseRef = context.payload.pull_request.base.ref;
- core.setOutput('use', legacyBranches.includes(baseRef) ? 'false' : 'true');
-
-
- name: Build
if: steps.docs-changes.outputs.has_docs == 'true'
uses: diplodoc-platform/docs-build-action@v3
@@ -98,7 +71,7 @@ jobs:
revision: "pr-${{ github.event.pull_request.number }}-${{ steps.sha.outputs.value }}"
src-root: "./ydb/docs"
cli-version: "4.59.12"
- use-ya-opensource: ${{ steps.ya.outputs.use }}
+ use-ya-opensource: true
- name: Fail on documentation build issues
if: steps.docs-changes.outputs.has_docs == 'true'
diff --git a/.github/workflows/docs_release.yaml b/.github/workflows/docs_release.yaml
index d1f91972be2..e808f8887a0 100644
--- a/.github/workflows/docs_release.yaml
+++ b/.github/workflows/docs_release.yaml
@@ -20,38 +20,13 @@ jobs:
- name: Checkout
uses: actions/checkout@v5
- # Branches listed below keep the old logic (use-ya-opensource: false).
- # All other branches use the new logic (use-ya-opensource: true).
- # This workflow runs on push / workflow_dispatch, so the relevant branch
- # is the one being built (context.ref), not a pull request base ref.
- - name: Determine ya-opensource flag
- id: ya
- uses: actions/github-script@v8
- with:
- script: |
- const legacyBranches = [
- 'stable-25-1',
- 'stable-25-2',
- 'stable-25-2-1',
- 'stable-25-3',
- 'stable-25-3-1',
- 'stable-25-4',
- 'stable-25-4-1',
- 'stable-26-1',
- 'stable-26-1-1',
- 'stable-26-2',
- 'stable-26-2-1',
- ];
- const branch = context.ref.replace('refs/heads/', '');
- core.setOutput('use', legacyBranches.includes(branch) ? 'false' : 'true');
-
- name: Build
uses: diplodoc-platform/docs-build-action@v3
with:
revision: "${{ github.sha }}"
src-root: ${{ vars.SRC_ROOT }}
cli-version: "4.59.12"
- use-ya-opensource: ${{ steps.ya.outputs.use }}
+ use-ya-opensource: true
upload:
needs: build
diff --git a/.github/workflows/ydbdoc-review.yml b/.github/workflows/ydbdoc-review.yml
index 88e943cc62c..32e0e93a119 100644
--- a/.github/workflows/ydbdoc-review.yml
+++ b/.github/workflows/ydbdoc-review.yml
@@ -21,11 +21,14 @@ jobs:
group: ydbdoc-review-${{ github.event.pull_request.number }}
cancel-in-progress: true
steps:
- - name: Checkout PR head
+ - name: Checkout translation source
uses: actions/checkout@v5
with:
- ref: ${{ github.event.pull_request.head.sha }}
+ # Merged PR (в т.ч. из форка): коммит уже в upstream (main).
+ # Open PR: head как раньше.
+ ref: ${{ github.event.pull_request.merged && github.event.pull_request.merge_commit_sha || github.event.pull_request.head.sha }}
fetch-depth: 0
+ allow-unsafe-pr-checkout: true
- name: Fetch merge base ref
run: |
diff --git a/.github/workflows/ydbdoc-verify.yml b/.github/workflows/ydbdoc-verify.yml
index 93b16919b9e..81160bd6316 100644
--- a/.github/workflows/ydbdoc-verify.yml
+++ b/.github/workflows/ydbdoc-verify.yml
@@ -21,11 +21,12 @@ jobs:
group: ydbdoc-verify-${{ github.event.pull_request.number }}
cancel-in-progress: true
steps:
- - name: Checkout PR head
+ - name: Checkout translation source
uses: actions/checkout@v5
with:
- ref: ${{ github.event.pull_request.head.sha }}
+ ref: ${{ github.event.pull_request.merged && github.event.pull_request.merge_commit_sha || github.event.pull_request.head.sha }}
fetch-depth: 0
+ allow-unsafe-pr-checkout: true
- name: Fetch merge base ref
run: |
diff --git a/ydb/core/cms/console/console_tenants_manager.cpp b/ydb/core/cms/console/console_tenants_manager.cpp
index d64c0ec597a..49594dcc953 100644
--- a/ydb/core/cms/console/console_tenants_manager.cpp
+++ b/ydb/core/cms/console/console_tenants_manager.cpp
@@ -4136,7 +4136,7 @@ void TTenantsManager::Handle(TEvHive::TEvShrinkStoragePoolDone::TPtr &ev, const
return;
}
auto pool = poolIt->second;
- if (pool->State != TStoragePool::EState::SHRINKING) {
+ if (pool->State != TStoragePool::EState::SHRINKING && pool->State != TStoragePool::EState::NOT_UPDATED) {
return;
}
diff --git a/ydb/core/driver_lib/run/kikimr_services_initializers.cpp b/ydb/core/driver_lib/run/kikimr_services_initializers.cpp
index e6ba3aef6b2..a9ca1d476b3 100644
--- a/ydb/core/driver_lib/run/kikimr_services_initializers.cpp
+++ b/ydb/core/driver_lib/run/kikimr_services_initializers.cpp
@@ -115,6 +115,7 @@
#include <ydb/core/nbs/cloud/blockstore/config/protos/storage.pb.h>
#include <ydb/core/nbs/cloud/blockstore/libs/storage/volume/volume.h>
#include <ydb/core/nbs/cloud/blockstore/libs/storage/partition_direct/partition_direct.h>
+#include <ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_controller.h>
#endif
#include <ydb/core/mon/mon.h>
@@ -1371,6 +1372,7 @@ void TLocalServiceInitializer::InitializeServices(
#if defined(YDB_EMBEDDED_NBS_ENABLED)
addToLocalConfig(TTabletTypes::BlockStoreVolumeDirect, &NYdb::NBS::NStorage::CreateVolumeTablet, TMailboxType::ReadAsFilled, appData->UserPoolId);
addToLocalConfig(TTabletTypes::BlockStorePartitionDirect, &NYdb::NBS::NBlockStore::NStorage::NPartitionDirect::CreatePartitionTablet, TMailboxType::ReadAsFilled, appData->UserPoolId);
+ addToLocalConfig(TTabletTypes::DbsController, &NYdb::NBS::NBlockStore::NStorage::NDbsController::CreateDbsControllerTablet, TMailboxType::ReadAsFilled, appData->UserPoolId);
#endif
if (Config.GetShutdownConfig().HasDrainTimeoutSeconds()) {
diff --git a/ydb/core/driver_lib/run/ya.make b/ydb/core/driver_lib/run/ya.make
index b896adb2491..8b8ffd0fa96 100644
--- a/ydb/core/driver_lib/run/ya.make
+++ b/ydb/core/driver_lib/run/ya.make
@@ -202,6 +202,7 @@ IF (OS_LINUX AND YDB_EMBEDDED_NBS_ENABLED)
PEERDIR(
ydb/core/nbs/cloud/blockstore/bootstrap
ydb/core/nbs/cloud/blockstore/config/protos
+ ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller
ydb/core/nbs/cloud/blockstore/libs/storage/ss_proxy
ydb/core/nbs/cloud/blockstore/libs/storage/volume
diff --git a/ydb/core/http_proxy/ut/sqs_topic_ut.cpp b/ydb/core/http_proxy/ut/sqs_topic_ut.cpp
index e73901b2549..3c486175ca1 100644
--- a/ydb/core/http_proxy/ut/sqs_topic_ut.cpp
+++ b/ydb/core/http_proxy/ut/sqs_topic_ut.cpp
@@ -2700,7 +2700,7 @@ Y_UNIT_TEST_SUITE(TestSqsTopicHttpProxy) {
auto json = CreateQueue({
{"QueueName", queueName},
- {"Attributes", NJson::TJsonMap{{"FifoQueue", "true"}, {"ContentBasedDeduplication", "true"}}}
+ {"Attributes", NJson::TJsonMap{{"FifoQueue", "true"}}}
});
UNIT_ASSERT(!GetByPath<TString>(json, "QueueUrl").empty());
@@ -2713,11 +2713,11 @@ Y_UNIT_TEST_SUITE(TestSqsTopicHttpProxy) {
UNIT_ASSERT_VALUES_EQUAL(
description.GetPartitionWriteSpeedMessagesPerSecond(),
- NPQ::CONTENT_BASED_DEDUPLICATION_MESSAGE_LIMIT
+ NPQ::FIFO_PARTITION_WRITE_SPEED_MESSAGES_PER_SECOND
);
UNIT_ASSERT_VALUES_EQUAL(
description.GetPartitionWriteBurstMessages(),
- NPQ::CONTENT_BASED_DEDUPLICATION_MESSAGE_BURST
+ NPQ::FIFO_PARTITION_WRITE_BURST_MESSAGES
);
driver.Stop(true);
@@ -2747,11 +2747,11 @@ Y_UNIT_TEST_SUITE(TestSqsTopicHttpProxy) {
UNIT_ASSERT_VALUES_EQUAL(
description.GetPartitionWriteSpeedMessagesPerSecond(),
- NPQ::CONTENT_BASED_DEDUPLICATION_MESSAGE_LIMIT
+ NPQ::FIFO_PARTITION_WRITE_SPEED_MESSAGES_PER_SECOND
);
UNIT_ASSERT_VALUES_EQUAL(
description.GetPartitionWriteBurstMessages(),
- NPQ::CONTENT_BASED_DEDUPLICATION_MESSAGE_BURST
+ NPQ::FIFO_PARTITION_WRITE_BURST_MESSAGES
);
driver.Stop(true);
@@ -2781,11 +2781,11 @@ Y_UNIT_TEST_SUITE(TestSqsTopicHttpProxy) {
UNIT_ASSERT_VALUES_EQUAL(
description.GetPartitionWriteSpeedMessagesPerSecond(),
- NPQ::DEFAULT_PARTITION_WRITE_SPEED_MESSAGES_PER_SECOND
+ NPQ::FIFO_PARTITION_WRITE_SPEED_MESSAGES_PER_SECOND
);
UNIT_ASSERT_VALUES_EQUAL(
description.GetPartitionWriteBurstMessages(),
- NPQ::DEFAULT_PARTITION_WRITE_SPEED_MESSAGES_PER_SECOND
+ NPQ::FIFO_PARTITION_WRITE_BURST_MESSAGES
);
driver.Stop(true);
diff --git a/ydb/core/http_proxy/ut/sqs_topic_xml_ut.cpp b/ydb/core/http_proxy/ut/sqs_topic_xml_ut.cpp
index 5b036d41f45..67c7f7a57ac 100644
--- a/ydb/core/http_proxy/ut/sqs_topic_xml_ut.cpp
+++ b/ydb/core/http_proxy/ut/sqs_topic_xml_ut.cpp
@@ -1336,7 +1336,7 @@ Y_UNIT_TEST_SUITE(TestSqsTopicHttpProxyXml) {
auto json = CreateQueueXml({
{"QueueName", queueName},
- {"Attributes", NJson::TJsonMap{{"FifoQueue", "true"}, {"ContentBasedDeduplication", "true"}}}
+ {"Attributes", NJson::TJsonMap{{"FifoQueue", "true"}}}
});
UNIT_ASSERT(!GetByPath<TString>(json, "QueueUrl").empty());
@@ -1349,11 +1349,11 @@ Y_UNIT_TEST_SUITE(TestSqsTopicHttpProxyXml) {
UNIT_ASSERT_VALUES_EQUAL(
description.GetPartitionWriteSpeedMessagesPerSecond(),
- NPQ::CONTENT_BASED_DEDUPLICATION_MESSAGE_LIMIT
+ NPQ::FIFO_PARTITION_WRITE_SPEED_MESSAGES_PER_SECOND
);
UNIT_ASSERT_VALUES_EQUAL(
description.GetPartitionWriteBurstMessages(),
- NPQ::CONTENT_BASED_DEDUPLICATION_MESSAGE_BURST
+ NPQ::FIFO_PARTITION_WRITE_BURST_MESSAGES
);
driver.Stop(true);
@@ -1383,11 +1383,11 @@ Y_UNIT_TEST_SUITE(TestSqsTopicHttpProxyXml) {
UNIT_ASSERT_VALUES_EQUAL(
description.GetPartitionWriteSpeedMessagesPerSecond(),
- NPQ::CONTENT_BASED_DEDUPLICATION_MESSAGE_LIMIT
+ NPQ::FIFO_PARTITION_WRITE_SPEED_MESSAGES_PER_SECOND
);
UNIT_ASSERT_VALUES_EQUAL(
description.GetPartitionWriteBurstMessages(),
- NPQ::CONTENT_BASED_DEDUPLICATION_MESSAGE_BURST
+ NPQ::FIFO_PARTITION_WRITE_BURST_MESSAGES
);
driver.Stop(true);
@@ -1417,11 +1417,11 @@ Y_UNIT_TEST_SUITE(TestSqsTopicHttpProxyXml) {
UNIT_ASSERT_VALUES_EQUAL(
description.GetPartitionWriteSpeedMessagesPerSecond(),
- NPQ::DEFAULT_PARTITION_WRITE_SPEED_MESSAGES_PER_SECOND
+ NPQ::FIFO_PARTITION_WRITE_SPEED_MESSAGES_PER_SECOND
);
UNIT_ASSERT_VALUES_EQUAL(
description.GetPartitionWriteBurstMessages(),
- NPQ::DEFAULT_PARTITION_WRITE_SPEED_MESSAGES_PER_SECOND
+ NPQ::FIFO_PARTITION_WRITE_BURST_MESSAGES
);
driver.Stop(true);
diff --git a/ydb/core/http_proxy/ut/ymq_ut.cpp b/ydb/core/http_proxy/ut/ymq_ut.cpp
index 51947810463..ccf90bb0c25 100644
--- a/ydb/core/http_proxy/ut/ymq_ut.cpp
+++ b/ydb/core/http_proxy/ut/ymq_ut.cpp
@@ -377,7 +377,7 @@ Y_UNIT_TEST_SUITE(TestYmqHttpProxy) {
const TString queueName = "CreateQueueContentBasedDeduplicationRateLimit.fifo";
auto json = CreateQueue({
{"QueueName", queueName},
- {"Attributes", NJson::TJsonMap{{"FifoQueue", "true"}, {"ContentBasedDeduplication", "true"}}}
+ {"Attributes", NJson::TJsonMap{{"FifoQueue", "true"}}}
});
const TString queueUrl = GetByPath<TString>(json, "QueueUrl");
UNIT_ASSERT(!queueUrl.empty());
@@ -385,11 +385,11 @@ Y_UNIT_TEST_SUITE(TestYmqHttpProxy) {
const auto description = DescribeMigrationTopicByQueueUrl(*this, queueUrl, queueName);
UNIT_ASSERT_VALUES_EQUAL(
description.GetPartitionWriteSpeedMessagesPerSecond(),
- NKikimr::NPQ::CONTENT_BASED_DEDUPLICATION_MESSAGE_LIMIT
+ NKikimr::NPQ::FIFO_PARTITION_WRITE_SPEED_MESSAGES_PER_SECOND
);
UNIT_ASSERT_VALUES_EQUAL(
description.GetPartitionWriteBurstMessages(),
- NKikimr::NPQ::CONTENT_BASED_DEDUPLICATION_MESSAGE_BURST
+ NKikimr::NPQ::FIFO_PARTITION_WRITE_BURST_MESSAGES
);
}
@@ -417,11 +417,11 @@ Y_UNIT_TEST_SUITE(TestYmqHttpProxy) {
const auto description = DescribeMigrationTopicByQueueUrl(*this, queueUrl, queueName);
UNIT_ASSERT_VALUES_EQUAL(
description.GetPartitionWriteSpeedMessagesPerSecond(),
- NKikimr::NPQ::CONTENT_BASED_DEDUPLICATION_MESSAGE_LIMIT
+ NKikimr::NPQ::FIFO_PARTITION_WRITE_SPEED_MESSAGES_PER_SECOND
);
UNIT_ASSERT_VALUES_EQUAL(
description.GetPartitionWriteBurstMessages(),
- NKikimr::NPQ::CONTENT_BASED_DEDUPLICATION_MESSAGE_BURST
+ NKikimr::NPQ::FIFO_PARTITION_WRITE_BURST_MESSAGES
);
}
@@ -449,11 +449,11 @@ Y_UNIT_TEST_SUITE(TestYmqHttpProxy) {
const auto description = DescribeMigrationTopicByQueueUrl(*this, queueUrl, queueName);
UNIT_ASSERT_VALUES_EQUAL(
description.GetPartitionWriteSpeedMessagesPerSecond(),
- NKikimr::NPQ::DEFAULT_PARTITION_WRITE_SPEED_MESSAGES_PER_SECOND
+ NKikimr::NPQ::FIFO_PARTITION_WRITE_SPEED_MESSAGES_PER_SECOND
);
UNIT_ASSERT_VALUES_EQUAL(
description.GetPartitionWriteBurstMessages(),
- NKikimr::NPQ::DEFAULT_PARTITION_WRITE_SPEED_MESSAGES_PER_SECOND
+ NKikimr::NPQ::FIFO_PARTITION_WRITE_BURST_MESSAGES
);
}
diff --git a/ydb/core/kqp/common/kqp_tx.cpp b/ydb/core/kqp/common/kqp_tx.cpp
index 198fa194166..cef634fd781 100644
--- a/ydb/core/kqp/common/kqp_tx.cpp
+++ b/ydb/core/kqp/common/kqp_tx.cpp
@@ -335,8 +335,7 @@ bool HasUncommittedChangesRead(THashSet<NKikimr::TTableId>& modifiedTables, cons
modifiedTables.insert(getTable(index.GetTable()));
}
- // For plans compatibility with old indexes. Don't need it for new.
- if (!settings.GetLookupColumns().empty() && tableModifiedBefore) {
+ if (settings.GetNeedLookup() && tableModifiedBefore) {
AFL_ENSURE(settings.GetType() != NKikimrKqp::TKqpTableSinkSettings::MODE_INSERT);
return true;
}
diff --git a/ydb/core/kqp/gateway/behaviour/external_data_source/manager.cpp b/ydb/core/kqp/gateway/behaviour/external_data_source/manager.cpp
index adb86b967cb..6ae388f4800 100644
--- a/ydb/core/kqp/gateway/behaviour/external_data_source/manager.cpp
+++ b/ydb/core/kqp/gateway/behaviour/external_data_source/manager.cpp
@@ -10,6 +10,7 @@
#include <ydb/library/conclusion/generic/result.h>
#include <ydb/library/actors/core/actor.h>
#include <ydb/core/external_sources/iceberg_fields.h>
+#include <ydb/services/scheme_secret/resolver.h>
namespace NKikimr::NKqp {
@@ -60,6 +61,18 @@ TString GetSecretName(const NYql::TCreateObjectSettings& settings, const TString
return GetOrEmpty(settings, secretKeyPrefix + "_path");
}
+[[nodiscard]] TYqlConclusionStatus CheckOldSecretCreationAllowed(
+ bool disableOldSecretCreation,
+ const TString& secretName)
+{
+ if (disableOldSecretCreation && secretName && !NSecret::IsSchemeSecret(secretName)) {
+ return TYqlConclusionStatus::Fail(
+ NYql::TIssuesIds::KIKIMR_BAD_REQUEST,
+ "Old secrets are disabled for creating new objects. Please use new secrets");
+ }
+ return TYqlConclusionStatus::Success();
+}
+
[[nodiscard]] TYqlConclusionStatus FillCreateExternalDataSourceDesc(
NKikimrSchemeOp::TExternalDataSourceDescription& externalDataSourceDesc,
const TString& name,
@@ -71,6 +84,9 @@ TString GetSecretName(const NYql::TCreateObjectSettings& settings, const TString
externalDataSourceDesc.SetLocation(GetOrEmpty(settings, "location"));
externalDataSourceDesc.SetInstallation(GetOrEmpty(settings, "installation"));
+ const bool disableOldSecretCreation = actorSystem &&
+ AppData(actorSystem)->FeatureFlags.GetDisableOldSecretCreation();
+
const TString& authMethod = GetOrEmpty(settings, "auth_method");
if (authMethod == "NONE") {
externalDataSourceDesc.MutableAuth()->MutableNone();
@@ -78,30 +94,78 @@ TString GetSecretName(const NYql::TCreateObjectSettings& settings, const TString
auto& sa = *externalDataSourceDesc.MutableAuth()->MutableServiceAccount();
sa.SetId(GetOrEmpty(settings, "service_account_id"));
sa.SetSecretName(GetSecretName(settings, "service_account_secret"));
+ if (const auto status =
+ CheckOldSecretCreationAllowed(
+ disableOldSecretCreation, sa.GetSecretName()); status.IsFail())
+ {
+ return status;
+ }
} else if (authMethod == "BASIC") {
auto& basic = *externalDataSourceDesc.MutableAuth()->MutableBasic();
basic.SetLogin(GetOrEmpty(settings, "login"));
basic.SetPasswordSecretName(GetSecretName(settings, "password_secret"));
+ if (const auto status =
+ CheckOldSecretCreationAllowed(
+ disableOldSecretCreation, basic.GetPasswordSecretName()); status.IsFail())
+ {
+ return status;
+ }
} else if (authMethod == "MDB_BASIC") {
auto& mdbBasic = *externalDataSourceDesc.MutableAuth()->MutableMdbBasic();
mdbBasic.SetServiceAccountId(GetOrEmpty(settings, "service_account_id"));
mdbBasic.SetServiceAccountSecretName(GetSecretName(settings, "service_account_secret"));
mdbBasic.SetLogin(GetOrEmpty(settings, "login"));
mdbBasic.SetPasswordSecretName(GetSecretName(settings, "password_secret"));
+ if (const auto status =
+ CheckOldSecretCreationAllowed(
+ disableOldSecretCreation, mdbBasic.GetServiceAccountSecretName()); status.IsFail())
+ {
+ return status;
+ }
+ if (const auto status =
+ CheckOldSecretCreationAllowed(
+ disableOldSecretCreation, mdbBasic.GetPasswordSecretName()); status.IsFail())
+ {
+ return status;
+ }
} else if (authMethod == "AWS") {
auto& aws = *externalDataSourceDesc.MutableAuth()->MutableAws();
aws.SetAwsAccessKeyIdSecretName(GetSecretName(settings, "aws_access_key_id_secret"));
aws.SetAwsSecretAccessKeySecretName(GetSecretName(settings, "aws_secret_access_key_secret"));
aws.SetAwsRegion(GetOrEmpty(settings, "aws_region"));
+ if (const auto status =
+ CheckOldSecretCreationAllowed(
+ disableOldSecretCreation, aws.GetAwsAccessKeyIdSecretName()); status.IsFail())
+ {
+ return status;
+ }
+ if (const auto status =
+ CheckOldSecretCreationAllowed(
+ disableOldSecretCreation, aws.GetAwsSecretAccessKeySecretName()); status.IsFail())
+ {
+ return status;
+ }
} else if (authMethod == "TOKEN") {
auto& token = *externalDataSourceDesc.MutableAuth()->MutableToken();
token.SetTokenSecretName(GetSecretName(settings, "token_secret"));
+ if (const auto status =
+ CheckOldSecretCreationAllowed(
+ disableOldSecretCreation, token.GetTokenSecretName()); status.IsFail())
+ {
+ return status;
+ }
} else if (authMethod == "IAM") {
auto& iam = *externalDataSourceDesc.MutableAuth()->MutableIam();
iam.SetServiceAccountId(GetOrEmpty(settings, "service_account_id"));
iam.SetInitialTokenSecretName(GetSecretName(settings, "initial_token_secret"));
// Note: user must not be allowed to specify resource_id;
// database authorization relies on resource_id lookup;
+ if (const auto status =
+ CheckOldSecretCreationAllowed(
+ disableOldSecretCreation, iam.GetInitialTokenSecretName()); status.IsFail())
+ {
+ return status;
+ }
} else {
return TYqlConclusionStatus::Fail(NYql::TIssuesIds::KIKIMR_INTERNAL_ERROR, TStringBuilder() << "Internal error. Unknown auth method: " << authMethod);
}
diff --git a/ydb/core/kqp/gateway/behaviour/external_data_source/ya.make b/ydb/core/kqp/gateway/behaviour/external_data_source/ya.make
index 7e5663a694b..03ad2926194 100644
--- a/ydb/core/kqp/gateway/behaviour/external_data_source/ya.make
+++ b/ydb/core/kqp/gateway/behaviour/external_data_source/ya.make
@@ -15,6 +15,7 @@ PEERDIR(
ydb/services/metadata/abstract
ydb/services/metadata/initializer
ydb/services/metadata/secret
+ ydb/services/scheme_secret
)
YQL_LAST_ABI_VERSION()
diff --git a/ydb/core/kqp/provider/yql_kikimr_exec.cpp b/ydb/core/kqp/provider/yql_kikimr_exec.cpp
index cfc69d14f9f..dd0dbee62b6 100644
--- a/ydb/core/kqp/provider/yql_kikimr_exec.cpp
+++ b/ydb/core/kqp/provider/yql_kikimr_exec.cpp
@@ -990,7 +990,7 @@ namespace {
bool ParseAsyncReplicationSettingsBase(
TReplicationSettingsBase& dstSettings, const TCoNameValueTupleList& srcSettings, TExprContext& ctx, TPositionHandle pos,
- const TString& objectName = "replication"
+ bool disableOldSecretCreation, const TString& objectName = "replication"
) {
for (auto setting : srcSettings) {
auto name = setting.Name().Value();
@@ -1089,14 +1089,40 @@ namespace {
return false;
}
+ if (disableOldSecretCreation) {
+ auto checkSecret = [&](const TString& secretName) {
+ if (secretName && !secretName.StartsWith('/')) {
+ ctx.AddError(
+ TIssue(
+ ctx.GetPosition(pos),
+ "Old secrets are disabled for creating new objects. Please use new secrets"
+ )
+ );
+ return false;
+ }
+ return true;
+ };
+
+ if (const auto& x = dstSettings.OAuthToken; x && !checkSecret(x->TokenSecretName)) {
+ return false;
+ }
+ if (const auto& x = dstSettings.StaticCredentials; x && !checkSecret(x->PasswordSecretName)) {
+ return false;
+ }
+ if (const auto& x = dstSettings.IamCredentials; x && !checkSecret(x->InitialToken.TokenSecretName)) {
+ return false;
+ }
+ }
+
return true;
}
bool ParseAsyncReplicationSettings(
- TReplicationSettings& dstSettings, const TCoNameValueTupleList& srcSettings, TExprContext& ctx, TPositionHandle pos
+ TReplicationSettings& dstSettings, const TCoNameValueTupleList& srcSettings, TExprContext& ctx, TPositionHandle pos,
+ bool disableOldSecretCreation
) {
- if (!ParseAsyncReplicationSettingsBase(dstSettings, srcSettings, ctx, pos)) {
+ if (!ParseAsyncReplicationSettingsBase(dstSettings, srcSettings, ctx, pos, disableOldSecretCreation)) {
return false;
}
@@ -1138,9 +1164,10 @@ namespace {
}
bool ParseTransferSettings(
- TTransferSettings& dstSettings, const TCoNameValueTupleList& srcSettings, TExprContext& ctx, TPositionHandle pos
+ TTransferSettings& dstSettings, const TCoNameValueTupleList& srcSettings, TExprContext& ctx, TPositionHandle pos,
+ bool disableOldSecretCreation
) {
- if (!ParseAsyncReplicationSettingsBase(dstSettings, srcSettings, ctx, pos, "transfer")) {
+ if (!ParseAsyncReplicationSettingsBase(dstSettings, srcSettings, ctx, pos, disableOldSecretCreation, "transfer")) {
return false;
}
@@ -3386,7 +3413,13 @@ public:
);
}
- if (!ParseAsyncReplicationSettings(settings.Settings, createReplication.ReplicationSettings(), ctx, createReplication.Pos())) {
+ if (!ParseAsyncReplicationSettings(
+ settings.Settings,
+ createReplication.ReplicationSettings(),
+ ctx,
+ createReplication.Pos(),
+ SessionCtx->Config().FeatureFlags.GetDisableOldSecretCreation())
+ ) {
return SyncError();
}
@@ -3442,7 +3475,10 @@ public:
TAlterReplicationSettings settings;
settings.Name = TString(alterReplication.Replication());
- if (!ParseAsyncReplicationSettings(settings.Settings, alterReplication.ReplicationSettings(), ctx, alterReplication.Pos())) {
+ if (!ParseAsyncReplicationSettings(
+ settings.Settings, alterReplication.ReplicationSettings(), ctx, alterReplication.Pos(),
+ SessionCtx->Config().FeatureFlags.GetDisableOldSecretCreation())
+ ) {
return SyncError();
}
@@ -3497,7 +3533,13 @@ public:
createTransfer.TransformLambda()
};
- if (!ParseTransferSettings(settings.Settings, createTransfer.TransferSettings(), ctx, createTransfer.Pos())) {
+ if (!ParseTransferSettings(
+ settings.Settings,
+ createTransfer.TransferSettings(),
+ ctx,
+ createTransfer.Pos(),
+ SessionCtx->Config().FeatureFlags.GetDisableOldSecretCreation())
+ ) {
return SyncError();
}
@@ -3554,7 +3596,9 @@ public:
settings.Name = TString(alterTransfer.Transfer());
settings.TranformLambda = alterTransfer.TransformLambda();
- if (!ParseTransferSettings(settings.Settings, alterTransfer.TransferSettings(), ctx, alterTransfer.Pos())) {
+ if (!ParseTransferSettings(settings.Settings, alterTransfer.TransferSettings(), ctx, alterTransfer.Pos(),
+ SessionCtx->Config().FeatureFlags.GetDisableOldSecretCreation())
+ ) {
return SyncError();
}
diff --git a/ydb/core/kqp/provider/yql_kikimr_opt_build.cpp b/ydb/core/kqp/provider/yql_kikimr_opt_build.cpp
index a3a598cd427..38e8b2f456d 100644
--- a/ydb/core/kqp/provider/yql_kikimr_opt_build.cpp
+++ b/ydb/core/kqp/provider/yql_kikimr_opt_build.cpp
@@ -486,7 +486,25 @@ bool ExploreNode(TExprBase node, TExprContext& ctx, const TKiDataSink& dataSink,
} else {
const auto& tableData = tablesData->ExistingTable(cluster, table);
YQL_ENSURE(tableData.Metadata);
- txRes.AddWriteOpToQueryBlock(node, tableData.Metadata->Name, tableData.Metadata->Indexes, tableOp & KikimrReadOps(), false, {});
+
+ bool needMainTableRead = bool(tableOp & KikimrReadOps());
+ if (!needMainTableRead && tableOp != TYdbOperation::Replace && !write.ReturningColumns().Empty()) {
+ auto inputColumnsSetting = GetSetting(write.Settings().Ref(), "input_columns");
+ YQL_ENSURE(inputColumnsSetting);
+ auto inputColumns = TCoNameValueTuple(inputColumnsSetting).Value().Cast<TCoAtomList>();
+ THashSet<TStringBuf> inputColumnsSet;
+ for (const auto& col : inputColumns) {
+ inputColumnsSet.insert(col.Value());
+ }
+ for (const auto& returnCol : write.ReturningColumns().Cast<TCoAtomList>()) {
+ if (!inputColumnsSet.contains(returnCol.Value())) {
+ needMainTableRead = true;
+ break;
+ }
+ }
+ }
+
+ txRes.AddWriteOpToQueryBlock(node, tableData.Metadata->Name, tableData.Metadata->Indexes, needMainTableRead, false, {});
}
if (!write.ReturningColumns().Empty()) {
@@ -577,7 +595,10 @@ bool ExploreNode(TExprBase node, TExprContext& ctx, const TKiDataSink& dataSink,
txRes.PrepareForResult();
}
- txRes.AddWriteOpToQueryBlock(node, tableData.Metadata->Name, tableData.Metadata->Indexes, tableOp & KikimrReadOps(), false, {});
+ const bool needMainTableRead = bool(tableOp & KikimrReadOps())
+ || !del.ReturningColumns().Empty(); // For RETURNING row existence must be checked.
+
+ txRes.AddWriteOpToQueryBlock(node, tableData.Metadata->Name, tableData.Metadata->Indexes, needMainTableRead, false, {});
if (!del.ReturningColumns().Empty()) {
txRes.AddResult(
Build<TResWrite>(ctx, del.Pos())
diff --git a/ydb/core/kqp/ut/federated_query/datastreams/streaming_ddl_ut.cpp b/ydb/core/kqp/ut/federated_query/datastreams/streaming_ddl_ut.cpp
index b512a370adb..5aae3622524 100644
--- a/ydb/core/kqp/ut/federated_query/datastreams/streaming_ddl_ut.cpp
+++ b/ydb/core/kqp/ut/federated_query/datastreams/streaming_ddl_ut.cpp
@@ -1366,10 +1366,12 @@ Y_UNIT_TEST_SUITE(KqpStreamingQueriesDdl) {
}
Y_UNIT_TEST_F(StreamingQueryWithPrecompute, TStreamingTestFixture) {
+ ExecQuery("GRANT ALL ON `/Root` TO `" BUILTIN_ACL_ROOT "`");
+
constexpr char inputTopicName[] = "streamingQueryWithPrecomputeInputTopic";
constexpr char outputTopicName[] = "streamingQueryWithPrecomputeOutputTopic";
constexpr char pqSourceName[] = "pqSourceName";
- CreateTopic(inputTopicName);
+ CreateTopic(inputTopicName, NTopic::TCreateTopicSettings().PartitioningSettings(2, 2));
CreateTopic(outputTopicName);
CreatePqSource(pqSourceName);
@@ -1415,6 +1417,25 @@ Y_UNIT_TEST_SUITE(KqpStreamingQueriesDdl) {
ReadTopicMessage(outputTopicName, "message-1-value-1");
Sleep(TDuration::Seconds(1)); // wait for checkpoint commit
+ const auto& result = ExecQuery("SELECT Plan, Ast FROM `.sys/streaming_queries`");
+ UNIT_ASSERT_VALUES_EQUAL(result.size(), 1);
+ CheckScriptResult(result[0], 2, 1, [&](TResultSetParser& resultSet) {
+ AstChecker(2, 3)(resultSet.ColumnParser("Ast").GetOptionalUtf8().value_or(""));
+
+ const auto planJson = resultSet.ColumnParser("Plan").GetOptionalUtf8().value_or("");
+ Cerr << "Plan: " << planJson << Endl;
+ NJson::TJsonValue plan;
+ UNIT_ASSERT(NJson::ReadJsonTree(planJson, &plan));
+
+ const auto& stagePlan = plan["Plan"]["Plans"][0]["Plans"][0];
+ UNIT_ASSERT_VALUES_EQUAL(stagePlan["Node Type"].GetStringSafe(), "Stage");
+ UNIT_ASSERT_VALUES_EQUAL(stagePlan["Stats"]["Tasks"].GetIntegerSafe(), 2);
+
+ const auto& sourceOp = stagePlan["Plans"][0]["Operators"].GetArraySafe()[0];
+ UNIT_ASSERT_VALUES_EQUAL(sourceOp["ExternalDataSource"].GetStringSafe(), pqSourceName);
+ UNIT_ASSERT_VALUES_EQUAL(sourceOp["SourceType"].GetStringSafe(), "pq");
+ });
+
ExecQuery(fmt::format(R"(
ALTER STREAMING QUERY `{query_name}` SET (
RUN = FALSE
diff --git a/ydb/core/kqp/ut/olap/indexes/indexes_ut.cpp b/ydb/core/kqp/ut/olap/indexes/indexes_ut.cpp
index f8f6820d4e0..a20e45cd0b4 100644
--- a/ydb/core/kqp/ut/olap/indexes/indexes_ut.cpp
+++ b/ydb/core/kqp/ut/olap/indexes/indexes_ut.cpp
@@ -1,4 +1,5 @@
+#include <ydb/core/base/counters.h>
#include <ydb/core/base/tablet_pipecache.h>
#include <ydb/core/formats/arrow/serializer/native.h>
#include <ydb/core/kqp/ut/common/olap_indexes_enums.h>
@@ -4122,6 +4123,38 @@ Y_UNIT_TEST(RenameLocalBloomIndex, EUseQueryService) {
SELECT COUNT(*) FROM `/Root/olapTableBloomWithDict` WHERE resource_id LIKE "alp%";
)"), "[[2u]]");
}
+
+ Y_UNIT_TEST(DataAndIndexBytesCounters) {
+ auto settings = TKikimrSettings().SetWithSampleTables(false).SetColumnShardAlterObjectEnabled(true);
+ TKikimrRunner kikimr(settings);
+
+ auto csController = NYDBTest::TControllers::RegisterCSControllerGuard<NYDBTest::NColumnShard::TController>();
+ csController->SetOverridePeriodicWakeupActivationPeriod(TDuration::Seconds(1));
+
+ auto helper = TLocalHelper(kikimr);
+ helper.CreateTestOlapTable();
+
+ // A MIN_MAX index gives compacted portions index blobs, so IndexBytes becomes non-zero.
+ ExecQuery(kikimr, false, R"(ALTER OBJECT `/Root/olapStore` (TYPE TABLESTORE) SET (ACTION=UPSERT_INDEX, NAME=index_uid, TYPE=MIN_MAX,
+ FEATURES=`{"column_name" : "uid"}`);)");
+
+ for (ui32 i = 0; i < 5; ++i) {
+ WriteTestData(kikimr, "/Root/olapStore/olapTable", 1000000 + i * 100000, 300000000 + i * 100000, 10000);
+ }
+
+ auto* runtime = kikimr.GetTestServer().GetRuntime();
+ auto appCounters = GetServiceCounters(runtime->GetAppData().Counters, "tablets")
+ ->GetSubgroup("type", "ColumnShard")
+ ->GetSubgroup("category", "app");
+ auto dataBytes = appCounters->GetCounter("SUM(ColumnShard/DataBytes)", false);
+ auto indexBytes = appCounters->GetCounter("SUM(ColumnShard/IndexBytes)", false);
+
+ // Index blobs are produced by async background compaction, and each shard's counters roll up into
+ // the SUM(...) aggregate sensor only periodically, so poll until both surface.
+ csController->WaitCondition(TDuration::Seconds(30), [&]() { return dataBytes->Val() > 0 && indexBytes->Val() > 0; });
+ UNIT_ASSERT_GT(dataBytes->Val(), 0);
+ UNIT_ASSERT_GT(indexBytes->Val(), 0);
+ }
}
} // namespace NKikimr::NKqp
diff --git a/ydb/core/kqp/ut/opt/kqp_concurrent_results_ut.cpp b/ydb/core/kqp/ut/opt/kqp_concurrent_results_ut.cpp
new file mode 100644
index 00000000000..31729e9fe71
--- /dev/null
+++ b/ydb/core/kqp/ut/opt/kqp_concurrent_results_ut.cpp
@@ -0,0 +1,571 @@
+#include <ydb/core/kqp/ut/common/kqp_ut_common.h>
+
+#include <ydb/public/lib/yson_value/ydb_yson_value.h>
+
+namespace NKikimr::NKqp {
+
+using namespace NYdb;
+using namespace NYdb::NQuery;
+
+inline NKikimrConfig::TAppConfig GetAppConfig(bool enableIndexStreamWrite) {
+ auto app = NKikimrConfig::TAppConfig();
+ app.MutableTableServiceConfig()->SetEnableIndexStreamWrite(enableIndexStreamWrite);
+ return app;
+}
+
+static TExecuteQuerySettings ConcurrentStreamSettings() {
+ return TExecuteQuerySettings()
+ .ConcurrentResultSets(true);
+}
+
+static TMap<ui64, TString> CollectConcurrentResults(TExecuteQueryIterator& it) {
+ TMap<ui64, TString> result;
+ TMap<ui64, TStringStream> streams;
+
+ for (;;) {
+ auto streamPart = it.ReadNext().GetValueSync();
+ if (!streamPart.IsSuccess()) {
+ UNIT_ASSERT_C(streamPart.EOS(), streamPart.GetIssues().ToString());
+ break;
+ }
+
+ if (streamPart.HasResultSet()) {
+ auto idx = streamPart.GetResultSetIndex();
+ auto resultSet = streamPart.ExtractResultSet();
+ NYson::TYsonWriter writer(&streams[idx], NYson::EYsonFormat::Text, ::NYson::EYsonType::Node, true);
+ NYdb::FormatResultSetYson(resultSet, writer);
+ }
+ }
+
+ for (auto& [idx, stream] : streams) {
+ result[idx] = stream.Str();
+ }
+
+ return result;
+}
+
+static void AssertConcurrentResult(const TMap<ui64, TString>& results, ui64 index, const TString& expectedYson) {
+ auto it = results.find(index);
+ UNIT_ASSERT_C(it != results.end(), TStringBuilder() << "result set index " << index << " not found");
+ CompareYson(expectedYson, it->second);
+}
+
+static void AssertConcurrentResultUnordered(const TMap<ui64, TString>& results, ui64 index, const TString& expectedYson) {
+ auto it = results.find(index);
+ UNIT_ASSERT_C(it != results.end(), TStringBuilder() << "result set index " << index << " not found");
+ CompareYsonUnordered(expectedYson, it->second);
+}
+
+static TString ReadTableViaQuery(TSession& session, const TString& table, const TString& columns, const TString& orderByColumn) {
+ auto query = TStringBuilder() << "SELECT " << columns << " FROM " << table << " ORDER BY " << orderByColumn << ";";
+ auto result = session.ExecuteQuery(query, TTxControl::BeginTx().CommitTx()).ExtractValueSync();
+ UNIT_ASSERT_VALUES_EQUAL_C(result.GetStatus(), EStatus::SUCCESS, result.GetIssues().ToString());
+ return FormatResultSetYson(result.GetResultSet(0));
+}
+
+static void ExecuteSchemeQuery(TSession& session, const TString& query) {
+ auto result = session.ExecuteQuery(query, TTxControl::NoTx()).ExtractValueSync();
+ UNIT_ASSERT_VALUES_EQUAL_C(result.GetStatus(), EStatus::SUCCESS, result.GetIssues().ToString());
+}
+
+static void ExecuteDataQuery(TSession& session, const TString& query) {
+ auto result = session.ExecuteQuery(query, TTxControl::BeginTx().CommitTx()).ExtractValueSync();
+ UNIT_ASSERT_VALUES_EQUAL_C(result.GetStatus(), EStatus::SUCCESS, result.GetIssues().ToString());
+}
+
+Y_UNIT_TEST_SUITE(KqpConcurrentResults) {
+
+Y_UNIT_TEST_TWIN(ConcurrentStreamSelectBetweenReturnings, EnableIndexStreamWrite) {
+ auto kikimr = DefaultKikimrRunner({}, GetAppConfig(EnableIndexStreamWrite));
+ auto db = kikimr.GetQueryClient();
+ auto session = db.GetSession().GetValueSync().GetSession();
+
+ ExecuteSchemeQuery(session, R"(
+ CREATE TABLE t1 (key Int32, val String, PRIMARY KEY(key));
+ CREATE TABLE t2 (key Int32, val String, PRIMARY KEY(key));
+ CREATE TABLE t3 (key Int32, val String, PRIMARY KEY(key));
+ )");
+
+ ExecuteDataQuery(session, R"(
+ INSERT INTO t1 (key, val) VALUES (1, "a");
+ INSERT INTO t2 (key, val) VALUES (10, "x");
+ INSERT INTO t3 (key, val) VALUES (20, "y");
+ )");
+
+ {
+ auto it = db.StreamExecuteQuery(R"(
+ UPSERT INTO t2 (key) VALUES (10) RETURNING key, val;
+ SELECT key, val FROM t1 ORDER BY key;
+ UPSERT INTO t3 (key) VALUES (20) RETURNING key, val;
+ )", TTxControl::BeginTx().CommitTx(), ConcurrentStreamSettings()).ExtractValueSync();
+ UNIT_ASSERT_VALUES_EQUAL_C(it.GetStatus(), EStatus::SUCCESS, it.GetIssues().ToString());
+
+ auto results = CollectConcurrentResults(it);
+ UNIT_ASSERT_VALUES_EQUAL(results.size(), 3u);
+ AssertConcurrentResult(results, 0, R"([[[10];["x"]]])");
+ AssertConcurrentResult(results, 1, R"([[[1];["a"]]])");
+ AssertConcurrentResult(results, 2, R"([[[20];["y"]]])");
+ }
+}
+
+Y_UNIT_TEST_TWIN(ConcurrentStreamMultipleSelectsAndReturnings, EnableIndexStreamWrite) {
+ auto kikimr = DefaultKikimrRunner({}, GetAppConfig(EnableIndexStreamWrite));
+ auto db = kikimr.GetQueryClient();
+ auto session = db.GetSession().GetValueSync().GetSession();
+
+ ExecuteSchemeQuery(session, R"(
+ CREATE TABLE t1 (key Int32, val String, PRIMARY KEY(key));
+ CREATE TABLE t2 (key Int32, val String, PRIMARY KEY(key));
+ CREATE TABLE t3 (key Int32, val String, PRIMARY KEY(key));
+ )");
+
+ ExecuteDataQuery(session, R"(
+ INSERT INTO t1 (key, val) VALUES (10, "x");
+ INSERT INTO t2 (key, val) VALUES (20, "y");
+ INSERT INTO t3 (key, val) VALUES (1, "a");
+ )");
+
+ {
+ auto it = db.StreamExecuteQuery(R"(
+ SELECT * FROM t3;
+ UPSERT INTO t1 (key) VALUES (10) RETURNING key, val;
+ SELECT * FROM t3;
+ UPSERT INTO t2 (key) VALUES (20) RETURNING key, val;
+ SELECT * FROM t3;
+ )", TTxControl::BeginTx().CommitTx(), ConcurrentStreamSettings()).ExtractValueSync();
+ UNIT_ASSERT_VALUES_EQUAL_C(it.GetStatus(), EStatus::SUCCESS, it.GetIssues().ToString());
+
+ auto results = CollectConcurrentResults(it);
+ UNIT_ASSERT_VALUES_EQUAL(results.size(), 5u);
+ AssertConcurrentResult(results, 0, R"([[[1];["a"]]])");
+ AssertConcurrentResult(results, 1, R"([[[10];["x"]]])");
+ AssertConcurrentResult(results, 2, R"([[[1];["a"]]])");
+ AssertConcurrentResult(results, 3, R"([[[20];["y"]]])");
+ AssertConcurrentResult(results, 4, R"([[[1];["a"]]])");
+ }
+}
+
+Y_UNIT_TEST_TWIN(ConcurrentStreamSameTableTwoReturnings, EnableIndexStreamWrite) {
+ auto kikimr = DefaultKikimrRunner({}, GetAppConfig(EnableIndexStreamWrite));
+ auto db = kikimr.GetQueryClient();
+ auto session = db.GetSession().GetValueSync().GetSession();
+
+ ExecuteSchemeQuery(session, R"(
+ CREATE TABLE t (key Int32, version Int32, val String, PRIMARY KEY(key));
+ )");
+
+ ExecuteDataQuery(session, R"(
+ INSERT INTO t (key, version, val) VALUES (1, 1, "first");
+ )");
+
+ {
+ auto it = db.StreamExecuteQuery(R"(
+ UPSERT INTO t (key, version) VALUES (1, 2) RETURNING key, version, val;
+ UPSERT INTO t (key, version) VALUES (1, 3) RETURNING key, version, val;
+ )", TTxControl::BeginTx().CommitTx(), ConcurrentStreamSettings()).ExtractValueSync();
+ UNIT_ASSERT_VALUES_EQUAL_C(it.GetStatus(), EStatus::SUCCESS, it.GetIssues().ToString());
+
+ auto results = CollectConcurrentResults(it);
+ UNIT_ASSERT_VALUES_EQUAL(results.size(), 2u);
+ AssertConcurrentResult(results, 0, R"([[[1];[2];["first"]]])");
+ AssertConcurrentResult(results, 1, R"([[[1];[3];["first"]]])");
+ }
+
+ CompareYson(R"([[[1];[3];["first"]]])", ReadTableViaQuery(session, "t", "key, version, val", "key"));
+}
+
+Y_UNIT_TEST_TWIN(ConcurrentStreamSameTableBlindWriteThenReturning, EnableIndexStreamWrite) {
+ auto kikimr = DefaultKikimrRunner({}, GetAppConfig(EnableIndexStreamWrite));
+ auto db = kikimr.GetQueryClient();
+ auto session = db.GetSession().GetValueSync().GetSession();
+
+ ExecuteSchemeQuery(session, R"(
+ CREATE TABLE t (key Int32, val String, PRIMARY KEY(key));
+ )");
+
+ ExecuteDataQuery(session, R"(
+ INSERT INTO t (key, val) VALUES (1, "existing");
+ )");
+
+ {
+ auto it = db.StreamExecuteQuery(R"(
+ UPSERT INTO t (key, val) VALUES (1, "overwritten");
+ UPSERT INTO t (key) VALUES (1) RETURNING key, val;
+ )", TTxControl::BeginTx().CommitTx(), ConcurrentStreamSettings()).ExtractValueSync();
+ UNIT_ASSERT_VALUES_EQUAL_C(it.GetStatus(), EStatus::SUCCESS, it.GetIssues().ToString());
+
+ auto results = CollectConcurrentResults(it);
+ UNIT_ASSERT_VALUES_EQUAL(results.size(), 1u);
+ AssertConcurrentResult(results, 0, R"([[[1];["overwritten"]]])");
+ }
+
+ CompareYson(R"([[[1];["overwritten"]]])", ReadTableViaQuery(session, "t", "key, val", "key"));
+}
+
+Y_UNIT_TEST_TWIN(ConcurrentStreamReturningThenReadSameTable, EnableIndexStreamWrite) {
+ auto kikimr = DefaultKikimrRunner({}, GetAppConfig(EnableIndexStreamWrite));
+ auto db = kikimr.GetQueryClient();
+ auto session = db.GetSession().GetValueSync().GetSession();
+
+ ExecuteSchemeQuery(session, R"(
+ CREATE TABLE t1 (key Int32, val String, PRIMARY KEY(key));
+ )");
+
+ ExecuteDataQuery(session, R"(
+ INSERT INTO t1 (key, val) VALUES (1, "original");
+ )");
+
+ {
+ auto it = db.StreamExecuteQuery(R"(
+ UPSERT INTO t1 (key, val) VALUES (1, "updated") RETURNING key, val;
+ SELECT val FROM t1 WHERE key = 1;
+ )", TTxControl::BeginTx().CommitTx(), ConcurrentStreamSettings()).ExtractValueSync();
+ UNIT_ASSERT_VALUES_EQUAL_C(it.GetStatus(), EStatus::SUCCESS, it.GetIssues().ToString());
+
+ auto results = CollectConcurrentResults(it);
+ UNIT_ASSERT_VALUES_EQUAL(results.size(), 2u);
+ AssertConcurrentResult(results, 0, R"([[[1];["updated"]]])");
+ AssertConcurrentResult(results, 1, R"([[["updated"]]])");
+ }
+}
+
+Y_UNIT_TEST_TWIN(ConcurrentStreamReturningWithIndex, EnableIndexStreamWrite) {
+ auto kikimr = DefaultKikimrRunner({}, GetAppConfig(EnableIndexStreamWrite));
+ auto db = kikimr.GetQueryClient();
+ auto session = db.GetSession().GetValueSync().GetSession();
+
+ ExecuteSchemeQuery(session, R"(
+ CREATE TABLE t (c0 Int64, c1 Int64, c2 Int64, PRIMARY KEY(c0),
+ INDEX idx GLOBAL SYNC ON (c2));
+ )");
+
+ ExecuteDataQuery(session, R"(
+ INSERT INTO t (c0, c1, c2) VALUES (1, 10, 20);
+ )");
+
+ {
+ auto it = db.StreamExecuteQuery(R"(
+ UPSERT INTO t (c0) VALUES (1) RETURNING c0, c1, c2;
+ )", TTxControl::BeginTx().CommitTx(), ConcurrentStreamSettings()).ExtractValueSync();
+ UNIT_ASSERT_VALUES_EQUAL_C(it.GetStatus(), EStatus::SUCCESS, it.GetIssues().ToString());
+
+ auto results = CollectConcurrentResults(it);
+ UNIT_ASSERT_VALUES_EQUAL(results.size(), 1u);
+ AssertConcurrentResult(results, 0, R"([[[1];[10];[20]]])");
+ }
+
+ CompareYson(R"([[[1];[10];[20]]])", ReadTableViaQuery(session, "t", "c0, c1, c2", "c0"));
+
+ ExecuteDataQuery(session, R"(
+ UPSERT INTO t (c0, c1, c2) VALUES (1, 100, 200);
+ )");
+
+ {
+ auto it = db.StreamExecuteQuery(R"(
+ $data = SELECT c0, c1, c2 FROM t WHERE c2 = 200 ORDER BY c0;
+ UPSERT INTO t SELECT c0, (c1 + 1) AS c1, c2 FROM $data RETURNING c0, c1, c2;
+ SELECT c0, c1, c2 FROM $data;
+ )", TTxControl::BeginTx().CommitTx(), ConcurrentStreamSettings()).ExtractValueSync();
+ UNIT_ASSERT_VALUES_EQUAL_C(it.GetStatus(), EStatus::SUCCESS, it.GetIssues().ToString());
+
+ auto results = CollectConcurrentResults(it);
+ UNIT_ASSERT_VALUES_EQUAL(results.size(), 2u);
+ AssertConcurrentResult(results, 0, R"([[[1];[101];[200]]])");
+ AssertConcurrentResult(results, 1, R"([[[1];[100];[200]]])");
+ }
+}
+
+Y_UNIT_TEST_TWIN(ConcurrentStreamNamedExprSharedByWriteAndSelect, EnableIndexStreamWrite) {
+ NKikimrConfig::TAppConfig app;
+ app.MutableTableServiceConfig()->SetEnableIndexStreamWrite(EnableIndexStreamWrite);
+ auto settings = TKikimrSettings(app).SetWithSampleTables(false);
+ TKikimrRunner kikimr(settings);
+ auto db = kikimr.GetQueryClient();
+ auto session = db.GetSession().GetValueSync().GetSession();
+
+ ExecuteSchemeQuery(session, R"(
+ CREATE TABLE Source (Key String, Value String, PRIMARY KEY(Key));
+ CREATE TABLE Dest (Key String, Value String, PRIMARY KEY(Key));
+ )");
+
+ ExecuteDataQuery(session, R"(
+ INSERT INTO Source (Key, Value) VALUES ("1", "a"), ("2", "b");
+ INSERT INTO Dest (Key, Value) VALUES ("1", "c"), ("2", "d");
+ )");
+
+ {
+ auto it = db.StreamExecuteQuery(R"(
+ $rows = SELECT Key, Value FROM Source ORDER BY Key;
+ UPSERT INTO Dest (Key) SELECT Key FROM $rows RETURNING Key, Value;
+ SELECT Key, Value FROM $rows;
+ SELECT Key, Value FROM Dest;
+ )", TTxControl::BeginTx().CommitTx(), ConcurrentStreamSettings()).ExtractValueSync();
+ UNIT_ASSERT_VALUES_EQUAL_C(it.GetStatus(), EStatus::SUCCESS, it.GetIssues().ToString());
+
+ auto results = CollectConcurrentResults(it);
+ UNIT_ASSERT_VALUES_EQUAL(results.size(), 3u);
+ AssertConcurrentResultUnordered(results, 0, R"([[["1"];["c"]];[["2"];["d"]]])");
+ AssertConcurrentResult(results, 1, R"([[["1"];["a"]];[["2"];["b"]]])");
+ AssertConcurrentResult(results, 2, R"([[["1"];["c"]];[["2"];["d"]]])");
+ }
+}
+
+Y_UNIT_TEST_TWIN(ConcurrentStreamNamedExprRandomConsistency, EnableIndexStreamWrite) {
+ NKikimrConfig::TAppConfig app;
+ app.MutableTableServiceConfig()->SetEnableIndexStreamWrite(EnableIndexStreamWrite);
+ auto settings = TKikimrSettings(app).SetWithSampleTables(false);
+ TKikimrRunner kikimr(settings);
+ auto db = kikimr.GetQueryClient();
+ auto session = db.GetSession().GetValueSync().GetSession();
+
+ ExecuteSchemeQuery(session, R"(
+ CREATE TABLE Source (Key String, Value String, PRIMARY KEY(Key));
+ CREATE TABLE Dest1 (Key String, Value String, PRIMARY KEY(Key));
+ CREATE TABLE Dest2 (Key String, Value String, PRIMARY KEY(Key));
+ )");
+
+ ExecuteDataQuery(session, R"(
+ INSERT INTO Source (Key, Value) VALUES ("1", "");
+ )");
+
+ {
+ auto it = db.StreamExecuteQuery(R"(
+ $rows = SELECT Key, CAST(RandomUuid(Key) AS String) AS Value FROM Source;
+ UPSERT INTO Dest1 SELECT * FROM $rows RETURNING Value;
+ UPSERT INTO Dest2 SELECT * FROM $rows RETURNING Value;
+ )", TTxControl::BeginTx().CommitTx(), ConcurrentStreamSettings()).ExtractValueSync();
+ UNIT_ASSERT_VALUES_EQUAL_C(it.GetStatus(), EStatus::SUCCESS, it.GetIssues().ToString());
+
+ auto results = CollectConcurrentResults(it);
+ UNIT_ASSERT_VALUES_EQUAL(results.size(), 2u);
+ UNIT_ASSERT_VALUES_EQUAL(results.at(0), results.at(1));
+ }
+}
+
+Y_UNIT_TEST_TWIN(ConcurrentStreamSameTableBlindWriteThenReturningUpsert, EnableIndexStreamWrite) {
+ auto kikimr = DefaultKikimrRunner({}, GetAppConfig(EnableIndexStreamWrite));
+ auto db = kikimr.GetQueryClient();
+ auto session = db.GetSession().GetValueSync().GetSession();
+
+ ExecuteSchemeQuery(session, R"(
+ CREATE TABLE t (key Int32, val String, PRIMARY KEY(key));
+ )");
+
+ ExecuteDataQuery(session, R"(
+ INSERT INTO t (key, val) VALUES (1, "existing");
+ )");
+
+ {
+ auto it = db.StreamExecuteQuery(R"(
+ UPSERT INTO t (key, val) VALUES (1, "overwritten");
+ UPSERT INTO t (key) VALUES (1) RETURNING key, val;
+ )", TTxControl::BeginTx().CommitTx(), ConcurrentStreamSettings()).ExtractValueSync();
+ UNIT_ASSERT_VALUES_EQUAL_C(it.GetStatus(), EStatus::SUCCESS, it.GetIssues().ToString());
+
+ auto results = CollectConcurrentResults(it);
+ UNIT_ASSERT_VALUES_EQUAL(results.size(), 1u);
+ AssertConcurrentResult(results, 0, R"([[[1];["overwritten"]]])");
+ }
+
+ CompareYson(R"([[[1];["overwritten"]]])", ReadTableViaQuery(session, "t", "key, val", "key"));
+}
+
+Y_UNIT_TEST_TWIN(ConcurrentStreamSameTableBlindWriteThenReturningReplace, EnableIndexStreamWrite) {
+ auto kikimr = DefaultKikimrRunner({}, GetAppConfig(EnableIndexStreamWrite));
+ auto db = kikimr.GetQueryClient();
+ auto session = db.GetSession().GetValueSync().GetSession();
+
+ ExecuteSchemeQuery(session, R"(
+ CREATE TABLE t (key Int32, val String, val2 String, PRIMARY KEY(key));
+ )");
+
+ ExecuteDataQuery(session, R"(
+ INSERT INTO t (key, val, val2) VALUES (1, "first", "second");
+ )");
+
+ {
+ auto it = db.StreamExecuteQuery(R"(
+ REPLACE INTO t (key, val) VALUES (1, "updated");
+ REPLACE INTO t (key, val) VALUES (1, "final") RETURNING key, val, val2;
+ )", TTxControl::BeginTx().CommitTx(), ConcurrentStreamSettings()).ExtractValueSync();
+ UNIT_ASSERT_VALUES_EQUAL_C(it.GetStatus(), EStatus::SUCCESS, it.GetIssues().ToString());
+
+ auto results = CollectConcurrentResults(it);
+ UNIT_ASSERT_VALUES_EQUAL(results.size(), 1u);
+ AssertConcurrentResult(results, 0, R"([[[1];["final"];#]])");
+ }
+
+ CompareYson(R"([[[1];["final"];#]])", ReadTableViaQuery(session, "t", "key, val, val2", "key"));
+}
+
+Y_UNIT_TEST_TWIN(ConcurrentStreamSameTableBlindWriteThenReturningUpdate, EnableIndexStreamWrite) {
+ auto kikimr = DefaultKikimrRunner({}, GetAppConfig(EnableIndexStreamWrite));
+ auto db = kikimr.GetQueryClient();
+ auto session = db.GetSession().GetValueSync().GetSession();
+
+ ExecuteSchemeQuery(session, R"(
+ CREATE TABLE t (key Int32, val String, PRIMARY KEY(key));
+ )");
+
+ ExecuteDataQuery(session, R"(
+ INSERT INTO t (key, val) VALUES (1, "existing");
+ )");
+
+ {
+ auto it = db.StreamExecuteQuery(R"(
+ UPSERT INTO t (key, val) VALUES (1, "overwritten");
+ UPDATE t SET val = "updated" WHERE key = 1 RETURNING key, val;
+ )", TTxControl::BeginTx().CommitTx(), ConcurrentStreamSettings()).ExtractValueSync();
+ UNIT_ASSERT_VALUES_EQUAL_C(it.GetStatus(), EStatus::SUCCESS, it.GetIssues().ToString());
+
+ auto results = CollectConcurrentResults(it);
+ UNIT_ASSERT_VALUES_EQUAL(results.size(), 1u);
+ AssertConcurrentResult(results, 0, R"([[[1];["updated"]]])");
+ }
+
+ CompareYson(R"([[[1];["updated"]]])", ReadTableViaQuery(session, "t", "key, val", "key"));
+}
+
+Y_UNIT_TEST_TWIN(ConcurrentStreamSameTableBlindWriteThenReturningDelete, EnableIndexStreamWrite) {
+ auto kikimr = DefaultKikimrRunner({}, GetAppConfig(EnableIndexStreamWrite));
+ auto db = kikimr.GetQueryClient();
+ auto session = db.GetSession().GetValueSync().GetSession();
+
+ ExecuteSchemeQuery(session, R"(
+ CREATE TABLE t (key Int32, val String, PRIMARY KEY(key));
+ )");
+
+ ExecuteDataQuery(session, R"(
+ INSERT INTO t (key, val) VALUES (1, "existing");
+ )");
+
+ {
+ auto it = db.StreamExecuteQuery(R"(
+ UPSERT INTO t (key, val) VALUES (1, "overwritten");
+ DELETE FROM t WHERE key = 1 RETURNING key, val;
+ )", TTxControl::BeginTx().CommitTx(), ConcurrentStreamSettings()).ExtractValueSync();
+ UNIT_ASSERT_VALUES_EQUAL_C(it.GetStatus(), EStatus::SUCCESS, it.GetIssues().ToString());
+
+ auto results = CollectConcurrentResults(it);
+ UNIT_ASSERT_VALUES_EQUAL(results.size(), 1u);
+ AssertConcurrentResult(results, 0, R"([[[1];["overwritten"]]])");
+ }
+
+ CompareYson(R"([])", ReadTableViaQuery(session, "t", "key, val", "key"));
+}
+
+Y_UNIT_TEST_TWIN(ConcurrentStreamSameTableBlindWriteThenReturningUpdateOn, EnableIndexStreamWrite) {
+ auto kikimr = DefaultKikimrRunner({}, GetAppConfig(EnableIndexStreamWrite));
+ auto db = kikimr.GetQueryClient();
+ auto session = db.GetSession().GetValueSync().GetSession();
+
+ ExecuteSchemeQuery(session, R"(
+ CREATE TABLE t (key Int32, val String, PRIMARY KEY(key));
+ )");
+
+ ExecuteDataQuery(session, R"(
+ INSERT INTO t (key, val) VALUES (1, "existing");
+ )");
+
+ {
+ auto it = db.StreamExecuteQuery(R"(
+ UPSERT INTO t (key, val) VALUES (1, "overwritten");
+ UPDATE t ON (key, val) VALUES (1, "updated") RETURNING key, val;
+ )", TTxControl::BeginTx().CommitTx(), ConcurrentStreamSettings()).ExtractValueSync();
+ UNIT_ASSERT_VALUES_EQUAL_C(it.GetStatus(), EStatus::SUCCESS, it.GetIssues().ToString());
+
+ auto results = CollectConcurrentResults(it);
+ UNIT_ASSERT_VALUES_EQUAL(results.size(), 1u);
+ AssertConcurrentResult(results, 0, R"([[[1];["updated"]]])");
+ }
+
+ CompareYson(R"([[[1];["updated"]]])", ReadTableViaQuery(session, "t", "key, val", "key"));
+}
+
+Y_UNIT_TEST_TWIN(ConcurrentStreamSameTableBlindWriteThenReturningAllColumnsInInput, EnableIndexStreamWrite) {
+ auto kikimr = DefaultKikimrRunner({}, GetAppConfig(EnableIndexStreamWrite));
+ auto db = kikimr.GetQueryClient();
+ auto session = db.GetSession().GetValueSync().GetSession();
+
+ ExecuteSchemeQuery(session, R"(
+ CREATE TABLE t (key Int32, val String, PRIMARY KEY(key));
+ )");
+
+ ExecuteDataQuery(session, R"(
+ INSERT INTO t (key, val) VALUES (1, "existing");
+ )");
+
+ {
+ auto it = db.StreamExecuteQuery(R"(
+ UPSERT INTO t (key, val) VALUES (1, "first");
+ UPSERT INTO t (key, val) VALUES (1, "overwritten") RETURNING key, val;
+ )", TTxControl::BeginTx().CommitTx(), ConcurrentStreamSettings()).ExtractValueSync();
+ UNIT_ASSERT_VALUES_EQUAL_C(it.GetStatus(), EStatus::SUCCESS, it.GetIssues().ToString());
+
+ auto results = CollectConcurrentResults(it);
+ UNIT_ASSERT_VALUES_EQUAL(results.size(), 1u);
+ AssertConcurrentResult(results, 0, R"([[[1];["overwritten"]]])");
+ }
+
+ CompareYson(R"([[[1];["overwritten"]]])", ReadTableViaQuery(session, "t", "key, val", "key"));
+}
+
+Y_UNIT_TEST_TWIN(ConcurrentStreamSameTableBlindWriteThenReturningDeleteOn, EnableIndexStreamWrite) {
+ auto kikimr = DefaultKikimrRunner({}, GetAppConfig(EnableIndexStreamWrite));
+ auto db = kikimr.GetQueryClient();
+ auto session = db.GetSession().GetValueSync().GetSession();
+
+ ExecuteSchemeQuery(session, R"(
+ CREATE TABLE t (key Int32, val String, PRIMARY KEY(key));
+ )");
+
+ ExecuteDataQuery(session, R"(
+ INSERT INTO t (key, val) VALUES (1, "existing");
+ )");
+
+ {
+ auto it = db.StreamExecuteQuery(R"(
+ UPSERT INTO t (key, val) VALUES (1, "overwritten");
+ DELETE FROM t ON (key) VALUES (1) RETURNING key, val;
+ )", TTxControl::BeginTx().CommitTx(), ConcurrentStreamSettings()).ExtractValueSync();
+ UNIT_ASSERT_VALUES_EQUAL_C(it.GetStatus(), EStatus::SUCCESS, it.GetIssues().ToString());
+
+ auto results = CollectConcurrentResults(it);
+ UNIT_ASSERT_VALUES_EQUAL(results.size(), 1u);
+ AssertConcurrentResult(results, 0, R"([[[1];["overwritten"]]])");
+ }
+
+ CompareYson(R"([])", ReadTableViaQuery(session, "t", "key, val", "key"));
+}
+
+Y_UNIT_TEST_TWIN(ConcurrentStreamSameTableBlindWriteThenReturningInsert, EnableIndexStreamWrite) {
+ auto kikimr = DefaultKikimrRunner({}, GetAppConfig(EnableIndexStreamWrite));
+ auto db = kikimr.GetQueryClient();
+ auto session = db.GetSession().GetValueSync().GetSession();
+
+ ExecuteSchemeQuery(session, R"(
+ CREATE TABLE t (key Int32, val String, val2 String, PRIMARY KEY(key));
+ )");
+
+ ExecuteDataQuery(session, R"(
+ INSERT INTO t (key, val, val2) VALUES (1, "existing", "other");
+ )");
+
+ {
+ auto it = db.StreamExecuteQuery(R"(
+ UPSERT INTO t (key, val, val2) VALUES (1, "overwritten", "other2");
+ INSERT INTO t (key, val) VALUES (2, "new") RETURNING key, val, val2;
+ )", TTxControl::BeginTx().CommitTx(), ConcurrentStreamSettings()).ExtractValueSync();
+ UNIT_ASSERT_VALUES_EQUAL_C(it.GetStatus(), EStatus::SUCCESS, it.GetIssues().ToString());
+
+ auto results = CollectConcurrentResults(it);
+ UNIT_ASSERT_VALUES_EQUAL(results.size(), 1u);
+ AssertConcurrentResult(results, 0, R"([[[2];["new"];#]])");
+ }
+
+ CompareYsonUnordered(R"([[[1];["overwritten"];["other2"]];[[2];["new"];#]])", ReadTableViaQuery(session, "t", "key, val, val2", "key"));
+}
+
+} // Y_UNIT_TEST_SUITE(KqpConcurrentResults)
+
+} // namespace NKikimr::NKqp
diff --git a/ydb/core/kqp/ut/opt/ya.make b/ydb/core/kqp/ut/opt/ya.make
index 384afc9f7e5..f8af28139ef 100644
--- a/ydb/core/kqp/ut/opt/ya.make
+++ b/ydb/core/kqp/ut/opt/ya.make
@@ -8,6 +8,7 @@ SIZE(MEDIUM)
SRCS(
kqp_agg_ut.cpp
+ kqp_concurrent_results_ut.cpp
kqp_extract_predicate_unpack_ut.cpp
kqp_hash_combine_ut.cpp
kqp_kv_ut.cpp
diff --git a/ydb/core/kqp/ut/scheme/kqp_scheme_ut.cpp b/ydb/core/kqp/ut/scheme/kqp_scheme_ut.cpp
index 6d6f6fcd192..420a6111383 100644
--- a/ydb/core/kqp/ut/scheme/kqp_scheme_ut.cpp
+++ b/ydb/core/kqp/ut/scheme/kqp_scheme_ut.cpp
@@ -12940,7 +12940,7 @@ Y_UNIT_TEST_SUITE(KqpScheme) {
RANK
);)").GetValueSync();
UNIT_ASSERT_VALUES_EQUAL(result.GetStatus(), EStatus::GENERIC_ERROR);
- UNIT_ASSERT_STRING_CONTAINS_C(result.GetIssues().ToString(), "The rank could not be set automatically, the maximum rank of the resource pool classifier is too high: 9223372036854775807", result.GetIssues().ToString());
+ UNIT_ASSERT_STRING_CONTAINS_C(result.GetIssues().ToString(), "Cannot reset property rank", result.GetIssues().ToString());
}
TString FetchResourcePoolClassifiers(TTestActorRuntime& runtime, ui32 nodeIndex) {
@@ -13117,11 +13117,11 @@ Y_UNIT_TEST_SUITE(KqpScheme) {
{
auto query = R"(
ALTER RESOURCE POOL CLASSIFIER MyResourcePoolClassifier
- RESET (RANK, MEMBER_NAME);
+ RESET (MEMBER_NAME);
)";
auto result = session.ExecuteSchemeQuery(query).GetValueSync();
UNIT_ASSERT_VALUES_EQUAL_C(result.GetStatus(), EStatus::SUCCESS, result.GetIssues().ToString());
- UNIT_ASSERT_VALUES_EQUAL(FetchResourcePoolClassifiers(kikimr), "{\"resource_pool_classifiers\":[{\"rank\":1042,\"name\":\"MyResourcePoolClassifier\",\"config\":{\"member_name\":\"\",\"resource_pool\":\"test_pool\"},\"database\":\"\\/Root\"},{\"rank\":42,\"name\":\"AnotherResourcePoolClassifier\",\"config\":{\"resource_pool\":\"test_pool\"},\"database\":\"\\/Root\"}]}");
+ UNIT_ASSERT_VALUES_EQUAL(FetchResourcePoolClassifiers(kikimr), "{\"resource_pool_classifiers\":[{\"rank\":20,\"name\":\"MyResourcePoolClassifier\",\"config\":{\"member_name\":\"\",\"resource_pool\":\"test_pool\"},\"database\":\"\\/Root\"},{\"rank\":42,\"name\":\"AnotherResourcePoolClassifier\",\"config\":{\"resource_pool\":\"test_pool\"},\"database\":\"\\/Root\"}]}");
}
}
@@ -15065,6 +15065,37 @@ END DO)",
}
}
+ Y_UNIT_TEST(CreateExternalDataSourceWithOldSecretDisabled) {
+ NKikimrConfig::TFeatureFlags featureFlags;
+ featureFlags.SetEnableExternalDataSources(true);
+ featureFlags.SetDisableOldSecretCreation(true);
+ featureFlags.SetDisableOldSecrets(true);
+
+ NKqp::TKikimrSettings settings;
+ settings.SetFeatureFlags(featureFlags);
+ settings.AppConfig.MutableQueryServiceConfig()->AddAvailableExternalDataSources("ObjectStorage");
+ TKikimrRunner kikimr(settings);
+
+ auto db = kikimr.GetTableClient();
+ auto session = db.CreateSession().GetValueSync().GetSession();
+
+ static const auto query = R"sql(
+ CREATE EXTERNAL DATA SOURCE `/Root/ExternalDataSource` WITH (
+ SOURCE_TYPE="ObjectStorage",
+ LOCATION="my-bucket",
+ AUTH_METHOD="SERVICE_ACCOUNT",
+ SERVICE_ACCOUNT_ID="mysa",
+ SERVICE_ACCOUNT_SECRET_NAME="OldSecret"
+ );
+ )sql";
+ const auto result = session.ExecuteSchemeQuery(query).GetValueSync();
+ UNIT_ASSERT_VALUES_EQUAL_C(result.GetStatus(), EStatus::BAD_REQUEST, result.GetIssues().ToString());
+ UNIT_ASSERT_STRING_CONTAINS_C(
+ result.GetIssues().ToString(),
+ "Old secrets are disabled for creating new objects. Please use new secrets",
+ result.GetIssues().ToString());
+ }
+
Y_UNIT_TEST(SimpleTruncateTableFullPathTableClient) {
TestTruncateTable("`/Root/TestTable`", false);
}
diff --git a/ydb/core/mind/hive/hive.cpp b/ydb/core/mind/hive/hive.cpp
index 900efc3527c..a7b2a9b161e 100644
--- a/ydb/core/mind/hive/hive.cpp
+++ b/ydb/core/mind/hive/hive.cpp
@@ -207,5 +207,9 @@ const std::unordered_map<TTabletTypes::EType, TString> TABLET_TYPE_SHORT_NAMES =
const std::unordered_map<TString, TTabletTypes::EType> TABLET_TYPE_BY_SHORT_NAME = MakeReverseMap(TABLET_TYPE_SHORT_NAMES);
+TFullTabletId ToFullTabletId(TTabletId tabletId) {
+ return {tabletId, 0};
+}
+
} // NHive
} // NKikimr
diff --git a/ydb/core/mind/hive/hive.h b/ydb/core/mind/hive/hive.h
index ad72e02e678..8c4fab48f50 100644
--- a/ydb/core/mind/hive/hive.h
+++ b/ydb/core/mind/hive/hive.h
@@ -435,6 +435,8 @@ struct TReassignOperation {
}
};
+TFullTabletId ToFullTabletId(TTabletId tabletId);
+
} // NHive
} // NKikimr
diff --git a/ydb/core/mind/hive/hive_events.h b/ydb/core/mind/hive/hive_events.h
index 340944c393f..f95e0405ba4 100644
--- a/ydb/core/mind/hive/hive_events.h
+++ b/ydb/core/mind/hive/hive_events.h
@@ -41,7 +41,7 @@ struct TEvPrivate {
EvUpdateBalanceCounters,
EvProcessTabletMetrics,
EvReassignInactiveGroupsComplete,
- EvCompactComplete,
+ EvMoveDataComplete,
EvEnd
};
@@ -157,11 +157,11 @@ struct TEvPrivate {
TEvReassignInactiveGroupsComplete(const TString& poolName) : PoolName(poolName) {};
};
- struct TEvCompactComplete : TEventLocal<TEvCompactComplete, EvCompactComplete> {
+ struct TEvMoveDataComplete : TEventLocal<TEvMoveDataComplete, EvMoveDataComplete> {
TString PoolName;
bool Success;
- TEvCompactComplete(const TString& poolName, bool success) : PoolName(poolName), Success(success) {}
+ TEvMoveDataComplete(const TString& poolName, bool success) : PoolName(poolName), Success(success) {}
};
};
diff --git a/ydb/core/mind/hive/hive_impl.cpp b/ydb/core/mind/hive/hive_impl.cpp
index e652fe28b8c..e09b79afad3 100644
--- a/ydb/core/mind/hive/hive_impl.cpp
+++ b/ydb/core/mind/hive/hive_impl.cpp
@@ -551,6 +551,9 @@ TVector<TTabletId> THive::UpdateStoragePools(const google::protobuf::RepeatedPtr
for (TTabletId tabletId : tabletsWaiting) {
tabletsToUpdate.emplace_back(tabletId);
}
+ if (storagePool.ShrinkRequest) {
+ Execute(CreateShrinkPool(std::move(storagePool.ShrinkRequest)));
+ }
}
return tabletsToUpdate;
}
@@ -3697,7 +3700,7 @@ void THive::ProcessEvent(std::unique_ptr<IEventHandle> event) {
hFunc(TEvHive::TEvShrinkStoragePoolReply, Handle);
hFunc(TEvHive::TEvShrinkStoragePoolDone, Handle);
hFunc(TEvPrivate::TEvReassignInactiveGroupsComplete, Handle);
- hFunc(TEvPrivate::TEvCompactComplete, Handle);
+ hFunc(TEvPrivate::TEvMoveDataComplete, Handle);
}
}
@@ -3817,7 +3820,7 @@ STFUNC(THive::StateWork) {
fFunc(TEvHive::TEvShrinkStoragePoolReply::EventType, EnqueueIncomingEvent);
fFunc(TEvHive::TEvShrinkStoragePoolDone::EventType, EnqueueIncomingEvent);
fFunc(TEvPrivate::TEvReassignInactiveGroupsComplete::EventType, EnqueueIncomingEvent);
- fFunc(TEvPrivate::TEvCompactComplete::EventType, EnqueueIncomingEvent);
+ fFunc(TEvPrivate::TEvMoveDataComplete::EventType, EnqueueIncomingEvent);
hFunc(TEvPrivate::TEvProcessIncomingEvent, Handle);
default:
if (!HandleDefaultEvents(ev, SelfId())) {
@@ -4265,7 +4268,15 @@ void THive::Handle(TEvHive::TEvShrinkStoragePool::TPtr& ev) {
}
}
- Execute(CreateShrinkPool(std::move(ev)));
+ if (pool.SetShrinkRequest(std::move(ev))) {
+ THolder<NKikimrBlobStorage::TEvControllerSelectGroups::TGroupParameters> item = pool.BuildRefreshRequest();
+ ++pool.RefreshRequestInFlight;
+ THolder<TEvBlobStorage::TEvControllerSelectGroups> request = MakeHolder<TEvBlobStorage::TEvControllerSelectGroups>();
+ NKikimrBlobStorage::TEvControllerSelectGroups& selectRecord = request->Record;
+ selectRecord.SetReturnAllMatchingGroups(true);
+ selectRecord.MutableGroupParameters()->AddAllocated(std::move(item).Release());
+ SendToBSControllerPipe(request.Release());
+ }
}
void THive::Handle(TEvHive::TEvShrinkStoragePoolReply::TPtr& ev) {
@@ -4274,17 +4285,17 @@ void THive::Handle(TEvHive::TEvShrinkStoragePoolReply::TPtr& ev) {
void THive::Handle(TEvPrivate::TEvReassignInactiveGroupsComplete::TPtr& ev) {
auto& pool = GetStoragePool(ev->Get()->PoolName);
- if (!CompactInactiveGroups(pool)) {
+ if (!MoveDataInactiveGroups(pool)) {
CheckRemainingHistory(pool);
}
}
-void THive::Handle(TEvPrivate::TEvCompactComplete::TPtr& ev) {
+void THive::Handle(TEvPrivate::TEvMoveDataComplete::TPtr& ev) {
auto& pool = GetStoragePool(ev->Get()->PoolName);
if (ev->Get()->Success) {
CheckRemainingHistory(pool);
} else {
- if (!CompactInactiveGroups(pool)) {
+ if (!MoveDataInactiveGroups(pool)) {
CheckRemainingHistory(pool);
}
}
@@ -4455,7 +4466,7 @@ void THive::StartShrinkPool(TStoragePoolInfo& pool) {
if (ReassignInactiveGroups(pool)) {
return;
}
- if (CompactInactiveGroups(pool)) {
+ if (MoveDataInactiveGroups(pool)) {
return;
}
CheckRemainingHistory(pool);
@@ -4498,9 +4509,9 @@ bool THive::ReassignInactiveGroups(TStoragePoolInfo& pool) {
}
}
-bool THive::CompactInactiveGroups(TStoragePoolInfo& pool) {
+bool THive::MoveDataInactiveGroups(TStoragePoolInfo& pool) {
std::unordered_set<TStorageGroupId> inactiveGroups(pool.InactiveGroups.begin(), pool.InactiveGroups.end());
- std::vector<TTabletId> tabletsToCompact;
+ std::vector<TTabletId> tabletsToMoveData;
if (pool.RemainingHistory.empty()) {
for (const auto& [tabletId, tablet] : Tablets) {
bool foundHistory = false;
@@ -4516,21 +4527,21 @@ bool THive::CompactInactiveGroups(TStoragePoolInfo& pool) {
}
}
if (foundHistory) {
- tabletsToCompact.push_back(tabletId);
+ tabletsToMoveData.push_back(tabletId);
}
}
} else {
auto tabletsWithHistory = pool.RemainingHistory | std::views::transform(&TStoragePoolInfo::THistoryEntry::Tablet);
std::unordered_set<TTabletId> uniqueTablets(tabletsWithHistory.begin(), tabletsWithHistory.end());
- tabletsToCompact.assign(uniqueTablets.begin(), uniqueTablets.end());
+ tabletsToMoveData.assign(uniqueTablets.begin(), uniqueTablets.end());
}
- if (tabletsToCompact.empty()) {
+ if (tabletsToMoveData.empty()) {
return false;
} else {
- YDB_LOG_INFO("ShrinkPool: starting compact for tablets",
+ YDB_LOG_INFO("ShrinkPool: starting move data for tablets",
{"logPrefix", GetLogPrefix()},
- {"tabletsToCompactCount", tabletsToCompact.size()});
- StartCompactActor(std::move(tabletsToCompact), pool.InactiveGroups, pool.Name);
+ {"tabletsToMoveDataCount", tabletsToMoveData.size()});
+ StartMoveDataActor(std::move(tabletsToMoveData), pool.InactiveGroups, pool.Name);
return true;
}
}
diff --git a/ydb/core/mind/hive/hive_impl.h b/ydb/core/mind/hive/hive_impl.h
index 1f6103e6d80..c28bdd5d05c 100644
--- a/ydb/core/mind/hive/hive_impl.h
+++ b/ydb/core/mind/hive/hive_impl.h
@@ -171,7 +171,7 @@ protected:
friend struct TNodeInfo;
friend struct TLeaderTabletInfo;
friend class TReassignTabletsActor;
- friend class TCompactActor;
+ friend class TMoveDataActor;
friend class TTxInitScheme;
friend class TTxDeleteBase;
@@ -258,7 +258,7 @@ protected:
void StartHiveStorageBalancer(TStorageBalancerSettings settings);
void StartReassignActor(std::vector<TReassignOperation> operations, const TActorId& source, ui32 maxInFlight, TString description, std::unique_ptr<IReassignCallback> callback);
void StartReassignActor(std::vector<TReassignOperation> operations);
- void StartCompactActor(std::vector<TTabletId> tablets, const std::vector<TStorageGroupId>& groups, const TString& poolName);
+ void StartMoveDataActor(std::vector<TTabletId> tablets, const std::vector<TStorageGroupId>& groups, const TString& poolName);
void CreateEvMonitoring(NMon::TEvRemoteHttpInfo::TPtr& ev, const TActorContext& ctx);
NJson::TJsonValue GetBalancerProgressJson();
ITransaction* CreateDeleteTablet(TEvHive::TEvDeleteTablet::TPtr& ev);
@@ -632,7 +632,7 @@ protected:
void Handle(TEvHive::TEvShrinkStoragePool::TPtr& ev);
void Handle(TEvHive::TEvShrinkStoragePoolReply::TPtr& ev);
void Handle(TEvPrivate::TEvReassignInactiveGroupsComplete::TPtr& ev);
- void Handle(TEvPrivate::TEvCompactComplete::TPtr& ev);
+ void Handle(TEvPrivate::TEvMoveDataComplete::TPtr& ev);
void Handle(TEvHive::TEvShrinkStoragePoolDone::TPtr& ev);
protected:
@@ -1149,7 +1149,7 @@ protected:
void StartShrinkPool(TStoragePoolInfo& pool);
bool ReassignInactiveGroups(TStoragePoolInfo& pool);
- bool CompactInactiveGroups(TStoragePoolInfo& pool);
+ bool MoveDataInactiveGroups(TStoragePoolInfo& pool);
void CheckRemainingHistory(TStoragePoolInfo& pool);
};
diff --git a/ydb/core/mind/hive/compact_actor.cpp b/ydb/core/mind/hive/move_data_actor.cpp
index df49f6368c3..583ac29b3d4 100644
--- a/ydb/core/mind/hive/compact_actor.cpp
+++ b/ydb/core/mind/hive/move_data_actor.cpp
@@ -6,8 +6,8 @@
namespace NKikimr::NHive {
-class TCompactActor
- : public TActorBootstrapped<TCompactActor>
+class TMoveDataActor
+ : public TActorBootstrapped<TMoveDataActor>
, public ISubActor
{
public:
@@ -21,10 +21,10 @@ public:
std::vector<TStorageGroupId> Groups;
TString PoolName;
std::vector<TPipeClient> PipeClients;
- i64 CompactsInFlight = 0;
+ i64 MoveDataInFlight = 0;
THive* Hive;
- TCompactActor(std::vector<TTabletId> tablets, const std::vector<TStorageGroupId>& groups, const TString& poolName, ui64 maxInFlight, THive* hive)
+ TMoveDataActor(std::vector<TTabletId> tablets, const std::vector<TStorageGroupId>& groups, const TString& poolName, ui64 maxInFlight, THive* hive)
: Tablets(std::move(tablets))
, NextTablet(Tablets.begin())
, Groups(groups)
@@ -48,21 +48,21 @@ public:
}
TString GetDescription() const override {
- return TStringBuilder() << "Compact(" << PoolName << ")";
+ return TStringBuilder() << "MoveData(" << PoolName << ")";
}
- void SendCompact(size_t index, TTabletId tablet) {
+ void SendMoveData(size_t index, TTabletId tablet) {
NTabletPipe::TClientConfig pipeConfig;
pipeConfig.RetryPolicy = {.RetryLimitCount = 13};
pipeConfig.CheckAliveness = true;
PipeClients[index] = {Register(NTabletPipe::CreateClient(SelfId(), tablet, pipeConfig)), tablet};
NTabletPipe::SendData(SelfId(), PipeClients[index].Client, new TEvTablet::TEvMoveData(Groups));
- ++CompactsInFlight;
+ ++MoveDataInFlight;
}
void CheckCompletion() {
- if (CompactsInFlight == 0 && NextTablet == Tablets.end()) {
- Send(Hive->SelfId(), new TEvPrivate::TEvCompactComplete(PoolName, true));
+ if (MoveDataInFlight == 0 && NextTablet == Tablets.end()) {
+ Send(Hive->SelfId(), new TEvPrivate::TEvMoveDataComplete(PoolName, true));
return PassAway();
}
}
@@ -70,7 +70,7 @@ public:
void Bootstrap() {
Become(&TThis::StateWork);
for (size_t i = 0; i < PipeClients.size() && NextTablet != Tablets.end(); ++i, ++NextTablet) {
- SendCompact(i, *NextTablet);
+ SendMoveData(i, *NextTablet);
}
return CheckCompletion();
}
@@ -80,9 +80,10 @@ public:
for (size_t i = 0; i < PipeClients.size(); ++i) {
if (PipeClients[i].Tablet == tablet) {
NTabletPipe::CloseClient(SelfId(), PipeClients[i].Client);
- --CompactsInFlight;
+ --MoveDataInFlight;
+ Hive->Execute(Hive->CreateRestartTablet(ToFullTabletId(tablet)));
if (NextTablet != Tablets.end()) {
- SendCompact(i, *(NextTablet++));
+ SendMoveData(i, *(NextTablet++));
break;
}
}
@@ -93,7 +94,7 @@ public:
void Handle(TEvTabletPipe::TEvClientConnected::TPtr& ev) {
if (ev->Get()->Status != NKikimrProto::OK) {
if (ev->Get()->Dead) {
- Send(Hive->SelfId(), new TEvPrivate::TEvCompactComplete(PoolName, false));
+ Send(Hive->SelfId(), new TEvPrivate::TEvMoveDataComplete(PoolName, false));
return PassAway();
} else {
Retry(ev->Get()->TabletId);
@@ -109,8 +110,8 @@ public:
for (size_t i = 0; i < PipeClients.size(); ++i) {
if (PipeClients[i].Tablet == tablet) {
NTabletPipe::CloseClient(SelfId(), PipeClients[i].Client);
- --CompactsInFlight;
- SendCompact(i, tablet);
+ --MoveDataInFlight;
+ SendMoveData(i, tablet);
break;
}
}
@@ -126,8 +127,8 @@ public:
}
};
-void THive::StartCompactActor(std::vector<TTabletId> tablets, const std::vector<TStorageGroupId>& groups, const TString& poolName) {
- auto* actor = new TCompactActor(std::move(tablets), groups, poolName, 1, this);
+void THive::StartMoveDataActor(std::vector<TTabletId> tablets, const std::vector<TStorageGroupId>& groups, const TString& poolName) {
+ auto* actor = new TMoveDataActor(std::move(tablets), groups, poolName, 1, this);
SubActors.emplace_back(actor);
RegisterWithSameMailbox(actor);
}
diff --git a/ydb/core/mind/hive/storage_pool_info.cpp b/ydb/core/mind/hive/storage_pool_info.cpp
index d401454e1dd..a092aafeb83 100644
--- a/ydb/core/mind/hive/storage_pool_info.cpp
+++ b/ydb/core/mind/hive/storage_pool_info.cpp
@@ -35,7 +35,14 @@ void TStoragePoolInfo::UpdateStorageGroup(TStorageGroupId groupId, const TEvCont
}
void TStoragePoolInfo::DeleteStorageGroup(TStorageGroupId groupId) {
- Groups.erase(groupId);
+ auto it = Groups.find(groupId);
+ if (it != Groups.end()) {
+ if (!it->second.IsActive()) {
+ auto inactiveIt = std::remove(InactiveGroups.begin(), InactiveGroups.end(), groupId);
+ InactiveGroups.erase(inactiveIt, InactiveGroups.end());
+ }
+ Groups.erase(it);
+ }
}
template <>
@@ -174,11 +181,17 @@ THolder<TEvControllerSelectGroups::TGroupParameters> TStoragePoolInfo::BuildRefr
}
bool TStoragePoolInfo::AddTabletToWait(TTabletId tabletId) {
- bool result = TabletsWaiting.empty();
+ bool result = TabletsWaiting.empty() && !(ShrinkRequest);
TabletsWaiting.emplace_back(tabletId);
return result;
}
+bool TStoragePoolInfo::SetShrinkRequest(TEvHive::TEvShrinkStoragePool::TPtr ev) {
+ bool result = TabletsWaiting.empty() && !(ShrinkRequest);
+ ShrinkRequest = std::move(ev);
+ return result;
+}
+
TVector<TTabletId> TStoragePoolInfo::PullWaitingTablets() {
return std::move(TabletsWaiting);
}
diff --git a/ydb/core/mind/hive/storage_pool_info.h b/ydb/core/mind/hive/storage_pool_info.h
index 0c67d9004a0..4acbeac1d61 100644
--- a/ydb/core/mind/hive/storage_pool_info.h
+++ b/ydb/core/mind/hive/storage_pool_info.h
@@ -74,6 +74,7 @@ struct TStoragePoolInfo {
ui64 ConsoleVersion = 0;
THashSet<THistoryEntry, THistoryHash> RemainingHistory;
bool NeedShrinkFromTenant = false;
+ TEvHive::TEvShrinkStoragePool::TPtr ShrinkRequest;
TStoragePoolInfo(const TString& name, THiveSharedSettings* hive);
TStoragePoolInfo(const TStoragePoolInfo&) = delete;
@@ -98,6 +99,7 @@ struct TStoragePoolInfo {
void Invalidate();
THolder<TEvControllerSelectGroups::TGroupParameters> BuildRefreshRequest() const;
bool AddTabletToWait(TTabletId tabletId);
+ bool SetShrinkRequest(TEvHive::TEvShrinkStoragePool::TPtr ev);
TVector<TTabletId> PullWaitingTablets();
template <NKikimrConfig::THiveConfig::EHiveStorageSelectStrategy Strategy>
size_t SelectGroup(const TVector<double>& groupCandidateUsages);
diff --git a/ydb/core/mind/hive/ya.make b/ydb/core/mind/hive/ya.make
index f609cdc1f27..4a9a0f632fd 100644
--- a/ydb/core/mind/hive/ya.make
+++ b/ydb/core/mind/hive/ya.make
@@ -6,7 +6,6 @@ SRCS(
boot_queue.cpp
boot_queue.h
bridge_pile_info.h
- compact_actor.cpp
data_center_info.h
domain_info.cpp
domain_info.h
@@ -30,6 +29,7 @@ SRCS(
metrics.h
monitoring.cpp
monitoring.h
+ move_data_actor.cpp
node_info.cpp
node_info.h
object_distribution.h
diff --git a/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_controller.cpp b/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_controller.cpp
new file mode 100644
index 00000000000..ec58390d838
--- /dev/null
+++ b/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_controller.cpp
@@ -0,0 +1,24 @@
+#include "dbs_controller.h"
+
+#include "dbs_controller_actor.h"
+
+#include <ydb/core/base/appdata_fwd.h>
+
+#include <ydb/library/actors/core/actor.h>
+
+namespace NYdb::NBS::NBlockStore::NStorage::NDbsController {
+
+using namespace NKikimr;
+
+////////////////////////////////////////////////////////////////////////////////
+
+IActor* CreateDbsControllerTablet(
+ const TActorId& tablet,
+ TTabletStorageInfo* info)
+{
+ return new TDbsControllerActor(tablet, info);
+}
+
+////////////////////////////////////////////////////////////////////////////////
+
+} // namespace NYdb::NBS::NBlockStore::NStorage::NDbsController
diff --git a/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_controller.h b/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_controller.h
new file mode 100644
index 00000000000..d2498d84b8c
--- /dev/null
+++ b/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_controller.h
@@ -0,0 +1,19 @@
+#pragma once
+
+#include <ydb/library/actors/core/actor.h>
+
+namespace NKikimr {
+class TTabletStorageInfo;
+} // namespace NKikimr
+
+namespace NYdb::NBS::NBlockStore::NStorage::NDbsController {
+
+////////////////////////////////////////////////////////////////////////////////
+
+NActors::IActor* CreateDbsControllerTablet(
+ const NActors::TActorId& tablet,
+ NKikimr::TTabletStorageInfo* info);
+
+////////////////////////////////////////////////////////////////////////////////
+
+} // namespace NYdb::NBS::NBlockStore::NStorage::NDbsController
diff --git a/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_controller_actor.cpp b/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_controller_actor.cpp
new file mode 100644
index 00000000000..1216c7e4d38
--- /dev/null
+++ b/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_controller_actor.cpp
@@ -0,0 +1,173 @@
+#include "dbs_controller_actor.h"
+
+#include "dbs_controller_database.h"
+
+#include <ydb/core/nbs/cloud/storage/core/libs/actors/helpers.h>
+
+#include <ydb/core/base/tablet_pipe.h>
+#include <ydb/core/node_whiteboard/node_whiteboard.h>
+
+namespace NYdb::NBS::NBlockStore::NStorage::NDbsController {
+
+using namespace NKikimr;
+using namespace NActors;
+
+////////////////////////////////////////////////////////////////////////////////
+
+TDbsControllerActor::TDbsControllerActor(
+ const TActorId& tablet,
+ NKikimr::TTabletStorageInfo* info)
+ : TActor(&TThis::StateInit)
+ , TTabletBase<TDbsControllerActor>(
+ tablet,
+ NKikimr::TTabletStorageInfoPtr(info),
+ nullptr)
+{
+ LOG_INFO(
+ TActivationContext::AsActorContext(),
+ NKikimrServices::DBS_CONTROLLER,
+ "[%lu] DbsController initialization started",
+ TabletID());
+}
+
+TDbsControllerActor::~TDbsControllerActor() = default;
+
+void TDbsControllerActor::OnDetach(const TActorContext& ctx)
+{
+ LOG_INFO(
+ ctx,
+ NKikimrServices::DBS_CONTROLLER,
+ "[%lu] OnDetach",
+ TabletID());
+ Die(ctx);
+}
+
+void TDbsControllerActor::OnTabletDead(
+ TEvTablet::TEvTabletDead::TPtr& ev,
+ const TActorContext& ctx)
+{
+ Y_UNUSED(ev);
+
+ LOG_INFO(
+ ctx,
+ NKikimrServices::DBS_CONTROLLER,
+ "[%lu] OnTabletDead",
+ TabletID());
+ Die(ctx);
+}
+
+void TDbsControllerActor::OnActivateExecutor(const TActorContext& ctx)
+{
+ Become(&TThis::StateWork);
+
+ LOG_INFO(
+ ctx,
+ NKikimrServices::DBS_CONTROLLER,
+ "[%lu] Started DbsController: actor id %s",
+ TabletID(),
+ SelfId().ToString().data());
+
+ if (!Executor()->GetStats().IsFollower()) {
+ ExecuteTx(ctx, CreateTx<TInitSchema>());
+ }
+
+ // allow pipes to connect
+ SignalTabletActive(ctx);
+}
+
+void TDbsControllerActor::DefaultSignalTabletActive(const TActorContext& ctx)
+{
+ Y_UNUSED(ctx);
+}
+
+void TDbsControllerActor::HandleServerConnected(
+ const TEvTabletPipe::TEvServerConnected::TPtr& ev,
+ const TActorContext& ctx)
+{
+ const auto* msg = ev->Get();
+
+ LOG_DEBUG(
+ ctx,
+ NKikimrServices::DBS_CONTROLLER,
+ "[%lu] Pipe client %s server %s connected",
+ TabletID(),
+ ToString(msg->ClientId).c_str(),
+ ToString(msg->ServerId).c_str());
+}
+
+void TDbsControllerActor::HandleServerDisconnected(
+ const TEvTabletPipe::TEvServerDisconnected::TPtr& ev,
+ const TActorContext& ctx)
+{
+ const auto* msg = ev->Get();
+
+ LOG_DEBUG(
+ ctx,
+ NKikimrServices::DBS_CONTROLLER,
+ "[%lu] Pipe client %s server %s disconnected",
+ TabletID(),
+ ToString(msg->ClientId).c_str(),
+ ToString(msg->ServerId).c_str());
+}
+
+void TDbsControllerActor::HandleServerDestroyed(
+ const TEvTabletPipe::TEvServerDestroyed::TPtr& ev,
+ const TActorContext& ctx)
+{
+ const auto* msg = ev->Get();
+
+ LOG_INFO(
+ ctx,
+ NKikimrServices::DBS_CONTROLLER,
+ "[%lu] Pipe client %s server %s got destroyed",
+ TabletID(),
+ ToString(msg->ClientId).c_str(),
+ ToString(msg->ServerId).c_str());
+}
+
+void TDbsControllerActor::ReportTabletState(const TActorContext& ctx)
+{
+ auto service =
+ NNodeWhiteboard::MakeNodeWhiteboardServiceId(SelfId().NodeId());
+
+ auto request = std::make_unique<
+ NNodeWhiteboard::TEvWhiteboard::TEvWhiteboard::TEvTabletStateUpdate>(
+ TabletID(),
+ STATE_WORK);
+
+ NYdb::NBS::Send(ctx, service, std::move(request));
+}
+
+////////////////////////////////////////////////////////////////////////////////
+
+void TDbsControllerActor::StateInit(TAutoPtr<NActors::IEventHandle>& ev)
+{
+ StateInitImpl(ev, SelfId());
+}
+
+STFUNC(TDbsControllerActor::StateWork)
+{
+ switch (ev->GetTypeRewrite()) {
+ cFunc(TEvents::TEvPoison::EventType, PassAway);
+
+ HFunc(TEvTabletPipe::TEvServerConnected, HandleServerConnected);
+ HFunc(TEvTabletPipe::TEvServerDisconnected, HandleServerDisconnected);
+ HFunc(TEvTabletPipe::TEvServerDestroyed, HandleServerDestroyed);
+
+ default:
+ if (!HandleDefaultEvents(ev, SelfId())) {
+ LOG_ERROR(
+ TActivationContext::AsActorContext(),
+ NKikimrServices::DBS_CONTROLLER,
+ "[%lu] Unhandled event type: %u event %s",
+ TabletID(),
+ ev->GetTypeRewrite(),
+ ev->ToString().c_str());
+ }
+ break;
+ }
+}
+
+////////////////////////////////////////////////////////////////////////////////
+
+} // namespace NYdb::NBS::NBlockStore::NStorage::NDbsController
diff --git a/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_controller_actor.h b/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_controller_actor.h
new file mode 100644
index 00000000000..3bccc105100
--- /dev/null
+++ b/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_controller_actor.h
@@ -0,0 +1,72 @@
+#pragma once
+
+#include "dbs_controller_counters.h"
+
+#include <ydb/core/nbs/cloud/blockstore/libs/storage/core/tablet.h>
+
+#include <ydb/core/base/tablet_pipe.h>
+#include <ydb/core/tablet_flat/tablet_flat_executed.h>
+
+#include <ydb/library/actors/core/actor.h>
+#include <ydb/library/services/services.pb.h>
+
+namespace NYdb::NBS::NBlockStore::NStorage::NDbsController {
+
+////////////////////////////////////////////////////////////////////////////////
+
+class TDbsControllerActor
+ : public NActors::TActor<TDbsControllerActor>
+ , public TTabletBase<TDbsControllerActor>
+{
+ enum EState
+ {
+ STATE_BOOT,
+ STATE_INIT,
+ STATE_WORK,
+ STATE_ZOMBIE,
+ STATE_MAX,
+ };
+
+public:
+ TDbsControllerActor(
+ const NActors::TActorId& tablet,
+ NKikimr::TTabletStorageInfo* info);
+
+ ~TDbsControllerActor() override;
+
+ static constexpr ui32 LogComponent = NKikimrServices::DBS_CONTROLLER;
+ using TCounters = TDbsControllerCounters;
+
+private:
+ void StateInit(TAutoPtr<NActors::IEventHandle>& ev);
+ STFUNC(StateWork);
+
+ void OnDetach(const NActors::TActorContext& ctx) override;
+ void OnTabletDead(
+ NKikimr::TEvTablet::TEvTabletDead::TPtr& ev,
+ const NActors::TActorContext& ctx) override;
+ void OnActivateExecutor(const NActors::TActorContext& ctx) override;
+ void DefaultSignalTabletActive(const NActors::TActorContext& ctx) override;
+
+ void HandleServerConnected(
+ const NKikimr::TEvTabletPipe::TEvServerConnected::TPtr& ev,
+ const NActors::TActorContext& ctx);
+
+ void HandleServerDisconnected(
+ const NKikimr::TEvTabletPipe::TEvServerDisconnected::TPtr& ev,
+ const NActors::TActorContext& ctx);
+
+ void HandleServerDestroyed(
+ const NKikimr::TEvTabletPipe::TEvServerDestroyed::TPtr& ev,
+ const NActors::TActorContext& ctx);
+
+ void ReportTabletState(const NActors::TActorContext& ctx);
+
+ BLOCKSTORE_DBS_CONTROLLER_TRANSACTIONS(
+ BLOCKSTORE_IMPLEMENT_TRANSACTION,
+ TTxDbsController)
+};
+
+////////////////////////////////////////////////////////////////////////////////
+
+} // namespace NYdb::NBS::NBlockStore::NStorage::NDbsController
diff --git a/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_controller_counters.h b/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_controller_counters.h
new file mode 100644
index 00000000000..40cbd601d78
--- /dev/null
+++ b/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_controller_counters.h
@@ -0,0 +1,24 @@
+#pragma once
+
+#include <ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_controller_tx.h>
+
+namespace NYdb::NBS::NBlockStore::NStorage::NDbsController {
+
+////////////////////////////////////////////////////////////////////////////////
+
+struct TDbsControllerCounters
+{
+ enum ETransactionType
+ {
+#define BLOCKSTORE_TRANSACTION_TYPE(name, ...) TX_##name,
+
+ BLOCKSTORE_DBS_CONTROLLER_TRANSACTIONS(BLOCKSTORE_TRANSACTION_TYPE)
+ TX_SIZE
+
+#undef BLOCKSTORE_TRANSACTION_TYPE
+ };
+};
+
+////////////////////////////////////////////////////////////////////////////////
+
+} // namespace NYdb::NBS::NBlockStore::NStorage::NDbsController
diff --git a/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_controller_database.cpp b/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_controller_database.cpp
new file mode 100644
index 00000000000..66be14d7e72
--- /dev/null
+++ b/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_controller_database.cpp
@@ -0,0 +1,19 @@
+#include "dbs_controller_database.h"
+
+#include "dbs_controller_schema.h"
+
+namespace NYdb::NBS::NBlockStore::NStorage::NDbsController {
+
+////////////////////////////////////////////////////////////////////////////////
+
+void TDbsControllerDatabase::InitSchema()
+{
+ Materialize<TDbsControllerSchema>();
+
+ TSchemaInitializer<TDbsControllerSchema::TTables>::InitStorage(
+ Database.Alter());
+}
+
+////////////////////////////////////////////////////////////////////////////////
+
+} // namespace NYdb::NBS::NBlockStore::NStorage::NDbsController
diff --git a/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_controller_database.h b/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_controller_database.h
new file mode 100644
index 00000000000..f48a05b6858
--- /dev/null
+++ b/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_controller_database.h
@@ -0,0 +1,26 @@
+#pragma once
+
+#include <ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/protos/dbs_controller.pb.h>
+
+#include <ydb/core/tablet_flat/flat_cxx_database.h>
+
+#include <util/generic/maybe.h>
+
+namespace NYdb::NBS::NBlockStore::NStorage::NDbsController {
+
+////////////////////////////////////////////////////////////////////////////////
+
+class TDbsControllerDatabase: public NKikimr::NIceDb::TNiceDb
+{
+ using TDbsControllerState =
+ ::NYdb::NBS::DbsController::NProto::TDbsControllerState;
+
+public:
+ TDbsControllerDatabase(NKikimr::NTable::TDatabase& database)
+ : NKikimr::NIceDb::TNiceDb(database)
+ {}
+
+ void InitSchema();
+};
+
+} // namespace NYdb::NBS::NBlockStore::NStorage::NDbsController
diff --git a/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_controller_schema.h b/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_controller_schema.h
new file mode 100644
index 00000000000..c6327a4d38d
--- /dev/null
+++ b/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_controller_schema.h
@@ -0,0 +1,45 @@
+#pragma once
+
+#include <ydb/core/nbs/cloud/blockstore/libs/storage/core/tablet_schema.h>
+#include <ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/protos/dbs_controller.pb.h>
+
+#include <ydb/core/tablet_flat/flat_cxx_database.h>
+
+namespace NYdb::NBS::NBlockStore::NStorage::NDbsController {
+
+////////////////////////////////////////////////////////////////////////////////
+
+struct TDbsControllerSchema: public NKikimr::NIceDb::Schema
+{
+ enum EChannels
+ {
+ SystemChannel,
+ LogChannel,
+ IndexChannel,
+ };
+
+ struct Dummy: public TTableSchema<1>
+ {
+ struct DummyA: public Column<1, NKikimr::NScheme::NTypeIds::Uint32>
+ {
+ };
+
+ struct DummyB: public Column<2, NKikimr::NScheme::NTypeIds::Uint32>
+ {
+ };
+
+ using TKey = TableKey<DummyA>;
+ using TColumns = TableColumns<DummyA, DummyB>;
+
+ using StoragePolicy = TStoragePolicy<IndexChannel>;
+ };
+
+ using TTables = SchemaTables<Dummy>;
+
+ using TSettings =
+ SchemaSettings<ExecutorLogBatching<true>, ExecutorLogFlushPeriod<0>>;
+};
+
+////////////////////////////////////////////////////////////////////////////////
+
+} // namespace NYdb::NBS::NBlockStore::NStorage::NDbsController
diff --git a/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_controller_tx.h b/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_controller_tx.h
new file mode 100644
index 00000000000..ef7b5c7560b
--- /dev/null
+++ b/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_controller_tx.h
@@ -0,0 +1,46 @@
+#pragma once
+
+#include <ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/protos/dbs_controller.pb.h>
+
+#include <util/generic/maybe.h>
+
+namespace NYdb::NBS::NBlockStore::NStorage::NDbsController {
+
+////////////////////////////////////////////////////////////////////////////////
+
+#define BLOCKSTORE_DBS_CONTROLLER_TRANSACTIONS(xxx, ...) \
+ xxx(InitSchema, __VA_ARGS__) \
+ xxx(LoadState, __VA_ARGS__)
+
+// BLOCKSTORE_DBS_CONTROLLER_TRANSACTIONS
+
+////////////////////////////////////////////////////////////////////////////////
+
+struct TTxDbsController
+{
+ //
+ // InitSchema
+ //
+ struct TInitSchema
+ {
+ explicit TInitSchema()
+ {}
+
+ void Clear()
+ {}
+ };
+
+ //
+ // LoadState
+ //
+ struct TLoadState
+ {
+ explicit TLoadState()
+ {}
+
+ void Clear()
+ {}
+ };
+};
+
+} // namespace NYdb::NBS::NBlockStore::NStorage::NDbsController
diff --git a/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_initschema.cpp b/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_initschema.cpp
new file mode 100644
index 00000000000..d2feda2ffd0
--- /dev/null
+++ b/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_initschema.cpp
@@ -0,0 +1,47 @@
+#include "dbs_controller_actor.h"
+#include "dbs_controller_database.h"
+
+namespace NYdb::NBS::NBlockStore::NStorage::NDbsController {
+
+using namespace NActors;
+using namespace NKikimr;
+using namespace NKikimr::NTabletFlatExecutor;
+
+////////////////////////////////////////////////////////////////////////////////
+
+bool TDbsControllerActor::PrepareInitSchema(
+ const TActorContext& ctx,
+ TTransactionContext& tx,
+ TTxDbsController::TInitSchema& args)
+{
+ Y_UNUSED(ctx);
+ Y_UNUSED(tx);
+ Y_UNUSED(args);
+
+ return true;
+}
+
+void TDbsControllerActor::ExecuteInitSchema(
+ const TActorContext& ctx,
+ TTransactionContext& tx,
+ TTxDbsController::TInitSchema& args)
+{
+ Y_UNUSED(ctx);
+ Y_UNUSED(args);
+
+ TDbsControllerDatabase db(tx.DB);
+ db.InitSchema();
+}
+
+void TDbsControllerActor::CompleteInitSchema(
+ const TActorContext& ctx,
+ TTxDbsController::TInitSchema& args)
+{
+ Y_UNUSED(args);
+
+ ExecuteTx(ctx, CreateTx<TLoadState>());
+}
+
+////////////////////////////////////////////////////////////////////////////////
+
+} // namespace NYdb::NBS::NBlockStore::NStorage::NDbsController
diff --git a/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_loadstate.cpp b/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_loadstate.cpp
new file mode 100644
index 00000000000..75bf7b577ab
--- /dev/null
+++ b/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_loadstate.cpp
@@ -0,0 +1,49 @@
+#include "dbs_controller_actor.h"
+#include "dbs_controller_database.h"
+
+namespace NYdb::NBS::NBlockStore::NStorage::NDbsController {
+
+using namespace NActors;
+using namespace NKikimr;
+using namespace NKikimr::NTabletFlatExecutor;
+
+////////////////////////////////////////////////////////////////////////////////
+
+bool TDbsControllerActor::PrepareLoadState(
+ const TActorContext& ctx,
+ TTransactionContext& tx,
+ TTxDbsController::TLoadState& args)
+{
+ Y_UNUSED(ctx);
+ Y_UNUSED(tx);
+ Y_UNUSED(args);
+
+ return true;
+}
+
+void TDbsControllerActor::ExecuteLoadState(
+ const TActorContext& ctx,
+ TTransactionContext& tx,
+ TTxDbsController::TLoadState& args)
+{
+ Y_UNUSED(ctx);
+ Y_UNUSED(tx);
+ Y_UNUSED(args);
+}
+
+void TDbsControllerActor::CompleteLoadState(
+ const TActorContext& ctx,
+ TTxDbsController::TLoadState& args)
+{
+ Y_UNUSED(args);
+
+ LOG_INFO(
+ ctx,
+ NKikimrServices::DBS_CONTROLLER,
+ "[%lu] State loaded, DbsController is ready",
+ TabletID());
+}
+
+////////////////////////////////////////////////////////////////////////////////
+
+} // namespace NYdb::NBS::NBlockStore::NStorage::NDbsController
diff --git a/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/protos/dbs_controller.proto b/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/protos/dbs_controller.proto
new file mode 100644
index 00000000000..bcdba308ea1
--- /dev/null
+++ b/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/protos/dbs_controller.proto
@@ -0,0 +1,10 @@
+syntax = "proto3";
+
+package NYdb.NBS.DbsController.NProto;
+option java_package = "ru.yandex.ydb.core.nbs.dbs_controller.proto";
+
+// Persisted controller state. Placeholder for now - fill with real fields as
+// the tablet gains functionality.
+message TDbsControllerState {
+ uint64 Version = 1;
+}
diff --git a/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/protos/ya.make b/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/protos/ya.make
new file mode 100644
index 00000000000..13f41bad6aa
--- /dev/null
+++ b/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/protos/ya.make
@@ -0,0 +1,14 @@
+PROTO_LIBRARY()
+
+EXCLUDE_TAGS(
+ GO_PROTO
+ JAVA_PROTO
+)
+
+SRCS(
+ dbs_controller.proto
+)
+
+#CPP_PROTO_PLUGIN0(validation ydb/public/lib/validation)
+
+END()
diff --git a/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/ut/dbs_controller_ut.cpp b/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/ut/dbs_controller_ut.cpp
new file mode 100644
index 00000000000..447f25fba9c
--- /dev/null
+++ b/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/ut/dbs_controller_ut.cpp
@@ -0,0 +1,37 @@
+#include <ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/dbs_controller_actor.h>
+
+#include <ydb/core/testlib/basics/runtime.h>
+#include <ydb/core/testlib/tablet_helpers.h>
+
+#include <library/cpp/testing/unittest/registar.h>
+
+namespace NYdb::NBS::NBlockStore::NStorage::NDbsController {
+
+using namespace NKikimr;
+
+////////////////////////////////////////////////////////////////////////////////
+
+Y_UNIT_TEST_SUITE(TDbsControllerTest)
+{
+ Y_UNIT_TEST(ShouldBoot)
+ {
+ TTestBasicRuntime runtime;
+ SetupTabletServices(runtime);
+
+ const ui64 tabletId = MakeTabletID(0, 0, 1);
+
+ CreateTestBootstrapper(
+ runtime,
+ CreateTestTabletInfo(tabletId, TTabletTypes::DbsController),
+ [](const TActorId& tablet, TTabletStorageInfo* info) -> IActor*
+ { return new TDbsControllerActor(tablet, info); });
+
+ TDispatchOptions options;
+ options.FinalEvents.emplace_back(TEvTablet::EvBoot, 1);
+ runtime.DispatchEvents(options);
+ }
+}
+
+////////////////////////////////////////////////////////////////////////////////
+
+} // namespace NYdb::NBS::NBlockStore::NStorage::NDbsController
diff --git a/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/ut/ya.make b/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/ut/ya.make
new file mode 100644
index 00000000000..9812a696381
--- /dev/null
+++ b/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/ut/ya.make
@@ -0,0 +1,15 @@
+UNITTEST_FOR(ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller)
+
+SRCS(
+ dbs_controller_ut.cpp
+)
+
+PEERDIR(
+ ydb/core/base
+ ydb/core/protos
+ ydb/core/testlib
+ ydb/core/testlib/basics
+ yql/essentials/sql/pg_dummy
+)
+
+END()
diff --git a/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/ya.make b/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/ya.make
new file mode 100644
index 00000000000..693a0c5de2f
--- /dev/null
+++ b/ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/ya.make
@@ -0,0 +1,29 @@
+LIBRARY()
+
+SRCS(
+ dbs_controller.cpp
+ dbs_controller_actor.cpp
+ dbs_controller_database.cpp
+ dbs_initschema.cpp
+ dbs_loadstate.cpp
+)
+
+PEERDIR(
+ ydb/core/nbs/cloud/blockstore/libs/storage/core
+ ydb/core/nbs/cloud/blockstore/libs/storage/dbs_controller/protos
+ ydb/core/base
+ ydb/core/protos
+ ydb/core/tablet_flat
+ ydb/library/actors/core
+ ydb/library/services
+)
+
+END()
+
+RECURSE(
+ protos
+)
+
+RECURSE_FOR_TESTS(
+ ut
+)
diff --git a/ydb/core/nbs/cloud/blockstore/libs/storage/partition_direct/part_monitoring.cpp b/ydb/core/nbs/cloud/blockstore/libs/storage/partition_direct/part_monitoring.cpp
index 1fe17501514..a835073e5bf 100644
--- a/ydb/core/nbs/cloud/blockstore/libs/storage/partition_direct/part_monitoring.cpp
+++ b/ydb/core/nbs/cloud/blockstore/libs/storage/partition_direct/part_monitoring.cpp
@@ -94,10 +94,16 @@ TTabletInfo TPartitionActor::MakeMonTabletInfo() const
};
}
-void TPartitionActor::HandleHttpInfo(
- NMon::TEvRemoteHttpInfo::TPtr& ev,
+bool TPartitionActor::OnRenderAppHtmlPage(
+ NMon::TEvRemoteHttpInfo::TPtr ev,
const TActorContext& ctx)
{
+ if (!ev) {
+ // Probe from the standard tablet page: report that the App page exists
+ // so its link is shown.
+ return true;
+ }
+
const auto& cgi = ev->Get()->Cgi();
const EMonPage page = ParsePage(cgi);
@@ -115,14 +121,14 @@ void TPartitionActor::HandleHttpInfo(
ctx.Send(
ev->Sender,
new NMon::TEvRemoteHttpInfoRes(RenderMonPage(data)));
- return;
+ return true;
}
// Local DB page: read the persisted state in a transaction;
// CompleteMonitoring renders and replies.
if (page == EMonPage::LocalDb) {
ExecuteTx(ctx, CreateTx<TMonitoring>(ev->Sender));
- return;
+ return true;
}
// VChunk page: no index - just the input form (synchronous); with an
@@ -137,7 +143,7 @@ void TPartitionActor::HandleHttpInfo(
ctx.Send(
ev->Sender,
new NMon::TEvRemoteHttpInfoRes(RenderMonPage(data)));
- return;
+ return true;
}
auto* actorSystem = TActivationContext::ActorSystem();
@@ -160,7 +166,7 @@ void TPartitionActor::HandleHttpInfo(
requester,
new NMon::TEvRemoteHttpInfoRes(RenderMonPage(data)));
});
- return;
+ return true;
}
const std::optional<size_t> selectedDbg = ParseSelectedDbg(cgi);
@@ -198,7 +204,7 @@ void TPartitionActor::HandleHttpInfo(
reply << "<meta http-equiv='refresh' content='0; ?TabletID="
<< TabletID() << "&page=dbg&dbg=" << *selectedDbg << "'/>";
ctx.Send(ev->Sender, new NMon::TEvRemoteHttpInfoRes(reply));
- return;
+ return true;
}
// DBG page: gather snapshots, then render + reply in the callback. Safe
@@ -230,6 +236,7 @@ void TPartitionActor::HandleHttpInfo(
requester,
new NMon::TEvRemoteHttpInfoRes(RenderMonPage(data)));
});
+ return true;
}
////////////////////////////////////////////////////////////////////////////////
diff --git a/ydb/core/nbs/cloud/blockstore/libs/storage/partition_direct/partition_direct_actor.cpp b/ydb/core/nbs/cloud/blockstore/libs/storage/partition_direct/partition_direct_actor.cpp
index 07d185ad5bf..16e32a56a1c 100644
--- a/ydb/core/nbs/cloud/blockstore/libs/storage/partition_direct/partition_direct_actor.cpp
+++ b/ydb/core/nbs/cloud/blockstore/libs/storage/partition_direct/partition_direct_actor.cpp
@@ -640,8 +640,6 @@ STFUNC(TPartitionActor::StateWork)
TEvPartitionDirectPrivate::TEvPoison,
HandlePoisonByBlockedGeneration);
- HFunc(NMon::TEvRemoteHttpInfo, HandleHttpInfo);
-
default:
if (!HandleDefaultEvents(ev, SelfId())) {
LOG_ERROR(
diff --git a/ydb/core/nbs/cloud/blockstore/libs/storage/partition_direct/partition_direct_actor.h b/ydb/core/nbs/cloud/blockstore/libs/storage/partition_direct/partition_direct_actor.h
index ceada066408..6c19978475b 100644
--- a/ydb/core/nbs/cloud/blockstore/libs/storage/partition_direct/partition_direct_actor.h
+++ b/ydb/core/nbs/cloud/blockstore/libs/storage/partition_direct/partition_direct_actor.h
@@ -82,9 +82,12 @@ private:
void StateInit(TAutoPtr<NActors::IEventHandle>& ev);
STFUNC(StateWork);
- void HandleHttpInfo(
- NActors::NMon::TEvRemoteHttpInfo::TPtr& ev,
- const NActors::TActorContext& ctx);
+ // The tablet's own monitoring page, reached via the standard tablet page's
+ // "App" link. The base class passes a null event to ask whether that link
+ // should appear - it always should.
+ bool OnRenderAppHtmlPage(
+ NActors::NMon::TEvRemoteHttpInfo::TPtr ev,
+ const NActors::TActorContext& ctx) override;
void OnDetach(const NActors::TActorContext& ctx) override;
void OnTabletDead(
diff --git a/ydb/core/nbs/cloud/blockstore/libs/storage/partition_direct/partition_direct_ut.cpp b/ydb/core/nbs/cloud/blockstore/libs/storage/partition_direct/partition_direct_ut.cpp
index 18f1b62b974..8c7c8aa1199 100644
--- a/ydb/core/nbs/cloud/blockstore/libs/storage/partition_direct/partition_direct_ut.cpp
+++ b/ydb/core/nbs/cloud/blockstore/libs/storage/partition_direct/partition_direct_ut.cpp
@@ -1439,6 +1439,46 @@ Y_UNIT_TEST_SUITE(TPartitionDirectTest)
UNIT_ASSERT_STRING_CONTAINS(html, "Overview");
}
+ Y_UNIT_TEST(StandardTabletPageRenders)
+ {
+ TEnvironmentSetup env{{
+ .NodeCount = 8,
+ .Erasure = TBlobStorageGroupType::Erasure4Plus2Block,
+ }};
+ auto& runtime = env.Runtime;
+
+ auto scopedService = SetupStorage(env, EWriteMode::DirectWrite);
+ const ui64 tabletId = CreatePartitionTablet(env);
+
+ const TActorId edge = runtime->AllocateEdgeActor(
+ env.Settings.ControllerNodeId,
+ __FILE__,
+ __LINE__);
+
+ // Empty path (not "/app") renders the standard flat-tablet page.
+ runtime->SendToPipe(
+ tabletId,
+ edge,
+ new NActors::NMon::TEvRemoteHttpInfo(
+ "?TabletID=" + ToString(tabletId)),
+ 0,
+ TTestActorSystem::GetPipeConfigWithRetries());
+
+ auto response =
+ env.WaitForEdgeActorEvent<NActors::NMon::TEvRemoteHttpInfoRes>(
+ edge);
+ UNIT_ASSERT(response);
+
+ const TString& html = response->Get()->Html;
+ // The standard tablet page: the info block, the Restart action, and the
+ // "App" link to this tablet's own monitoring page.
+ UNIT_ASSERT_STRING_CONTAINS(html, "Tablet generation");
+ UNIT_ASSERT_STRING_CONTAINS(
+ html,
+ "RestartTabletID=" + ToString(tabletId));
+ UNIT_ASSERT_STRING_CONTAINS(html, "tablets/app?");
+ }
+
Y_UNIT_TEST(ShouldSuicideOnPoisonByBlockedGeneration)
{
TEnvironmentSetup env{{
diff --git a/ydb/core/nbs/cloud/blockstore/libs/storage/partition_direct/protos/ya.make b/ydb/core/nbs/cloud/blockstore/libs/storage/partition_direct/protos/ya.make
index 9ecb8949d1e..de0a3186df9 100644
--- a/ydb/core/nbs/cloud/blockstore/libs/storage/partition_direct/protos/ya.make
+++ b/ydb/core/nbs/cloud/blockstore/libs/storage/partition_direct/protos/ya.make
@@ -11,6 +11,4 @@ PEERDIR(
ydb/core/protos
)
-#CPP_PROTO_PLUGIN0(validation ydb/public/lib/validation)
-
END()
diff --git a/ydb/core/nbs/cloud/blockstore/libs/storage/ya.make b/ydb/core/nbs/cloud/blockstore/libs/storage/ya.make
index b9990b4fb2d..d9a1882df0b 100644
--- a/ydb/core/nbs/cloud/blockstore/libs/storage/ya.make
+++ b/ydb/core/nbs/cloud/blockstore/libs/storage/ya.make
@@ -1,5 +1,6 @@
RECURSE(
api
+ dbs_controller
partition_direct
storage_transport
testlib
diff --git a/ydb/core/persqueue/public/constants.h b/ydb/core/persqueue/public/constants.h
index 4b330799ed2..f73c64de0a9 100644
--- a/ydb/core/persqueue/public/constants.h
+++ b/ydb/core/persqueue/public/constants.h
@@ -25,7 +25,11 @@ constexpr i32 MAX_READ_RULES_COUNT = 3000;
constexpr i32 MAX_SUPPORTED_CODECS_COUNT = 100;
constexpr ui64 DEFAULT_PARTITION_WRITE_SPEED_MESSAGES_PER_SECOND = 1024 * 1024;
-constexpr ui32 CONTENT_BASED_DEDUPLICATION_MESSAGE_LIMIT = 1000;
-constexpr ui32 CONTENT_BASED_DEDUPLICATION_MESSAGE_BURST = 1000;
+// FIFO queues use a lower per-partition message rate because of ordering/dedup work.
+constexpr ui32 FIFO_PARTITION_WRITE_SPEED_MESSAGES_PER_SECOND = 1000;
+constexpr ui32 FIFO_PARTITION_WRITE_BURST_MESSAGES = 1000;
+
+constexpr ui32 CONTENT_BASED_DEDUPLICATION_MESSAGE_LIMIT = FIFO_PARTITION_WRITE_SPEED_MESSAGES_PER_SECOND;
+constexpr ui32 CONTENT_BASED_DEDUPLICATION_MESSAGE_BURST = FIFO_PARTITION_WRITE_BURST_MESSAGES;
} // namespace NKikimr::NPQ
diff --git a/ydb/core/protos/counters_columnshard.proto b/ydb/core/protos/counters_columnshard.proto
index d2714dbdd29..aee63e7972f 100644
--- a/ydb/core/protos/counters_columnshard.proto
+++ b/ydb/core/protos/counters_columnshard.proto
@@ -50,6 +50,8 @@ enum ESimpleCounters {
COUNTER_EVICTED_RAW_BYTES = 40 [(CounterOpts) = {Name: "Index/EvictedBytesRaw"}];
COUNTER_WRITES_IN_FLY = 41 [(CounterOpts) = {Name: "WritesInFly"}];
COUNTER_TX_COMPLETE_LAG = 42 [(CounterOpts) = {Name: "TxCompleteLag"}];
+ COUNTER_DATA_BYTES = 43 [(CounterOpts) = {Name: "DataBytes"}];
+ COUNTER_INDEX_BYTES = 44 [(CounterOpts) = {Name: "IndexBytes"}];
}
enum ECumulativeCounters {
diff --git a/ydb/core/protos/feature_flags.proto b/ydb/core/protos/feature_flags.proto
index 127175149f6..8679a47d7ec 100644
--- a/ydb/core/protos/feature_flags.proto
+++ b/ydb/core/protos/feature_flags.proto
@@ -356,4 +356,5 @@ message TFeatureFlags {
optional bool EnableHasPredicatesInResourcePoolClassifiers = 303 [default = true];
optional bool EnableRejectActionInResourcePoolClassifiers = 304 [default = true];
optional bool EnableMoveWithColumnTableReplace = 305 [default = false];
+ optional bool DisableOldSecrets = 306 [default = false];
}
diff --git a/ydb/core/protos/tablet.proto b/ydb/core/protos/tablet.proto
index 2c725df68be..fb66a804784 100644
--- a/ydb/core/protos/tablet.proto
+++ b/ydb/core/protos/tablet.proto
@@ -54,14 +54,15 @@ message TTabletTypes {
BlockStorePartitionDirect = 43;
BlockStoreVolumeDirect = 44;
NbsLoadTablet = 45;
+ DbsController = 46;
// when adding a new tablet type and keeping parse compatibility with the old version
// rename existing reserved item to desired one, and add new reserved item to
// the end of reserved list
- Reserved46 = 46;
Reserved47 = 47;
Reserved48 = 48;
Reserved49 = 49;
+ Reserved50 = 50;
UserTypeStart = 255;
TypeInvalid = -1;
diff --git a/ydb/core/tx/columnshard/columnshard.cpp b/ydb/core/tx/columnshard/columnshard.cpp
index 4495181de3a..02cf8bb0bcf 100644
--- a/ydb/core/tx/columnshard/columnshard.cpp
+++ b/ydb/core/tx/columnshard/columnshard.cpp
@@ -341,6 +341,10 @@ void TColumnShard::UpdateIndexCounters() {
const std::shared_ptr<const TTabletCountersHandle>& counters = Counters.GetTabletCounters();
counters->SetCounter(COUNTER_INDEX_TABLES, Counters.GetPortionIndexCounters()->GetTablesCount());
+ auto diskUsedStats = Counters.GetPortionIndexCounters()->GetTotalStats(TPortionIndexStats::TDiskUsedPortions());
+ counters->SetCounter(COUNTER_DATA_BYTES, diskUsedStats.GetDataBlobBytes());
+ counters->SetCounter(COUNTER_INDEX_BYTES, diskUsedStats.GetIndexBlobBytes());
+
auto insertedStats =
Counters.GetPortionIndexCounters()->GetTotalStats(TPortionIndexStats::TPortionsByType<NOlap::NPortion::EProduced::INSERTED>());
counters->SetCounter(COUNTER_INSERTED_PORTIONS, insertedStats.GetCount());
diff --git a/ydb/core/tx/columnshard/counters/portions.cpp b/ydb/core/tx/columnshard/counters/portions.cpp
index 31ac9b5eda0..38a0d61a310 100644
--- a/ydb/core/tx/columnshard/counters/portions.cpp
+++ b/ydb/core/tx/columnshard/counters/portions.cpp
@@ -27,6 +27,7 @@ namespace NKikimr::NOlap {
void TSimplePortionsGroupInfo::RemovePortion(const TPortionInfo& p) {
BlobBytes.Sub(p.GetTotalBlobBytes());
+ IndexBlobBytes.Sub(p.GetIndexBlobBytes());
RawBytes.Sub(p.GetTotalRawBytes());
Count.Sub(1);
RecordsCount.Sub(p.GetRecordsCount());
@@ -34,6 +35,7 @@ void TSimplePortionsGroupInfo::RemovePortion(const TPortionInfo& p) {
void TSimplePortionsGroupInfo::AddPortion(const TPortionInfo& p) {
BlobBytes.Add(p.GetTotalBlobBytes());
+ IndexBlobBytes.Add(p.GetIndexBlobBytes());
RawBytes.Add(p.GetTotalRawBytes());
Count.Inc();
RecordsCount.Add(p.GetRecordsCount());
diff --git a/ydb/core/tx/columnshard/counters/portions.h b/ydb/core/tx/columnshard/counters/portions.h
index e0acb3e19e7..d7547511158 100644
--- a/ydb/core/tx/columnshard/counters/portions.h
+++ b/ydb/core/tx/columnshard/counters/portions.h
@@ -19,6 +19,7 @@ class TSimplePortionsGroupInfo {
private:
using TCountByChannel = THashMap<ui16, i64>;
TPositiveControlInteger BlobBytes;
+ TPositiveControlInteger IndexBlobBytes;
TPositiveControlInteger RawBytes;
TPositiveControlInteger Count;
TPositiveControlInteger RecordsCount;
@@ -26,6 +27,7 @@ private:
protected:
void Add(const TSimplePortionsGroupInfo& item) {
BlobBytes.Add(item.BlobBytes);
+ IndexBlobBytes.Add(item.IndexBlobBytes);
RawBytes.Add(item.RawBytes);
Count.Add(item.Count);
RecordsCount.Add(item.RecordsCount);
@@ -44,6 +46,17 @@ public:
return BlobBytes.Val();
}
+ ui64 GetIndexBlobBytes() const {
+ return IndexBlobBytes.Val();
+ }
+
+ ui64 GetDataBlobBytes() const {
+ const ui64 blob = BlobBytes.Val();
+ const ui64 index = IndexBlobBytes.Val();
+ AFL_VERIFY(blob >= index)("blob", blob)("index", index);
+ return blob - index;
+ }
+
ui64 GetRawBytes() const {
return RawBytes.Val();
}
@@ -51,6 +64,7 @@ public:
NJson::TJsonValue SerializeToJson() const {
NJson::TJsonValue result = NJson::JSON_MAP;
result.InsertValue("blob_bytes", BlobBytes.Val());
+ result.InsertValue("index_blob_bytes", IndexBlobBytes.Val());
result.InsertValue("raw_bytes", RawBytes.Val());
result.InsertValue("count", Count.Val());
result.InsertValue("records_count", RecordsCount.Val());
@@ -66,8 +80,8 @@ public:
}
TString DebugString() const {
- return TStringBuilder() << "{blob_bytes=" << BlobBytes.Val() << ";raw_bytes=" << RawBytes.Val() << ";count=" << Count.Val()
- << ";records=" << RecordsCount.Val() << "}";
+ return TStringBuilder() << "{blob_bytes=" << BlobBytes.Val() << ";index_blob_bytes=" << IndexBlobBytes.Val()
+ << ";raw_bytes=" << RawBytes.Val() << ";count=" << Count.Val() << ";records=" << RecordsCount.Val() << "}";
}
TSimplePortionsGroupInfo& operator+=(const TSimplePortionsGroupInfo& item) {
@@ -98,6 +112,7 @@ public:
bool IsEmpty() const {
if (!Count.Val()) {
AFL_VERIFY(!BlobBytes.Val())("this", DebugString());
+ AFL_VERIFY(!IndexBlobBytes.Val())("this", DebugString());
AFL_VERIFY(!RawBytes.Val())("this", DebugString());
AFL_VERIFY(!RecordsCount.Val())("this", DebugString());
return true;
diff --git a/ydb/core/tx/replication/controller/secret_resolver.cpp b/ydb/core/tx/replication/controller/secret_resolver.cpp
index 966556d8141..0924762cdd5 100644
--- a/ydb/core/tx/replication/controller/secret_resolver.cpp
+++ b/ydb/core/tx/replication/controller/secret_resolver.cpp
@@ -60,6 +60,9 @@ class TSecretResolver: public TActorBootstrapped<TSecretResolver> {
future.Subscribe([actorSystem, replyActorId](const NThreading::TFuture<NKqp::TEvDescribeSecretsResponse::TDescription>& result) {
actorSystem->Send(replyActorId, new NKqp::TEvDescribeSecretsResponse(result.GetValue()));
});
+ } else if (AppData()->FeatureFlags.GetDisableOldSecrets()) {
+ // Just in case - when we disable old secrets, we'll make sure they are not needed any more
+ return Reply(false, "Usage of old secrets is disabled now. Please use new secrets");
} else {
Send(NMetadata::NProvider::MakeServiceId(SelfId().NodeId()),
new NMetadata::NProvider::TEvAskSnapshot(SnapshotFetcher()));
diff --git a/ydb/core/tx/schemeshard/ut_export_reboots_s3/ya.make b/ydb/core/tx/schemeshard/ut_export_reboots_s3/ya.make
index 0fc638a8da1..33b7777de85 100644
--- a/ydb/core/tx/schemeshard/ut_export_reboots_s3/ya.make
+++ b/ydb/core/tx/schemeshard/ut_export_reboots_s3/ya.make
@@ -2,7 +2,7 @@ UNITTEST_FOR(ydb/core/tx/schemeshard)
FORK_SUBTESTS()
-SPLIT_FACTOR(50)
+SPLIT_FACTOR(200)
REQUIREMENTS(ram:32 cpu:4)
diff --git a/ydb/core/ymq/actor/create_topic_tx.cpp b/ydb/core/ymq/actor/create_topic_tx.cpp
index 9dff4ec8717..d345845c627 100644
--- a/ydb/core/ymq/actor/create_topic_tx.cpp
+++ b/ydb/core/ymq/actor/create_topic_tx.cpp
@@ -27,10 +27,10 @@ Ydb::Topic::CreateTopicRequest BuildCreateTopicTx(
if (params.HasContentBasedDeduplication) {
request.set_content_based_deduplication(params.ContentBasedDeduplication);
- if (params.ContentBasedDeduplication) {
- request.set_partition_write_speed_messages_per_second(NPQ::CONTENT_BASED_DEDUPLICATION_MESSAGE_LIMIT);
- request.set_partition_write_burst_messages(NPQ::CONTENT_BASED_DEDUPLICATION_MESSAGE_BURST);
- }
+ }
+ if (isFifo) {
+ request.set_partition_write_speed_messages_per_second(NPQ::FIFO_PARTITION_WRITE_SPEED_MESSAGES_PER_SECOND);
+ request.set_partition_write_burst_messages(NPQ::FIFO_PARTITION_WRITE_BURST_MESSAGES);
}
auto* partitioningSettings = request.mutable_partitioning_settings();
diff --git a/ydb/core/ymq/actor/set_queue_attributes.cpp b/ydb/core/ymq/actor/set_queue_attributes.cpp
index 8ff37df5205..317367b7183 100644
--- a/ydb/core/ymq/actor/set_queue_attributes.cpp
+++ b/ydb/core/ymq/actor/set_queue_attributes.cpp
@@ -5,7 +5,6 @@
#include "params.h"
#include "serviceid.h"
-#include <ydb/core/persqueue/public/constants.h>
#include <ydb/core/persqueue/public/schema/schema.h>
#include <ydb/core/ymq/base/limits.h>
#include <ydb/core/ymq/base/dlq_helpers.h>
@@ -173,13 +172,6 @@ private:
if (ValidatedAttributes_.ContentBasedDeduplication) {
request.set_set_content_based_deduplication(*ValidatedAttributes_.ContentBasedDeduplication);
- if (*ValidatedAttributes_.ContentBasedDeduplication) {
- request.set_set_partition_write_speed_messages_per_second(NPQ::CONTENT_BASED_DEDUPLICATION_MESSAGE_LIMIT);
- request.set_set_partition_write_burst_messages(NPQ::CONTENT_BASED_DEDUPLICATION_MESSAGE_BURST);
- } else {
- request.set_set_partition_write_speed_messages_per_second(NPQ::DEFAULT_PARTITION_WRITE_SPEED_MESSAGES_PER_SECOND);
- request.set_set_partition_write_burst_messages(NPQ::DEFAULT_PARTITION_WRITE_SPEED_MESSAGES_PER_SECOND);
- }
}
auto* type = consumer->mutable_alter_shared_consumer_type();
diff --git a/ydb/docs/en/core/_assets/resources_weight.drawio b/ydb/docs/en/core/_assets/resources_weight.drawio
index 5e37ed89cbf..6156ef34bda 100644
--- a/ydb/docs/en/core/_assets/resources_weight.drawio
+++ b/ydb/docs/en/core/_assets/resources_weight.drawio
@@ -40,7 +40,7 @@
<mxPoint x="490" y="230" as="targetPoint" />
</mxGeometry>
</mxCell>
- <mxCell id="VCuo0HpmyH9Wwde20rOg-12" value="RESOURCES_WEIGHT&lt;div&gt;=100&lt;/div&gt;" style="edgeLabel;html=1;align=center;verticalAlign=middle;resizable=0;points=[];" parent="VCuo0HpmyH9Wwde20rOg-11" vertex="1" connectable="0">
+ <mxCell id="VCuo0HpmyH9Wwde20rOg-12" value="RESOURCE_WEIGHT&lt;div&gt;=100&lt;/div&gt;" style="edgeLabel;html=1;align=center;verticalAlign=middle;resizable=0;points=[];" parent="VCuo0HpmyH9Wwde20rOg-11" vertex="1" connectable="0">
<mxGeometry x="-0.1738" y="6" relative="1" as="geometry">
<mxPoint x="65" y="-12" as="offset" />
</mxGeometry>
@@ -146,10 +146,10 @@
<mxPoint x="-97" y="7" as="offset" />
</mxGeometry>
</mxCell>
- <mxCell id="VCuo0HpmyH9Wwde20rOg-36" value="&lt;div style=&quot;text-align: center;&quot;&gt;&lt;span style=&quot;color: rgb(0, 0, 0); font-family: Helvetica; font-size: 12px; font-style: normal; font-variant-ligatures: normal; font-variant-caps: normal; font-weight: 400; letter-spacing: normal; orphans: 2; text-indent: 0px; text-transform: none; widows: 2; word-spacing: 0px; -webkit-text-stroke-width: 0px; white-space: nowrap; background-color: rgb(251, 251, 251); text-decoration-thickness: initial; text-decoration-style: initial; text-decoration-color: initial; float: none; display: inline !important;&quot;&gt;Old limit (RESOURCES_WEIGHT=100):&lt;/span&gt;&lt;/div&gt;&lt;span style=&quot;color: rgb(0, 0, 0); font-family: Helvetica; font-size: 12px; font-style: normal; font-variant-ligatures: normal; font-variant-caps: normal; font-weight: 400; letter-spacing: normal; orphans: 2; text-align: center; text-indent: 0px; text-transform: none; widows: 2; word-spacing: 0px; -webkit-text-stroke-width: 0px; white-space: nowrap; background-color: rgb(251, 251, 251); text-decoration-thickness: initial; text-decoration-style: initial; text-decoration-color: initial; display: inline !important; float: none;&quot;&gt;TOTAL_CPU_LIMIT_PERCENT_PER_NODE = 30 = 3 vCPU&lt;/span&gt;&lt;br style=&quot;forced-color-adjust: none; color: rgb(0, 0, 0); font-family: Helvetica; font-size: 12px; font-style: normal; font-variant-ligatures: normal; font-variant-caps: normal; font-weight: 400; letter-spacing: normal; orphans: 2; text-align: center; text-indent: 0px; text-transform: none; widows: 2; word-spacing: 0px; -webkit-text-stroke-width: 0px; white-space: nowrap; background-color: rgb(251, 251, 251); text-decoration-thickness: initial; text-decoration-style: initial; text-decoration-color: initial;&quot;&gt;&lt;span style=&quot;color: rgb(0, 0, 0); font-family: Helvetica; font-size: 12px; font-style: normal; font-variant-ligatures: normal; font-variant-caps: normal; font-weight: 400; letter-spacing: normal; orphans: 2; text-align: center; text-indent: 0px; text-transform: none; widows: 2; word-spacing: 0px; -webkit-text-stroke-width: 0px; white-space: nowrap; background-color: rgb(251, 251, 251); text-decoration-thickness: initial; text-decoration-style: initial; text-decoration-color: initial; display: inline !important; float: none;&quot;&gt;QUERY_CPU_LIMIT_PERCENT_PER_NODE = 50 = 1.5&lt;/span&gt;&lt;div&gt;&lt;span style=&quot;color: rgb(0, 0, 0); font-family: Helvetica; font-size: 12px; font-style: normal; font-variant-ligatures: normal; font-variant-caps: normal; font-weight: 400; letter-spacing: normal; orphans: 2; text-align: center; text-indent: 0px; text-transform: none; widows: 2; word-spacing: 0px; -webkit-text-stroke-width: 0px; white-space: nowrap; background-color: rgb(251, 251, 251); text-decoration-thickness: initial; text-decoration-style: initial; text-decoration-color: initial; display: inline !important; float: none;&quot;&gt;&lt;br&gt;&lt;/span&gt;&lt;/div&gt;&lt;div style=&quot;text-align: center;&quot;&gt;&lt;span style=&quot;text-wrap: nowrap;&quot;&gt;New limit (&lt;/span&gt;&lt;span style=&quot;text-wrap: nowrap; background-color: initial;&quot;&gt;(RESOURCES_WEIGHT=100)&lt;/span&gt;&lt;span style=&quot;text-wrap: nowrap; background-color: initial;&quot;&gt;:&lt;/span&gt;&lt;/div&gt;&lt;div style=&quot;text-align: center;&quot;&gt;&lt;span style=&quot;text-wrap: nowrap;&quot;&gt;TOTAL_CPU_LIMIT_PERCENT_PER_NODE = 70/3 ~= 2.3 vCPU&lt;/span&gt;&lt;br style=&quot;text-wrap: nowrap;&quot;&gt;&lt;span style=&quot;text-wrap: nowrap;&quot;&gt;QUERY_CPU_LIMIT_PERCENT_PER_NODE = 50 ~= 1.15&amp;nbsp; vCPU&lt;/span&gt;&lt;span style=&quot;text-wrap: nowrap;&quot;&gt;&lt;br&gt;&lt;/span&gt;&lt;/div&gt;" style="text;whiteSpace=wrap;html=1;" parent="1" vertex="1">
+ <mxCell id="VCuo0HpmyH9Wwde20rOg-36" value="&lt;div style=&quot;text-align: center;&quot;&gt;&lt;span style=&quot;color: rgb(0, 0, 0); font-family: Helvetica; font-size: 12px; font-style: normal; font-variant-ligatures: normal; font-variant-caps: normal; font-weight: 400; letter-spacing: normal; orphans: 2; text-indent: 0px; text-transform: none; widows: 2; word-spacing: 0px; -webkit-text-stroke-width: 0px; white-space: nowrap; background-color: rgb(251, 251, 251); text-decoration-thickness: initial; text-decoration-style: initial; text-decoration-color: initial; float: none; display: inline !important;&quot;&gt;Old limit (RESOURCE_WEIGHT=100):&lt;/span&gt;&lt;/div&gt;&lt;span style=&quot;color: rgb(0, 0, 0); font-family: Helvetica; font-size: 12px; font-style: normal; font-variant-ligatures: normal; font-variant-caps: normal; font-weight: 400; letter-spacing: normal; orphans: 2; text-align: center; text-indent: 0px; text-transform: none; widows: 2; word-spacing: 0px; -webkit-text-stroke-width: 0px; white-space: nowrap; background-color: rgb(251, 251, 251); text-decoration-thickness: initial; text-decoration-style: initial; text-decoration-color: initial; display: inline !important; float: none;&quot;&gt;TOTAL_CPU_LIMIT_PERCENT_PER_NODE = 30 = 3 vCPU&lt;/span&gt;&lt;br style=&quot;forced-color-adjust: none; color: rgb(0, 0, 0); font-family: Helvetica; font-size: 12px; font-style: normal; font-variant-ligatures: normal; font-variant-caps: normal; font-weight: 400; letter-spacing: normal; orphans: 2; text-align: center; text-indent: 0px; text-transform: none; widows: 2; word-spacing: 0px; -webkit-text-stroke-width: 0px; white-space: nowrap; background-color: rgb(251, 251, 251); text-decoration-thickness: initial; text-decoration-style: initial; text-decoration-color: initial;&quot;&gt;&lt;span style=&quot;color: rgb(0, 0, 0); font-family: Helvetica; font-size: 12px; font-style: normal; font-variant-ligatures: normal; font-variant-caps: normal; font-weight: 400; letter-spacing: normal; orphans: 2; text-align: center; text-indent: 0px; text-transform: none; widows: 2; word-spacing: 0px; -webkit-text-stroke-width: 0px; white-space: nowrap; background-color: rgb(251, 251, 251); text-decoration-thickness: initial; text-decoration-style: initial; text-decoration-color: initial; display: inline !important; float: none;&quot;&gt;QUERY_CPU_LIMIT_PERCENT_PER_NODE = 50 = 1.5&lt;/span&gt;&lt;div&gt;&lt;span style=&quot;color: rgb(0, 0, 0); font-family: Helvetica; font-size: 12px; font-style: normal; font-variant-ligatures: normal; font-variant-caps: normal; font-weight: 400; letter-spacing: normal; orphans: 2; text-align: center; text-indent: 0px; text-transform: none; widows: 2; word-spacing: 0px; -webkit-text-stroke-width: 0px; white-space: nowrap; background-color: rgb(251, 251, 251); text-decoration-thickness: initial; text-decoration-style: initial; text-decoration-color: initial; display: inline !important; float: none;&quot;&gt;&lt;br&gt;&lt;/span&gt;&lt;/div&gt;&lt;div style=&quot;text-align: center;&quot;&gt;&lt;span style=&quot;text-wrap: nowrap;&quot;&gt;New limit (&lt;/span&gt;&lt;span style=&quot;text-wrap: nowrap; background-color: initial;&quot;&gt;(RESOURCE_WEIGHT=100)&lt;/span&gt;&lt;span style=&quot;text-wrap: nowrap; background-color: initial;&quot;&gt;:&lt;/span&gt;&lt;/div&gt;&lt;div style=&quot;text-align: center;&quot;&gt;&lt;span style=&quot;text-wrap: nowrap;&quot;&gt;TOTAL_CPU_LIMIT_PERCENT_PER_NODE = 70/3 ~= 2.3 vCPU&lt;/span&gt;&lt;br style=&quot;text-wrap: nowrap;&quot;&gt;&lt;span style=&quot;text-wrap: nowrap;&quot;&gt;QUERY_CPU_LIMIT_PERCENT_PER_NODE = 50 ~= 1.15&amp;nbsp; vCPU&lt;/span&gt;&lt;span style=&quot;text-wrap: nowrap;&quot;&gt;&lt;br&gt;&lt;/span&gt;&lt;/div&gt;" style="text;whiteSpace=wrap;html=1;" parent="1" vertex="1">
<mxGeometry x="170.00000000000009" y="820" width="360" height="50" as="geometry" />
</mxCell>
- <mxCell id="VCuo0HpmyH9Wwde20rOg-37" value="&lt;div style=&quot;text-align: center;&quot;&gt;&lt;span style=&quot;color: rgb(0, 0, 0); font-family: Helvetica; font-size: 12px; font-style: normal; font-variant-ligatures: normal; font-variant-caps: normal; font-weight: 400; letter-spacing: normal; orphans: 2; text-indent: 0px; text-transform: none; widows: 2; word-spacing: 0px; -webkit-text-stroke-width: 0px; white-space: nowrap; background-color: rgb(251, 251, 251); text-decoration-thickness: initial; text-decoration-style: initial; text-decoration-color: initial; float: none; display: inline !important;&quot;&gt;Old limit (&lt;/span&gt;&lt;span style=&quot;text-wrap: nowrap; background-color: initial;&quot;&gt;(RESOURCES_WEIGHT=200)&lt;/span&gt;&lt;span style=&quot;text-wrap: nowrap;&quot;&gt;:&lt;/span&gt;&lt;/div&gt;&lt;span style=&quot;color: rgb(0, 0, 0); font-family: Helvetica; font-size: 12px; font-style: normal; font-variant-ligatures: normal; font-variant-caps: normal; font-weight: 400; letter-spacing: normal; orphans: 2; text-align: center; text-indent: 0px; text-transform: none; widows: 2; word-spacing: 0px; -webkit-text-stroke-width: 0px; white-space: nowrap; background-color: rgb(251, 251, 251); text-decoration-thickness: initial; text-decoration-style: initial; text-decoration-color: initial; display: inline !important; float: none;&quot;&gt;TOTAL_CPU_LIMIT_PERCENT_PER_NODE = 30 = 3 vCPU&lt;/span&gt;&lt;br style=&quot;forced-color-adjust: none; color: rgb(0, 0, 0); font-family: Helvetica; font-size: 12px; font-style: normal; font-variant-ligatures: normal; font-variant-caps: normal; font-weight: 400; letter-spacing: normal; orphans: 2; text-align: center; text-indent: 0px; text-transform: none; widows: 2; word-spacing: 0px; -webkit-text-stroke-width: 0px; white-space: nowrap; background-color: rgb(251, 251, 251); text-decoration-thickness: initial; text-decoration-style: initial; text-decoration-color: initial;&quot;&gt;&lt;span style=&quot;color: rgb(0, 0, 0); font-family: Helvetica; font-size: 12px; font-style: normal; font-variant-ligatures: normal; font-variant-caps: normal; font-weight: 400; letter-spacing: normal; orphans: 2; text-align: center; text-indent: 0px; text-transform: none; widows: 2; word-spacing: 0px; -webkit-text-stroke-width: 0px; white-space: nowrap; background-color: rgb(251, 251, 251); text-decoration-thickness: initial; text-decoration-style: initial; text-decoration-color: initial; display: inline !important; float: none;&quot;&gt;QUERY_CPU_LIMIT_PERCENT_PER_NODE = 50 = 1.5&lt;/span&gt;&lt;div&gt;&lt;span style=&quot;color: rgb(0, 0, 0); font-family: Helvetica; font-size: 12px; font-style: normal; font-variant-ligatures: normal; font-variant-caps: normal; font-weight: 400; letter-spacing: normal; orphans: 2; text-align: center; text-indent: 0px; text-transform: none; widows: 2; word-spacing: 0px; -webkit-text-stroke-width: 0px; white-space: nowrap; background-color: rgb(251, 251, 251); text-decoration-thickness: initial; text-decoration-style: initial; text-decoration-color: initial; display: inline !important; float: none;&quot;&gt;&lt;br&gt;&lt;/span&gt;&lt;/div&gt;&lt;div style=&quot;text-align: center;&quot;&gt;&lt;span style=&quot;text-wrap: nowrap;&quot;&gt;New limit (&lt;/span&gt;&lt;span style=&quot;text-wrap: nowrap; background-color: initial;&quot;&gt;(RESOURCES_WEIGHT=200)&lt;/span&gt;&lt;span style=&quot;text-wrap: nowrap; background-color: initial;&quot;&gt;:&lt;/span&gt;&lt;/div&gt;&lt;div style=&quot;text-align: center;&quot;&gt;&lt;span style=&quot;text-wrap: nowrap;&quot;&gt;TOTAL_CPU_LIMIT_PERCENT_PER_NODE = 30 = 3 vCPU&lt;/span&gt;&lt;br style=&quot;text-wrap: nowrap;&quot;&gt;&lt;span style=&quot;text-wrap: nowrap;&quot;&gt;QUERY_CPU_LIMIT_PERCENT_PER_NODE = 50 = 1.5 vCPU&lt;/span&gt;&lt;span style=&quot;text-wrap: nowrap;&quot;&gt;&lt;br&gt;&lt;/span&gt;&lt;/div&gt;" style="text;whiteSpace=wrap;html=1;" parent="1" vertex="1">
+ <mxCell id="VCuo0HpmyH9Wwde20rOg-37" value="&lt;div style=&quot;text-align: center;&quot;&gt;&lt;span style=&quot;color: rgb(0, 0, 0); font-family: Helvetica; font-size: 12px; font-style: normal; font-variant-ligatures: normal; font-variant-caps: normal; font-weight: 400; letter-spacing: normal; orphans: 2; text-indent: 0px; text-transform: none; widows: 2; word-spacing: 0px; -webkit-text-stroke-width: 0px; white-space: nowrap; background-color: rgb(251, 251, 251); text-decoration-thickness: initial; text-decoration-style: initial; text-decoration-color: initial; float: none; display: inline !important;&quot;&gt;Old limit (&lt;/span&gt;&lt;span style=&quot;text-wrap: nowrap; background-color: initial;&quot;&gt;(RESOURCE_WEIGHT=200)&lt;/span&gt;&lt;span style=&quot;text-wrap: nowrap;&quot;&gt;:&lt;/span&gt;&lt;/div&gt;&lt;span style=&quot;color: rgb(0, 0, 0); font-family: Helvetica; font-size: 12px; font-style: normal; font-variant-ligatures: normal; font-variant-caps: normal; font-weight: 400; letter-spacing: normal; orphans: 2; text-align: center; text-indent: 0px; text-transform: none; widows: 2; word-spacing: 0px; -webkit-text-stroke-width: 0px; white-space: nowrap; background-color: rgb(251, 251, 251); text-decoration-thickness: initial; text-decoration-style: initial; text-decoration-color: initial; display: inline !important; float: none;&quot;&gt;TOTAL_CPU_LIMIT_PERCENT_PER_NODE = 30 = 3 vCPU&lt;/span&gt;&lt;br style=&quot;forced-color-adjust: none; color: rgb(0, 0, 0); font-family: Helvetica; font-size: 12px; font-style: normal; font-variant-ligatures: normal; font-variant-caps: normal; font-weight: 400; letter-spacing: normal; orphans: 2; text-align: center; text-indent: 0px; text-transform: none; widows: 2; word-spacing: 0px; -webkit-text-stroke-width: 0px; white-space: nowrap; background-color: rgb(251, 251, 251); text-decoration-thickness: initial; text-decoration-style: initial; text-decoration-color: initial;&quot;&gt;&lt;span style=&quot;color: rgb(0, 0, 0); font-family: Helvetica; font-size: 12px; font-style: normal; font-variant-ligatures: normal; font-variant-caps: normal; font-weight: 400; letter-spacing: normal; orphans: 2; text-align: center; text-indent: 0px; text-transform: none; widows: 2; word-spacing: 0px; -webkit-text-stroke-width: 0px; white-space: nowrap; background-color: rgb(251, 251, 251); text-decoration-thickness: initial; text-decoration-style: initial; text-decoration-color: initial; display: inline !important; float: none;&quot;&gt;QUERY_CPU_LIMIT_PERCENT_PER_NODE = 50 = 1.5&lt;/span&gt;&lt;div&gt;&lt;span style=&quot;color: rgb(0, 0, 0); font-family: Helvetica; font-size: 12px; font-style: normal; font-variant-ligatures: normal; font-variant-caps: normal; font-weight: 400; letter-spacing: normal; orphans: 2; text-align: center; text-indent: 0px; text-transform: none; widows: 2; word-spacing: 0px; -webkit-text-stroke-width: 0px; white-space: nowrap; background-color: rgb(251, 251, 251); text-decoration-thickness: initial; text-decoration-style: initial; text-decoration-color: initial; display: inline !important; float: none;&quot;&gt;&lt;br&gt;&lt;/span&gt;&lt;/div&gt;&lt;div style=&quot;text-align: center;&quot;&gt;&lt;span style=&quot;text-wrap: nowrap;&quot;&gt;New limit (&lt;/span&gt;&lt;span style=&quot;text-wrap: nowrap; background-color: initial;&quot;&gt;(RESOURCE_WEIGHT=200)&lt;/span&gt;&lt;span style=&quot;text-wrap: nowrap; background-color: initial;&quot;&gt;:&lt;/span&gt;&lt;/div&gt;&lt;div style=&quot;text-align: center;&quot;&gt;&lt;span style=&quot;text-wrap: nowrap;&quot;&gt;TOTAL_CPU_LIMIT_PERCENT_PER_NODE = 30 = 3 vCPU&lt;/span&gt;&lt;br style=&quot;text-wrap: nowrap;&quot;&gt;&lt;span style=&quot;text-wrap: nowrap;&quot;&gt;QUERY_CPU_LIMIT_PERCENT_PER_NODE = 50 = 1.5 vCPU&lt;/span&gt;&lt;span style=&quot;text-wrap: nowrap;&quot;&gt;&lt;br&gt;&lt;/span&gt;&lt;/div&gt;" style="text;whiteSpace=wrap;html=1;" parent="1" vertex="1">
<mxGeometry x="555" y="820" width="360" height="50" as="geometry" />
</mxCell>
</root>
diff --git a/ydb/docs/en/core/_assets/resources_weight.png b/ydb/docs/en/core/_assets/resources_weight.png
index cb9f2d19145..9853f0606a4 100644
--- a/ydb/docs/en/core/_assets/resources_weight.png
+++ b/ydb/docs/en/core/_assets/resources_weight.png
Binary files differ
diff --git a/ydb/docs/en/core/_includes/olap-data-types.md b/ydb/docs/en/core/_includes/olap-data-types.md
deleted file mode 100644
index 0c41154a604..00000000000
--- a/ydb/docs/en/core/_includes/olap-data-types.md
+++ /dev/null
@@ -1,24 +0,0 @@
-| Data type | Can be used in<br/>column-oriented tables | Can be used<br/>as primary key |
----|---|---
-| `Bool` | ✓ | ✓ |
-| `Int8` | ✓ | ☓ |
-| `Int16` | ✓ | ☓ |
-| `Int32` | ✓ | ✓ |
-| `Int64` | ✓ | ✓ |
-| `Uint8` | ✓ | ✓ |
-| `Uint16` | ✓ | ✓ |
-| `Uint32` | ✓ | ✓ |
-| `Uint64` | ✓ | ✓ |
-| `Float` | ✓ | ☓ |
-| `Double` | ✓ | ☓ |
-| `Decimal` | ✓ | ☓ |
-| `String` | ✓ | ✓ |
-| `Utf8` | ✓ | ✓ |
-| `Json` | ✓ | ☓ |
-| `JsonDocument` | ✓ | ☓ |
-| `Yson` | ✓ | ☓ |
-| `Uuid` | ☓ | ☓ |
-| `Date` | ✓ | ✓ |
-| `Datetime` | ✓ | ✓ |
-| `Timestamp` | ✓ | ✓ |
-| `Interval` | ☓ | ☓ |
diff --git a/ydb/docs/en/core/concepts/_includes/secondary_indexes.md b/ydb/docs/en/core/concepts/_includes/secondary_indexes.md
index 5ef6631d223..a9259b5abf4 100644
--- a/ydb/docs/en/core/concepts/_includes/secondary_indexes.md
+++ b/ydb/docs/en/core/concepts/_includes/secondary_indexes.md
@@ -1,90 +1,94 @@
-# Secondary Indexes
+# Secondary indexes
-{{ ydb-short-name }} automatically creates a primary key index, which is why selection by primary key is always efficient, affecting only the rows needed. Selections by criteria applied to one or more non-key columns typically result in a full table scan. To make these selections efficient, use _secondary indexes_ — global structures backed by a separate index table.
+In {{ ydb-short-name }}, an index is automatically created on the primary key, so queries with a condition on the primary key are always executed efficiently, affecting only the required rows. A query with a condition on one or more non-key columns typically results in a full table scan. To make such queries efficient, you should use _secondary indexes_ — global structures with a separate index table.
-[Local indexes](../glossary.md#local-index) are a separate kind of auxiliary structure: they are stored with table data and applied in the storage layer on read, without materializing a separate index table (see [Local indexes](#bloom-skip-index) below).
+Separately, there are [local indexes](../glossary.md#local-index): auxiliary structures that are stored together with the table data and are used when reading on the storage side, without materializing a separate index table (see the [Local indexes](#bloom-skip-index) section below).
-The current version of {{ ydb-short-name }} implements _synchronous_ and _asynchronous_ global secondary indexes. Each index is a hidden table that is updated:
+In the current version of {{ ydb-short-name }}, _synchronous_ and _asynchronous_ global secondary indexes are implemented. Each index is a hidden table that is updated:
-* For synchronous indexes: Transactionally when the main table changes.
-* For asynchronous indexes: In the background while getting the necessary changes from the main table.
+* for synchronous indexes — transactionally when the main table is modified;
+* for asynchronous indexes — in the background, receiving the necessary changes from the main table.
-When a user sends an SQL query to insert, modify, or delete data, the database transparently generates commands to modify the index table. A table may have multiple secondary indexes. An index may include multiple columns, and the sequence of columns in an index matters. A single column may be included in multiple indexes. In addition to the specified columns, every index implicitly stores the table primary key columns to enable navigation from an index record to the table row.
+When a user sends an SQL query to insert, modify, or delete data, the database transparently generates commands to modify the index table. A table can have multiple secondary indexes. An index can include multiple columns, and the order of columns in the index is important. A single column can be included in multiple indexes. In addition to the specified columns, the values of the table's primary key columns are always implicitly stored in the index, so that from a found index entry you can navigate to the table entry.
-## Synchronous Secondary Index {#sync}
+## Synchronous secondary index {#sync}
-A synchronous index is updated simultaneously with the table that it indexes. This index ensures [strict consistency](https://en.wikipedia.org/wiki/Consistency_model) through [distributed transactions](../transactions.md#distributed-tx). While reads and blind writes to a table with no index can be performed without a planning stage, significantly reducing delays, such optimization is impossible when writing data to a table with a synchronous index.
+A synchronous index is updated simultaneously with the table it indexes. Such an index provides [strong data consistency](https://en.wikipedia.org/wiki/Consistency_model) and uses the [distributed transactions](../transactions.md#distributed-tx) mechanism for this. Thus, while read and blind write operations on a table without an index can be performed without a planning stage, thereby significantly reducing latency, such optimization is not possible for writes to a table with a synchronous index.
-## Asynchronous Secondary Index {#async}
+## Asynchronous secondary index {#async}
-Unlike a synchronous index, an asynchronous index doesn't use distributed transactions. Instead, it receives changes from an indexed table in the background. Write transactions to a table using this index are performed with no planning overheads due to reduced guarantees: an asynchronous index provides [eventual consistency](https://en.wikipedia.org/wiki/Eventual_consistency), but no strict consistency. You can only use asynchronous indexes in read transactions in [Stale Read Only](../transactions.md#modes) mode.
+An asynchronous index, unlike a synchronous one, does not use the distributed transaction mechanism, but receives changes from the indexed table in the background. Write transactions to a table with such an index are performed without additional planning overhead, at the cost of reduced guarantees: an asynchronous index provides [eventual data consistency](https://en.wikipedia.org/wiki/Eventual_consistency), but not strong consistency. Using an asynchronous index in read transactions is only possible in [Stale Read Only](../transactions.md#modes) mode.
-## Covering Secondary Index {#covering}
+## Covering secondary index {#covering}
-You can copy the contents of columns into a covering index. This eliminates the need to read data from the main table when performing reads by index and significantly reduces delays. At the same time, such denormalization leads to increased usage of disk space and may slow down inserts and updates due to the need for additional data copying.
+It is possible to copy the contents of columns into the index (covering index), thus eliminating the need to read from the main table in index read operations, which significantly reduces latency. At the same time, such denormalization leads to increased disk space consumption and possible slowdown of insert and update operations due to the need for additional data copying.
-## Unique Secondary Index {#unique}
+## Unique secondary index {#unique}
-This type of index enforces unique constraint behavior and, just like a regular secondary index, allows efficient point lookup queries. {{ ydb-short-name }} uses it to perform additional checks, ensuring that each distinct value in the indexed column set appears in the table no more than once. If a modifying query violates the constraint, it is aborted with a `PRECONDITION_FAILED` status. Therefore, client code must be prepared to handle this status.
+This type of index implements the semantics of a unique value in a column or set of columns, and, like other indexes, allows efficient point reads on the set of indexed columns. {{ ydb-short-name }} uses it to perform additional checks to ensure that each unique value of the indexed columns appears in the table no more than once. If a modifying query violates this constraint, it is aborted with the status `PRECONDITION_FAILED`. Therefore, user code must be prepared to handle this status.
-A unique secondary index is a synchronous index, so from a transactional perspective, the update process is the same as for the [synchronous secondary index](#sync) described above.
+A unique secondary index is a synchronous index, so from a transactional perspective, its update process is the same as that of the [synchronous secondary index](#sync) described above.
-## Vector Index
+## Vector index {#vector}
-[Vector Index](../../dev/vector-indexes.md) is a special type of secondary index.
+[Vector index](../../dev/vector-indexes.md) is a special type of secondary index.
-Unlike traditional secondary indexes, which optimize equality or range searches, vector indexes allow [vector search](../query_execution/vector_search.md) based on distance or similarity functions.
+Unlike traditional secondary indexes, which optimize equality or range search, vector indexes allow performing [vector search](../query_execution/vector_search.md) based on distance or similarity functions.
-## Fulltext Index
+## Full-text index {#fulltext}
-[Fulltext index](../../dev/fulltext-indexes.md) is a special type of secondary index.
+[Full-text index](../../dev/fulltext-indexes.md) is a special type of secondary index.
-Unlike traditional secondary indexes, which optimize equality or range searches, fulltext indexes allow scalable text search by words and phrases (and, with n-grams, by substrings). See also: [Fulltext search](../query_execution/fulltext_search.md).
+Unlike traditional secondary indexes, which optimize search by equality or range, full-text indexes allow scalable text search for words and phrases (and when using [N-grams](https://en.wikipedia.org/wiki/N-gram), also for substrings). See also: [Full-text search](../query_execution/fulltext_search.md).
-## Local indexes {#bloom-skip-index}
+## JSON-index {#json}
+
+[JSON index](../../dev/json-indexes.md) is a special type of secondary index, like full-text index — both are built on top of an [inverted index](https://en.wikipedia.org/wiki/Inverted_index), but use different tokenizers.
-[Local indexes](../query_execution/local_indexes.md) are auxiliary structures stored together with table data and applied while reading in the storage layer. They do not materialize a separate index table. Currently, [Bloom skip indexes](../../dev/bloom-skip-indexes.md) are implemented; other kinds are planned.
+JSON indexes allow you to speed up predicates with the [JSON_EXISTS](../../yql/reference/builtins/json.md) and [JSON_VALUE](../../yql/reference/builtins/json.md) functions on the content of a column of type `Json` or `JsonDocument`. The index is built by splitting JSON documents into path tokens and pairs of the form "path + value", which allows you to find matching rows by [JsonPath](../../yql/reference/builtins/json.md#jsonpath) paths without a full table scan. See also: [JSON search](../query_execution/json_search.md).
-## Creating a Secondary Index Online {#index-add}
+## Local indexes {#bloom-skip-index}
-{{ ydb-short-name }} lets you create new and delete existing secondary indexes without stopping the service. For a single table, you can only create one index at a time.
+[Local indexes](../query_execution/local_indexes.md) — auxiliary structures stored together with table data and used for reads on the storage side. They do not materialize a separate index table. Currently, [Bloom indexes](../../dev/bloom-skip-indexes.md) are implemented, and other types are planned in the future.
-Online index creation consists of the following steps:
+## Online creation of a secondary index {#index-add}
-1. Taking a snapshot of a data table and creating an index table marked that writes are available.
+In {{ ydb-short-name }}, you can create a secondary index and delete an existing secondary index without stopping service. For one table, only one index can be created at a time.
- After this step, write transactions are distributed, writing to the main table and the index, respectively. The index is not yet available to the user.
+The online index creation operation consists of the following steps:
-1. Reading the snapshot of the main table and writing data to the index.
+1. Taking a snapshot of the table with data, creating an index table marked as available for writes.
- "Writes to the past" are implemented: situations where data updates in step 1 change the data written in step 2 are resolved.
+ After this step, write transactions become distributed, and data is written to the main table and the index. The index is not yet available to the user.
+2. Reading a snapshot of the main table and writing to the index.
-1. Publishing the results and deleting the snapshot.
+ Implements "write to the past": situations are allowed where data updates in step 1 change data written in step 2.
+3. Publishing the result, deleting the snapshot.
The index is ready to use.
Possible impact on user transactions:
-* There may be an increase in delays because transactions are now distributed (when creating a synchronous index).
-* There may be an enhanced background of `OVERLOADED` errors because index table automatic shard splitting is actively running during data writes.
+* Increased latency may be observed because transactions become distributed (when creating a synchronous index).
+* An increased background of `OVERLOADED` errors is possible because automatic partitioning of index table shards is actively working during data writes.
{% note info %}
-The data write rate is chosen to minimize the impact of the write process on user transactions. To control the rate, configure limits for the corresponding queue in the [resource broker](../../reference/configuration/resource_broker_config.md#resource-broker-config).
+The data write speed is chosen to minimize the impact of the write process on user transactions. To control the speed, configure limits for the corresponding queue of the [resource broker](../../reference/configuration/resource_broker_config.md#resource-broker-config).
{% endnote %}
-Creating an index is an asynchronous operation. If the client-server connection is interrupted after the operation has started, index building continues. You can manage asynchronous operations using the {{ ydb-short-name }} CLI.
+Creating an index is an asynchronous operation. If a client-server connection break occurs after starting the operation, the index building will continue. You can manage the asynchronous operation via {{ ydb-short-name }} CLI.
-## Creating and Deleting Secondary Indexes {#ddl}
+## Creating and deleting secondary indexes {#ddl}
A secondary index can be:
-- Created when creating a table with the YQL [`CREATE TABLE`](../../yql/reference/syntax/create_table/index.md) statement.
-- Added to an existing table with the YQL [`ALTER TABLE`](../../yql/reference/syntax/alter_table/index.md) statement or the YDB CLI [`table index add`](../../reference/ydb-cli/commands/secondary_index.md#add) command.
-- Deleted from an existing table with the YQL [`ALTER TABLE`](../../yql/reference/syntax/alter_table/index.md) statement or the YDB CLI [`table index drop`](../../reference/ydb-cli/commands/secondary_index.md#drop) command.
-- Deleted together with the table using the YQL [`DROP TABLE`](../../yql/reference/syntax/drop_table.md) statement or the YDB CLI `table drop` command.
+- Created when creating a table using the YQL [CREATE TABLE](../../yql/reference/syntax/create_table/index.md) command.
+- Added to an existing table by the YQL [ALTER TABLE](../../yql/reference/syntax/alter_table/index.md) command or by the {{ ydb-short-name }} CLI [table index add](../../reference/ydb-cli/commands/secondary_index.md#add) command.
+- Deleted from an existing table by the YQL command [ALTER TABLE](../../yql/reference/syntax/alter_table/index.md) or by the {{ ydb-short-name }} CLI command [table index drop](../../reference/ydb-cli/commands/secondary_index.md#drop).
+- Deleted together with the table by the YQL command [DROP TABLE](../../yql/reference/syntax/drop_table.md) or by the {{ ydb-short-name }} CLI command `table drop`.
-## Using Secondary Indexes {#use}
+## Using secondary indexes {#use}
-For detailed information on using secondary indexes in applications, refer to the [relevant article](../../dev/secondary-indexes.md) in the documentation section for developers.
+Detailed information on using secondary indexes in applications is available in [the article about them](../../dev/secondary-indexes.md) in the developer documentation section.
diff --git a/ydb/docs/en/core/concepts/glossary.md b/ydb/docs/en/core/concepts/glossary.md
index 9a947d1053d..d74ff77d9d7 100644
--- a/ydb/docs/en/core/concepts/glossary.md
+++ b/ydb/docs/en/core/concepts/glossary.md
@@ -4,13 +4,13 @@ This article provides an overview of terms and definitions used in {{ ydb-short-
## Key terminology {#key-terminology}
-This section describes terms that are useful to anyone working with {{ ydb-short-name }}, regardless of their role and use case.
+This section describes terms that are useful to anyone working with {{ ydb-short-name }}, regardless of their role or use case.
### Cluster {#cluster}
-A **cluster** {{ ydb-short-name }} is a set of interconnected [nodes](#node) {{ ydb-short-name }} that exchange data to execute user queries and reliably store data. These nodes form one of the supported [cluster topologies](#topology), which directly affects its reliability and performance characteristics.
+A **cluster** {{ ydb-short-name }} is a set of interconnected [nodes](#node) {{ ydb-short-name }} that exchange data to execute user queries and ensure reliable data storage. These nodes form one of the supported [cluster topologies](#topology), which directly affects its reliability and performance characteristics.
-Clusters {{ ydb-short-name }} are multi-tenant and can contain multiple isolated [databases](#database).
+Clusters {{ ydb-short-name }} are multi-tenant and can contain several isolated [databases](#database).
### Database {#database}
@@ -20,31 +20,31 @@ Another important characteristic of {{ ydb-short-name }} databases is that they
### Node {#node}
-{{ ydb-short-name }} **node** is a server process that runs an executable file called `ydbd`. Multiple nodes {{ ydb-short-name }} can run on a single physical server or virtual machine, which is common practice. Thus, in the context of {{ ydb-short-name }}, nodes are **not** synonymous with hosts.
+{{ ydb-short-name }} A **node** is a server process that runs an executable called `ydbd`. Multiple nodes {{ ydb-short-name }} can run on a single physical server or virtual machine, which is common practice. Thus, in the context of {{ ydb-short-name }}, nodes are **not** synonymous with hosts.
-Since {{ ydb-short-name }} uses a storage and compute separation approach, `ydbd` has several operating modes that determine the node type. The available node types are described below.
+Since {{ ydb-short-name }} uses an approach with separate storage and compute layers (storage and compute separation), `ydbd` has several operation modes that define the node type. The available node types are described below.
#### Database node {#database-node}
-**Database nodes** (also known as **tenant nodes** or **compute nodes**) process user queries addressed to a specific logical [database](#database). Their state resides only in RAM and can be restored from [distributed storage](#distributed-storage). The set of database nodes of a given [cluster {{ ydb-short-name }}](topology.md) can be considered the compute layer of that cluster. Thus, adding database nodes and allocating additional resources (CPU and RAM) to them are the main ways to increase the compute resources of a database.
+**Database nodes** (also known as **tenant nodes** or **compute nodes**) process user queries addressed to a specific logical [database](#database). Their state is stored only in RAM and can be restored from [distributed storage](#distributed-storage). The set of database nodes of a given [cluster {{ ydb-short-name }}](topology.md) can be considered the compute layer of that cluster. Thus, adding database nodes and allocating additional resources (CPU and RAM) to them are the main ways to increase the compute resources of a database.
The main role of database nodes is to run various [tablets](#tablet) and [actors](#actor), as well as to receive incoming requests over the network.
#### Storage node {#storage-node}
-**Storage nodes** are stateful nodes responsible for long-term storage of data fragments. The set of storage nodes in a given [cluster {{ ydb-short-name }}](#cluster) is called [distributed storage](#distributed-storage) and can be considered as the storage layer of that cluster. Thus, adding additional storage nodes and their disks is the primary way to increase the cluster's storage capacity and I/O throughput.
+**Storage nodes** are stateful nodes responsible for long-term storage of data fragments. The set of storage nodes in a given [cluster {{ ydb-short-name }}](#cluster) is called [distributed storage](#distributed-storage) and can be considered as the storage layer of that cluster. Thus, adding additional storage nodes and their disks is the primary way to increase the storage capacity and I/O throughput of the cluster.
#### Hybrid node {#hybrid-mode}
-A **hybrid node** is a process that simultaneously performs both the roles of a [database node](#database-node) and a [storage node](#storage-node). Hybrid nodes are often used for development purposes. For example, you can run a container with a full-featured {{ ydb-short-name }} containing only one `ydbd` process in hybrid mode. They are rarely used in production environments.
+A **Hybrid node** is a process that simultaneously performs both the roles of a [database node](#database-node) and a [storage node](#storage-node). Hybrid nodes are often used for development purposes. For example, you can run a container with a full-featured {{ ydb-short-name }} containing only one `ydbd` process in hybrid mode. They are rarely used in production environments.
#### Static node {#static-node}
-**Static nodes** are configured manually during initial cluster initialization or reconfiguration. Typically, they serve as [storage nodes](#storage-node), but technically it is possible to configure them as [database nodes](#database-node).
+**Static nodes** are configured manually during initial cluster initialization or reconfiguration. Typically, they serve as [storage nodes](#storage-node), but they can technically be configured as [database nodes](#database-node).
#### Dynamic node {#dynamic}
-**Dynamic nodes** are added and removed from the cluster on the fly. They can only serve as [database nodes](#database-node).
+**Dynamic nodes** are added and removed from the cluster on the fly. They can only act as [database nodes](#database-node).
### Distributed storage {#distributed-storage}
@@ -54,45 +54,45 @@ Many terms related to the [implementation of distributed storage](#distributed-s
### Storage group {#storage-group}
-A **storage group** is a place for reliable data storage, similar to [RAID](https://en.wikipedia.org/wiki/RAID), but using disks from multiple servers. Depending on the selected [cluster topology](#topology), storage groups use different algorithms to ensure high availability, similar to [standard RAID levels](https://en.wikipedia.org/wiki/Standard_RAID_levels).
+**Storage group**, **distributed storage group**, or **Blob storage group** is a place for reliable data storage, similar to [RAID](https://en.wikipedia.org/wiki/RAID), but using disks from multiple servers. Depending on the selected [cluster topology](#topology), storage groups use different algorithms to ensure high availability, similar to [standard RAID levels](https://en.wikipedia.org/wiki/Standard_RAID_levels).
[Distributed storage](#distributed-storage) typically manages a large number of relatively small storage groups. Each group can be assigned to a specific [database](#database) to increase the disk space capacity and I/O throughput available to that database.
-[Static](#static-group) and [dynamic](#dynamic-group) storage groups are physical, meaning their data is placed directly on [VDisks](#vdisk).
+[Static](#static-group) and [dynamic](#dynamic-group) storage groups are physical, meaning their data is placed directly on [VDisk](#vdisk)s.
#### Static group {#static-group}
-A **static group** is a special [storage group](#storage-group) created during the initial cluster deployment. Its main role is to store data of system [tablets](#tablet), which can be considered as cluster-level metadata.
+A **static group** is a special [storage group](#storage-group) created during the initial deployment of the cluster. Its main role is to store data of system [tablets](#tablet), which can be considered as cluster-level metadata.
A static group may require special attention during major cluster maintenance, such as decommissioning an [availability zone](#regions-az).
#### Dynamic group {#dynamic-group}
-Ordinary storage groups that are not [static](#static-group) are called **dynamic groups**. They are called dynamic because they can be created and deleted on the fly while the [cluster](#cluster) is running.
+Ordinary storage groups that are not [static](#static-group) are called **dynamic groups**. They are called dynamic because they can be created and removed on the fly while the [cluster](#cluster) is running.
#### Virtual storage group {#virtual-storage-groups}
-A **virtual storage group** is an entity that is not actually a [storage group](#storage-group) but appears to be one from the outside (provides a similar external interface). It can store its data in other storage groups or in S3.
+A **virtual storage group** is an entity that is not actually a [storage group](#storage-group) but appears as one from the outside (provides a similar external interface). It can store its data in other storage groups or in S3.
### Storage pool {#storage-pool}
-A **storage pool** is a set of data storage devices with similar characteristics. Each storage pool is assigned a unique name within the {{ ydb-short-name }} cluster. Technically, each storage pool consists of multiple physical disks ( [PDisk](#pdisk)). Each [storage group](#storage-group) is created in a specific storage pool, which determines the performance characteristics of the storage group through the selection of appropriate storage devices. Typically, separate storage pools are created for devices of different types (e.g., NVMe, SSD, and HDD) or for specific models of these devices that have different capacity and access speed.
+A **storage pool** is a set of data storage devices with similar characteristics. Each storage pool is assigned a unique name within the {{ ydb-short-name }} cluster. Technically, each storage pool consists of multiple physical disks ( [PDisk](#pdisk)). Each [storage group](#storage-group) is created in a specific storage pool, which determines the performance characteristics of the storage group through the selection of appropriate storage devices. Typically, separate storage pools are created for devices of different types (e.g., NVMe, SSD, and HDD) or for specific models of these devices with different capacity and access speed.
### Actor {#actor}
-The [actor model](https://en.wikipedia.org/wiki/Actor_model) is one of the main approaches to execution parallelism used in {{ ydb-short-name }}. In this model, **actors** are lightweight user-space processes that can have and modify their private state, but can only influence each other indirectly through message passing. {{ ydb-short-name }} has its own implementation of this model, which is described [below](#actor-implementation).
+The [Actor model](https://en.wikipedia.org/wiki/Actor_model) is one of the main approaches to execution parallelism used in {{ ydb-short-name }}. In this model, **actors** are lightweight user-space processes that can have and modify their private state, but can only influence each other indirectly through message passing. {{ ydb-short-name }} has its own implementation of this model, which is described [below](#actor-implementation).
In {{ ydb-short-name }}, actors with reliably persisted state are called [tablets](#tablet).
### Tablet {#tablet}
-A **tablet** is one of the main building blocks and abstractions of {{ ydb-short-name }}. It represents an entity responsible for a relatively small segment of user or system data. Typically, a tablet manages up to several gigabytes of data, but some types of tablets can handle larger volumes.
+A **tablet** is one of the main building blocks and abstractions of {{ ydb-short-name }}. It represents an entity responsible for a relatively small segment of user or system data. Typically, a tablet manages up to several gigabytes of data, although some types of tablets can handle larger volumes.
For example, a [string user table](#row-oriented-table) is managed by one or more tablets of type [DataShard](#data-shard), with each tablet responsible for a continuous range of [primary keys](#primary-key) and their corresponding data.
End users who send queries to the {{ ydb-short-name }} cluster for execution do not need to know the details of tablets, their types, or how they work, but this knowledge can be useful, for example, for performance optimization.
-Technically, tablets are [actors](#actor) with state reliably stored in [distributed storage](#distributed-storage). This state allows the tablet to continue operating on another [database node](#database-node) if the previous one fails or becomes overloaded.
+Technically, tablets are [actors](#actor) with state reliably stored in [distributed storage](#distributed-storage). This state allows a tablet to continue operating on a different [database node](#database-node) if the previous one fails or becomes overloaded.
[Tablet implementation details](#tablet-implementation) and related terms, as well as [main tablet types](#tablet-types), are discussed below.
@@ -100,7 +100,7 @@ Technically, tablets are [actors](#actor) with state reliably stored in [distrib
{{ ydb-short-name }} implements **transactions** at two main levels:
-* [Local database](#local-database) and the rest of the [tablet infrastructure](#tablet-implementation) allow [tablets](#tablet) to manipulate their state using **local transactions** with [serializable isolation level](https://en.wikipedia.org/wiki/Isolation_(database_systems)#Serializable_(%D1%83%D0%BF%D0%BE%D1%80%D1%8F%D0%B4%D0%BE%D1%87%D0%B8%D0%B2%D0%B0%D0%B5%D0%BC%D0%BE%D1%81%D1%82%D1%8C)). Technically, they are not local to a single node, since this state is stored remotely in [distributed storage](#distributed-storage).
+* [Local database](#local-database) and the rest of the [tablet infrastructure](#tablet-implementation) allow [tablets](#tablet) to manipulate their state using **local transactions** with [serializable isolation level](https://en.wikipedia.org/wiki/Isolation_(database_systems)#Serializable). Technically, they are not local to a single node, since this state is stored remotely in [distributed storage](#distributed-storage).
* In the context of {{ ydb-short-name }}, the term **distributed transactions** usually refers to transactions that span multiple tablets. For example, transactions between tables or even rows of the same table are often distributed.
* **Single-shard** transactions cover one tablet and execute faster. For example, transactions between rows of the same table partition are often single-shard.
@@ -114,79 +114,59 @@ The implementation of distributed transactions is discussed in a separate articl
### Sessions
-Logical connections to the database that store the context needed for executing queries and managing transactions. Sessions are described in more detail in the section [{#T}](query_execution/execution_process.md#sessions).
+Logical connections to the database that store the context needed for executing queries and managing transactions. Sessions are described in more detail in the section [{#T}](query_execution/index.md#sessions).
### Client-side timeout {#client-timeout}
-**Client-side timeout** — a time limit that the application or {{ ydb-short-name }} SDK waits for a database operation to complete (for example, executing a query or receiving a response via a gRPC call). After this time expires, the client usually aborts the wait: closes the connection or data stream, receives a transport or SDK error — before the server has returned an explicit response (see [codes](../reference/ydb-sdk/ydb-status-codes.md) of the {{ ydb-short-name }} server response).
+**Client-side timeout** — a time limit that an application or {{ ydb-short-name }} SDK waits for a database operation to complete (for example, executing a query or receiving a response via a gRPC call). After this time expires, the client usually aborts the wait: closes the connection or data stream, receives a transport or SDK error — even before the server has returned an explicit response (see {{ ydb-short-name }} server [response codes](../reference/ydb-sdk/ydb-status-codes.md)).
-If the client-side timeout is shorter than the execution time of the query on the {{ ydb-short-name }} side, then due to the specifics of query processing in the cluster, a query aborted on the client may continue to execute on the server for some time. If such a situation occurs on a large scale, the server becomes overloaded with queries for which the client is not waiting for a response. Therefore, frequent retries of the same query immediately after a timeout can exacerbate the overload. For more details, see the articles [{#T}](../troubleshooting/performance/queries/retry-cascade.md) and [{#T}](../troubleshooting/performance/queries/overloaded-errors.md); retry policies in the SDK are described in the section [{#T}](../reference/ydb-sdk/error_handling.md).
-
-### Transaction retry {#transaction-retry}
-
-**Transaction retry** — a client practice of re-executing a [transaction](#transactions) entirely from the beginning upon a retryable error (e.g., a temporary network failure or optimistic locking conflict). In {{ ydb-short-name }}, retries should be performed at the transaction level, not at the level of individual queries within it. Built-in retry policies in the {{ ydb-short-name }} SDK and integrations (e.g., [spring-ydb-retry](../integrations/spring/spring-retry.md)) implement this approach. For more details, see [{#T}](../reference/ydb-sdk/error_handling.md).
-
-### Exponential backoff {#exponential-backoff}
-
-**Exponential backoff** (also known as **backoff**) — a strategy for pausing between [transaction retry](#transaction-retry) attempts: the wait interval increases exponentially with each attempt, usually with an upper limit. The {{ ydb-short-name }} SDK often uses two levels of backoff — fast and slow — depending on the error type. For more details, see [{#T}](../reference/ydb-sdk/error_handling.md#handling-retryable-errors).
-
-### Jitter {#jitter}
-
-**Jitter** is a small random variation added to [exponential backoff](#exponential-backoff) delays. It helps avoid simultaneous retries by many clients after a common failure (a "retry storm") and distributes the load more evenly.
-
-### Idempotency {#idempotency}
-
-**Idempotency** is a property of an operation: repeated execution produces the same effect as a single execution (for example, `UPSERT` with a deterministic primary key or read operations). [Transaction retries](#transaction-retry) are safe only for idempotent operations or for retry errors where the server guarantees that the transaction was not committed. SDKs and client libraries {{ ydb-short-name }} can extend the set of retryable status codes if the calling code marks the operation as idempotent.
-
-### Transaction interceptor {#transaction-interceptor}
-
-**Transaction interceptor** is a Spring Framework component that wraps methods annotated with `@Transactional` and manages transaction boundaries. Modules like [spring-ydb-retry](../integrations/spring/spring-retry.md) replace the standard Spring interceptor, adding [transaction retry](#transaction-retry) logic around transactional methods.
+If the client-side timeout is shorter than the query execution time on the {{ ydb-short-name }} side, then due to the specifics of query processing in the cluster, a query interrupted on the client side may continue to execute on the server for some time. If this situation occurs on a large scale, the server becomes overloaded with queries for which the client is not waiting for a response. Therefore, frequent retries of the same query immediately after a timeout can exacerbate the overload. For more details, see the articles [{#T}](../troubleshooting/performance/queries/retry-cascade.md) and [{#T}](../troubleshooting/performance/queries/overloaded-errors.md); retry policies in the SDK are described in the section [{#T}](../reference/ydb-sdk/error_handling.md).
### Implicit transactions {#implicit-transactions}
-**Implicit transaction** is a query execution mode where the [transaction mode](transactions.md#modes) is not specified. In this case, {{ ydb-short-name }} independently determines whether to wrap them in a transaction. This mode is described in more detail in [{#T}](transactions.md#implicit).
+**Implicit transaction** is a query execution mode in which the [transaction mode](transactions.md#modes) is not specified. In this case, {{ ydb-short-name }} independently determines whether to wrap them in a transaction. This mode is described in more detail in [{#T}](transactions.md#implicit).
### Multi-version concurrency control {#mvcc}
-[**Multi-version concurrency control**](https://en.wikipedia.org/wiki/Multiversion_concurrency_control), also known as **MVCC**, is a method used by {{ ydb-short-name }} to allow multiple concurrent transactions to access the database without interfering with each other. It is described in more detail in a separate article [{#T}](query_execution/mvcc.md).
+[**Multi-version concurrency control**](https://en.wikipedia.org/wiki/Multiversion_concurrency_control), **multi-version concurrency control** or **MVCC** is a method used by {{ ydb-short-name }} for concurrent access of multiple parallel transactions to the database without interfering with each other. It is described in more detail in a separate article [{#T}](query_execution/mvcc.md).
### Streaming queries {#streaming-query}
-A type of query designed for [stream processing](https://en.wikipedia.org/wiki/Stream_processing) of an unbounded data stream. Unlike regular queries, streaming queries have no execution time limits, automatically restart on errors, and periodically save their state as [checkpoints](#streaming-queries-checkpoints) for fault tolerance.
+A type of query designed for [stream processing](https://en.wikipedia.org/wiki/Stream_processing) of an unbounded data stream. Unlike regular queries, streaming queries have no restrictions on execution duration, automatically restart on errors, and periodically save their state as [checkpoints](#streaming-queries-checkpoints) to ensure fault tolerance.
Streaming queries are described in more detail in a separate article [{#T}](streaming-query.md).
### Streaming query checkpoints {#streaming-queries-checkpoints}
-The periodically saved state of a [streaming query](#streaming-query), necessary for automatically restoring its operation after failures in a distributed system. More about checkpoints in the article [{#T}](../dev/streaming-query/checkpoints.md).
+A periodically saved state of a [streaming query](#streaming-query), necessary for automatically restoring its operation after failures in a distributed system. For more details about checkpoints, see the article [{#T}](../dev/streaming-query/checkpoints.md).
### Topology {#topology}
-{{ ydb-short-name }} supports several [cluster](#cluster) **topologies** (or **topology**), described in more detail in a separate article [{#T}](topology.md). Below are explanations of several related terms.
+{{ ydb-short-name }} supports several **topologies** of a [cluster](#cluster) (or **topology**), described in more detail in a separate article [{#T}](topology.md). Below, several related terms are explained.
#### Availability zones and regions {#regions-az}
An **availability zone** is a data center or its isolated segment with minimal physical distance between nodes and minimal risk of failure simultaneously with other availability zones. Thus, availability zones should not share common infrastructure such as power supply, cooling, or external network connections.
-A **region** is a large geographic area containing multiple availability zones. The distance between availability zones within a region should be about 500 km or less. {{ ydb-short-name }} writes data to each availability zone in the region synchronously, ensuring reasonable latency and uninterrupted operation in case one availability zone fails.
+A **region** is a large geographic area containing several availability zones. The distance between availability zones in one region should be about 500 km or less. {{ ydb-short-name }} performs data writes to each availability zone in the region synchronously, ensuring reasonable latency and uninterrupted operation in case of failure of one of the availability zones.
#### Rack {#rack}
-A **rack**, also known as a **server rack**, is equipment used to organize the placement of multiple servers. Servers in the same rack are more likely to become unavailable simultaneously due to rack-level issues related to power supply, cooling, etc. {{ ydb-short-name }} can take into account which server is in which rack when placing each data fragment in environments based on physical servers.
+A **rack** or **server rack** is equipment used to organize the placement of multiple servers. Servers in the same rack are more likely to become unavailable simultaneously due to rack-level issues related to power supply, cooling, etc. {{ ydb-short-name }} can take into account information about which server is in which rack when placing each data fragment in environments based on physical servers.
#### Pile {#pile}
-A **Pile** is a set of nodes that can fail or be shut down simultaneously while keeping other parts of the cluster operational. A Pile can remain operational when other cluster nodes are shut down. Piles are used in [bridge mode](#bridge) to split the cluster into several parts between which synchronous replication is performed. A Pile can consist of nodes from one or more regions.
+A **Pile** is a set of nodes that can fail or be shut down simultaneously while maintaining the operability of other parts of the cluster (pile). A Pile can remain operational when other cluster nodes are shut down. Piles are used in [bridge mode](#bridge) to split the cluster into several parts between which synchronous replication is performed. A Pile can consist of nodes from one or more regions.
#### Bridge mode {#bridge}
-**Bridge mode** is a special cluster topology in which data is stored with synchronous replication between several [piles](#pile). The features of this mode are described in [{#T}](topology.md#bridge) and [{#T}](bridge.md).
+**Bridge mode** is a special cluster topology in which data is stored with synchronous replication between several [piles](#pile). The features of this mode are described in [{#T}](topology.md#bridge) and also in [{#T}](bridge.md).
### Table {#table}
A **table** is a structured piece of information organized into rows and columns. Each row represents a single record or item, and each column is a specific attribute or field with a defined data type.
-There are two main approaches to representing table data in memory or on disk: [row-oriented (row by row)](#row-oriented-table) and [column-oriented (column by column)](#column-oriented-table). The chosen approach greatly affects the performance characteristics of operations on this data: the former is more suitable for transactional workloads (OLTP), and the latter for analytical workloads (OLAP). {{ ydb-short-name }} supports both approaches.
+There are two main approaches to representing tabular data in memory or on disks: [row-based (row by row)](#row-oriented-table) and [column-based (column by column)](#column-oriented-table). The chosen approach greatly affects the performance characteristics of operations on this data: the former is more suitable for transactional workloads (OLTP), and the latter for analytical workloads (OLAP). {{ ydb-short-name }} supports both approaches.
#### Row-oriented table {#row-oriented-table}
@@ -194,42 +174,42 @@ There are two main approaches to representing table data in memory or on disk: [
#### Column-oriented table {#column-oriented-table}
-**Column-oriented tables** (also known as **columnar tables**) store data for each column separately. They are optimized for building aggregates over a small number of columns, but are less suitable for accessing specific rows, as rows need to be reconstructed from their cells on the fly. They are described in more detail in [{#T}](datamodel/table.md#column-oriented-tables).
+**Column-oriented tables** or **columnar tables** store data for each column separately. They are optimized for building aggregates over a small number of columns, but are less suitable for accessing specific rows, as rows need to be reconstructed from their cells on the fly. They are described in more detail in [{#T}](datamodel/table.md#column-oriented-tables).
#### Primary key {#primary-key}
A **primary key** is an ordered list of columns whose values uniquely identify a row. It is used to create the table's [primary index](#primary-index). It is set by the {{ ydb-short-name }} user when [creating a table](../yql/reference/syntax/create_table/index.md) and significantly affects the performance of operations on that table.
-A guide on choosing primary keys is provided in [{#T}](../dev/primary-key/index.md).
+Guidance on choosing primary keys is provided in [{#T}](../dev/primary-key/index.md).
#### Primary index {#primary-index}
-A **primary index** (also known as a **primary key index**) is the main data structure used to find rows in a table. It is created based on the chosen [primary key](#primary-key) and determines the physical order of rows in the table; thus, each table can have only one primary index. The primary index is unique.
+**Primary index** or **primary key index** is the main data structure used to find rows in a table. It is created based on the selected [primary key](#primary-key) and determines the physical order of rows in the table; thus, each table can have only one primary index. The primary index is unique.
#### Secondary index {#secondary-index}
-A **secondary index** is an additional data structure used to find rows in a table, typically when this cannot be done efficiently using the [primary index](#primary-index). Unlike the primary index, secondary indexes are managed independently of the table's main data. Thus, a table can have multiple secondary indexes for different scenarios. The capabilities of {{ ydb-short-name }} regarding secondary indexes are described in a separate article [{#T}](query_execution/secondary_indexes.md). A secondary index can be either unique or non-unique.
+A **secondary index** is an additional data structure used to find rows in a table, typically when this cannot be done efficiently using the [primary index](#primary-index). Unlike a primary index, secondary indexes are managed independently of the table's main data. Thus, a table can have multiple secondary indexes for different scenarios. The capabilities of {{ ydb-short-name }} regarding secondary indexes are described in a separate article [{#T}](query_execution/secondary_indexes.md). A secondary index can be either unique or non-unique.
-Special types of secondary indexes are distinguished separately: [vector index](#vector-index), [full-text index](#fulltext-index), and [JSON index](#json-index).
+Special types of secondary indexes include [vector index](#vector-index), [full-text index](#fulltext-index), and [JSON index](#json-index).
#### Vector index {#vector-index}
-A **vector index** is an additional data structure used to speed up the solution of the [vector search](query_execution/vector_search.md) problem when there is a large amount of data and [exact vector search without an index](../yql/reference/udf/list/knn.md) does not work satisfactorily.
+**Vector index** is an additional data structure used to speed up the solution of the [vector search](query_execution/vector_search.md) problem when there is a lot of data and [exact vector search without an index](../yql/reference/udf/list/knn.md) does not work satisfactorily.
The capabilities of {{ ydb-short-name }} for approximate nearest neighbor search (ANN search) using vector indexes are described in a separate article [{#T}](../dev/vector-indexes.md).
-**Vector index** is a specialized type of [secondary index](#secondary-index) designed for similarity search, unlike traditional secondary indexes optimized for equality or range search.
+A **vector index** is a specialized type of [secondary index](#secondary-index) designed for similarity search, as opposed to traditional secondary indexes optimized for equality or range search.
#### Full-text index {#fulltext-index}
-A **full-text index** is an additional data structure used to speed up text search on a table column (by words and phrases, and when using N-grams, also by substrings).
+**Full-text index** is an additional data structure used to speed up text search on a table column (by words and phrases, and when using N-grams, also by substrings).
-The full-text search capabilities and index parameters are described in the articles [{#T}](../dev/fulltext-indexes.md) and [{#T}](query_execution/fulltext_search.md).
+The features of full-text search and index parameters are described in the articles [{#T}](../dev/fulltext-indexes.md) and [{#T}](query_execution/fulltext_search.md).
-#### JSON index {#json-index}
+#### JSON-index {#json-index}
-**JSON index** is an additional data structure used to speed up predicates with the functions [JSON_EXISTS](../yql/reference/builtins/json.md#json_exists) and [JSON_VALUE](../yql/reference/builtins/json.md#json_value) on a column of type `Json` or `JsonDocument`. Unlike traditional secondary indexes optimized for equality or range searches on individual table columns, the JSON index works with arbitrary [JsonPath](../yql/reference/builtins/json.md#jsonpath) paths within a JSON document.
+**JSON index** is an additional data structure used to speed up predicates with the [JSON_EXISTS](../yql/reference/builtins/json.md#json_exists) and [JSON_VALUE](../yql/reference/builtins/json.md#json_value) functions on a column of type `Json` or `JsonDocument`. Unlike traditional secondary indexes, which are optimized for equality or range searches on individual table columns, the JSON index works with arbitrary [JsonPath](../yql/reference/builtins/json.md#jsonpath) paths within a JSON document.
-JSON index, like the [full-text index](#fulltext-index), is implemented on top of an [inverted index](https://en.wikipedia.org/wiki/Inverted_index), but uses its own tokenizer for JSON documents. JSON search capabilities are described in the articles {#T} and {#T}.
+A JSON index, like a [full-text index](#fulltext-index), is built on top of an [inverted index](https://en.wikipedia.org/wiki/Inverted_index), but uses its own JSON document tokenizer. JSON search capabilities are described in the articles [{#T}](../dev/json-indexes.md) and [{#T}](query_execution/json_search.md).
#### Local index {#local-index}
@@ -237,106 +217,106 @@ A local index is an auxiliary structure that is stored together with the table d
#### Bloom filter {#bloom-filter}
-A **Bloom filter** is a [probabilistic data structure](https://en.wikipedia.org/wiki/Bloom_filter) that allows you to quickly check whether an element belongs to a set. False positives are possible, but false negatives are not.
+A Bloom filter is a [probabilistic data structure](https://en.wikipedia.org/wiki/Bloom_filter) that allows you to quickly check whether an element belongs to a set. False positives are possible, but false negatives are not.
-#### Local Bloom index {#local-bloom-skip-index}
+#### Local bloom index {#local-bloom-skip-index}
-**Local Bloom index** is a special case of [local index](#local-index): a probabilistic filter over column values based on the [Bloom filter](https://en.wikipedia.org/wiki/Bloom_filter) that speeds up selective queries by skipping data fragments where the searched value is guaranteed to be absent. For more information, see [Bloom indexes](../dev/bloom-skip-indexes.md), [local indexes](query_execution/local_indexes.md).
+A local Bloom index is a special case of a [local index](#local-index): a probabilistic filter on column values based on a [Bloom filter](https://en.wikipedia.org/wiki/Bloom_filter) that speeds up selective queries by skipping data fragments where the searched value is guaranteed to be absent. For more information, see [Bloom indexes](../dev/bloom-skip-indexes.md) and [local indexes](query_execution/local_indexes.md).
#### Column family {#column-family}
-**Column family** or **column group** is a feature that allows you to store subsets of columns of a [row table](#row-oriented-table) in a separate family or group. The main use case is storing some columns on other disk types (moving less important columns to HDD) or with different compression settings. If the workload requires many column families, consider using [column-oriented tables](#column-oriented-table).
+**Column family** or **column group** is a feature that allows storing subsets of columns of a [row table](#row-oriented-table) separately in a family or group. The main use case is storing some columns on other disk types (moving less important columns to HDD) or with different compression settings. If the workload requires many column families, consider using [column tables](#column-oriented-table).
#### Column encoding {#column-encoding}
-**Column encoding** is a mechanism for optimizing data storage in table columns, which reduces the amount of disk space used and speeds up the execution of certain operations.
+**Column encoding** is a data storage optimization mechanism for table columns that reduces disk space usage and speeds up some operations.
-#### Time to Live {#ttl}
+#### Time to live {#ttl}
**Time to live** or **TTL** is a mechanism for automatically deleting old rows from a table asynchronously in the background. It is described in a separate article [{#T}](ttl.md).
### View {#view}
-**view** is a way to save a query and access its results as a real table. The view itself does not store any data except the query text. The query stored in the view is executed each time a SELECT is performed on it, generating the returned result. Any changes to the tables referenced by the view are immediately reflected in the results of reading from it.
+**View** is a way to save a query and access its results as if they were a real table. The view itself does not store any data except the query text. The query stored in the view is executed each time a SELECT is performed on it, generating the returned result. Any changes to the tables referenced by the view are immediately reflected in the read results.
{% if feature_view %}
-Views can be user or system.
+Views can be user-defined or system.
-#### User views {#user-view}
+#### User-defined views {#user-view}
-**User views** are created by the user using the [{#T}](../yql/reference/syntax/create-view.md) command. They are described in more detail in [{#T}](../concepts/datamodel/view.md).
+**User-defined views** are created by the user using the [{#T}](../yql/reference/syntax/create-view.md) command. They are described in more detail in [{#T}](../concepts/datamodel/view.md).
{% endif %}
#### System views {#system-view}
-**System views** are special views automatically created by the system for monitoring the state of a database and cluster. They are located in a special directory `.sys`, which is in the root folder of each database. System views for databases are described in [{#T}](../dev/system-views.md); system views for the cluster and access management issues are described in [{#T}](../devops/observability/system-views.md).
+**System views** are special views automatically created by the system for monitoring the state of the database and cluster. They are located in a special directory `.sys` in the root folder of each database. System views for databases are described in [{#T}](../dev/system-views.md); system views for the cluster and access management issues are described in [{#T}](../devops/observability/system-views.md).
### Topic {#topic}
-A **message queue** is used for reliable asynchronous communication between different systems by passing messages. {{ ydb-short-name }} provides infrastructure that ensures exactly-once semantics in such communications. Using it, you can achieve a guarantee of no lost messages and no accidental duplicates.
+**Message queue** is used for reliable asynchronous communication between different systems by passing messages. {{ ydb-short-name }} provides infrastructure that ensures exactly-once semantics in such communications. Using it, you can guarantee no lost messages or accidental duplicates.
-A **topic** is a named entity in a message queue designed for interaction between [writers](#producer) and [readers](#consumer).
+**Topic** is a named entity in a message queue for interaction between [writers](#producer) and [readers](#consumer).
-Several terms related to topics are given below. How topics work in {{ ydb-short-name }} is explained in more detail in a separate article [{#T}](datamodel/topic.md).
+Several terms related to topics are listed below. How topics work in {{ ydb-short-name }} is explained in more detail in a separate article [{#T}](datamodel/topic.md).
#### Partition {#partition}
-For horizontal scaling, topics are divided into individual elements called **partitions**. Thus, partitions are the unit of parallelism within a topic. Messages within each partition are ordered.
+For horizontal scaling, topics are divided into separate elements called **partitions**. Thus, partitions are the unit of parallelism within a topic. Messages within each partition are ordered.
However, subsets of data managed by a single [data shard](#data-shard) or [column shard](#column-shard) may also be called partitions.
#### Offset {#offset}
-An **offset** is a sequence number that identifies a message within a [partition](#partition).
+**Offset** is a sequence number that identifies a message within a [partition](#partition).
#### Writer {#producer}
-A **producer** (also known as a **writer**) is an entity that writes new messages to a topic.
+**Writer** or **producer** is an entity that writes new messages to a topic.
#### Reader {#consumer}
-A **consumer** (also known as a **reader**) is an entity that reads messages from a topic.
+**Reader** or **consumer** is an entity that reads messages from a topic.
-### Change Data Capture {#cdc}
+### Change data capture {#cdc}
-**Change Data Capture** (CDC) is a mechanism that allows you to subscribe to a **change feed** in a specific [table](#table). Technically, it is implemented on top of [topics](#topic). It is described in more detail in a separate article [{#T}](cdc.md).
+**Change data capture** or **CDC** is a mechanism that allows subscribing to a **change feed** for a specific [table](#table). Technically, it is implemented on top of [topics](#topic). It is described in more detail in a separate article [{#T}](cdc.md).
-#### Change Feed {#changefeed}
+#### Change feed {#changefeed}
-A **change feed** is an ordered list of changes to a [table](#table) placed in a [topic](#topic).
+**Change feed** is an ordered list of changes to a [table](#table) placed in a [topic](#topic).
-### Backup Collection {#backup-collection}
+### Backup collection {#backup-collection}
-A **backup collection** is a [schema object](#scheme-object) that organizes full and incremental [backups](#backup) for selected [row tables](#row-oriented-table). Collections provide [point-in-time recovery](https://en.wikipedia.org/wiki/Point-in-time_recovery) by maintaining [backup chains](#backup-chain) and ensuring consistent recovery of multiple tables. A table can belong to only one backup collection at a time.
+**Backup collection** is a [schema object](#scheme-object) that organizes full and incremental [backups](#backup) for selected [row tables](#row-oriented-table). Collections provide [point-in-time recovery](https://en.wikipedia.org/wiki/Point-in-time_recovery) by maintaining [backup chains](#backup-chain) and ensuring consistent recovery of multiple tables. A table can belong to only one backup collection at a time.
For more information, see [{#T}](datamodel/backup-collection.md).
#### Backup {#backup}
-A **backup** is a copy of data at a specific point in time that can be used for data recovery. In the context of [backup collections](#backup-collection), there are two types:
+**Backup** is a copy of data at a specific point in time that can be used for data recovery. In the context of [backup collections](#backup-collection), there are two types:
-- **Full backup**: A complete snapshot of all data in the collection. Serves as the basis for [backup chains](#backup-chain) and can be restored independently.
+- **Full backup**: A complete snapshot of all data in the collection. It serves as the basis for [backup chains](#backup-chain) and can be restored independently.
- **Incremental backup**: Captures only changes (inserts, updates, deletes) since the previous backup. Requires the entire backup chain for restoration.
-#### Backup Chain {#backup-chain}
+#### Backup chain {#backup-chain}
-A **backup chain** is an ordered sequence of [backups](#backup) starting with a full backup, followed by zero or more incremental backups. Each incremental backup depends on all previous backups in the chain. Deleting any backup in the chain makes subsequent incremental backups unrecoverable.
+**Backup chain** is an ordered sequence of [backups](#backup) starting with a full backup followed by zero or more incremental backups. Each incremental backup depends on all previous backups in the chain. Deleting any backup in the chain makes subsequent incremental backups unrecoverable.
{% if feature_async_replication == true %}
### Async replication instance {#async-replication-instance}
-An **async replication instance** is a named entity that stores [async replication](async-replication.md) settings (connection settings, list of replicated objects, etc.). It can also be used to obtain information about the state of async replication: [initial scan progress](async-replication.md#initial-scan), [lag](async-replication.md#replication-of-changes), [errors](async-replication.md#error-handling), etc.
+**Async replication instance** is a named entity that stores [async replication](async-replication.md) settings (connection settings, list of replicated objects, etc.). It can also be used to obtain information about the async replication status: [initial scan progress](async-replication.md#initial-scan), [lag](async-replication.md#replication-of-changes), [errors](async-replication.md#error-handling), etc.
#### Replicated object {#replicated-object}
-A **replicated object** is an object (for example, a table) for which asynchronous replication is configured.
+**Replicated object** is an object (for example, a table) for which async replication is configured.
#### Replica object {#replica-object}
-A **replica object** is a mirror copy of the replicated object, automatically created by an asynchronous replication instance. It is typically read-only.
+**Replica object** is a mirror copy of the replicated object, automatically created by the async replication instance. Typically, it is read-only.
{% endif %}
@@ -344,27 +324,27 @@ A **replica object** is a mirror copy of the replicated object, automatically cr
### Transfer instance {#transfer-instance}
-A **transfer instance** is a named entity that stores [transfer](transfer.md) settings, including connection settings and data transformation rules. It can also be used to obtain information about the transfer status, such as [errors](transfer.md#error-handling).
+**Transfer instance** is a named entity that stores [transfer](transfer.md) settings, including connection settings and data transformation rules. It can also be used to obtain information about the transfer status, for example [errors](transfer.md#error-handling).
{% endif %}
### Coordination node {#coordination-node}
-A **coordination node** is a schema object that allows client applications to create semaphores to coordinate their actions. Coordination nodes are used to implement distributed locks, service discovery, leader election, and other scenarios. For more information about [coordination nodes](./datamodel/coordination-node.md).
+**Coordination node** is a schema object that allows client applications to create semaphores for coordinating their actions. Coordination nodes are used to implement distributed locks, service discovery, leader election, and other scenarios. For more information about [coordination nodes](./datamodel/coordination-node.md).
#### Semaphore {#semaphore}
-A **semaphore** is an object inside a [coordination node](#coordination-node) that provides a synchronization mechanism for distributed applications. Semaphores can be persistent or temporary and support create, acquire, release, and monitor operations. For more information about [semaphores in {{ ydb-short-name }}](./datamodel/coordination-node.md#semaphore).
+**Semaphore** is an object inside a [coordination node](#coordination-node) that provides a synchronization mechanism for distributed applications. Semaphores can be persistent or temporary and support create, acquire, release, and monitor operations. For more information about [semaphores in {{ ydb-short-name }}](./datamodel/coordination-node.md#semaphore).
{% if feature_resource_pool == true and feature_resource_pool_classifier == true %}
### Resource pool {#resource-pool}
-A **resource pool** is a schema object that describes the limits imposed on resources (CPU, RAM, etc.) available for executing queries in this resource pool. A query is always executed in some resource pool. By `default`, all queries are executed in the resource pool named , which does not impose any limits. For more information about using resource pools, see the article [{#T}](../dev/resource-consumption-management.md).
+**Resource pool** is a schema object that describes the limits imposed on resources (CPU, RAM, etc.) available for executing queries in this resource pool. A query is always executed in some resource pool. By default, all queries are executed in a resource pool named `default`, which does not impose any limits. For more information about using resource pools, see [{#T}](../dev/resource-consumption-management.md).
### Resource pool classifier {#resource-pool-classifier}
-A **resource pool classifier** is an object designed to manage the distribution of queries among [resource pools](#resource-pool). It describes the rules by which a resource pool is selected for each query. These classifiers are global for the entire [database](#database) and apply to all queries that come into it. For more information about their use, see the article [{#T}](../dev/resource-consumption-management.md).
+**Resource pool classifier** is an object designed to manage the distribution of queries among [resource pools](#resource-pool). It describes the rules by which a resource pool is selected for each query. These classifiers are global for the entire [database](#database) and apply to all queries that come into it. For more information about their usage, see [{#T}](../dev/resource-consumption-management.md).
{% endif %}
@@ -374,25 +354,25 @@ A **resource pool classifier** is an object designed to manage the distribution
### Federated queries {#federated-queries}
-**Federated queries** is a feature that allows you to execute queries against data stored in systems external to the {{ ydb-short-name }} cluster.
+**Federated queries** is a feature that allows executing queries to data stored in systems external to the {{ ydb-short-name }} cluster.
Below are explanations of several terms related to federated queries. How federated queries work in {{ ydb-short-name }} is explained in more detail in a separate article [{#T}](query_execution/federated_query/index.md).
#### External data source {#external-data-source}
-An **external data source** or **external connection** is metadata that describes how to connect to a supported external system to execute [federated queries](#federated-queries).
+**External data source** or **external connection** is metadata describing how to connect to a supported external system to execute [federated queries](#federated-queries).
#### External table {#external-table}
-An **external table** is metadata that describes a specific data set that can be retrieved from an [external data source](#external-data-source).
+**External table** is metadata describing a specific dataset that can be retrieved from an [external data source](#external-data-source).
#### Secret {#secret}
-A **secret** is confidential metadata that requires special handling. For example, secrets can be used in definitions of [external data sources](#external-data-source) and represent entities such as passwords and tokens.
+**Secret** is confidential metadata that requires special handling. For example, secrets can be used in definitions of [external data sources](#external-data-source) and represent entities such as passwords and tokens.
### Authentication token {#auth-token}
-An **authentication token** (or **auth token**) is a token used for [authentication](../security/authentication.md) in {{ ydb-short-name }}.
+**Auth token** is a token used for [authentication](../security/authentication.md) in {{ ydb-short-name }}.
{{ ydb-short-name }} supports [different types of authentication](../security/authentication.md) and various token types.
@@ -410,7 +390,7 @@ The **{{ ydb-short-name }} cluster schema** is the hierarchical namespace of the
### Schema root {#scheme-root}
-**Cluster schema root** is the root element of the [namespace {{ ydb-short-name }}](datamodel/cluster-namespace.md), whose child elements are [databases](#database).
+**Cluster schema root** is the root element of the [{{ ydb-short-name }} namespace](datamodel/cluster-namespace.md), whose child elements are [databases](#database).
### Schema object {#scheme-object}
@@ -428,13 +408,13 @@ Folders can contain subfolders, and such nesting can be of arbitrary depth.
An **access object** during [authorization](../security/authorization.md) is an entity for which access rights and restrictions are configured. In {{ ydb-short-name }}, access objects are [schema objects](#scheme-object).
-Each [schema object](#scheme-object) has an [owner](#access-owner) and an [access control list](#access-control-list) for that object, granted to users and groups ([access subjects](#access-subject)).
+Each [schema object](#scheme-object) has an [owner](#access-owner) and an [access control list](#access-control-list) on that object, granted to users and groups ([access subjects](#access-subject)).
### Access subject {#access-subject}
An **access subject** is an entity that can access [access objects](#access-object) and perform certain actions in the system.
-Obtaining access during these requests and actions depends on the configured [access control lists](#access-control-list) and the subject's [access level](#access-level).
+Gaining access during these requests and actions depends on the configured [access control lists](#access-control-list) and the [access level](#access-level) of the subject.
An access subject can be a [user](#access-user) or a [group](#access-group).
@@ -444,11 +424,11 @@ An **[access right](../security/authorization.md#right)** is an entity that refl
### Access right inheritance {#access-right-inheritance}
-**Access right inheritance** is a mechanism where [access rights](#access-right) granted on parent [access objects](#access-object) are inherited by child objects in the hierarchical database structure. This ensures that permissions granted at a higher level of the hierarchy apply to all lower levels, unless they are [explicitly overridden](../reference/ydb-cli/commands/scheme-permissions.md#clear-inheritance).
+**Access rights inheritance** is a mechanism where [access rights](#access-right) granted on parent [access objects](#access-object) are inherited by child objects in the hierarchical database structure. This ensures that permissions granted at a higher level of the hierarchy apply to all lower levels unless they are [explicitly overridden](../reference/ydb-cli/commands/scheme-permissions.md#clear-inheritance).
-### Access control list {#access-control-list}
+### Permissions list {#access-control-list}
-An **[access control list](../security/authorization.md#right)** (ACL) is a list of all [rights](#access-right) granted to [access subjects](#access-subject) (users and groups) on a specific [access object](#access-object).
+**[access control list](../security/authorization.md#right)** or **ACL** — a list of all [rights](#access-right) granted to [access subjects](#access-subject) (users and groups) on a specific [access object](#access-object).
### Access level {#access-level}
@@ -457,32 +437,32 @@ An **access level** provides an [access subject](#access-subject) with additiona
- Database
- Viewer
- Monitoring
-- Administration
+- Administration.
-The access level for a subject is configured using [access level lists](#access-level-list).
+The access level for a subject is configured using [access control lists](#access-level-list).
### Access level list {#access-level-list}
-An **access level list** or **permission list** is a list of [SID](#access-sid)s of [access subjects](#access-subject) that are allowed a specific [access level](#access-level).
+**Access Control List** or **Permission List** — a list of [SID](#access-sid)s of [access subjects](#access-subject) that are allowed a certain [access level](#access-level).
In {{ ydb-short-name }}, there are [several such lists](../reference/configuration/security_config.md#security-access-levels) that define who has which [access levels](#access-level).
-Detailed information about access level lists, their hierarchy, and how they work is provided in the [Access Level Lists](../security/authorization.md#access-level-lists) section of the authorization documentation.
+Detailed information about access control lists, their hierarchy, and operating principles is provided in the [Access control lists](../security/authorization.md#access-level-lists) section of the authorization documentation.
### Owner {#access-owner}
-An **[owner](../security/authorization.md#owner)** is an [access subject](#access-subject) ([user](#access-user) or [group](#access-group)) that has full rights to a specific [access object](#access-object).
+**[Owner](../security/authorization.md#owner)** — an [access subject](#access-subject) ([user](#access-user) or [group](#access-group)) that has full rights to a specific [access object](#access-object).
### User {#access-user}
-A **[user](../security/authorization.md#user)** is a person who uses {{ ydb-short-name }} to perform a specific function.
+**[User](../security/authorization.md#user)** — a person who uses {{ ydb-short-name }} to perform a specific function.
In {{ ydb-short-name }}, there are different types of users depending on the creation method:
-- local users in {{ ydb-short-name }} databases
-- external users from third-party directories
+- Local users in databases {{ ydb-short-name }}.
+- external users from third-party directories.
-A user is identified by a [SID](#access-sid).
+A user is identified by [SID](#access-sid).
#### Local user {#local-user}
@@ -490,120 +470,120 @@ A user whose account is created directly in {{ ydb-short-name }} using the YQL c
#### External user {#external-user}
-A {{ ydb-short-name }} user whose account is created in a third-party directory, for example, in an LDAP directory or IAM system.
+A user {{ ydb-short-name }} whose account is created in an external directory, for example, an LDAP directory or IAM system.
### Group {#access-group}
-**[Group](../security/authorization.md#group)** or **access group** — a named set of [users](#access-user) and other groups with equal capabilities for their members.
+**[Group](../security/authorization.md#group)** or **access group** - a named set of [users](#access-user) and other groups with equal permissions for their members.
-A group is identified by a [SID](#access-sid).
+A group is identified by [SID](#access-sid).
### Role {#access-role}
A role is a named set of [access rights](#access-right) used to assign to [users](#access-user) or [groups](#access-group) of users.
-Roles in {{ ydb-short-name }} are implemented using [groups](#access-group), which are created during the initial cluster deployment and assigned a specific [access list](#access-right) on the cluster schema root. For more details about roles, see the article [{#T}](../security/builtin-security.md).
+Roles in {{ ydb-short-name }} are implemented using [groups](#access-group), which are created during the initial deployment of the cluster and are assigned a specific [access rights list](#access-right) at the cluster schema root. For more information about roles, see the [{#T}](../security/builtin-security.md) article.
### SID {#access-sid}
-**SID** or **security identifier** — a string of the form `<name>` or `<name>@<auth-domain>` that identifies an [access subject](../concepts/glossary.md#access-subject). It is used in [authentication](../security/authentication.md), [authorization](../security/authorization.md), [access lists](#access-control-list), and [access level lists](#access-level-list).
+**SID** or **security identifier** — a string of the form `<name>` or `<name>@<auth-domain>` that identifies an [access subject](../concepts/glossary.md#access-subject). It is used in [authentication](../security/authentication.md), [authorization](../security/authorization.md), [access control lists](#access-control-list), and [access level lists](#access-level-list).
-A SID identifies an individual [user](#access-user) or [group of users](#access-group).
+SID identifies an individual [user](#access-user) or [group of users](#access-group).
-The optional suffix `@<auth-domain>` identifies the source of the access subject, i.e., the external directory or system from which it was obtained. For example, users or groups from an LDAP directory may have the suffix `@ldap`. The absence of a suffix means that the user or group is created and exists directly in {{ ydb-short-name }}.
+The optional suffix `@<auth-domain>` identifies the source of the access subject, i.e., the external directory or system from which it was obtained. For example, users or groups from an LDAP directory may have the suffix `@ldap`. The absence of a suffix means that the user or group was created and exists directly in {{ ydb-short-name }}.
### Query optimizer {#optimizer}
-[**Query optimizer**](https://en.wikipedia.org/wiki/Query_optimization) — a set of {{ ydb-short-name }} components responsible for converting the logical representation of a query into a specific physically executable plan for obtaining the requested result. The main goal of the optimizer is to select, among all possible query execution plans, one that is sufficiently efficient in terms of predicted execution time and cluster resource consumption. It is described in more detail in a separate article [{#T}](query_execution/optimizer.md).
+[**Query optimizer**](https://en.wikipedia.org/wiki/Query_optimization) is a set of {{ ydb-short-name }} components responsible for converting the logical representation of a query into a specific physically executable plan for obtaining the requested result. The main goal of the optimizer is to select, among all possible query execution plans, one that is sufficiently efficient in terms of predicted execution time and cluster resource consumption. It is described in more detail in a separate article [{#T}](query_execution/optimizer.md).
### Compilation cache {#compile-cache}
-**Compilation cache** or **compile cache** — a cache of compiled queries on each [node](#node) of the cluster. It is used to avoid recompilation: if the query text is already in the node's cache, no additional compilation is performed. For more details, see the section [Query compilation cache](../dev/system-views.md#compile-cache-queries).
+**compile cache** is a cache of compiled queries on each [node](#node) of the cluster. It is used to avoid recompilation: if the query text is already in the node's cache, no additional compilation is performed. For more details, see the [Query Compilation Cache](../dev/system-views.md#compile-cache-queries) section.
-## Advanced terminology {#advanced-terminology}
+## Advanced Terminology {#advanced-terminology}
-This section explains terms that are useful for [{{ ydb-short-name }} contributors](../contributor/index.md) and users who want to gain a deeper understanding of what happens inside the system.
+This section explains terms that are useful for [{{ ydb-short-name }} contributors](../contributor/index.md) and users who want to understand what happens inside the system more deeply.
-### Actor implementation {#actor-implementation}
+### Actor Implementation {#actor-implementation}
-#### Actor system {#actor-system}
+#### Actor System {#actor-system}
-**Actor system** — a C++ library with an [implementation](https://github.com/ydb-platform/ydb/tree/main/ydb/library/actors) of the [actor model](https://en.wikipedia.org/wiki/Actor_model) for {{ ydb-short-name }} needs.
+**actor system** is a C++ library with an [implementation](https://github.com/ydb-platform/ydb/tree/main/ydb/library/actors) of the [actor model](https://en.wikipedia.org/wiki/Actor_model) for the needs of {{ ydb-short-name }}.
-#### Actor service {#actor-service}
+#### Actor Service {#actor-service}
-**Actor service** — an [actor](#actor) that has a well-known name and typically runs as a single instance on a [node](#node).
+**actor service** is an [actor](#actor) that has a well-known name and usually runs as a single instance on a [node](#node).
#### ActorId {#actorid}
-**ActorId** — a unique identifier of an actor or [tablet](#tablet) in a [cluster](#cluster).
+**ActorId** is a unique identifier of an actor or a [tablet](#tablet) in a [cluster](#cluster).
-#### Actor system interconnect {#actor-system-interconnect}
+#### Actor System Interconnect {#actor-system-interconnect}
-**Actor system interconnect** or **interconnect** — the internal network layer of a [cluster](#cluster). All [actors](#actor) communicate with each other in the system via the interconnect.
+**actor system interconnect** or **interconnect** is the internal network layer of the [cluster](#cluster). All [actors](#actor) communicate with each other in the system via the interconnect.
#### Local {#local}
-**Local** — an [actor service](#actor-service) running on each [node](#node). It directly manages [tablets](#tablet) on its node and interacts with [Hive](#hive). It registers with Hive and receives commands to start tablets.
+**local** is an [actor service](#actor-service) running on each [node](#node). It directly manages [tablets](#tablet) on its node and interacts with [Hive](#hive). It registers with Hive and receives commands to start tablets.
-### Tablet implementation {#tablet-implementation}
+### Tablet Implementation {#tablet-implementation}
-[**Tablet**](#tablet) — an [actor](#actor) with persistent state. It includes a set of data for which this tablet is responsible and a state machine through which the tablet's data (or state) is modified. The tablet is a fault-tolerant entity because the tablet's data is stored in [distributed storage](#distributed-storage), which survives disk and node failures. The tablet is automatically restarted on another [node](#node) in case of failure or overload of the previous one. Data in the tablet is modified sequentially, as the system infrastructure guarantees that there is no more than one [tablet leader](#tablet-leader) through which tablet data changes are performed.
+[**tablet**](#tablet) is an [actor](#actor) with persistent state. It includes a set of data that this tablet is responsible for and a state machine through which the tablet's data (or state) is modified. The tablet is a fault-tolerant entity because the tablet's data is stored in [distributed storage](#distributed-storage), which survives disk and node failures. The tablet is automatically restarted on another [node](#node) in case of failure or overload of the previous one. Data in the tablet is modified sequentially, as the system infrastructure guarantees that there is no more than one [tablet leader](#tablet-leader) through which tablet data changes are performed.
-A tablet solves the same problem as the [Paxos](https://en.wikipedia.org/wiki/Paxos_(computer_science)) and [Raft](https://en.wikipedia.org/wiki/Raft_(algorithm)) algorithms in other systems, namely the problem of [distributed consensus](https://en.wikipedia.org/wiki/Consensus_(computer_science)). From a technical standpoint, the implementation of a tablet can be described as a replicated state machine (RSM) on top of a shared log, since the tablet's state is fully described by an ordered log of commands stored in a distributed and fault-tolerant storage.
+A tablet solves the same problem as the [Paxos](https://en.wikipedia.org/wiki/Paxos_(computer_science)) and [Raft](https://en.wikipedia.org/wiki/Raft_(algorithm)) algorithms in other systems, namely the problem of [distributed consensus](https://en.wikipedia.org/wiki/Consensus_(computer_science)). From a technical point of view, the tablet implementation can be described as a replicated state machine (RSM) on top of a shared log, since the tablet state is fully described by an ordered log of commands stored in a distributed and fault-tolerant storage.
-During execution, the tablet's state machine is managed by three components:
+At runtime, the tablet state machine is managed by three components:
1. The common tablet part ensures log consistency and recovery in case of failures.
-2. **Executor** is an abstraction of a local database, namely data structures and code that organize work with data stored by the tablet.
+2. **executor** is an abstraction of a local database, namely data structures and code that organize work with data stored by the tablet.
3. An actor with user code that implements the specific logic of a particular tablet type.
In {{ ydb-short-name }}, there are several types of specialized tablets that store various data for different tasks. Many {{ ydb-short-name }} features, such as [tables](#table) and [topics](#topic), are implemented as different types of tablets. Thus, reusing the tablet infrastructure is one of the key means of extensibility of {{ ydb-short-name }} as a platform.
-Typically, in a {{ ydb-short-name }} cluster, there are orders of magnitude more tablets compared to processes or threads that other systems would use for a cluster of similar size. In a {{ ydb-short-name }} cluster, hundreds of thousands or millions of tablets can easily run simultaneously.
+Typically, a {{ ydb-short-name }} cluster runs orders of magnitude more tablets compared to the processes or threads that other systems would use for a cluster of similar size. In a {{ ydb-short-name }} cluster, hundreds of thousands or millions of tablets can easily run simultaneously.
-Since a tablet stores its state in [distributed storage](#distributed-storage), it can be (re)started on any node of the cluster. Tablets are identified by a [TabletID](#tabletid), a 64-bit number assigned when the tablet is created.
+Since a tablet stores its state in [distributed storage](#distributed-storage), it can be (re)started on any node of the cluster. Tablets are identified by [TabletID](#tabletid), a 64-bit number assigned when the tablet is created.
-### Tablet leader {#tablet-leader}
+### Tablet Leader {#tablet-leader}
-A **tablet leader** is the current active leader of a given tablet. The tablet leader accepts commands, assigns them an order, and confirms them to the outside world. It is guaranteed that at any given time there is at most one leader for each tablet.
+**Tablet leader** is the current active leader of a given tablet. The tablet leader accepts commands, assigns them an order, and confirms them to the outside world. It is guaranteed that at any given time there is at most one leader for each tablet.
### Tablet candidate {#tablet-candidate}
-A **tablet candidate** is one of the election participants that wants to become the [leader](#tablet-leader) of a given tablet. If the candidate wins the election, it becomes the tablet leader.
+**Tablet candidate** is one of the election participants that wants to become the [leader](#tablet-leader) of a given tablet. If the candidate wins the election, it becomes the tablet leader.
### Tablet replica {#tablet-follower}
-A **tablet follower** or **hot standby** is a copy of the [tablet leader](#tablet-leader) that applies the log of commands accepted by the leader (with some delay). A tablet can have zero or more followers. Followers perform two main functions:
+**Tablet follower** or **hot standby** is a copy of the [tablet leader](#tablet-leader) that applies the command log accepted by the leader (with some delay). A tablet can have zero or more replicas. Replicas perform two main functions:
-* In case of termination or failure of the leader, followers are preferred [candidates](#tablet-candidate) for the new leader role, as they can become leader much faster than other candidates because they have applied most of the log.
-* Followers can respond to read-only requests if the client explicitly chooses an optional relaxed transaction mode that allows stale reads.
+* In case of termination or failure of the leader, replicas are preferred [candidates](#tablet-candidate) for the new leader role, as they can become the leader much faster than other candidates because they have applied most of the log.
+* Replicas can answer read-only requests if the client explicitly chooses an optional relaxed transaction mode that allows stale reads.
### Tablet generation {#tablet-generation}
-A **tablet generation** is a number that identifies the reincarnation of the tablet leader. It changes only when a new leader is elected and always increases.
+**Tablet generation** is a number that identifies the reincarnation of the tablet leader. It changes only when a new leader is elected and always increases.
### Tablet local database {#local-database}
-A **tablet local database** or **local database** is a set of data structures and associated code that manage the tablet's state and the data it stores. Logically, the state of the local database is represented by a set of tables, very similar to relational tables. Modification of the local database state is performed by local tablet transactions created by the tablet's user actor.
+**Tablet local database** or **local database** is a set of data structures and associated code that manage the state of the tablet and the data it stores. Logically, the state of the local database is represented by a set of tables, very similar to relational tables. Modification of the local database state is performed by local tablet transactions created by the user tablet actor.
Each table of the local database is stored as an [LSM tree](#lsm-tree).
#### Log-structured merge-tree {#lsm-tree}
-A **[Log-structured merge-tree](https://en.wikipedia.org/wiki/Log-structured_merge-tree)** or **LSM tree** is a data structure designed to optimize write and read performance in storage systems. It is used in {{ ydb-short-name }} to store [local database](#local-database) tables and [VDisks](#vdisk) data.
+**[Log-structured merge-tree](https://en.wikipedia.org/wiki/Log-structured_merge-tree)** is a data structure designed to optimize write and read performance in storage systems. It is used in {{ ydb-short-name }} to store tables of the [local database](#local-database) and data of [VDisks](#vdisk).
#### MemTable {#memtable}
-All data written to tables of a [local database](#local-database) is initially stored in an in-memory data structure called a **MemTable**. When the MemTable reaches a specified size, it is flushed to disk as an immutable data structure called an [SST](#sst).
+All data written to the tables of the [local database](#local-database) is initially stored in an in-memory data structure called **MemTable**. When the MemTable reaches a specified size, it is flushed to disk as an immutable data structure [SST](#sst).
#### Sorted string table {#sst}
-**Sorted string table** or **SST** is an immutable data structure that stores table rows sorted by key, facilitating efficient key lookup and range scans. Each SST consists of a contiguous series of small data pages, typically about 7 KiB each, which further optimizes disk read operations. An SST is usually part of an [LSM tree](#lsm-tree).
+**Sorted string table** is an immutable data structure that stores table rows sorted by key, facilitating efficient key lookup and range scans. Each SST consists of a continuous series of small data pages, typically about 7 KiB each, which further optimizes reading data from disk. An SST is usually part of an [LSM tree](#lsm-tree).
#### Tablet pipe {#tablet-pipe}
-**Tablet pipe** or **TabletPipe** is a virtual connection that can be established with a tablet. It involves looking up the [tablet leader](#tablet-leader) by [TabletID](#tabletid). This is the recommended way to interact with a tablet. The term **open a pipe to a tablet** describes the process of resolving (finding) a tablet in the cluster and establishing a virtual communication channel with it.
+**Tablet pipe** or **TabletPipe** is a virtual connection that can be established with a tablet. It involves finding the [tablet leader](#tablet-leader) by [TabletID](#tabletid). This is the recommended way to work with a tablet. The term **open a pipe to a tablet** describes the process of resolving (finding) a tablet in the cluster and establishing a virtual communication channel with it.
#### TabletID {#tabletid}
@@ -611,11 +591,11 @@ All data written to tables of a [local database](#local-database) is initially s
#### Bootstrapper {#bootstrapper}
-**Bootstrapper** is the primary mechanism for starting tablets, used for system tablets (e.g., [Hive](#hive), [DS controller](#ds-controller), the root [SchemeShard](#scheme-shard)). [Hive](#hive) initializes the remaining tablets.
+**Bootstrapper** is the main mechanism for starting tablets, used for system tablets (e.g., [Hive](#hive), [DS controller](#ds-controller), root [SchemeShard](#scheme-shard)). [Hive](#hive) initializes the other tablets.
### Shared cache {#shared-cache}
-**Shared cache** is an [actor](#actor) that stores data pages recently read from [distributed storage](#distributed-storage). Caching these pages reduces disk I/O operations and speeds up data retrieval, improving overall system performance.
+**Shared cache** is an [actor](#actor) that stores data pages recently read from [distributed storage](#distributed-storage). Caching these pages reduces the number of disk I/O operations and speeds up data retrieval, improving overall system performance.
### Memory controller {#memory-controller}
@@ -623,23 +603,23 @@ All data written to tables of a [local database](#local-database) is initially s
### Spilling {#spilling}
-**Spilling** is a memory management mechanism in {{ ydb-short-name }} that temporarily offloads intermediate query data to external storage when such data exceeds the available RAM of a node. In {{ ydb-short-name }}, disk is currently used for spilling.
+**Spilling** is a memory management mechanism in {{ ydb-short-name }} that temporarily offloads intermediate query data to external storage when such data exceeds the available RAM of the node. In {{ ydb-short-name }}, disk is currently used for spilling.
For more details about spilling, see [{#T}](query_execution/spilling.md).
### Tablet types {#tablet-types}
-[Tablets](#tablet) can be thought of as a framework for building reliable components that operate in a distributed system. Many components of {{ ydb-short-name }} — both system and user-data components — are implemented using this framework; the main ones are listed below.
+[Tablets](#tablet) can be considered as a framework for building reliable components operating in a distributed system. Many components of {{ ydb-short-name }} — both system and those working with user data — are implemented using this framework; the main ones are listed below.
#### SchemeShard {#scheme-shard}
**SchemeShard** or **Scheme shard** is a system tablet that stores the database schema, including metadata of user [tables](#table), [topics](#topic), etc.
-Additionally, there is a **root SchemeShard** that stores information about databases created in the cluster.
+In addition, there is a **root SchemeShard** that stores information about databases created in the cluster.
#### DataShard {#data-shard}
-**DataShard** or **Data shard** is a tablet that manages a segment of a [row-based user table](datamodel/table.md#row-oriented-tables). A logical user table is divided into segments by contiguous ranges of the table's primary key. Each such range is handled by a separate DataShard tablet. The range itself is also called a [partition](#partition). The DataShard tablet stores data row by row, which is efficient for OLTP workloads.
+**DataShard** or **Data shard** is a tablet that manages a segment of a [row-based user table](datamodel/table.md#row-oriented-tables). A logical user table is divided into segments by continuous ranges of the table's primary key. Each such range is managed by a separate DataShard tablet. The range itself is also called a [partition](#partition). The DataShard tablet stores data row by row, which is efficient for OLTP workloads.
#### ColumnShard {#column-shard}
@@ -647,7 +627,7 @@ Additionally, there is a **root SchemeShard** that stores information about data
#### KeyValue Tablet {#kv-tablet}
-**KeyValue** or **KV Tablet** is a tablet that implements a simple key→value mapping, where keys and values are strings. It also has several specific features, such as locks.
+**KeyValue**, **KV Tablet** is a tablet that implements a simple key → value mapping, where keys and values are strings. It also has several specific features, such as locks.
#### PersQueue Tablet {#pq-tablet}
@@ -659,7 +639,7 @@ Additionally, there is a **root SchemeShard** that stores information about data
#### Coordinator {#coordinator}
-**Coordinator** is a system tablet that ensures the global order of transactions. The coordinator's task is to assign a logical time [PlanStep](#planstep) to each transaction planned through this coordinator. Each transaction is assigned exactly one coordinator, selected by hashing its [TxId](#txid).
+**Coordinator** is a system tablet that ensures global ordering of transactions. The coordinator's task is to assign a logical time [PlanStep](#planstep) to each transaction planned through this coordinator. Each transaction is assigned exactly one coordinator, selected by hashing its [TxId](#txid).
#### Mediator {#mediator}
@@ -667,23 +647,23 @@ Additionally, there is a **root SchemeShard** that stores information about data
#### Hive {#hive}
-**Hive** is a system tablet responsible for starting and managing other tablets. Its responsibilities include moving tablets between nodes in case of a [node](#node) failure or overload.{% if audience != "corp" %} You can learn more about Hive in a [separate article](../contributor/hive.md).{% endif %}
+**Hive** is a system tablet responsible for launching and managing other tablets. Its responsibilities include moving tablets between nodes in case of failure or overload of a [node](#node).{% if audience != "corp" %} For more details about Hive, see the [dedicated article](../contributor/hive.md).{% endif %}
#### CMS {#cms}
-**CMS** or **cluster management system** is a system tablet responsible for managing information about the current state of the [cluster {{ ydb-short-name }}](#cluster). This information is used to perform gradual cluster restarts without impacting user workloads, maintenance, cluster reconfiguration, etc.
+**CMS** or **cluster management system** is a system tablet responsible for managing information about the current state of the [{{ ydb-short-name }} cluster](#cluster). This information is used for performing rolling restarts of the cluster without affecting user workloads, maintenance, cluster reconfiguration, etc.
#### NodeBroker {#node-broker}
-**NodeBroker** is a system tablet responsible for registering [dynamic nodes](#dynamic) in the cluster.
+**NodeBroker** is a system tablet that is responsible for registering [dynamic nodes](#dynamic) in the cluster.
#### BSController {#ds-controller}
-**BSController**, **blob storage controller** manages the dynamic configuration of distributed storage, including information about [PDisk](#pdisk), [VDisk](#vdisk), and [storage groups](#storage-group). It interacts with [node warden](#node-warden) to start various distributed storage components. It interacts with [Hive](#hive) to allocate [channels](#channel) to tablets.
+**BSController** (also known as **blob storage controller**) manages the dynamic configuration of the distributed storage, including information about [PDisk](#pdisk), [VDisk](#vdisk), and [storage groups](#storage-group). It interacts with [node warden](#node-warden) to start various distributed storage components. It interacts with [Hive](#hive) to allocate [channels](#channel) to tablets.
#### Console {#console}
-**Console** is a system tablet responsible for storing [dynamic configuration](../devops/configuration-management/configuration-v1/dynamic-config.md) and delivering it to cluster nodes.
+**Console** is a system tablet responsible for storing the [dynamic configuration](../devops/configuration-management/configuration-v1/dynamic-config.md) and delivering it to cluster nodes.
#### Kesus {#kesus}
@@ -711,42 +691,36 @@ Additionally, there is a **root SchemeShard** that stores information about data
#### StatisticsAggregator {#statistics-aggregator}
-**StatisticsAggregator** is a tablet responsible for collecting statistics used in cost-based optimization.
+**StatisticsAggregator** is a tablet responsible for collecting statistics used in cost optimization.
### Slot {#slot}
**Slot** in {{ ydb-short-name }} can be used in two contexts:
-* A **slot** is a portion of server resources allocated to run one {{ ydb-short-name }} [node](#node). The typical slot size is 10 CPU cores and 50 GB of RAM. Slots are used if the {{ ydb-short-name }} cluster is deployed on servers or virtual machines with sufficient resources to host multiple slots.
-* A **VDisk slot** or **VSlot** is a portion of a [PDisk](#pdisk) that can be allocated to one of the [VDisk](#vdisk).
+* **Slot** is a portion of server resources allocated to run one [node](#node) of {{ ydb-short-name }}. A typical slot size is 10 CPU cores and 50 GB of RAM. Slots are used when the {{ ydb-short-name }} cluster is deployed on servers or virtual machines with sufficient resources to host multiple slots.
+* **VDisk slot** or **VSlot** is a portion of a [PDisk](#pdisk) that can be allocated to one of the [VDisk](#vdisk) instances.
### State storage {#state-storage}
-**State storage** or **StateStorage** is a distributed service that stores information about tablets, namely:
+**State storage** (also known as **StateStorage**) is a distributed service that stores information about tablets, namely:
* The current tablet leader or its absence.
* Tablet replicas.
* Tablet generation and step `(generation:step)`.
-State storage is used as a service for tablet name resolution, i.e., to obtain an [ActorId](#actorid) from a [TabletID](#tabletid). StateStorage is also used in the [tablet leader](#tablet-leader) election process.
+State storage is used as a service for resolving tablet names, i.e., to obtain an [ActorId](#actorid) from a [TabletID](#tabletid). StateStorage is also used in the [tablet leader](#tablet-leader) election process.
-Information in the State Storage is volatile. Thus, it is lost when power is turned off or the process is restarted. Despite its name, this service is not a permanent long-term storage. It only contains information that is easy to recover and that should not be durable. However, the State Storage stores information on multiple nodes to minimize the impact of node failures. This service can also be used to gather a quorum, which is used for tablet leader election.
+The information in the state storage is volatile. Thus, it is lost on power failure or process restart. Despite its name, this service is not a permanent long-term storage. It only contains information that is easy to recover and that does not need to be durable. However, the state storage stores information on multiple nodes to minimize the impact of node failures. This service can also be used to gather a quorum, which is used for tablet leader election.
-Due to its nature, the State Storage service operates on a best-effort basis. For example, the absence of multiple tablet leaders is guaranteed through a leader election protocol on [Distributed Storage](#distributed-storage), not on the State Storage.
-
-For more details about the StateStorage and related subsystems, see the Metadata Distribution Services section.
+Due to its nature, the state storage service operates on a best-effort basis. For example, the absence of multiple tablet leaders is guaranteed through the leader election protocol on the [distributed storage](#distributed-storage), not on the state storage.
### Board {#board}
-**Board** is a distributed service designed for storing metadata as key-value pairs. It is used, among other things, for storing information about [endpoints](../concepts/connect.md#endpoint).
-
-For more details about the Board and related subsystems, see the Metadata Distribution Services section.
+**Board** is a distributed service designed to store metadata as key-value pairs. It is used, among other things, to store information about [endpoints](../concepts/connect.md#endpoint).
### SchemeBoard {#scheme-board}
-**SchemeBoard** is a distributed service designed for storing metadata as key-value pairs. It is used, among other things, for storing information about [schemas](#global-schema).
-
-For more details about the SchemeBoard and related subsystems, see the Metadata Distribution Services section.
+**SchemeBoard** is a distributed service designed to store metadata as key-value pairs. It is used, among other things, to store information about [schemas](#global-schema).
#### Compaction {#compaction}
@@ -754,57 +728,57 @@ For more details about the SchemeBoard and related subsystems, see the Metadata
#### gRPC proxy {#grpc-proxy}
-**gRPC proxy** is a proxy system for external user requests. Client requests enter the system via the [gRPC](https://grpc.io) protocol, then the proxy component translates them into internal calls to execute these requests, transmitted via [interconnect](#actor-system-interconnect). This proxy provides an interface for both request-response and bidirectional streaming data.
+**gRPC proxy** — a proxy system for external user requests. Client requests enter the system via the [gRPC](https://grpc.io) protocol, then the proxy component translates them into internal calls to execute these requests, transmitted via [interconnect](#actor-system-interconnect). This proxy provides an interface for both request-response and bidirectional streaming data.
### Distributed configuration {#distributed-configuration}
-**Distributed configuration** or **DistConf** is an internal [configuration](../devops/configuration-management/configuration-v2/config-overview.md) mechanism of the cluster that ensures the startup and configuration of [static nodes](#static-node), automatic management of the [static storage group](#static-group) and [State Storage](../concepts/glossary.md#state-storage). The distributed configuration starts before any [tablets](#tablet), [storage groups](#storage-group), and [State Storage](../concepts/glossary.md#state-storage).
+**Distributed configuration** or **DistConf** — an internal mechanism of cluster [configuration](../devops/configuration-management/configuration-v2/config-overview.md) that ensures the startup and configuration of [static nodes](#static-node), automatic management of the [static storage group](#static-group) and [State Storage](../concepts/glossary.md#state-storage). Distributed configuration starts before any [tablets](#tablet), [storage groups](#storage-group), or [State Storage](../concepts/glossary.md#state-storage).
For more details about the distributed configuration, see [{#T}](../contributor/configuration-v2.md).
-### Distributed Storage Implementation {#distributed-storage-implementation}
+### Distributed storage implementation {#distributed-storage-implementation}
-**Distributed Storage** is a distributed fault-tolerant data storage layer that stores binary records called [LogoBlob](#logoblob), addressed using a specific type of identifier called [LogoBlobID](#logoblobid). Thus, Distributed Storage is a key-value store that maps a LogoBlobID to a string of up to 10 MB. Distributed Storage consists of many [storage groups](#storage-group), each of which is an independent data repository.
+**Distributed storage** — a distributed fault-tolerant data storage layer that stores binary records called [LogoBlob](#logoblob), addressed by a specific type of identifier called [LogoBlobID](#logoblobid). Thus, Distributed storage is a key-value store that maps a LogoBlobID to a string of up to 10 MB. Distributed storage consists of many [storage groups](#storage-group), each of which is an independent data repository.
-Distributed storage stores immutable data, with each immutable data block identified by a specific LogoBlobID key. The Distributed storage API is very specific and is intended only for use by [tablets](#tablet) to store their data and change logs. Thus, it is not intended for general-purpose data storage. Data in Distributed storage is deleted using special barrier commands. Due to the absence of mutations in its interface, Distributed storage can be implemented without implementing [distributed consensus](https://en.wikipedia.org/wiki/Consensus_(computer_science)). Distributed storage is just one of the components that tablets use to implement distributed consensus.
+Distributed storage stores immutable data, with each immutable data block identified by a specific LogoBlobID key. The Distributed storage API is very specific, intended only for use by [tablets](#tablet) to store their data and change logs. Thus, it is not intended for general-purpose data storage. Data in Distributed storage is deleted using special barrier commands. Due to the absence of mutations in its interface, Distributed storage can be implemented without implementing [distributed consensus](https://en.wikipedia.org/wiki/Consensus_(computer_science)). Distributed storage is just one of the components that tablets use to implement distributed consensus.
#### LogoBlob {#logoblob}
-**LogoBlob** is a set of binary immutable data identified by a [LogoBlobID](#logoblobid) and stored in [Distributed storage](#distributed-storage). The data block size is limited at the [VDisk](#vdisk) level and above in the stack. Currently, the maximum data block size that VDisks can handle is 10 MB.
+**LogoBlob** — a set of binary immutable data, identified by [LogoBlobID](#logoblobid) and stored in [Distributed storage](#distributed-storage). The data block size is limited at the [VDisk](#vdisk) level and above in the stack. Currently, the maximum data block size that a VDisk can handle is 10 MB.
#### LogoBlobID {#logoblobid}
-**LogoBlobID** is an identifier of a [LogoBlob](#logoblob) in [Distributed storage](#distributed-storage). It has a structure of the form `[TabletID, Generation, Step, Channel, Cookie, BlobSize, PartID]`. The main elements of LogoBlobID are:
+**LogoBlobID** — an identifier of a [LogoBlob](#logoblob) in [Distributed storage](#distributed-storage). It has a structure of the form `[TabletID, Generation, Step, Channel, Cookie, BlobSize, PartID]`. The main elements of LogoBlobID:
-* `TabletID` is the [ID](#tabletid) of the tablet that owns the LogoBlob.
-* `Generation` is the generation of the tablet in which the data block was written.
-* `Channel` is the [channel](#channel) of the tablet on which the LogoBlob is written.
-* `Step` is an incremental counter, usually within the tablet generation.
-* `Cookie` is a unique identifier of the data block within a single `Step`. The Cookie is typically used when writing multiple data blocks to a single `Step`.
-* `BlobSize` is the size of the LogoBlob.
-* `PartID` is the identifier of the data block part. It is important when the original LogoBlob is split into parts using [error correction coding](#erasure-coding), and the parts are written to the corresponding [VDisk](#vdisk) and [storage groups](#storage-group).
+* `TabletID` — the [ID](#tabletid) of the tablet that owns the LogoBlob.
+* `Generation` — the generation of the tablet in which the data block was written.
+* `Channel` — the [channel](#channel) of the tablet on which the LogoBlob is written.
+* `Step` — an incremental counter, usually within the tablet generation.
+* `Cookie` — a unique identifier of a data block within a single `Step`. Cookie is typically used when writing multiple data blocks into one `Step`.
+* `BlobSize` — the size of the LogoBlob.
+* `PartID` — the identifier of a data block part. It is important when the original LogoBlob is split into parts using [erasure coding](#erasure-coding), and the parts are written to the corresponding [VDisk](#vdisk) and [storage groups](#storage-group).
#### Replication {#replication}
-**Replication** is a process that ensures a sufficient number of copies (replicas) of data to maintain the desired availability characteristics of the {{ ydb-short-name }} cluster. It is typically used in geo-distributed {{ ydb-short-name }} clusters.
+**Replication** — a process that ensures a sufficient number of copies (replicas) of data to maintain the desired availability characteristics of the cluster {{ ydb-short-name }}. Typically used in geo-distributed clusters {{ ydb-short-name }}.
#### Erasure coding {#erasure-coding}
-[**Erasure coding**](https://en.wikipedia.org/wiki/Erasure_code) is a data encoding method in which the original data is supplemented with redundancy and split into multiple fragments, enabling recovery of the original data if one or more fragments are lost. It is widely used in {{ ydb-short-name }} clusters with a single [availability zone](#regions-az), as opposed to [replication](#replication) with 3 replicas. For example, the most popular erasure coding scheme 4+2 provides the same reliability as three replicas, with a space overhead of 1.5 versus 3.
+[**Erasure coding**](https://en.wikipedia.org/wiki/Erasure_code) is a data encoding method where the original data is supplemented with redundancy and split into multiple fragments, enabling recovery of the original data if one or more fragments are lost. It is widely used in clusters {{ ydb-short-name }} with a single [availability zone](#regions-az), as opposed to [replication](#replication) with 3 replicas. For example, the most popular erasure coding scheme 4+2 provides the same reliability as three replicas, with a space overhead of 1.5 versus 3.
#### PDisk {#pdisk}
-**PDisk** or **physical disk** is a component that controls a physical disk drive (block device). In other words, PDisk is a subsystem that implements an abstraction similar to a specialized file system on top of block devices (or files emulating a block device for testing purposes). PDisk provides data integrity control (including [error correction coding](#erasure-coding) of sector groups to recover data on individual damaged sectors, integrity control using checksums), transparent encryption of all data on the disk, and transactional guarantees for disk operations (write confirmation strictly after `fsync`).
+**PDisk**, **physical disk** is a component that controls a physical disk drive (block device). In other words, PDisk is a subsystem that implements an abstraction similar to a specialized file system on top of block devices (or files emulating a block device for testing purposes). PDisk provides data integrity control (including [erasure coding](#erasure-coding) of sector groups to recover data on individual damaged sectors, integrity checking using checksums), transparent encryption of all data on the disk, and transactional guarantees for disk operations (write confirmation strictly after `fsync`).
-PDisk contains a scheduler that ensures sharing of device bandwidth among multiple clients ( [VDisk](#vdisk)). PDisk divides the block device into blocks called [slots](#slot) (about 128 megabytes in size; smaller blocks are also allowed). At any given time, no more than one VDisk can own each slot. PDisk also maintains a recovery log shared by PDisk service records and all VDisks.
+PDisk contains a scheduler that ensures shared use of the device's bandwidth among multiple clients ( [VDisk](#vdisk)). PDisk divides the block device into blocks called [slots](#slot) (about 128 megabytes in size; smaller blocks are also allowed). At any given time, no more than one VDisk can own each slot. PDisk also maintains a recovery log shared by PDisk service records and all VDisks.
#### VDisk {#vdisk}
-**VDisk** or **virtual disk** is a component that implements data storage of [distributed storage](#distributed-storage) [LogoBlob](#logoblob) on [PDisk](#pdisk). VDisk stores all its data on PDisk. One VDisk corresponds to one PDisk, but typically several VDisks are associated with one PDisk. Unlike PDisk, which hides blocks and journals behind it, VDisk provides an interface at the LogoBlob and [LogoBlobID](#logoblobid) level, for example, writing a LogoBlob, reading LogoBlobID data, and deleting a set of LogoBlobs using a special command. VDisk is a member of a [storage group](#storage-group). The VDisk itself is local, but many VDisks in a given group provide reliable data storage. VDisks in a group synchronize data with each other and replicate data in case of losses. The set of VDisks in a storage group forms a distributed RAID.
+**VDisk**, **virtual disk** is a component that implements data storage of [distributed storage](#distributed-storage) [LogoBlob](#logoblob) on [PDisk](#pdisk). VDisk stores all its data on PDisk. One VDisk corresponds to one PDisk, but typically several VDisks are associated with one PDisk. Unlike PDisk, which hides blocks and logs behind it, VDisk provides an interface at the LogoBlob and [LogoBlobID](#logoblobid) level, for example writing a LogoBlob, reading LogoBlobID data, and deleting a set of LogoBlobs using a special command. VDisk is a member of a [storage group](#storage-group). VDisk itself is local, but many VDisks in a given group provide reliable data storage. VDisks in a group synchronize data with each other and replicate data in case of losses. The set of VDisks in a storage group forms a distributed RAID.
#### Yard {#yard}
-**Yard** is the name of the [PDisk](#pdisk) API. It allows [VDisk](#vdisk) to read and write data to blocks and journals, reserve blocks, delete blocks, and transactionally acquire and release block ownership. In some contexts, Yard can be considered a synonym for PDisk.
+**Yard** is the name of the [PDisk](#pdisk) API. It allows [VDisk](#vdisk) to read and write data to blocks and logs, reserve blocks, delete blocks, and transactionally acquire and release ownership of blocks. In some contexts, Yard can be considered a synonym for PDisk.
#### Skeleton {#skeleton}
@@ -816,58 +790,58 @@ PDisk contains a scheduler that ensures sharing of device bandwidth among multip
#### Proxy {#ds-proxy}
-**Distributed storage proxy**, **DS-proxy**, or **BS-proxy** acts as a client library for performing operations with [distributed storage](#distributed-storage). The users of the DS-proxy are [tablets](#tablet), which write to and read from distributed storage. The DS-proxy hides the distributed nature of distributed storage from the user. The task of the DS-proxy is to write to a quorum of [VDisks](#vdisk), retry when necessary, and control the write/read flow to prevent VDisk overload.
+**Distributed storage proxy**, **DS-proxy**, or **BS-proxy** acts as a client library for performing operations with [distributed storage](#distributed-storage). The users of DS-proxy are [tablets](#tablet) that write to and read from distributed storage. DS-proxy hides the distributed nature of distributed storage from the user. The task of DS-proxy is to write to a quorum of [VDisk](#vdisk), perform retries when necessary, and control the write/read flow to prevent VDisk overload.
-Technically, the DS-proxy is implemented as an [actor service](#actor-service) launched by [node warden](#node-warden) on each node for each storage group, handling all requests to the group (writing, reading, and deleting [LogoBlob](#logoblob), group locking). When writing data, the DS-proxy performs [erasure coding](#erasure-coding) of the data, splitting the LogoBlob into parts that are then sent to the corresponding VDisks. The DS-proxy performs the reverse process when reading, receiving parts from VDisks and reconstructing the LogoBlob from them.
+Technically, the DS proxy is implemented as an [actor service](#actor-service) launched by [node warden](#node-warden) on each node for each storage group, handling all requests to the group (writing, reading, and deleting [LogoBlob](#logoblob), and group locking). When writing data, the DS proxy performs [erasure coding](#erasure-coding) of the data, splitting the LogoBlob into parts that are then sent to the corresponding VDisks. The DS proxy performs the reverse process when reading, receiving parts from VDisks and reconstructing the LogoBlob from them.
#### Node warden {#node-warden}
-**Node warden** or `BS_NODE` is an [actor service](#actor-service) on each cluster node that launches [PDisks](#pdisk), [VDisks](#vdisk), and [DS proxies](#ds-proxy) of [static storage groups](#static-group) when the node starts. It also interacts with the [DS controller](#ds-controller) to launch PDisk, VDisk, and DS proxies of [dynamic groups](#dynamic-group). The DS proxy of dynamic groups is launched on demand: node warden processes "undelivered" messages to DS proxies, launches the corresponding DS proxies, and receives group configuration from the DS controller.
+**Node warden** or `BS_NODE` is an [actor service](#actor-service) on each cluster node that launches [PDisks](#pdisk), [VDisks](#vdisk), and [DS proxies](#ds-proxy) of [static storage groups](#static-group) when the node starts. It also interacts with the [DS controller](#ds-controller) to launch PDisk, VDisk, and DS proxies of [dynamic groups](#dynamic-group). The DS proxy of dynamic groups is launched on demand: node warden processes undelivered messages to DS proxies, launches the corresponding DS proxies, and receives group configuration from the DS controller.
-#### Fail realm {#fail-realm}
+#### Failure realm {#fail-realm}
-A **fail realm** is a set of [fail domains](#fail-domain) that can fail simultaneously due to a common cause. A correlated failure of two [VDisks](#vdisk) in the same fail realm is more likely than a failure of two VDisks from different fail realms.
+A **fail realm** is a set of [failure domains](#fail-domain) that can fail simultaneously due to a common cause. A correlated failure of two [VDisks](#vdisk) in the same fail realm is more likely than a failure of two VDisks from different fail realms.
-An example of a fail realm is a set of equipment located in one [data center, or availability zone](#regions-az), which can fail entirely due to a natural disaster, large-scale power outage, or other similar event.
+An example of a fail realm is a set of equipment located in a single [data center, or availability zone](#regions-az), which can fail entirely due to a natural disaster, large-scale power outage, or other similar event.
#### Fail domain {#fail-domain}
A **fail domain** is a set of equipment that can fail simultaneously. A correlated failure of two [VDisk](#vdisk) within the same fail domain is more likely than a failure of two VDisks from different fail domains. In the case of different fail domains, the probability of simultaneous failure also depends on whether the domains in question belong to the same fail realm or different ones.
-An example of a fail domain is a set of disks connected to a single server, since all disks of a particular server may become unavailable if the server's power supply or network controller fails. Typically, all servers located in one [server rack](#rack) are considered part of a common fail domain, because power or network issues at the rack level cause all equipment in it to become unavailable. Thus, a typical fail domain corresponds to a server rack (if the [cluster](#cluster) is configured with rack-level topology awareness) or to an individual server.
+An example of a fail domain is a set of disks connected to a single server, since all disks of a particular server may become unavailable if the server's power supply or network controller fails. Typically, all servers located in a single [server rack](#rack) are considered a common fail domain, because power or network issues at the rack level cause all equipment in it to become unavailable. Thus, a typical fail domain corresponds to a server rack (if the [cluster](#cluster) is configured with rack-aware topology) or an individual server.
-Fail domain-level failures are automatically handled by {{ ydb-short-name }} without stopping the cluster.
+Failures at the fail domain level are automatically handled by {{ ydb-short-name }} without stopping the cluster.
#### Distributed storage channel {#channel}
A **distributed storage channel**, **DS channel**, or **channel** is a logical connection between a [tablet](#tablet) and a [distributed storage](#distributed-storage) group. A tablet can write data to different channels, and each channel maps to a specific [storage group](#storage-group). Having multiple channels allows a tablet to:
-* Write more data than a single storage group can contain.
-* Store different [LogoBlob](#logoblob) in different storage groups, with different properties such as erasure coding or on different media (HDD, SSD, NVMe).
+* Write more data than a single storage group can hold.
+* Store different [LogoBlob](#logoblob)s in different storage groups, with different properties, such as erasure coding or on different media (HDD, SSD, NVMe).
### Distributed transaction implementation {#transaction-implementation}
-The terms related to the implementation of [distributed transactions](#transactions) are explained below.{% if oss == true %} The implementation itself is described in a separate article [{#T}](../contributor/datashard-distributed-txs.md).{% endif %}
+Below are the terms related to the implementation of [distributed transactions](#transactions).{% if oss == true %} The implementation itself is described in a separate article [{#T}](../contributor/datashard-distributed-txs.md).{% endif %}
#### Deterministic transactions {#deterministic-transactions}
-Distributed transactions {{ ydb-short-name }} are inspired by the research paper [Building Deterministic Transaction Processing Systems without Deterministic Thread Scheduling](http://cs-www.cs.yale.edu/homes/dna/papers/transactions-wodet11.pdf) by Alexander Thomson and Daniel J. Abadi from Yale University. The paper introduces the concept of **deterministic transaction processing**, which allows efficient processing of distributed transactions. The original paper imposed restrictions on the types of operations that can be performed this way. Since these restrictions hindered real user scenarios, {{ ydb-short-name }} developed its own algorithms to execute them, using deterministic transactions as stages of user transaction execution with additional orchestration and locking.
+Distributed transactions in {{ ydb-short-name }} are inspired by the research paper [Building Deterministic Transaction Processing Systems without Deterministic Thread Scheduling](http://cs-www.cs.yale.edu/homes/dna/papers/transactions-wodet11.pdf) by Alexander Thomson and Daniel J. Abadi from Yale University. The paper introduced the concept of **deterministic transaction processing**, which allows efficient processing of distributed transactions. The original paper imposed restrictions on the types of operations that could be performed this way. Since these restrictions hindered real-world user scenarios, {{ ydb-short-name }} evolved its algorithms to handle them, using deterministic transactions as stages for executing user transactions with additional orchestration and locking.
#### Optimistic locking {#optimistic-locking}
-As in many other database management systems, queries {{ ydb-short-name }} can place locks on certain data fragments, such as table rows, to ensure that concurrent changes do not lead to an inconsistent state. However, {{ ydb-short-name }} checks these locks not at the beginning of transactions, but when attempting to commit them. The first approach is called **pessimistic locking** (used, for example, in PostgreSQL), and the second is called **optimistic locking** (used in {{ ydb-short-name }}).
+As in many other database management systems, queries in {{ ydb-short-name }} can place locks on certain data fragments, such as table rows, to ensure that concurrent changes do not lead to an inconsistent state. However, {{ ydb-short-name }} checks these locks not at the start of transactions, but when attempting to commit them. The first approach is called **pessimistic locking** (for example, used in PostgreSQL), and the second is called **optimistic locking** (used in {{ ydb-short-name }}).
#### Transaction lock invalidation {#tli}
-**Transaction Lock Invalidation** (Transaction Lock Invalidation, **TLI**) is the normal behavior of {{ ydb-short-name }} when parallel transactions conflict under [optimistic locking](#optimistic-locking). If one transaction (the violator) writes data and thereby breaks the locks of another transaction (the victim), {{ ydb-short-name }} detects this when the victim commits and rolls it back with error `transaction locks invalidated`. For more details on TLI diagnostics, see [{#T}](../troubleshooting/performance/queries/transaction-lock-invalidation.md).
+**Transaction Lock Invalidation** (**TLI**) is the normal behavior of {{ ydb-short-name }} when parallel transactions conflict under [optimistic locks](#optimistic-locking). If one transaction (the violator) writes data and thereby breaks the locks of another transaction (the victim), {{ ydb-short-name }} detects this when the victim commits and aborts it with error `transaction locks invalidated`. For more details on TLI diagnostics, see [{#T}](../troubleshooting/performance/queries/transaction-lock-invalidation.md).
#### Preparation phase {#prepare-stage}
-**Preparation phase** is a transaction phase during which the transaction body is registered on all participating shards.
+**Preparation phase** is the transaction phase during which the transaction body is registered on all participating shards.
#### Execution phase {#execute-stage}
-**Execution phase** is a transaction phase during which the scheduled transaction is executed and a response is generated.
+**Execution phase** is the transaction phase during which the scheduled transaction is executed and a response is generated.
In some cases, instead of [preparation](#prepare-stage) and execution, the transaction is executed immediately and a response is generated. For example, this happens for transactions that affect only one shard or for consistent reads from a data snapshot.
@@ -877,31 +851,31 @@ In the case of read-only transactions, similar to "read uncommitted" in other da
#### Read-write set {#rw-set}
-**Read-write set**, **RW-set** is a set of data that will participate in the execution of a [distributed transaction](#transactions). It combines the read set data that will be read and the write set for which modifications will be performed.
+**Read-write set** or **RW-set** is a data set that will participate in the execution of a [distributed transaction](#transactions). It combines the read set data that will be read and the write set for which modifications will be performed.
#### Read set {#read-set}
-**Read set**, **ReadSet data** is what participating shards send during transaction execution. In the case of data transactions, it may contain information about the state of [optimistic locks](#optimistic-locking), shard readiness to commit, or a decision to abort the transaction.
+**Read set** or **ReadSet data** is what participating shards send during transaction execution. In the case of data transactions, it may contain information about the state of [optimistic locks](#optimistic-locking), the shard's readiness to commit, or a decision to abort the transaction.
#### Transaction proxies {#transaction-proxy}
-**Transaction proxy** or `TX_PROXY` is a service that orchestrates the execution of many [distributed transactions](#transactions): sequential phases, phase execution, planning, and result aggregation. In the case of direct orchestration by other actors (e.g., QP data transactions), it is used for caching and allocating unique [TxIDs](#txid).
+**Transaction proxy** or `TX_PROXY` is a service that orchestrates the execution of many [distributed transactions](#transactions): sequential phases, phase execution, scheduling, and result aggregation. In the case of direct orchestration by other actors (for example, QP data transactions), it is used for caching and allocating unique [TxIDs](#txid).
#### Transaction flags {#txflags}
-**Transaction flags** or **TxFlags** is a bitmask of flags that modify the execution of a transaction in some way.
+**Transaction flags** or **TxFlags** is a bitmask of flags that somehow modify the execution of a transaction.
#### Transaction ID {#txid}
-**Transaction ID** or **TxID** is a unique identifier assigned to each transaction when it is accepted by {{ ydb-short-name }}.
+**TxID** is a unique identifier assigned to each transaction when it is accepted by {{ ydb-short-name }}.
#### Transaction order ID {#transaction-order-id}
-**Transaction order ID** is a unique identifier assigned to each transaction during planning. It consists of [PlanStep](#planstep) and [Transaction ID](#txid).
+**Transaction order id** is a unique identifier assigned to each transaction during scheduling. It consists of [PlanStep](#planstep) and [Transaction ID](#txid).
#### Plan step {#planstep}
-**Plan step**, **PlanStep**, **Step** is the logical time at which the execution of a set of transactions is scheduled.
+**PlanStep** or **Step** is the logical time at which the execution of a set of transactions is scheduled.
#### Mediator time {#mediator-time}
@@ -909,18 +883,18 @@ During the execution of distributed transactions, **mediator time** is the logic
#### MiniKQL {#minikql}
-**MiniKQL** is a language that allows expressing a single [deterministic transaction](#deterministic-transactions) in the system. It is a functional, strongly typed language. Conceptually, the language describes a graph of reading from the database, performing computations on the read data, and writing results to the database and/or to a special document representing the query result (for display to the user). A MiniKQL transaction must explicitly specify its read set (data to be read) and assume deterministic branching (e.g., no randomness).
+**MiniKQL** is a language that allows expressing a single [deterministic transaction](#deterministic-transactions) in the system. It is a functional, strongly typed language. Conceptually, the language describes a graph of reading from the database, performing computations on the read data, and writing results to the database and/or to a special document representing the query result (for display to the user). A MiniKQL transaction must explicitly specify its read set (data to be read) and assume deterministic selection of execution branches (for example, no randomness).
MiniKQL is a low-level language. End users of the system only see queries in the [YQL](#yql) language, which relies on MiniKQL in its implementation.
#### Query Processor {#kqp}
-**Query Processor** or **QP** (formerly **KQP**) is a component of {{ ydb-short-name }} responsible for orchestrating the execution of user queries and generating the final response.
+**Query Processor** or **QP** (formerly **KQP**) is a {{ ydb-short-name }} component responsible for orchestrating the execution of user queries and generating the final response.
### Global schema {#global-schema}
-**Global scheme**, **global schema**, or **database schema** is the schema of all data stored in the [database](#database). It consists of [tables](#table) and other entities such as [topics](#topic). The metadata about these entities is called the global schema. The term is used in contrast to **local schema**, which refers to the data schema inside a [tablet](#tablet). Users of {{ ydb-short-name }} never see the local schema and work only with the global schema.
+**Global scheme**, **global schema**, or **database schema** is the schema of all data stored in a [database](#database). It consists of [tables](#table) and other entities such as [topics](#topic). The metadata about these entities is called the global schema. The term is used in contrast to **local schema**, which refers to the data schema inside a [tablet](#tablet). {{ ydb-short-name }} users never see the local schema and work only with the global schema.
### KiKiMR {#kikimr}
-**KiKiMR** is the outdated name of {{ ydb-short-name }}, used before it became an [open source product](https://github.com/ydb-platform/ydb). It can still be found in source code, old articles, videos, etc.
+**KiKiMR** is the former name of {{ ydb-short-name }}, used before it became an [open source product](https://github.com/ydb-platform/ydb). It can still be found in source code, old articles, videos, etc.
diff --git a/ydb/docs/en/core/concepts/query_execution/json_search.md b/ydb/docs/en/core/concepts/query_execution/json_search.md
new file mode 100644
index 00000000000..219a4abb4e5
--- /dev/null
+++ b/ydb/docs/en/core/concepts/query_execution/json_search.md
@@ -0,0 +1,73 @@
+# Searching JSON document contents
+
+JSON search is a way to find table rows by the content of a JSON document stored in a column of type `Json` or `JsonDocument`: by the existence of a path in the document and by the value located at the specified path. The path is specified using a [JsonPath](../../yql/reference/builtins/json.md#jsonpath) expression, and the checks are performed using the existing functions [JSON_EXISTS](../../yql/reference/builtins/json.md#json_exists) and [JSON_VALUE](../../yql/reference/builtins/json.md#json_value). Typical scenarios:
+
+* Search for documents with a specific nested field
+* Search for documents whose field at a given path equals the required value.
+* filtering semi-structured data without a pre-defined schema.
+
+In {{ ydb-short-name }}, JSON search can be performed in two main ways:
+
+* Without an index: scanning the table and applying the `JSON_EXISTS` or `JSON_VALUE` functions to each row. The approach is simple but scales poorly: the amount of scanning work grows with the table size.
+* with a JSON index — creating a [JSON index](../../dev/json-indexes.md) on a JSON column. This approach is designed for scalable search.
+
+## JSON search with a JSON index {#json-search-index}
+
+The key idea of a [JSON index](../../dev/json-indexes.md) is building an [inverted index](https://en.wikipedia.org/wiki/Inverted_index) over paths and "path + value" pairs. Each path of a JSON document (and for equality checks, the scalar value at that path) is encoded into a token. For each token, the index stores a list of primary key values of the corresponding table rows. Queries to the index are reduced to an inverted search over these tokens — following the same scheme as a [full-text index](../../dev/fulltext-indexes.md#basic), but with its own JSON tokenizer.
+
+The JSON index does not store a copy of the entire JSON document, but decomposes it into individual tokens:
+
+* For each path in the JSON tree, a token 'path exists' is created.
+* if the path leads to a scalar value (string, number, boolean, or `null`), a «path + value» token is additionally created.
+
+Arrays are transparent in this case: array elements are indexed under the same path as the array itself, so the index responds equally to queries to `$.items`, `$.items[0]`, and `$.items[*]`.
+
+For example, for a document:
+
+
+```json
+{
+ "id": 42042,
+ "name": "Michael",
+ "email": null,
+ "items": [null, "str"],
+ "parts": {
+ "key": "k1",
+ "value": false
+ }
+}
+```
+
+
+Conceptually, the following tokens are indexed (without regard to the specifics of the internal representation):
+
+| Token | What it finds |
+| --- | --- |
+| `$` | document is not equal to `NULL` |
+| `$.id` / `$.id == 42042` | path `id` exists / equals `42042` |
+| `$.name` / `$.name == "Michael"` | path `name` exists / equals `"Michael"` |
+| `$.email` / `$.email == null` | path `email` exists / equals `null` |
+| `$.items` / `$.items == null` / `$.items == "str"` | path `items` exists / contains element `null` / contains element `"str"` |
+| `$.parts` | path `parts` exists |
+| `$.parts.key` / `$.parts.key == "k1"` | nested path exists / equals `"k1"` |
+| `$.parts.value` / `$.parts.value == false` | nested path exists or equals `false` |
+
+JSON index allows:
+
+* find rows by path existence using [JSON_EXISTS](../../yql/reference/builtins/json.md#json_exists)
+* find rows by a value in a path using [JSON_VALUE](../../yql/reference/builtins/json.md#json_value) or a filter predicate within `JSON_EXISTS`.
+
+When executing a query, the JSON index can be automatically used by the [optimizer](../glossary.md#optimizer). It is enough to write a regular condition `WHERE` with expressions `JSON_EXISTS` or `JSON_VALUE` on an indexed JSON column. The optimizer recognizes such a predicate and applies reading by the JSON index instead of scanning records of the source table. Additionally, the JSON index, like any other [secondary index](../glossary.md#secondary-index), can be forcefully used by specifying its name in the section `VIEW IndexName`.
+
+If the predicate cannot be converted into an index access, the behavior of {{ ydb-short-name }} depends on whether the required JSON index was explicitly specified in the query:
+
+* When the optimizer automatically selects an index, the query is simply executed without index acceleration (the result remains correct).
+* When explicitly specifying an index via the `VIEW` expression, an error is returned.
+
+For more information, see [{#T}](../../yql/reference/syntax/select/json_index.md).
+
+Additional information:
+
+* [JSON indexes](../../dev/json-indexes.md) — an overview of supported capabilities, predicates, and limitations.
+* [JSON functions](../../yql/reference/builtins/json.md) — reference on `JSON_EXISTS`, `JSON_VALUE`, and the JsonPath language.
+* [VIEW (JSON index)](../../yql/reference/syntax/select/json_index.md) — syntax for queries with a JSON index.
diff --git a/ydb/docs/en/core/concepts/query_execution/toc_i.yaml b/ydb/docs/en/core/concepts/query_execution/toc_i.yaml
index a046a7ca930..42ee92bb1eb 100644
--- a/ydb/docs/en/core/concepts/query_execution/toc_i.yaml
+++ b/ydb/docs/en/core/concepts/query_execution/toc_i.yaml
@@ -1,8 +1,6 @@
items:
- name: Overview
href: index.md
- - name: Query Execution Process
- href: execution_process.md
- name: Query optimizer
href: optimizer.md
- name: Multi-Version Concurrency Control (MVCC)
@@ -15,6 +13,8 @@ items:
href: fulltext_search.md
- name: Hybrid search
href: hybrid_search.md
+ - name: JSON Search
+ href: json_search.md
- name: Local indexes
href: local_indexes.md
- name: Federated query
@@ -25,5 +25,7 @@ items:
href: spilling.md
- name: Scan queries
href: scan_query.md
+ - name: Query Execution Process
+ href: execution_process.md
- name: YQL queries to topics
href: topics.md
diff --git a/ydb/docs/en/core/concepts/query_execution/topics.md b/ydb/docs/en/core/concepts/query_execution/topics.md
index 5fb695a8bda..eed3a4d8481 100644
--- a/ydb/docs/en/core/concepts/query_execution/topics.md
+++ b/ydb/docs/en/core/concepts/query_execution/topics.md
@@ -1,14 +1,14 @@
# YQL queries to topics {#yql-syntax}
-To read and write messages to [topics](../datamodel/topic.md), familiar YQL constructs are used: [SELECT](../../yql/reference/syntax/select/index.md) for reading and [INSERT](../../yql/reference/syntax/insert_into.md) for writing.
+To read and write messages in [topics](../datamodel/topic.md), familiar YQL constructs are used: [SELECT](../../yql/reference/syntax/select/index.md) for reading and [INSERT](../../yql/reference/syntax/insert_into.md) for writing.
## Local and external topics {#local-external-topics}
-YQL queries to topics work the same regardless of whether the topic is in the current database or in another {{ ydb-short-name }} database. The source and receiver of messages can be either a topic **in the same database** where the query is executed, or a topic **in another database**.
+YQL queries to topics work the same regardless of whether the topic is in the current database or in another {{ ydb-short-name }} database. The source and destination of messages can be either a topic **in the same database** where the query is executed, or a topic **in another database**.
### Local topics {#local-topics}
-**Local topics** are topics created in the **same {{ ydb-short-name }} database** as the query being executed.
+**Local topics** are topics created in the **same {{ ydb-short-name }} database** as the executed query.
In the query text, they are referred to **by a short name** — just like a table in the current database:
@@ -29,7 +29,7 @@ INSERT INTO output_topic SELECT ...;
Access to them is performed only through a pre-created [external data source](../datamodel/external_data_source.md) with the YDB source type.
-After creating a source, for example named `ext_source`, accessing topic `input_topic` in an external database is written as follows:
+After creating a source, for example named `ext_source`, accessing the `input_topic` topic in an external database is written as follows:
```yql
@@ -41,11 +41,11 @@ The name `ext_source` in the documentation is **conditional** — in your databa
## Reading from a topic {#topic-read}
-Reading from a topic can be performed in [table](#table-read) and [streaming](#streaming-read) modes (not to be confused with streaming queries).
+Reading from a topic can be limited to only the current data of the topic or work with waiting for new written messages.
-### Table reading {#table-read}
+### Reading current data {#table-read}
-In table mode, reading is performed from the first to the last offset stored in the topic at the time the query is started. If data continues to be written to the topic, the query will stop after reaching the last offset known at startup. Specifying filters on [Service fields](#system-metadata) speeds up reading, as reading occurs only over the specified ranges.
+In this mode, reading is performed from the first to the last offset stored in the topic at the time the query is started. If data continues to be written to the topic, the query will stop after reaching the last offset known at startup. Specifying filters on [Service fields](#system-metadata) speeds up reading, as reading occurs only over the specified ranges.
```yql
@@ -57,9 +57,9 @@ LIMIT 10;
```
-### Streaming reading {#streaming-read}
+### Reading with data waiting {#streaming-read}
-To read new messages, use the `WITH (STREAMING = "TRUE")` option — see more in the [Streaming reading of data from a topic](../../yql/reference/syntax/select/streaming.md) section. Reading starts from the current moment and continues until the number of messages specified in the `LIMIT` expression is read. The `LIMIT` parameter is required — without it, the query will not complete, as it will wait for new messages indefinitely.
+To wait for new messages, use the `WITH (STREAMING = "TRUE")` option. Reading starts from the current moment and continues until the number of messages specified in the `LIMIT` expression is read. The `LIMIT` parameter is mandatory — without it, the query will not complete, as it will wait for new messages indefinitely.
```yql
@@ -148,7 +148,19 @@ FROM
### Service fields {#system-metadata}
-When reading, you can request service fields:
+When reading, you can request service fields and [user message attributes](../datamodel/topic.md#message):
+
+| Field | [Type](../../yql/reference/types/index.md) | Description |
+| --- | --- | --- |
+| `__ydb_create_time` | `Timestamp` | Message creation time |
+| `__ydb_write_time` | `Timestamp` | Message write time to topic |
+| `__ydb_offset` | `Uint64` | Message offset in partition |
+| `__ydb_partition_id` | `Uint64` | Partition number |
+| `__ydb_message_group_id` | `String` | Message group ID |
+| `__ydb_seq_no` | `Uint64` | Message sequence number within group |
+| `__ydb_user_attributes` | `Dict<String,String>` | [User message attributes](../datamodel/topic.md#message) |
+
+Example of using service fields:
```yql
@@ -158,7 +170,7 @@ SELECT
__ydb_write_time AS WriteTime, -- message write time
__ydb_offset AS Offset, -- message offset in topic
__ydb_partition_id AS Partition, -- partition number
- __ydb_message_group_id AS MessageGroupId, -- message group identifier
+ __ydb_message_group_id AS MessageGroupId, -- message group ID
__ydb_seq_no AS SeqNo -- sequence number within partition
FROM
input_topic -- local topic; for external: ext_source.input_topic
@@ -166,7 +178,7 @@ LIMIT 10;
```
-Filters on service fields are evaluated before reading data from the topic and significantly reduce the volume of messages read. Supported are comparison operators (`=`, `<>`, `<`, `<=`, `>`, `>=`, `IN`), logical conditions (`AND`, `OR`), and fields `partition_id`, `write_time`, `offset`. Predicates on other service fields do not limit the read volume.
+Filters on service fields are evaluated before reading data from the topic and significantly reduce the volume of messages read. Comparison operators (`=`, `<>`, `<`, `<=`, `>`, `>=`, `IN`), logical conditions (`AND`, `OR`), and fields `partition_id`, `write_time`, `offset` are supported. Predicates on other service fields do not limit the read volume.
```yql
@@ -182,6 +194,20 @@ WHERE
```
+Example of using custom attributes:
+
+
+```yql
+SELECT
+ COUNT(*) AS ErrorCount
+FROM
+ input_topic -- local topic; for external: ext_source.input_topic
+WHERE
+ __ydb_user_attributes["type"] = "log"
+ AND __ydb_user_attributes["level"] = "error";
+```
+
+
## Writing to a topic {#topic-write}
### Writing a single message {#simple-write}
@@ -214,13 +240,13 @@ FROM
{% note warning %}
-Reading and writing [user attributes](../datamodel/topic.md#message) are not supported.
+Writing [custom attributes](../datamodel/topic.md#message) via YQL is not supported.
{% endnote %}
{% note warning %}
-Transactional writes via YQL/`INSERT INTO` are not supported — partial query results may appear in the topic.
+Transactional writing via YQL/`INSERT INTO` is not supported — partial query results may appear in the topic.
{% endnote %}
diff --git a/ydb/docs/en/core/concepts/toc_i.yaml b/ydb/docs/en/core/concepts/toc_i.yaml
index b44c1842f22..433520c005d 100644
--- a/ydb/docs/en/core/concepts/toc_i.yaml
+++ b/ydb/docs/en/core/concepts/toc_i.yaml
@@ -5,11 +5,6 @@ items:
include:
path: analytics/toc_p.yaml
mode: link
-- name: Architecture
- href: architecture/index.md
- include:
- path: architecture/toc_p.yaml
- mode: link
- name: Connecting to a database
href: connect.md
- name: Schema objects
@@ -19,8 +14,6 @@ items:
mode: link
- name: Cluster topology
href: topology.md
-- name: Bridge mode
- href: bridge.md
- name: Query execution
href: query_execution/index.md
include:
@@ -47,3 +40,13 @@ items:
when: feature_transfer
- name: Streaming queries
href: streaming-query.md
+- name: Architecture
+ href: architecture/index.md
+ include:
+ path: architecture/toc_p.yaml
+ mode: link
+- name: Architecture
+ href: architecture/index.md
+ include:
+ path: architecture/toc_p.yaml
+ mode: link
diff --git a/ydb/docs/en/core/concepts/topology.md b/ydb/docs/en/core/concepts/topology.md
index ffa0406d854..6b6a860054b 100644
--- a/ydb/docs/en/core/concepts/topology.md
+++ b/ydb/docs/en/core/concepts/topology.md
@@ -33,6 +33,7 @@ Fault-tolerant operation modes of distributed storage require a significant amou
| `mirror-3-dc` *(3 nodes)*, can stand a failure of a single server, or a failure of a data center | 3 | 3 | Server | Data center | 3 | Doesn't matter |
| `block-4-2`, can stand a failure of 2 racks | 1.5 | 8 ([10 recommended](*recommended-node-count)) | Rack | Data center | 1 | 8 |
| `block-4-2` *(reduced)*, can stand a failure of 1 rack | 1.5 | 10 | ½ a rack | Data center | 1 | 5 |
+| `block-4-2` *(reduced fault-tolerant)*, can stand a failure of 1 node | 1.5 | 4 | Server | Data center | 1 | Doesn't matter |
| `none`, no fault tolerance | 1 | 1 | Node | Node | 1 | 1 |
{% note info %}
@@ -63,7 +64,11 @@ Cluster response time in bridge mode for most operations is limited by the respo
If it is impossible to use the [recommended amount](#cluster-config) of hardware, you can divide servers within a single rack into two dummy fail domains. In this configuration, the failure of one rack results in the failure of two domains instead of just one. In such reduced configurations, {{ ydb-short-name }} will continue to operate if two domains fail. The minimum number of racks in a cluster is five for `block-4-2` mode and two per data center (e.g., six in total) for `mirror-3-dc` mode.
-The minimal fault-tolerant configuration of a {{ ydb-short-name }} cluster uses the 3 nodes variant of `mirror-3-dc` operating mode, which requires only three servers with three disks each. In this configuration, each server acts as both a fail domain and a fail realm, and the cluster can withstand the failure of only a single server. Each server must be located in an independent data center to provide reasonable fault tolerance.
+There are 2 variants of the minimal fault-tolerant configuration of a {{ ydb-short-name }} cluster:
+
+- The 3 nodes variant of the `mirror-3-dc` operating mode, which requires only three servers with three disks each. In this configuration, each server acts as both a fail domain and a fail realm, and the cluster can withstand the failure of only a single server. Each server must be located in an independent data center to provide reasonable fault tolerance.
+
+- The 4 nodes variant of the `block-4-2` operating mode, which requires 4 servers with 2 or more disks each. In this configuration, the disks of each server are manually divided into 2 fail domains using the [`disk_scope`](../reference/configuration/host_configs.md#disk-scope) attribute, which gives a total of 8 fail domains required for the `block-4-2` mode. Such a cluster remains operational when a single server fails.
{{ ydb-short-name }} clusters configured with one of these approaches can be used for production environments if they don't require stronger fault tolerance guarantees.
diff --git a/ydb/docs/en/core/contributor/hive-booting.md b/ydb/docs/en/core/contributor/hive-booting.md
index 7d416de4ea3..40a401f75f6 100644
--- a/ydb/docs/en/core/contributor/hive-booting.md
+++ b/ydb/docs/en/core/contributor/hive-booting.md
@@ -21,7 +21,7 @@ The boot queue, or *Boot queue*, is stored in Hive's memory and is prioritized.
1. [Resource consumption metrics](hive.md#resources) — tablets with higher consumption have higher priority.
1. Tablets that restart frequently have lower priority.
-When processing the queue, a limited number of tablets are processed at once (`max_boot_batch_size` in [configuration](../reference/configuration/hive.md#boot)). This is necessary so that when starting a large number of tablets, Hive does not stop responding to other requests for a long time.
+When processing the queue, a limited number of tablets are processed at once (`max_boot_batch_size` in [configuration](../reference/configuration/hive_config.md#boot)). This is necessary so that when starting a large number of tablets, Hive does not stop responding to other requests for a long time.
If when processing a tablet it turns out that it cannot be started on any of the nodes, then this tablet is postponed to a separate *Wait queue*. When node availability changes (a new node connects, or a restriction is removed from a node in [Hive UI](../reference/embedded-ui/hive.md)), Hive returns to these tablets and when processing the boot queue alternates tablets from the Boot Queue and tablets from the Wait Queue.
@@ -40,7 +40,7 @@ stateDiagram-v2
{% note warning %}
-Simultaneous startup of many tablets can create increased load on a node. Therefore, the maximum number of simultaneously starting tablets on one node is limited by the `max_tablets_scheduled` value from [configuration](../reference/configuration/hive.md#boot). At the same time, if one of the nodes hits this limit, Hive stops starting new tablets on other nodes too, so that this does not affect the uniformity of distribution. This behavior can be controlled using the [`boot_strategy`](../reference/configuration/hive.md#boot) parameter.
+Simultaneous startup of many tablets can create increased load on a node. Therefore, the maximum number of simultaneously starting tablets on one node is limited by the `max_tablets_scheduled` value from [configuration](../reference/configuration/hive_config.md#boot). At the same time, if one of the nodes hits this limit, Hive stops starting new tablets on other nodes too, so that this does not affect the uniformity of distribution. This behavior can be controlled using the [`boot_strategy`](../reference/configuration/hive_config.md#boot) parameter.
{% endnote %}
@@ -48,7 +48,7 @@ Simultaneous startup of many tablets can create increased load on a node. Theref
There are strict restrictions on which nodes are allowed to start a tablet: not every node can start every type of tablet; tablets of a certain database can only be started on nodes of that database. Additionally, when **moving** tablets, overloaded nodes are not considered.
-1. From all suitable nodes, nodes with maximum priority are selected. Priority is determined based on the data center where the node is located. You can explicitly specify data center priorities in the [`default_tablet_preference`](../reference/configuration/hive.md#boot) subsection in the configuration. For [coordinators](../concepts/glossary.md#coordinator) and [mediators](../concepts/glossary.md#mediator), priorities are determined dynamically to maintain them in the same data center when possible. Additionally, if a tablet terminates with an error on a certain node, the priority of that node is lowered for the next start of this tablet.
+1. From all suitable nodes, nodes with maximum priority are selected. Priority is determined based on the data center where the node is located. You can explicitly specify data center priorities in the [`default_tablet_preference`](../reference/configuration/hive_config.md#boot) subsection in the configuration. For [coordinators](../concepts/glossary.md#coordinator) and [mediators](../concepts/glossary.md#mediator), priorities are determined dynamically to maintain them in the same data center when possible. Additionally, if a tablet terminates with an error on a certain node, the priority of that node is lowered for the next start of this tablet.
1. For nodes with maximum priority, a target metric is calculated, which almost matches the [Node usage](hive.md#node-usage) metric. It differs in that only those resources consumed by this particular tablet are considered, as well as the presence of a penalty for the number of tablets of the same [schema object](../concepts/glossary.md#schema-object).
diff --git a/ydb/docs/en/core/dev/json-indexes.md b/ydb/docs/en/core/dev/json-indexes.md
new file mode 100644
index 00000000000..981d907aa35
--- /dev/null
+++ b/ydb/docs/en/core/dev/json-indexes.md
@@ -0,0 +1,244 @@
+# JSON indexes
+
+JSON indexes are a type of secondary index implemented on top of an [inverted index](https://en.wikipedia.org/wiki/Inverted_index) that speeds up filtering table rows by conditions imposed on the contents of columns of type `Json` and `JsonDocument`. The index is used if the `WHERE` predicate uses the functions [JSON_EXISTS](../yql/reference/builtins/json.md) and [JSON_VALUE](../yql/reference/builtins/json.md) with [JsonPath](../yql/reference/builtins/json.md#jsonpath) expressions. Unlike traditional secondary indexes optimized for equality or range searches on individual table columns, a JSON index works with arbitrary paths within a JSON document.
+
+For a general description of JSON search and the structure of an inverted index on JSON document paths, see the [{#T}](../concepts/query_execution/json_search.md) section.
+
+## Characteristics of JSON indexes {#characteristics}
+
+JSON indexes in {{ ydb-short-name }} allow:
+
+* Quickly filter rows by [JSON_EXISTS](#json-exists) and [JSON_VALUE](#json-value) with [JsonPath](../yql/reference/builtins/json.md#jsonpath) expressions
+* combine indexed conditions with `AND` and `OR` operators
+* use query parameter values passed by the application when processing checked predicates.
+
+A JSON index is a [global synchronous](../concepts/glossary.md#secondary-index) index — its data is always consistent with the base table.
+
+When executing a query, a JSON index can be applied:
+
+- explicitly — via the `<table_name> VIEW <index_name>` operator
+- automatically by the [optimizer](../concepts/glossary.md#optimizer), if the predicate matches the formal rules.
+
+## Syntax of JSON indexes {#syntax}
+
+Creating a JSON index:
+
+* when creating a table: [INDEX (CREATE TABLE)](../yql/reference/syntax/create_table/json_index.md)
+* Adding to an existing table: [ALTER TABLE](../yql/reference/syntax/alter_table/indexes.md#add-index).
+
+Deleting JSON indexes is done via [ALTER TABLE](../yql/reference/syntax/alter_table/indexes.md#drop-index):
+
+
+```yql
+ALTER TABLE documents DROP INDEX json_idx
+```
+
+
+Query syntax with explicit JSON index specification:
+
+* [VIEW (JSON index)](../yql/reference/syntax/select/json_index.md).
+
+Functions and expressions for working with JSON in predicates:
+
+* [JSON functions](../yql/reference/builtins/json.md) — `JSON_EXISTS` and `JSON_VALUE`
+* [JsonPath](../yql/reference/builtins/json.md#jsonpath) — a query language for accessing values inside JSON.
+
+Ready-made use cases are collected in the [JSON document search recipes](../recipes/json-search/index.md) section.
+
+## Updating JSON indexes {#update}
+
+JSON indexes are automatically maintained when data is modified and are updated synchronously together with the main table. Tables with JSON indexes support:
+
+* `INSERT`
+* `UPSERT`
+* `REPLACE`
+* `UPDATE`
+* `DELETE`
+
+Batch operations (`BATCH UPDATE` and `BATCH DELETE`) are not supported for tables with JSON indexes. When attempting to execute such a query on a table for which a JSON index has been created, the query will be rejected, a corresponding error will be returned to the application, and the data will remain unchanged.
+
+Additionally, tables with JSON indexes do not support:
+
+* Bulk data loading via a `BulkUpsert` call — the requested operation will be rejected with a corresponding error message.
+* automatic deletion of rows by [TTL](../concepts/ttl.md) — errors are returned when attempting to create a table with both a TTL policy and a JSON index, as well as when trying to retrieve such a combination of properties using the `ALTER TABLE` commands.
+
+## Supported predicates {#predicates}
+
+For execution via JSON indexes, only expressions based on the `JSON_EXISTS` and `JSON_VALUE` functions in the `WHERE` block, combined by the `AND` / `OR` operators according to the rules below, are supported.
+
+### JSON_EXISTS {#json-exists}
+
+Checking the existence of a path or value inside a JsonPath filter.
+
+**Allowed:**
+
+
+```yql
+-- Document root (value not NULL)
+WHERE JSON_EXISTS(doc, '$')
+
+-- Key chain; array indexes are 'transparent'
+WHERE JSON_EXISTS(doc, '$.user.name')
+WHERE JSON_EXISTS(doc, '$.items[*].sku')
+WHERE JSON_EXISTS(doc, '$.items[0 to last].active')
+
+-- Filter ? (...) — predicates inside the filter are allowed
+WHERE JSON_EXISTS(doc, '$.items ? (@.price == 100)')
+WHERE JSON_EXISTS(doc, '$.items ? (@.qty >= 1 && @.qty <= 10)')
+WHERE JSON_EXISTS(doc, '$.items ? (@.tag == $t)' PASSING "sale" AS t)
+
+-- JsonPath methods (path is indexed up to the method; exact check is performed by post-filter)
+WHERE JSON_EXISTS(doc, '$.value.type()')
+WHERE JSON_EXISTS(doc, '$.arr.size()')
+
+-- Combinations on one column
+WHERE JSON_EXISTS(doc, '$.a') AND JSON_EXISTS(doc, '$.b')
+WHERE JSON_EXISTS(doc, '$.a') OR JSON_EXISTS(doc, '$.b')
+```
+
+
+**Prohibited** (error when using the `VIEW` statement or index auto-selection failure):
+
+
+```yql
+-- Comparison predicates at the top level of the path (outside ? (...))
+WHERE JSON_EXISTS(doc, '$.key == 10')
+WHERE JSON_EXISTS(doc, 'exists($.key)')
+WHERE JSON_EXISTS(doc, '$.key starts with "a"')
+
+-- Negation in JsonPath
+WHERE JSON_EXISTS(doc, '!($.key == 10)')
+
+-- ON ERROR TRUE
+WHERE JSON_EXISTS(doc, '$.key' TRUE ON ERROR)
+
+-- Path without context operator ($) — the passed document is not used
+WHERE JSON_EXISTS(doc, '1')
+```
+
+
+{% note info %}
+
+The `JSON_EXISTS` function returns `true` for any non-empty JsonPath result. The `$.key == 10` predicate specified at the top level would give "path existence" even when the comparison is false, which does not match the expected semantics. Comparisons should be moved into `JSON_VALUE` calls or into a filter of the form `? (...)`.
+
+{% endnote %}
+
+### JSON_VALUE {#json-value}
+
+Extracting a scalar value with a required `RETURNING <type>`.
+
+To compare a value using `JSON_VALUE`, you must always specify `RETURNING` with the required type. By default, `JSON_VALUE` returns type `Utf8`, which leads to incorrect comparison during query execution — values of different types are compared as strings:
+
+
+```yql
+$tmp = Json(@@["1", 1]@@);
+SELECT JSON_VALUE($tmp, '$[0]') == "1"; -- true: correct, string compared with string
+SELECT JSON_VALUE($tmp, '$[1]') == "1"; -- true: incorrect, number compared with string
+```
+
+
+Supported types for the `RETURNING` section: `Int8` … `Int64`, `Uint8` … `Uint64`, `Float`, `Double`, `Bytes` (`String`), `Text` (`Utf8`), `Bool`.
+
+**Examples of predicates applied via a JSON index:**
+
+
+```yql
+-- Equality (path + value are included in the index)
+WHERE JSON_VALUE(doc, '$.user.age' RETURNING Int32) = 25
+WHERE JSON_VALUE(doc, '$.flag' RETURNING Bool) = true
+WHERE JSON_VALUE(doc, '$.name' RETURNING Utf8) = "Alice"u
+
+-- Implicit comparison with true for Bool
+WHERE JSON_VALUE(doc, '$.active' RETURNING Bool)
+
+-- Parameters
+WHERE JSON_VALUE(doc, '$.user.id' RETURNING Int64) = $id
+WHERE JSON_VALUE(doc, '$.tag' RETURNING Utf8) = $tag
+
+-- Comparisons (only the path is used in the index, comparison is performed by post-filter)
+WHERE JSON_VALUE(doc, '$.score' RETURNING Int64) > 0
+WHERE JSON_VALUE(doc, '$.score' RETURNING Int64) != 100
+WHERE JSON_VALUE(doc, '$.score' RETURNING Int64) BETWEEN 1 AND 10
+WHERE JSON_VALUE(doc, '$.score' RETURNING Int64) NOT BETWEEN 0 AND 5
+
+-- IN: list of literals
+WHERE JSON_VALUE(doc, '$.status' RETURNING Utf8) IN ("open"u, "pending"u)
+
+-- IN: specified parameter of type List<Utf8>
+WHERE JSON_VALUE(doc, '$.status' RETURNING Utf8) IN $status_list
+
+-- PASSING for JsonPath variables
+WHERE JSON_VALUE(doc, '$.x ? (@.y == $v)' RETURNING Int64 PASSING 42 AS v) = 10
+
+-- JsonPath predicates inside the path.
+-- Unlike JSON_EXISTS, predicates at the top level are allowed.
+WHERE JSON_VALUE(doc, '$.user ? (@.role == "admin")' RETURNING Utf8) = "ok"u
+WHERE JSON_VALUE(doc, '$.code starts with "A"' RETURNING String) != ""
+WHERE JSON_VALUE(doc, 'exists($.meta)' RETURNING Bool)
+
+-- AND / OR combinations on one column
+WHERE JSON_VALUE(doc, '$.a' RETURNING Int32) = 1
+ OR JSON_VALUE(doc, '$.b' RETURNING Int32) = 2
+WHERE JSON_EXISTS(doc, '$.a') AND JSON_VALUE(doc, '$.a' RETURNING Int32) = 10
+```
+
+
+**Examples of predicates that cannot be applied via a JSON index:**
+
+
+```yql
+-- JSON_VALUE call without RETURNING
+WHERE JSON_VALUE(doc, '$.key') = "x"
+
+-- DEFAULT with ON EMPTY / ON ERROR (except NULL)
+WHERE JSON_VALUE(doc, '$.k' RETURNING Utf8 DEFAULT "x" ON ERROR) = "y"
+
+-- Unsupported data type in RETURNING
+WHERE JSON_VALUE(doc, '$.ts' RETURNING Timestamp) = ...
+
+-- RETURNING Bool with comparison operators
+WHERE JSON_VALUE(doc, '$.flag' RETURNING Bool) >= true
+
+-- IS NULL / IS NOT NULL — semantically contradict the 'path existence' index
+WHERE JSON_VALUE(doc, '$.k' RETURNING Utf8) IS NULL
+
+-- Comparison of two JSON_VALUE from different columns
+WHERE JSON_VALUE(doc1, '$.k' RETURNING Utf8) = JSON_VALUE(doc2, '$.k' RETURNING Utf8)
+
+-- Nested JSON_* in arguments
+WHERE JSON_VALUE(JSON_QUERY(doc, '$.a'), '$.b' RETURNING Utf8) = "x"
+```
+
+
+{% note info %}
+
+To check 'value equals `false`' or 'value equals `null`', use a JsonPath filter inside `JSON_EXISTS`, for example `JSON_EXISTS(doc, '$.k ? (@ == false)')` or `JSON_EXISTS(doc, '$.k ? (@ == null)')`, not `JSON_VALUE(...) IS NULL`.
+
+{% endnote %}
+
+## Limitations {#limitations}
+
+* JSON indexes are supported only for [row tables](../concepts/datamodel/table.md#row-oriented-tables).
+* The table's primary key must consist of a single column of an integer type (`Uint64`, `Uint32`, `Int64`, or `Int32`). This is a temporary limitation that will be removed in future development.
+* A single JSON index indexes exactly one column of type `Json` or `JsonDocument`.
+* The `COVER` expression is not supported for JSON indexes.
+* A number of data modification operations and mechanisms are [not supported](#update) for tables with JSON indexes.
+* The parameter type of a read query from the index cannot be wrapped in `Optional<T>` — optional parameters are not supported.
+* Equality comparison with an integer literal whose absolute value exceeds 2⁵³ is not accelerated by the value index (such numbers do not fit into the numeric type used in `Json` and `JsonDocument`) and is reduced to a path existence check.
+* Casting floating-point literals (`Float`, `Double`) to integer types during comparison is not performed — such comparison is not accelerated by the index.
+
+## Recipes {#recipes}
+
+Ready-made scenarios for working with a JSON index:
+
+* [{#T}](../recipes/json-search/json-index-quickstart.md) — quick start.
+* [{#T}](../recipes/json-search/json-index-catalog.md) — product catalog with nested attributes.
+* [{#T}](../recipes/json-search/json-index-parameters.md) — parameterized queries and JsonPath variables.
+* [{#T}](../recipes/json-search/json-index-typecheck.md) — field type check and path existence.
+
+## Related materials {#see-also}
+
+- [JSON functions](../yql/reference/builtins/json.md) — `JSON_EXISTS`, `JSON_VALUE`, `JSON_QUERY`, JsonPath syntax.
+- [Secondary indexes](secondary-indexes.md) — general information about global indexes and `VIEW`.
+- [Full-text indexes](fulltext-indexes.md) — a related mechanism built on top of an inverted index of words and phrases.
+- [INDEX (CREATE TABLE)](../yql/reference/syntax/create_table/json_index.md) and [VIEW (JSON index)](../yql/reference/syntax/select/json_index.md) — syntax reference.
diff --git a/ydb/docs/en/core/dev/resource-consumption-management.md b/ydb/docs/en/core/dev/resource-consumption-management.md
index 3490227ea3b..2137b972a9d 100644
--- a/ydb/docs/en/core/dev/resource-consumption-management.md
+++ b/ydb/docs/en/core/dev/resource-consumption-management.md
@@ -23,14 +23,14 @@ CREATE RESOURCE POOL olap WITH (
CONCURRENT_QUERY_LIMIT=10,
QUEUE_SIZE=1000,
DATABASE_LOAD_CPU_THRESHOLD=80,
- RESOURCES_WEIGHT=100,
+ RESOURCE_WEIGHT=100,
QUERY_CPU_LIMIT_PERCENT_PER_NODE=50,
TOTAL_CPU_LIMIT_PERCENT_PER_NODE=70
)
```
-You can find the full list of resource pool parameters in the [{#T}](../yql/reference/syntax/create-resource-pool.md#parameters) reference. Some parameters are global for the entire database (for example, `CONCURRENT_QUERY_LIMIT`, `QUEUE_SIZE`, `DATABASE_LOAD_CPU_THRESHOLD`), while others apply only to a single compute node (for example, `QUERY_CPU_LIMIT_PERCENT_PER_NODE`, `TOTAL_CPU_LIMIT_PERCENT_PER_NODE`, `TOTAL_MEMORY_LIMIT_PERCENT_PER_NODE`). CPU can be shared among all pools in case of oversubscription on a single compute node using `RESOURCES_WEIGHT`.
+You can find the full list of resource pool parameters in the [{#T}](../yql/reference/syntax/create-resource-pool.md#parameters) reference. Some parameters are global for the entire database (for example, `CONCURRENT_QUERY_LIMIT`, `QUEUE_SIZE`, `DATABASE_LOAD_CPU_THRESHOLD`), while others apply only to a single compute node (for example, `QUERY_CPU_LIMIT_PERCENT_PER_NODE`, `TOTAL_CPU_LIMIT_PERCENT_PER_NODE`, `TOTAL_MEMORY_LIMIT_PERCENT_PER_NODE`). CPU can be shared among all pools in case of oversubscription on a single compute node using `RESOURCE_WEIGHT`.
![resource_pools](../_assets/resource_pool.png)
@@ -68,18 +68,18 @@ When a query enters a resource pool for which `DATABASE_LOAD_CPU_THRESHOLD` is s
As with `CONCURRENT_QUERY_LIMIT`, when the specified load threshold is exceeded, queries are sent to the waiting queue.
-### Resource allocation according to RESOURCES_WEIGHT {#resources_weight}
+### Resource allocation according to RESOURCE_WEIGHT {#resources_weight}
![resource_pools](../_assets/resources_weight.png)
-The `RESOURCES_WEIGHT` parameter only takes effect in case of oversubscription and when there is more than one resource pool in the system. In the current implementation, `RESOURCES_WEIGHT` only affects the allocation of `vCPU` resources. When queries appear in a resource pool, it starts participating in resource allocation. For this, the pools recalculate their limits according to the [Max-min fairness](https://en.wikipedia.org/wiki/Max-min_fairness) algorithm. The actual resource redistribution is performed on each compute node individually, as shown in the figure above.
+The `RESOURCE_WEIGHT` parameter only takes effect in case of oversubscription and when there is more than one resource pool in the system. In the current implementation, `RESOURCE_WEIGHT` only affects the allocation of `vCPU` resources. When queries appear in a resource pool, it starts participating in resource allocation. For this, the pools recalculate their limits according to the [Max-min fairness](https://en.wikipedia.org/wiki/Max-min_fairness) algorithm. The actual resource redistribution is performed on each compute node individually, as shown in the figure above.
Suppose we have a node in the system with $10 vCPU$ available. The following limits are set:
- $TOTAL_CPU_LIMIT_PERCENT_PER_NODE = 30$,
- $QUERY_CPU_LIMIT_PERCENT_PER_NODE = 50$.
-In this case, the resource pool will have a limit of $3 vCPU$ per node and $1.5 vCPU$ per query in this pool (figure *a*). If there are 4 such pools in the system and they all try to use maximum resources, this would amount to $12 vCPU$, which exceeds the limit of available resources on the node ($10 vCPU$). In this case, `RESOURCES_WEIGHT` takes effect, and each pool will be allocated $2.5 vCPU$ (figure *b*).
+In this case, the resource pool will have a limit of $3 vCPU$ per node and $1.5 vCPU$ per query in this pool (figure *a*). If there are 4 such pools in the system and they all try to use maximum resources, this would amount to $12 vCPU$, which exceeds the limit of available resources on the node ($10 vCPU$). In this case, `RESOURCE_WEIGHT` takes effect, and each pool will be allocated $2.5 vCPU$ (figure *b*).
If you need to increase the allocated resources for a specific pool, you can change its weight, for example, to 200. Then this pool will get $3 vCPU$, and the remaining pools will equally share the remaining $7 vCPU$, which amounts to $\frac{7}{3} vCPU$ per pool (figure *c*).
@@ -99,7 +99,7 @@ CREATE RESOURCE POOL default WITH (
CONCURRENT_QUERY_LIMIT=-1,
QUEUE_SIZE=-1,
DATABASE_LOAD_CPU_THRESHOLD=-1,
- RESOURCES_WEIGHT=-1,
+ RESOURCE_WEIGHT=-1,
TOTAL_MEMORY_LIMIT_PERCENT_PER_NODE=-1,
QUERY_CPU_LIMIT_PERCENT_PER_NODE=-1,
TOTAL_CPU_LIMIT_PERCENT_PER_NODE=-1
@@ -200,7 +200,7 @@ CREATE RESOURCE POOL olap WITH (
CONCURRENT_QUERY_LIMIT=20,
QUEUE_SIZE=100,
DATABASE_LOAD_CPU_THRESHOLD=80,
- RESOURCES_WEIGHT=20,
+ RESOURCE_WEIGHT=20,
QUERY_CPU_LIMIT_PERCENT_PER_NODE=80,
TOTAL_CPU_LIMIT_PERCENT_PER_NODE=100
);
@@ -208,7 +208,7 @@ CREATE RESOURCE POOL olap WITH (
CREATE RESOURCE POOL the_ceo WITH (
CONCURRENT_QUERY_LIMIT=20,
QUEUE_SIZE=100,
- RESOURCES_WEIGHT=100,
+ RESOURCE_WEIGHT=100,
QUERY_CPU_LIMIT_PERCENT_PER_NODE=100,
TOTAL_CPU_LIMIT_PERCENT_PER_NODE=100
);
@@ -303,11 +303,11 @@ The following query outputs information about all active queries in the system:
```yql
select
- Query, -- Запрос
- WmPoolId, -- Идентификатор пула
- WmState, -- Статус запроса в WM
- WmEnterTime, -- Время, когда запрос перешел в статус PENDING или DELAYED
- WmExitTime -- Время, когда запрос передан на выполнение
+ Query, -- Query
+ WmPoolId, -- Pool ID
+ WmState, -- Query status in WM
+ WmEnterTime, -- Time when the query transitioned to PENDING or DELAYED status
+ WmExitTime -- Time when the query was submitted for execution
from `.sys/query_sessions`
where State = 'EXECUTING'
```
diff --git a/ydb/docs/en/core/dev/toc_p.yaml b/ydb/docs/en/core/dev/toc_p.yaml
index a02a2f17fb5..52fddc0e4ed 100644
--- a/ydb/docs/en/core/dev/toc_p.yaml
+++ b/ydb/docs/en/core/dev/toc_p.yaml
@@ -31,6 +31,8 @@ items:
href: fulltext-indexes.md
- name: Hybrid search
href: hybrid-search.md
+- name: JSON indexes
+ href: json-indexes.md
- name: Local indexes
href: local-indexes/index.md
include:
@@ -51,11 +53,11 @@ items:
href: system-views.md
- name: Change Data Capture
href: cdc.md
-- name: Structure and limitations of a shared topic reader
- href: shared-consumer-internals.md
- name: Terraform
href: terraform.md
- name: Custom attributes
href: custom-attributes.md
- name: Resource consumption management
href: resource-consumption-management.md
+- name: Structure and limitations of a shared topic reader
+ href: shared-consumer-internals.md
diff --git a/ydb/docs/en/core/devops/enterprise-manager/ai-assistant.md b/ydb/docs/en/core/devops/enterprise-manager/ai-assistant.md
new file mode 100644
index 00000000000..c9fcb5cd8ab
--- /dev/null
+++ b/ydb/docs/en/core/devops/enterprise-manager/ai-assistant.md
@@ -0,0 +1,205 @@
+# Configuring AI assistant in YDB EM
+
+This guide shows how to enable AI assistant in {{ ydb-short-name }} Enterprise Manager (YDB EM). After the setup, users will see the assistant in the YDB EM web interface. The assistant sends model requests through [Gateway](index.md#architecture) and can use Model Context Protocol (MCP) tools provided by Gateway.
+
+## Before You Start {#before-start}
+
+You can use this guide before the first YDB EM deployment or when updating an existing installation. For a new deployment, add the variables to the inventory before running the initial setup playbook. For deployment instructions, see [{#T}](initial-deployment.md).
+
+Make sure that you have:
+
+1. Access to the Ansible inventory used to deploy YDB EM.
+1. An endpoint of an OpenAI-compatible model that will be reachable from the Gateway host.
+1. Permissions to modify the [Gateway token file](#configure-model-access). During setup you will need to write the model access secret into the token file. Gateway reads this file and uses the `Token` value of the entry whose `Name` equals `ydb_em_ai_model_token_name`, for example `model-token`, as the upstream `Authorization` header.
+1. If you are updating an existing installation, access to the deployed Gateway host.
+
+For an existing installation, first confirm that the Gateway host is managed by the same Ansible inventory. Check the active service manager unit, process owner, config path, token file, and listening port before applying the playbook. If the host uses a custom or manual Gateway layout, apply the same settings through that installation's operational procedure instead of running the playbook blindly.
+
+{% note warning %}
+
+Do not put model API keys, OAuth tokens, or other secrets into `ydb_em_ai_assistant_client_runtime_config`. Gateway returns this value to the browser from `GET /meta/ai_assistant_client_config`. Keep secrets in the token file.
+
+{% endnote %}
+
+## How It Works {#how-it-works}
+
+The browser works with the assistant through Gateway:
+
+1. The UI checks `GET /capabilities`. The assistant button is shown when `Settings.Proxy.Model` is `true` and the user setting for AI assistant is enabled. When the backend capability first appears, YDB EM initializes this user setting to enabled.
+1. The UI gets runtime settings from `GET /meta/ai_assistant_client_config`.
+1. The UI sends model requests to `/proxy/model/...`.
+1. Gateway forwards model requests to `ydb_em_ai_model_endpoint` and adds the Authorization header from tokenator.
+1. The UI sends MCP requests to `/meta/mcp`. Gateway exposes MCP tools through this endpoint and runs them against YDB EM APIs and cluster proxy endpoints. When documentation search is configured, `search_docs` is exposed through the same MCP endpoint and can be used by the assistant.
+
+## Configure Model Access {#configure-model-access}
+
+`ydb_em_ai_model_token_name` is the [tokenator](../../concepts/glossary.md#tokenator) entry name that Gateway uses for model requests.
+
+Example tokenator entry:
+
+```textproto
+StaticTokenInfo {
+ Name: "model-token"
+ Token: "Bearer <model-api-token>"
+}
+```
+
+The token delivery method depends on your deployment process. The requirement is that the Gateway tokenator file contains an entry whose `Name` matches `ydb_em_ai_model_token_name`.
+
+{% note info %}
+
+The default YDB EM Ansible token template renders only the `meta` credentials entry. If your installation uses that template unchanged, extend or override the tokenator file through your deployment process so that it also contains `model-token` and, if different, the embeddings token.
+
+{% endnote %}
+
+## Configure Ansible Variables {#configure-ansible-variables}
+
+Add the AI assistant variables to the YDB EM inventory, for example to `examples/inventory/50-inventory.yaml` or another inventory file used by your deployment.
+
+```yaml
+ydb_em_ai_assistant_enabled: true
+ydb_em_ai_model_endpoint: "https://llm.example.com"
+ydb_em_ai_model_token_name: "model-token"
+
+ydb_em_ai_assistant_client_runtime_config:
+ llm:
+ baseURL: "/proxy/model/v1"
+ model: "<model-name>"
+ temperature: 0.2
+ mcp:
+ - id: "ydb-meta"
+ url: "/meta/mcp"
+ transport: "streamable"
+ toolCallTimeoutMs: 600000
+```
+
+Parameter | Description
+--- | ---
+`ydb_em_ai_assistant_enabled` | Enables AI assistant settings in Gateway.
+`ydb_em_ai_model_endpoint` | Upstream model API endpoint used by Gateway.
+`ydb_em_ai_model_token_name` | Tokenator entry name for model requests.
+`ydb_em_ai_assistant_client_runtime_config.llm.baseURL` | Browser-visible model URL. Use Gateway proxy, usually `/proxy/model/v1`.
+`ydb_em_ai_assistant_client_runtime_config.llm.model` | Model name sent to the OpenAI-compatible API.
+`ydb_em_ai_assistant_client_runtime_config.mcp` | MCP servers available to the assistant. For YDB EM, use `/meta/mcp`.
+
+Gateway appends the request suffix from `/proxy/model/...` to `ydb_em_ai_model_endpoint`. With `baseURL: "/proxy/model/v1"`, a chat completion request is forwarded to `/v1/chat/completions` on the upstream endpoint. Configure the endpoint so that this path is valid for your model provider.
+
+## Configure Documentation Search {#configure-docs-search}
+
+It is recommended to enable documentation search so that the assistant gets the `search_docs` MCP tool. Gateway calls an OpenAI-compatible embeddings endpoint for this tool and appends `/embeddings` to the configured base URL when the suffix is missing. When documentation search is enabled, the assistant gets `search_docs` through the configured `/meta/mcp` server.
+
+```yaml
+ydb_em_docs_search_enabled: true
+ydb_em_docs_search_embeddings_upstream_base_url: "https://llm.example.com/v1"
+ydb_em_docs_search_embeddings_token_name: "model-token"
+ydb_em_docs_search_embeddings_model: "<embeddings-model-name>"
+ydb_em_docs_search_vector_size: 0
+ydb_em_docs_search_limit: 6
+ydb_em_docs_search_score: 0.6
+```
+
+Parameter | Description
+--- | ---
+`ydb_em_docs_search_enabled` | Configures the semantic docs search backend and exposes `search_docs` through MCP when all required search settings are valid.
+`ydb_em_docs_search_embeddings_upstream_base_url` | OpenAI-compatible embeddings endpoint base URL.
+`ydb_em_docs_search_embeddings_token_name` | Tokenator entry name for embeddings requests.
+`ydb_em_docs_search_embeddings_model` | Embeddings model name.
+`ydb_em_docs_search_vector_size` | Optional embedding vector dimension override. Leave it unset to use the Ansible default `0`. With `0`, Gateway omits the OpenAI-compatible `dimensions` field and skips the response size check. Set a positive value only when the embeddings provider and model support explicit dimensions and the expected size is known.
+`ydb_em_docs_search_limit` | Maximum number of documents returned.
+`ydb_em_docs_search_score` | Minimum similarity score from `0` to `1`. Documents with a lower score are filtered out.
+
+## Apply The Configuration {#apply-configuration}
+
+After updating the inventory, run the YDB EM playbook from the directory with your inventory. The same playbook is used for initial deployment and for applying this change to an existing installation:
+
+```bash
+ansible-playbook ydb_platform.ydb_em.initial_setup
+```
+
+If your vault file is encrypted, add the vault option used in your deployment:
+
+```bash
+ansible-playbook ydb_platform.ydb_em.initial_setup --ask-vault-pass
+```
+
+The role renders the Gateway config and the default token file and ensures that the Gateway service is started. If you deliver additional model credentials through an overridden token file or another deployment step, make sure the deployed token file contains the configured entry name. When applying these settings to an already running deployment, restart the Gateway service using your operational procedure if your Ansible run does not restart it after config or token changes.
+
+## Verify The Setup {#verify}
+
+Open the YDB EM web interface:
+
+```text
+https://<gateway-host>:8789/ui/clusters
+```
+
+Check that model proxy is enabled:
+
+```bash
+curl -k https://<gateway-host>:8789/capabilities
+```
+
+The response should contain:
+
+```json
+{
+ "Settings": {
+ "Proxy": {
+ "Model": true
+ }
+ }
+}
+```
+
+Check the runtime config returned to the UI:
+
+```bash
+curl -k https://<gateway-host>:8789/meta/ai_assistant_client_config
+```
+
+The response should contain the `llm` block and should not contain model secrets.
+
+Check the model proxy:
+
+```bash
+curl -k https://<gateway-host>:8789/proxy/model/v1/chat/completions \
+ -H 'Content-Type: application/json' \
+ -d '{"model":"<model-name>","messages":[{"role":"user","content":"ping"}],"max_tokens":1}'
+```
+
+If your deployment requires user authentication, add the same authentication headers that are used for the YDB EM UI.
+
+If documentation search is enabled, also check that `/meta/mcp` exposes `search_docs` and run a harmless documentation-search request through your MCP client or assistant tooling. This verifies both the MCP registration and the embeddings and metabase path.
+
+## Troubleshooting {#troubleshooting}
+
+### Assistant Button Is Not Shown {#button-not-shown}
+
+Check that `ydb_em_ai_assistant_enabled` is `true` and that `/capabilities` contains `Settings.Proxy.Model: true`. Also check the user setting for AI assistant in the YDB EM UI: YDB EM initializes it to enabled when the backend capability is present, but a user can turn it off.
+
+### Runtime Config Is Invalid {#runtime-config-invalid}
+
+Check `GET /meta/ai_assistant_client_config`. For an enabled assistant, the response must be a JSON object with `llm.baseURL` and `llm.model` string fields. A `null` response means that Gateway did not load an enabled AI assistant client config.
+
+### Model Proxy Returns An Authorization Error {#model-authorization-error}
+
+Check that `ydb_em_ai_model_token_name` matches a tokenator entry and that the entry returns the Authorization header expected by the model endpoint. If tokenator cannot find the entry, Gateway sends the upstream request without the expected Authorization header.
+
+### Model Proxy Calls The Wrong Upstream Path {#wrong-upstream-path}
+
+Check `ydb_em_ai_model_endpoint` together with `llm.baseURL`. Gateway appends the path from `/proxy/model/...` to the configured endpoint. A duplicated `/v1` path often causes upstream `404` errors.
+
+### MCP Tools Are Unavailable {#mcp-tools-unavailable}
+
+Check that the runtime config contains `url: "/meta/mcp"` and that user requests can reach `/meta/mcp` through Gateway. Gateway registers `/meta/mcp` only when MCP is enabled; it is enabled by default, but custom Gateway config can disable it.
+
+### Documentation Search Is Unavailable {#docs-search-unavailable}
+
+Check the `ydb_em_docs_search_*` settings, the embeddings endpoint path, and the tokenator entry used for embeddings requests. Also make sure documentation vectors are available in the YDB EM metabase. If docs search is not configured, the `search_docs` tool is not exposed.
+
+### Embeddings Endpoint Rejects Vector Size {#embeddings-vector-size}
+
+If `search_docs` fails with `variable embedding size not supported`, set `ydb_em_docs_search_vector_size: 0` and verify the embeddings endpoint and model.
+
+### Documentation Search Returns 504 {#docs-search-504}
+
+If `search_docs` returns `504 Gateway Timeout` and Gateway logs for `/meta/docs` contain `Member not found: doc_id`, check the YDB EM metabase table `ydb/Docs.db`. The current Gateway docs search query uses the `doc_id`, `embedding`, and `payload` columns. For useful results, the table should also contain rows and be built for an embeddings model compatible with `ydb_em_docs_search_embeddings_model`.
diff --git a/ydb/docs/en/core/devops/enterprise-manager/index.md b/ydb/docs/en/core/devops/enterprise-manager/index.md
index 3c17a8910bb..539c14ba7ee 100644
--- a/ydb/docs/en/core/devops/enterprise-manager/index.md
+++ b/ydb/docs/en/core/devops/enterprise-manager/index.md
@@ -81,3 +81,4 @@ The user interacts with Gateway through a browser or API. Gateway forwards reque
- [{#T}](initial-deployment.md)
- [{#T}](s3-backups.md)
+- [{#T}](ai-assistant.md)
diff --git a/ydb/docs/en/core/devops/enterprise-manager/toc_p.yaml b/ydb/docs/en/core/devops/enterprise-manager/toc_p.yaml
index 62bbdf45692..9d215745076 100644
--- a/ydb/docs/en/core/devops/enterprise-manager/toc_p.yaml
+++ b/ydb/docs/en/core/devops/enterprise-manager/toc_p.yaml
@@ -3,3 +3,5 @@ items:
href: initial-deployment.md
- name: Configuring S3 backups
href: s3-backups.md
+- name: Configuring AI assistant
+ href: ai-assistant.md
diff --git a/ydb/docs/en/core/downloads/ydb-ansible.md b/ydb/docs/en/core/downloads/ydb-ansible.md
index 1bbaaec590b..bf92da7df0c 100644
--- a/ydb/docs/en/core/downloads/ydb-ansible.md
+++ b/ydb/docs/en/core/downloads/ydb-ansible.md
@@ -4,6 +4,7 @@ A set of automated playbooks for installing and maintaining the server side of [
| Version | Release date | Download | Changelog |
| ------ | ------------ | ------- | ----------------- |
+| v2.1.0 | 29.06.2026 | [ydb-ansible-2.1.0.tar.gz](https://github.com/ydb-platform/ydb-ansible/releases/download/v2.1.0/ydb_platform-ydb-2.1.0.tar.gz) | |
| v2.0.0 | 23.12.2025 | [ydb-ansible-2.0.0.tar.gz](https://github.com/ydb-platform/ydb-ansible/releases/download/v2.0.0/ydb_platform-ydb-2.0.0.tar.gz) | |
| v1.3.2 | 02.12.2025 | [ydb-ansible-1.3.2.tar.gz](https://github.com/ydb-platform/ydb-ansible/releases/download/v1.3.2/ydb_platform-ydb-1.3.2.tar.gz) | |
| v1.3.1 | 01.12.2025 | [ydb-ansible-1.3.1.tar.gz](https://github.com/ydb-platform/ydb-ansible/releases/download/v1.3.1/ydb_platform-ydb-1.3.1.tar.gz) | |
diff --git a/ydb/docs/en/core/recipes/json-search/index.md b/ydb/docs/en/core/recipes/json-search/index.md
new file mode 100644
index 00000000000..b6568a26d8f
--- /dev/null
+++ b/ydb/docs/en/core/recipes/json-search/index.md
@@ -0,0 +1,12 @@
+# Recipes for searching JSON documents
+
+This section contains ready-made recipes for using [JSON indexes](../../dev/json-indexes.md) to efficiently select data based on the contents of columns of type `Json` and `JsonDocument`.
+
+The recipes are built around the [JSON_EXISTS](../../yql/reference/builtins/json.md) and [JSON_VALUE](../../yql/reference/builtins/json.md) functions with [JsonPath](../../yql/reference/builtins/json.md#jsonpath) expressions. To guarantee index usage, all examples use explicit access via `VIEW IndexName`.
+
+* [{#T}](json-index-quickstart.md) — minimal scenario: creating a table, an index, inserting data, and basic queries.
+* [{#T}](json-index-catalog.md) — example of a product catalog with nested attributes, range conditions, and searching nested arrays.
+* [{#T}](json-index-parameters.md) — parameterized queries and passing variables to JsonPath via the `PASSING` section.
+* [{#T}](json-index-typecheck.md) — checking the field type and path existence using JsonPath methods.
+
+Before running the examples, make sure that [JSON index support is enabled](../../reference/configuration/feature_flags.md) on the cluster.
diff --git a/ydb/docs/en/core/recipes/json-search/json-index-catalog.md b/ydb/docs/en/core/recipes/json-search/json-index-catalog.md
new file mode 100644
index 00000000000..f9d6fcda059
--- /dev/null
+++ b/ydb/docs/en/core/recipes/json-search/json-index-catalog.md
@@ -0,0 +1,133 @@
+# Catalog with nested attributes
+
+This recipe shows how to use a [JSON index](../../dev/json-indexes.md) to speed up access to a product catalog where product characteristics are stored as a JSON document with an arbitrary set of fields. This schema is convenient when:
+
+* The set of product attributes is unknown in advance or differs for different categories.
+* Adding new attributes via `ALTER TABLE ADD COLUMN` is undesirable.
+* You need to efficiently filter products by various combinations of attributes.
+
+A JSON index enables efficient retrieval by any path and value inside a JSON document without a full table scan.
+
+## Create a table and index
+
+
+```yql
+CREATE TABLE products (
+ sku_id Uint64,
+ attrs JsonDocument,
+ PRIMARY KEY (sku_id),
+ INDEX attrs_json_idx GLOBAL USING json ON (attrs)
+);
+```
+
+
+In this schema:
+
+* `sku_id` — numeric product identifier.
+* `attrs` — product attributes in `JsonDocument` format. Storing as `JsonDocument` saves space and speeds up deserialization compared to `Json`.
+* `attrs_json_idx` — JSON index on column `attrs`. It is updated synchronously together with the main table.
+
+## Load test data
+
+
+```yql
+UPSERT INTO products (sku_id, attrs) VALUES
+ (10, JsonDocument(@@{
+ "brand": "ACME",
+ "price": 49.90,
+ "category": "tools",
+ "warehouses": [{"id": 1, "stock": 12}, {"id": 2, "stock": 0}]
+ }@@)),
+ (11, JsonDocument(@@{
+ "brand": "ACME",
+ "price": 199.00,
+ "category": "electronics",
+ "warehouses": [{"id": 1, "stock": 3}]
+ }@@)),
+ (12, JsonDocument(@@{
+ "brand": "Globex",
+ "price": 25.00,
+ "category": "tools",
+ "warehouses": [{"id": 2, "stock": 0}]
+ }@@));
+```
+
+
+## Filter by brand and price range
+
+
+```yql
+SELECT sku_id, attrs
+FROM products VIEW attrs_json_idx
+WHERE JSON_VALUE(attrs, '$.brand' RETURNING Utf8) = "ACME"u
+ AND JSON_VALUE(attrs, '$.price' RETURNING Double) BETWEEN 10.0 AND 100.0;
+```
+
+
+What happens:
+
+* For the `$.brand = "ACME"` condition, the token «path + value» goes into the index — this gives an exact match and maximum selectivity.
+* For the `BETWEEN 10.0 AND 100.0` condition, only the path token `$.price` goes into the index. This narrows the set of rows to those where the `price` field is present, after which the query execution engine checks the range with an exact comparison (post-filter).
+* The conditions are combined via `AND` — this allows the index to be used to check both fragments at once, resulting in minimal index reads.
+
+Result:
+
+
+```text
+sku_id attrs
+10 {"brand":"ACME","price":49.9,...}
+```
+
+
+## Search by nested array
+
+JsonPath supports accessing array elements and filters within the path. For example, you can find products that have stock in at least one warehouse:
+
+
+```yql
+SELECT sku_id
+FROM products VIEW attrs_json_idx
+WHERE JSON_EXISTS(attrs, '$.warehouses ? (@.stock > 0)');
+```
+
+
+For index search, the path token `$.warehouses.stock` is used, and the `@.stock > 0` condition is checked by a post-filter on each found record. This is efficient when the corresponding field is present in a relatively small portion of documents.
+
+Result:
+
+
+```text
+sku_id
+10
+11
+```
+
+
+## Search by category and attribute presence
+
+Conditions can be combined with any `AND` and `OR` operators. Search for products in category `tools` that have the `price` field specified:
+
+
+```yql
+SELECT sku_id
+FROM products VIEW attrs_json_idx
+WHERE JSON_VALUE(attrs, '$.category' RETURNING Utf8) = "tools"u
+ AND JSON_EXISTS(attrs, '$.price');
+```
+
+
+Result:
+
+
+```text
+sku_id
+10
+12
+```
+
+
+## See also
+
+* [Supported JSON index predicates](../../dev/json-indexes.md#predicates) — full rules for which expressions are indexed.
+* [AND and OR handling](../../dev/json-indexes.md#predicates) — nuances of combining conditions, including with non-indexable predicates.
+* [{#T}](json-index-parameters.md) — parameterized versions of the same queries.
diff --git a/ydb/docs/en/core/recipes/json-search/json-index-parameters.md b/ydb/docs/en/core/recipes/json-search/json-index-parameters.md
new file mode 100644
index 00000000000..6759f2215b3
--- /dev/null
+++ b/ydb/docs/en/core/recipes/json-search/json-index-parameters.md
@@ -0,0 +1,139 @@
+# Parameterized queries and JsonPath variables
+
+In most applications, query input data is not substituted into SQL text but is passed via [parameters](../../yql/reference/syntax/declare.md). Typical use cases for parameters when working with [JSON indexes](../../dev/json-indexes.md):
+
+1. direct comparison of the result of `JSON_VALUE` with a parameter.
+2. checking whether the result is in a list of values using the `IN` expression.
+3. passing a parameter to a JsonPath expression via the `PASSING` clause.
+
+## Prerequisites
+
+Below is an example of a table, a JSON index, and data population.
+
+
+```yql
+CREATE TABLE documents (
+ id Uint64,
+ payload JsonDocument,
+ PRIMARY KEY (id),
+ INDEX json_idx GLOBAL USING json ON (payload)
+);
+
+UPSERT INTO documents (id, payload) VALUES
+ (1, JsonDocument(@@{
+ "owner_id": 100,
+ "tag": "active",
+ "archived": false,
+ "content": {"x": 1, "y": 1}
+ }@@)),
+ (2, JsonDocument(@@{
+ "owner_id": 100,
+ "tag": "draft",
+ "archived": false,
+ "content": {"x": 1, "y": 2}
+ }@@)),
+ (3, JsonDocument(@@{
+ "owner_id": 101,
+ "tag": "active",
+ "archived": true,
+ "content": {"x": 2, "y": 1}
+ }@@)),
+ (4, JsonDocument(@@{
+ "owner_id": 102,
+ "tag": "pending",
+ "archived": false,
+ "content": {"x": 2, "y": 2}
+ }@@));
+```
+
+
+## Direct comparison with a parameter
+
+The most common case is that the parameter is substituted as the right operand of the comparison, and the left operand is a call to `JSON_VALUE` with an explicit `RETURNING`:
+
+
+```yql
+DECLARE $owner_id AS Int64;
+
+SELECT id, payload
+FROM documents VIEW json_idx
+WHERE JSON_VALUE(payload, '$.owner_id' RETURNING Int64) = $owner_id
+ AND JSON_EXISTS(payload, '$.archived ? (@ == false)');
+```
+
+
+The index receives a token «path + value» (`$.owner_id = $owner_id`) — the parameter is treated as a regular value. This allows performing a query with the same selectivity as when comparing with a literal.
+
+Running with `$owner_id = 100` returns rows `1` and `2`.
+
+## Search by a list of values
+
+To search by multiple values of a single field, you can use `IN`:
+
+
+```yql
+DECLARE $tags AS List<Utf8>;
+
+SELECT id
+FROM documents VIEW json_idx
+WHERE JSON_VALUE(payload, '$.tag' RETURNING Utf8) IN $tags;
+```
+
+
+The behavior of the index differs depending on whether it is a literal or a parameter:
+
+* `IN ("active"u, "pending"u)` — a list of literals is converted into `OR` of several «path + value» tokens, one for each value in the list.
+* `IN $tags` — for each parameter value during query execution, a «path + value» token is formed (`$.tag` + value from the list). At the compilation stage, the parameter values are not yet known, so in the query plan (`EXPLAIN`) such a condition is displayed as a «path + parameter» pair (`{"path": "$.tag", "param": "$tags"}`).
+
+Any collection of scalar values can serve as a parameter for `IN`: `List<T>`, `Tuple<T, ...>`, `Dict<K, V>`, or `Set<T>`.
+
+Running with `$tags = ["active"u, "pending"u]` returns rows `1`, `3`, and `4`.
+
+## Parameters inside JsonPath (PASSING)
+
+If a parameter must be used inside a JsonPath filter (`? (...)`), it is passed in the `PASSING` clause:
+
+
+```yql
+DECLARE $min_stock AS Int64;
+
+SELECT id
+FROM documents VIEW json_idx
+WHERE JSON_EXISTS(
+ payload,
+ '$.content ? (@.y > $threshold)'
+ PASSING $min_stock AS threshold
+);
+```
+
+
+For index search, the path token `$.content.y` is used, and the condition `@.y > $threshold` is checked by a post-filter.
+
+Similarly, `PASSING` works in `JSON_VALUE`:
+
+
+```yql
+DECLARE $v AS Int64;
+
+SELECT id
+FROM documents VIEW json_idx
+WHERE JSON_VALUE(
+ payload,
+ '$.content ? (@.y == $val)'
+ PASSING $v AS val
+ RETURNING Int64
+) = 10;
+```
+
+
+## Supported parameter types
+
+For all three methods, parameters with the following types are supported: `Int8` … `Int64`, `Uint8` … `Uint64`, `Float`, `Double`, `Bytes` (`String`), `Text` (`Utf8`), `Bool`. Optional types (`Optional<T>`) are not supported in parameters.
+
+For more details about types, see [{#T}](../../dev/json-indexes.md#json-value).
+
+## See also
+
+* [{#T}](json-index-quickstart.md) — basic use case for a JSON index.
+* [Passing parameters to JSON index predicates](../../dev/json-indexes.md#json-value) — detailed description of all options.
+* [JsonPath](../../yql/reference/builtins/json.md#jsonpath) — JsonPath language syntax.
diff --git a/ydb/docs/en/core/recipes/json-search/json-index-quickstart.md b/ydb/docs/en/core/recipes/json-search/json-index-quickstart.md
new file mode 100644
index 00000000000..4a15dc1d7db
--- /dev/null
+++ b/ydb/docs/en/core/recipes/json-search/json-index-quickstart.md
@@ -0,0 +1,110 @@
+# JSON index – quick start
+
+This guide shows how to create a [JSON index](../../dev/json-indexes.md) and run queries using the [JSON_EXISTS](../../yql/reference/builtins/json.md#json_exists) and [JSON_VALUE](../../yql/reference/builtins/json.md#json_value) functions in {{ ydb-short-name }}.
+
+## Create a table and a JSON index
+
+
+```yql
+CREATE TABLE documents (
+ id Uint64,
+ payload JsonDocument,
+ PRIMARY KEY (id),
+ INDEX json_idx GLOBAL USING json ON (payload)
+);
+```
+
+
+The column type `JsonDocument` stores JSON in a compact binary format and is preferred for an indexed column. Alternatively, the type `Json` (text representation) can be used.
+
+The primary key of the table must consist of a single column of an integer type (`Uint64`, `Uint32`, `Int64`, or `Int32`) — this is a [current limitation](../../dev/json-indexes.md#limitations) of the JSON index implementation.
+
+## Add test data
+
+
+```yql
+UPSERT INTO documents (id, payload) VALUES
+ (1, JsonDocument(@@{"user": {"id": 100, "name": "Alice"}, "active": true}@@)),
+ (2, JsonDocument(@@{"user": {"id": 101, "name": "Bob"}, "active": false}@@)),
+ (3, JsonDocument(@@{"user": {"id": 102, "name": "Charlie"}, "archived": true}@@));
+```
+
+
+Here, the `@@...@@` construct is a [multiline string literal](../../yql/reference/syntax/expressions.md), convenient for writing JSON without escaping quotes. The `JsonDocument(...)` function converts text into a value of type `JsonDocument`.
+
+## Filter by the presence of a path in the document
+
+The [JSON_EXISTS](../../yql/reference/builtins/json.md#json_exists) function checks whether a path specified by a JsonPath expression exists in the document.
+
+
+```yql
+SELECT id
+FROM documents VIEW json_idx
+WHERE JSON_EXISTS(payload, '$.user.id');
+```
+
+
+Result:
+
+
+```text
+id
+1
+2
+3
+```
+
+
+The path token `$.user.id` is used for index search. The index returns the result without scanning the main table.
+
+## Selecting rows with a specific document field value
+
+The [JSON_VALUE](../../yql/reference/builtins/json.md#json_value) function extracts a scalar value by JsonPath; to use the index, you must specify the return type in the `RETURNING` clause:
+
+
+```yql
+SELECT id
+FROM documents VIEW json_idx
+WHERE JSON_VALUE(payload, '$.user.name' RETURNING Utf8) = "Alice"u;
+```
+
+
+Result:
+
+
+```text
+id
+1
+```
+
+
+When checking equality, the token «path + value» (`$.user.name = "Alice"`) is placed into the index, which ensures the highest selectivity.
+
+## Combination of conditions
+
+Multiple calls `JSON_EXISTS` / `JSON_VALUE` on a single indexed JSON column can be combined using the `AND` and `OR` operators:
+
+
+```yql
+SELECT id
+FROM documents VIEW json_idx
+WHERE JSON_EXISTS(payload, '$.user.id')
+ AND JSON_VALUE(payload, '$.active' RETURNING Bool);
+```
+
+
+Result:
+
+
+```text
+id
+1
+```
+
+
+## See also
+
+* [JSON indexes](../../dev/json-indexes.md) — a complete overview of capabilities and limitations.
+* [VIEW (JSON index)](../../yql/reference/syntax/select/json_index.md) — query syntax through `VIEW`.
+* [INDEX (CREATE TABLE)](../../yql/reference/syntax/create_table/json_index.md) — syntax for creating a JSON index.
+* [{#T}](json-index-catalog.md) — example of a product catalog with nested attributes.
diff --git a/ydb/docs/en/core/recipes/json-search/json-index-typecheck.md b/ydb/docs/en/core/recipes/json-search/json-index-typecheck.md
new file mode 100644
index 00000000000..1ec8fc63af4
--- /dev/null
+++ b/ydb/docs/en/core/recipes/json-search/json-index-typecheck.md
@@ -0,0 +1,94 @@
+# Checking field type and path existence
+
+This recipe shows how a [JSON index](../../dev/json-indexes.md) is used to check document structure: the presence of a specific path, the type of value at the path, etc. These tasks are common when working with heterogeneous JSON documents where some fields are optional or may have different types.
+
+## Prerequisites
+
+
+```yql
+CREATE TABLE documents (
+ id Uint64,
+ payload JsonDocument,
+ PRIMARY KEY (id),
+ INDEX json_idx GLOBAL USING json ON (payload)
+);
+
+UPSERT INTO documents (id, payload) VALUES
+ (1, JsonDocument(@@{"archived": false, "value": 1, "data": [1, 2, 3]}@@)),
+ (2, JsonDocument(@@{"archived": false, "value": 2, "data": "plain text"}@@)),
+ (3, JsonDocument(@@{"archived": true, "value": 3, "data": {"nested": true}}@@)),
+ (4, JsonDocument(@@{"archived": false, "value": null, "meta": "no data field"}@@));
+```
+
+
+## Find documents where a field contains an array
+
+The JsonPath method `.type()` returns a string name of the value type at the specified path. This allows filtering documents by content type:
+
+
+```yql
+SELECT id
+FROM documents VIEW json_idx
+WHERE JSON_VALUE(payload, '$.data.type()' RETURNING Utf8) = "array"u;
+```
+
+
+The path token `$.data` is added to the index — the method `.type()` completes the path token construction. The exact check of the string value `"array"` is performed by a post-filter.
+
+Result:
+
+
+```text
+id
+1
+```
+
+
+The full list of values returned by the method `.type()` is given in the [JsonPath](../../yql/reference/builtins/json.md#jsonpath) syntax description.
+
+## Find documents where a field is a non-empty array
+
+The method `.size()` returns the number of array elements (or 1 for a scalar, 0 for a missing path). In combination with a type filter:
+
+
+```yql
+SELECT id
+FROM documents VIEW json_idx
+WHERE JSON_VALUE(payload, '$.data.type()' RETURNING Utf8) = "array"u
+ AND JSON_VALUE(payload, '$.data.size()' RETURNING Int64) > 0;
+```
+
+
+Both fragments are indexed by the corresponding paths, and the exact values are checked by a post-filter. Result: `id = 1`.
+
+## Find documents with value `false` or `null`
+
+To check for "value equals `false`" or "value equals `null`", use JsonPath inside `JSON_EXISTS`, not `JSON_VALUE(...) IS NULL`:
+
+
+```yql
+-- The value of the 'archived' field is false
+SELECT id
+FROM documents VIEW json_idx
+WHERE JSON_EXISTS(payload, '$.archived ? (@ == false)');
+
+-- The value of the 'value' field is null
+SELECT id
+FROM documents VIEW json_idx
+WHERE JSON_EXISTS(payload, '$.value ? (@ == null)');
+```
+
+
+The index search operation includes the path `$.archived` (or `$.value`) and the values `false` or `null`, respectively.
+
+{% note info %}
+
+The above conditions select documents where the specified attribute is explicitly set to false or null. Checking for the absence of an attribute cannot be done using a JSON index.
+
+{% endnote %}
+
+## See also
+
+* [JSON_EXISTS](../../dev/json-indexes.md#json-exists) — what is allowed in JsonPath expressions.
+* [JSON_VALUE](../../dev/json-indexes.md#json-value) — what is allowed when extracting values.
+* [JsonPath: methods](../../yql/reference/builtins/json.md#jsonpath) — list of methods (`type`, `size`, `keyvalue`, ...) and JsonPath predicates.
diff --git a/ydb/docs/en/core/recipes/json-search/toc_p.yaml b/ydb/docs/en/core/recipes/json-search/toc_p.yaml
new file mode 100644
index 00000000000..3b1d05ecbe1
--- /dev/null
+++ b/ydb/docs/en/core/recipes/json-search/toc_p.yaml
@@ -0,0 +1,9 @@
+items:
+- name: JSON index — quick start
+ href: json-index-quickstart.md
+- name: Directory with nested attributes
+ href: json-index-catalog.md
+- name: Parameterized queries and JsonPath variables
+ href: json-index-parameters.md
+- name: Checking field type and path existence
+ href: json-index-typecheck.md
diff --git a/ydb/docs/en/core/recipes/toc_p.yaml b/ydb/docs/en/core/recipes/toc_p.yaml
index e6abaf532e0..b4552b3e512 100644
--- a/ydb/docs/en/core/recipes/toc_p.yaml
+++ b/ydb/docs/en/core/recipes/toc_p.yaml
@@ -32,6 +32,11 @@ items:
include:
mode: link
path: fulltext-search/toc_p.yaml
+- name: JSON search
+ href: json-search/index.md
+ include:
+ mode: link
+ path: json-search/toc_p.yaml
- name: Local indexes
href: local-indexes/index.md
include:
diff --git a/ydb/docs/en/core/recipes/ydb-sdk/debug-jaeger.md b/ydb/docs/en/core/recipes/ydb-sdk/debug-jaeger.md
deleted file mode 100644
index bdb57ffd9f7..00000000000
--- a/ydb/docs/en/core/recipes/ydb-sdk/debug-jaeger.md
+++ /dev/null
@@ -1,170 +0,0 @@
-# Enabling tracing in Jaeger
-
-Below are code examples for enabling tracing in Jaeger in different {{ ydb-short-name }} SDK.
-
-{% list tabs %}
-
-- C++
-
- This functionality is not currently supported.
-
-- Go
-
- {% list tabs %}
-
- - Native SDK
-
- ```go
- package main
-
- import (
- "context"
- "time"
-
- "github.com/opentracing/opentracing-go"
- jaegerConfig "github.com/uber/jaeger-client-go/config"
-
- "github.com/ydb-platform/ydb-go-sdk/v3"
- "github.com/ydb-platform/ydb-go-sdk/v3/trace"
-
- tracing "github.com/ydb-platform/ydb-go-sdk-opentracing"
- )
-
- const (
- tracerURL = "localhost:5775"
- serviceName = "ydb-go-sdk"
- )
-
- func main() {
- tracer, closer, err := jaegerConfig.Configuration{
- ServiceName: serviceName,
- Sampler: &jaegerConfig.SamplerConfig{
- Type: "const",
- Param: 1,
- },
- Reporter: &jaegerConfig.ReporterConfig{
- LogSpans: true,
- BufferFlushInterval: 1 * time.Second,
- LocalAgentHostPort: tracerURL,
- },
- }.NewTracer()
- if err != nil {
- panic(err)
- }
-
- defer closer.Close()
-
- // set global tracer of this application
- opentracing.SetGlobalTracer(tracer)
-
- span, ctx := opentracing.StartSpanFromContext(context.Background(), "client")
- defer span.Finish()
-
- db, err := ydb.Open(ctx,
- os.Getenv("YDB_CONNECTION_STRING"),
- tracing.WithTraces(tracing.WithDetails(trace.DetailsAll)),
- )
- if err != nil {
- panic(err)
- }
- defer db.Close(ctx)
- ...
- }
- ```
-
- - database/sql
-
- ```go
- package main
-
- import (
- "context"
- "database/sql"
- "time"
-
- "github.com/opentracing/opentracing-go"
- jaegerConfig "github.com/uber/jaeger-client-go/config"
-
- "github.com/ydb-platform/ydb-go-sdk/v3"
- "github.com/ydb-platform/ydb-go-sdk/v3/trace"
-
- tracing "github.com/ydb-platform/ydb-go-sdk-opentracing"
- )
-
- const (
- tracerURL = "localhost:5775"
- serviceName = "ydb-go-sdk"
- )
-
- func main() {
- tracer, closer, err := jaegerConfig.Configuration{
- ServiceName: serviceName,
- Sampler: &jaegerConfig.SamplerConfig{
- Type: "const",
- Param: 1,
- },
- Reporter: &jaegerConfig.ReporterConfig{
- LogSpans: true,
- BufferFlushInterval: 1 * time.Second,
- LocalAgentHostPort: tracerURL,
- },
- }.NewTracer()
- if err != nil {
- panic(err)
- }
-
- defer closer.Close()
-
- // set global tracer of this application
- opentracing.SetGlobalTracer(tracer)
-
- span, ctx := opentracing.StartSpanFromContext(context.Background(), "client")
- defer span.Finish()
-
- nativeDriver, err := ydb.Open(ctx,
- os.Getenv("YDB_CONNECTION_STRING"),
- tracing.WithTraces(tracing.WithDetails(trace.DetailsAll)),
- )
- if err != nil {
- panic(err)
- }
- defer nativeDriver.Close(ctx)
-
- connector, err := ydb.Connector(nativeDriver)
- if err != nil {
- panic(err)
- }
-
- db := sql.OpenDB(connector)
- defer db.Close()
- ...
- }
- ```
-
- {% endlist %}
-
-- Java
-
- {% include [feature-not-supported](../../_includes/feature-not-supported.md) %}
-
-- Python
-
- {% include [feature-not-supported](../../_includes/feature-not-supported.md) %}
-
-- C#
-
- {% include [feature-not-supported](../../_includes/feature-not-supported.md) %}
-
-- JavaScript
-
- {% include [feature-not-supported](../../_includes/feature-not-supported.md) %}
-
-- Rust
-
- {% include [feature-not-supported](../../_includes/feature-not-supported.md) %}
-
-- PHP
-
- {% include [feature-not-supported](../../_includes/feature-not-supported.md) %}
-
-{% endlist %}
diff --git a/ydb/docs/en/core/recipes/ydb-sdk/debug-logs-otel.md b/ydb/docs/en/core/recipes/ydb-sdk/debug-logs-otel.md
deleted file mode 100644
index c90d1578c90..00000000000
--- a/ydb/docs/en/core/recipes/ydb-sdk/debug-logs-otel.md
+++ /dev/null
@@ -1,352 +0,0 @@
-# Export logs to OpenTelemetry
-
-Each {{ ydb-short-name }} SDK writes its internal logs (driver initialization, session pool, query execution, retries, etc.) through the standard logging facility of its language. Instead of outputting logs only to the console (see [Enable logging](debug-logs.md)), they can be redirected to the [OpenTelemetry](https://opentelemetry.io/) Logs SDK and exported via the standard OTLP protocol to a collector. The collector then forwards the records to the chosen backend for log storage and viewing.
-
-The principle is the same in all SDKs:
-
-1. Create an OpenTelemetry `LoggerProvider` with an OTLP log exporter and resource attribute `service.name`.
-2. Redirect the SDK logger to this provider via an adapter (log appender / bridge) that converts each SDK log record into an OTel log record. For the list of adapters and log support status in different languages, see the [OpenTelemetry logs documentation](https://opentelemetry.io/docs/concepts/signals/logs/).
-3. Run the workload — all internal SDK logs are now sent to the collector as OTLP log records.
-
-{% note info %}
-
-## Principles {#principles}
-
-The log collection method depends on the language and infrastructure — there is no single format:
-
-* **Programmatic bridges (log appender / bridge).** The application itself sends logs to the collector via OTLP directly from the process ("direct-to-Collector" workflow). This is the approach shown in the examples below.
-* **Collector agents.** The application writes logs to a file or `stdout`, and a separate agent (e.g., [filelog receiver](https://github.com/open-telemetry/opentelemetry-collector-contrib/tree/main/receiver/filelogreceiver) in [OpenTelemetry Collector](https://opentelemetry.io/docs/collector/)) reads, parses, and forwards them to the backend — without changing application code.
-
-{% endnote %}
-
-## Connecting to the SDK {#integration}
-
-{% list tabs %}
-
-- Go
-
- For the {{ ydb-short-name }} Go SDK, there is a ready-made adapter [ydb-go-sdk-otel](https://github.com/ydb-platform/ydb-go-sdk-otel) that converts SDK events into OpenTelemetry signals: traces (`WithTracer`), metrics (`WithMetrics`), and logs (`WithLogger`). The adapter does not configure exporters itself — you create `LoggerProvider` with an OTLP log exporter, get a logger from it, and pass it to the `ydbOtel.WithLogger` option when calling `ydb.Open`. Each internal SDK log record (driver initialization, session pool, query execution, retries, etc.) is then sent to the collector as an OTLP log record.
-
-
- ```bash
- go get github.com/ydb-platform/ydb-go-sdk-otel
- go get go.opentelemetry.io/otel/sdk/log
- go get go.opentelemetry.io/otel/exporters/otlp/otlplog/otlploggrpc
- ```
-
-
- ```go
- package main
-
- import (
- "context"
- "os"
-
- "github.com/ydb-platform/ydb-go-sdk/v3"
- "github.com/ydb-platform/ydb-go-sdk/v3/trace"
-
- "go.opentelemetry.io/otel/attribute"
- "go.opentelemetry.io/otel/exporters/otlp/otlplog/otlploggrpc"
- sdklog "go.opentelemetry.io/otel/sdk/log"
- "go.opentelemetry.io/otel/sdk/resource"
-
- ydbOtel "github.com/ydb-platform/ydb-go-sdk-otel"
- )
-
- func main() {
- ctx := context.Background()
-
- // 1. Configuring the OTel log provider with an OTLP exporter.
- exporter, err := otlploggrpc.New(ctx,
- otlploggrpc.WithEndpoint("localhost:4317"),
- otlploggrpc.WithInsecure(),
- )
- if err != nil {
- panic(err)
- }
- res, _ := resource.Merge(resource.Default(), resource.NewSchemaless(
- attribute.String("service.name", "ydb-go-sdk-otel-logs-sample"),
- ))
- lp := sdklog.NewLoggerProvider(
- sdklog.WithResource(res),
- sdklog.WithProcessor(sdklog.NewBatchProcessor(exporter)),
- )
- defer lp.Shutdown(ctx)
-
- // 2. Opening the YDB driver with the ydb-go-sdk-otel adapter.
- // WithLogger forwards SDK log events to OTel log records.
- logger := lp.Logger("ydb-go-sdk")
- db, err := ydb.Open(ctx,
- os.Getenv("YDB_CONNECTION_STRING"),
- ydbOtel.WithLogger(logger, ydbOtel.WithDetailer(trace.DetailsAll)),
- )
- if err != nil {
- panic(err)
- }
- defer db.Close(ctx)
- // ... use db ...
- }
- ```
-
-- Python
-
- {{ ydb-short-name }} Python SDK writes logs through the standard `logging` module (loggers named `ydb.*`). Use the OpenTelemetry built-in `LoggingHandler` to redirect these records to `LoggerProvider` and export via OTLP:
-
-
- ```bash
- pip install opentelemetry-sdk opentelemetry-exporter-otlp-proto-grpc
- ```
-
-
- ```python
- import logging
-
- import ydb
- from opentelemetry._logs import set_logger_provider
- from opentelemetry.exporter.otlp.proto.grpc._log_exporter import OTLPLogExporter
- from opentelemetry.sdk._logs import LoggerProvider, LoggingHandler
- from opentelemetry.sdk._logs.export import BatchLogRecordProcessor
- from opentelemetry.sdk.resources import Resource
-
- resource = Resource(attributes={"service.name": "ydb-otel-logs-example"})
- logger_provider = LoggerProvider(resource=resource)
- logger_provider.add_log_record_processor(
- BatchLogRecordProcessor(OTLPLogExporter(endpoint="http://localhost:4317"))
- )
- set_logger_provider(logger_provider)
-
- # Bridge stdlib logging -> OpenTelemetry. Binding a handler to the root logger
- # intercepts everything the SDK writes via logging.getLogger("ydb...").
- otel_handler = LoggingHandler(level=logging.NOTSET, logger_provider=logger_provider)
- logging.basicConfig(level=logging.INFO, handlers=[otel_handler])
- logging.getLogger("ydb").setLevel(logging.INFO)
-
- with ydb.Driver(endpoint="grpc://localhost:2136", database="/local") as driver:
- driver.wait(timeout=5)
- with ydb.QuerySessionPool(driver) as pool:
- pool.execute_with_retries("SELECT 1")
-
- logger_provider.shutdown()
- ```
-
-- C#
-
- {{ ydb-short-name }} C# SDK writes logs through the `ILoggerFactory` passed to it. Create a factory based on the OpenTelemetry logging provider with an OTLP exporter and pass it to the data source:
-
-
- ```bash
- dotnet add package OpenTelemetry
- dotnet add package OpenTelemetry.Exporter.OpenTelemetryProtocol
- ```
-
-
- ```csharp
- using Microsoft.Extensions.Logging;
- using OpenTelemetry.Logs;
- using OpenTelemetry.Resources;
- using Ydb.Sdk.Ado;
-
- var resourceBuilder = ResourceBuilder.CreateDefault()
- .AddService("ydb-sdk-otel-logs-sample");
-
- using var loggerFactory = LoggerFactory.Create(builder =>
- {
- builder.AddOpenTelemetry(options =>
- {
- options.SetResourceBuilder(resourceBuilder);
- options.AddOtlpExporter(o => o.Endpoint = new Uri("http://localhost:4317"));
- });
- });
-
- // Pass the factory to the SDK: each internal log (driver initialization, session pool,
- // query execution, retries, ...) is exported to the collector as an OTLP log record.
- await using var dataSource = new YdbDataSource(
- new YdbConnectionStringBuilder("Host=localhost;Port=2136;Database=/local")
- {
- LoggerFactory = loggerFactory
- });
-
- await using var connection = await dataSource.OpenConnectionAsync();
- await new YdbCommand("SELECT 1", connection).ExecuteNonQueryAsync();
- ```
-
-- Java
-
- {{ ydb-short-name }} Java SDK writes logs through `slf4j`. If logback is used as the `slf4j` implementation, connect the ready-made `opentelemetry-logback-appender-1.0` appender, which forwards each event to the OpenTelemetry Logs SDK:
-
-
- ```xml
- <dependency>
- <groupId>io.opentelemetry</groupId>
- <artifactId>opentelemetry-sdk</artifactId>
- <version>${otel.version}</version>
- </dependency>
- <dependency>
- <groupId>io.opentelemetry</groupId>
- <artifactId>opentelemetry-exporter-otlp</artifactId>
- <version>${otel.version}</version>
- </dependency>
- <dependency>
- <groupId>io.opentelemetry.instrumentation</groupId>
- <artifactId>opentelemetry-logback-appender-1.0</artifactId>
- <version>${otel.instrumentation.version}</version>
- </dependency>
- <dependency>
- <groupId>ch.qos.logback</groupId>
- <artifactId>logback-classic</artifactId>
- <version>${logback.version}</version>
- </dependency>
- ```
-
-
- Connect the appender in `logback.xml`:
-
-
- ```xml
- <configuration>
- <appender name="OTEL" class="io.opentelemetry.instrumentation.logback.appender.v1_0.OpenTelemetryAppender">
- </appender>
-
- <root level="INFO">
- <appender-ref ref="OTEL"/>
- </root>
- </configuration>
- ```
-
-
- Build an instance of the OpenTelemetry SDK with an OTLP log exporter and set it in the appender by calling `OpenTelemetryAppender.install(...)` before starting the workload:
-
-
- ```java
- import java.util.concurrent.TimeUnit;
-
- import io.opentelemetry.api.OpenTelemetry;
- import io.opentelemetry.api.common.Attributes;
- import io.opentelemetry.exporter.otlp.logs.OtlpGrpcLogRecordExporter;
- import io.opentelemetry.instrumentation.logback.appender.v1_0.OpenTelemetryAppender;
- import io.opentelemetry.sdk.OpenTelemetrySdk;
- import io.opentelemetry.sdk.logs.SdkLoggerProvider;
- import io.opentelemetry.sdk.logs.export.BatchLogRecordProcessor;
- import io.opentelemetry.sdk.resources.Resource;
- import tech.ydb.core.grpc.GrpcTransport;
- import tech.ydb.query.QueryClient;
-
- String serviceName = "ydb-java-sdk-otel-logs-sample";
- Resource resource = Resource.getDefault().merge(
- Resource.create(Attributes.builder().put("service.name", serviceName).build()));
-
- SdkLoggerProvider loggerProvider = SdkLoggerProvider.builder()
- .setResource(resource)
- .addLogRecordProcessor(BatchLogRecordProcessor.builder(
- OtlpGrpcLogRecordExporter.builder().setEndpoint("http://localhost:4317").build()
- ).build())
- .build();
-
- OpenTelemetry openTelemetry = OpenTelemetrySdk.builder()
- .setLoggerProvider(loggerProvider)
- .build();
-
- // Pass the OpenTelemetry SDK to the appender declared in logback.xml.
- OpenTelemetryAppender.install(openTelemetry);
-
- try (GrpcTransport transport = GrpcTransport.forConnectionString("grpc://localhost:2136/local").build();
- QueryClient queryClient = QueryClient.newClient(transport).build()) {
- // ... use queryClient ...
- } finally {
- loggerProvider.forceFlush().join(10, TimeUnit.SECONDS);
- loggerProvider.shutdown().join(10, TimeUnit.SECONDS);
- }
- ```
-
-- C++
-
- {{ ydb-short-name }} C++ SDK writes logs through `TLogBackend` from `util`. Implement a backend that turns each record into an OTel log record, and pass it to the driver via `TDriverConfig::SetLog`. See [Getting Started](https://opentelemetry.io/docs/languages/cpp/getting-started/) for building the OpenTelemetry C++ SDK.
-
-
- ```cpp
- #include <ydb-cpp-sdk/client/driver/driver.h>
- #include <opentelemetry/exporters/otlp/otlp_grpc_log_record_exporter_factory.h>
- #include <opentelemetry/exporters/otlp/otlp_grpc_log_record_exporter_options.h>
- #include <opentelemetry/logs/provider.h>
- #include <opentelemetry/sdk/logs/batch_log_record_processor_factory.h>
- #include <opentelemetry/sdk/logs/logger_provider_factory.h>
- #include <opentelemetry/sdk/resource/resource.h>
- #include <library/cpp/logger/backend.h>
- #include <library/cpp/logger/record.h>
- #include <memory>
-
- namespace otlp = opentelemetry::exporter::otlp;
- namespace logs_sdk = opentelemetry::sdk::logs;
- namespace logs_api = opentelemetry::logs;
- namespace resource = opentelemetry::sdk::resource;
-
- namespace {
-
- class TOtelLogBackend final : public TLogBackend {
- public:
- explicit TOtelLogBackend(opentelemetry::nostd::shared_ptr<logs_api::Logger> logger)
- : Logger_(std::move(logger)) {}
-
- void WriteData(const TLogRecord& rec) override {
- Logger_->EmitLogRecord(MapSeverity(rec.Priority),
- opentelemetry::nostd::string_view(rec.Data, rec.Len));
- }
-
- void ReopenLog() override {}
-
- private:
- static logs_api::Severity MapSeverity(ELogPriority p) {
- switch (p) {
- case TLOG_EMERG:
- case TLOG_ALERT:
- case TLOG_CRIT: return logs_api::Severity::kFatal;
- case TLOG_ERR: return logs_api::Severity::kError;
- case TLOG_WARNING: return logs_api::Severity::kWarn;
- case TLOG_NOTICE:
- case TLOG_INFO: return logs_api::Severity::kInfo;
- case TLOG_DEBUG: return logs_api::Severity::kDebug;
- default: return logs_api::Severity::kTrace;
- }
- }
-
- opentelemetry::nostd::shared_ptr<logs_api::Logger> Logger_;
- };
- } // namespace
-
- int main() {
- // 1. Configuring the OTel log provider with an OTLP exporter
- otlp::OtlpGrpcLogRecordExporterOptions exporterOpts;
- exporterOpts.endpoint = "localhost:4317";
- auto exporter = otlp::OtlpGrpcLogRecordExporterFactory::Create(exporterOpts);
- auto processor = logs_sdk::BatchLogRecordProcessorFactory::Create(std::move(exporter));
- auto res = resource::Resource::Create({{"service.name", "ydb-cpp-sdk-otel-logs-sample"}});
-
- std::shared_ptr<logs_api::LoggerProvider> provider(
- logs_sdk::LoggerProviderFactory::Create(std::move(processor), res));
- logs_api::Provider::SetLoggerProvider(provider);
-
- auto logger = provider->GetLogger("ydb-cpp-sdk");
-
- // 2. Pass the bridge to the YDB driver
- auto config = NYdb::TDriverConfig()
- .SetEndpoint("localhost:2136")
- .SetDatabase("/local")
- .SetLog(std::make_unique<TOtelLogBackend>(logger));
-
- NYdb::TDriver driver(config);
- // ... use driver ...
- driver.Stop(true);
- }
- ```
-
-- JavaScript
-
- {% include [feature-not-supported](../../_includes/feature-not-supported.md) %}
-
-- Rust
-
- {% include [feature-not-supported](../../_includes/feature-not-supported.md) %}
-
- Track progress or vote for Rust SDK support: [ydb-rs-sdk#268](https://github.com/ydb-platform/ydb-rs-sdk/issues/268)
-
-{% endlist %}
diff --git a/ydb/docs/en/core/recipes/ydb-sdk/debug-logs.md b/ydb/docs/en/core/recipes/ydb-sdk/debug-logs.md
deleted file mode 100644
index b164a73bb7f..00000000000
--- a/ydb/docs/en/core/recipes/ydb-sdk/debug-logs.md
+++ /dev/null
@@ -1,411 +0,0 @@
-# Enabling logging
-
-Below are examples of code that enables logging in different {{ ydb-short-name }} SDKs.
-
-{% list tabs %}
-
-- Go
-
- {% list tabs %}
-
- - Native SDK
-
- There are several ways to enable logs in an application that uses `ydb-go-sdk`:
-
- {% cut "Using the `YDB_LOG_SEVERITY_LEVEL` environment variable" %}
-
- This environment variable enables the built-in `ydb-go-sdk` logger (synchronous, non-block) and prints to the standard output stream.
- You can set the environment variable as follows:
-
- ```shell
- export YDB_LOG_SEVERITY_LEVEL=info
- ```
-
- (possible values: `trace`, `debug`, `info`, `warn`, `error`, `fatal`, and `quiet`, defaults to `quiet`).
-
- {% endcut %}
-
- {% cut "Enable a third-party logger `go.uber.org/zap`" %}
-
- ```go
- package main
-
- import (
- "context"
- "os"
-
- "go.uber.org/zap"
-
- ydbZap "github.com/ydb-platform/ydb-go-sdk-zap"
- "github.com/ydb-platform/ydb-go-sdk/v3"
- "github.com/ydb-platform/ydb-go-sdk/v3/trace"
- )
-
- func main() {
- ctx, cancel := context.WithCancel(context.Background())
- defer cancel()
- var log *zap.Logger // zap-logger with init out of this scope
- db, err := ydb.Open(ctx,
- os.Getenv("YDB_CONNECTION_STRING"),
- ydbZap.WithTraces(
- log,
- trace.DetailsAll,
- ),
- )
- if err != nil {
- panic(err)
- }
- defer db.Close(ctx)
- ...
- }
- ```
-
- {% endcut %}
-
- {% cut "Enable a third-party logger `github.com/rs/zerolog`" %}
-
- ```go
- package main
-
- import (
- "context"
- "os"
-
- "github.com/rs/zerolog"
-
- ydbZerolog "github.com/ydb-platform/ydb-go-sdk-zerolog"
- "github.com/ydb-platform/ydb-go-sdk/v3"
- "github.com/ydb-platform/ydb-go-sdk/v3/trace"
- )
-
- func main() {
- ctx, cancel := context.WithCancel(context.Background())
- defer cancel()
- var log zerolog.Logger // zap-logger with init out of this scope
- db, err := ydb.Open(ctx,
- os.Getenv("YDB_CONNECTION_STRING"),
- ydbZerolog.WithTraces(
- &log,
- trace.DetailsAll,
- ),
- )
- if err != nil {
- panic(err)
- }
- defer db.Close(ctx)
- ...
- }
- ```
-
- {% endcut %}
-
- {% include [overlay](_includes/debug-logs-go-appendix.md) %}
-
- {% cut "Enable a custom logger implementation `github.com/ydb-platform/ydb-go-sdk/v3/log.Logger`" %}
-
- ```go
- package main
-
- import (
- "context"
- "os"
-
- "github.com/ydb-platform/ydb-go-sdk/v3"
- "github.com/ydb-platform/ydb-go-sdk/v3/log"
- "github.com/ydb-platform/ydb-go-sdk/v3/trace"
- )
-
- func main() {
- ctx, cancel := context.WithCancel(context.Background())
- defer cancel()
- var logger log.Logger // logger implementation with init out of this scope
- db, err := ydb.Open(ctx,
- os.Getenv("YDB_CONNECTION_STRING"),
- ydb.WithLogger(
- logger,
- trace.DetailsAll,
- ),
- )
- if err != nil {
- panic(err)
- }
- defer db.Close(ctx)
- ...
- }
- ```
-
- {% endcut %}
-
- {% cut "Implement your own logging package" %}
-
- You can implement your own logging package based on the driver events in the `github.com/ydb-platform/ydb-go-sdk/v3/trace` tracing package. The `github.com/ydb-platform/ydb-go-sdk/v3/trace` tracing package describes all logged driver events.
-
- {% endcut %}
-
- {% cut "Iterate over server errors with `IterateByIssues`" %}
-
- When using {{ ydb-short-name }} through the Go SDK, you can enable logging of requests and responses and also obtain detailed information about server errors (issues) — extra messages YDB returns in the response when operations fail. To iterate over issues in the server response, use [IterateByIssues](https://pkg.go.dev/github.com/ydb-platform/ydb-go-sdk/v3#IterateByIssues).
-
- {% endcut %}
-
- - database/sql
-
- There are several ways to enable logs in an application that uses `ydb-go-sdk`:
-
- {% cut "Using the `YDB_LOG_SEVERITY_LEVEL` environment variable" %}
-
- This environment variable enables the built-in `ydb-go-sdk` logger (synchronous, non-block) and prints to the standard output stream.
- You can set the environment variable as follows:
-
- ```shell
- export YDB_LOG_SEVERITY_LEVEL=info
- ```
-
- (possible values: `trace`, `debug`, `info`, `warn`, `error`, `fatal`, and `quiet`, defaults to `quiet`).
-
- {% endcut %}
-
- {% cut "Enable a third-party logger `go.uber.org/zap`" %}
-
- ```go
- package main
-
- import (
- "context"
- "database/sql"
- "os"
-
- "go.uber.org/zap"
-
- ydbZap "github.com/ydb-platform/ydb-go-sdk-zap"
- "github.com/ydb-platform/ydb-go-sdk/v3"
- "github.com/ydb-platform/ydb-go-sdk/v3/trace"
- )
-
- func main() {
- ctx, cancel := context.WithCancel(context.Background())
- defer cancel()
- var log *zap.Logger // zap-logger with init out of this scope
- nativeDriver, err := ydb.Open(ctx,
- os.Getenv("YDB_CONNECTION_STRING"),
- ydbZap.WithTraces(
- log,
- trace.DetailsAll,
- ),
- )
- if err != nil {
- panic(err)
- }
- defer nativeDriver.Close(ctx)
-
- connector, err := ydb.Connector(nativeDriver)
- if err != nil {
- panic(err)
- }
- defer connector.Close()
-
- db := sql.OpenDB(connector)
- defer db.Close()
- ...
- }
- ```
-
- {% endcut %}
-
- {% cut "Enable a third-party logger `github.com/rs/zerolog`" %}
-
- ```go
- package main
-
- import (
- "context"
- "database/sql"
- "os"
-
- "github.com/rs/zerolog"
-
- ydbZerolog "github.com/ydb-platform/ydb-go-sdk-zerolog"
- "github.com/ydb-platform/ydb-go-sdk/v3"
- "github.com/ydb-platform/ydb-go-sdk/v3/trace"
- )
-
- func main() {
- ctx, cancel := context.WithCancel(context.Background())
- defer cancel()
- var log zerolog.Logger // zap-logger with init out of this scope
- nativeDriver, err := ydb.Open(ctx,
- os.Getenv("YDB_CONNECTION_STRING"),
- ydbZerolog.WithTraces(
- &log,
- trace.DetailsAll,
- ),
- )
- if err != nil {
- panic(err)
- }
- defer nativeDriver.Close(ctx)
-
- connector, err := ydb.Connector(nativeDriver)
- if err != nil {
- panic(err)
- }
- defer connector.Close()
-
- db := sql.OpenDB(connector)
- defer db.Close()
- ...
- }
- ```
-
- {% endcut %}
-
- {% include [overlay](_includes/debug-logs-go-sql-appendix.md) %}
-
- {% cut "Enable a custom logger implementation `github.com/ydb-platform/ydb-go-sdk/v3/log.Logger`" %}
-
- ```go
- package main
-
- import (
- "context"
- "database/sql"
- "os"
-
- "github.com/ydb-platform/ydb-go-sdk/v3"
- "github.com/ydb-platform/ydb-go-sdk/v3/log"
- "github.com/ydb-platform/ydb-go-sdk/v3/trace"
- )
-
- func main() {
- ctx, cancel := context.WithCancel(context.Background())
- defer cancel()
- var logger log.Logger // logger implementation with init out of this scope
- nativeDriver, err := ydb.Open(ctx,
- os.Getenv("YDB_CONNECTION_STRING"),
- ydb.WithLogger(
- logger,
- trace.DetailsAll,
- ),
- )
- if err != nil {
- panic(err)
- }
- defer nativeDriver.Close(ctx)
-
- connector, err := ydb.Connector(nativeDriver)
- if err != nil {
- panic(err)
- }
- defer connector.Close()
-
- db := sql.OpenDB(connector)
- defer db.Close()
- ...
- }
- ```
-
- {% endcut %}
-
- {% cut "Implement your own logging package" %}
-
- You can implement your own logging package based on the driver events in the `github.com/ydb-platform/ydb-go-sdk/v3/trace` tracing package. The `github.com/ydb-platform/ydb-go-sdk/v3/trace` tracing package describes all logged driver events.
-
- {% endcut %}
-
- {% endlist %}
-
-
-- Java
-
- {% list tabs %}
-
- - Native SDK
-
- For logging purposes, the {{ ydb-short-name }} Java SDK uses the slf4j library, which supports multiple logging levels (`error`, `warn`, `info`, `debug`, `trace`) for one or many loggers. The current implementation supports the following loggers:
-
- * The `tech.ydb.core.grpc` logger provides information about the internal gRPC implementation
- * The `debug` level logs all gRPC operations; use it only for debugging
- * The `info` level is recommended by default
- * On the `debug` level, the `tech.ydb.table.impl` logger lets you track the internal state of the {{ ydb-short-name }} driver, including the session pool
- * On the `debug` level, the `tech.ydb.table.SessionRetryContext` logger reports the number of retries, query results, per-retry duration, and total operation time
- * On the `debug` level, the `tech.ydb.table.Session` logger provides the query text, response status, and execution time for session operations
-
- Enabling and configuring the Java SDK loggers depends on the `slf4j-api` implementation you use.
- Here is an example `log4j2` configuration for the `log4j-slf4j-impl` library:
-
- ```xml
- <Configuration status="WARN">
- <Appenders>
- <Console name="Console" target="SYSTEM_OUT">
- <PatternLayout pattern="%d{HH:mm:ss.SSS} [%t] %-5level %logger{36} - %msg%n"/>
- </Console>
- </Appenders>
-
- <Loggers>
- <Logger name="io.netty" level="warn" additivity="false">
- <AppenderRef ref="Console"/>
- </Logger>
- <Logger name="io.grpc.netty" level="warn" additivity="false">
- <AppenderRef ref="Console"/>
- </Logger>
- <Logger name="tech.ydb.core.grpc" level="info" additivity="false">
- <AppenderRef ref="Console"/>
- </Logger>
- <Logger name="tech.ydb.table.impl" level="info" additivity="false">
- <AppenderRef ref="Console"/>
- </Logger>
- <Logger name="tech.ydb.table.SessionRetryContext" level="debug" additivity="false">
- <AppenderRef ref="Console"/>
- </Logger>
- <Logger name="tech.ydb.table.Session" level="debug" additivity="false">
- <AppenderRef ref="Console"/>
- </Logger>
-
- <Root level="debug" >
- <AppenderRef ref="Console"/>
- </Root>
- </Loggers>
- </Configuration>
- ```
-
- - JDBC
-
- The JDBC driver uses the same logging stack via `slf4j`; configure the `tech.ydb.*` loggers the same way as in the native SDK.
-
- The same debug logs are available in other stacks built on JDBC (Spring Boot, ORMs, connection pools, and so on): they reach {{ ydb-short-name }} through this driver, so it is enough to attach the same `slf4j` / log4j2 / logback configuration in the application.
-
- {% endlist %}
-
-- PHP
-
- For logging purposes, you need to use a class, that implements `\Psr\Log\LoggerInterface`.
- `ydb-php-sdk` has build-in loggers in `YdbPlatform\Ydb\Logger` namespace:
-
- * `NullLogger` - default logger, which writes nothing
- * `SimpleStdLogger($level)` - logger, which writes to logs in stderr.
-
- Usage example:
-
- ```php
- $config = [
- 'logger' => new \YdbPlatform\Ydb\Logger\SimpleStdLogger(\YdbPlatform\Ydb\Logger\SimpleStdLogger::INFO)
- ]
- $ydb = new \YdbPlatform\Ydb\Ydb($config);
- ```
-
-- Python
-
- The Python SDK uses the standard `logging` library. To enable a specific logging level:
-
- ```python
- import logging
-
- logging.getLogger('ydb').setLevel(logging.DEBUG)
- ```
-
-- JavaScript
-
- The SDK uses the [debug](https://www.npmjs.com/package/debug) library for logging.
- To enable logs, set the `DEBUG` environment variable to filter SDK events, for example `DEBUG=ydbjs:*`.
-
-{% endlist %}
diff --git a/ydb/docs/en/core/recipes/ydb-sdk/debug-otel-metrics.md b/ydb/docs/en/core/recipes/ydb-sdk/debug-otel-metrics.md
deleted file mode 100644
index 791ac9a4feb..00000000000
--- a/ydb/docs/en/core/recipes/ydb-sdk/debug-otel-metrics.md
+++ /dev/null
@@ -1,321 +0,0 @@
-# Metrics with OpenTelemetry
-
-{{ ydb-short-name }} SDK instruments Query Service operations with [OpenTelemetry](https://opentelemetry.io/) metrics, allowing you to observe the client state — the duration and number of operations, the state of the session pool — from the application code to gRPC calls to YDB. The metrics are exported via the standard OTLP protocol and are compatible with Prometheus, Grafana, VictoriaMetrics, and any other backend that supports OpenTelemetry.
-
-## List of metrics {#metrics-list}
-
-### Operation metrics {#operation-metrics}
-
-| Name | Type | Unit | Description |
-| --- | --- | --- | --- |
-| `ydb.client.operation.duration` | Histogram | `s` | Duration of a single attempt of a client operation (`ExecuteQuery`, `Commit`, `Rollback`, `CreateSession`). |
-| `ydb.client.operation.failed` | Counter | `{operation}` | Number of unsuccessful client operations. |
-
-### Session pool metrics {#session-pool-metrics}
-
-| Name | Type | Unit | Description |
-| --- | --- | --- | --- |
-| `ydb.query.session.create_time` | Histogram | `s` | Duration of creating a new session. |
-| `ydb.query.session.pending_requests` | Counter | `{request}` | Monotonic counter of session acquisition queries that have entered the wait queue since the pool was created. |
-| `ydb.query.session.timeouts` | Counter | `{timeout}` | Monotonic counter of timeouts when waiting for a free session, since the pool creation. |
-| `ydb.query.session.count` | Gauge | `{session}` | Current number of sessions in the pool, broken down by state. |
-| `ydb.query.session.min` | Gauge | `{session}` | Configured minimum size of the session pool. |
-| `ydb.query.session.max` | Gauge | `{session}` | Configured maximum size of the session pool. |
-
-### Retry metrics {#retry-metrics}
-
-| Name | Type | Unit | Description |
-| --- | --- | --- | --- |
-| `ydb.client.retry.duration` | Histogram | `s` | Total client-visible duration of a logical operation executed through a retry policy, including all attempts and backoff delays. |
-| `ydb.client.retry.attempts` | Histogram | `{attempt}` | Distribution of the number of attempts per logical operation. The value `1` means success on the first attempt. |
-
-### Additional metrics (JavaScript) {#js-metrics}
-
-In the [JavaScript SDK](https://github.com/ydb-platform/ydb-js-sdk), the `@ydbjs/telemetry` package uses its own names and semantics for retry metrics — different from `ydb.client.retry.*` above:
-
-| Name | Type | Unit | Description |
-| --- | --- | --- | --- |
-| `ydb.retry.attempts` | Counter | `{attempt}` | Monotonic counter of retry attempts with outcome tag `ydb.retry.outcome`. Unlike histogram `ydb.client.retry.attempts`, it does not show the distribution of the number of attempts per query. |
-| `ydb.retry.duration` | Histogram | `s` | Total retry cycle duration, including backoff (equivalent of `ydb.client.retry.duration`). |
-
-## Attributes {#attributes}
-
-| Name | Applies to | Value |
-| --- | --- | --- |
-| `database` | `ydb.client.operation.duration`, `ydb.client.operation.failed` | Database name {{ ydb-short-name }}. |
-| `endpoint` | `ydb.client.operation.duration`, `ydb.client.operation.failed` | Discovery-endpoint in the `host:port` format. |
-| `operation.name` | `ydb.client.operation.duration`, `ydb.client.operation.failed`, `ydb.client.retry.duration`, `ydb.client.retry.attempts` | Client operation name: `ExecuteQuery`, `Commit`, `Rollback`, `CreateSession`. |
-| `status_code` | `ydb.client.operation.failed` | Status code {{ ydb-short-name }} (for example, `BAD_REQUEST`, `SCHEME_ERROR`). |
-| `ydb.query.session.pool.name` | All `ydb.query.session.*` metrics | Session pool name. By default, it is formed as `<endpoint>/<database>`; configured through the API of a specific SDK. |
-| `ydb.query.session.state` | `ydb.query.session.count` | Session state: `idle` or `used`. |
-| `ydb.retry.outcome` | `ydb.retry.attempts` (JavaScript) | Outcome of a retry attempt: for example, `success`, `retried`. |
-
-## Connecting to the SDK {#integration}
-
-{% list tabs %}
-
-- Go
-
- Install the OpenTelemetry adapter for the {{ ydb-short-name }} Go SDK:
-
-
- ```bash
- go get github.com/ydb-platform/ydb-go-sdk-otel
- ```
-
-
- Configure `MeterProvider`, obtain `Meter` from it, and pass it to the `ydbOtel.WithMetrics` adapter, which connects to the driver via the `ydb.Open` option. The granularity of collected metrics can be configured via `ydbOtel.WithDetailer`:
-
-
- ```go
- package main
-
- import (
- "context"
- "os"
-
- "go.opentelemetry.io/otel"
- "go.opentelemetry.io/otel/attribute"
- "go.opentelemetry.io/otel/exporters/otlp/otlpmetric/otlpmetricgrpc"
- "go.opentelemetry.io/otel/sdk/metric"
- "go.opentelemetry.io/otel/sdk/resource"
-
- "github.com/ydb-platform/ydb-go-sdk/v3"
- "github.com/ydb-platform/ydb-go-sdk/v3/trace"
- ydbOtel "github.com/ydb-platform/ydb-go-sdk-otel"
- )
-
- func main() {
- ctx := context.Background()
-
- exporter, err := otlpmetricgrpc.New(ctx,
- otlpmetricgrpc.WithEndpoint("localhost:4317"),
- otlpmetricgrpc.WithInsecure(),
- )
- if err != nil {
- panic(err)
- }
- res, _ := resource.Merge(resource.Default(), resource.NewSchemaless(
- attribute.String("service.name", "my-service"),
- ))
- mp := metric.NewMeterProvider(
- metric.WithReader(metric.NewPeriodicReader(exporter)),
- metric.WithResource(res),
- )
- defer mp.Shutdown(ctx)
- otel.SetMeterProvider(mp)
-
- db, err := ydb.Open(ctx,
- os.Getenv("YDB_CONNECTION_STRING"),
- ydbOtel.WithMetrics(
- mp.Meter("ydb-go-sdk"),
- ydbOtel.WithDetailer(trace.DetailsAll),
- ),
- )
- if err != nil {
- panic(err)
- }
- defer db.Close(ctx)
- }
- ```
-
-- Python
-
- {% include [feature-not-supported](../../_includes/feature-not-supported.md) %}
-
-- C#
-
- Add the NuGet package:
-
-
- ```bash
- dotnet add package Ydb.Sdk.OpenTelemetry
- ```
-
-
- Register the {{ ydb-short-name }} instrumentation in the OpenTelemetry metrics pipeline:
-
-
- ```csharp
- services.AddOpenTelemetry()
- .WithMetrics(builder => builder
- .AddYdb()
- .AddOtlpExporter());
- ```
-
-
- Or using standalone `MeterProvider`:
-
-
- ```csharp
- using var meterProvider = Sdk.CreateMeterProviderBuilder()
- .SetResourceBuilder(ResourceBuilder.CreateDefault().AddService("my-service"))
- .AddYdb()
- .AddOtlpExporter()
- .Build();
- ```
-
-
- The session pool name is set by the `PoolName=` parameter in the `YdbDataSource` connection string. A full example with load and Grafana is in the SDK repository ( [Ydb.Sdk.AdoNet.OpenTelemetry/Metrics](https://github.com/ydb-platform/ydb-dotnet-sdk/blob/main/examples/Ydb.Sdk.AdoNet.OpenTelemetry/Metrics/Program.cs)).
-
-- Java
-
- {% note info %}
-
- Currently, only [session pool metrics](#session-pool-metrics) are supported in the Java SDK.
-
- {% endnote %}
-
- Pass your `OpenTelemetry` to the SDK via the `OpenTelemetryMeter` adapter and the `QueryClient.Builder#withMeter` method:
-
-
- ```java
- import io.opentelemetry.api.OpenTelemetry;
- import io.opentelemetry.exporter.otlp.metrics.OtlpGrpcMetricExporter;
- import io.opentelemetry.sdk.OpenTelemetrySdk;
- import io.opentelemetry.sdk.metrics.SdkMeterProvider;
- import io.opentelemetry.sdk.metrics.export.PeriodicMetricReader;
-
- import tech.ydb.core.grpc.GrpcTransport;
- import tech.ydb.core.metrics.OpenTelemetryMeter;
- import tech.ydb.query.QueryClient;
-
- SdkMeterProvider meterProvider = SdkMeterProvider.builder()
- .registerMetricReader(PeriodicMetricReader
- .builder(OtlpGrpcMetricExporter.getDefault())
- .build())
- .build();
-
- OpenTelemetry openTelemetry = OpenTelemetrySdk.builder()
- .setMeterProvider(meterProvider)
- .build();
-
- try (GrpcTransport transport = GrpcTransport
- .forConnectionString(System.getenv("YDB_CONNECTION_STRING"))
- .build();
- QueryClient queryClient = QueryClient.newClient(transport)
- .withMeter(OpenTelemetryMeter.fromOpenTelemetry(openTelemetry))
- .sessionPoolName("my-app")
- .build()) {
- // Use queryClient
- }
- meterProvider.close();
- ```
-
-
- If a global `GlobalOpenTelemetry` is already configured, you can use `OpenTelemetryMeter.createGlobal()`. `TableClient` also supports `withMeter(...)`.
-
-- C++
-
- Include the OpenTelemetry metrics header from {{ ydb-short-name }} C++ SDK and register `MetricRegistry` in `TDriverConfig`:
-
-
- ```cpp
- #include <ydb-cpp-sdk/client/driver/driver.h>
- #include <ydb-cpp-sdk/open_telemetry/metrics.h>
-
- #include <opentelemetry/exporters/otlp/otlp_http_metric_exporter_factory.h>
- #include <opentelemetry/exporters/otlp/otlp_http_metric_exporter_options.h>
- #include <opentelemetry/sdk/metrics/meter_provider.h>
- #include <opentelemetry/sdk/metrics/view/view_registry.h>
- #include <opentelemetry/sdk/metrics/export/periodic_exporting_metric_reader_factory.h>
- #include <opentelemetry/sdk/resource/resource.h>
- #include <opentelemetry/metrics/provider.h>
-
- namespace sdkmetrics = opentelemetry::sdk::metrics;
- namespace otlp = opentelemetry::exporter::otlp;
- namespace resource = opentelemetry::sdk::resource;
- using namespace NYdb;
-
- // 1. Initialize the OTel metrics provider
- otlp::OtlpHttpMetricExporterOptions opts;
- opts.url = "http://localhost:4318/v1/metrics";
-
- auto exporter = otlp::OtlpHttpMetricExporterFactory::Create(opts);
-
- sdkmetrics::PeriodicExportingMetricReaderOptions readerOpts;
- readerOpts.export_interval_millis = std::chrono::milliseconds(5000);
- auto reader = sdkmetrics::PeriodicExportingMetricReaderFactory::Create(
- std::move(exporter), readerOpts);
-
- auto res = resource::Resource::Create({{"service.name", "my-service"}});
- auto rawProvider = std::make_shared<sdkmetrics::MeterProvider>(
- std::unique_ptr<sdkmetrics::ViewRegistry>(new sdkmetrics::ViewRegistry()), res);
- rawProvider->AddMetricReader(std::move(reader));
-
- std::shared_ptr<opentelemetry::metrics::MeterProvider> meterProvider = rawProvider;
- opentelemetry::metrics::Provider::SetMeterProvider(meterProvider);
-
- // 2. Wrap in the YDB metrics registrar
- auto ydbMetricRegistry = NMetrics::CreateOtelMetricRegistry(meterProvider);
-
- // 3. Create a YDB driver with metrics enabled
- auto driverConfig = TDriverConfig()
- .SetEndpoint("localhost:2136")
- .SetDatabase("/local")
- .SetMetricRegistry(ydbMetricRegistry);
-
- TDriver driver(driverConfig);
- ```
-
-
- Metrics and tracing can be connected together by registering both `MetricRegistry` and `TraceProvider` in `TDriverConfig` simultaneously.
-
-- JavaScript
-
- Install the {{ ydb-short-name }} JavaScript SDK telemetry package and the OpenTelemetry SDK:
-
-
- ```bash
- npm install @ydbjs/telemetry @opentelemetry/sdk-node @opentelemetry/exporter-metrics-otlp-http
- ```
-
-
- Initialize `NodeSDK` before creating `Driver` and register the `@ydbjs/telemetry` instrumentation:
-
-
- ```ts
- import { NodeSDK } from '@opentelemetry/sdk-node'
- import { OTLPMetricExporter } from '@opentelemetry/exporter-metrics-otlp-http'
- import { PeriodicExportingMetricReader } from '@opentelemetry/sdk-metrics'
- import { Driver } from '@ydbjs/core'
- import { query } from '@ydbjs/query'
- import { register } from '@ydbjs/telemetry'
-
- const sdk = new NodeSDK({
- serviceName: 'my-service',
- metricReader: new PeriodicExportingMetricReader({
- exporter: new OTLPMetricExporter({
- url: 'http://localhost:4318/v1/metrics',
- }),
- }),
- })
- sdk.start()
-
- // Must be called BEFORE creating the Driver: the package subscribes to
- // diagnostics_channel SDK events and converts them into OTel metrics.
- const instrumentation = register()
-
- const driver = new Driver(process.env.YDB_CONNECTION_STRING)
- await driver.ready()
-
- const sql = query(driver)
- await sql`SELECT 1`
-
- instrumentation.disable()
- await driver.close()
- await sdk.shutdown()
- ```
-
-
- The `@ydbjs/telemetry` package subscribes to `node:diagnostics_channel` events from `@ydbjs/core`, `@ydbjs/query`, `@ydbjs/auth`, and `@ydbjs/retry` and publishes metrics from the [list above](#metrics-list) and [additional JavaScript SDK metrics](#js-metrics). Operation duration is exported as `db.client.operation.duration` (OpenTelemetry semantic conventions), not `ydb.client.operation.duration`. The current number of requests waiting for a session is gauge `ydb.query.session.acquire.pending`, not counter `ydb.query.session.pending_requests`. For more information, see the [JavaScript SDK repository](https://github.com/ydb-platform/ydb-js-sdk).
-
-- Rust
-
- {% include [feature-not-supported](../../_includes/feature-not-supported.md) %}
-
-- PHP
-
- {% include [feature-not-supported](../../_includes/feature-not-supported.md) %}
-
-{% endlist %}
diff --git a/ydb/docs/en/core/recipes/ydb-sdk/debug-otel-tracing.md b/ydb/docs/en/core/recipes/ydb-sdk/debug-otel-tracing.md
deleted file mode 100644
index 4bce69ac974..00000000000
--- a/ydb/docs/en/core/recipes/ydb-sdk/debug-otel-tracing.md
+++ /dev/null
@@ -1,371 +0,0 @@
-# Tracing with OpenTelemetry
-
-{{ ydb-short-name }} SDK instruments Query Service operations with [OpenTelemetry](https://opentelemetry.io/) spans, providing distributed tracing from application code to each gRPC call to YDB. The spans are exported via the standard OTLP protocol and are compatible with Jaeger, Grafana Tempo, Zipkin, and any other backend that supports OpenTelemetry.
-
-## Created spans {#spans}
-
-Currently, spans are supported for Query Service operations. Support for topics and other services is planned for future SDK versions.
-
-The following spans are created:
-
-| Span | Type | Description |
-| --- | --- | --- |
-| `ydb.RunWithRetry` | `Internal` | Covers the entire retry cycle for a single operation |
-| `ydb.Try` | `Internal` | One span per attempt, including the first; child RPC spans are attached to it |
-| `ydb.CreateSession` | `Client` | Session creation via gRPC `CreateSession` and `AttachStream` |
-| `ydb.ExecuteQuery` | `Client` | Execution of a single YQL query |
-| `ydb.BeginTransaction` | `Client` | Explicit call to start a transaction |
-| `ydb.Commit` | `Client` | Transaction commit |
-| `ydb.Rollback` | `Client` | Transaction rollback |
-| `ydb.Driver.Initialize` | `Internal` | Initial driver initialization: cluster discovery and authentication |
-
-A typical span tree for a transactional operation with retry looks as follows:
-
-
-```text
-ydb.RunWithRetry (Internal)
-├─ ydb.Try (Internal) ← 1st attempt: ERROR
-│ ├─ ydb.ExecuteQuery (Client)
-│ ├─ ydb.ExecuteQuery (Client)
-│ └─ ydb.Commit (Client) ← ERROR: Transaction Lock Invalidated
-└─ ydb.Try (Internal) ← 2nd attempt: SUCCESS, ydb.retry.backoff_ms=50
- ├─ ydb.ExecuteQuery (Client)
- ├─ ydb.ExecuteQuery (Client)
- └─ ydb.Commit (Client)
-```
-
-
-`ydb.RunWithRetry` is the parent span for the entire retry operation. For each execution attempt, a separate child span `ydb.Try` is created: the first `ydb.Try` corresponds to the first attempt, the second to the first **retry** attempt, and so on. RPC spans of a specific attempt, for example `ydb.ExecuteQuery` and `ydb.Commit`, are created inside the corresponding `ydb.Try`.
-
-{% note info %}
-
-If an attempt fails, its span `ydb.Try` ends with an error status. On a retry, a new `ydb.Try` is created; starting from the second attempt, the attribute `ydb.retry.backoff_ms` is set on it — the wait time before this attempt in milliseconds. This wait is included in the duration of the next span `ydb.Try`: the span starts before the backoff pause, then after the pause the RPC calls of this attempt are executed.
-
-{% endnote %}
-
-## Span attributes {#attributes}
-
-The SDK uses both standard OpenTelemetry semantic convention attributes and YDB-specific extensions.
-
-### Standard OpenTelemetry attributes
-
-The following attributes belong to the stable [OpenTelemetry semantic conventions](https://opentelemetry.io/docs/specs/semconv/) and can be processed by tracing backends as standard:
-
-| Attribute | Where set | Description |
-| --- | --- | --- |
-| `db.system.name` | RPC spans | Always `"ydb"` |
-| `db.namespace` | RPC spans | YDB database path |
-| `server.address` | RPC spans | Primary host from the connection string |
-| `server.port` | RPC spans | Primary port from the connection string |
-| `network.peer.address` | RPC spans | Actual gRPC endpoint used for the call |
-| `network.peer.port` | RPC spans | Actual port of the gRPC endpoint used for the call |
-| `error.type` | Spans that ended with an error | Error type. For example: `"transport_error"`, `"ydb_error"`, or the full exception class name |
-| `db.response.status_code` | RPC spans when `YdbException` | Text name of the YDB status from the error, for example `ABORTED`, `UNAVAILABLE`, `OVERLOADED` |
-
-### YDB-specific attributes
-
-The following attributes are YDB extensions on top of the standard semantic conventions:
-
-| Attribute | Where set | Description |
-| --- | --- | --- |
-| `ydb.node.id` | RPC spans | ID of the YDB node that processed the request |
-| `ydb.node.dc` | RPC spans | Datacenter of the YDB node that processed the request |
-| `ydb.retry.backoff_ms` | Spans `ydb.Try`, starting from the second attempt | Wait time before a retry attempt in milliseconds |
-
-## W3C trace context {#w3c}
-
-The SDK automatically propagates the W3C `traceparent` header in every outgoing gRPC call. This allows the YDB server to trace internal operations within the same trace — without additional configuration. For more details on server-side tracing, see the section [Passing an external trace-id to {{ ydb-short-name }}](../../reference/observability/tracing/external-traces.md).
-
-## Connecting to the SDK {#integration}
-
-{% list tabs %}
-
-- Go
-
- Install the OpenTelemetry adapter for the {{ ydb-short-name }} Go SDK:
-
-
- ```bash
- go get github.com/ydb-platform/ydb-go-sdk-otel
- ```
-
-
- Configure `TracerProvider` and pass the adapter to `ydb.Open`:
-
-
- ```go
- package main
-
- import (
- "context"
- "os"
-
- "go.opentelemetry.io/otel"
- "go.opentelemetry.io/otel/attribute"
- "go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracegrpc"
- "go.opentelemetry.io/otel/sdk/resource"
- sdktrace "go.opentelemetry.io/otel/sdk/trace"
-
- "github.com/ydb-platform/ydb-go-sdk/v3"
- ydbOtel "github.com/ydb-platform/ydb-go-sdk-otel"
- )
-
- func main() {
- ctx := context.Background()
-
- exporter, err := otlptracegrpc.New(ctx,
- otlptracegrpc.WithEndpoint("localhost:4317"),
- otlptracegrpc.WithInsecure(),
- )
- if err != nil {
- panic(err)
- }
- res, _ := resource.Merge(resource.Default(), resource.NewSchemaless(
- attribute.String("service.name", "my-service"),
- ))
- tp := sdktrace.NewTracerProvider(
- sdktrace.WithBatcher(exporter),
- sdktrace.WithResource(res),
- )
- defer tp.Shutdown(ctx)
- otel.SetTracerProvider(tp)
-
- db, err := ydb.Open(ctx,
- os.Getenv("YDB_CONNECTION_STRING"),
- ydbOtel.WithTraces(
- ydbOtel.WithTracer(tp.Tracer("ydb-go-sdk")),
- ),
- )
- if err != nil {
- panic(err)
- }
- defer db.Close(ctx)
- }
- ```
-
-- Python
-
- Install the additional dependencies `opentelemetry` and the OTLP exporter:
-
-
- ```bash
- pip install ydb[opentelemetry]
- pip install opentelemetry-exporter-otlp-proto-grpc
- ```
-
-
- Call `enable_tracing()` after configuring the global `TracerProvider`:
-
-
- ```python
- from opentelemetry import trace
- from opentelemetry.sdk.trace import TracerProvider
- from opentelemetry.sdk.trace.export import BatchSpanProcessor
- from opentelemetry.exporter.otlp.proto.grpc.trace_exporter import OTLPSpanExporter
- from opentelemetry.sdk.resources import Resource
-
- import ydb
- from ydb.opentelemetry import enable_tracing
-
- resource = Resource(attributes={"service.name": "my-service"})
- provider = TracerProvider(resource=resource)
- provider.add_span_processor(
- BatchSpanProcessor(OTLPSpanExporter(endpoint="http://localhost:4317"))
- )
- trace.set_tracer_provider(provider)
-
- enable_tracing()
-
- with ydb.Driver(endpoint="grpc://localhost:2136", database="/local") as driver:
- driver.wait(timeout=5)
- with ydb.QuerySessionPool(driver) as pool:
- pool.execute_with_retries("SELECT 1")
-
- provider.shutdown()
- ```
-
-- C#
-
- Add the NuGet package:
-
-
- ```bash
- dotnet add package Ydb.Sdk.OpenTelemetry
- ```
-
-
- Register the {{ ydb-short-name }} instrumentation when configuring OpenTelemetry in your service:
-
-
- ```csharp
- services.AddOpenTelemetry()
- .WithTracing(builder => builder
- .AddYdb()
- .AddOtlpExporter());
- ```
-
-- Java
-
- Add the YDB SDK and OpenTelemetry dependencies (example for Maven):
-
-
- ```xml
- <dependency>
- <groupId>tech.ydb</groupId>
- <artifactId>ydb-sdk-core</artifactId>
- <version>${ydb.sdk.version}</version>
- </dependency>
- <dependency>
- <groupId>io.opentelemetry</groupId>
- <artifactId>opentelemetry-sdk</artifactId>
- <version>${otel.version}</version>
- </dependency>
- <dependency>
- <groupId>io.opentelemetry</groupId>
- <artifactId>opentelemetry-exporter-otlp</artifactId>
- <version>${otel.version}</version>
- </dependency>
- ```
-
-
- Create an instance of the OpenTelemetry SDK and pass it to the transport via `OpenTelemetryTracer`:
-
-
- ```java
- import io.opentelemetry.api.OpenTelemetry;
- import io.opentelemetry.exporter.otlp.trace.OtlpGrpcSpanExporter;
- import io.opentelemetry.sdk.OpenTelemetrySdk;
- import io.opentelemetry.sdk.resources.Resource;
- import io.opentelemetry.sdk.trace.SdkTracerProvider;
- import io.opentelemetry.sdk.trace.export.BatchSpanProcessor;
- import io.opentelemetry.semconv.resource.attributes.ResourceAttributes;
- import tech.ydb.core.auth.CloudAuthHelper;
- import tech.ydb.core.grpc.GrpcTransport;
- import tech.ydb.core.opentelemetry.OpenTelemetryTracer;
- import tech.ydb.query.QueryClient;
-
- Resource resource = Resource.getDefault().toBuilder()
- .put(ResourceAttributes.SERVICE_NAME, "my-service")
- .build();
-
- SdkTracerProvider tracerProvider = SdkTracerProvider.builder()
- .setResource(resource)
- .addSpanProcessor(BatchSpanProcessor.builder(
- OtlpGrpcSpanExporter.builder()
- .setEndpoint("http://localhost:4317")
- .build()
- ).build())
- .build();
-
- OpenTelemetry openTelemetry = OpenTelemetrySdk.builder()
- .setTracerProvider(tracerProvider)
- .build();
-
- try (GrpcTransport transport = GrpcTransport.forConnectionString(connectionString)
- .withAuthProvider(CloudAuthHelper.getAuthProviderFromEnviron())
- .withTracer(OpenTelemetryTracer.fromOpenTelemetry(openTelemetry))
- .build();
- QueryClient queryClient = QueryClient.newClient(transport).build()) {
- // Use queryClient here
- }
- ```
-
-
- When using the JDBC driver, simply add the `enableOpenTelemetryTracer=true` parameter to the connection string — the driver will pick up the global OTel provider automatically:
-
-
- ```text
- jdbc:ydb://<host>:<port>/<database>?enableOpenTelemetryTracer=true
- ```
-
-- C++
-
- Include the OpenTelemetry tracing header from the {{ ydb-short-name }} C++ SDK and add a dependency on the OTel C++ SDK:
-
-
- ```cpp
- #include <ydb-cpp-sdk/client/driver/driver.h>
- #include <ydb-cpp-sdk/open_telemetry/trace.h>
-
- #include <opentelemetry/exporters/otlp/otlp_http_exporter_factory.h>
- #include <opentelemetry/exporters/otlp/otlp_http_exporter_options.h>
- #include <opentelemetry/sdk/trace/tracer_provider.h>
- #include <opentelemetry/sdk/trace/simple_processor_factory.h>
- #include <opentelemetry/sdk/resource/resource.h>
- #include <opentelemetry/trace/provider.h>
-
- namespace sdktrace = opentelemetry::sdk::trace;
- namespace otlp = opentelemetry::exporter::otlp;
- namespace resource = opentelemetry::sdk::resource;
- using namespace NYdb;
-
- // 1. Initialize the OTel tracing provider
- otlp::OtlpHttpExporterOptions opts;
- opts.url = "http://localhost:4318/v1/traces";
- auto exporter = otlp::OtlpHttpExporterFactory::Create(opts);
- auto processor = sdktrace::SimpleSpanProcessorFactory::Create(std::move(exporter));
- auto res = resource::Resource::Create({{"service.name", "my-service"}});
- auto otelProvider = std::make_shared<sdktrace::TracerProvider>(
- std::move(processor), res);
- opentelemetry::trace::Provider::SetTracerProvider(otelProvider);
-
- // 2. Wrap in the YDB tracing provider
- auto ydbTraceProvider = NTrace::CreateOtelTraceProvider(otelProvider);
-
- // 3. Create the YDB driver with tracing enabled
- auto driverConfig = TDriverConfig()
- .SetEndpoint("localhost:2136")
- .SetDatabase("/local")
- .SetTraceProvider(ydbTraceProvider);
-
- TDriver driver(driverConfig);
- ```
-
-- JavaScript
-
- Install `@ydbjs/telemetry` together with the OpenTelemetry Node SDK and the OTLP exporter:
-
-
- ```bash
- npm install @ydbjs/telemetry @opentelemetry/sdk-node @opentelemetry/exporter-trace-otlp-http
- ```
-
-
- Initialize `NodeSDK` before creating the driver and call `register()` from `@ydbjs/telemetry`:
-
-
- ```js
- import { NodeSDK } from '@opentelemetry/sdk-node'
- import { OTLPTraceExporter } from '@opentelemetry/exporter-trace-otlp-http'
- import { Driver } from '@ydbjs/core'
- import { query } from '@ydbjs/query'
- import { register } from '@ydbjs/telemetry'
-
- const sdk = new NodeSDK({
- serviceName: 'my-service',
- traceExporter: new OTLPTraceExporter({ url: 'http://localhost:4318/v1/traces' }),
- })
- sdk.start()
-
- // Must be called BEFORE creating the Driver — W3C propagation middleware
- // trace context is set once during driver construction.
- const instrumentation = register()
-
- using driver = new Driver(process.env.YDB_CONNECTION_STRING)
- await driver.ready()
- await using sql = query(driver)
- // ...
-
- instrumentation.disable()
- await sdk.shutdown()
- ```
-
-
- Alternatively, via `--import` for auto-loading before the application starts:
-
-
- ```bash
- node --import @opentelemetry/sdk-node/register --import @ydbjs/telemetry/register your-app.js
- ```
-
-{% endlist %}
diff --git a/ydb/docs/en/core/recipes/ydb-sdk/debug-otel.md b/ydb/docs/en/core/recipes/ydb-sdk/debug-otel.md
deleted file mode 100644
index e8b21dec519..00000000000
--- a/ydb/docs/en/core/recipes/ydb-sdk/debug-otel.md
+++ /dev/null
@@ -1,371 +0,0 @@
-# Tracing with OpenTelemetry
-
-{{ ydb-short-name }} SDK instruments Query Service operations with [OpenTelemetry](https://opentelemetry.io/) spans, providing distributed tracing from application code to each gRPC call to YDB. The spans are exported via the standard OTLP protocol and are compatible with Jaeger, Grafana Tempo, Zipkin, and any other backend that supports OpenTelemetry.
-
-## Created spans {#spans}
-
-Currently, spans are supported for Query Service operations. Support for topics and other services is planned for future SDK versions.
-
-The following spans are created:
-
-| Span | Type | Description |
-| --- | --- | --- |
-| `ydb.RunWithRetry` | `Internal` | Covers the entire retry cycle for a single operation |
-| `ydb.Try` | `Internal` | One span per attempt, including the first; child RPC spans are attached to it |
-| `ydb.CreateSession` | `Client` | Creating a session via gRPC `CreateSession` and `AttachStream` |
-| `ydb.ExecuteQuery` | `Client` | Execution of one YQL query |
-| `ydb.BeginTransaction` | `Client` | Explicit call to start a transaction |
-| `ydb.Commit` | `Client` | Transaction commit |
-| `ydb.Rollback` | `Client` | Transaction rollback |
-| `ydb.Driver.Initialize` | `Internal` | Initial driver initialization: cluster discovery and authentication |
-
-A typical span tree for a transactional operation with retry looks as follows:
-
-
-```text
-ydb.RunWithRetry (Internal)
-├─ ydb.Try (Internal) ← 1st attempt: ERROR
-│ ├─ ydb.ExecuteQuery (Client)
-│ ├─ ydb.ExecuteQuery (Client)
-│ └─ ydb.Commit (Client) ← ERROR: Transaction Lock Invalidated
-└─ ydb.Try (Internal) ← 2nd attempt: SUCCESS, ydb.retry.backoff_ms=50
- ├─ ydb.ExecuteQuery (Client)
- ├─ ydb.ExecuteQuery (Client)
- └─ ydb.Commit (Client)
-```
-
-
-`ydb.RunWithRetry` is the parent span for the entire retry operation. For each execution attempt, a separate child span `ydb.Try` is created: the first `ydb.Try` corresponds to the first attempt, the second to the first **retry** attempt, and so on. RPC spans of a specific attempt, for example `ydb.ExecuteQuery` and `ydb.Commit`, are created inside the corresponding `ydb.Try`.
-
-{% note info %}
-
-If an attempt ends with an error, its span `ydb.Try` ends with an error status. On a retry, a new `ydb.Try` is created; starting from the second attempt, the attribute `ydb.retry.backoff_ms` — the wait time before this attempt in milliseconds — is set on it. This wait time is included in the duration of the next span `ydb.Try`: the span starts before the backoff pause, then after the pause, the RPC calls of this attempt are executed.
-
-{% endnote %}
-
-## Span attributes {#attributes}
-
-The SDK uses both standard OpenTelemetry semantic conventions attributes and YDB-specific extensions.
-
-### Standard OpenTelemetry attributes
-
-The following attributes belong to the stable [OpenTelemetry semantic conventions](https://opentelemetry.io/docs/specs/semconv/) and can be processed by tracing backends as standard:
-
-| Attribute | Where installed | Description |
-| --- | --- | --- |
-| `db.system.name` | RPC spans | Always `"ydb"` |
-| `db.namespace` | RPC spans | Path to YDB database |
-| `server.address` | RPC spans | Main host from the connection string |
-| `server.port` | RPC spans | Main port from the connection string |
-| `network.peer.address` | RPC spans | Actual gRPC endpoint used for the call |
-| `network.peer.port` | RPC spans | Actual gRPC endpoint port used for the call |
-| `error.type` | Spans that failed | Error type. For example: `"transport_error"`, `"ydb_error"` or the full exception class name |
-| `db.response.status_code` | RPC spans during `YdbException` | Textual name of the YDB status from the error, for example `ABORTED`, `UNAVAILABLE`, `OVERLOADED` |
-
-### YDB-specific attributes
-
-The following attributes are YDB extensions on top of the standard semantic conventions:
-
-| Attribute | Where it is installed | Description |
-| --- | --- | --- |
-| `ydb.node.id` | RPC spans | ID of the YDB node that processed the query |
-| `ydb.node.dc` | RPC spans | Data center of the YDB node that processed the query |
-| `ydb.retry.backoff_ms` | Spans `ydb.Try`, starting from the second attempt | Wait time before retry in milliseconds |
-
-## W3C trace context {#w3c}
-
-The SDK automatically propagates the W3C `traceparent` header in every outgoing gRPC call. This allows the YDB server to trace internal operations within the same trace, without additional configuration. For more details about server-side tracing, see the section [Passing an external trace-id to {{ ydb-short-name }}](../../reference/observability/tracing/external-traces.md).
-
-## Connecting to the SDK {#integration}
-
-{% list tabs %}
-
-- Go
-
- Install the OpenTelemetry adapter for the {{ ydb-short-name }} Go SDK:
-
-
- ```bash
- go get github.com/ydb-platform/ydb-go-sdk-otel
- ```
-
-
- Configure `TracerProvider` and pass the adapter to `ydb.Open`:
-
-
- ```go
- package main
-
- import (
- "context"
- "os"
-
- "go.opentelemetry.io/otel"
- "go.opentelemetry.io/otel/attribute"
- "go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracegrpc"
- "go.opentelemetry.io/otel/sdk/resource"
- sdktrace "go.opentelemetry.io/otel/sdk/trace"
-
- "github.com/ydb-platform/ydb-go-sdk/v3"
- ydbOtel "github.com/ydb-platform/ydb-go-sdk-otel"
- )
-
- func main() {
- ctx := context.Background()
-
- exporter, err := otlptracegrpc.New(ctx,
- otlptracegrpc.WithEndpoint("localhost:4317"),
- otlptracegrpc.WithInsecure(),
- )
- if err != nil {
- panic(err)
- }
- res, _ := resource.Merge(resource.Default(), resource.NewSchemaless(
- attribute.String("service.name", "my-service"),
- ))
- tp := sdktrace.NewTracerProvider(
- sdktrace.WithBatcher(exporter),
- sdktrace.WithResource(res),
- )
- defer tp.Shutdown(ctx)
- otel.SetTracerProvider(tp)
-
- db, err := ydb.Open(ctx,
- os.Getenv("YDB_CONNECTION_STRING"),
- ydbOtel.WithTraces(
- ydbOtel.WithTracer(tp.Tracer("ydb-go-sdk")),
- ),
- )
- if err != nil {
- panic(err)
- }
- defer db.Close(ctx)
- }
- ```
-
-- Python
-
- Install additional dependencies `opentelemetry` and the OTLP exporter:
-
-
- ```bash
- pip install ydb[opentelemetry]
- pip install opentelemetry-exporter-otlp-proto-grpc
- ```
-
-
- Call `enable_tracing()` after configuring the global `TracerProvider`:
-
-
- ```python
- from opentelemetry import trace
- from opentelemetry.sdk.trace import TracerProvider
- from opentelemetry.sdk.trace.export import BatchSpanProcessor
- from opentelemetry.exporter.otlp.proto.grpc.trace_exporter import OTLPSpanExporter
- from opentelemetry.sdk.resources import Resource
-
- import ydb
- from ydb.opentelemetry import enable_tracing
-
- resource = Resource(attributes={"service.name": "my-service"})
- provider = TracerProvider(resource=resource)
- provider.add_span_processor(
- BatchSpanProcessor(OTLPSpanExporter(endpoint="http://localhost:4317"))
- )
- trace.set_tracer_provider(provider)
-
- enable_tracing()
-
- with ydb.Driver(endpoint="grpc://localhost:2136", database="/local") as driver:
- driver.wait(timeout=5)
- with ydb.QuerySessionPool(driver) as pool:
- pool.execute_with_retries("SELECT 1")
-
- provider.shutdown()
- ```
-
-- C#
-
- Add the NuGet package:
-
-
- ```bash
- dotnet add package Ydb.Sdk.OpenTelemetry
- ```
-
-
- Register the {{ ydb-short-name }} instrumentation when configuring OpenTelemetry in your service:
-
-
- ```csharp
- services.AddOpenTelemetry()
- .WithTracing(builder => builder
- .AddYdb()
- .AddOtlpExporter());
- ```
-
-- Java
-
- Add YDB SDK and OpenTelemetry dependencies (example for Maven):
-
-
- ```xml
- <dependency>
- <groupId>tech.ydb</groupId>
- <artifactId>ydb-sdk-core</artifactId>
- <version>${ydb.sdk.version}</version>
- </dependency>
- <dependency>
- <groupId>io.opentelemetry</groupId>
- <artifactId>opentelemetry-sdk</artifactId>
- <version>${otel.version}</version>
- </dependency>
- <dependency>
- <groupId>io.opentelemetry</groupId>
- <artifactId>opentelemetry-exporter-otlp</artifactId>
- <version>${otel.version}</version>
- </dependency>
- ```
-
-
- Create an instance of the OpenTelemetry SDK and pass it to the transport via `OpenTelemetryTracer`:
-
-
- ```java
- import io.opentelemetry.api.OpenTelemetry;
- import io.opentelemetry.exporter.otlp.trace.OtlpGrpcSpanExporter;
- import io.opentelemetry.sdk.OpenTelemetrySdk;
- import io.opentelemetry.sdk.resources.Resource;
- import io.opentelemetry.sdk.trace.SdkTracerProvider;
- import io.opentelemetry.sdk.trace.export.BatchSpanProcessor;
- import io.opentelemetry.semconv.resource.attributes.ResourceAttributes;
- import tech.ydb.core.auth.CloudAuthHelper;
- import tech.ydb.core.grpc.GrpcTransport;
- import tech.ydb.core.opentelemetry.OpenTelemetryTracer;
- import tech.ydb.query.QueryClient;
-
- Resource resource = Resource.getDefault().toBuilder()
- .put(ResourceAttributes.SERVICE_NAME, "my-service")
- .build();
-
- SdkTracerProvider tracerProvider = SdkTracerProvider.builder()
- .setResource(resource)
- .addSpanProcessor(BatchSpanProcessor.builder(
- OtlpGrpcSpanExporter.builder()
- .setEndpoint("http://localhost:4317")
- .build()
- ).build())
- .build();
-
- OpenTelemetry openTelemetry = OpenTelemetrySdk.builder()
- .setTracerProvider(tracerProvider)
- .build();
-
- try (GrpcTransport transport = GrpcTransport.forConnectionString(connectionString)
- .withAuthProvider(CloudAuthHelper.getAuthProviderFromEnviron())
- .withTracer(OpenTelemetryTracer.fromOpenTelemetry(openTelemetry))
- .build();
- QueryClient queryClient = QueryClient.newClient(transport).build()) {
- // Use queryClient here
- }
- ```
-
-
- When using the JDBC driver, simply add the `enableOpenTelemetryTracer=true` parameter to the connection string — the driver will pick up the global OTel provider automatically:
-
-
- ```text
- jdbc:ydb://<host>:<port>/<database>?enableOpenTelemetryTracer=true
- ```
-
-- C++
-
- Include the OpenTelemetry tracing header from the {{ ydb-short-name }} C++ SDK and add a dependency on the OTel C++ SDK:
-
-
- ```cpp
- #include <ydb-cpp-sdk/client/driver/driver.h>
- #include <ydb-cpp-sdk/open_telemetry/trace.h>
-
- #include <opentelemetry/exporters/otlp/otlp_http_exporter_factory.h>
- #include <opentelemetry/exporters/otlp/otlp_http_exporter_options.h>
- #include <opentelemetry/sdk/trace/tracer_provider.h>
- #include <opentelemetry/sdk/trace/simple_processor_factory.h>
- #include <opentelemetry/sdk/resource/resource.h>
- #include <opentelemetry/trace/provider.h>
-
- namespace sdktrace = opentelemetry::sdk::trace;
- namespace otlp = opentelemetry::exporter::otlp;
- namespace resource = opentelemetry::sdk::resource;
- using namespace NYdb;
-
- // 1. Initialize the OTel tracing provider
- otlp::OtlpHttpExporterOptions opts;
- opts.url = "http://localhost:4318/v1/traces";
- auto exporter = otlp::OtlpHttpExporterFactory::Create(opts);
- auto processor = sdktrace::SimpleSpanProcessorFactory::Create(std::move(exporter));
- auto res = resource::Resource::Create({{"service.name", "my-service"}});
- auto otelProvider = std::make_shared<sdktrace::TracerProvider>(
- std::move(processor), res);
- opentelemetry::trace::Provider::SetTracerProvider(otelProvider);
-
- // 2. Wrap in the YDB tracing provider
- auto ydbTraceProvider = NTrace::CreateOtelTraceProvider(otelProvider);
-
- // 3. Create a YDB driver with tracing enabled
- auto driverConfig = TDriverConfig()
- .SetEndpoint("localhost:2136")
- .SetDatabase("/local")
- .SetTraceProvider(ydbTraceProvider);
-
- TDriver driver(driverConfig);
- ```
-
-- JavaScript
-
- Install `@ydbjs/telemetry` together with the OpenTelemetry Node SDK and OTLP exporter:
-
-
- ```bash
- npm install @ydbjs/telemetry @opentelemetry/sdk-node @opentelemetry/exporter-trace-otlp-http
- ```
-
-
- Initialize `NodeSDK` before creating the driver and call `register()` from `@ydbjs/telemetry`:
-
-
- ```js
- import { NodeSDK } from '@opentelemetry/sdk-node'
- import { OTLPTraceExporter } from '@opentelemetry/exporter-trace-otlp-http'
- import { Driver } from '@ydbjs/core'
- import { query } from '@ydbjs/query'
- import { register } from '@ydbjs/telemetry'
-
- const sdk = new NodeSDK({
- serviceName: 'my-service',
- traceExporter: new OTLPTraceExporter({ url: 'http://localhost:4318/v1/traces' }),
- })
- sdk.start()
-
- // Must be called BEFORE creating the Driver — W3C propagation middleware
- // trace context is set once during driver construction.
- const instrumentation = register()
-
- using driver = new Driver(process.env.YDB_CONNECTION_STRING)
- await driver.ready()
- await using sql = query(driver)
- // ...
-
- instrumentation.disable()
- await sdk.shutdown()
- ```
-
-
- Alternatively, use `--import` for auto-loading before starting the application:
-
-
- ```bash
- node --import @opentelemetry/sdk-node/register --import @ydbjs/telemetry/register your-app.js
- ```
-
-{% endlist %}
diff --git a/ydb/docs/en/core/recipes/ydb-sdk/debug-prometheus.md b/ydb/docs/en/core/recipes/ydb-sdk/debug-prometheus.md
deleted file mode 100644
index 855ae9ff88d..00000000000
--- a/ydb/docs/en/core/recipes/ydb-sdk/debug-prometheus.md
+++ /dev/null
@@ -1,106 +0,0 @@
-# Enabling metrics in Prometheus
-
-Below are examples of the code for enabling metrics in Prometheus in different {{ ydb-short-name }} SDKs.
-
-{% list tabs %}
-
-- Go
-
- {% list tabs %}
-
- - Native SDK
-
- ```go
- package main
-
- import (
- "context"
-
- "github.com/prometheus/client_golang/prometheus"
- metrics "github.com/ydb-platform/ydb-go-sdk-prometheus/v2"
- "github.com/ydb-platform/ydb-go-sdk/v3"
- "github.com/ydb-platform/ydb-go-sdk/v3/trace"
- )
-
- func main() {
- ctx := context.Background()
- registry := prometheus.NewRegistry()
- db, err := ydb.Open(ctx,
- os.Getenv("YDB_CONNECTION_STRING"),
- metrics.WithTraces(
- registry,
- metrics.WithDetails(trace.DetailsAll),
- metrics.WithSeparator("_"),
- ),
- )
- if err != nil {
- panic(err)
- }
- defer db.Close(ctx)
- ...
- }
- ```
-
- - database/sql
-
- ```go
- package main
-
- import (
- "context"
- "database/sql"
-
- "github.com/prometheus/client_golang/prometheus"
- metrics "github.com/ydb-platform/ydb-go-sdk-prometheus/v2"
- "github.com/ydb-platform/ydb-go-sdk/v3"
- "github.com/ydb-platform/ydb-go-sdk/v3/trace"
- )
-
- func main() {
- ctx := context.Background()
- registry := prometheus.NewRegistry()
- nativeDriver, err := ydb.Open(ctx,
- os.Getenv("YDB_CONNECTION_STRING"),
- metrics.WithTraces(
- registry,
- metrics.WithDetails(trace.DetailsAll),
- metrics.WithSeparator("_"),
- ),
- )
- if err != nil {
- panic(err)
- }
- defer nativeDriver.Close(ctx)
-
- connector, err := ydb.Connector(nativeDriver)
- if err != nil {
- panic(err)
- }
-
- db := sql.OpenDB(connector)
- defer db.Close()
- ...
- }
- ```
-
- {% endlist %}
-
-- Java
-
- This functionality is not currently supported.
-
-- Python
-
- This functionality is not currently supported.
-
-- JavaScript
-
- {% include [work-in-progress](../../_includes/work-in-progress.md) %}
-
-- Rust
-
- {% include [feature-not-supported](../../_includes/feature-not-supported.md) %}
-
- Track progress or vote for Rust SDK support: [ydb-rs-sdk#267](https://github.com/ydb-platform/ydb-rs-sdk/issues/267)
-
-{% endlist %}
diff --git a/ydb/docs/en/core/recipes/ydb-sdk/debug.md b/ydb/docs/en/core/recipes/ydb-sdk/debug.md
deleted file mode 100644
index 6b54b61c897..00000000000
--- a/ydb/docs/en/core/recipes/ydb-sdk/debug.md
+++ /dev/null
@@ -1,12 +0,0 @@
-# Diagnosing problems
-
-When diagnosing problems related to {{ ydb-short-name }}, diagnostic tools help: logging, metrics, and distributed tracing. It is recommended to enable them in advance, before problems occur, so that when investigating an incident, you can see the full picture of the system state before, during, and after the failure.
-
-This section contains code recipes for enabling diagnostic tools in different {{ ydb-short-name }} SDKs.
-
-Contents:
-
-- [Enable logging](debug-logs.md)
-- [Connect metrics to Prometheus](debug-prometheus.md)
-- [Tracing with OpenTelemetry](debug-otel.md)
-- [Connect tracing to Jaeger](debug-jaeger.md)
diff --git a/ydb/docs/en/core/recipes/ydb-sdk/index.md b/ydb/docs/en/core/recipes/ydb-sdk/index.md
index 650af008091..c1217c4f34f 100644
--- a/ydb/docs/en/core/recipes/ydb-sdk/index.md
+++ b/ydb/docs/en/core/recipes/ydb-sdk/index.md
@@ -31,11 +31,8 @@ Contents:
- [Service discovery](service-discovery.md)
- [Configuration publishing](config-publication.md)
- [Leader election](leader-election.md)
-- [Problem diagnostics](debug.md)
- - [Enable logging](debug-logs.md)
- - [Connect metrics to Prometheus](debug-prometheus.md)
- - [Tracing with OpenTelemetry](debug-otel.md)
+Connecting {{ ydb-short-name }} SDK diagnostics — logging, metrics, and distributed tracing — is described in [{#T}](../../reference/ydb-sdk/observability/index.md).
See also:
diff --git a/ydb/docs/en/core/recipes/ydb-sdk/toc_i.yaml b/ydb/docs/en/core/recipes/ydb-sdk/toc_i.yaml
index cb7d29a231c..de23ade42c8 100644
--- a/ydb/docs/en/core/recipes/ydb-sdk/toc_i.yaml
+++ b/ydb/docs/en/core/recipes/ydb-sdk/toc_i.yaml
@@ -55,15 +55,3 @@ items:
href: service-discovery.md
- name: Configuration publication
href: config-publication.md
-- name: Troubleshooting
- items:
- - name: Overview
- href: debug.md
- - name: Enable logging
- href: debug-logs.md
- - name: Enable metrics in Prometheus
- href: debug-prometheus.md
- - name: Tracing with OpenTelemetry
- href: debug-otel.md
- - name: Connect tracing in Jaeger
- href: debug-jaeger.md
diff --git a/ydb/docs/en/core/reference/configuration/feature_flags.md b/ydb/docs/en/core/reference/configuration/feature_flags.md
index 72af624eca5..92867eb88c7 100644
--- a/ydb/docs/en/core/reference/configuration/feature_flags.md
+++ b/ydb/docs/en/core/reference/configuration/feature_flags.md
@@ -1,29 +1,35 @@
# feature_flags
-The `feature_flags` section enables or disables specific {{ ydb-short-name }} features using boolean flags. To enable a feature, set the corresponding feature flag to `true` in the cluster configuration. For example, to enable support for auto-partitioning of topics in the CDC, you need to add the following lines to the configuration:
+The `feature_flags` section enables or disables certain {{ ydb-short-name }} features using boolean flags. To enable a feature, set the corresponding feature flag to `true` in the cluster configuration. For example, to enable support for auto-partitioning of topics in CDC, add the following lines to the configuration:
+
```yaml
feature_flags:
enable_topic_autopartitioning_for_cdc: true
```
-## Feature Flags
-| Flag | Feature |
-|---------------------------| ----------------------------------------------------|
-| `enable_fulltext_index` | [Fulltext index](../../dev/fulltext-indexes.md) for fulltext search |
-| `enable_local_bloom_filter_index` | [Local Bloom skip index](../../dev/bloom-skip-indexes.md#types) of type `bloom_filter` |
-| `enable_local_bloom_ngram_filter_index` | [Local Bloom skip index](../../dev/bloom-skip-indexes.md#types) of type `bloom_ngram_filter` |
-| `enable_topic_autopartitioning_for_cdc` | [Auto-partitioning topics](../../concepts/cdc.md#topic-partitions) for row-oriented tables in CDC |
-| `enable_access_to_index_impl_tables` | Support for [followers (read replicas)](../../yql/reference/syntax/alter_table/indexes.md) for covered secondary indexes |
-| `enable_changefeeds_export`, `enable_changefeeds_import` | Support for changefeeds in backup and restore operations |
-| `enable_view_export` | Support for views in backup and restore operations |
-| `enable_export_auto_dropping` | Automatic cleanup of temporary tables and directories during export to S3 |
-| `enable_followers_stats` | System views with information about [history of overloaded partitions](../../dev/system-views.md#top-overload-partitions) |
-| `enable_strict_acl_check` | Strict ACL checks — do not allow granting rights to non-existent users and delete users with permissions |
-| `enable_strict_user_management` | Strict checks for local users — only the cluster or database administrator can administer local users |
-| `enable_database_admin` | The role of a database administrator |
-| `enable_kafka_native_balancing` | Client balancing of partitions when reading using the [Kafka protocol](https://kafka.apache.org/documentation/#consumerconfigs_partition.assignment.strategy) |
-| `enable_topic_compactification_by_key` | Enabling topic compactification in the [YDB Topics Kafka API](../../reference/kafka-api/index.md) |
-| `enable_kafka_transactions` | Enabling transactions in the [YDB Topics Kafka API](../../reference/kafka-api/index.md) |
-| `enable_grpc_audit` | Enabling [audit](../../security/audit-log.md#grpc-connection) of gRPC connection state changes |
+## Feature flags
+
+| Flag | Function |
+| --- | --- |
+| `enable_json_index` | [JSON indexes](../../dev/json-indexes.md) to speed up search in JSON fields |
+| `enable_json_index_auto_select` | Automatic selection of [JSON indexes](../../dev/json-indexes.md) when executing queries |
+| `enable_fulltext_index` | [Full-text index](../../dev/fulltext-indexes.md) for full-text search |
+| `enable_local_bloom_filter_index` | [Local Bloom index](../../dev/bloom-skip-indexes.md#types) of type `bloom_filter` |
+| `enable_local_bloom_ngram_filter_index` | [Local Bloom index](../../dev/bloom-skip-indexes.md#types) of type `bloom_ngram_filter` |
+| `enable_topic_autopartitioning_for_cdc` | [Auto-partitioning of topics](../../concepts/cdc.md#topic-partitions) in CDC for row tables |
+| `enable_access_to_index_impl_tables` | Ability to [specify the number of replicas](../../yql/reference/syntax/alter_table/indexes.md) for a secondary index |
+| `enable_changefeeds_export`, `enable_changefeeds_import` | Support for change feeds (changefeed) in backup and restore operations |
+| `enable_view_export` | Support for views (`VIEW`) in backup and restore operations |
+| `enable_export_auto_dropping` | Auto-deletion of temporary directories and tables when exporting to S3 |
+| `enable_followers_stats` | System views with information about [history of overloaded partitions](../../dev/system-views#top-overload-partitions) |
+| `enable_strict_acl_check` | Prohibition on granting permissions to non-existent users and on deleting users who have been granted permissions |
+| `enable_strict_user_management` | Strict rules for administering local users (i.e., only the cluster or database administrator can administer local users) |
+| `enable_database_admin` | Adding the database administrator role |
+| `enable_kafka_native_balancing` | Client-side partition balancing when reading via [Kafka protocol](https://kafka.apache.org/documentation/#consumerconfigs_partition.assignment.strategy) |
+| `enable_topic_compactification_by_key` | Enabling topic compaction in [YDB Topics Kafka API](../../reference/kafka-api/index.md) |
+| `enable_kafka_transactions` | Enabling transactions in [YDB Topics Kafka API](../../reference/kafka-api/index.md) |
+| `enable_external_data_sources` | Enabling [external data sources](../../concepts/datamodel/external_data_source.md) |
+| `enable_grpc_audit` | Enabling [audit](../../security/audit-log.md#grpc-connection) of gRPC connection state changes |
+| `enable_fs_backups` | Enabling [backup and restore operations to a network file system](../../concepts/backup.md#nfs) |
diff --git a/ydb/docs/en/core/reference/configuration/hive.md b/ydb/docs/en/core/reference/configuration/hive_config.md
index 8f65c4e39ab..8f65c4e39ab 100644
--- a/ydb/docs/en/core/reference/configuration/hive.md
+++ b/ydb/docs/en/core/reference/configuration/hive_config.md
diff --git a/ydb/docs/en/core/reference/configuration/host_configs.md b/ydb/docs/en/core/reference/configuration/host_configs.md
index b26b51119b8..43e236ece91 100644
--- a/ydb/docs/en/core/reference/configuration/host_configs.md
+++ b/ydb/docs/en/core/reference/configuration/host_configs.md
@@ -10,16 +10,19 @@ host_configs:
drive:
- path: <path_to_device>
type: <type>
+ disk_scope: <disk_scope> # optional attribute
- path: ...
- host_config_id: 2
...
```
-The `host_config_id` attribute specifies a numeric configuration ID. The `drive` attribute contains a collection of descriptions of connected drives. Each description consists of two attributes:
+The `host_config_id` attribute specifies a numeric configuration ID. The `drive` attribute contains a collection of descriptions of connected drives. Each description consists of two mandatory attributes:
- `path`: Path to the mounted block device, for example, `/dev/disk/by-partlabel/ydb_disk_ssd_01`
- `type`: Type of the device's physical media: `ssd`, `nvme`, or `rot` (rotational - HDD)
+Additionally, an optional `disk_scope` attribute can be specified — a label for calculating fail domains for some reduced configurations, see [Configuring disk_scope](#disk-scope).
+
## Examples
One configuration with ID 1 and one SSD disk accessible via `/dev/disk/by-partlabel/ydb_disk_ssd_01`:
@@ -52,6 +55,54 @@ host_configs:
type: SSD
```
+## Configuring disk_scope {#disk-scope}
+
+`disk_scope` is an optional string attribute of a disk that defines a finer-grained failure zone within a single node. It is taken into account when calculating [fail domains](../../concepts/glossary.md#fail-domain) during the selection of disks for placing [VDisks](../../concepts/glossary.md#vdisk) of [storage groups](../../concepts/glossary.md#storage-group).
+
+A fail domain is determined by the physical location of a disk (data center, rack, server, physical device), so either no two VDisks of the same group can be placed on the same node (if the fail domain corresponds to a server or a server rack), or several VDisks of the same group can be placed on the same node on different disks (if the fail domain corresponds to an individual physical device). In the first case, it is impossible to create a configuration in `block-4-2` mode with fewer than 8 storage nodes, and in the second case the configuration might not stand the failure of a single node. To limit the number of VDisks of the same group placed on a single node, physical devices can be labeled with the `disk_scope` attribute, and fail domain calculation can be enabled at the `disk_scope` level. This makes it possible to build some fault-tolerant configurations in installations that have fewer servers than the number of fail domains required by the chosen [fault tolerance mode](../../concepts/topology.md).
+
+### Example {#disk-scope-example}
+
+A cluster of 4 servers with 4 disks on each server, in the `block-4-2` fault tolerance mode:
+
+``` yaml
+host_configs:
+- host_config_id: 1
+ drive:
+ - path: /dev/disk/by-partlabel/ydb_disk_ssd_01
+ type: SSD
+ disk_scope: fail-domain-1
+ - path: /dev/disk/by-partlabel/ydb_disk_ssd_02
+ type: SSD
+ disk_scope: fail-domain-1
+ - path: /dev/disk/by-partlabel/ydb_disk_ssd_03
+ type: SSD
+ disk_scope: fail-domain-2
+ - path: /dev/disk/by-partlabel/ydb_disk_ssd_04
+ type: SSD
+ disk_scope: fail-domain-2
+```
+
+To calculate fail domains at the `disk_scope` level for the static group and other storage groups, specify the corresponding fail domain type in the yaml config (top-level key):
+
+``` yaml
+fail_domain_type: disk_scope
+```
+
+And in the domain configuration:
+
+```yaml
+domains:
+- domain_name: <domain name>
+ ...
+ storage_pool_kinds:
+ - kind: <type of physical devices used>
+ fail_domain_type: disk_scope
+ ...
+```
+
+With this configuration, two VDisks of each storage group will be placed on each server, and the maximum tolerated failure in such a configuration is the failure of a single server.
+
## Kubernetes Features {#host-configs-k8s}
The {{ ydb-short-name }} Kubernetes operator mounts NBS disks for Storage nodes at the path `/dev/kikimr_ssd_00`. To use them, the following `host_configs` configuration must be specified:
diff --git a/ydb/docs/en/core/reference/configuration/index.md b/ydb/docs/en/core/reference/configuration/index.md
index 76ab6831c2f..af930764d40 100644
--- a/ydb/docs/en/core/reference/configuration/index.md
+++ b/ydb/docs/en/core/reference/configuration/index.md
@@ -21,10 +21,10 @@ The following top-level configuration sections are available, listed in alphabet
|| [{#T}](domains_config.md) | No | Cluster domain configuration including Blob Storage and State Storage ||
|| [{#T}](feature_flags.md) | No | Feature flags to enable or disable specific {{ ydb-short-name }} features ||
|| [{#T}](healthcheck_config.md) | No | Health check service thresholds and timeout settings ||
-|| [{#T}](hive.md) | No | Hive component configuration for tablet management ||
+|| [{#T}](hive_config.md) | No | Hive component configuration for tablet management ||
|| [{#T}](host_configs.md) | No | Typical host configurations for cluster nodes ||
|| [{#T}](hosts.md) | Yes | Static cluster nodes configuration ||
-|| [{#T}](kafka.md) | No | [Kafka Proxy](../../reference/kafka-api/index.md) configuration ||
+|| [{#T}](kafka_proxy_config.md) | No | [Kafka Proxy](../../reference/kafka-api/index.md) configuration ||
|| [{#T}](log_config.md) | No | Logging configuration and parameters ||
|| [{#T}](memory_controller_config.md) | No | Memory allocation and limits for database components ||
|| [{#T}](node_broker_config.md) | No | Stable node names configuration ||
diff --git a/ydb/docs/en/core/reference/configuration/kafka.md b/ydb/docs/en/core/reference/configuration/kafka_proxy_config.md
index a4593cd1d6e..065a248530b 100644
--- a/ydb/docs/en/core/reference/configuration/kafka.md
+++ b/ydb/docs/en/core/reference/configuration/kafka_proxy_config.md
@@ -8,6 +8,7 @@ The `kafka_proxy_config` section of the {{ ydb-short-name }} configuration file
| --- | --- | --- | --- |
| `enable_kafka_proxy` | bool | `false` | Enables or disables Kafka Proxy. |
| `listening_port` | int32 | `9092` | The port on which the Kafka API will be available. |
+| `listening_address` | string | `[::]` | The network address on which Kafka Proxy listens for incoming connections. Use `[::]` to listen on all interfaces (dual-stack, requires IPv6 to be enabled), `127.0.0.1` or `[::1]` to restrict access to localhost. |
| `transaction_timeout_ms` | uint32 | `300000` (5 minutes) | The maximum timeout for Kafka transactions, after which the transaction will be cancelled. |
| `auto_create_topics_enable` | bool | `false` | Enables automatic creation of topics when they are accessed. Analogous to [the same option](https://kafka.apache.org/documentation/#brokerconfigs_auto.create.topics.enable) in Apache Kafka. |
| `auto_create_consumers_enable` | bool | `true` | Enables automatic registration of consumers when they are accessed. |
@@ -22,6 +23,7 @@ The `kafka_proxy_config` section of the {{ ydb-short-name }} configuration file
kafka_proxy_config:
enable_kafka_proxy: true
listening_port: 9092
+ listening_address: "[::]"
transaction_timeout_ms: 300000 # 5 minutes
auto_create_topics_enable: true
auto_create_consumers_enable: true
diff --git a/ydb/docs/en/core/reference/configuration/monitoring_config.md b/ydb/docs/en/core/reference/configuration/monitoring_config.md
new file mode 100644
index 00000000000..ea734d88044
--- /dev/null
+++ b/ydb/docs/en/core/reference/configuration/monitoring_config.md
@@ -0,0 +1,47 @@
+# monitoring_config
+
+The `monitoring_config` section of the {{ ydb-short-name }} configuration file configures [YDB Monitoring](../embedded-ui/ydb-monitoring.md). The parameters below control [authentication](../../security/authentication.md) on individual embedded monitoring pages.
+
+```yaml
+monitoring_config:
+ # authentication on the /counters and /healthcheck pages
+ require_counters_authentication: false
+ require_healthcheck_authentication: false
+```
+
+## Authentication on Monitoring Pages {#authentication}
+
+#|
+|| Parameter | Description ||
+|| `require_counters_authentication` | Selects mandatory [authentication](../../security/authentication.md) mode for the `/counters` and `/counters/hosts` pages.
+
+Valid values:
+
+- `true`: Access to `/counters` and `/counters/hosts` requires an [auth token](../../concepts/glossary.md#auth-token). Requests undergo authentication and authorization.
+
+ The `true` value is allowed only when mandatory [authentication](../../security/authentication.md) is enabled in the [security_config](./security_config.md) section of the {{ ydb-short-name }} configuration file.
+
+- `false`: Requests to `/counters` and `/counters/hosts` can be made without an [auth token](../../concepts/glossary.md#auth-token).
+
+Default value: `false`.
+ ||
+|| `require_healthcheck_authentication` | Adds an [authentication](../../security/authentication.md) requirement for the `/healthcheck` endpoint on top of the cluster-wide rules.
+
+Valid values:
+
+- `true`: Any `/healthcheck` response, including [Prometheus format](https://prometheus.io/docs/instrumenting/exposition_formats/) output (the `format=prometheus` parameter), is returned only for requests with an [auth token](../../concepts/glossary.md#auth-token). Requests undergo authentication and authorization.
+
+ The `true` value is allowed only when mandatory [authentication](../../security/authentication.md) is enabled in the [security_config](./security_config.md) section of the {{ ydb-short-name }} configuration file.
+
+- `false`: When mandatory authentication is enabled in the cluster, requests to `/healthcheck` without a token are still allowed if [Prometheus format](https://prometheus.io/docs/instrumenting/exposition_formats/) output is requested (`format=prometheus`). Cluster-wide rules apply to all other `/healthcheck` response formats (see the note below).
+
+Default value: `false`.
+
+{% note info %}
+
+If mandatory [authentication](../../security/authentication.md) is enabled in [security_config](./security_config.md), an auth token is required for `/healthcheck` responses in any format other than Prometheus, regardless of the `require_healthcheck_authentication` value.
+
+{% endnote %}
+
+||
+|#
diff --git a/ydb/docs/en/core/reference/configuration/toc_p.yaml b/ydb/docs/en/core/reference/configuration/toc_p.yaml
index a5b0c7aca1c..530013aa1b7 100644
--- a/ydb/docs/en/core/reference/configuration/toc_p.yaml
+++ b/ydb/docs/en/core/reference/configuration/toc_p.yaml
@@ -21,14 +21,20 @@ items:
href: query_service_config.md
- name: healthcheck_config
href: healthcheck_config.md
+- name: hive_config
+ href: hive_config.md
- name: host_configs
href: host_configs.md
- name: hosts
href: hosts.md
+- name: kafka_proxy_config
+ href: kafka_proxy_config.md
- name: log_config
href: log_config.md
- name: memory_controller_config
href: memory_controller_config.md
+- name: monitoring_config
+ href: monitoring_config.md
- name: node_broker_config
href: node_broker_config.md
- name: resource_broker_config
@@ -43,7 +49,3 @@ items:
href: table_service_config.md
- name: tli_config
href: tli_config.md
-- name: hive_config
- href: hive.md
-- name: kafka_proxy_config
- href: kafka.md
diff --git a/ydb/docs/en/core/reference/kafka-api/auth.md b/ydb/docs/en/core/reference/kafka-api/auth.md
index 6645cf358f3..e493f1cdfdc 100644
--- a/ydb/docs/en/core/reference/kafka-api/auth.md
+++ b/ydb/docs/en/core/reference/kafka-api/auth.md
@@ -154,7 +154,7 @@ ssl.endpoint.identification.algorithm=
#### YDB configuration
-It is necessary to specify the required fields in the [kafka_proxy_config](../configuration/kafka.md).
+It is necessary to specify the required fields in the [kafka_proxy_config](../configuration/kafka_proxy_config.md).
```yaml
kafka_proxy_config:
diff --git a/ydb/docs/en/core/reference/observability/tracing/external-traces.md b/ydb/docs/en/core/reference/observability/tracing/external-traces.md
index a293348daa5..5dec01a970b 100644
--- a/ydb/docs/en/core/reference/observability/tracing/external-traces.md
+++ b/ydb/docs/en/core/reference/observability/tracing/external-traces.md
@@ -19,4 +19,4 @@ If the [`external_throttling`](./setup.md#external-throttling) section is presen
## SDK support
-Some SDKs support trace-id passing; you can find their list and usage examples on the [{#T}](../../../recipes/ydb-sdk/debug-otel.md) page.
+Some SDKs support trace-id passing; you can find their list and usage examples on the [{#T}](../../ydb-sdk/observability/tracing/opentelemetry.md) page.
diff --git a/ydb/docs/en/core/yql/reference/syntax/_includes/index_grammar_explanation.md b/ydb/docs/en/core/yql/reference/syntax/_includes/index_grammar_explanation.md
index 5a526a3acaf..d1b89d58256 100644
--- a/ydb/docs/en/core/yql/reference/syntax/_includes/index_grammar_explanation.md
+++ b/ydb/docs/en/core/yql/reference/syntax/_includes/index_grammar_explanation.md
@@ -1,24 +1,22 @@
-* `GLOBAL/LOCAL` — global or local index; depending on the index type (`<index_type>`), only one of them may be available:
-
- * `GLOBAL` — an index implemented as a separate table or set of tables. Synchronous updates to such an index require distributed transactions.
- * `LOCAL` — a local index within a shard of a row-oriented or column-oriented table. Does not require distributed transactions for updates, but does not provide pruning during search.
+* `GLOBAL/LOCAL` — global or local index, depending on the index type (`<index_type>`), only one of them may be available:
-* `<index_name>` — a unique name of the index that will be used to access data.
-* `UNIQUE` — required to create a [unique index](../../../../concepts/query_execution/secondary_indexes.md#unique). A unique index must always be created as global and synchronous (`GLOBAL UNIQUE SYNC`) and must not specify a `USING <index_type>` clause.
-* `SYNC/ASYNC` — the index synchronization mode.
+ * `GLOBAL` — an index implemented as a separate table or a set of tables. Synchronous update of such an index requires distributed transactions.
+ * `LOCAL` — a local index within a shard of a columnar or row-based table, does not require distributed transactions during update, but does not provide pruning during search.
+* `<index_name>` — unique index name by which data can be accessed.
+* `SYNC/ASYNC` — indicator of index synchrony.
- * `SYNC` — a [synchronous](../../../../concepts/query_execution/secondary_indexes.md#sync) index. This is the default value.
- * `ASYNC` — an [asynchronous](../../../../concepts/query_execution/secondary_indexes.md#async) index.
+ * `SYNC` - [synchronous](../../../../concepts/query_execution/secondary_indexes.md#sync) index. Default value.
+ * `ASYNC` - [asynchronous](../../../../concepts/query_execution/secondary_indexes.md#async) index.
+* `UNIQUE` — indicator of a [unique secondary index](../../../../concepts/query_execution/secondary_indexes.md#unique). A unique index must be global synchronous (`GLOBAL UNIQUE SYNC`) and must not contain the `USING <index_type>` construct.
+* `<index_type>` - index type, currently supported:
-* `<index_type>` — index type, currently supported:
-
- * `secondary` — secondary index. Only `GLOBAL` mode is available for secondary indexes. This is the default index type.
- * `vector_kmeans_tree` — vector index. Described in detail in [{#T}](../create_table/vector_index.md).
- * `fulltext_plain` — basic fulltext index. Described in detail in [{#T}](../create_table/fulltext_index.md).
- * `fulltext_relevance` — fulltext index with [BM25](https://en.wikipedia.org/wiki/Okapi_BM25) statistics for relevance scoring. Described in detail in [{#T}](../create_table/fulltext_index.md).
- * `bloom_filter` — local Bloom skip index. Only `LOCAL` is available. See [ALTER TABLE ADD INDEX](../alter_table/indexes.md#local-bloom).
- * `bloom_ngram_filter` — local N-gram Bloom skip index. Only `LOCAL` is available. See [ALTER TABLE ADD INDEX](../alter_table/indexes.md#local-bloom).
-
-* `<index_columns>` — comma-separated list of column names for the table being created. This list defines the composition and order of columns included in the index key. Must be specified. The index key will include both the columns listed and the columns from the table's primary key.
-* `<cover_columns>` — comma-separated list of column names from the created table that will be saved in the index in addition to index key columns, providing the ability to get additional data without accessing the table. Empty by default.
-* `<parameter_name>` and `<parameter_value>` — index parameters specific to a particular `<index_type>`.
+ * `secondary` — secondary index. Only the `GLOBAL` mode is available for secondary indexes. This is the default index type.
+ * `vector_kmeans_tree` — vector index. More details are described in the [{#T}](../create_table/vector_index.md) section.
+ * `fulltext_plain` — basic full-text index. More details are described in [{#T}](../create_table/fulltext_index.md).
+ * `fulltext_relevance` — full-text index with [BM25](https://en.wikipedia.org/wiki/Okapi_BM25) statistics for relevance calculation. More details are described in [{#T}](../create_table/fulltext_index.md).
+ * `json` — JSON index to speed up predicates `JSON_EXISTS` and `JSON_VALUE` on a column of type `Json` or `JsonDocument`. More details are described in [{#T}](../create_table/json_index.md).
+ * `bloom_filter` — local Bloom index. Only `LOCAL` is available. See [ALTER TABLE ADD INDEX](../alter_table/indexes.md#local-bloom).
+ * `bloom_ngram_filter` — local N-gram Bloom index. Only `LOCAL` is available. See [ALTER TABLE ADD INDEX](../alter_table/indexes.md#local-bloom).
+* `<index_columns>` — comma-separated list of column names of the table being created, which determines the composition and order of columns included in the index key. Must be specified. The index key will consist of these columns with the addition of the table's primary key columns.
+* `<cover_columns>` — comma-separated list of column names of the table being created that will be stored in the index in addition to the index key columns, allowing you to get additional data without accessing the table. Empty by default.
+* `<parameter_name>` and `<parameter_value>` are index parameters specific to a particular `<index_type>`.
diff --git a/ydb/docs/en/core/yql/reference/syntax/alter-resource-pool-classifier.md b/ydb/docs/en/core/yql/reference/syntax/alter-resource-pool-classifier.md
index 4414c69b316..6c3ebd41593 100644
--- a/ydb/docs/en/core/yql/reference/syntax/alter-resource-pool-classifier.md
+++ b/ydb/docs/en/core/yql/reference/syntax/alter-resource-pool-classifier.md
@@ -1,6 +1,6 @@
# ALTER RESOURCE POOL CLASSIFIER
-`ALTER RESOURCE POOL CLASSIFIER` changes the definition of a [resource pool classifier](../../../concepts/glossary.md#resource-pool-classifier).
+`ALTER RESOURCE POOL CLASSIFIER` changes the definition of a [resource pool classifier](../../../concepts/glossary.md#resource-pool-classifier.md).
## Syntax
@@ -8,47 +8,57 @@
The syntax for changing any resource pool classifier parameter is as follows:
+
```yql
ALTER RESOURCE POOL CLASSIFIER <name> SET (<key> = <value>);
```
+
`<key>` is the parameter name, `<value>` is its new value.
For example, the following command changes the user to which the rule applies:
+
```yql
ALTER RESOURCE POOL CLASSIFIER olap_classifier SET (MEMBER_NAME = "user2@domain");
```
+
### Resetting parameters
The command to reset a resource pool classifier parameter is as follows:
+
```yql
ALTER RESOURCE POOL CLASSIFIER <name> RESET (<key>);
```
+
`<key>` is the parameter name.
For example, the following command resets the `MEMBER_NAME` setting:
+
```yql
ALTER RESOURCE POOL CLASSIFIER olap_classifier RESET (MEMBER_NAME);
```
+
## Permissions
The `ALL` [permission](grant.md#permissions-list) on the database is required. Example of granting it:
+
```yql
GRANT 'ALL' ON `/my_db` TO `user1@domain`;
```
+
## Parameters
* `RANK` (Int64) — Optional field that defines the order in which resource pool classifiers are chosen. If omitted, the maximum existing `RANK` is taken and 1000 is added. Allowed values: a unique number in the range $[0, 2^{63}-1]$.
* `RESOURCE_POOL` (String) — Required field: name of the resource pool to which queries matching the classifier criteria are sent.
-* `MEMBER_NAME` (String) — Optional field specifying which user or group is routed to the given resource pool. If omitted, the classifier ignores `MEMBER_NAME` and classification uses other criteria.
+* `MEMBER_NAME` (String) — Optional field specifying which user or group is routed to the given resource pool. If omitted, the classifier ignores ⟦C2⟧ and classification uses other criteria.
## See also
diff --git a/ydb/docs/en/core/yql/reference/syntax/alter_table/indexes.md b/ydb/docs/en/core/yql/reference/syntax/alter_table/indexes.md
index 83eee7b18e5..fc382b2728c 100644
--- a/ydb/docs/en/core/yql/reference/syntax/alter_table/indexes.md
+++ b/ydb/docs/en/core/yql/reference/syntax/alter_table/indexes.md
@@ -1,8 +1,9 @@
-# Adding, removing, and renaming a index
+# Adding, deleting, and renaming an index
## Adding an index {#add-index}
-`ADD INDEX` — adds an index with the specified name and type for a given set of columns. Grammar:
+`ADD INDEX` — adds an index with the specified name and type for the given set of columns in {% if backend_name == "YDB" and oss == true %}row tables.{% else %}tables.{% endif %} Grammar:
+
```yql
ALTER TABLE `<table_name>`
@@ -17,14 +18,16 @@ ALTER TABLE `<table_name>`
[, ...]
```
+
{% include [index_grammar_explanation.md](../_includes/index_grammar_explanation.md) %}
Parameters for all index types:
-* `parallel` - maximum number of parallel [partition](../../../../concepts/glossary.md#partition)-based workers used during index build (an integer between `1` and `MaxBuildIndexShardsInFlight` from `SchemeShardConfig`).
- - If not specified, currently defaults to `32` or `MaxBuildIndexShardsInFlight` if it's lower. Default `MaxBuildIndexShardsInFlight` is `1000`. Default parallelism selection logic may be changed in future versions.
- - You may set a smaller limit to reduce the impact of index build on the DB performance.
- - You may also set a larger limit to speed up the index build if you have enough hardware resources.
+* - maximum number of `parallel` handlers based on [partitions](../../../../concepts/glossary.md#partition) involved in index building (an integer between `1` and `MaxBuildIndexShardsInFlight` from `SchemeShardConfig`).
+
+- If the parameter is not specified, the default value `32` or `MaxBuildIndexShardsInFlight` is currently used, whichever is smaller. `MaxBuildIndexShardsInFlight` defaults to `1000`. In future versions, the default parallelism selection logic may change.
+- You can set a lower limit to reduce the impact of index building on database performance.
+- You can also set a higher limit to speed up index building if you have sufficient hardware resources.
Parameters specific to vector indexes:
@@ -32,29 +35,29 @@ Parameters specific to vector indexes:
{% note info %}
-For vector indexes, the vector_type and vector_dimension parameters can be omitted if the table is not empty — they are determined automatically based on the row contents. The levels and clusters parameters are also determined automatically, and the table may be empty for them, but this is highly unrecommended because the default values in that case are levels=1, clusters=2. It is far better to create the index on a table that already has data loaded, so that the values can be determined correctly.
+For vector indexes, the `vector_type` and `vector_dimension` parameters can be omitted if the table is not empty — they are determined automatically from the row contents. The `levels` and `clusters` parameters are also determined automatically, and for them the table can be empty, but doing so is strongly not recommended because the default values in this case are `levels`=1, `clusters`=2; it is much better to create the index on a table that already has data loaded, so that the values can be correctly determined.
{% endnote %}
-Parameters specific to fulltext indexes:
+Parameters specific to full-text indexes:
{% include [fulltext_index_parameters.md](../_includes/fulltext_index_parameters.md) %}
-### Local Bloom skip index parameters {#local-bloom}
+### Parameters of local bloom indexes {#local-bloom}
{% include [bloom_skip_index_parameters.md](../_includes/bloom_skip_index_parameters.md) %}
-{% if backend_name == "YDB" %}
+{% if backend_name == "YDB" and oss == true %}
-You can also add a secondary index using the {{ ydb-short-name }} CLI [table index](../../../../reference/ydb-cli/commands/secondary_index.md#add) command.
+You can also add a secondary index using the [table index](../../../../reference/ydb-cli/commands/secondary_index.md#add) {{ ydb-short-name }} CLI command.
{% endif %}
### Limitations
-The `ADD INDEX` operation for creating global secondary (`GLOBAL`, `UNIQUE`, and so on) and vector indexes is supported only for row-oriented tables. For [column-oriented tables](../../../../concepts/datamodel/table.md#column-oriented-tables), `ADD INDEX` [supports only local Bloom skip indexes](#local-bloom).
+The `ADD INDEX` operation for creating global secondary (`GLOBAL`, `UNIQUE`, etc.) and vector indexes is supported only for row tables. For [columnar tables](../../../../concepts/datamodel/table.md#column-oriented-tables), via `ADD INDEX`, [only local bloom indexes are supported](#local-bloom).
-Local Bloom skip index behavior:
+Features of local bloom indexes:
{% include [bloom_skip_index_features.md](../_includes/bloom_skip_index_features.md) %}
@@ -66,7 +69,8 @@ Local Bloom skip index behavior:
### Examples
-A regular secondary index:
+Secondary index:
+
```yql
ALTER TABLE `series`
@@ -74,8 +78,10 @@ ALTER TABLE `series`
GLOBAL ON (`title`);
```
+
[Vector index](../../../../dev/vector-indexes.md):
+
```yql
ALTER TABLE `series`
ADD INDEX emb_cosine_idx GLOBAL SYNC USING vector_kmeans_tree
@@ -85,7 +91,9 @@ ALTER TABLE `series`
);
```
-A fulltext index:
+
+Full-text index:
+
```yql
ALTER TABLE `series`
@@ -94,7 +102,19 @@ ALTER TABLE `series`
WITH (tokenizer=standard, use_filter_lowercase=true);
```
-A bloom index:
+
+[JSON index](../../../../dev/json-indexes.md):
+
+
+```yql
+ALTER TABLE `series`
+ ADD INDEX json_idx GLOBAL USING json
+ ON (metadata);
+```
+
+
+[Bloom index](../../../../dev/bloom-skip-indexes.md):
+
```yql
ALTER TABLE `/Root/Table`
@@ -103,7 +123,9 @@ ALTER TABLE `/Root/Table`
WITH (false_positive_probability = 0.01);
```
-A bloom ngram index:
+
+Bloom n-gram index:
+
```yql
ALTER TABLE `/Root/Table`
@@ -114,55 +136,58 @@ ALTER TABLE `/Root/Table`
false_positive_probability = 0.01,
case_sensitive = true
);
-```
-## Altering an index {#alter-index}
+# # Changing index parameters {#alter-index}
-Indexes have type-specific parameters that can be tuned. Global indexes, whether [synchronous]({{ concept_secondary_index }}#sync) or [asynchronous]({{ concept_secondary_index }}#async), are implemented as hidden tables, and their automatic partitioning and followers settings can be adjusted just like those of regular tables.
+Индексы имеют параметры, зависящие от типа, которые можно настраивать. Глобальные индексы, [синхронные](yfmvar-0-yfmvarend#sync) или [асинхронные](yfmvar-1-yfmvarend#async), реализованы в виде скрытых таблиц, и их параметры автоматического партиционирования и реплик можно регулировать так же, как и настройки обычных таблиц.
{% note info %}
-Currently, specifying secondary index partitioning settings during index creation is not supported in either the [`ALTER TABLE ADD INDEX`](#add-index) or the [`CREATE TABLE INDEX`](../create_table/secondary_index.md) statements.
+В настоящее время задание настроек партиционирования вторичных индексов при создании индекса не поддерживается ни в операторе [`ALTER TABLE ADD INDEX`](#add-index), ни в операторе [`CREATE TABLE INDEX`](../create_table/secondary_index.md).
{% endnote %}
-```sql
+```yql
ALTER TABLE <table_name> ALTER INDEX <index_name> SET <setting_name> <value>;
ALTER TABLE <table_name> ALTER INDEX <index_name> SET (<setting_name_1> = <value_1>, ...);
```
-* `<table_name>`: The name of the table whose index is to be modified.
-* `<index_name>`: The name of the index to be modified.
-* `<setting_name>`: The name of the setting to be modified. Allowed settings depend on the index type:
- * for global secondary indexes:
- * [AUTO_PARTITIONING_BY_SIZE]({{ concept_table }}#auto_partitioning_by_size)
- * [AUTO_PARTITIONING_BY_LOAD]({{ concept_table }}#auto_partitioning_by_load)
- * [AUTO_PARTITIONING_PARTITION_SIZE_MB]({{ concept_table }}#auto_partitioning_partition_size_mb)
- * [AUTO_PARTITIONING_MIN_PARTITIONS_COUNT]({{ concept_table }}#auto_partitioning_min_partitions_count)
- * [AUTO_PARTITIONING_MAX_PARTITIONS_COUNT]({{ concept_table }}#auto_partitioning_max_partitions_count)
- * [READ_REPLICAS_SETTINGS]({{ concept_table }}#read_only_replicas)
- * for local Bloom skip indexes (see [Local Bloom skip index parameters](#local-bloom)):
- * `FALSE_POSITIVE_PROBABILITY`
- * `NGRAM_SIZE` and `CASE_SENSITIVE` (for `bloom_ngram_filter` only)
+* `<table_name>` - name of the table whose index needs to be changed.
+* `<index_name>` - name of the index to change.
+* `<setting_name>` - name of the parameter to change. The set of allowed parameters depends on the index type:
+
+ * for global secondary indexes:
+
+ * [AUTO_PARTITIONING_BY_SIZE]({{ concept_table }}#auto_partitioning_by_size)
+ * [AUTO_PARTITIONING_BY_LOAD]({{ concept_table }}#auto_partitioning_by_load)
+ * [AUTO_PARTITIONING_PARTITION_SIZE_MB]({{ concept_table }}#auto_partitioning_partition_size_mb)
+ * [AUTO_PARTITIONING_MIN_PARTITIONS_COUNT]({{ concept_table }}#auto_partitioning_min_partitions_count)
+ * [AUTO_PARTITIONING_MAX_PARTITIONS_COUNT]({{ concept_table }}#auto_partitioning_max_partitions_count)
+ * [READ_REPLICAS_SETTINGS]({{ concept_table }}#read_only_replicas)
+ * for local bloom indexes (see [Parameters of local bloom indexes](#local-bloom)):
+
+ * `FALSE_POSITIVE_PROBABILITY`
+ * `NGRAM_SIZE` and `CASE_SENSITIVE` (only for `bloom_ngram_filter`)
{% note info %}
-`RESET` is not supported for `ALTER INDEX`.
+The `RESET` operation for `ALTER INDEX` is not supported.
{% endnote %}
-* `<value>`: The new value for the setting. Possible values include:
- * `ENABLED` or `DISABLED` for the `AUTO_PARTITIONING_BY_SIZE` and `AUTO_PARTITIONING_BY_LOAD` settings
- * `"PER_AZ:<count>"` or `"ANY_AZ:<count>"` where `<count>` is the number of replicas for the `READ_REPLICAS_SETTINGS`
- * An integer of `Uint64` type for the other settings
- * A floating-point value in `(0, 1)` for `FALSE_POSITIVE_PROBABILITY`; smaller values usually reduce false positives but increase index size
- * An integer value from `3` to `8` for `NGRAM_SIZE` (a typical starting point is `3`)
- * `true` or `false` for `CASE_SENSITIVE`
+* `<value>` - new parameter value. Possible values include:
+
+ * `ENABLED` or `DISABLED` for the `AUTO_PARTITIONING_BY_SIZE` and `AUTO_PARTITIONING_BY_LOAD` parameters
+ * `"PER_AZ:<count>"` or `"ANY_AZ:<count>"` where `<count>` is the number of replicas for `READ_REPLICAS_SETTINGS`
+ * for other parameters — an integer of type `Uint64`
+ * for `FALSE_POSITIVE_PROBABILITY` — a floating-point number in the range `(0, 1)`; a smaller value usually reduces the number of false positives but increases the index size
+ * for `NGRAM_SIZE` — an integer in the range from `3` to `8` (usually recommended to start with `3`)
+ * for `CASE_SENSITIVE` — `true` or `false`
### Example
-The query in the following example enables automatic partitioning by load for the index named `title_index` of the table `series`, sets its minimum partition count to 5, and enables one follower per AZ for every partition:
+The code in the following example enables automatic partitioning by load for the index named `title_index` in the table `series`, sets the minimum number of partitions to 5, and starts one replica in each availability zone (AZ) for each partition:
```yql
@@ -173,7 +198,9 @@ ALTER TABLE `series` ALTER INDEX `title_index` SET (
);
```
-For local Bloom skip indexes, you can also alter index-specific parameters, for example:
+
+For local bloom indexes, you can also change their specific parameters, for example:
+
```yql
ALTER TABLE `/Root/Table` ALTER INDEX idx_ngram SET (
@@ -183,35 +210,37 @@ ALTER TABLE `/Root/Table` ALTER INDEX idx_ngram SET (
);
```
+
## Deleting an index {#drop-index}
-`DROP INDEX`: Deletes the index with the specified name. The code below deletes the index named `title_index`.
+`DROP INDEX` — deletes the index with the specified name. The code below will delete the index named `title_index`.
+
```yql
ALTER TABLE `series` DROP INDEX `title_index`;
```
-{% if backend_name == "YDB" %}
-You can also remove a index using the {{ ydb-short-name }} CLI [table index](../../../../reference/ydb-cli/commands/secondary_index.md#drop) command.
+{% if backend_name == "YDB" and oss == true %}
-{% endif %}
+You can also delete an index using the [table index](../../../../reference/ydb-cli/commands/secondary_index.md#drop) {{ ydb-short-name }} CLI command.
-## Renaming an index {#rename-index}
+{% endif %}
-`RENAME INDEX`: Renames the index with the specified name.
+## Renaming a secondary index {#rename-secondary-index}
-If an index with the new name exists, an error is returned.
+`RENAME INDEX` — renames the index with the specified name. If an index with the new name already exists, an error will be returned.
-{% if backend_name == "YDB" %}
+{% if backend_name == "YDB" and oss == true %}
-Atomically replacing an index under load is supported by the [{{ ydb-cli }} table index rename](../../../../reference/ydb-cli/commands/secondary_index.md#rename) command in the {{ ydb-short-name }} CLI and by {{ ydb-short-name }} SDK methods.
+The ability to atomically replace an index under load is supported by the [{{ ydb-cli }} table index rename](../../../../reference/ydb-cli/commands/secondary_index.md#rename) {{ ydb-short-name }} CLI command and specialized {{ ydb-short-name }} SDK methods.
-This applies to global secondary indexes (the hidden index table and the `--replace` mode). Local Bloom skip indexes are not covered by this atomic under-load replacement flow.
+This applies to global secondary indexes (hidden index table and `--replace` mode). Local bloom indexes are not applicable to such atomic replacement under load.
{% endif %}
-Example of index renaming:
+Example of renaming an index:
+
```yql
ALTER TABLE `series` RENAME INDEX `title_index` TO `title_index_new`;
diff --git a/ydb/docs/en/core/yql/reference/syntax/create-resource-pool-classifier.md b/ydb/docs/en/core/yql/reference/syntax/create-resource-pool-classifier.md
index 25d76d45304..eb4a7c06ebc 100644
--- a/ydb/docs/en/core/yql/reference/syntax/create-resource-pool-classifier.md
+++ b/ydb/docs/en/core/yql/reference/syntax/create-resource-pool-classifier.md
@@ -1,43 +1,59 @@
# CREATE RESOURCE POOL CLASSIFIER
-`CREATE RESOURCE POOL CLASSIFIER` creates a [resource pool classifier](../../../concepts/glossary.md#resource-pool-classifier).
+`CREATE RESOURCE POOL CLASSIFIER` creates a [resource pool classifier](../../../concepts/glossary.md#resource-pool-classifier.md).
## Syntax
+
```yql
CREATE RESOURCE POOL CLASSIFIER <name>
WITH ( <parameter_name> [= <parameter_value>] [, ... ] )
```
-- `name` — name of the resource pool classifier to create. Must be unique and must not contain characters forbidden for schema objects.
-- `WITH ( <parameter_name> [= <parameter_value>] [, ... ] )` — parameters that define classifier behavior.
+
+- `name` — the name of the resource pool classifier being created. It must be unique. The name must not contain characters prohibited for schema objects.
+- `WITH ( <parameter_name> [= <parameter_value>] [, ... ] )` — allows you to set parameter values that define the behavior of the resource pool classifier.
### Parameters
-* `RANK` (Int64) — Optional: order in which classifiers are evaluated. If omitted, the maximum existing `RANK` plus 1000 is used. Allowed values: a unique number in $[0, 2^{63}-1]$.
-* `RESOURCE_POOL` (String) — Required: name of the resource pool for queries that match the classifier.
-* `MEMBER_NAME` (String) — Optional: user or group routed to that pool. If omitted, the classifier ignores `MEMBER_NAME` and uses other criteria.
+* `RANK` (Int64) — an optional field that specifies the selection order of the resource pool classifier. If the value is not specified, the maximum existing `RANK` is taken and 1000 is added to it. Valid values: a unique number in the range $[0, 2^{63}-1]$.
+* `RESOURCE_POOL` (String) — a required field that specifies the name of the resource pool to which queries that meet the classifier criteria will be sent.
+* `MEMBER_NAME` (String) — an optional field that determines which user or group of users will be sent to the specified resource pool. The value is compared with the user's SID or any group SID from their authentication token; see [below](#member-name-format) for details on the format. If the field is not specified, the classifier ignores `MEMBER_NAME`, and classification is performed based on other criteria.
+
+### MEMBER_NAME format {#member-name-format}
+
+`MEMBER_NAME` is compared character by character with the user's [SID](../../../concepts/glossary.md#access-sid) or any group SID from their authentication token. The SID format depends on how the user logged into the system.
+
+- **Built-in users {{ ydb-short-name }} (login/password)** — the SID matches the username, without a suffix. For example, `user1`. For more information, see [{#T}](../../../security/authentication.md#static-credentials).
+- **Cloud users (Access Service)** — the SID has the form `<subject_id>@as`, where `<subject_id>` is the user ID in IAM. The suffix is set by the [`access_service_domain`](../../../reference/configuration/auth_config.md#iam-auth-config) parameter (default `as`). For example, `ajeb89hv69nujke769fa@as`. For more information, see [{#T}](../../../security/authentication.md#iam).
+- **LDAP** — the SID has the form `<login>@<domain>`, where the domain is set by the [`ldap_authentication_domain`](../../../reference/configuration/auth_config.md#ldap-auth-config) parameter (default `ldap`). For example, `user1@ldap`. For more information, see [{#T}](../../../security/authentication.md#ldap).
+- **External identity providers (OIDC)** — the SID has the form `<login>@<domain>`, where the domain is set by the `external_idp_authentication_domain` parameter in the [authentication configuration](../../../reference/configuration/auth_config.md) (default `sso`). For example, `user1@sso`.
+
+You can specify either the SID of a specific user or the SID of a group. The group `all-users@well-known` is automatically added to all authenticated users — it is convenient to use if you need to direct queries from all authenticated clients to the pool.
## Notes {#remarks}
-If `RANK` is omitted in the DDL, the default is $RANK = MAX(existing\_ranks) + 1000$. All `RANK` values must be unique so pool choice is deterministic when rules conflict. This allows inserting new classifiers between existing ones.
+If `RANK` is not specified in the DDL for creating a resource pool classifier, it will be assigned the default value $RANK = MAX(existing_ranks) + 1000$. All `RANK` values must be unique to ensure a strictly deterministic order of resource pool selection in case of conflicting conditions. This behavior is chosen to allow adding new resource pool classifiers between existing ones.
-A classifier may reference a non-existent pool or a pool the user cannot access; such classifiers are skipped.
+It is also possible to have a classifier that references a non-existent resource pool or one to which the user does not have access. In such a case, such classifiers will be skipped.
-Classifier count limits are described on the [limits](../../../concepts/limits-ydb.md#resource_pool) page.
+For limitations on the number of classifiers, see the [limitations](../../../../concepts/limits-ydb#resource_pool) page.
## Permissions
-The `ALL` [permission](grant.md#permissions-list) on the database is required.
+The [permission](./grant.md#permissions-list) `ALL` on the database is required.
+
+Example of granting such a permission:
-Example:
```yql
GRANT 'ALL' ON `/my_db` TO `user1@domain`;
```
+
## Examples {#examples}
+
```yql
CREATE RESOURCE POOL CLASSIFIER olap_classifier WITH (
RANK=1000,
@@ -46,7 +62,8 @@ CREATE RESOURCE POOL CLASSIFIER olap_classifier WITH (
)
```
-The example above creates a resource pool classifier named `olap_classifier` that routes queries from user `user1@domain` to the resource pool named `olap`. Queries from all other users go to the `default` resource pool, assuming no other resource pool classifiers exist.
+
+In the example above, a resource pool classifier named `olap_classifier` is created, which directs queries from user `user1@domain` to a resource pool named `olap`. Queries from all other users will be sent to the resource pool `default`, provided that no other resource pool classifiers exist.
## See also
diff --git a/ydb/docs/en/core/yql/reference/syntax/create_table/index.md b/ydb/docs/en/core/yql/reference/syntax/create_table/index.md
index 6f77c0b80a9..cd5a9ee304d 100644
--- a/ydb/docs/en/core/yql/reference/syntax/create_table/index.md
+++ b/ydb/docs/en/core/yql/reference/syntax/create_table/index.md
@@ -2,16 +2,17 @@
{% if feature_bulk_tables %}
-The table is automatically created upon the first [INSERT INTO](../insert_into.md){% if feature_mapreduce %} in the database specified by the [USE](../use.md) operator{% endif %}. The schema is defined automatically in this process.
+The table is created automatically on the first [INSERT INTO](../insert_into.md){% if feature_mapreduce %}, in the database specified by the [USE](../use.md) statement{% endif %}. The schema is determined automatically.
{% else %}
-The invocation of `CREATE TABLE` creates {% if concept_table %}a [table]({{ concept_table }}){% else %}a table{% endif %} with the specified data schema{% if feature_map_tables %} and primary key columns (`PRIMARY KEY`){% endif %}.{% if feature_secondary_index == true %} It also allows defining secondary indexes on the created table.
+The `CREATE TABLE` call creates {% if concept_table %} [a table]({{ concept_table }}){% else %}a table{% endif %} with the specified data schema{% if feature_map_tables %} and key columns (`PRIMARY KEY`){% endif %}.{% if feature_secondary_index == true %} It allows defining secondary indexes on the created table.
{% endif %}
{% endif %}
+
```yql
CREATE TABLE [IF NOT EXISTS] <table_name> (
[<column_name> <column_data_type>] [FAMILY <family_name>] [NULL | NOT NULL] [DEFAULT <default_value>]
@@ -36,90 +37,100 @@ CREATE TABLE [IF NOT EXISTS] <table_name> (
[AS SELECT ...]
```
+
{% if oss == true and backend_name == "YDB" %}
-## Request parameters {#request-parameters}
+## Query parameters
### table_name
-The path of the table to be created.
+Path of the table being created.
-When choosing a name for the table, consider the common [schema object naming rules](../../../../concepts/datamodel/cluster-namespace.md#object-naming-rules).
+When choosing a table name, follow the general [naming rules for schema objects](../../../../concepts/datamodel/cluster-namespace.md#object-naming-rules).
### IF NOT EXISTS
-If the table with the specified name already exists, the execution of the operator is completely skipped — no checks or schema matching is performed, and no error occurs. Note that the existing table may differ in structure from the one you would like to create with this query — no comparison or equivalence check is performed.
+If a table with the specified name already exists, the statement execution is completely skipped — no checks or schema comparison are performed, and no error occurs. Note that the existing table may differ in structure from the one you intended to create with this query — no comparison or equivalence check is performed.
### column_name
-The name of the column to be created in the new table.
+Name of the column being created in the new table.
-When choosing a name for the column, consider the common [column naming rules](../../../../concepts/datamodel/table.md#column-naming-rules).
+When choosing a column name, follow the general [column naming rules](../../../../concepts/datamodel/table.md#column-naming-rules).
### column_data_type
-The data type of the column. The complete list of data types supported by {{ ydb-short-name }} is available in the [{#T}](../../types/index.md) section.
+Data type of the column. The full list of data types supported by {{ ydb-short-name }} is available in the [{#T}](../../types/index.md) section.
{% include [column_option_list.md](../_includes/column_option_list.md) %}
### INDEX
-Definition of an index on the table. [Secondary indexes](secondary_index.md), [vector indexes](vector_index.md), [fulltext indexes](fulltext_index.md), and [Bloom skip indexes](bloom_skip_index.md) are supported.
+Defining an index on the table. Supported types:
+
+* [secondary indexes](secondary_index.md),
+* [vector indexes](vector_index.md),
+* [full-text indexes](fulltext_index.md),
+* [Bloom indexes](bloom_skip_index.md),
+* [JSON indexes](json_index.md).
### PRIMARY KEY
-Definition of the primary key of the table. Specifies the columns that make up the primary key in the order of enumeration. For more information on selecting a primary key, see the [{#T}](../../../../dev/primary-key/index.md) article.
+Defining the table's primary key. Specifies the columns that make up the primary key in the order listed. For more details on choosing a primary key, see the [{#T}](../../../../dev/primary-key/index.md) section.
### PARTITION BY HASH
-Definition of the columns on which partitioning will occur for **column-oriented** tables. Specifies the columns on which [partitioning](../../../../concepts/glossary.md#partition) will occur using the hash function. The columns must be part of the primary key. The columns do not necessarily have to be a prefix or suffix — the requirement is to be part of the primary key.
+Defining partitioning keys for **column-oriented** tables. Specifies the columns by whose hash the data [partitioning](../../../../concepts/glossary.md#partition) is performed. The columns must be part of the primary key. However, the columns do not necessarily have to be a prefix or suffix — the requirement is to be part of the primary key.
-If the parameter is not specified, the table will be partitioned on the same columns as those included in the primary key. For more information on selecting and working with partition keys in column-oriented tables, see the [{#T}](../../../../dev/primary-key/column-oriented.md) article.
+If the parameter is not specified, the table will be split into partitions by the same columns that are part of the primary key. For guidance on how to choose partitioning keys for column-oriented tables, see the article [{#T}](../../../../dev/primary-key/column-oriented.md).
-For more information on partitioning column-oriented tables, see the [{#T}](../../../../concepts/datamodel/table.md#olap-tables-partitioning) section.
+For more details on partitioning column-oriented tables, see the [{#T}](../../../../concepts/datamodel/table.md#olap-tables-partitioning) section.
-### FAMILY <column_family> (column group setting)
+### FAMILY <column_family> (column group settings)
-Definition of a column group with specified parameters. For more information, see the [{#T}](family.md) section.
+Defining a column group with specified parameters. For more details, see the [{#T}](family.md) section.
### WITH
-Additional parameters for creating a table. For more information, see the [{#T}](with.md) section.
+Additional table creation parameters. For more details, see the [{#T}](with.md) section.
{% note info %}
{{ ydb-short-name }} supports two types of tables:
-* [Row-oriented](../../../../concepts/datamodel/table.md#row-oriented-tables) tables.
-* [Column-oriented](../../../../concepts/datamodel/table.md#column-oriented-tables) tables.
+* [Row-oriented](../../../../concepts/datamodel/table.md#row-oriented-tables).
+* [Column-oriented](../../../../concepts/datamodel/table.md#column-oriented-tables).
+
+The table type when created is specified by the `STORE` parameter in the `WITH` block, where `ROW` means [row-oriented table](../../../../concepts/datamodel/table.md#row-oriented-tables) and `COLUMN` means [column-oriented table](../../../../concepts/datamodel/table.md#column-oriented-tables):
-The table type is specified by the `STORE` parameter in the `WITH` clause, where `ROW` indicates a [row-oriented](../../../../concepts/datamodel/table.md#row-oriented-tables) table and `COLUMN` indicates a [column-oriented](../../../../concepts/datamodel/table.md#column-oriented-tables) table:
```yql
CREATE <table_name> (
columns
...
)
+
WITH (
STORE = COLUMN -- Default value ROW
)
```
+
By default, if the `STORE` parameter is not specified, a row-oriented table is created.
{% endnote %}
{% note info %}
-When choosing a name for the table, consider the common [schema object naming rules](../../../../concepts/datamodel/cluster-namespace.md#object-naming-rules).
+When choosing a table name, follow the general [naming rules for schema objects](../../../../concepts/datamodel/cluster-namespace.md#object-naming-rules).
{% endnote %}
### AS SELECT
-Creating and filling a table with data from a `SELECT` query. For more information, see the [{#T}](as_select.md) section.
+Creating and populating a table based on the results of the `SELECT` query. For more details, see the [{#T}](as_select.md) section.
-## Examples of table creation
+## Table creation examples
{% list tabs %}
@@ -135,7 +146,7 @@ Creating and filling a table with data from a `SELECT` query. For more informati
d "List<List<Int32>>"
PRIMARY KEY (a, b)
);
- ```
+ ```
{% else %}
@@ -146,11 +157,12 @@ Creating and filling a table with data from a `SELECT` query. For more informati
c Float,
PRIMARY KEY (a, b)
);
- ```
+ ```
{% endif %}
- Example of creating a table with a DEFAULT value:
+ Example of creating a table using a default value (DEFAULT):
+
```yql
CREATE TABLE table_with_default (
@@ -161,15 +173,16 @@ Creating and filling a table with data from a `SELECT` query. For more informati
);
```
+
{% if feature_column_container_type == true %}
- For non-key columns, any data types are allowed{% if feature_serial %}, except [serial](../../types/serial.md) types{% endif %}, whereas for key columns only [primitive](../../types/primitive.md) types{% if feature_serial %} and [serial](../../types/serial.md) types{% endif %} are permitted. When specifying complex types (for example, `List<String>`), the type should be enclosed in double quotes.
+ For non-key columns, any data types are allowed{% if feature_serial %}, except [serial](../../types/serial.md) {% endif %}; for key columns, only [primitive](../../types/primitive.md){% if feature_serial %} and [serial](../../types/serial.md){% endif %} types are allowed. When specifying complex types (e.g., `List<String>`), the type is enclosed in double quotes.
{% else %}
{% if feature_serial %}
- For key columns, only [primitive](../../types/primitive.md) and [serial](../../types/serial.md) data types are allowed; for non-key columns, only [primitive](../../types/primitive.md) types are allowed.
+ For key columns, only [primitive](../../types/primitive.md) and [serial](../../types/serial.md) data types are allowed; for non-key columns, only [primitive](../../types/primitive.md) data types are allowed.
{% else %}
@@ -181,17 +194,17 @@ Creating and filling a table with data from a `SELECT` query. For more informati
{% if feature_not_null == true %}
- Without additional modifiers, a column acquires an [optional](../../types/optional.md) type and allows `NULL` values. To designate a non-optional type, use the `NOT NULL` constraint.
+ Without additional modifiers, the column acquires an [optional type](../../types/optional.md) and allows `NULL` to be written as values. To obtain a non-optional type, use `NOT NULL`.
{% else %}
{% if feature_not_null_for_pk %}
- By default, all columns are [optional](../../types/optional.md) and can have `NULL` values. The `NOT NULL` constraint can only be specified for columns that are part of the primary key.
+ By default, all columns are [optional](../../types/optional.md) and can have a NULL value. The `NOT NULL` constraint can only be specified for columns that are part of the primary key.
{% else %}
- All columns allow NULL values, meaning they are [optional](../../types/optional.md).
+ All columns allow `NULL` as a value, meaning they are [optional](../../types/optional.md).
{% endif %}
@@ -199,11 +212,12 @@ Creating and filling a table with data from a `SELECT` query. For more informati
{% if feature_map_tables %}
- Specifying a `PRIMARY KEY` with a non-empty list of columns is mandatory. These columns become part of the key in the order they are listed.
+ It is mandatory to specify `PRIMARY KEY` with a non-empty list of columns. These columns become part of the key in the order they are listed.
{% endif %}
- Example of creating a row-oriented table using partitioning options:
+ Example of creating a row table with partitioning options:
+
```yql
CREATE TABLE <table_name> (
@@ -218,9 +232,10 @@ Creating and filling a table with data from a `SELECT` query. For more informati
);
```
- Such code will create a row-oriented table with automatic partitioning by partition size (`AUTO_PARTITIONING_BY_SIZE`) enabled, and with the preferred size of each partition (`AUTO_PARTITIONING_PARTITION_SIZE_MB`) set to 512 megabytes. The full list of row-oriented table partitioning options can be found in the [Partitioning Row-Oriented Tables](../../../../concepts/datamodel/table.md#partitioning_row_table) section of the [{#T}](../../../../concepts/datamodel/table.md) article.
-- Creating a column-oriented table
+ This code will create a row table with automatic partitioning enabled by partition size (`AUTO_PARTITIONING_BY_SIZE`) and a preferred partition size (`AUTO_PARTITIONING_PARTITION_SIZE_MB`) of 512 megabytes. The full list of row table partitioning options is in the [Row table partitioning](../../../../concepts/datamodel/table.md#partitioning_row_table) section of the [{#T}](../../../../concepts/datamodel/table.md) article.
+
+- Creating a column table
```yql
CREATE TABLE table_name (
@@ -235,9 +250,11 @@ Creating and filling a table with data from a `SELECT` query. For more informati
);
```
- For column-oriented tables, you can explicitly specify the columns on which partitioning will occur using the `PARTITION BY HASH` construct. Usually, these are columns of the primary key with a large number of unique values, such as `Timestamp`. If `PARTITION BY HASH` is not specified, partitioning will occur automatically on all columns included in the primary key. For more information on selecting and working with partition keys in column-oriented tables, see the [{#T}](../../../../dev/primary-key/column-oriented.md) article.
- Column-oriented tables do not currently support automatic repartitioning, so it is important to specify the correct number of partitions when creating a table with the `AUTO_PARTITIONING_MIN_PARTITIONS_COUNT` parameter:
+ For column tables, you can explicitly specify which columns will be used for partitioning using the `PARTITION BY HASH` construct. Typically, primary key columns with a large number of unique values are chosen for this, for example, `Timestamp`. If `PARTITION BY HASH` is not specified, partitioning will occur automatically across all columns that are part of the primary key. For more details on selecting and working with partitioning keys in column tables, see the [{#T}](../../../../dev/primary-key/column-oriented.md) article.
+
+ Currently, column tables do not support automatic repartitioning, so it is important to specify the correct number of partitions when creating a table using the `AUTO_PARTITIONING_MIN_PARTITIONS_COUNT` parameter:
+
```yql
CREATE TABLE table_name (
@@ -253,8 +270,8 @@ Creating and filling a table with data from a `SELECT` query. For more informati
);
```
- This code will create a column-oriented table with 10 partitions. The full list of column-oriented table partitioning options can be found in the [{#T}](../../../../concepts/datamodel/table.md#olap-tables-partitioning) section of the [{#T}](../../../../concepts/datamodel/table.md) article.
+ This code will create a column table with 10 partitions. For a full list of column table partitioning options, see the [{#T}](../../../../concepts/datamodel/table.md#olap-tables-partitioning) section of the [{#T}](../../../../concepts/datamodel/table.md) article.
{% endlist %}
@@ -262,27 +279,27 @@ Creating and filling a table with data from a `SELECT` query. For more informati
{% if feature_column_container_type == true %}
-For non-key columns, any data types are allowed, whereas for key columns only [primitive](../../types/primitive.md) types are permitted. When specifying complex types (for example, `List<String>`), the type should be enclosed in double quotes.
+For non-key columns, any data types are allowed; for key columns, only [primitive](../../types/primitive.md) types are allowed. When specifying complex types (e.g., `List<String>`), the type is enclosed in double quotes.
{% else %}
-For both key and non-key columns, only [primitive](../../types/primitive.md) data types are allowed.
+Only [primitive](../../types/primitive.md) data types are allowed for both key and non-key columns.
{% endif %}
{% if feature_not_null == true %}
-Without additional modifiers, a column acquires an [optional](../../types/optional.md) type and allows `NULL` values. To designate a non-optional type, use the `NOT NULL` constraint.
+Without additional modifiers, a column acquires an [optional type](../../types/optional.md) and allows `NULL` as a value. To get a non-optional type, use `NOT NULL`.
{% else %}
{% if feature_not_null_for_pk %}
-By default, all columns are [optional](../../types/optional.md) and can have `NULL` values. The `NOT NULL` constraint can only be specified for columns that are part of the primary key.
+By default, all columns are [optional](../../types/optional.md) and can have a NULL value. The `NOT NULL` constraint can only be specified for columns that are part of the primary key.
{% else %}
-All columns allow NULL values, meaning they are [optional](../../types/optional.md).
+All columns allow `NULL` as a value, meaning they are [optional](../../types/optional.md).
{% endif %}
@@ -290,12 +307,13 @@ All columns allow NULL values, meaning they are [optional](../../types/optional.
{% if feature_map_tables %}
-Specifying a `PRIMARY KEY` with a non-empty list of columns is mandatory. These columns become part of the key in the order they are listed.
+It is mandatory to specify `PRIMARY KEY` with a non-empty list of columns. These columns become part of the key in the order they are listed.
{% endif %}
Example:
+
```yql
CREATE TABLE <table_name> (
a Uint64,
@@ -309,21 +327,22 @@ CREATE TABLE <table_name> (
{% if backend_name == "YDB" and oss == true %}
-When creating row-oriented tables, it is possible to specify:
+When creating row tables, you can specify:
-* [A secondary index](secondary_index.md).
-* [A vector index](vector_index.md).
-* [A fulltext index](fulltext_index.md).
-* [A Bloom skip index](bloom_skip_index.md).
+* [Secondary index](secondary_index.md).
+* [Vector index](vector_index.md).
+* [Full-text index](fulltext_index.md).
+* [JSON index](json_index.md).
+* [Bloom index](bloom_skip_index.md).
* [Column groups](family.md).
* [Additional parameters](with.md).
-* [Creating a table filled with query results](as_select.md).
+* [Creating and populating a table based on query results](as_select.md).
-When creating column-oriented tables, it is possible to specify:
+For column tables, when creating them, you can specify:
-* [A Bloom skip index](bloom_skip_index.md).
+* [Bloom index](bloom_skip_index.md).
* [Column groups](family.md).
* [Additional parameters](with.md).
-* [Creating a table filled with query results](as_select.md).
+* [Creating and populating a table based on query results](as_select.md).
{% endif %}
diff --git a/ydb/docs/en/core/yql/reference/syntax/create_table/json_index.md b/ydb/docs/en/core/yql/reference/syntax/create_table/json_index.md
new file mode 100644
index 00000000000..d53b623719a
--- /dev/null
+++ b/ydb/docs/en/core/yql/reference/syntax/create_table/json_index.md
@@ -0,0 +1,42 @@
+# JSON-index
+
+{% if backend_name == 'YDB' %} [JSON indexes](../../../../dev/json-indexes.md){% else %}JSON indexes{% endif %} in {% if backend_name == 'YDB' %}[row](../../../../concepts/datamodel/table.md#row-oriented-tables){% else %}row{% endif %} tables are created using the same syntax as [secondary indexes](secondary_index.md), by specifying `json` as the index type. A subset of the syntax available for JSON indexes:
+
+
+```yql
+CREATE TABLE `<table_name>` (
+ ...
+ INDEX `<index_name>`
+ GLOBAL
+ [SYNC]
+ USING json
+ ON ( <json_column> )
+ [, ...]
+)
+```
+
+
+Where:
+
+* `<index_name>` — unique index name for data access.
+* `SYNC` — specifies synchronous index update. This is the only mode available for JSON indexes; explicit specification is not required.
+* `<json_column>` — table column of type `Json` or `JsonDocument`. A JSON index is built on a single column only.
+
+A JSON index does not support the `COVER` expression — attempting to specify it will result in an error.
+
+{% include [not_allow_for_olap](../../../../_includes/not_allow_for_olap_note.md) %}
+
+## Example
+
+
+```yql
+CREATE TABLE documents (
+ id Uint64 NOT NULL,
+ payload JsonDocument NOT NULL,
+ INDEX json_idx GLOBAL USING json ON (payload),
+ PRIMARY KEY (id)
+)
+```
+
+
+In this example, a table `documents` is created with a JSON index `json_idx` on column `payload`. The index will be used by queries whose `WHERE` predicate contains calls to `JSON_EXISTS` or `JSON_VALUE` on column `payload`.
diff --git a/ydb/docs/en/core/yql/reference/syntax/create_table/toc_i.yaml b/ydb/docs/en/core/yql/reference/syntax/create_table/toc_i.yaml
index d51714fb9d8..6644cea4776 100644
--- a/ydb/docs/en/core/yql/reference/syntax/create_table/toc_i.yaml
+++ b/ydb/docs/en/core/yql/reference/syntax/create_table/toc_i.yaml
@@ -1,8 +1,9 @@
items:
-- { name: Overview, href: index.md }
+- { name: Overview, href: index.md }
- { name: SECONDARY INDEX, href: secondary_index.md }
- { name: VECTOR INDEX, href: vector_index.md }
- { name: FULLTEXT INDEX, href: fulltext_index.md }
+- { name: JSON INDEX, href: json_index.md }
- { name: BLOOM INDEX, href: bloom_skip_index.md }
- { name: FAMILY, href: family.md }
- { name: WITH, href: with.md }
diff --git a/ydb/docs/en/core/yql/reference/syntax/discard.md b/ydb/docs/en/core/yql/reference/syntax/discard.md
index 480fe5dedaa..f19f75e553f 100644
--- a/ydb/docs/en/core/yql/reference/syntax/discard.md
+++ b/ydb/docs/en/core/yql/reference/syntax/discard.md
@@ -4,6 +4,34 @@ Calculates {% if select_command == "SELECT STREAM" %}[`SELECT STREAM`](select_st
It's good to combine it with [`Ensure`](../builtins/basic.md#ensure) to check the final calculation result against the user's criteria.
+{% if backend_name == "YDB" %}
+
+A query with `DISCARD` is executed in full — with all its filters, aggregations, and `Ensure` checks — but the result set is not returned to the client. The data is not sent over the network either: the client receives only the query execution status. For large outputs, this saves a noticeable amount of traffic and memory.
+
+In a query consisting of multiple statements, `DISCARD` applies to an individual statement:
+
+```yql
+SELECT 1; -- returned
+DISCARD SELECT 2; -- executed, but not included in the response
+SELECT 3; -- returned
+```
+
+The client will receive two result sets — from the first and third statements.
+
+`DISCARD` applies only to a statement as a whole and cannot be used inside expressions. In particular, it is not allowed:
+
+* in subqueries — `SELECT * FROM (DISCARD SELECT 1)`;
+* in `WHERE ... IN (...)` — `SELECT * FROM my_table WHERE Key IN (DISCARD SELECT 1)`;
+* in `UNION ALL` operands — `SELECT 1 UNION ALL DISCARD SELECT 2`.
+
+{% note info %}
+
+`DISCARD` is supported for queries executed via the Query Service starting from {{ ydb-short-name }} version 26.2. When a query is executed via the legacy interfaces (Table Service, scan queries), `DISCARD` is ignored and the result is returned to the client — this behavior is preserved for backward compatibility.
+
+{% endnote %}
+
+{% endif %}
+
{% if select_command != true or select_command == "SELECT" %}
## Examples
@@ -12,6 +40,20 @@ It's good to combine it with [`Ensure`](../builtins/basic.md#ensure) to check th
DISCARD SELECT 1;
```
+{% if backend_name == "YDB" %}
+
+```yql
+DISCARD SELECT Ensure(
+ Data,
+ Data < 1000000,
+ "Value too big"
+) FROM `result_table`;
+```
+
+If the `Ensure` condition is violated, the query fails with an error. Otherwise, the query completes successfully, and the client does not receive the result set that would otherwise have to be downloaded in full.
+
+{% else %}
+
```yql
INSERT INTO result_table WITH TRUNCATE
SELECT * FROM
@@ -29,3 +71,4 @@ DISCARD SELECT Ensure(
{% endif %}
+{% endif %}
diff --git a/ydb/docs/en/core/yql/reference/syntax/index.md b/ydb/docs/en/core/yql/reference/syntax/index.md
index ca8bb8acc76..c251f9d48b8 100644
--- a/ydb/docs/en/core/yql/reference/syntax/index.md
+++ b/ydb/docs/en/core/yql/reference/syntax/index.md
@@ -79,12 +79,8 @@
{% endif %}
-{% if backend_name != "YDB" %}
-
* [DISCARD](discard.md)
-{% endif %}
-
* [INTO RESULT](into_result.md)
{% if feature_mapreduce %}
diff --git a/ydb/docs/en/core/yql/reference/syntax/select/index.md b/ydb/docs/en/core/yql/reference/syntax/select/index.md
index db34322bed5..a416ce4b424 100644
--- a/ydb/docs/en/core/yql/reference/syntax/select/index.md
+++ b/ydb/docs/en/core/yql/reference/syntax/select/index.md
@@ -1,112 +1,234 @@
-<!-- markdownlint-disable blanks-around-fences -->
-
-# SELECT
+## SELECT
Returns the result of evaluating the expressions specified after `SELECT`.
-It can be used in combination with other operations to obtain other effect.
+Can be used in combination with other operations to achieve a different effect.
+
+### Examples
-## Examples
```yql
SELECT "Hello, world!";
```
+
```yql
SELECT 2 + 2;
```
-## SELECT execution procedure {#selectexec}
-The `SELECT` query result is calculated as follows:
+## Procedure for executing SELECT {#selectexec}
+
+The result of the `SELECT` query is computed as follows:
+
+* the set of input tables is determined: expressions after [FROM](../select/from.md) are evaluated
-* Determine the set of input tables by evaluating the [FROM](from.md) clauses.
{% if feature_match_recogznize==true %}
-* Apply [MATCH_RECOGNIZE](match_recognize.md) to input tables.
-{% endif %}
-{% if feature_tablesample==true %}
-* Evaluate [SAMPLE](sample.md)/[TABLESAMPLE](sample.md).
+
+* [MATCH_RECOGNIZE](match_recognize.md) is applied to the input tables
+
{% endif %}
-* Execute [FLATTEN COLUMNS](flatten.md#flatten-columns) or [FLATTEN BY](flatten.md); aliases set in `FLATTEN BY` become visible after this point.
+
+* is computed [SAMPLE](sample.md) / [TABLESAMPLE](sample.md)
+* [FLATTEN COLUMNS](flatten.md#flatten-columns) or [FLATTEN BY](flatten.md) is performed; aliases specified in `FLATTEN BY` become visible after this point.
+
{% if feature_join %}
-* Execute every [JOIN](join.md).
+* All [JOIN](join.md) are executed.
{% endif %}
-* Add to (or replace in) the data the columns listed in [GROUP BY ... AS ...](group-by.md).
-* Execute [WHERE](where.md) &mdash; Discard all the data mismatching the predicate.
-* Execute [GROUP BY](group-by.md), evaluate aggregate functions.
-* Apply the filter [HAVING](group-by.md#having).
+
+* columns specified in [GROUP BY ... AS ...](group-by.md) are added (or replaced) to the resulting data
+* the [WHERE](where.md) clause is executed: all data that does not satisfy the predicate is filtered out.
+* [GROUP BY](group-by.md) is performed, aggregate function values are computed.
+* Filtering is performed using [HAVING](group-by.md#having)
+
{% if feature_window_functions %}
-* Evaluate [window functions](window.md);
+* Values of [window functions](window.md) are computed
{% endif %}
-* Evaluate expressions in `SELECT`.
-* Assign names set by aliases to expressions in `SELECT`.
-* Apply top-level [DISTINCT](distinct.md) to the resulting columns.
-* Execute similarly every subquery inside [UNION ALL](union.md#union-all), combine them (see [PRAGMA AnsiOrderByLimitInUnionAll](../pragma.md#pragmas)).
-* Perform sorting with [ORDER BY](order_by.md).
-* Apply [OFFSET and LIMIT](limit_offset.md) to the result.
+
+* Expressions in `SELECT` are evaluated.
+* expressions in `SELECT` are assigned names defined by aliases.
+* a top-level [DISTINCT](distinct.md) is applied to the columns obtained in this way
+* All subqueries in [UNION ALL](union.md#union-all) are computed in the same way and combined (see [PRAGMA AnsiOrderByLimitInUnionAll](../pragma.md#pragmas)).
+* Sorting is performed according to [ORDER BY](order_by.md)
+* [OFFSET and LIMIT](limit_offset.md) are applied to the result.
## Column order in YQL {#orderedcolumns}
-The standard SQL is sensitive to the order of columns in projections (that is, in `SELECT`). While the order of columns must be preserved in the query results or when writing data to a new table, some SQL constructs use this order.
-This applies, for example, to [UNION ALL](union.md#union-all) and positional [ORDER BY](order_by.md) (ORDER BY ordinal).
+In standard SQL, the order of columns specified in the projection (in `SELECT`) matters. Besides the fact that the column order must be preserved when displaying query results or when writing to a new table, some SQL constructs use this order. This applies in particular to [UNION ALL](union.md#union-all) and positional [ORDER BY](order_by.md) (ORDER BY ordinal).
-The column order is ignored in YQL by default:
+By default, the order of columns is ignored in YQL:
-* The order of columns in the output tables and query results is undefined
-* The data scheme of the `UNION ALL` result is output by column names rather than positions
+* the order of columns in output tables and in query results is undefined
+* The data schema of the `UNION ALL` result is output by column names, not by positions.
-If you enable `PRAGMA OrderedColumns;`, the order of columns is preserved in the query results and is derived from the order of columns in the input tables using the following rules:
+When `PRAGMA OrderedColumns;` is enabled, the order of columns is preserved in the query results and is derived from the order of columns in the input tables according to the following rules:
-* `SELECT`: an explicit column enumeration dictates the result order.
+* `SELECT` with explicit column enumeration sets the corresponding order.
* `SELECT` with an asterisk (`SELECT * FROM ...`) inherits the order from its input.
+
{% if feature_join %}
-* The order of columns after [JOIN](join.md): First output the left-hand columns, then the right-hand ones. If the column order in any of the sides in the `JOIN` output is undefined, the column order in the result is also undefined.
+* the order of columns after [JOIN](join.md): first the columns from the left side, then from the right. If the order of either side present in the output `JOIN` is not defined, the order of the result columns is also not defined;
{% endif %}
-* The order in `UNION ALL` depends on the [UNION ALL](union.md#union-all) execution mode.
-* The column order for [AS_TABLE](from_as_table.md) is undefined.
-
-{% note warning %}
-
-In the YT table schema, key columns always precede non-key columns. The order of key columns is determined by the order of the composite key.
-When `PRAGMA OrderedColumns;` is enabled, non-key columns preserve their output order.
+* the order of depends on the execution mode of [`UNION ALL`](union.md#union-all).
+* Column order for [AS_TABLE](from_as_table.md) is not defined.
-{% endnote %}
+## Combination of queries {#combining-queries}
-## Combining queries {#combining-queries}
+Results of multiple SELECT (or subqueries) can be combined using the keywords `UNION` and `UNION ALL`.
-Results of several SELECT statements (or subqueries) can be combined using `UNION` and `UNION ALL` keywords.
```yql
query1 UNION [ALL] query2 (UNION [ALL] query3 ...)
```
-Union of more than two queries is interpreted as a left-associative operation, that is
+
+Union of more than two queries is interpreted as a left-associative operation, i.e.
+
```yql
query1 UNION query2 UNION ALL query3
```
+
is interpreted as
+
```yql
(query1 UNION query2) UNION ALL query3
```
-If the underlying queries have one of the `ORDER BY/LIMIT/DISCARD/INTO RESULT` operators, the following rules apply:
-* `ORDER BY/LIMIT/INTO RESULT` is only allowed after the last query
-* `DISCARD` is only allowed before the first query
-* the operators apply to the `UNION [ALL]` as a whole, instead of referring to one of the queries
-* to apply the operator to one of the queries, enclose the query in parentheses
+If `ORDER BY/LIMIT/DISCARD/INTO RESULT` is present in the combined subqueries, the following rules apply:
+
+* `ORDER BY/LIMIT/INTO RESULT` is allowed only after the last subquery
+* `DISCARD` is allowed only before the first subquery.
+* the specified operators act on the result `UNION [ALL]`, not on the subquery
+* to apply an operator to a subquery, the subquery must be enclosed in parentheses.
+
+## Accessing multiple tables in a single query
+
+In standard SQL, [UNION ALL](union.md#union-all) is used to query multiple tables, which combines the results of two or more `SELECT`. This is not very convenient for a use case where you need to run the same query across multiple tables (for example, containing data for different dates). In YQL, for convenience, in `SELECT` after `FROM` you can specify not only a single table or subquery, but also call built-in functions that allow combining data from multiple tables.
+
+The following functions are defined for these purposes:
+
+``` CONCAT(`table1`, `table2`, `table3` VIEW view_name, ...) ``` — combines all tables listed in the arguments.
+
+`EACH($list_of_strings)` or `EACH($list_of_strings VIEW view_name)` — combines all tables whose names are listed in the string list. Optionally, you can pass multiple lists in separate arguments, similar to `CONCAT`.
+
+``` RANGE(`prefix`, `min`, `max`, `suffix`, `view`) ```: combines a range of tables. Arguments:
+
+* prefix — directory for searching tables, specified without a trailing slash. The only required argument; if only it is specified, all tables in this directory are used.
+* min, max — the next two arguments specify a range of names for including tables. The range is inclusive on both ends. If the range is not specified, all tables in the prefix directory are used. Names of tables or directories located in the directory specified in prefix are compared with the range `[min, max]` lexicographically, not concatenated, so it is important to specify the range without leading slashes.
+* suffix — table name. Expected without a leading slash. If suffix is not specified, the arguments `[min, max]` specify a range of table names. If suffix is specified, the arguments `[min, max]` specify a range of folders in which a table with the name specified in the suffix argument exists.
+
+``` LIKE(`prefix`, `pattern`, `suffix`, `view`)` и `REGEXP(`prefix`, `pattern`, `suffix`, `view`) ``` — the pattern argument is specified in a format similar to the binary operators of the same name: [LIKE](../expressions.md#like) and [REGEXP](../expressions.md#regexp).
+
+``` FILTER(`prefix`, `callable`, `suffix`, `view`) ``` — the callable argument must be a callable expression with signature `(String)->Bool`, which will be called for each table/subdirectory in the prefix directory. Only those tables for which the callable value returned `true` will participate in the query. As a callable value, it is most convenient to use [lambda functions](../expressions.md#lambda){% if yql == true %}, or UDFs in Python or JavaScript{% endif %}.
+
+{% note warning %}
+
+The order in which tables are merged by all the above functions is not guaranteed.
+
+The list of tables is computed **before** the query itself is executed. Therefore, tables created during the query will not be included in the function results.
+
+{% endnote %}
+
+By default, schemas of all participating tables are merged according to the rules of [UNION ALL](union.md#union-all). If schema merging is not desired, you can use functions with the suffix `_STRICT`, for example `CONCAT_STRICT` or `RANGE_STRICT`, which work exactly like the original ones but treat any discrepancy in table schemas as an error.
+
+To specify the cluster of the merged tables, you need to specify it before the function name.
+
+All arguments of the functions described above can be declared separately using [named expressions](../expressions.md#named-nodes). In this case, simple expressions are also allowed in them by implicitly calling [EvaluateExpr](../../builtins/basic.md#evaluate_expr_atom).
+
+The name of the source table from which each row was originally obtained can be obtained using the [TablePath()](../../builtins/basic.md#tablepath) function.
+
+### Examples
+
+
+```yql
+SELECT * FROM CONCAT(
+ `table1`,
+ `table2`,
+ `table3`);
+```
-## Clauses supported in SELECT
+
+```yql
+$indices = ListFromRange(1, 4);
+$tables = ListMap($indices, ($index) -> {
+ RETURN "table" || CAST($index AS String);
+});
+SELECT * FROM EACH($tables); -- identical to the previous example
+```
+
+
+```yql
+SELECT * FROM RANGE(`my_folder`);
+```
+
+
+```yql
+SELECT * FROM some_cluster.RANGE( -- The cluster can be specified before the function name
+ `my_folder`,
+ `from_table`,
+ `to_table`);
+```
+
+
+```yql
+SELECT * FROM RANGE(
+ `my_folder`,
+ `from_folder`,
+ `to_folder`,
+ `my_table`);
+```
+
+
+```yql
+SELECT * FROM RANGE(
+ `my_folder`,
+ `from_table`,
+ `to_table`,
+ ``,
+ `my_view`);
+```
+
+
+```yql
+SELECT * FROM LIKE(
+ `my_folder`,
+ "2017-03-%"
+);
+```
+
+
+```yql
+SELECT * FROM REGEXP(
+ `my_folder`,
+ "2017-03-1[2-4]?"
+);
+```
+
+
+```yql
+$callable = ($table_name) -> {
+ return $table_name > "2017-03-13";
+};
+
+SELECT * FROM FILTER(
+ `my_folder`,
+ $callable
+);
+```
+
+
+## Supported constructs in SELECT
* [FROM](from.md)
* [FROM AS_TABLE](from_as_table.md)
@@ -114,26 +236,34 @@ If the underlying queries have one of the `ORDER BY/LIMIT/DISCARD/INTO RESULT` o
* [DISTINCT](distinct.md)
* [UNIQUE DISTINCT](unique_distinct_hints.md)
* [UNION](union.md)
-* [WITH](with.md)
+* WITH
* [WITHOUT](without.md)
* [WHERE](where.md)
* [ORDER BY](order_by.md)
* [ASSUME ORDER BY](assume_order_by.md)
* [LIMIT OFFSET](limit_offset.md)
-{% if feature_tablesample==true %}
* [SAMPLE](sample.md)
* [TABLESAMPLE](sample.md)
-{% endif %}
+
{% if feature_match_recogznize==true %}
+
* [MATCH_RECOGNIZE](match_recognize.md)
+
{% endif %}
+
{% if feature_join %}
+
* [JOIN](join.md)
+
{% endif %}
+
* [GROUP BY](group-by.md)
* [FLATTEN](flatten.md)
+
{% if feature_window_functions %}
+
* [WINDOW](window.md)
+
{% endif %}
{% if yt %}
@@ -164,8 +294,8 @@ If the underlying queries have one of the `ORDER BY/LIMIT/DISCARD/INTO RESULT` o
{% if feature_secondary_index %}
* [VIEW secondary_index](secondary_index.md)
-
* [VIEW vector_index](vector_index.md)
* [VIEW fulltext_index](fulltext_index.md)
+* [VIEW json_index](json_index.md)
{% endif %}
diff --git a/ydb/docs/en/core/yql/reference/syntax/select/json_index.md b/ydb/docs/en/core/yql/reference/syntax/select/json_index.md
new file mode 100644
index 00000000000..e9985894235
--- /dev/null
+++ b/ydb/docs/en/core/yql/reference/syntax/select/json_index.md
@@ -0,0 +1,125 @@
+# VIEW (JSON index)
+
+To execute a `SELECT` query on a row table with explicit use of a [JSON index](../../../../dev/json-indexes.md), use the `VIEW` expression:
+
+
+```yql
+SELECT ...
+FROM documents VIEW json_idx
+WHERE <предикат на основе JSON_EXISTS / JSON_VALUE>
+ORDER BY ...
+```
+
+
+In the query example above, `documents` is the name of the table containing a column of type `Json` or `JsonDocument`, and `json_idx` is the name of the JSON index created on that column.
+
+If the predicate is not supported for execution via a JSON index, a query with an explicit `VIEW` expression fails with a compile-time error. Without the `VIEW` expression, the optimizer cannot select this index for such a predicate, and the query will be executed by another method (for example, selecting a different index or performing a full scan of the base table).
+
+{% note info %}
+
+A JSON index can be selected automatically by the [optimizer](../../../../concepts/glossary.md#optimizer) if the predicate meets the requirements for index usage. For debugging and guaranteed index usage, specify it explicitly using `VIEW IndexName`.
+
+For a description of supported predicates, see the [{#T}](../../../../dev/json-indexes.md#predicates) section.
+
+{% endnote %}
+
+## Automatic index selection {#auto}
+
+If the `WHERE` predicate contains `JSON_EXISTS` / `JSON_VALUE` calls on an indexed JSON column, the [optimizer](../../../../concepts/glossary.md#optimizer) may use the JSON index without an explicit `VIEW`. Automatic selection follows these rules:
+
+* The JSON index is considered by the optimizer with the lowest priority — it is a fallback option that is selected only after other access methods have been ruled out.
+* The JSON index is not selected if the query can already be served by a [primary key](../../../../concepts/glossary.md#primary-index) or another, more specific [secondary index](../../../../concepts/glossary.md#secondary-index).
+* Explicit specification of `VIEW` overrides the optimizer's decision and forces index usage.
+* All indexed subexpressions in a single query must refer to the same indexed JSON column. If predicates on different indexed columns are combined via `AND` / `OR`, automatic selection is not applied.
+
+{% note info %}
+
+Automatic JSON index selection is rule-based: it uses the query structure and schema metadata, without data statistics. The selection logic may change in future versions of {{ ydb-short-name }}, so for guaranteed index usage, specify it explicitly via `VIEW`.
+
+{% endnote %}
+
+## JSON_EXISTS
+
+[JSON_EXISTS(doc, jsonpath)](../../builtins/json.md#json_exists) checks for the existence of a path or value inside a JSON document:
+
+
+```yql
+SELECT id, payload
+FROM documents VIEW json_idx
+WHERE JSON_EXISTS(payload, '$.user.id');
+```
+
+
+The JSON index supports almost all JsonPath syntax, except for nested predicates and boolean expressions at the context object level (`$`) in `JSON_EXISTS`. A more complex example of a supported expression:
+
+
+```yql
+SELECT id, payload
+FROM documents VIEW json_idx
+WHERE JSON_EXISTS(payload, '$.user ? (@.id > 100)');
+```
+
+
+## JSON_VALUE
+
+[JSON_VALUE(doc, jsonpath RETURNING <type>)](../../builtins/json.md#json_value) extracts a scalar value by a JsonPath and returns it in the specified type. To use the JSON index, the `RETURNING <type>` clause is required:
+
+
+```yql
+SELECT id, payload
+FROM documents VIEW json_idx
+WHERE JSON_VALUE(payload, '$.user.name' RETURNING Utf8) = "Charlie"u;
+```
+
+
+When checking equality conditions or whether a value is in a given list (`IN`), the index search is performed by the 'path + value' token, which provides the highest selectivity. For other comparisons (`!=`, `<`, `>=`, `BETWEEN`, etc.), only the path token is used, and the final comparison accuracy is ensured by a post-filter.
+
+### Parameters
+
+Three ways to pass query parameters are supported:
+
+1. Direct comparison of the `JSON_VALUE` result with a parameter:
+
+
+ ```yql
+ DECLARE $id AS Int64;
+ SELECT * FROM documents VIEW json_idx
+ WHERE JSON_VALUE(payload, '$.owner_id' RETURNING Int64) = $id;
+ ```
+
+2. Checking whether the result is in a list of values:
+
+
+ ```yql
+ DECLARE $tags AS List<Utf8>;
+ SELECT * FROM documents VIEW json_idx
+ WHERE JSON_VALUE(payload, '$.tag' RETURNING Utf8) IN $tags;
+ ```
+
+3. Passing a parameter to JsonPath via the `PASSING` clause:
+
+
+ ```yql
+ DECLARE $v AS Int64;
+ SELECT * FROM documents VIEW json_idx
+ WHERE JSON_EXISTS(payload, '$.k ? (@ == $v)' PASSING $v AS v);
+ ```
+
+## AND and OR combinations
+
+`JSON_EXISTS` and `JSON_VALUE` on the same JSON column can be combined in a single `WHERE` using the `AND` and `OR` operators:
+
+
+```yql
+SELECT id, payload
+FROM documents VIEW json_idx
+WHERE JSON_VALUE(payload, '$.user.name' RETURNING Utf8) = "Charlie"u
+ AND JSON_VALUE(payload, '$.user.id' RETURNING Int64) BETWEEN 100 AND 200;
+```
+
+
+{% note info %}
+
+Query execution via a JSON index can only use a subset of expressions based on `JSON_EXISTS` and `JSON_VALUE` (see [{#T}](../../../../dev/json-indexes.md#predicates)). Conditions that do not fall under these rules (for example, negations, comparisons with another column, comparison of two `JSON_VALUE` from different columns) are not indexed; in `AND` they go into a post-filter, in `OR` they lead to rejection of index usage for the entire group of conditions combined via `OR`.
+
+{% endnote %}
diff --git a/ydb/docs/en/core/yql/reference/syntax/select/toc_i.yaml b/ydb/docs/en/core/yql/reference/syntax/select/toc_i.yaml
index abb7e9b1c22..8a6bee82b9b 100644
--- a/ydb/docs/en/core/yql/reference/syntax/select/toc_i.yaml
+++ b/ydb/docs/en/core/yql/reference/syntax/select/toc_i.yaml
@@ -1,5 +1,5 @@
items:
-- { name: Overview, href: index.md }
+- { name: Overview, href: index.md }
- { name: TEMPORARY TABLE, href: temporary_table.md, when: feature_temp_table }
- { name: FROM, href: from.md }
- { name: FROM AS_TABLE, href: from_as_table.md }
@@ -18,16 +18,13 @@ items:
- { name: VIEW SECONDARY INDEX, href: secondary_index.md, when: feature_secondary_index }
- { name: VIEW VECTOR INDEX, href: vector_index.md, when: feature_secondary_index }
- { name: VIEW FULLTEXT INDEX, href: fulltext_index.md, when: feature_secondary_index }
+- { name: VIEW JSON INDEX, href: json_index.md, when: feature_secondary_index }
- { name: ORDER BY HybridRank, href: hybrid_search.md, when: feature_secondary_index }
-- name: WITH
- href: with.md
- include: { mode: link, path: with/toc_p.yaml }
- { name: WITHOUT, href: without.md }
- { name: WHERE, href: where.md }
- { name: ORDER BY, href: order_by.md }
- { name: ASSUME ORDER BY, href: assume_order_by.md }
- { name: LIMIT OFFSET, href: limit_offset.md }
-- { name: SAMPLE, href: sample.md, when: feature_tablesample }
-- { name: TABLESAMPLE, href: sample.md, when: feature_tablesample }
+- { name: SAMPLE, href: sample.md }
- { name: MATCH_RECOGNIZE, href: match_recognize.md }
- { name: STREAMING, href: streaming.md }
diff --git a/ydb/docs/en/core/yql/reference/syntax/toc_i.yaml b/ydb/docs/en/core/yql/reference/syntax/toc_i.yaml
index d0c57876b81..b9fd62bf1f2 100644
--- a/ydb/docs/en/core/yql/reference/syntax/toc_i.yaml
+++ b/ydb/docs/en/core/yql/reference/syntax/toc_i.yaml
@@ -44,7 +44,7 @@ items:
- { name: COMMIT, href: commit.md }
- { name: DECLARE, href: declare.md }
- { name: DELETE, href: delete.md, when: feature_map_tables }
-- { name: DISCARD, href: discard.md, when: backend_name != "YDB" }
+- { name: DISCARD, href: discard.md }
- { name: DROP ASYNC REPLICATION, href: drop-async-replication.md, when: feature_async_replication }
- { name: DROP BACKUP COLLECTION, href: drop-backup-collection.md, when: feature_backup_collections }
- { name: DROP GROUP, href: drop-group.md, when: feature_user_and_group }
diff --git a/ydb/docs/en/core/yql/reference/types/primitive.md b/ydb/docs/en/core/yql/reference/types/primitive.md
index e68db6350b7..d41ee113642 100644
--- a/ydb/docs/en/core/yql/reference/types/primitive.md
+++ b/ydb/docs/en/core/yql/reference/types/primitive.md
@@ -60,7 +60,7 @@ Floating-point number with fixed precision, 16 bytes in size. Precision is the m
|| `DyNumber` |
Binary representation of a floating-point number with up to 38 digits of precision.
Valid values: positive from 1×10<sup>-130</sup> to 1×10<sup>126</sup>–1, negative from -1×10<sup>126</sup>–1 to -1×10<sup>-130</sup> and 0.
-Compatible with the `Number` type of AWS DynamoDB. Not recommended for use in {{ backend_name_lower }}-native applications. | Not supported in columnar tables
+Compatible with the `Number` type of AWS DynamoDB. Not recommended for use in {{ backend_name_lower }}-native applications. |
||
{% endif %}
|#
@@ -95,7 +95,7 @@ Does not support comparison{% if feature_map_tables %}, cannot be used in the pr
Does not support comparison{% if feature_map_tables %}, cannot be used in the primary key and in the columns that form the secondary index key{% endif %}
||
|| `Uuid` |
-Universal identifier [UUID](https://tools.ietf.org/html/rfc4122) | Not supported in columnar tables
+Universal identifier [UUID](https://tools.ietf.org/html/rfc4122) |
||
|#
{% note info "Size limitations" %}
@@ -199,7 +199,6 @@ from -136 years to +136 years
|
8
|
-Not supported in columnar tables
||
||
@@ -211,7 +210,6 @@ from -292277 years to +292277 years
|
8
|
-Not supported in columnar tables
||
||
diff --git a/ydb/docs/en/core/yql/toc_i.yaml b/ydb/docs/en/core/yql/toc_i.yaml
index f5b02d179a6..28a10223cb7 100644
--- a/ydb/docs/en/core/yql/toc_i.yaml
+++ b/ydb/docs/en/core/yql/toc_i.yaml
@@ -1,6 +1,3 @@
items:
-- name: Overview
- href: reference/index.md
-- include: { mode: link, path: reference/toc_i.yaml }
- name: Query plans
href: query_plans.md
diff --git a/ydb/docs/en/core/yql/toc_p.yaml b/ydb/docs/en/core/yql/toc_p.yaml
index 50818e81233..0be250eb322 100644
--- a/ydb/docs/en/core/yql/toc_p.yaml
+++ b/ydb/docs/en/core/yql/toc_p.yaml
@@ -1,3 +1,5 @@
items:
-- include: { mode: link, path: toc_i.yaml }
+- name: Overview
+ href: reference/index.md
- include: { mode: link, path: reference/toc_p.yaml }
+- include: { mode: link, path: toc_i.yaml }
diff --git a/ydb/docs/redirects.yaml b/ydb/docs/redirects.yaml
index c0f191371f2..e738be43d12 100644
--- a/ydb/docs/redirects.yaml
+++ b/ydb/docs/redirects.yaml
@@ -544,20 +544,38 @@ ru:
to: /reference/ydb-sdk/observability/metrics/opentelemetry.md
- from: /recipes/ydb-sdk/debug-jaeger.md
to: /reference/ydb-sdk/observability/tracing/jaeger.md
+ - from: /recipes/ydb-sdk/debug-otel.md
+ to: /reference/ydb-sdk/observability/tracing/opentelemetry.md
- from: /recipes/ydb-sdk/debug-otel-tracing.md
to: /reference/ydb-sdk/observability/tracing/opentelemetry.md
- from: /integrations/sql-dialect-converter.md
to: /integrations/sql-translation/sql-dialect-converter.md
en:
- - from: /reference/ydb-sdk/recipes/debug-jaeger.md
- to: /recipes/ydb-sdk/debug-jaeger.md
+ - from: /reference/ydb-sdk/recipes/debug.md
+ to: /reference/ydb-sdk/observability/index.md
- from: /reference/ydb-sdk/recipes/debug-logs.md
- to: /recipes/ydb-sdk/debug-logs.md
+ to: /reference/ydb-sdk/observability/logging/logging.md
- from: /reference/ydb-sdk/recipes/debug-prometheus.md
- to: /recipes/ydb-sdk/debug-prometheus.md
- - from: /reference/ydb-sdk/recipes/debug.md
- to: /recipes/ydb-sdk/debug.md
+ to: /reference/ydb-sdk/observability/metrics/prometheus.md
+ - from: /reference/ydb-sdk/recipes/debug-jaeger.md
+ to: /reference/ydb-sdk/observability/tracing/jaeger.md
+ - from: /recipes/ydb-sdk/debug.md
+ to: /reference/ydb-sdk/observability/index.md
+ - from: /recipes/ydb-sdk/debug-logs.md
+ to: /reference/ydb-sdk/observability/logging/logging.md
+ - from: /recipes/ydb-sdk/debug-logs-otel.md
+ to: /reference/ydb-sdk/observability/logging/opentelemetry.md
+ - from: /recipes/ydb-sdk/debug-prometheus.md
+ to: /reference/ydb-sdk/observability/metrics/prometheus.md
+ - from: /recipes/ydb-sdk/debug-otel-metrics.md
+ to: /reference/ydb-sdk/observability/metrics/opentelemetry.md
+ - from: /recipes/ydb-sdk/debug-jaeger.md
+ to: /reference/ydb-sdk/observability/tracing/jaeger.md
+ - from: /recipes/ydb-sdk/debug-otel.md
+ to: /reference/ydb-sdk/observability/tracing/opentelemetry.md
+ - from: /recipes/ydb-sdk/debug-otel-tracing.md
+ to: /reference/ydb-sdk/observability/tracing/opentelemetry.md
- from: /devops/deployment-options/manual/initial-deployment.md
to: /devops/deployment-options/manual/initial-deployment/index.md
- from: /devops/manual/initial-deployment.md
diff --git a/ydb/docs/ru/core/_assets/resources_weight.drawio b/ydb/docs/ru/core/_assets/resources_weight.drawio
index 9fe02ceb6cf..ef2a1a4bdd1 100644
--- a/ydb/docs/ru/core/_assets/resources_weight.drawio
+++ b/ydb/docs/ru/core/_assets/resources_weight.drawio
@@ -40,7 +40,7 @@
<mxPoint x="490" y="230" as="targetPoint" />
</mxGeometry>
</mxCell>
- <mxCell id="cv_4T4Yt_JHGhEiulPHa-16" value="RESOURCES_WEIGHT&lt;div&gt;=100&lt;/div&gt;" style="edgeLabel;html=1;align=center;verticalAlign=middle;resizable=0;points=[];" vertex="1" connectable="0" parent="cv_4T4Yt_JHGhEiulPHa-15">
+ <mxCell id="cv_4T4Yt_JHGhEiulPHa-16" value="RESOURCE_WEIGHT&lt;div&gt;=100&lt;/div&gt;" style="edgeLabel;html=1;align=center;verticalAlign=middle;resizable=0;points=[];" vertex="1" connectable="0" parent="cv_4T4Yt_JHGhEiulPHa-15">
<mxGeometry x="-0.1738" y="6" relative="1" as="geometry">
<mxPoint x="65" y="-12" as="offset" />
</mxGeometry>
@@ -146,10 +146,10 @@
<mxPoint x="-97" y="7" as="offset" />
</mxGeometry>
</mxCell>
- <mxCell id="cv_4T4Yt_JHGhEiulPHa-42" value="&lt;div style=&quot;text-align: center;&quot;&gt;&lt;span style=&quot;color: rgb(0, 0, 0); font-family: Helvetica; font-size: 12px; font-style: normal; font-variant-ligatures: normal; font-variant-caps: normal; font-weight: 400; letter-spacing: normal; orphans: 2; text-indent: 0px; text-transform: none; widows: 2; word-spacing: 0px; -webkit-text-stroke-width: 0px; white-space: nowrap; background-color: rgb(251, 251, 251); text-decoration-thickness: initial; text-decoration-style: initial; text-decoration-color: initial; float: none; display: inline !important;&quot;&gt;Cтарый лимит (RESOURCES_WEIGHT=100):&lt;/span&gt;&lt;/div&gt;&lt;span style=&quot;color: rgb(0, 0, 0); font-family: Helvetica; font-size: 12px; font-style: normal; font-variant-ligatures: normal; font-variant-caps: normal; font-weight: 400; letter-spacing: normal; orphans: 2; text-align: center; text-indent: 0px; text-transform: none; widows: 2; word-spacing: 0px; -webkit-text-stroke-width: 0px; white-space: nowrap; background-color: rgb(251, 251, 251); text-decoration-thickness: initial; text-decoration-style: initial; text-decoration-color: initial; display: inline !important; float: none;&quot;&gt;TOTAL_CPU_LIMIT_PERCENT_PER_NODE = 30 = 3 vCPU&lt;/span&gt;&lt;br style=&quot;forced-color-adjust: none; color: rgb(0, 0, 0); font-family: Helvetica; font-size: 12px; font-style: normal; font-variant-ligatures: normal; font-variant-caps: normal; font-weight: 400; letter-spacing: normal; orphans: 2; text-align: center; text-indent: 0px; text-transform: none; widows: 2; word-spacing: 0px; -webkit-text-stroke-width: 0px; white-space: nowrap; background-color: rgb(251, 251, 251); text-decoration-thickness: initial; text-decoration-style: initial; text-decoration-color: initial;&quot;&gt;&lt;span style=&quot;color: rgb(0, 0, 0); font-family: Helvetica; font-size: 12px; font-style: normal; font-variant-ligatures: normal; font-variant-caps: normal; font-weight: 400; letter-spacing: normal; orphans: 2; text-align: center; text-indent: 0px; text-transform: none; widows: 2; word-spacing: 0px; -webkit-text-stroke-width: 0px; white-space: nowrap; background-color: rgb(251, 251, 251); text-decoration-thickness: initial; text-decoration-style: initial; text-decoration-color: initial; display: inline !important; float: none;&quot;&gt;QUERY_CPU_LIMIT_PERCENT_PER_NODE = 50 = 1.5&lt;/span&gt;&lt;div&gt;&lt;span style=&quot;color: rgb(0, 0, 0); font-family: Helvetica; font-size: 12px; font-style: normal; font-variant-ligatures: normal; font-variant-caps: normal; font-weight: 400; letter-spacing: normal; orphans: 2; text-align: center; text-indent: 0px; text-transform: none; widows: 2; word-spacing: 0px; -webkit-text-stroke-width: 0px; white-space: nowrap; background-color: rgb(251, 251, 251); text-decoration-thickness: initial; text-decoration-style: initial; text-decoration-color: initial; display: inline !important; float: none;&quot;&gt;&lt;br&gt;&lt;/span&gt;&lt;/div&gt;&lt;div style=&quot;text-align: center;&quot;&gt;&lt;span style=&quot;text-wrap: nowrap;&quot;&gt;Новый лимит (&lt;/span&gt;&lt;span style=&quot;text-wrap: nowrap; background-color: initial;&quot;&gt;(RESOURCES_WEIGHT=100)&lt;/span&gt;&lt;span style=&quot;text-wrap: nowrap; background-color: initial;&quot;&gt;:&lt;/span&gt;&lt;/div&gt;&lt;div style=&quot;text-align: center;&quot;&gt;&lt;span style=&quot;text-wrap: nowrap;&quot;&gt;TOTAL_CPU_LIMIT_PERCENT_PER_NODE = 70/3 ~= 2.3 vCPU&lt;/span&gt;&lt;br style=&quot;text-wrap: nowrap;&quot;&gt;&lt;span style=&quot;text-wrap: nowrap;&quot;&gt;QUERY_CPU_LIMIT_PERCENT_PER_NODE = 50 ~= 1.15&amp;nbsp; vCPU&lt;/span&gt;&lt;span style=&quot;text-wrap: nowrap;&quot;&gt;&lt;br&gt;&lt;/span&gt;&lt;/div&gt;" style="text;whiteSpace=wrap;html=1;" vertex="1" parent="1">
+ <mxCell id="cv_4T4Yt_JHGhEiulPHa-42" value="&lt;div style=&quot;text-align: center;&quot;&gt;&lt;span style=&quot;color: rgb(0, 0, 0); font-family: Helvetica; font-size: 12px; font-style: normal; font-variant-ligatures: normal; font-variant-caps: normal; font-weight: 400; letter-spacing: normal; orphans: 2; text-indent: 0px; text-transform: none; widows: 2; word-spacing: 0px; -webkit-text-stroke-width: 0px; white-space: nowrap; background-color: rgb(251, 251, 251); text-decoration-thickness: initial; text-decoration-style: initial; text-decoration-color: initial; float: none; display: inline !important;&quot;&gt;Cтарый лимит (RESOURCE_WEIGHT=100):&lt;/span&gt;&lt;/div&gt;&lt;span style=&quot;color: rgb(0, 0, 0); font-family: Helvetica; font-size: 12px; font-style: normal; font-variant-ligatures: normal; font-variant-caps: normal; font-weight: 400; letter-spacing: normal; orphans: 2; text-align: center; text-indent: 0px; text-transform: none; widows: 2; word-spacing: 0px; -webkit-text-stroke-width: 0px; white-space: nowrap; background-color: rgb(251, 251, 251); text-decoration-thickness: initial; text-decoration-style: initial; text-decoration-color: initial; display: inline !important; float: none;&quot;&gt;TOTAL_CPU_LIMIT_PERCENT_PER_NODE = 30 = 3 vCPU&lt;/span&gt;&lt;br style=&quot;forced-color-adjust: none; color: rgb(0, 0, 0); font-family: Helvetica; font-size: 12px; font-style: normal; font-variant-ligatures: normal; font-variant-caps: normal; font-weight: 400; letter-spacing: normal; orphans: 2; text-align: center; text-indent: 0px; text-transform: none; widows: 2; word-spacing: 0px; -webkit-text-stroke-width: 0px; white-space: nowrap; background-color: rgb(251, 251, 251); text-decoration-thickness: initial; text-decoration-style: initial; text-decoration-color: initial;&quot;&gt;&lt;span style=&quot;color: rgb(0, 0, 0); font-family: Helvetica; font-size: 12px; font-style: normal; font-variant-ligatures: normal; font-variant-caps: normal; font-weight: 400; letter-spacing: normal; orphans: 2; text-align: center; text-indent: 0px; text-transform: none; widows: 2; word-spacing: 0px; -webkit-text-stroke-width: 0px; white-space: nowrap; background-color: rgb(251, 251, 251); text-decoration-thickness: initial; text-decoration-style: initial; text-decoration-color: initial; display: inline !important; float: none;&quot;&gt;QUERY_CPU_LIMIT_PERCENT_PER_NODE = 50 = 1.5&lt;/span&gt;&lt;div&gt;&lt;span style=&quot;color: rgb(0, 0, 0); font-family: Helvetica; font-size: 12px; font-style: normal; font-variant-ligatures: normal; font-variant-caps: normal; font-weight: 400; letter-spacing: normal; orphans: 2; text-align: center; text-indent: 0px; text-transform: none; widows: 2; word-spacing: 0px; -webkit-text-stroke-width: 0px; white-space: nowrap; background-color: rgb(251, 251, 251); text-decoration-thickness: initial; text-decoration-style: initial; text-decoration-color: initial; display: inline !important; float: none;&quot;&gt;&lt;br&gt;&lt;/span&gt;&lt;/div&gt;&lt;div style=&quot;text-align: center;&quot;&gt;&lt;span style=&quot;text-wrap: nowrap;&quot;&gt;Новый лимит (&lt;/span&gt;&lt;span style=&quot;text-wrap: nowrap; background-color: initial;&quot;&gt;(RESOURCE_WEIGHT=100)&lt;/span&gt;&lt;span style=&quot;text-wrap: nowrap; background-color: initial;&quot;&gt;:&lt;/span&gt;&lt;/div&gt;&lt;div style=&quot;text-align: center;&quot;&gt;&lt;span style=&quot;text-wrap: nowrap;&quot;&gt;TOTAL_CPU_LIMIT_PERCENT_PER_NODE = 70/3 ~= 2.3 vCPU&lt;/span&gt;&lt;br style=&quot;text-wrap: nowrap;&quot;&gt;&lt;span style=&quot;text-wrap: nowrap;&quot;&gt;QUERY_CPU_LIMIT_PERCENT_PER_NODE = 50 ~= 1.15&amp;nbsp; vCPU&lt;/span&gt;&lt;span style=&quot;text-wrap: nowrap;&quot;&gt;&lt;br&gt;&lt;/span&gt;&lt;/div&gt;" style="text;whiteSpace=wrap;html=1;" vertex="1" parent="1">
<mxGeometry x="170.00000000000009" y="820" width="360" height="50" as="geometry" />
</mxCell>
- <mxCell id="cv_4T4Yt_JHGhEiulPHa-43" value="&lt;div style=&quot;text-align: center;&quot;&gt;&lt;span style=&quot;color: rgb(0, 0, 0); font-family: Helvetica; font-size: 12px; font-style: normal; font-variant-ligatures: normal; font-variant-caps: normal; font-weight: 400; letter-spacing: normal; orphans: 2; text-indent: 0px; text-transform: none; widows: 2; word-spacing: 0px; -webkit-text-stroke-width: 0px; white-space: nowrap; background-color: rgb(251, 251, 251); text-decoration-thickness: initial; text-decoration-style: initial; text-decoration-color: initial; float: none; display: inline !important;&quot;&gt;Cтарый лимит (&lt;/span&gt;&lt;span style=&quot;text-wrap: nowrap; background-color: initial;&quot;&gt;(RESOURCES_WEIGHT=200)&lt;/span&gt;&lt;span style=&quot;text-wrap: nowrap;&quot;&gt;:&lt;/span&gt;&lt;/div&gt;&lt;span style=&quot;color: rgb(0, 0, 0); font-family: Helvetica; font-size: 12px; font-style: normal; font-variant-ligatures: normal; font-variant-caps: normal; font-weight: 400; letter-spacing: normal; orphans: 2; text-align: center; text-indent: 0px; text-transform: none; widows: 2; word-spacing: 0px; -webkit-text-stroke-width: 0px; white-space: nowrap; background-color: rgb(251, 251, 251); text-decoration-thickness: initial; text-decoration-style: initial; text-decoration-color: initial; display: inline !important; float: none;&quot;&gt;TOTAL_CPU_LIMIT_PERCENT_PER_NODE = 30 = 3 vCPU&lt;/span&gt;&lt;br style=&quot;forced-color-adjust: none; color: rgb(0, 0, 0); font-family: Helvetica; font-size: 12px; font-style: normal; font-variant-ligatures: normal; font-variant-caps: normal; font-weight: 400; letter-spacing: normal; orphans: 2; text-align: center; text-indent: 0px; text-transform: none; widows: 2; word-spacing: 0px; -webkit-text-stroke-width: 0px; white-space: nowrap; background-color: rgb(251, 251, 251); text-decoration-thickness: initial; text-decoration-style: initial; text-decoration-color: initial;&quot;&gt;&lt;span style=&quot;color: rgb(0, 0, 0); font-family: Helvetica; font-size: 12px; font-style: normal; font-variant-ligatures: normal; font-variant-caps: normal; font-weight: 400; letter-spacing: normal; orphans: 2; text-align: center; text-indent: 0px; text-transform: none; widows: 2; word-spacing: 0px; -webkit-text-stroke-width: 0px; white-space: nowrap; background-color: rgb(251, 251, 251); text-decoration-thickness: initial; text-decoration-style: initial; text-decoration-color: initial; display: inline !important; float: none;&quot;&gt;QUERY_CPU_LIMIT_PERCENT_PER_NODE = 50 = 1.5&lt;/span&gt;&lt;div&gt;&lt;span style=&quot;color: rgb(0, 0, 0); font-family: Helvetica; font-size: 12px; font-style: normal; font-variant-ligatures: normal; font-variant-caps: normal; font-weight: 400; letter-spacing: normal; orphans: 2; text-align: center; text-indent: 0px; text-transform: none; widows: 2; word-spacing: 0px; -webkit-text-stroke-width: 0px; white-space: nowrap; background-color: rgb(251, 251, 251); text-decoration-thickness: initial; text-decoration-style: initial; text-decoration-color: initial; display: inline !important; float: none;&quot;&gt;&lt;br&gt;&lt;/span&gt;&lt;/div&gt;&lt;div style=&quot;text-align: center;&quot;&gt;&lt;span style=&quot;text-wrap: nowrap;&quot;&gt;Новый лимит (&lt;/span&gt;&lt;span style=&quot;text-wrap: nowrap; background-color: initial;&quot;&gt;(RESOURCES_WEIGHT=200)&lt;/span&gt;&lt;span style=&quot;text-wrap: nowrap; background-color: initial;&quot;&gt;:&lt;/span&gt;&lt;/div&gt;&lt;div style=&quot;text-align: center;&quot;&gt;&lt;span style=&quot;text-wrap: nowrap;&quot;&gt;TOTAL_CPU_LIMIT_PERCENT_PER_NODE = 30 = 3 vCPU&lt;/span&gt;&lt;br style=&quot;text-wrap: nowrap;&quot;&gt;&lt;span style=&quot;text-wrap: nowrap;&quot;&gt;QUERY_CPU_LIMIT_PERCENT_PER_NODE = 50 = 1.5 vCPU&lt;/span&gt;&lt;span style=&quot;text-wrap: nowrap;&quot;&gt;&lt;br&gt;&lt;/span&gt;&lt;/div&gt;" style="text;whiteSpace=wrap;html=1;" vertex="1" parent="1">
+ <mxCell id="cv_4T4Yt_JHGhEiulPHa-43" value="&lt;div style=&quot;text-align: center;&quot;&gt;&lt;span style=&quot;color: rgb(0, 0, 0); font-family: Helvetica; font-size: 12px; font-style: normal; font-variant-ligatures: normal; font-variant-caps: normal; font-weight: 400; letter-spacing: normal; orphans: 2; text-indent: 0px; text-transform: none; widows: 2; word-spacing: 0px; -webkit-text-stroke-width: 0px; white-space: nowrap; background-color: rgb(251, 251, 251); text-decoration-thickness: initial; text-decoration-style: initial; text-decoration-color: initial; float: none; display: inline !important;&quot;&gt;Cтарый лимит (&lt;/span&gt;&lt;span style=&quot;text-wrap: nowrap; background-color: initial;&quot;&gt;(RESOURCE_WEIGHT=200)&lt;/span&gt;&lt;span style=&quot;text-wrap: nowrap;&quot;&gt;:&lt;/span&gt;&lt;/div&gt;&lt;span style=&quot;color: rgb(0, 0, 0); font-family: Helvetica; font-size: 12px; font-style: normal; font-variant-ligatures: normal; font-variant-caps: normal; font-weight: 400; letter-spacing: normal; orphans: 2; text-align: center; text-indent: 0px; text-transform: none; widows: 2; word-spacing: 0px; -webkit-text-stroke-width: 0px; white-space: nowrap; background-color: rgb(251, 251, 251); text-decoration-thickness: initial; text-decoration-style: initial; text-decoration-color: initial; display: inline !important; float: none;&quot;&gt;TOTAL_CPU_LIMIT_PERCENT_PER_NODE = 30 = 3 vCPU&lt;/span&gt;&lt;br style=&quot;forced-color-adjust: none; color: rgb(0, 0, 0); font-family: Helvetica; font-size: 12px; font-style: normal; font-variant-ligatures: normal; font-variant-caps: normal; font-weight: 400; letter-spacing: normal; orphans: 2; text-align: center; text-indent: 0px; text-transform: none; widows: 2; word-spacing: 0px; -webkit-text-stroke-width: 0px; white-space: nowrap; background-color: rgb(251, 251, 251); text-decoration-thickness: initial; text-decoration-style: initial; text-decoration-color: initial;&quot;&gt;&lt;span style=&quot;color: rgb(0, 0, 0); font-family: Helvetica; font-size: 12px; font-style: normal; font-variant-ligatures: normal; font-variant-caps: normal; font-weight: 400; letter-spacing: normal; orphans: 2; text-align: center; text-indent: 0px; text-transform: none; widows: 2; word-spacing: 0px; -webkit-text-stroke-width: 0px; white-space: nowrap; background-color: rgb(251, 251, 251); text-decoration-thickness: initial; text-decoration-style: initial; text-decoration-color: initial; display: inline !important; float: none;&quot;&gt;QUERY_CPU_LIMIT_PERCENT_PER_NODE = 50 = 1.5&lt;/span&gt;&lt;div&gt;&lt;span style=&quot;color: rgb(0, 0, 0); font-family: Helvetica; font-size: 12px; font-style: normal; font-variant-ligatures: normal; font-variant-caps: normal; font-weight: 400; letter-spacing: normal; orphans: 2; text-align: center; text-indent: 0px; text-transform: none; widows: 2; word-spacing: 0px; -webkit-text-stroke-width: 0px; white-space: nowrap; background-color: rgb(251, 251, 251); text-decoration-thickness: initial; text-decoration-style: initial; text-decoration-color: initial; display: inline !important; float: none;&quot;&gt;&lt;br&gt;&lt;/span&gt;&lt;/div&gt;&lt;div style=&quot;text-align: center;&quot;&gt;&lt;span style=&quot;text-wrap: nowrap;&quot;&gt;Новый лимит (&lt;/span&gt;&lt;span style=&quot;text-wrap: nowrap; background-color: initial;&quot;&gt;(RESOURCE_WEIGHT=200)&lt;/span&gt;&lt;span style=&quot;text-wrap: nowrap; background-color: initial;&quot;&gt;:&lt;/span&gt;&lt;/div&gt;&lt;div style=&quot;text-align: center;&quot;&gt;&lt;span style=&quot;text-wrap: nowrap;&quot;&gt;TOTAL_CPU_LIMIT_PERCENT_PER_NODE = 30 = 3 vCPU&lt;/span&gt;&lt;br style=&quot;text-wrap: nowrap;&quot;&gt;&lt;span style=&quot;text-wrap: nowrap;&quot;&gt;QUERY_CPU_LIMIT_PERCENT_PER_NODE = 50 = 1.5 vCPU&lt;/span&gt;&lt;span style=&quot;text-wrap: nowrap;&quot;&gt;&lt;br&gt;&lt;/span&gt;&lt;/div&gt;" style="text;whiteSpace=wrap;html=1;" vertex="1" parent="1">
<mxGeometry x="555" y="820" width="360" height="50" as="geometry" />
</mxCell>
</root>
diff --git a/ydb/docs/ru/core/_assets/resources_weight.png b/ydb/docs/ru/core/_assets/resources_weight.png
index 6e84265c170..25e7b38ba52 100644
--- a/ydb/docs/ru/core/_assets/resources_weight.png
+++ b/ydb/docs/ru/core/_assets/resources_weight.png
Binary files differ
diff --git a/ydb/docs/ru/core/_includes/olap-data-types.md b/ydb/docs/ru/core/_includes/olap-data-types.md
deleted file mode 100644
index ca4b37293e0..00000000000
--- a/ydb/docs/ru/core/_includes/olap-data-types.md
+++ /dev/null
@@ -1,24 +0,0 @@
-Тип данных | Можно использовать<br/>в колоночных таблицах | Можно использовать<br/>в качестве первичного ключа
----|---|---
-`Bool` | ✓ | ✓
-`Int8` | ✓ | ☓
-`Int16` | ✓ | ☓
-`Int32` | ✓ | ✓
-`Int64` | ✓ | ✓
-`Uint8` | ✓ | ✓
-`Uint16` | ✓ | ✓
-`Uint32` | ✓ | ✓
-`Uint64` | ✓ | ✓
-`Float` | ✓ | ☓
-`Double` | ✓ | ☓
-`Decimal` | ✓ | ☓
-`String` | ✓ | ✓
-`Utf8` | ✓ | ✓
-`Json` | ✓ | ☓
-`JsonDocument` | ✓ | ☓
-`Yson` | ✓ | ☓
-`Uuid` | ☓ | ☓
-`Date` | ✓ | ✓
-`Datetime` | ✓ | ✓
-`Timestamp` | ✓ | ✓
-`Interval` | ☓ | ☓
diff --git a/ydb/docs/ru/core/concepts/query_execution/topics.md b/ydb/docs/ru/core/concepts/query_execution/topics.md
index 3672673f790..f1a1be7f90d 100644
--- a/ydb/docs/ru/core/concepts/query_execution/topics.md
+++ b/ydb/docs/ru/core/concepts/query_execution/topics.md
@@ -135,7 +135,19 @@ FROM
### Служебные поля {#system-metadata}
-При чтении можно запрашивать служебные поля:
+При чтении можно запрашивать служебные поля и [пользовательские атрибуты сообщений](../datamodel/topic.md#message):
+
+| Поле | [Тип](../../yql/reference/types/index.md) | Описание |
+|------|-----|----------|
+| `__ydb_create_time` | `Timestamp` | Время создания сообщения |
+| `__ydb_write_time` | `Timestamp` | Время записи сообщения в топик |
+| `__ydb_offset` | `Uint64` | Смещение сообщения в партиции |
+| `__ydb_partition_id` | `Uint64` | Номер партиции |
+| `__ydb_message_group_id` | `String` | Идентификатор группы сообщений |
+| `__ydb_seq_no` | `Uint64` | Порядковый номер сообщения внутри группы |
+| `__ydb_user_attributes` | `Dict<String,String>` | [Пользовательские атрибуты сообщения](../datamodel/topic.md#message) |
+
+Пример использования служебных полей:
```yql
SELECT
@@ -165,6 +177,18 @@ WHERE
AND __ydb_write_time > CurrentUtcTimestamp() - Interval("PT2H");
```
+Пример использования пользовательских атрибутов:
+
+```yql
+SELECT
+ COUNT(*) AS ErrorCount
+FROM
+ input_topic -- локальный топик; для внешнего: ext_source.input_topic
+WHERE
+ __ydb_user_attributes["type"] = "log"
+ AND __ydb_user_attributes["level"] = "error";
+```
+
## Запись в топик {#topic-write}
### Запись одного сообщения {#simple-write}
@@ -193,7 +217,7 @@ FROM
{% note warning %}
-Чтение и запись [пользовательских атрибутов](../datamodel/topic.md#message) не поддерживаются.
+Запись [пользовательских атрибутов](../datamodel/topic.md#message) через YQL не поддерживается.
{% endnote %}
diff --git a/ydb/docs/ru/core/concepts/streaming-query/watermarks.md b/ydb/docs/ru/core/concepts/streaming-query/watermarks.md
index 9da73888d6e..aead307dbc2 100644
--- a/ydb/docs/ru/core/concepts/streaming-query/watermarks.md
+++ b/ydb/docs/ru/core/concepts/streaming-query/watermarks.md
@@ -2,9 +2,13 @@
Watermark в потоковой обработке данных ([stream processing](https://en.wikipedia.org/wiki/Stream_processing)) — это монотонно возрастающая нижняя оценка времён событий, которые ещё могут поступить в поток. Когда watermark достигает значения X, система объявляет, что все события с временем меньше X с высокой вероятностью уже получены.
+{{ ydb-short-name }} реализует механизм watermarks в [потоковых запросах](streaming-query.md): они используются для корректного закрытия временных окон агрегации ([HoppingWindow](../../yql/reference/syntax/select/group-by.md#group-by-hopping_window)) и гарантируют, что результат окна выдаётся только тогда, когда система убедилась в полноте входных данных за этот период.
+
В потоковой обработке каждое событие имеет две временны́е метки: **время события** (event time) — момент, когда событие произошло в реальном мире, и **время обработки** (processing time) — момент, когда система получила событие. Из-за сетевых задержек, сбоев и неравномерной нагрузки эти два значения могут существенно расходиться. Именно поэтому системе нужен механизм watermarks: без него она не знает, когда можно считать прошедший временной диапазон достаточно полным, чтобы выдать результат.
-## Компромисс точности и задержки {#tradeoff}
+## Компромисс Точности и Задержки {#tradeoff}
+
+В {{ ydb-short-name }} этот компромисс регулируется явно: параметр [`WATERMARK`](../../dev/streaming-query/watermarks.md#configuration) в предложении `GROUP BY` задаёт выражение для вычисления watermark, в том числе величину отставания от времени последнего события. Это позволяет подобрать баланс между актуальностью результатов и полнотой учёта запоздавших событий.
Watermark не может одновременно учитывать сколь угодно большие задержки событий и продвигаться быстро: чем дольше система ждёт опоздавших событий, тем позже она выдаёт результаты. Это фундаментальный компромисс потоковой обработки.
diff --git a/ydb/docs/ru/core/concepts/topology.md b/ydb/docs/ru/core/concepts/topology.md
index 4339c5ced2d..2b42b5e97c8 100644
--- a/ydb/docs/ru/core/concepts/topology.md
+++ b/ydb/docs/ru/core/concepts/topology.md
@@ -31,6 +31,7 @@
| `mirror-3-dc` *(3 узла)*, переживает отказ 1 узла или 1 дата-центра | 3 | 3 | Сервер | Дата-центр | 3 | Не важно |
| `block-4-2`, переживает отказ 2 стоек | 1.5 | 8 ([10 рекомендовано](*recommended-node-count)) | Стойка | Дата-центр | 1 | 8 |
| `block-4-2` *(упрощённый)*, переживает отказ 1 стойки | 1.5 | 10 | ½ стойки | Дата-центр | 1 | 5 |
+| `block-4-2` *(упрощённый отказоустойчивый)*, переживает отказ 1 узла | 1.5 | 4 | Сервер | Дата-центр | 1 | Не важно |
| `none`, избыточность отсутствует | 1 | 1 | Узел | Узел | 1 | 1 |
{% note info %}
@@ -61,7 +62,11 @@
В случаях, когда невозможно использовать [рекомендованное количество](#cluster-config) оборудования, можно разделить серверы одной стойки на 2 фиктивных домена отказа. В такой конфигурации отказ одной стойки будет означать отказ не одного, а сразу двух доменов. При использовании таких упрощённых конфигураций {{ ydb-short-name }} сохраняет работоспособность при отказе сразу двух доменов. Минимальное количество стоек в кластере для режима `block-4-2` составляет 5, а для `mirror-3-dc` — по 2 в каждом дата-центре (т.е. суммарно 6 стоек).
-Минимальная отказоустойчивая конфигурация кластера {{ ydb-short-name }} представляет собой вариант режима работы `mirror-3-dc` с 3 узлами, который требует всего три сервера с тремя дисками на каждом. В этой конфигурации каждый сервер выступает как домен отказа, так и как область отказа. Таким образом, кластер может выдержать отказ только одного сервера. Каждый сервер должен находиться в своём независимом дата-центре для обеспечения должного уровня отказоустойчивости.
+Есть 2 варианта минимальной отказоустойчивой конфигурации кластера {{ ydb-short-name }}:
+
+- Вариант режима работы `mirror-3-dc` с 3 узлами, который требует всего три сервера с тремя дисками на каждом. В этой конфигурации каждый сервер выступает как домен отказа, так и как область отказа. Таким образом, кластер может выдержать отказ только одного сервера. Каждый сервер должен находиться в своём независимом дата-центре для обеспечения должного уровня отказоустойчивости.
+
+- Вариант режима работы `block-4-2` с 4 узлами, который требует 4 сервера с 2 или более дисками на каждом. В этой конфигурации диски каждого сервера делятся на 2 домена отказа с помощью атрибута [`disk_scope`](../reference/configuration/host_configs.md#disk-scope), что в сумме даёт 8 доменов отказа, необходимых для работы режима `block-4-2`. Такой кластер сохраняет работоспособность при отказе одного сервера.
Кластеры {{ ydb-short-name }}, настроенные с использованием одного из этих подходов, могут использоваться в production окружениях, если в них не требуются повышенные гарантии отказоустойчивости.
diff --git a/ydb/docs/ru/core/dev/resource-consumption-management.md b/ydb/docs/ru/core/dev/resource-consumption-management.md
index 01f41e1638c..c33c75eb4f6 100644
--- a/ydb/docs/ru/core/dev/resource-consumption-management.md
+++ b/ydb/docs/ru/core/dev/resource-consumption-management.md
@@ -22,13 +22,13 @@ CREATE RESOURCE POOL olap WITH (
CONCURRENT_QUERY_LIMIT=10,
QUEUE_SIZE=1000,
DATABASE_LOAD_CPU_THRESHOLD=80,
- RESOURCES_WEIGHT=100,
+ RESOURCE_WEIGHT=100,
QUERY_CPU_LIMIT_PERCENT_PER_NODE=50,
TOTAL_CPU_LIMIT_PERCENT_PER_NODE=70
)
```
-Ознакомиться с полным перечнем параметров пулов ресурсов можно в справке по [{#T}](../yql/reference/syntax/create-resource-pool.md#parameters). Часть параметров является глобальными для всей базы данных (например, `CONCURRENT_QUERY_LIMIT`, `QUEUE_SIZE`, `DATABASE_LOAD_CPU_THRESHOLD`), а другие — применяются только к одному вычислительному узлу (например, `QUERY_CPU_LIMIT_PERCENT_PER_NODE`, `TOTAL_CPU_LIMIT_PERCENT_PER_NODE`, `TOTAL_MEMORY_LIMIT_PERCENT_PER_NODE`). Между всеми пулами может разделяться CPU в случае переподписки на одном вычислительном узле с помощью `RESOURCES_WEIGHT`.
+Ознакомиться с полным перечнем параметров пулов ресурсов можно в справке по [{#T}](../yql/reference/syntax/create-resource-pool.md#parameters). Часть параметров является глобальными для всей базы данных (например, `CONCURRENT_QUERY_LIMIT`, `QUEUE_SIZE`, `DATABASE_LOAD_CPU_THRESHOLD`), а другие — применяются только к одному вычислительному узлу (например, `QUERY_CPU_LIMIT_PERCENT_PER_NODE`, `TOTAL_CPU_LIMIT_PERCENT_PER_NODE`, `TOTAL_MEMORY_LIMIT_PERCENT_PER_NODE`). Между всеми пулами может разделяться CPU в случае переподписки на одном вычислительном узле с помощью `RESOURCE_WEIGHT`.
![resource_pools](../_assets/resource_pool.png)
@@ -64,18 +64,18 @@ Issues:
Как и в случае `CONCURRENT_QUERY_LIMIT`, при превышении указанного порога нагрузки запросы отправляются в очередь ожидания.
-### Распределение ресурсов в соответствии с RESOURCES_WEIGHT {#resources_weight}
+### Распределение ресурсов в соответствии с RESOURCE_WEIGHT {#resources_weight}
![resource_pools](../_assets/resources_weight.png)
-Параметр `RESOURCES_WEIGHT` начинает работать только в случае переподписки и при наличии более одного пула ресурсов в системе. В текущей реализации `RESOURCES_WEIGHT` влияет только на распределение ресурсов `vCPU`. Когда в пуле ресурсов появляются запросы, он начинает участвовать в распределении ресурсов. Для этого в пулах происходит перерасчёт лимитов согласно алгоритму [Max-min fairness](https://en.wikipedia.org/wiki/Max-min_fairness). Само перераспределение ресурсов выполняется на каждом вычислительном узле индивидуально, как показано на рисунке выше.
+Параметр `RESOURCE_WEIGHT` начинает работать только в случае переподписки и при наличии более одного пула ресурсов в системе. В текущей реализации `RESOURCE_WEIGHT` влияет только на распределение ресурсов `vCPU`. Когда в пуле ресурсов появляются запросы, он начинает участвовать в распределении ресурсов. Для этого в пулах происходит перерасчёт лимитов согласно алгоритму [Max-min fairness](https://en.wikipedia.org/wiki/Max-min_fairness). Само перераспределение ресурсов выполняется на каждом вычислительном узле индивидуально, как показано на рисунке выше.
Допустим, у нас есть узел в системе с доступными $10 vCPU$. Установлены ограничения:
- $TOTAL\_CPU\_LIMIT\_PERCENT\_PER\_NODE = 30$,
- $QUERY\_CPU\_LIMIT\_PERCENT\_PER\_NODE = 50$.
-В этом случае у пула ресурсов будет ограничение $3 vCPU$ на узел и $1.5 vCPU$ на один запрос в этом пуле (рисунок *a*). Если в системе существует 4 таких пула, и все они пытаются использовать максимальные ресурсы, это составит $12 vCPU$, что превышает предел доступных ресурсов на узле ($10 vCPU$). В этом случае начинают действовать `RESOURCES_WEIGHT`, и каждому пулу будет выделено по $2.5 vCPU$ (рисунок *b*).
+В этом случае у пула ресурсов будет ограничение $3 vCPU$ на узел и $1.5 vCPU$ на один запрос в этом пуле (рисунок *a*). Если в системе существует 4 таких пула, и все они пытаются использовать максимальные ресурсы, это составит $12 vCPU$, что превышает предел доступных ресурсов на узле ($10 vCPU$). В этом случае начинают действовать `RESOURCE_WEIGHT`, и каждому пулу будет выделено по $2.5 vCPU$ (рисунок *b*).
Если необходимо увеличить выделяемые ресурсы для конкретного пула, можно изменить его вес, например, на 200. Тогда этот пул получит $3 vCPU$, а остальные пулы поделят поровну оставшиеся $7 vCPU$, что составит $\frac{7}{3} vCPU$ на каждый пул (рисунок *c*).
@@ -94,7 +94,7 @@ CREATE RESOURCE POOL default WITH (
CONCURRENT_QUERY_LIMIT=-1,
QUEUE_SIZE=-1,
DATABASE_LOAD_CPU_THRESHOLD=-1,
- RESOURCES_WEIGHT=-1,
+ RESOURCE_WEIGHT=-1,
TOTAL_MEMORY_LIMIT_PERCENT_PER_NODE=-1,
QUERY_CPU_LIMIT_PERCENT_PER_NODE=-1,
TOTAL_CPU_LIMIT_PERCENT_PER_NODE=-1
@@ -136,7 +136,7 @@ WITH (
```
- `RESOURCE_POOL` — имя пула ресурсов, в который будет отправлен запрос, удовлетворяющий требованиям, заданным в классификаторе пула ресурсов.
-- `MEMBER_NAME` — группа пользователей или пользователь, запросы которых будут отправлены в указанный пул ресурсов.
+- `MEMBER_NAME` — группа пользователей или пользователь, запросы которых будут отправлены в указанный пул ресурсов. Подробнее о формате значения см. [CREATE RESOURCE POOL CLASSIFIER](../yql/reference/syntax/create-resource-pool-classifier.md#member-name-format).
## Управление ACL классификатора пула ресурсов
@@ -183,7 +183,7 @@ CREATE RESOURCE POOL olap WITH (
CONCURRENT_QUERY_LIMIT=20,
QUEUE_SIZE=100,
DATABASE_LOAD_CPU_THRESHOLD=80,
- RESOURCES_WEIGHT=20,
+ RESOURCE_WEIGHT=20,
QUERY_CPU_LIMIT_PERCENT_PER_NODE=80,
TOTAL_CPU_LIMIT_PERCENT_PER_NODE=100
);
@@ -191,7 +191,7 @@ CREATE RESOURCE POOL olap WITH (
CREATE RESOURCE POOL the_ceo WITH (
CONCURRENT_QUERY_LIMIT=20,
QUEUE_SIZE=100,
- RESOURCES_WEIGHT=100,
+ RESOURCE_WEIGHT=100,
QUERY_CPU_LIMIT_PERCENT_PER_NODE=100,
TOTAL_CPU_LIMIT_PERCENT_PER_NODE=100
);
diff --git a/ydb/docs/ru/core/dev/streaming-query/watermarks.md b/ydb/docs/ru/core/dev/streaming-query/watermarks.md
index 21ed92747aa..d8ed9c1bab6 100644
--- a/ydb/docs/ru/core/dev/streaming-query/watermarks.md
+++ b/ydb/docs/ru/core/dev/streaming-query/watermarks.md
@@ -4,7 +4,7 @@ Watermark — монотонно возрастающая нижняя оцен�
## Время события {#event-time}
-В потоковой обработке каждое событие имеет временную метку, по которой система отслеживает прогресс времени в потоке. В текущей реализации источником времени события может быть только время записи события в [топик](../../concepts/datamodel/topic.md), доступное через системную колонку `__ydb_write_time`.
+В потоковой обработке каждое событие имеет временную метку, по которой система отслеживает прогресс времени в потоке. В текущей реализации источником времени события может быть только время записи события в [топик](../../concepts/datamodel/topic.md), доступное через системную колонку [`__ydb_write_time`](../../concepts/query_execution/topics.md#system-metadata).
{% note info %}
@@ -35,7 +35,7 @@ sequenceDiagram
## Вычисление watermark {#watermark-computation}
-Когда система получает событие, она обновляет watermark — **продвигает** его вперёд по временной оси. Watermark вычисляется как `максимальное наблюдаемое время события − delay`, где `delay` — величина отставания, заданная в выражении [`WATERMARK`](#configuration) (например, `Interval("PT5S")` в `WATERMARK = __ydb_write_time - Interval("PT5S")`).
+Когда система получает событие, она обновляет watermark — **продвигает** его вперёд по временной оси. Watermark вычисляется как `максимальное наблюдаемое время события − delay`, где `delay` — величина отставания, заданная в выражении [`WATERMARK`](#configuration) (например, `Interval("PT5S")` в `WATERMARK = __ydb_write_time - Interval("PT5S")`; [`__ydb_write_time`](../../concepts/query_execution/topics.md#system-metadata) — служебная колонка топика).
События в потоке могут приходить не в хронологическом порядке: событие с временем 10:00:03 может быть обработано после события с временем 10:00:05. Причины: расхождение часов в распределённой системе, сетевые задержки, неравномерная нагрузка на [партиции](../../concepts/datamodel/topic.md#partitioning) топика.
@@ -61,7 +61,7 @@ Watermarks включаются и настраиваются в секции [W
{% note warning %}
-При использовании [HoppingWindow](../../yql/reference/syntax/select/group-by.md#group-by-hopping_window) первый параметр (time extractor) и источник времени в выражении WATERMARK должны совпадать. В текущей реализации оба должны использовать `__ydb_write_time`.
+При использовании [HoppingWindow](../../yql/reference/syntax/select/group-by.md#group-by-hopping_window) первый параметр (time extractor) и источник времени в выражении WATERMARK должны совпадать. В текущей реализации оба должны использовать [`__ydb_write_time`](../../concepts/query_execution/topics.md#system-metadata).
{% endnote %}
@@ -120,7 +120,7 @@ END DO;
Где:
- [`CREATE STREAMING QUERY`](../../yql/reference/syntax/create-streaming-query.md) — создаёт именованный потоковый запрос.
-- `__ydb_write_time` — системная колонка, содержащая время записи события в [топик](../../concepts/datamodel/topic.md).
+- [`__ydb_write_time`](../../concepts/query_execution/topics.md#system-metadata) — системная колонка, содержащая время записи события в [топик](../../concepts/datamodel/topic.md).
- `FORMAT = json_each_row` — [формат данных](streaming-query-formats.md) в топике, каждая строка содержит отдельный JSON-объект.
- `WATERMARK = __ydb_write_time - Interval("PT5S")` — watermark с отставанием 5 секунд. `Interval("PT5S")` задаёт интервал в формате [ISO 8601](https://en.wikipedia.org/wiki/ISO_8601#Durations).
- [`AGGREGATE_LIST`](../../yql/reference/builtins/aggregation.md#agg-list) — агрегатная функция, собирающая значения в список.
diff --git a/ydb/docs/ru/core/devops/deployment-options/manual/initial-deployment/deployment-configuration-v1.md b/ydb/docs/ru/core/devops/deployment-options/manual/initial-deployment/deployment-configuration-v1.md
index 8efa15c447b..433145ac6e4 100644
--- a/ydb/docs/ru/core/devops/deployment-options/manual/initial-deployment/deployment-configuration-v1.md
+++ b/ydb/docs/ru/core/devops/deployment-options/manual/initial-deployment/deployment-configuration-v1.md
@@ -403,53 +403,51 @@
- drive:
- path: /dev/disk/by-partlabel/ydb_disk_ssd_01
type: SSD
- - path: /dev/disk/by-partlabel/ydb_disk_ssd_02
- type: SSD
host_config_id: 1
hosts:
- - host: ydb-node-zone-a-1.local
+ - host: static-node-1.ydb-cluster.com
host_config_id: 1
walle_location:
body: 1
data_center: 'zone-a'
rack: '1'
- - host: ydb-node-zone-a-2.local
+ - host: static-node-2.ydb-cluster.com
host_config_id: 1
walle_location:
body: 2
data_center: 'zone-a'
rack: '2'
- - host: ydb-node-zone-a-3.local
+ - host: static-node-3.ydb-cluster.com
host_config_id: 1
walle_location:
body: 3
data_center: 'zone-a'
rack: '3'
- - host: ydb-node-zone-a-4.local
+ - host: static-node-4.ydb-cluster.com
host_config_id: 1
walle_location:
body: 4
data_center: 'zone-a'
rack: '4'
- - host: ydb-node-zone-a-5.local
+ - host: static-node-5.ydb-cluster.com
host_config_id: 1
walle_location:
body: 5
data_center: 'zone-a'
rack: '5'
- - host: ydb-node-zone-a-6.local
+ - host: static-node-6.ydb-cluster.com
host_config_id: 1
walle_location:
body: 6
data_center: 'zone-a'
rack: '6'
- - host: ydb-node-zone-a-7.local
+ - host: static-node-7.ydb-cluster.com
host_config_id: 1
walle_location:
body: 7
data_center: 'zone-a'
rack: '7'
- - host: ydb-node-zone-a-8.local
+ - host: static-node-8.ydb-cluster.com
host_config_id: 1
walle_location:
body: 8
@@ -458,11 +456,21 @@
domains_config:
security_config:
enforce_user_token_requirement: true
- default_users:
- - name: "root"
- password: ""
- default_access:
- - "+(F):root"
+ monitoring_allowed_sids:
+ - "root"
+ - "ADMINS"
+ - "DATABASE-ADMINS"
+ administration_allowed_sids:
+ - "root"
+ - "ADMINS"
+ - "DATABASE-ADMINS"
+ viewer_allowed_sids:
+ - "root"
+ - "ADMINS"
+ - "DATABASE-ADMINS"
+ register_dynamic_node_allowed_sids:
+ - databaseNodes@cert
+ - root@builtin
domain:
- name: Root
storage_pool_types:
@@ -513,55 +521,55 @@
rings:
- fail_domains:
- vdisk_locations:
- - node_id: "ydb-node-zone-a-1.local"
+ - node_id: "static-node-1.ydb-cluster.com"
pdisk_category: SSD
path: /dev/disk/by-partlabel/ydb_disk_ssd_01
- fail_domains:
- vdisk_locations:
- - node_id: "ydb-node-zone-a-2.local"
+ - node_id: "static-node-2.ydb-cluster.com"
pdisk_category: SSD
path: /dev/disk/by-partlabel/ydb_disk_ssd_01
- fail_domains:
- vdisk_locations:
- - node_id: "ydb-node-zone-a-3.local"
+ - node_id: "static-node-3.ydb-cluster.com"
pdisk_category: SSD
path: /dev/disk/by-partlabel/ydb_disk_ssd_01
- fail_domains:
- vdisk_locations:
- - node_id: "ydb-node-zone-a-4.local"
+ - node_id: "static-node-4.ydb-cluster.com"
pdisk_category: SSD
path: /dev/disk/by-partlabel/ydb_disk_ssd_01
- fail_domains:
- vdisk_locations:
- - node_id: "ydb-node-zone-a-5.local"
+ - node_id: "static-node-5.ydb-cluster.com"
pdisk_category: SSD
path: /dev/disk/by-partlabel/ydb_disk_ssd_01
- fail_domains:
- vdisk_locations:
- - node_id: "ydb-node-zone-a-6.local"
+ - node_id: "static-node-6.ydb-cluster.com"
pdisk_category: SSD
path: /dev/disk/by-partlabel/ydb_disk_ssd_01
- fail_domains:
- vdisk_locations:
- - node_id: "ydb-node-zone-a-7.local"
+ - node_id: "static-node-7.ydb-cluster.com"
pdisk_category: SSD
path: /dev/disk/by-partlabel/ydb_disk_ssd_01
- fail_domains:
- vdisk_locations:
- - node_id: "ydb-node-zone-a-8.local"
+ - node_id: "static-node-8.ydb-cluster.com"
pdisk_category: SSD
path: /dev/disk/by-partlabel/ydb_disk_ssd_01
channel_profile_config:
profile:
- channel:
- erasure_species: block-4-2
- pdisk_category: 1
+ pdisk_category: 0
storage_pool_kind: ssd
- erasure_species: block-4-2
- pdisk_category: 1
+ pdisk_category: 0
storage_pool_kind: ssd
- erasure_species: block-4-2
- pdisk_category: 1
+ pdisk_category: 0
storage_pool_kind: ssd
profile_id: 0
interconnect_config:
@@ -579,7 +587,7 @@
client_certificate_authorization:
request_client_certificate: true
client_certificate_definitions:
- - member_groups: ["registerNode@cert"]
+ - member_groups: ["databaseNodes@cert"]
subject_terms:
- short_name: "O"
values: ["YDB"]
diff --git a/ydb/docs/ru/core/devops/enterprise-manager/ai-assistant.md b/ydb/docs/ru/core/devops/enterprise-manager/ai-assistant.md
new file mode 100644
index 00000000000..e61f084664d
--- /dev/null
+++ b/ydb/docs/ru/core/devops/enterprise-manager/ai-assistant.md
@@ -0,0 +1,205 @@
+# Настройка AI-ассистента в YDB EM
+
+В этой инструкции описано, как включить AI-ассистента в {{ ydb-short-name }} Enterprise Manager (YDB EM). После настройки пользователи увидят ассистента в веб-интерфейсе YDB EM. Ассистент отправляет запросы к модели через [Gateway](index.md#architecture) и может использовать инструменты Model Context Protocol (MCP), предоставляемые Gateway.
+
+## Перед началом {#before-start}
+
+Эту инструкцию можно использовать до первого развёртывания YDB EM или для уже работающей установки. При первом развёртывании добавьте переменные в inventory до запуска initial setup playbook. Инструкцию по развёртыванию см. в разделе [{#T}](initial-deployment.md).
+
+Убедитесь, что у вас есть:
+
+1. Доступ к Ansible inventory, который используется для развёртывания YDB EM.
+1. Endpoint OpenAI-compatible модели, который будет доступен с хоста Gateway.
+1. Права на изменение [token-файла Gateway](#configure-model-access). В рамках настройки будет необходимо прописать секрет для доступа к модели в token-файл. Gateway читает этот файл и использует значение поля `Token` записи, у которой поле `Name` равно `ydb_em_ai_model_token_name`, например `model-token`, как upstream-заголовок `Authorization`.
+1. Если вы обновляете существующую установку, доступ к развёрнутому хосту Gateway.
+
+Для уже работающей установки сначала убедитесь, что хост Gateway управляется тем же Ansible inventory. Перед запуском playbook проверьте активный service manager unit, владельца процесса, путь к конфигурации, token-файл и порт, который слушает Gateway. Если на хосте используется custom или ручной layout Gateway, применяйте те же настройки через операционную процедуру этой установки, а не запускайте playbook вслепую.
+
+{% note warning %}
+
+Не помещайте API-ключи модели, OAuth-токены и другие секреты в `ydb_em_ai_assistant_client_runtime_config`. Gateway возвращает это значение в браузер через `GET /meta/ai_assistant_client_config`. Секреты держите в token-файле.
+
+{% endnote %}
+
+## Как это работает {#how-it-works}
+
+Браузер работает с ассистентом через Gateway:
+
+1. UI проверяет `GET /capabilities`. Кнопка ассистента отображается, если `Settings.Proxy.Model` равно `true` и включена пользовательская настройка AI-ассистента. Когда backend capability появляется впервые, YDB EM инициализирует эту пользовательскую настройку включённой.
+1. UI получает runtime-настройки из `GET /meta/ai_assistant_client_config`.
+1. UI отправляет запросы к модели в `/proxy/model/...`.
+1. Gateway перенаправляет запросы к модели в `ydb_em_ai_model_endpoint` и добавляет заголовок Authorization из tokenator.
+1. UI отправляет MCP-запросы в `/meta/mcp`. Gateway публикует MCP-инструменты через этот endpoint и выполняет их через API YDB EM и proxy endpoints кластеров. Если поиск по документации настроен, `search_docs` публикуется через тот же MCP endpoint и доступен ассистенту.
+
+## Настройте доступ к модели {#configure-model-access}
+
+`ydb_em_ai_model_token_name` — это имя записи [tokenator](../../concepts/glossary.md#tokenator), которую Gateway использует для запросов к модели.
+
+Пример записи tokenator:
+
+```textproto
+StaticTokenInfo {
+ Name: "model-token"
+ Token: "Bearer <model-api-token>"
+}
+```
+
+Способ доставки токена зависит от вашего процесса развёртывания. Главное требование: tokenator-файл Gateway содержит запись, у которой `Name` совпадает с `ydb_em_ai_model_token_name`.
+
+{% note info %}
+
+Стандартный Ansible-шаблон токенов YDB EM генерирует только запись `meta` для доступа к metabase. Если в вашей установке используется этот шаблон без изменений, расширьте или переопределите tokenator-файл через ваш процесс развёртывания, чтобы в нём также были `model-token` и, если он отличается, токен для embeddings.
+
+{% endnote %}
+
+## Настройте Ansible-переменные {#configure-ansible-variables}
+
+Добавьте переменные AI-ассистента в inventory YDB EM, например в `examples/inventory/50-inventory.yaml` или другой inventory-файл вашего развёртывания.
+
+```yaml
+ydb_em_ai_assistant_enabled: true
+ydb_em_ai_model_endpoint: "https://llm.example.com"
+ydb_em_ai_model_token_name: "model-token"
+
+ydb_em_ai_assistant_client_runtime_config:
+ llm:
+ baseURL: "/proxy/model/v1"
+ model: "<model-name>"
+ temperature: 0.2
+ mcp:
+ - id: "ydb-meta"
+ url: "/meta/mcp"
+ transport: "streamable"
+ toolCallTimeoutMs: 600000
+```
+
+Параметр | Описание
+--- | ---
+`ydb_em_ai_assistant_enabled` | Включает настройки AI-ассистента в Gateway.
+`ydb_em_ai_model_endpoint` | Upstream model API endpoint, который использует Gateway.
+`ydb_em_ai_model_token_name` | Имя записи tokenator для запросов к модели.
+`ydb_em_ai_assistant_client_runtime_config.llm.baseURL` | URL модели, который видит браузер. Используйте Gateway proxy, обычно `/proxy/model/v1`.
+`ydb_em_ai_assistant_client_runtime_config.llm.model` | Имя модели, которое отправляется в OpenAI-compatible API.
+`ydb_em_ai_assistant_client_runtime_config.mcp` | MCP-серверы, доступные ассистенту. Для YDB EM используйте `/meta/mcp`.
+
+Gateway добавляет путь из `/proxy/model/...` к `ydb_em_ai_model_endpoint`. При `baseURL: "/proxy/model/v1"` запрос chat completions будет перенаправлен в `/v1/chat/completions` на upstream endpoint. Настройте endpoint так, чтобы этот путь был корректен для вашего provider.
+
+## Настройте поиск по документации {#configure-docs-search}
+
+Рекомендуется включить поиск по документации, чтобы ассистент получил MCP-инструмент `search_docs`. Для этого инструмента Gateway вызывает OpenAI-compatible embeddings endpoint и добавляет `/embeddings` к настроенному base URL, если suffix отсутствует. Когда поиск по документации включён, ассистент получает `search_docs` через настроенный MCP-сервер `/meta/mcp`.
+
+```yaml
+ydb_em_docs_search_enabled: true
+ydb_em_docs_search_embeddings_upstream_base_url: "https://llm.example.com/v1"
+ydb_em_docs_search_embeddings_token_name: "model-token"
+ydb_em_docs_search_embeddings_model: "<embeddings-model-name>"
+ydb_em_docs_search_vector_size: 0
+ydb_em_docs_search_limit: 6
+ydb_em_docs_search_score: 0.6
+```
+
+Параметр | Описание
+--- | ---
+`ydb_em_docs_search_enabled` | Настраивает backend семантического поиска по документации и публикует `search_docs` через MCP, если все обязательные настройки поиска валидны.
+`ydb_em_docs_search_embeddings_upstream_base_url` | Базовый URL OpenAI-совместимого endpoint для embeddings.
+`ydb_em_docs_search_embeddings_token_name` | Имя записи tokenator для embeddings-запросов.
+`ydb_em_docs_search_embeddings_model` | Имя embeddings-модели.
+`ydb_em_docs_search_vector_size` | Необязательное переопределение размерности вектора эмбеддинга. Не задавайте переменную, чтобы использовать Ansible default `0`. При `0` Gateway не отправляет OpenAI-compatible поле `dimensions` и отключает проверку размера ответа. Положительное значение задавайте только если embeddings provider и модель поддерживают explicit dimensions и ожидаемый размер известен.
+`ydb_em_docs_search_limit` | Максимальное количество возвращаемых документов.
+`ydb_em_docs_search_score` | Минимальная оценка релевантности от `0` до `1`. Документы с меньшей оценкой не возвращаются.
+
+## Примените конфигурацию {#apply-configuration}
+
+После обновления inventory запустите playbook YDB EM из каталога с вашим inventory. Один и тот же playbook используется для первого развёртывания и для применения этого изменения к существующей установке:
+
+```bash
+ansible-playbook ydb_platform.ydb_em.initial_setup
+```
+
+Если vault-файл зашифрован, добавьте vault-параметр, который используется в вашем развёртывании:
+
+```bash
+ansible-playbook ydb_platform.ydb_em.initial_setup --ask-vault-pass
+```
+
+Роль сгенерирует конфигурацию и стандартный token-файл Gateway и убедится, что сервис Gateway запущен. Если дополнительные учётные данные модели доставляются через переопределённый token-файл или отдельный шаг развёртывания, убедитесь, что развёрнутый token-файл содержит настроенное имя записи. Если вы применяете эти настройки к уже работающему развёртыванию, перезапустите Gateway по вашей операционной процедуре, если ваш Ansible-запуск не перезапускает сервис после изменения конфигурации или token-файла.
+
+## Проверьте настройку {#verify}
+
+Откройте веб-интерфейс YDB EM:
+
+```text
+https://<gateway-host>:8789/ui/clusters
+```
+
+Проверьте, что model proxy включён:
+
+```bash
+curl -k https://<gateway-host>:8789/capabilities
+```
+
+Ответ должен содержать:
+
+```json
+{
+ "Settings": {
+ "Proxy": {
+ "Model": true
+ }
+ }
+}
+```
+
+Проверьте runtime-конфигурацию, которую получает UI:
+
+```bash
+curl -k https://<gateway-host>:8789/meta/ai_assistant_client_config
+```
+
+Ответ должен содержать блок `llm` и не должен содержать секреты модели.
+
+Проверьте model proxy:
+
+```bash
+curl -k https://<gateway-host>:8789/proxy/model/v1/chat/completions \
+ -H 'Content-Type: application/json' \
+ -d '{"model":"<model-name>","messages":[{"role":"user","content":"ping"}],"max_tokens":1}'
+```
+
+Если в вашем развёртывании Gateway требует аутентификацию пользователя, добавьте те же authentication headers, которые используются для UI YDB EM.
+
+Если включён поиск по документации, также проверьте, что `/meta/mcp` публикует `search_docs`, и выполните безопасный поисковый запрос через ваш MCP-клиент или tooling ассистента. Это проверяет и регистрацию MCP-инструмента, и путь embeddings/metabase.
+
+## Устранение неполадок {#troubleshooting}
+
+### Кнопка ассистента не отображается {#button-not-shown}
+
+Проверьте, что `ydb_em_ai_assistant_enabled` равно `true`, а `/capabilities` содержит `Settings.Proxy.Model: true`. Также проверьте пользовательскую настройку AI-ассистента в UI YDB EM: YDB EM инициализирует её включённой, когда backend capability присутствует, но пользователь может её выключить.
+
+### Runtime-конфигурация невалидна {#runtime-config-invalid}
+
+Проверьте `GET /meta/ai_assistant_client_config`. Для включённого ассистента ответ должен быть JSON-объектом со строковыми полями `llm.baseURL` и `llm.model`. Ответ `null` означает, что Gateway не загрузил включённую клиентскую конфигурацию AI-ассистента.
+
+### Model proxy возвращает ошибку авторизации {#model-authorization-error}
+
+Проверьте, что `ydb_em_ai_model_token_name` совпадает с именем записи tokenator, а запись возвращает заголовок Authorization, который ожидает endpoint модели. Если tokenator не находит запись, Gateway отправляет upstream-запрос без ожидаемого заголовка Authorization.
+
+### Model proxy вызывает неправильный upstream path {#wrong-upstream-path}
+
+Проверьте `ydb_em_ai_model_endpoint` вместе с `llm.baseURL`. Gateway добавляет путь из `/proxy/model/...` к настроенному endpoint. Дублирование `/v1` часто приводит к upstream `404`.
+
+### MCP-инструменты недоступны {#mcp-tools-unavailable}
+
+Проверьте, что runtime-конфигурация содержит `url: "/meta/mcp"` и что пользовательские запросы могут обращаться к `/meta/mcp` через Gateway. Gateway регистрирует `/meta/mcp` только если MCP включён; по умолчанию он включён, но custom Gateway config может его отключить.
+
+### Поиск по документации недоступен {#docs-search-unavailable}
+
+Проверьте настройки `ydb_em_docs_search_*`, путь embeddings endpoint и запись tokenator для embeddings-запросов. Также убедитесь, что documentation vectors доступны в metabase YDB EM. Если поиск по документации не настроен, инструмент `search_docs` не публикуется.
+
+### Embeddings endpoint отклоняет размер вектора {#embeddings-vector-size}
+
+Если `search_docs` падает с `variable embedding size not supported`, установите `ydb_em_docs_search_vector_size: 0` и проверьте embeddings endpoint и модель.
+
+### Поиск по документации возвращает 504 {#docs-search-504}
+
+Если `search_docs` возвращает `504 Gateway Timeout`, а в логах Gateway для `/meta/docs` есть `Member not found: doc_id`, проверьте таблицу metabase `ydb/Docs.db`. Текущий запрос Gateway docs search использует колонки `doc_id`, `embedding` и `payload`. Для полезных результатов в таблице также должны быть строки, а сама таблица должна быть построена для embeddings model, совместимой с `ydb_em_docs_search_embeddings_model`.
diff --git a/ydb/docs/ru/core/devops/enterprise-manager/index.md b/ydb/docs/ru/core/devops/enterprise-manager/index.md
index 3fda6e30032..bbc397de54a 100644
--- a/ydb/docs/ru/core/devops/enterprise-manager/index.md
+++ b/ydb/docs/ru/core/devops/enterprise-manager/index.md
@@ -81,3 +81,4 @@ flowchart LR
- [{#T}](initial-deployment.md)
- [{#T}](s3-backups.md)
+- [{#T}](ai-assistant.md)
diff --git a/ydb/docs/ru/core/devops/enterprise-manager/toc_p.yaml b/ydb/docs/ru/core/devops/enterprise-manager/toc_p.yaml
index d5d01a59696..58cc63c869c 100644
--- a/ydb/docs/ru/core/devops/enterprise-manager/toc_p.yaml
+++ b/ydb/docs/ru/core/devops/enterprise-manager/toc_p.yaml
@@ -3,3 +3,5 @@ items:
href: initial-deployment.md
- name: Настройка резервного копирования в S3
href: s3-backups.md
+- name: Настройка AI-ассистента
+ href: ai-assistant.md
diff --git a/ydb/docs/ru/core/downloads/ydb-ansible.md b/ydb/docs/ru/core/downloads/ydb-ansible.md
index d424f9e3a1c..5b9f36e95a7 100644
--- a/ydb/docs/ru/core/downloads/ydb-ansible.md
+++ b/ydb/docs/ru/core/downloads/ydb-ansible.md
@@ -4,6 +4,7 @@
| Версия | Дата выпуска | Скачать | Список изменений |
| ------ | ------------ | ------- | ----------------- |
+| v2.1.0 | 29.06.2026 | [ydb-ansible-2.1.0.tar.gz](https://github.com/ydb-platform/ydb-ansible/releases/download/v2.1.0/ydb_platform-ydb-2.1.0.tar.gz) | |
| v2.0.0 | 23.12.2025 | [ydb-ansible-2.0.0.tar.gz](https://github.com/ydb-platform/ydb-ansible/releases/download/v2.0.0/ydb_platform-ydb-2.0.0.tar.gz) | |
| v1.3.2 | 02.12.2025 | [ydb-ansible-1.3.2.tar.gz](https://github.com/ydb-platform/ydb-ansible/releases/download/v1.3.2/ydb_platform-ydb-1.3.2.tar.gz) | |
| v1.3.1 | 01.12.2025 | [ydb-ansible-1.3.1.tar.gz](https://github.com/ydb-platform/ydb-ansible/releases/download/v1.3.1/ydb_platform-ydb-1.3.1.tar.gz) | |
diff --git a/ydb/docs/ru/core/recipes/ydb-sdk/debug-jaeger.md b/ydb/docs/ru/core/recipes/ydb-sdk/debug-jaeger.md
deleted file mode 100644
index 591aa6cb092..00000000000
--- a/ydb/docs/ru/core/recipes/ydb-sdk/debug-jaeger.md
+++ /dev/null
@@ -1,172 +0,0 @@
-# Включение трассировки в Jaeger
-
-Ниже приведены примеры кода для включения трассировки в Jaeger в различных {{ ydb-short-name }} SDK.
-
-{% list tabs %}
-
-- C++
-
- Функциональность в настоящее время не поддерживается.
-
-- Go
-
- {% list tabs %}
-
- - Нативный SDK
-
- ```go
- package main
-
- import (
- "context"
- "time"
-
- "github.com/opentracing/opentracing-go"
- jaegerConfig "github.com/uber/jaeger-client-go/config"
-
- "github.com/ydb-platform/ydb-go-sdk/v3"
- "github.com/ydb-platform/ydb-go-sdk/v3/trace"
-
- tracing "github.com/ydb-platform/ydb-go-sdk-opentracing"
- )
-
- const (
- tracerURL = "localhost:5775"
- serviceName = "ydb-go-sdk"
- )
-
- func main() {
- tracer, closer, err := jaegerConfig.Configuration{
- ServiceName: serviceName,
- Sampler: &jaegerConfig.SamplerConfig{
- Type: "const",
- Param: 1,
- },
- Reporter: &jaegerConfig.ReporterConfig{
- LogSpans: true,
- BufferFlushInterval: 1 * time.Second,
- LocalAgentHostPort: tracerURL,
- },
- }.NewTracer()
- if err != nil {
- panic(err)
- }
-
- defer closer.Close()
-
- // set global tracer of this application
- opentracing.SetGlobalTracer(tracer)
-
- span, ctx := opentracing.StartSpanFromContext(context.Background(), "client")
- defer span.Finish()
-
- db, err := ydb.Open(ctx,
- os.Getenv("YDB_CONNECTION_STRING"),
- tracing.WithTraces(tracing.WithDetails(trace.DetailsAll)),
- )
- if err != nil {
- panic(err)
- }
- defer db.Close(ctx)
- ...
- }
- ```
-
- - database/sql
-
- ```go
- package main
-
- import (
- "context"
- "database/sql"
- "time"
-
- "github.com/opentracing/opentracing-go"
- jaegerConfig "github.com/uber/jaeger-client-go/config"
-
- "github.com/ydb-platform/ydb-go-sdk/v3"
- "github.com/ydb-platform/ydb-go-sdk/v3/trace"
-
- tracing "github.com/ydb-platform/ydb-go-sdk-opentracing"
- )
-
- const (
- tracerURL = "localhost:5775"
- serviceName = "ydb-go-sdk"
- )
-
- func main() {
- tracer, closer, err := jaegerConfig.Configuration{
- ServiceName: serviceName,
- Sampler: &jaegerConfig.SamplerConfig{
- Type: "const",
- Param: 1,
- },
- Reporter: &jaegerConfig.ReporterConfig{
- LogSpans: true,
- BufferFlushInterval: 1 * time.Second,
- LocalAgentHostPort: tracerURL,
- },
- }.NewTracer()
- if err != nil {
- panic(err)
- }
-
- defer closer.Close()
-
- // set global tracer of this application
- opentracing.SetGlobalTracer(tracer)
-
- span, ctx := opentracing.StartSpanFromContext(context.Background(), "client")
- defer span.Finish()
-
- nativeDriver, err := ydb.Open(ctx,
- os.Getenv("YDB_CONNECTION_STRING"),
- tracing.WithTraces(tracing.WithDetails(trace.DetailsAll)),
- )
- if err != nil {
- panic(err)
- }
- defer nativeDriver.Close(ctx)
-
- connector, err := ydb.Connector(nativeDriver)
- if err != nil {
- panic(err)
- }
-
- db := sql.OpenDB(connector)
- defer db.Close()
- ...
- }
- ```
-
- {% endlist %}
-
-- Java
-
- {% include [feature-not-supported](../../_includes/feature-not-supported.md) %}
-
-- Python
-
- {% include [feature-not-supported](../../_includes/feature-not-supported.md) %}
-
-- C#
-
- {% include [feature-not-supported](../../_includes/feature-not-supported.md) %}
-
-- JavaScript
-
- {% include [feature-not-supported](../../_includes/feature-not-supported.md) %}
-
-- Rust
-
- {% include [feature-not-supported](../../_includes/feature-not-supported.md) %}
-
- Используйте экосистему [`tracing`](https://docs.rs/tracing) и экспорт OpenTelemetry ([#268](https://github.com/ydb-platform/ydb-rs-sdk/issues/268)).
-
-- PHP
-
- {% include [feature-not-supported](../../_includes/feature-not-supported.md) %}
-
-{% endlist %}
diff --git a/ydb/docs/ru/core/recipes/ydb-sdk/debug-logs-otel.md b/ydb/docs/ru/core/recipes/ydb-sdk/debug-logs-otel.md
deleted file mode 100644
index 8b294c69ac2..00000000000
--- a/ydb/docs/ru/core/recipes/ydb-sdk/debug-logs-otel.md
+++ /dev/null
@@ -1,352 +0,0 @@
-# Экспорт логов в OpenTelemetry
-
-Каждый {{ ydb-short-name }} SDK записывает свои внутренние логи (инициализация драйвера, пул сессий, выполнение запросов, повторные попытки и т.д.) через стандартный механизм логирования своего языка. Вместо вывода логов только в консоль (см. [Включение логирования](debug-logs.md)), их можно перенаправить в [OpenTelemetry](https://opentelemetry.io/) Logs SDK и экспортировать по стандартному протоколу OTLP в коллектор. Затем коллектор пересылает записи в выбранный бэкенд для хранения и просмотра логов.
-
-Принцип одинаков для всех SDK:
-
-1. Создайте OpenTelemetry `LoggerProvider` с экспортером логов OTLP и атрибутом ресурса `service.name`.
-2. Перенаправьте логгер SDK в этот провайдер через адаптер (log appender / bridge), который преобразует каждую запись лога SDK в запись лога OTel. Список адаптеров и статус поддержки логов в разных языках см. в [документации по логам OpenTelemetry](https://opentelemetry.io/docs/concepts/signals/logs/).
-3. Запустите рабочую нагрузку — теперь все внутренние логи SDK отправляются в коллектор как записи логов OTLP.
-
-{% note info %}
-
-## Принципы {#principles}
-
-Метод сбора логов зависит от языка и инфраструктуры — единого формата нет:
-
-* **Программные мосты (log appender / bridge).** Приложение само отправляет логи в коллектор по OTLP непосредственно из процесса (рабочий процесс "direct-to-Collector"). Именно этот подход показан в примерах ниже.
-* **Агенты коллектора.** Приложение записывает логи в файл или `stdout`, а отдельный агент (например, [filelog receiver](https://github.com/open-telemetry/opentelemetry-collector-contrib/tree/main/receiver/filelogreceiver) в [OpenTelemetry Collector](https://opentelemetry.io/docs/collector/)) читает, разбирает и пересылает их в бэкенд — без изменения кода приложения.
-
-{% endnote %}
-
-## Подключение к SDK {#integration}
-
-{% list tabs %}
-
-- Go
-
- Для {{ ydb-short-name }} Go SDK существует готовый адаптер [ydb-go-sdk-otel](https://github.com/ydb-platform/ydb-go-sdk-otel), который преобразует события SDK в сигналы OpenTelemetry: трассы (`WithTracer`), метрики (`WithMetrics`) и логи (`WithLogger`). Адаптер сам не настраивает экспортеры — вы создаете `LoggerProvider` с OTLP-экспортером логов, получаете из него логгер и передаете его в опцию `ydbOtel.WithLogger` при вызове `ydb.Open`. Каждая внутренняя запись лога SDK (инициализация драйвера, пул сессий, выполнение запросов, повторные попытки и т.д.) затем отправляется в коллектор как запись лога OTLP.
-
-
- ```bash
- go get github.com/ydb-platform/ydb-go-sdk-otel
- go get go.opentelemetry.io/otel/sdk/log
- go get go.opentelemetry.io/otel/exporters/otlp/otlplog/otlploggrpc
- ```
-
-
- ```go
- package main
-
- import (
- "context"
- "os"
-
- "github.com/ydb-platform/ydb-go-sdk/v3"
- "github.com/ydb-platform/ydb-go-sdk/v3/trace"
-
- "go.opentelemetry.io/otel/attribute"
- "go.opentelemetry.io/otel/exporters/otlp/otlplog/otlploggrpc"
- sdklog "go.opentelemetry.io/otel/sdk/log"
- "go.opentelemetry.io/otel/sdk/resource"
-
- ydbOtel "github.com/ydb-platform/ydb-go-sdk-otel"
- )
-
- func main() {
- ctx := context.Background()
-
- // 1. Configuring the OTel log provider with an OTLP exporter.
- exporter, err := otlploggrpc.New(ctx,
- otlploggrpc.WithEndpoint("localhost:4317"),
- otlploggrpc.WithInsecure(),
- )
- if err != nil {
- panic(err)
- }
- res, _ := resource.Merge(resource.Default(), resource.NewSchemaless(
- attribute.String("service.name", "ydb-go-sdk-otel-logs-sample"),
- ))
- lp := sdklog.NewLoggerProvider(
- sdklog.WithResource(res),
- sdklog.WithProcessor(sdklog.NewBatchProcessor(exporter)),
- )
- defer lp.Shutdown(ctx)
-
- // 2. Opening the YDB driver with the ydb-go-sdk-otel adapter.
- // WithLogger forwards SDK log events to OTel log records.
- logger := lp.Logger("ydb-go-sdk")
- db, err := ydb.Open(ctx,
- os.Getenv("YDB_CONNECTION_STRING"),
- ydbOtel.WithLogger(logger, ydbOtel.WithDetailer(trace.DetailsAll)),
- )
- if err != nil {
- panic(err)
- }
- defer db.Close(ctx)
- // ... use db ...
- }
- ```
-
-- Python
-
- {{ ydb-short-name }} Python SDK записывает логи через стандартный модуль `logging` (логгеры с именем `ydb.*`). Используйте встроенный `LoggingHandler` OpenTelemetry для перенаправления этих записей в `LoggerProvider` и экспорта через OTLP:
-
-
- ```bash
- pip install opentelemetry-sdk opentelemetry-exporter-otlp-proto-grpc
- ```
-
-
- ```python
- import logging
-
- import ydb
- from opentelemetry._logs import set_logger_provider
- from opentelemetry.exporter.otlp.proto.grpc._log_exporter import OTLPLogExporter
- from opentelemetry.sdk._logs import LoggerProvider, LoggingHandler
- from opentelemetry.sdk._logs.export import BatchLogRecordProcessor
- from opentelemetry.sdk.resources import Resource
-
- resource = Resource(attributes={"service.name": "ydb-otel-logs-example"})
- logger_provider = LoggerProvider(resource=resource)
- logger_provider.add_log_record_processor(
- BatchLogRecordProcessor(OTLPLogExporter(endpoint="http://localhost:4317"))
- )
- set_logger_provider(logger_provider)
-
- # Bridge stdlib logging -> OpenTelemetry. Binding a handler to the root logger
- # intercepts everything the SDK writes via logging.getLogger("ydb...").
- otel_handler = LoggingHandler(level=logging.NOTSET, logger_provider=logger_provider)
- logging.basicConfig(level=logging.INFO, handlers=[otel_handler])
- logging.getLogger("ydb").setLevel(logging.INFO)
-
- with ydb.Driver(endpoint="grpc://localhost:2136", database="/local") as driver:
- driver.wait(timeout=5)
- with ydb.QuerySessionPool(driver) as pool:
- pool.execute_with_retries("SELECT 1")
-
- logger_provider.shutdown()
- ```
-
-- C#
-
- {{ ydb-short-name }} C# SDK записывает логи через переданный ему `ILoggerFactory`. Создайте фабрику на основе провайдера логирования OpenTelemetry с экспортером OTLP и передайте ее источнику данных:
-
-
- ```bash
- dotnet add package OpenTelemetry
- dotnet add package OpenTelemetry.Exporter.OpenTelemetryProtocol
- ```
-
-
- ```csharp
- using Microsoft.Extensions.Logging;
- using OpenTelemetry.Logs;
- using OpenTelemetry.Resources;
- using Ydb.Sdk.Ado;
-
- var resourceBuilder = ResourceBuilder.CreateDefault()
- .AddService("ydb-sdk-otel-logs-sample");
-
- using var loggerFactory = LoggerFactory.Create(builder =>
- {
- builder.AddOpenTelemetry(options =>
- {
- options.SetResourceBuilder(resourceBuilder);
- options.AddOtlpExporter(o => o.Endpoint = new Uri("http://localhost:4317"));
- });
- });
-
- // Pass the factory to the SDK: each internal log (driver initialization, session pool,
- // query execution, retries, ...) is exported to the collector as an OTLP log record.
- await using var dataSource = new YdbDataSource(
- new YdbConnectionStringBuilder("Host=localhost;Port=2136;Database=/local")
- {
- LoggerFactory = loggerFactory
- });
-
- await using var connection = await dataSource.OpenConnectionAsync();
- await new YdbCommand("SELECT 1", connection).ExecuteNonQueryAsync();
- ```
-
-- Java
-
- {{ ydb-short-name }} Java SDK записывает логи через `slf4j`. Если logback используется в качестве реализации `slf4j`, подключите готовый аппендер `opentelemetry-logback-appender-1.0`, который пересылает каждое событие в OpenTelemetry Logs SDK:
-
-
- ```xml
- <dependency>
- <groupId>io.opentelemetry</groupId>
- <artifactId>opentelemetry-sdk</artifactId>
- <version>${otel.version}</version>
- </dependency>
- <dependency>
- <groupId>io.opentelemetry</groupId>
- <artifactId>opentelemetry-exporter-otlp</artifactId>
- <version>${otel.version}</version>
- </dependency>
- <dependency>
- <groupId>io.opentelemetry.instrumentation</groupId>
- <artifactId>opentelemetry-logback-appender-1.0</artifactId>
- <version>${otel.instrumentation.version}</version>
- </dependency>
- <dependency>
- <groupId>ch.qos.logback</groupId>
- <artifactId>logback-classic</artifactId>
- <version>${logback.version}</version>
- </dependency>
- ```
-
-
- Подключите аппендер в `logback.xml`:
-
-
- ```xml
- <configuration>
- <appender name="OTEL" class="io.opentelemetry.instrumentation.logback.appender.v1_0.OpenTelemetryAppender">
- </appender>
-
- <root level="INFO">
- <appender-ref ref="OTEL"/>
- </root>
- </configuration>
- ```
-
-
- Создайте экземпляр OpenTelemetry SDK с экспортером логов OTLP и установите его в аппендере, вызвав `OpenTelemetryAppender.install(...)` перед началом рабочей нагрузки:
-
-
- ```java
- import java.util.concurrent.TimeUnit;
-
- import io.opentelemetry.api.OpenTelemetry;
- import io.opentelemetry.api.common.Attributes;
- import io.opentelemetry.exporter.otlp.logs.OtlpGrpcLogRecordExporter;
- import io.opentelemetry.instrumentation.logback.appender.v1_0.OpenTelemetryAppender;
- import io.opentelemetry.sdk.OpenTelemetrySdk;
- import io.opentelemetry.sdk.logs.SdkLoggerProvider;
- import io.opentelemetry.sdk.logs.export.BatchLogRecordProcessor;
- import io.opentelemetry.sdk.resources.Resource;
- import tech.ydb.core.grpc.GrpcTransport;
- import tech.ydb.query.QueryClient;
-
- String serviceName = "ydb-java-sdk-otel-logs-sample";
- Resource resource = Resource.getDefault().merge(
- Resource.create(Attributes.builder().put("service.name", serviceName).build()));
-
- SdkLoggerProvider loggerProvider = SdkLoggerProvider.builder()
- .setResource(resource)
- .addLogRecordProcessor(BatchLogRecordProcessor.builder(
- OtlpGrpcLogRecordExporter.builder().setEndpoint("http://localhost:4317").build()
- ).build())
- .build();
-
- OpenTelemetry openTelemetry = OpenTelemetrySdk.builder()
- .setLoggerProvider(loggerProvider)
- .build();
-
- // Pass the OpenTelemetry SDK to the appender declared in logback.xml.
- OpenTelemetryAppender.install(openTelemetry);
-
- try (GrpcTransport transport = GrpcTransport.forConnectionString("grpc://localhost:2136/local").build();
- QueryClient queryClient = QueryClient.newClient(transport).build()) {
- // ... use queryClient ...
- } finally {
- loggerProvider.forceFlush().join(10, TimeUnit.SECONDS);
- loggerProvider.shutdown().join(10, TimeUnit.SECONDS);
- }
- ```
-
-- C++
-
- {{ ydb-short-name }}C++ SDK записывает логи через `TLogBackend` из `util`. Реализуйте бэкенд, который преобразует каждую запись в лог-запись OTel, и передайте его драйверу через `TDriverConfig::SetLog`. См. [Начало работы](https://opentelemetry.io/docs/languages/cpp/getting-started/) для сборки OpenTelemetry C++ SDK.
-
-
- ```cpp
- #include <ydb-cpp-sdk/client/driver/driver.h>
- #include <opentelemetry/exporters/otlp/otlp_grpc_log_record_exporter_factory.h>
- #include <opentelemetry/exporters/otlp/otlp_grpc_log_record_exporter_options.h>
- #include <opentelemetry/logs/provider.h>
- #include <opentelemetry/sdk/logs/batch_log_record_processor_factory.h>
- #include <opentelemetry/sdk/logs/logger_provider_factory.h>
- #include <opentelemetry/sdk/resource/resource.h>
- #include <library/cpp/logger/backend.h>
- #include <library/cpp/logger/record.h>
- #include <memory>
-
- namespace otlp = opentelemetry::exporter::otlp;
- namespace logs_sdk = opentelemetry::sdk::logs;
- namespace logs_api = opentelemetry::logs;
- namespace resource = opentelemetry::sdk::resource;
-
- namespace {
-
- class TOtelLogBackend final : public TLogBackend {
- public:
- explicit TOtelLogBackend(opentelemetry::nostd::shared_ptr<logs_api::Logger> logger)
- : Logger_(std::move(logger)) {}
-
- void WriteData(const TLogRecord& rec) override {
- Logger_->EmitLogRecord(MapSeverity(rec.Priority),
- opentelemetry::nostd::string_view(rec.Data, rec.Len));
- }
-
- void ReopenLog() override {}
-
- private:
- static logs_api::Severity MapSeverity(ELogPriority p) {
- switch (p) {
- case TLOG_EMERG:
- case TLOG_ALERT:
- case TLOG_CRIT: return logs_api::Severity::kFatal;
- case TLOG_ERR: return logs_api::Severity::kError;
- case TLOG_WARNING: return logs_api::Severity::kWarn;
- case TLOG_NOTICE:
- case TLOG_INFO: return logs_api::Severity::kInfo;
- case TLOG_DEBUG: return logs_api::Severity::kDebug;
- default: return logs_api::Severity::kTrace;
- }
- }
-
- opentelemetry::nostd::shared_ptr<logs_api::Logger> Logger_;
- };
- } // namespace
-
- int main() {
- // 1. Configuring the OTel log provider with an OTLP exporter
- otlp::OtlpGrpcLogRecordExporterOptions exporterOpts;
- exporterOpts.endpoint = "localhost:4317";
- auto exporter = otlp::OtlpGrpcLogRecordExporterFactory::Create(exporterOpts);
- auto processor = logs_sdk::BatchLogRecordProcessorFactory::Create(std::move(exporter));
- auto res = resource::Resource::Create({{"service.name", "ydb-cpp-sdk-otel-logs-sample"}});
-
- std::shared_ptr<logs_api::LoggerProvider> provider(
- logs_sdk::LoggerProviderFactory::Create(std::move(processor), res));
- logs_api::Provider::SetLoggerProvider(provider);
-
- auto logger = provider->GetLogger("ydb-cpp-sdk");
-
- // 2. Pass the bridge to the YDB driver
- auto config = NYdb::TDriverConfig()
- .SetEndpoint("localhost:2136")
- .SetDatabase("/local")
- .SetLog(std::make_unique<TOtelLogBackend>(logger));
-
- NYdb::TDriver driver(config);
- // ... use driver ...
- driver.Stop(true);
- }
- ```
-
-- JavaScript
-
- {% include [feature-not-supported](../../_includes/feature-not-supported.md) %}
-
-- Rust
-
- {% include [feature-not-supported](../../_includes/feature-not-supported.md) %}
-
- Отслеживайте прогресс или голосуйте за поддержку Rust SDK: [ydb-rs-sdk#268](https://github.com/ydb-platform/ydb-rs-sdk/issues/268)
-
-{% endlist %}
diff --git a/ydb/docs/ru/core/recipes/ydb-sdk/debug-prometheus.md b/ydb/docs/ru/core/recipes/ydb-sdk/debug-prometheus.md
deleted file mode 100644
index b466129199e..00000000000
--- a/ydb/docs/ru/core/recipes/ydb-sdk/debug-prometheus.md
+++ /dev/null
@@ -1,106 +0,0 @@
-# Включение метрик в Prometheus
-
-Ниже приведены примеры кода для включения метрик в Prometheus в различных {{ ydb-short-name }} SDK.
-
-{% list tabs %}
-
-- Go
-
- {% list tabs %}
-
- - Нативный SDK
-
- ```go
- package main
-
- import (
- "context"
-
- "github.com/prometheus/client_golang/prometheus"
- metrics "github.com/ydb-platform/ydb-go-sdk-prometheus/v2"
- "github.com/ydb-platform/ydb-go-sdk/v3"
- "github.com/ydb-platform/ydb-go-sdk/v3/trace"
- )
-
- func main() {
- ctx := context.Background()
- registry := prometheus.NewRegistry()
- db, err := ydb.Open(ctx,
- os.Getenv("YDB_CONNECTION_STRING"),
- metrics.WithTraces(
- registry,
- metrics.WithDetails(trace.DetailsAll),
- metrics.WithSeparator("_"),
- ),
- )
- if err != nil {
- panic(err)
- }
- defer db.Close(ctx)
- ...
- }
- ```
-
- - database/sql
-
- ```go
- package main
-
- import (
- "context"
- "database/sql"
-
- "github.com/prometheus/client_golang/prometheus"
- metrics "github.com/ydb-platform/ydb-go-sdk-prometheus/v2"
- "github.com/ydb-platform/ydb-go-sdk/v3"
- "github.com/ydb-platform/ydb-go-sdk/v3/trace"
- )
-
- func main() {
- ctx := context.Background()
- registry := prometheus.NewRegistry()
- nativeDriver, err := ydb.Open(ctx,
- os.Getenv("YDB_CONNECTION_STRING"),
- metrics.WithTraces(
- registry,
- metrics.WithDetails(trace.DetailsAll),
- metrics.WithSeparator("_"),
- ),
- )
- if err != nil {
- panic(err)
- }
- defer nativeDriver.Close(ctx)
-
- connector, err := ydb.Connector(nativeDriver)
- if err != nil {
- panic(err)
- }
-
- db := sql.OpenDB(connector)
- defer db.Close()
- ...
- }
- ```
-
- {% endlist %}
-
-- Java
-
- Данная функциональность в настоящее время не поддерживается.
-
-- Python
-
- Данная функциональность в настоящее время не поддерживается.
-
-- JavaScript
-
- {% include [work-in-progress](../../_includes/work-in-progress.md) %}
-
-- Rust
-
- {% include [feature-not-supported](../../_includes/feature-not-supported.md) %}
-
- Отслеживайте прогресс или голосуйте за поддержку Rust SDK: [ydb-rs-sdk#267](https://github.com/ydb-platform/ydb-rs-sdk/issues/267)
-
-{% endlist %}
diff --git a/ydb/docs/ru/core/reference/configuration/host_configs.md b/ydb/docs/ru/core/reference/configuration/host_configs.md
index f99b5214c65..eac783db2dc 100644
--- a/ydb/docs/ru/core/reference/configuration/host_configs.md
+++ b/ydb/docs/ru/core/reference/configuration/host_configs.md
@@ -10,16 +10,19 @@ host_configs:
drive:
- path: <path_to_device>
type: <type>
+ disk_scope: <disk_scope> # необязательный атрибут
- path: ...
- host_config_id: 2
...
```
-Атрибут `host_config_id` задает числовой идентификатор конфигурации. В атрибуте `drive` содержится коллекция описаний подключенных дисков. Каждое описание состоит из двух атрибутов:
+Атрибут `host_config_id` задает числовой идентификатор конфигурации. В атрибуте `drive` содержится коллекция описаний подключенных дисков. Каждое описание состоит из двух обязательных атрибутов:
- `path` : Путь к смонтированному блочному устройству, например `/dev/disk/by-partlabel/ydb_disk_ssd_01`
- `type` : Тип физического носителя устройства: `ssd`, `nvme` или `rot` (rotational - HDD)
+Дополнительно может быть указан необязательный атрибут `disk_scope` — метка для вычисления домена отказа для некоторых упрощенных конфигураций, см. [Конфигурирование disk_scope](#disk-scope).
+
## Примеры
Одна конфигурация с идентификатором 1, с одним диском типа SSD, доступным по пути `/dev/disk/by-partlabel/ydb_disk_ssd_01`:
@@ -52,6 +55,54 @@ host_configs:
type: SSD
```
+## Конфигурирование disk_scope {#disk-scope}
+
+`disk_scope` — необязательный строковый атрибут диска, задающий более детальную зону отказа внутри одного узла. Он учитывается при вычислении [доменов отказа](../../concepts/glossary.md#fail-domain) во время выбора дисков для размещения [VDisk'ов](../../concepts/glossary.md#vdisk) [групп хранения](../../concepts/glossary.md#storage-group).
+
+Домен отказа определяется физическим расположением диска (дата-центр, стойка, сервер, физическое устройство), поэтому либо никакие два VDisk'а одной группы не могут быть размещены на одном узле (если домен отказа соответствует серверу или серверной стойке), либо несколько VDisk'ов одной группы может быть размещено на одном узле на разных дисках (если домен отказа соответствует отдельному физическому устройству). В первом случае невозможно создать конфигурацию в режиме `block-4-2` менее чем с 8 узлами хранения, а во втором случае конфигурация может не переживать отказ 1 узла. Чтобы ограничить количество VDisk'ов одной группы, размещаемых на одном узле, физические устройства можно пометить атрибутом `disk_scope` и включить расчёт доменов отказа по уровню `disk_scope`. Это позволит собирать некоторые отказоустойчивые конфигурации в инсталляциях, в которых серверов меньше, чем требуется доменов отказа для выбранного [режима отказоустойчивости](../../concepts/topology.md).
+
+### Пример {#disk-scope-example}
+
+Кластер из 4 серверов, на каждом сервере 4 диска в режиме отказоустойчивости `block-4-2`:
+
+``` yaml
+host_configs:
+- host_config_id: 1
+ drive:
+ - path: /dev/disk/by-partlabel/ydb_disk_ssd_01
+ type: SSD
+ disk_scope: fail-domain-1
+ - path: /dev/disk/by-partlabel/ydb_disk_ssd_02
+ type: SSD
+ disk_scope: fail-domain-1
+ - path: /dev/disk/by-partlabel/ydb_disk_ssd_03
+ type: SSD
+ disk_scope: fail-domain-2
+ - path: /dev/disk/by-partlabel/ydb_disk_ssd_04
+ type: SSD
+ disk_scope: fail-domain-2
+```
+
+Чтобы домены отказа рассчитывались по уровню `disk_scope` для статической группы и других групп хранения, укажите соответствующий тип домена отказа в yaml-конфиге (верхнеуровневый ключ):
+
+``` yaml
+fail_domain_type: disk_scope
+```
+
+И в конфигурации домена:
+
+```yaml
+domains:
+- domain_name: <имя домена>
+ ...
+ storage_pool_kinds:
+ - kind: <тип используемых физических устройств>
+ fail_domain_type: disk_scope
+ ...
+```
+
+При такой конфигурации на каждом сервере будут размещены по два VDisk'а каждой группы хранения, и максимальным отказом в такой конфигурации будет отказ одного сервера.
+
## Особенности Kubernetes {#host-configs-k8s}
{{ ydb-short-name }} Kubernetes operator монтирует NBS диски для Storage узлов на путь `/dev/kikimr_ssd_00`. Для их использования должна быть указана следующая конфигурация `host_configs`:
diff --git a/ydb/docs/ru/core/reference/configuration/kafka_proxy_config.md b/ydb/docs/ru/core/reference/configuration/kafka_proxy_config.md
index adf90e80d3a..1cc5a76a55c 100644
--- a/ydb/docs/ru/core/reference/configuration/kafka_proxy_config.md
+++ b/ydb/docs/ru/core/reference/configuration/kafka_proxy_config.md
@@ -8,6 +8,7 @@
|| Параметр | Тип | Значение по умолчанию | Описание ||
|| `enable_kafka_proxy` | bool | `false` | Включает или отключает Kafka Proxy. ||
|| `listening_port` | int32 | `9092` | Порт, на котором будет доступен Kafka API. ||
+|| `listening_address` | string | `[::]` | Сетевой адрес, на котором Kafka Proxy принимает входящие соединения. Значение `[::]` — все интерфейсы (dual-stack, должна быть включена поддержка IPv6), `127.0.0.1` или `[::1]` — только localhost. ||
|| `transaction_timeout_ms` | uint32 | `300000` (5 минут) | Максимальный таймаут для Kafka транзакций, после которого транзакция будет отменена. ||
|| `auto_create_topics_enable` | bool | `false` | Включает автоматическое создание топиков при обращении к ним. Аналог [такой же опции](https://kafka.apache.org/documentation/#brokerconfigs_auto.create.topics.enable) в Apache Kafka. ||
|| `auto_create_consumers_enable` | bool | `true` | Включает автоматическое заведение консьюмеров при обращении к ним. ||
@@ -26,6 +27,7 @@
kafka_proxy_config:
enable_kafka_proxy: true
listening_port: 9092
+ listening_address: "[::]"
transaction_timeout_ms: 300000 # 5 минут
auto_create_topics_enable: true
auto_create_consumers_enable: true
diff --git a/ydb/docs/ru/core/yql/reference/syntax/alter-resource-pool-classifier.md b/ydb/docs/ru/core/yql/reference/syntax/alter-resource-pool-classifier.md
index b5b42e2dc5b..97a9326e7dd 100644
--- a/ydb/docs/ru/core/yql/reference/syntax/alter-resource-pool-classifier.md
+++ b/ydb/docs/ru/core/yql/reference/syntax/alter-resource-pool-classifier.md
@@ -48,7 +48,7 @@ GRANT 'ALL' ON `/my_db` TO `user1@domain`;
* `RANK` (Int64) — опциональное поле, задающее порядок выбора классификатора пула ресурсов. Если значение не указано, берётся максимальный существующий `RANK` и к нему прибавляется 1000. Допустимые значения: уникальное число в диапазоне $[0, 2^{63}-1]$.
* `RESOURCE_POOL` (String) — обязательное поле, задающее имя пула ресурсов, в который будут отправлены запросы, удовлетворяющие критериям классификатора.
-* `MEMBER_NAME` (String) — опциональное поле, определяющее, какой пользователь или группа пользователей будут отправлены в указанный пул ресурсов. Если поле не указано, классификатор игнорирует `MEMBER_NAME`, и классификация осуществляется по другим признакам.
+* `MEMBER_NAME` (String) — опциональное поле, определяющее, какой пользователь или группа пользователей будут отправлены в указанный пул ресурсов. Формат значения описан в [CREATE RESOURCE POOL CLASSIFIER](create-resource-pool-classifier.md#member-name-format).
## См. также
diff --git a/ydb/docs/ru/core/yql/reference/syntax/create-resource-pool-classifier.md b/ydb/docs/ru/core/yql/reference/syntax/create-resource-pool-classifier.md
index f14947dd664..aa1645c6f59 100644
--- a/ydb/docs/ru/core/yql/reference/syntax/create-resource-pool-classifier.md
+++ b/ydb/docs/ru/core/yql/reference/syntax/create-resource-pool-classifier.md
@@ -16,7 +16,18 @@ WITH ( <parameter_name> [= <parameter_value>] [, ... ] )
* `RANK` (Int64) — опциональное поле, задающее порядок выбора классификатора пула ресурсов. Если значение не указано, берётся максимальный существующий `RANK` и к нему прибавляется 1000. Допустимые значения: уникальное число в диапазоне $[0, 2^{63}-1]$.
* `RESOURCE_POOL` (String) — обязательное поле, задающее имя пула ресурсов, в который будут отправлены запросы, удовлетворяющие критериям классификатора.
-* `MEMBER_NAME` (String) — опциональное поле, определяющее, какой пользователь или группа пользователей будут отправлены в указанный пул ресурсов. Если поле не указано, классификатор игнорирует `MEMBER_NAME`, и классификация осуществляется по другим признакам.
+* `MEMBER_NAME` (String) — опциональное поле, определяющее, какой пользователь или группа пользователей будут отправлены в указанный пул ресурсов. Значение сравнивается с SID пользователя или любым SID группы из его токена аутентификации; подробнее о формате см. [ниже](#member-name-format). Если поле не указано, классификатор игнорирует `MEMBER_NAME`, и классификация осуществляется по другим признакам.
+
+### Формат MEMBER_NAME {#member-name-format}
+
+`MEMBER_NAME` сравнивается посимвольно с [SID](../../../concepts/glossary.md#access-sid) пользователя или любым SID группы из его токена аутентификации. Формат SID зависит от того, каким способом пользователь пришёл в систему.
+
+- **Встроенные пользователи {{ ydb-short-name }} (логин/пароль)** — SID совпадает с именем пользователя, без суффикса. Например, `user1`. Подробнее — [{#T}](../../../security/authentication.md#static-credentials).
+- **Облачные пользователи (Access Service)** — SID имеет вид `<subject_id>@as`, где `<subject_id>` — идентификатор пользователя в IAM. Суффикс задаётся параметром [`access_service_domain`](../../../reference/configuration/auth_config.md#iam-auth-config) (по умолчанию `as`). Например, `ajeb89hv69nujke769fa@as`. Подробнее — [{#T}](../../../security/authentication.md#iam).
+- **LDAP** — SID имеет вид `<логин>@<домен>`, где домен задаётся параметром [`ldap_authentication_domain`](../../../reference/configuration/auth_config.md#ldap-auth-config) (по умолчанию `ldap`). Например, `user1@ldap`. Подробнее — [{#T}](../../../security/authentication.md#ldap).
+- **Внешние провайдеры идентификации (OIDC)** — SID имеет вид `<логин>@<домен>`, где домен задаётся параметром `external_idp_authentication_domain` в [конфигурации аутентификации](../../../reference/configuration/auth_config.md) (по умолчанию `sso`). Например, `user1@sso`.
+
+В качестве значения можно указать как SID конкретного пользователя, так и SID группы. Ко всем аутентифицированным пользователям автоматически добавляется группа `all-users@well-known` — её удобно использовать, если нужно направить в пул запросы от всех аутентифицированных клиентов.
## Замечания {#remarks}
diff --git a/ydb/docs/ru/core/yql/reference/syntax/discard.md b/ydb/docs/ru/core/yql/reference/syntax/discard.md
index 4ccf9cc0567..d6299aece2f 100644
--- a/ydb/docs/ru/core/yql/reference/syntax/discard.md
+++ b/ydb/docs/ru/core/yql/reference/syntax/discard.md
@@ -1,18 +1,60 @@
# DISCARD
-Вычисляет {% if select_command == "SELECT STREAM" %}[`SELECT STREAM`](select_stream.md){% else %}[`SELECT`](select/index.md){% endif %}{% if feature_mapreduce %}{% if reduce_command %}, [`{{ reduce_command }}`](reduce.md){% endif %} или [`{{ process_command }}`](process.md){% endif %}, но не возвращает результат ни в клиент, ни в таблицу. {% if feature_mapreduce %}Не может быть задано одновременно с [INTO RESULT](into_result.md).{% endif %}
+Вычисляет {% if select_command == "SELECT STREAM" %}[`SELECT STREAM`](select_stream.md){% else %}[`SELECT`](select/index.md){% endif %}{% if feature_mapreduce %}{% if reduce_command %}, [`{{ reduce_command }}`](reduce.md){% endif %} или [`{{ process_command }}`](process.md){% endif %}, но не возвращает результат ни в клиент, ни в таблицу.{% if feature_mapreduce %} Не может быть задано одновременно с [INTO RESULT](into_result.md).{% endif %}
Полезно использовать в сочетании с [`Ensure`](../builtins/basic.md#ensure) для проверки выполнения пользовательских условий на финальный результат вычислений.
+{% if backend_name == "YDB" %}
+
+Запрос с `DISCARD` выполняется полностью — со всеми фильтрами, агрегациями и проверками `Ensure`, — но результирующий набор (result set) не возвращается клиенту. Данные при этом не пересылаются по сети: клиент получает только статус выполнения запроса. На больших выборках это заметно экономит трафик и память.
+
+В запросе из нескольких операторов `DISCARD` действует на отдельный оператор:
+
+```yql
+SELECT 1; -- вернётся
+DISCARD SELECT 2; -- выполнится, но в ответ не попадёт
+SELECT 3; -- вернётся
+```
+
+Клиент получит два результирующих набора — от первого и третьего оператора.
+
+`DISCARD` применяется только к оператору целиком, использовать его внутри выражений нельзя. В частности, недопустимо:
+
+* в подзапросах — `SELECT * FROM (DISCARD SELECT 1)`;
+* в `WHERE ... IN (...)` — `SELECT * FROM my_table WHERE Key IN (DISCARD SELECT 1)`;
+* в операндах `UNION ALL` — `SELECT 1 UNION ALL DISCARD SELECT 2`.
+
+{% note info %}
+
+`DISCARD` поддерживается при выполнении запросов через сервис исполнения запросов (Query Service) начиная с версии {{ ydb-short-name }} 26.2. При выполнении через устаревшие интерфейсы (Table Service, скан-запросы) `DISCARD` игнорируется, и результат возвращается клиенту — поведение сохранено для обратной совместимости.
+
+{% endnote %}
+
+{% endif %}
+
{% if select_command != true or select_command == "SELECT" %}
-### Примеры
+## Примеры
```yql
DISCARD SELECT 1;
```
+{% if backend_name == "YDB" %}
+
+```yql
+DISCARD SELECT Ensure(
+ Data,
+ Data < 1000000,
+ "Value too big"
+) FROM `result_table`;
+```
+
+Если условие `Ensure` нарушено, запрос завершится ошибкой. Если всё в порядке — запрос отработает успешно, а клиент не получит результирующий набор, который иначе пришлось бы выкачивать целиком.
+
+{% else %}
+
```yql
INSERT INTO result_table WITH TRUNCATE
SELECT * FROM
@@ -30,3 +72,5 @@ DISCARD SELECT Ensure(
```
{% endif %}
+
+{% endif %}
diff --git a/ydb/docs/ru/core/yql/reference/syntax/index.md b/ydb/docs/ru/core/yql/reference/syntax/index.md
index 243e56f2d92..c5d3ea27465 100644
--- a/ydb/docs/ru/core/yql/reference/syntax/index.md
+++ b/ydb/docs/ru/core/yql/reference/syntax/index.md
@@ -75,12 +75,8 @@
{% endif %}
-{% if backend_name != "YDB" %}
-
* [DISCARD](discard.md)
-{% endif %}
-
* [INTO RESULT](into_result.md)
{% if feature_mapreduce %}
diff --git a/ydb/docs/ru/core/yql/reference/syntax/select/toc_i.yaml b/ydb/docs/ru/core/yql/reference/syntax/select/toc_i.yaml
index f35b392aa9d..82cf0b75811 100644
--- a/ydb/docs/ru/core/yql/reference/syntax/select/toc_i.yaml
+++ b/ydb/docs/ru/core/yql/reference/syntax/select/toc_i.yaml
@@ -4,6 +4,7 @@ items:
- { name: FROM, href: from.md }
- { name: FROM AS_TABLE, href: from_as_table.md }
- { name: FROM SELECT, href: from_select.md }
+- { name: FROM Topic, href: topics.md }
- { name: FOLDER, href: folder.md, when: yt }
- { name: FLATTEN, href: flatten.md }
- { name: GROUP BY, href: group-by.md }
@@ -31,4 +32,3 @@ items:
- { name: SAMPLE, href: sample.md, when: feature_tablesample }
- { name: TABLESAMPLE, href: sample.md, when: feature_tablesample }
- { name: MATCH_RECOGNIZE, href: match_recognize.md, when: feature_match_recogznize }
-- { name: FROM Topic, href: topics.md }
diff --git a/ydb/docs/ru/core/yql/reference/syntax/toc_i.yaml b/ydb/docs/ru/core/yql/reference/syntax/toc_i.yaml
index 72752ddc1de..384ca5771bd 100644
--- a/ydb/docs/ru/core/yql/reference/syntax/toc_i.yaml
+++ b/ydb/docs/ru/core/yql/reference/syntax/toc_i.yaml
@@ -41,7 +41,7 @@ items:
- { name: COMMIT, href: commit.md }
- { name: DECLARE, href: declare.md }
- { name: DELETE, href: delete.md, when: feature_map_tables }
-- { name: DISCARD, href: discard.md, when: backend_name != "YDB" }
+- { name: DISCARD, href: discard.md }
- { name: DROP ASYNC REPLICATION, href: drop-async-replication.md }
- { name: DROP BACKUP COLLECTION, href: drop-backup-collection.md, when: feature_backup_collections }
- { name: DROP GROUP, href: drop-group.md, when: feature_user_and_group }
diff --git a/ydb/docs/ru/core/yql/reference/types/primitive.md b/ydb/docs/ru/core/yql/reference/types/primitive.md
index a48af7d1150..e01ed7ff615 100644
--- a/ydb/docs/ru/core/yql/reference/types/primitive.md
+++ b/ydb/docs/ru/core/yql/reference/types/primitive.md
@@ -61,7 +61,7 @@
|| `DyNumber` |
Бинарное представление вещественного числа точностью до 38 знаков.
Допустимые значения: положительные от 1×10<sup>-130</sup> до 1×10<sup>126</sup>-1, отрицательные от -1×10<sup>126</sup>-1 до -1×10<sup>-130</sup> и 0.
-Совместим с типом `Number` AWS DynamoDB. Не рекомендуется для использования в {{ backend_name_lower }}-native приложениях. | Не поддерживается в колоночных таблицах
+Совместим с типом `Number` AWS DynamoDB. Не рекомендуется для использования в {{ backend_name_lower }}-native приложениях. |
||
{% endif %}
|#
@@ -97,7 +97,7 @@
Не поддерживает возможность сравнения{% if feature_map_tables %}, не может использоваться в первичном ключе и в колонках, формирующих ключ вторичного индекса{% endif %}
||
|| `Uuid` |
-Универсальный идентификатор [UUID](https://tools.ietf.org/html/rfc4122) | Не поддерживается в колоночных таблицах
+Универсальный идентификатор [UUID](https://tools.ietf.org/html/rfc4122) |
||
|#
@@ -205,7 +205,6 @@
|
8
|
-Не поддерживается в колоночных таблицах
||
||
@@ -217,7 +216,6 @@
|
8
|
-Не поддерживается в колоночных таблицах
||
||
@@ -265,7 +263,6 @@
|
8 и метка таймзоны
|
-—
||
||
diff --git a/ydb/docs/ru/core/yql/toc_i.yaml b/ydb/docs/ru/core/yql/toc_i.yaml
index 2d19742b88a..4fac8bfcd8f 100644
--- a/ydb/docs/ru/core/yql/toc_i.yaml
+++ b/ydb/docs/ru/core/yql/toc_i.yaml
@@ -1,8 +1,3 @@
items:
-- name: Обзор
- href: reference/index.md
-- include:
- mode: link
- path: reference/toc_i.yaml
- name: Планы запросов
href: query_plans.md
diff --git a/ydb/docs/ru/core/yql/toc_p.yaml b/ydb/docs/ru/core/yql/toc_p.yaml
index 50818e81233..1ab6f7ee53c 100644
--- a/ydb/docs/ru/core/yql/toc_p.yaml
+++ b/ydb/docs/ru/core/yql/toc_p.yaml
@@ -1,3 +1,5 @@
items:
-- include: { mode: link, path: toc_i.yaml }
+- name: Обзор
+ href: reference/index.md
- include: { mode: link, path: reference/toc_p.yaml }
+- include: { mode: link, path: toc_i.yaml }
diff --git a/ydb/library/actors/core/actor.cpp b/ydb/library/actors/core/actor.cpp
index 804f1846cb8..6e7b0aae725 100644
--- a/ydb/library/actors/core/actor.cpp
+++ b/ydb/library/actors/core/actor.cpp
@@ -93,6 +93,16 @@ namespace NActors {
return TActivationContext::Send(ev);
}
+ bool IActor::SendActorLivenessCheck(const TActorId& target, ui64 cookie) const noexcept {
+ return Send(new IEventHandle(
+ TEvents::TSystem::CheckActorLiveness,
+ TEvents::TEvCheckActorLiveness::RequestFlags,
+ target,
+ SelfId(),
+ nullptr,
+ cookie));
+ }
+
bool IActor::Send(const TActorId& recipient, IEventBase* ev, ui32 flags, ui64 cookie, NWilson::TTraceId traceId) const noexcept {
return SelfActorId.Send(recipient, ev, flags, cookie, std::move(traceId));
}
diff --git a/ydb/library/actors/core/actor.h b/ydb/library/actors/core/actor.h
index 3db485c8bd9..ede3d97ca7a 100644
--- a/ydb/library/actors/core/actor.h
+++ b/ydb/library/actors/core/actor.h
@@ -816,6 +816,7 @@ namespace NActors {
void Describe(IOutputStream&) const override;
bool Send(TAutoPtr<IEventHandle> ev) const noexcept;
+ bool SendActorLivenessCheck(const TActorId& target, ui64 cookie = 0) const noexcept;
bool Send(const TActorId& recipient, IEventBase* ev, TEventFlags flags = 0, ui64 cookie = 0, NWilson::TTraceId traceId = {}) const noexcept final;
bool Send(const TActorId& recipient, THolder<IEventBase> ev, TEventFlags flags = 0, ui64 cookie = 0, NWilson::TTraceId traceId = {}) const{
return Send(recipient, ev.Release(), flags, cookie, std::move(traceId));
diff --git a/ydb/library/actors/core/ut/actor_ut.cpp b/ydb/library/actors/core/ut/actor_ut.cpp
index dca02644085..fe7efe2ed7b 100644
--- a/ydb/library/actors/core/ut/actor_ut.cpp
+++ b/ydb/library/actors/core/ut/actor_ut.cpp
@@ -782,7 +782,7 @@ Y_UNIT_TEST_SUITE(TestActorLiveness) {
void Bootstrap() {
Become(&TThis::StateWork);
- CheckActorLiveness(AliveCookie);
+ SendActorLivenessCheck(Target, AliveCookie);
}
STRICT_STFUNC(StateWork,
@@ -796,7 +796,7 @@ Y_UNIT_TEST_SUITE(TestActorLiveness) {
Y_ABORT_UNLESS(ev->Cookie == AliveCookie);
Send(Target, new TEvents::TEvPoison());
- CheckActorLiveness(DeadCookie);
+ SendActorLivenessCheck(Target, DeadCookie);
}
void Handle(TEvents::TEvActorDead::TPtr& ev) {
@@ -813,16 +813,6 @@ Y_UNIT_TEST_SUITE(TestActorLiveness) {
PassAway();
}
- void CheckActorLiveness(ui64 cookie) {
- Send(new IEventHandle(
- TEvents::TSystem::CheckActorLiveness,
- TEvents::TEvCheckActorLiveness::RequestFlags,
- Target,
- SelfId(),
- nullptr,
- cookie));
- }
-
const TActorId Target;
TThreadParkPad* const DonePad;
std::atomic<bool>* const DoneFlag;
@@ -839,13 +829,7 @@ Y_UNIT_TEST_SUITE(TestActorLiveness) {
void Bootstrap() {
Become(&TThis::StateWork);
- Send(new IEventHandle(
- TEvents::TSystem::CheckActorLiveness,
- TEvents::TEvCheckActorLiveness::RequestFlags,
- Target,
- SelfId(),
- nullptr,
- Cookie));
+ SendActorLivenessCheck(Target, Cookie);
}
STRICT_STFUNC(StateWork,
diff --git a/ydb/library/actors/helpers/actor_liveness_checker.cpp b/ydb/library/actors/helpers/actor_liveness_checker.cpp
new file mode 100644
index 00000000000..45bae9cd4be
--- /dev/null
+++ b/ydb/library/actors/helpers/actor_liveness_checker.cpp
@@ -0,0 +1,96 @@
+#include "actor_liveness_checker.h"
+
+#include <ydb/library/actors/core/hfunc.h>
+
+namespace NActors {
+ TActorLivenessChecker::TActorLivenessChecker(TVector<TActorLivenessCheckTarget> targets) {
+ for (const auto& target : targets) {
+ PendingTargets.emplace(target.ActorId, target.Cookie);
+ }
+ }
+
+ void TActorLivenessChecker::Bootstrap() {
+ if (PendingTargets.empty()) {
+ return Complete();
+ }
+
+ Become(&TThis::StateFunc);
+ // Every local liveness probe eventually produces ActorAlive or ActorDead.
+ // Remote probes currently produce ActorLivenessUnsure.
+ for (const auto& [actorId, cookie] : PendingTargets) {
+ SendActorLivenessCheck(actorId, cookie);
+ }
+ }
+
+ STFUNC(TActorLivenessChecker::StateFunc) {
+ switch (ev->GetTypeRewrite()) {
+ hFunc(TEvents::TEvActorAlive, Handle)
+ hFunc(TEvents::TEvActorDead, Handle)
+ hFunc(TEvents::TEvActorLivenessUnsure, Handle)
+ }
+ }
+
+ void TActorLivenessChecker::OnAlive(const TActorLivenessCheckTarget&) {
+ }
+
+ void TActorLivenessChecker::OnDead(const TActorLivenessCheckTarget&) {
+ }
+
+ void TActorLivenessChecker::OnUnsure(const TActorLivenessCheckTarget&) {
+ }
+
+ void TActorLivenessChecker::OnFinish() {
+ }
+
+ void TActorLivenessChecker::Handle(TEvents::TEvActorAlive::TPtr& ev) {
+ TActorLivenessCheckTarget target;
+ if (ExtractTarget(ev->Sender, target)) {
+ OnAlive(target);
+ CompleteIfDone();
+ }
+ }
+
+ void TActorLivenessChecker::Handle(TEvents::TEvActorDead::TPtr& ev) {
+ TActorLivenessCheckTarget target;
+ if (ExtractTarget(ev->Sender, target)) {
+ OnDead(target);
+ CompleteIfDone();
+ }
+ }
+
+ void TActorLivenessChecker::Handle(TEvents::TEvActorLivenessUnsure::TPtr& ev) {
+ TActorLivenessCheckTarget target;
+ if (ExtractTarget(ev->Sender, target)) {
+ OnUnsure(target);
+ CompleteIfDone();
+ }
+ }
+
+ bool TActorLivenessChecker::ExtractTarget(
+ const TActorId& actorId,
+ TActorLivenessCheckTarget& target) {
+ const auto it = PendingTargets.find(actorId);
+ if (it == PendingTargets.end()) {
+ return false;
+ }
+
+ target = {
+ .ActorId = it->first,
+ .Cookie = it->second,
+ };
+ PendingTargets.erase(it);
+ return true;
+ }
+
+ void TActorLivenessChecker::CompleteIfDone() {
+ if (PendingTargets.empty()) {
+ Complete();
+ }
+ }
+
+ void TActorLivenessChecker::Complete() {
+ OnFinish();
+ PassAway();
+ }
+
+}
diff --git a/ydb/library/actors/helpers/actor_liveness_checker.h b/ydb/library/actors/helpers/actor_liveness_checker.h
new file mode 100644
index 00000000000..02465256cfb
--- /dev/null
+++ b/ydb/library/actors/helpers/actor_liveness_checker.h
@@ -0,0 +1,47 @@
+#pragma once
+
+#include <ydb/library/actors/core/actor_bootstrapped.h>
+#include <ydb/library/actors/core/events.h>
+
+#include <util/generic/hash.h>
+#include <util/generic/vector.h>
+
+namespace NActors {
+
+ struct TActorLivenessCheckTarget {
+ TActorId ActorId;
+ ui64 Cookie = 0;
+ };
+
+ class TActorLivenessChecker
+ : public TActorBootstrapped<TActorLivenessChecker>
+ {
+ public:
+ explicit TActorLivenessChecker(TVector<TActorLivenessCheckTarget> targets);
+
+ void Bootstrap();
+
+ STFUNC(StateFunc);
+
+ protected:
+ // Hooks run synchronously in the checker actor context. OnFinish is
+ // called exactly once after every target has produced a response.
+ virtual void OnAlive(const TActorLivenessCheckTarget& target);
+ virtual void OnDead(const TActorLivenessCheckTarget& target);
+ virtual void OnUnsure(const TActorLivenessCheckTarget& target);
+ virtual void OnFinish();
+
+ private:
+ void Handle(TEvents::TEvActorAlive::TPtr& ev);
+ void Handle(TEvents::TEvActorDead::TPtr& ev);
+ void Handle(TEvents::TEvActorLivenessUnsure::TPtr& ev);
+
+ bool ExtractTarget(const TActorId& actorId, TActorLivenessCheckTarget& target);
+ void CompleteIfDone();
+ void Complete();
+
+ private:
+ THashMap<TActorId, ui64> PendingTargets;
+ };
+
+}
diff --git a/ydb/library/actors/helpers/actor_liveness_checker_ut.cpp b/ydb/library/actors/helpers/actor_liveness_checker_ut.cpp
new file mode 100644
index 00000000000..60a0795c19d
--- /dev/null
+++ b/ydb/library/actors/helpers/actor_liveness_checker_ut.cpp
@@ -0,0 +1,154 @@
+#include "actor_liveness_checker.h"
+
+#include <ydb/library/actors/core/actor_bootstrapped.h>
+#include <ydb/library/actors/testlib/test_runtime.h>
+
+#include <library/cpp/testing/unittest/registar.h>
+
+#include <utility>
+
+namespace NActors {
+namespace {
+
+ class TTargetActor
+ : public TActorBootstrapped<TTargetActor>
+ {
+ public:
+ void Bootstrap() {
+ Become(&TThis::StateFunc);
+ }
+
+ private:
+ STRICT_STFUNC(StateFunc,
+ cFunc(TEvents::TSystem::Poison, PassAway)
+ )
+ };
+
+ struct TCheckResult {
+ TVector<TActorLivenessCheckTarget> AliveTargets;
+ TVector<TActorLivenessCheckTarget> DeadTargets;
+ TVector<TActorLivenessCheckTarget> UnsureTargets;
+ size_t FinishCount = 0;
+ };
+
+ class TTestActorLivenessChecker
+ : public TActorLivenessChecker
+ {
+ public:
+ TTestActorLivenessChecker(
+ TVector<TActorLivenessCheckTarget> targets,
+ TCheckResult& result)
+ : TActorLivenessChecker(std::move(targets))
+ , Result(result)
+ {
+ }
+
+ private:
+ void OnAlive(const TActorLivenessCheckTarget& target) override {
+ Result.AliveTargets.push_back(target);
+ }
+
+ void OnDead(const TActorLivenessCheckTarget& target) override {
+ Result.DeadTargets.push_back(target);
+ }
+
+ void OnUnsure(const TActorLivenessCheckTarget& target) override {
+ Result.UnsureTargets.push_back(target);
+ }
+
+ void OnFinish() override {
+ ++Result.FinishCount;
+ }
+
+ private:
+ TCheckResult& Result;
+ };
+
+}
+
+Y_UNIT_TEST_SUITE(TActorLivenessCheckerTest) {
+ Y_UNIT_TEST(KeepsLiveActor) {
+ TTestActorRuntimeBase runtime;
+ runtime.Initialize();
+
+ const TActorId target = runtime.Register(new TTargetActor());
+ TCheckResult result;
+ runtime.Register(new TTestActorLivenessChecker(
+ {{
+ .ActorId = target,
+ .Cookie = 10,
+ }},
+ result));
+
+ TDispatchOptions options;
+ options.CustomFinalCondition = [&result] {
+ return result.FinishCount;
+ };
+ runtime.DispatchEvents(options);
+
+ UNIT_ASSERT_VALUES_EQUAL(result.FinishCount, 1);
+ UNIT_ASSERT_VALUES_EQUAL(result.AliveTargets.size(), 1);
+ UNIT_ASSERT_VALUES_EQUAL(result.AliveTargets.front().ActorId, target);
+ UNIT_ASSERT_VALUES_EQUAL(result.AliveTargets.front().Cookie, 10);
+ UNIT_ASSERT(result.DeadTargets.empty());
+ UNIT_ASSERT(result.UnsureTargets.empty());
+ }
+
+ Y_UNIT_TEST(ReportsDeadActorWithCookie) {
+ TTestActorRuntimeBase runtime;
+ runtime.Initialize();
+
+ const TActorId target(runtime.GetNodeId(), 0, Max<ui64>(), 0);
+ constexpr ui64 Cookie = 42;
+ TCheckResult result;
+ runtime.Register(new TTestActorLivenessChecker(
+ {{
+ .ActorId = target,
+ .Cookie = Cookie,
+ }},
+ result));
+
+ TDispatchOptions options;
+ options.CustomFinalCondition = [&result] {
+ return result.FinishCount;
+ };
+ runtime.DispatchEvents(options);
+
+ UNIT_ASSERT_VALUES_EQUAL(result.FinishCount, 1);
+ UNIT_ASSERT(result.AliveTargets.empty());
+ UNIT_ASSERT_VALUES_EQUAL(result.DeadTargets.size(), 1);
+ UNIT_ASSERT_VALUES_EQUAL(result.DeadTargets.front().ActorId, target);
+ UNIT_ASSERT_VALUES_EQUAL(result.DeadTargets.front().Cookie, Cookie);
+ UNIT_ASSERT(result.UnsureTargets.empty());
+ }
+
+ Y_UNIT_TEST(KeepsRemoteActorWhenLivenessIsUnsure) {
+ TTestActorRuntimeBase runtime;
+ runtime.Initialize();
+
+ const TActorId target(runtime.GetNodeId() + 1, 0, 1, 0);
+ constexpr ui64 Cookie = 10;
+ TCheckResult result;
+ runtime.Register(new TTestActorLivenessChecker(
+ {{
+ .ActorId = target,
+ .Cookie = Cookie,
+ }},
+ result));
+
+ TDispatchOptions options;
+ options.CustomFinalCondition = [&result] {
+ return result.FinishCount;
+ };
+ runtime.DispatchEvents(options);
+
+ UNIT_ASSERT_VALUES_EQUAL(result.FinishCount, 1);
+ UNIT_ASSERT(result.AliveTargets.empty());
+ UNIT_ASSERT(result.DeadTargets.empty());
+ UNIT_ASSERT_VALUES_EQUAL(result.UnsureTargets.size(), 1);
+ UNIT_ASSERT_VALUES_EQUAL(result.UnsureTargets.front().ActorId, target);
+ UNIT_ASSERT_VALUES_EQUAL(result.UnsureTargets.front().Cookie, Cookie);
+ }
+}
+
+}
diff --git a/ydb/library/actors/helpers/ut/ya.make b/ydb/library/actors/helpers/ut/ya.make
index 4ff76005089..76509caf89c 100644
--- a/ydb/library/actors/helpers/ut/ya.make
+++ b/ydb/library/actors/helpers/ut/ya.make
@@ -20,6 +20,7 @@ PEERDIR(
)
SRCS(
+ actor_liveness_checker_ut.cpp
selfping_actor_ut.cpp
)
diff --git a/ydb/library/actors/helpers/ya.make b/ydb/library/actors/helpers/ya.make
index 41dbece82c1..2b06263750c 100644
--- a/ydb/library/actors/helpers/ya.make
+++ b/ydb/library/actors/helpers/ya.make
@@ -3,6 +3,8 @@ LIBRARY()
SRCS(
activeactors.cpp
activeactors.h
+ actor_liveness_checker.cpp
+ actor_liveness_checker.h
collector_counters.cpp
future_callback.h
mon_histogram_helper.h
@@ -20,4 +22,3 @@ END()
RECURSE_FOR_TESTS(
ut
)
-
diff --git a/ydb/library/actors/interconnect/interconnect_common.h b/ydb/library/actors/interconnect/interconnect_common.h
index b70d17f2a87..01e5bac6d97 100644
--- a/ydb/library/actors/interconnect/interconnect_common.h
+++ b/ydb/library/actors/interconnect/interconnect_common.h
@@ -84,6 +84,7 @@ namespace NActors {
// 5s * 2^8 = 1280s, about 21 minutes with the current RDMA retry base delay.
ui32 MaxRdmaRetryBackoffLevel = 8;
bool CollectSubscriptionStackTrace = false;
+ TDuration SubscriberLivenessCheckInterval = TDuration::Hours(1);
bool UseUring = false;
bool EnableUringSQPOLL = false; // only effective when UseUring is set
diff --git a/ydb/library/actors/interconnect/interconnect_tcp_session.cpp b/ydb/library/actors/interconnect/interconnect_tcp_session.cpp
index 38d63517d8e..3cbb63c9f8b 100644
--- a/ydb/library/actors/interconnect/interconnect_tcp_session.cpp
+++ b/ydb/library/actors/interconnect/interconnect_tcp_session.cpp
@@ -2,6 +2,7 @@
#include "interconnect_tcp_session.h"
#include "interconnect_handshake.h"
#include "interconnect_zc_processor.h"
+#include "subscriber_liveness_checker.h"
#include <ydb/library/actors/core/probes.h>
#include <ydb/library/actors/core/log.h>
@@ -85,6 +86,10 @@ namespace NActors {
Pool = std::make_unique<TEventHolderPool>(Proxy->Common, std::move(destroyCallback));
ChannelScheduler.ConstructInPlace(Proxy->PeerNodeId, Proxy->Common->ChannelsConfig, Proxy->Metrics,
Proxy->Common->Settings.MaxSerializedEventSize, Params, Proxy->Common->RdmaMemPool);
+ if (const TDuration interval = Proxy->Common->Settings.SubscriberLivenessCheckInterval;
+ interval != TDuration::Zero()) {
+ Schedule(interval, new TEvCheckSubscriberLiveness);
+ }
LOG_INFO(*TlsActivationContext, NActorsServices::INTERCONNECT_STATUS, "[%u] session created", Proxy->PeerNodeId);
SetPrefix(Sprintf("Session %s [node %" PRIu32 "]", SelfId().ToString().data(), Proxy->PeerNodeId));
@@ -323,6 +328,16 @@ namespace NActors {
}
}
+ void TInterconnectSessionTCP::CheckSubscriberLiveness() {
+ const TDuration interval = Proxy->Common->Settings.SubscriberLivenessCheckInterval;
+ if (interval == TDuration::Zero()) {
+ return;
+ }
+
+ RegisterSubscriberLivenessChecker(SelfId(), Subscribers);
+ Schedule(interval, new TEvCheckSubscriberLiveness);
+ }
+
void TInterconnectSessionTCP::UpdateSubscriber(const TActorId& actorId, ui64 cookie, ui32 activityIndex, TString eventTypeName,
TString stackTrace) {
auto updateInfo = [&] (TSubscriberInfo& info) {
diff --git a/ydb/library/actors/interconnect/interconnect_tcp_session.h b/ydb/library/actors/interconnect/interconnect_tcp_session.h
index 58f2bbf08aa..33c3f268d34 100644
--- a/ydb/library/actors/interconnect/interconnect_tcp_session.h
+++ b/ydb/library/actors/interconnect/interconnect_tcp_session.h
@@ -466,10 +466,13 @@ namespace NActors {
EvRam,
EvTerminate,
EvFreeItems,
+ EvCheckSubscriberLiveness,
};
struct TEvCheckCloseOnIdle : TEventLocal<TEvCheckCloseOnIdle, EvCheckCloseOnIdle> {};
struct TEvCheckLostConnection : TEventLocal<TEvCheckLostConnection, EvCheckLostConnection> {};
+ struct TEvCheckSubscriberLiveness
+ : TEventLocal<TEvCheckSubscriberLiveness, EvCheckSubscriberLiveness> {};
struct TEvRam : TEventLocal<TEvRam, EvRam> {
const bool Batching;
@@ -563,6 +566,7 @@ namespace NActors {
void ForwardDelayed();
void Subscribe(STATEFN_SIG);
void Unsubscribe(STATEFN_SIG);
+ void CheckSubscriberLiveness();
void EnqueueForward(TAutoPtr<IEventHandle> ev);
void UpdateSubscriber(const TActorId& actorId, ui64 cookie, ui32 activityIndex = Max<ui32>(),
TString eventTypeName = {},
@@ -581,6 +585,7 @@ namespace NActors {
fFunc(TEvInterconnect::TEvConnectNode::EventType, Subscribe)
fFunc(TEvents::TEvSubscribe::EventType, Subscribe)
fFunc(TEvents::TEvUnsubscribe::EventType, Unsubscribe)
+ cFunc(TEvCheckSubscriberLiveness::EventType, CheckSubscriberLiveness)
cFunc(TEvFlush::EventType, HandleFlush)
hFunc(TEvPollerReady, Handle)
hFunc(TEvPollerRegisterResult, Handle)
diff --git a/ydb/library/actors/interconnect/interconnect_tcp_session_v2.cpp b/ydb/library/actors/interconnect/interconnect_tcp_session_v2.cpp
index a0f5ee860ae..5f7cf547c6d 100644
--- a/ydb/library/actors/interconnect/interconnect_tcp_session_v2.cpp
+++ b/ydb/library/actors/interconnect/interconnect_tcp_session_v2.cpp
@@ -1,5 +1,6 @@
#include "interconnect_tcp_session_v2.h"
#include "interconnect_tcp_proxy.h"
+#include "subscriber_liveness_checker.h"
#include <util/stream/str.h>
#include <util/string/cast.h>
@@ -56,6 +57,10 @@ namespace NActors {
Proxy->Metrics->SetPeerScopeId(Params.PeerScopeId);
Proxy->Metrics->SetConnected(0);
SetPrefix(Sprintf("SessionV2 %s [node %" PRIu32 "]", SelfId().ToString().data(), Proxy->PeerNodeId));
+ if (const TDuration interval = Proxy->Common->Settings.SubscriberLivenessCheckInterval;
+ interval != TDuration::Zero()) {
+ Schedule(interval, new TEvPrivate::TEvCheckSubscriberLiveness);
+ }
LOG_INFO_IC_SESSION("ICS90", "v2 session created");
}
@@ -107,8 +112,8 @@ namespace NActors {
XdcSocket->Shutdown(SHUT_RDWR);
}
- for (const auto& [actorId, cookie] : Subscribers) {
- Send(actorId, new TEvInterconnect::TEvNodeDisconnected(Proxy->PeerNodeId), 0, cookie);
+ for (const auto& [actorId, info] : Subscribers) {
+ Send(actorId, new TEvInterconnect::TEvNodeDisconnected(Proxy->PeerNodeId), 0, info.Cookie);
}
Subscribers.clear();
@@ -144,8 +149,11 @@ namespace NActors {
Terminate(TDisconnectReason::UserRequest());
}
- void TInterconnectSessionTCPv2::AddSubscriber(const TActorId& actorId, ui64 cookie) {
- Subscribers[actorId] = cookie;
+ void TInterconnectSessionTCPv2::AddSubscriber(const TActorId& actorId, ui64 cookie, ui32 activityIndex) {
+ Subscribers[actorId] = {
+ .Cookie = cookie,
+ .ActivityIndex = activityIndex,
+ };
}
IEventBase* TInterconnectSessionTCPv2::MakeNodeConnectedEvent() const {
@@ -172,7 +180,7 @@ namespace NActors {
Proxy->ValidateEvent(ev, "ForwardWithSubscribe");
auto msg = ev->Release<TEvForwardSubscribeSession>();
Y_ABORT_UNLESS(msg->Event);
- AddSubscriber(msg->Event->Sender, msg->Event->Cookie);
+ AddSubscriber(msg->Event->Sender, msg->Event->Cookie, msg->ActivityIndex);
Send(msg->Event->Sender, MakeNodeConnectedEvent(), 0, msg->Event->Cookie);
EnqueueOutgoing(TAutoPtr<IEventHandle>(msg->Event.Release()));
}
@@ -188,6 +196,16 @@ namespace NActors {
Subscribers.erase(ev->Sender);
}
+ void TInterconnectSessionTCPv2::CheckSubscriberLiveness() {
+ const TDuration interval = Proxy->Common->Settings.SubscriberLivenessCheckInterval;
+ if (interval == TDuration::Zero()) {
+ return;
+ }
+
+ RegisterSubscriberLivenessChecker(SelfId(), Subscribers);
+ Schedule(interval, new TEvPrivate::TEvCheckSubscriberLiveness);
+ }
+
void TInterconnectSessionTCPv2::HandlePoison() {
Terminate(TDisconnectReason::UserRequest());
}
@@ -204,6 +222,7 @@ namespace NActors {
// str << "<tr><td>OutstandingWrites</td><td>" << PendingBatches.size() << "</td></tr>";
str << "<tr><td>BytesSent</td><td>" << BytesSent << "</td></tr>";
str << "<tr><td>BytesReceived</td><td>" << BytesReceived << "</td></tr>";
+ str << "<tr><td>Subscribers.size()</td><td>" << Subscribers.size() << "</td></tr>";
str << "</table>";
str << "</div></div>";
TActivationContext::Send(new IEventHandle(ev->Recipient, ev->Sender, new NMon::TEvHttpInfoRes(str.Str())));
diff --git a/ydb/library/actors/interconnect/interconnect_tcp_session_v2.h b/ydb/library/actors/interconnect/interconnect_tcp_session_v2.h
index 59a27aa7fee..535f0249902 100644
--- a/ydb/library/actors/interconnect/interconnect_tcp_session_v2.h
+++ b/ydb/library/actors/interconnect/interconnect_tcp_session_v2.h
@@ -72,6 +72,7 @@ namespace NActors {
struct TEvPrivate {
enum {
EvTerminate = EventSpaceBegin(TEvents::ES_PRIVATE),
+ EvCheckSubscriberLiveness,
};
struct TEvTerminate : TEventLocal<TEvTerminate, EvTerminate> {
@@ -79,6 +80,9 @@ namespace NActors {
TEvTerminate(TDisconnectReason reason) : Reason(reason) {}
};
+
+ struct TEvCheckSubscriberLiveness
+ : TEventLocal<TEvCheckSubscriberLiveness, EvCheckSubscriberLiveness> {};
};
STATEFN(StateFunc) {
@@ -88,6 +92,7 @@ namespace NActors {
fFunc(TEvInterconnect::TEvConnectNode::EventType, HandleSubscribe)
fFunc(TEvents::TEvSubscribe::EventType, HandleSubscribe)
fFunc(TEvents::TEvUnsubscribe::EventType, HandleUnsubscribe)
+ cFunc(TEvPrivate::TEvCheckSubscriberLiveness::EventType, CheckSubscriberLiveness)
cFunc(TEvents::TEvPoisonPill::EventType, HandlePoison)
cFunc(TEvInterconnect::EvForwardDelayed, IgnoreForwardDelayed)
hFunc(TEvPrivate::TEvTerminate, [&](auto& ev) { Terminate(ev->Get()->Reason); });
@@ -98,12 +103,13 @@ namespace NActors {
void ForwardWithSubscribe(STATEFN_SIG);
void HandleSubscribe(STATEFN_SIG);
void HandleUnsubscribe(STATEFN_SIG);
+ void CheckSubscriberLiveness();
void HandlePoison();
void IgnoreForwardDelayed() {}
void EnqueueOutgoing(TAutoPtr<IEventHandle> ev);
- void AddSubscriber(const TActorId& actorId, ui64 cookie);
+ void AddSubscriber(const TActorId& actorId, ui64 cookie, ui32 activityIndex = Max<ui32>());
IEventBase* MakeNodeConnectedEvent() const;
private:
@@ -123,8 +129,13 @@ namespace NActors {
ui64 BytesSent = 0;
ui64 BytesReceived = 0;
- // subscribers awaiting connection state notifications (actor id -> cookie)
- THashMap<TActorId, ui64> Subscribers;
+ struct TSubscriberInfo {
+ ui64 Cookie = 0;
+ ui32 ActivityIndex = Max<ui32>();
+ };
+
+ // subscribers awaiting connection state notifications
+ THashMap<TActorId, TSubscriberInfo> Subscribers;
std::shared_ptr<std::atomic<int64_t>> ClockSkew = std::make_shared<std::atomic<int64_t>>();
std::shared_ptr<std::atomic<uint64_t>> PingRTT = std::make_shared<std::atomic<uint64_t>>();
diff --git a/ydb/library/actors/interconnect/interconnect_uring_engine.cpp b/ydb/library/actors/interconnect/interconnect_uring_engine.cpp
index 6dc9a8c97fc..f84153a4002 100644
--- a/ydb/library/actors/interconnect/interconnect_uring_engine.cpp
+++ b/ydb/library/actors/interconnect/interconnect_uring_engine.cpp
@@ -83,6 +83,8 @@ namespace NActors {
size_t UnsentBytes = 0;
int ReadPendingRingIdx = -1;
ui32 PreferredRingIdx = 0;
+ std::atomic_uint64_t IncomingSeqNo{1};
+ ui64 ExpectedSeqNo = 1;
EMigrateState MigrateState = EMigrateState::None;
ui32 MigrateTargetShard = 0;
@@ -94,6 +96,8 @@ namespace NActors {
THashMap<TActorId, TIntrusivePtr<IReceiveCallback>> ReceiveCallbacks;
NMonitoring::TDynamicCounters::TCounterPtr EventsReceived;
+ std::vector<TIncomingEventQueue::TRecord> PendingRecordsHeap;
+
TRegisteredSession(ui32 shardIdx, TIntrusivePtr<NInterconnect::TStreamSocket> socket,
TActorId sessionId, bool checksumming, TScopeId peerScopeId,
std::function<void(TDisconnectReason)> onDisconnectCallback, TActorSystem *actorSystem,
@@ -378,6 +382,8 @@ namespace NActors {
NMonitoring::TDynamicCounters::TCounterPtr WriteUnavail;
NMonitoring::TDynamicCounters::TCounterPtr SessionsMigratedOut;
NMonitoring::TDynamicCounters::TCounterPtr SessionsMigratedIn;
+ NMonitoring::TDynamicCounters::TCounterPtr OutOfOrderCameIn;
+ NMonitoring::TDynamicCounters::TCounterPtr OutOfOrderProcessed;
NMonitoring::TDynamicCounters::TCounterPtr OtherTotalTime;
NMonitoring::TDynamicCounters::TCounterPtr CompleteWaitTotalTime;
@@ -494,6 +500,8 @@ namespace NActors {
, COUNTER(WriteUnavail, true)
, COUNTER(SessionsMigratedOut, true)
, COUNTER(SessionsMigratedIn, true)
+ , COUNTER(OutOfOrderCameIn, true)
+ , COUNTER(OutOfOrderProcessed, true)
#define TOTAL_TIME(NAME) NAME(shardCounters->GetCounter("TotalTime/" #NAME, true))
, TOTAL_TIME(OtherTotalTime)
, TOTAL_TIME(CompleteWaitTotalTime)
@@ -561,42 +569,71 @@ namespace NActors {
void Register(std::unique_ptr<TRegisteredSession> session) {
++*SessionsRegistered;
+
+ // this would be the session's first event, so its sequencing is not the problem
SendInternal(reinterpret_cast<ui64>(session.release()), static_cast<ui32>(ENetwork::EvRegisterSession),
- {}, nullptr);
+ {}, nullptr, false);
}
void AcceptMigrated(std::unique_ptr<TRegisteredSession> session) {
++*SessionsMigratedIn;
+
+ // this event isn't the first one, but its sequence doesn't matter, because all further events are kept
+ // in order and forwarded to the new processor
SendInternal(reinterpret_cast<ui64>(session.release()), static_cast<ui32>(ENetwork::EvRegisterSession),
- {}, nullptr);
+ {}, nullptr, false);
}
- void Enqueue(ui64 conn, std::unique_ptr<IEventHandle> ev, TIntrusivePtr<IReceiveCallback> replyCallback) {
- SendImpl(conn, std::move(ev), std::move(replyCallback));
+ void Enqueue(TIncomingEventQueue::TRecord&& record) {
+ const bool first = IncomingEventQueue.Push(std::move(record));
+ if (first) {
+ ++*PushedAsFirst;
+ }
+ ++*PushedTotal;
+ if (first && WaitingForCQ.load(std::memory_order_acquire)) {
+ // first command while waiting on CQ: kick the worker via the pipe on ring 0
+ const ui64 value = 1; // this commands adds 1 to the counter stored in eventfd
+ ssize_t res;
+ while ((res = write(EventFd, &value, sizeof(value))) != sizeof(value)) {
+ if (res == -1 && errno == EINTR) {
+ continue;
+ } else {
+ Y_ABORT("write() to eventfd failed: %s", strerror(errno));
+ }
+ }
+ ++*EventWakeups;
+ }
}
void Send(ui64 conn, std::unique_ptr<IEventHandle> ev, TIntrusivePtr<IReceiveCallback> replyCallback) {
++*EventsSent;
- SendImpl(conn, std::move(ev), std::move(replyCallback));
+
+ // this event is strictly sequenced
+ SendImpl(conn, std::move(ev), std::move(replyCallback), true);
}
void Unregister(ui64 conn) {
++*SessionsUnregistered;
- SendInternal(conn, static_cast<ui32>(ENetwork::EvUnregisterSession), {}, nullptr);
+
+ // this event is sequenced too
+ SendInternal(conn, static_cast<ui32>(ENetwork::EvUnregisterSession), {}, nullptr, true);
}
void RegisterReceiveCallback(ui64 conn, TActorId localActorId, TIntrusivePtr<IReceiveCallback> callback) {
++*(callback ? DirectReceiveCallbacksRegistered : DirectReceiveCallbacksUnregistered);
- SendInternal(conn, static_cast<ui32>(ENetwork::EvRegisterCallback), localActorId, std::move(callback));
+
+ // this event's ordering is important
+ SendInternal(conn, static_cast<ui32>(ENetwork::EvRegisterCallback), localActorId, std::move(callback),
+ true);
}
void NotifyMigrateDone(ui64 conn) {
- SendInternal(conn, static_cast<ui32>(ENetwork::EvMigrateDone), {}, nullptr);
+ SendInternal(conn, static_cast<ui32>(ENetwork::EvMigrateDone), {}, nullptr, false);
}
void Stop() {
if (Worker.joinable()) {
- SendInternal(0, static_cast<ui32>(ENetwork::EvStop), {}, nullptr);
+ SendInternal(0, static_cast<ui32>(ENetwork::EvStop), {}, nullptr, false);
Worker.join();
// The worker is stopped, so it is now safe to touch the sessions directly. Shut every
// socket down so the peer observes the disconnect promptly instead of only when this shard
@@ -615,45 +652,42 @@ namespace NActors {
// the ownership handling of the worker loop: destroys the embedded TEventPayload and reclaims a
// TRegisteredSession handed off via an unprocessed EvRegisterSession.
void DrainQueue() {
- for (;;) {
- if (auto&& [ev, conn, callback, timestamp] = IncomingEventQueue.Pop(); ev) {
- if (ev->Type == static_cast<ui32>(ENetwork::EvRegisterSession)) {
- delete reinterpret_cast<TRegisteredSession*>(conn);
- }
- } else {
- break;
+ while (auto record = IncomingEventQueue.Pop()) {
+ if (record->Ev->Type == static_cast<ui32>(ENetwork::EvRegisterSession)) {
+ delete reinterpret_cast<TRegisteredSession*>(record->Conn);
}
}
}
- void SendImpl(ui64 conn, std::unique_ptr<IEventHandle> ev, TIntrusivePtr<IReceiveCallback> replyCallback) {
- const bool first = IncomingEventQueue.Push(std::move(ev), conn, std::move(replyCallback));
- if (first) {
- ++*PushedAsFirst;
- }
- ++*PushedTotal;
- if (first && WaitingForCQ.load(std::memory_order_relaxed)) {
- // first command while waiting on CQ: kick the worker via the pipe on ring 0
- const ui64 value = 1; // this commands adds 1 to the counter stored in eventfd
- ssize_t res;
- while ((res = write(EventFd, &value, sizeof(value))) != sizeof(value)) {
- if (res == -1 && errno == EINTR) {
- continue;
- } else {
- Y_ABORT("write() to eventfd failed: %s", strerror(errno));
- }
- }
- ++*EventWakeups;
+ void SendImpl(ui64 conn, std::unique_ptr<IEventHandle> ev, TIntrusivePtr<IReceiveCallback> replyCallback,
+ bool ensureSequence) {
+ ui64 seqNo = 0;
+ if (Y_LIKELY(ensureSequence)) {
+ Y_DEBUG_ABORT_UNLESS(conn);
+ auto& session = *reinterpret_cast<TRegisteredSession*>(conn);
+ seqNo = session.IncomingSeqNo.fetch_add(1);
}
+ Enqueue(TIncomingEventQueue::TRecord{
+ std::move(ev),
+ conn,
+ std::move(replyCallback),
+ GetCycleCountFast(),
+ seqNo,
+ });
}
- void SendInternal(ui64 conn, ui32 type, TActorId sender, TIntrusivePtr<IReceiveCallback> callback) {
- SendImpl(conn, std::make_unique<IEventHandle>(type, 0, TActorId(), sender, nullptr, 0), std::move(callback));
+ void SendInternal(ui64 conn, ui32 type, TActorId sender, TIntrusivePtr<IReceiveCallback> callback,
+ bool ensureSequence) {
+ SendImpl(conn, std::make_unique<IEventHandle>(type, 0, TActorId(), sender, nullptr, 0),
+ std::move(callback), ensureSequence);
}
- bool ForwardIfMigratingOut(ui64 conn, std::unique_ptr<IEventHandle>& ev, TIntrusivePtr<IReceiveCallback>& callback) {
- if (const auto it = MigratingOut.find(conn); it != MigratingOut.end()) {
- Engine.Shards[it->second]->Enqueue(conn, std::move(ev), std::move(callback));
+ bool ForwardIfMigratingOut(TIncomingEventQueue::TRecord *record) {
+ if (Y_LIKELY(MigratingOut.empty())) {
+ return false;
+ }
+ if (const auto it = MigratingOut.find(record->Conn); it != MigratingOut.end()) {
+ Engine.Shards[it->second]->Enqueue(std::move(*record));
return true;
}
return false;
@@ -772,7 +806,7 @@ namespace NActors {
ui64 waitStartTimestamp = 0;
if (!progress) { // wait for something to happen -- no progress were made in this loop
- WaitingForCQ.store(true, std::memory_order_relaxed);
+ WaitingForCQ.store(true, std::memory_order_release);
// it is critical we first set WaitingForCQ, and then rechecking the queue
if (IncomingEventQueue.IsEmpty()) {
@@ -807,99 +841,128 @@ namespace NActors {
bool progress = false;
ui64 cycleCountOnEnter = 0;
- for (;;) {
- auto&& [ev, conn, callback, cycleCountOnSend] = IncomingEventQueue.Pop();
- if (!ev) {
+ while (auto record = IncomingEventQueue.Pop()) {
+ progress = true;
+ if (record->Ev->Type == static_cast<ui32>(ENetwork::EvStop)) {
+ *stopping = true;
break;
}
- progress = true;
if (!cycleCountOnEnter) {
cycleCountOnEnter = GetCycleCountFast();
}
- switch (ev->Type) {
- case static_cast<ui32>(ENetwork::EvRegisterCallback):
- if (ForwardIfMigratingOut(conn, ev, callback)) {
- break;
- }
- if (TRegisteredSession& session = GetSession(conn); callback) {
- session.ReceiveCallbacks[ev->Sender] = std::move(callback);
- } else {
- session.ReceiveCallbacks.erase(ev->Sender);
+ if (Y_LIKELY(record->SeqNo)) {
+ if (ForwardIfMigratingOut(&record.value())) {
+ // this record is just sent to the other thread
+ } else if (auto& session = GetSession(record->Conn); record->SeqNo != session.ExpectedSeqNo) {
+ Y_DEBUG_ABORT_UNLESS(session.ExpectedSeqNo < record->SeqNo);
+ session.PendingRecordsHeap.push_back(std::move(*record));
+ std::ranges::push_heap(session.PendingRecordsHeap, std::greater<ui64>{},
+ &TIncomingEventQueue::TRecord::SeqNo);
+ ++*OutOfOrderCameIn;
+ } else {
+ // process this event
+ const bool isUnregister = record->Ev->Type == static_cast<ui32>(ENetwork::EvUnregisterSession);
+ ProcessIncomingEvent(&record.value());
+ if (!isUnregister) {
+ ++session.ExpectedSeqNo;
+
+ // check if there are other events in the process queue
+ if (auto& heap = session.PendingRecordsHeap; Y_UNLIKELY(!heap.empty())) {
+ while (!heap.empty() && heap.front().SeqNo == session.ExpectedSeqNo) {
+ std::ranges::pop_heap(heap, std::greater<ui64>{}, &TIncomingEventQueue::TRecord::SeqNo);
+ ProcessIncomingEvent(&heap.back());
+ ++session.ExpectedSeqNo;
+ heap.pop_back();
+ ++*OutOfOrderProcessed;
+ }
+ if (heap.empty()) {
+ heap.shrink_to_fit();
+ }
+ }
}
- break;
+ }
+ } else {
+ // process unsequenced event
+ ProcessIncomingEvent(&record.value());
+ }
- case static_cast<ui32>(ENetwork::EvRegisterSession): {
- std::unique_ptr<TRegisteredSession> session(reinterpret_cast<TRegisteredSession*>(conn));
- const bool migrated = session->MigrateState == EMigrateState::HandedOff;
- const ui32 sourceShard = session->MigrateSourceShard;
- session->OwnerShard.store(ShardIdx, std::memory_order_release);
+ const ui64 cycleCountOnExit = GetCycleCountFast();
+ CommandDeliveryTime->Collect((cycleCountOnEnter - record->ReceivedTimestamp) * Freq);
+ CommandExecTime->Collect((cycleCountOnExit - cycleCountOnEnter) * Freq);
+ cycleCountOnEnter = cycleCountOnExit;
+ }
+
+ return progress;
+ }
+
+ void ProcessIncomingEvent(TIncomingEventQueue::TRecord *record) {
+ switch (record->Ev->Type) {
+ case static_cast<ui32>(ENetwork::EvRegisterCallback):
+ if (TRegisteredSession& session = GetSession(record->Conn); record->Callback) {
+ session.ReceiveCallbacks[record->Ev->Sender] = std::move(record->Callback);
+ } else {
+ session.ReceiveCallbacks.erase(record->Ev->Sender);
+ }
+ break;
+
+ case static_cast<ui32>(ENetwork::EvRegisterSession): {
+ std::unique_ptr<TRegisteredSession> session(reinterpret_cast<TRegisteredSession*>(record->Conn));
+ if (session->MigrateState == EMigrateState::HandedOff) {
+ session->OwnerShard.store(ShardIdx, std::memory_order_release); // all new events will arrive here from now on
+ Engine.Shards[session->MigrateSourceShard]->NotifyMigrateDone(record->Conn);
session->MigrateState = EMigrateState::None;
- session->MigrateSourceShard = ShardIdx;
- session->PreferredRingIdx = OpShift++ % Rings.size();
- const auto [it, inserted] = Sessions.emplace(std::move(session));
- Y_ABORT_UNLESS(inserted);
- (*it)->EventsReceived = EventsReceived;
- IssueReadForSession(**it);
- if (migrated && sourceShard != ShardIdx) {
- Engine.Shards[sourceShard]->NotifyMigrateDone(conn);
- }
- break;
+ } else {
+ Y_DEBUG_ABORT_UNLESS(session->MigrateState == EMigrateState::None);
}
+ session->PreferredRingIdx = OpShift++ % Rings.size();
+ const auto [it, inserted] = Sessions.emplace(std::move(session));
+ Y_ABORT_UNLESS(inserted);
+ (*it)->EventsReceived = EventsReceived;
+ IssueReadForSession(**it);
+ break;
+ }
- case static_cast<ui32>(ENetwork::EvUnregisterSession): {
- if (ForwardIfMigratingOut(conn, ev, callback)) {
- break;
- }
- TRegisteredSession& session = GetSession(conn);
- // Do NOT free the session while it still has an armed recv or an in-flight
- // writev: their io_uring completions carry a raw pointer to this object and
- // would dereference freed memory. Mark it terminated (so no new ops are armed)
- // and erase only once both are drained. The session actor has already shut the
- // socket down before requesting unregistration, so the pending ops complete
- // promptly (EOF/EPIPE).
- if (session.MigrateState != EMigrateState::None) {
- session.MigrateState = EMigrateState::None; // unregister wins over migrate
- }
- session.Terminated = true;
- session.UnregisterRequested = true;
- if (session.ReadPending) { // cancel pending read in order to unregister the session
- CancelOp(session, kOpRead, session.ReadPendingRingIdx);
- }
- MaybeEraseSession(session);
- break;
+ case static_cast<ui32>(ENetwork::EvUnregisterSession): {
+ TRegisteredSession& session = GetSession(record->Conn);
+ // Do NOT free the session while it still has an armed recv or an in-flight
+ // writev: their io_uring completions carry a raw pointer to this object and
+ // would dereference freed memory. Mark it terminated (so no new ops are armed)
+ // and erase only once both are drained. The session actor has already shut the
+ // socket down before requesting unregistration, so the pending ops complete
+ // promptly (EOF/EPIPE).
+ if (session.MigrateState != EMigrateState::None) {
+ session.MigrateState = EMigrateState::None; // unregister wins over migrate
}
+ session.Terminated = true;
+ session.UnregisterRequested = true;
+ if (session.ReadPending) { // cancel pending read in order to unregister the session
+ CancelOp(session, kOpRead, session.ReadPendingRingIdx);
+ }
+ MaybeEraseSession(session);
+ break;
+ }
- case static_cast<ui32>(ENetwork::EvMigrateDone):
- MigratingOut.erase(conn);
- break;
+ case static_cast<ui32>(ENetwork::EvMigrateDone): {
+ const size_t num = MigratingOut.erase(record->Conn);
+ Y_DEBUG_ABORT_UNLESS(num == 1);
+ break;
+ }
- case static_cast<ui32>(ENetwork::EvStop):
- *stopping = true;
- return true;
+ case static_cast<ui32>(ENetwork::EvStop):
+ Y_ABORT();
- default: {
- if (ForwardIfMigratingOut(conn, ev, callback)) {
- break;
- }
- TRegisteredSession& session = GetSession(conn);
- if (callback) { // register callback coming along with the message
- session.ReceiveCallbacks[ev->Sender] = std::move(callback);
- }
- session.Serializer.Push(std::move(ev));
- IssueWritesForSession(session);
- break;
+ default: {
+ TRegisteredSession& session = GetSession(record->Conn);
+ if (record->Callback) { // register callback coming along with the message
+ session.ReceiveCallbacks[record->Ev->Sender] = std::move(record->Callback);
}
+ session.Serializer.Push(std::move(record->Ev));
+ IssueWritesForSession(session);
+ break;
}
-
- const ui64 cycleCountOnExit = GetCycleCountFast();
- CommandDeliveryTime->Collect(NHPTimer::GetSeconds(cycleCountOnEnter - cycleCountOnSend) * 1e9);
- CommandExecTime->Collect(NHPTimer::GetSeconds(cycleCountOnExit - cycleCountOnEnter) * 1e9);
- cycleCountOnEnter = cycleCountOnExit;
}
-
- return progress;
}
////////////////////////////////////////////////////////////////////////////////////////////////////////////
@@ -982,8 +1045,6 @@ namespace NActors {
return;
}
- return; // TODO(alexvru): check for out-of-order when processing old shard queue when new shard is registered
-
ui32 bestTarget = ShardIdx;
ui32 bestBusy = Max<ui32>();
for (ui32 i = 0; i < Engine.Shards.size(); ++i) {
@@ -1013,6 +1074,7 @@ namespace NActors {
candidate->MigrateState = EMigrateState::Draining;
candidate->MigrateTargetShard = bestTarget;
candidate->MigrateSourceShard = ShardIdx;
+ Y_DEBUG_ABORT_UNLESS(candidate->MigrateTargetShard != candidate->MigrateSourceShard);
if (candidate->ReadPending) {
CancelOp(*candidate, kOpRead, candidate->ReadPendingRingIdx);
diff --git a/ydb/library/actors/interconnect/interconnect_uring_event_queue.h b/ydb/library/actors/interconnect/interconnect_uring_event_queue.h
index e4849f8a614..da1a88643f5 100644
--- a/ydb/library/actors/interconnect/interconnect_uring_event_queue.h
+++ b/ydb/library/actors/interconnect/interconnect_uring_event_queue.h
@@ -15,6 +15,21 @@ namespace NActors {
};
static_assert(sizeof(TEventPayload) <= sizeof(TActorId));
+ struct TEventPayload2 {
+ ui64 ReceivedTimestamp; // GetCycleCountFast()
+ ui64 SeqNo;
+ };
+ static_assert(sizeof(TEventPayload2) <= sizeof(TScopeId));
+
+ public:
+ struct TRecord {
+ std::unique_ptr<IEventHandle> Ev;
+ ui64 Conn;
+ TIntrusivePtr<IReceiveCallback> Callback;
+ ui64 ReceivedTimestamp;
+ ui64 SeqNo;
+ };
+
public:
TIncomingEventQueue() {
Stub.NextLinkPtr.store(0, std::memory_order_relaxed);
@@ -24,15 +39,19 @@ namespace NActors {
Y_DEBUG_ABORT_UNLESS(IsEmpty()); // ensure this event queue has been properly drained by owner
}
- bool Push(std::unique_ptr<IEventHandle> ev, ui64 conn, TIntrusivePtr<IReceiveCallback> replyCallback) {
+ bool Push(TRecord&& record) {
+ IEventHandle *last = record.Ev.release();
+
// store some metadata in event's unmatching fields
- new(const_cast<TActorId*>(&ev->InterconnectSession)) TEventPayload{
- .Conn = conn,
- .Callback = std::move(replyCallback),
+ new(const_cast<TActorId*>(&last->InterconnectSession)) TEventPayload{
+ .Conn = record.Conn,
+ .Callback = std::move(record.Callback),
+ };
+ new(const_cast<TScopeId*>(&last->OriginScopeId)) TEventPayload2{
+ .ReceivedTimestamp = record.ReceivedTimestamp,
+ .SeqNo = record.SeqNo,
};
- reinterpret_cast<ui64&>(const_cast<TScopeId&>(ev->OriginScopeId)) = GetCycleCountFast();
- IEventHandle *last = ev.release();
last->NextLinkPtr.store(0, std::memory_order_relaxed);
IEventHandle *prev = Head.exchange(last, std::memory_order_acq_rel);
prev->NextLinkPtr.store(reinterpret_cast<uintptr_t>(last), std::memory_order_release);
@@ -46,14 +65,19 @@ namespace NActors {
return tail == &Stub && !next && tail == head;
}
- std::tuple<std::unique_ptr<IEventHandle>, ui64, TIntrusivePtr<IReceiveCallback>, ui64> Pop() {
+ std::optional<TRecord> Pop() {
auto decompose = [&](IEventHandle *ev) {
auto& payload = reinterpret_cast<TEventPayload&>(const_cast<TActorId&>(ev->InterconnectSession));
ui64 conn = payload.Conn;
TIntrusivePtr<IReceiveCallback> callback = std::move(payload.Callback);
payload.~TEventPayload();
- const ui64 timestamp = reinterpret_cast<const ui64&>(ev->OriginScopeId);
- return std::make_tuple(std::unique_ptr<IEventHandle>(ev), conn, std::move(callback), timestamp);
+
+ auto& payload2 = reinterpret_cast<TEventPayload2&>(const_cast<TScopeId&>(ev->OriginScopeId));
+ ui64 receivedTimestamp = payload2.ReceivedTimestamp;
+ ui64 seqNo = payload2.SeqNo;
+ payload2.~TEventPayload2();
+
+ return TRecord{std::unique_ptr<IEventHandle>(ev), conn, std::move(callback), receivedTimestamp, seqNo};
};
for (;;) {
@@ -66,7 +90,7 @@ namespace NActors {
if (head = Head.load(std::memory_order_acquire); tail != head) {
continue;
} else {
- return {};
+ return std::nullopt;
}
}
Tail.store(next, std::memory_order_relaxed);
diff --git a/ydb/library/actors/interconnect/subscriber_liveness_checker.cpp b/ydb/library/actors/interconnect/subscriber_liveness_checker.cpp
new file mode 100644
index 00000000000..6626447bbdf
--- /dev/null
+++ b/ydb/library/actors/interconnect/subscriber_liveness_checker.cpp
@@ -0,0 +1,77 @@
+#include "subscriber_liveness_checker.h"
+
+#include <ydb/library/actors/core/actor_bootstrapped.h>
+#include <ydb/library/actors/core/events.h>
+#include <ydb/library/actors/core/hfunc.h>
+#include <ydb/library/actors/core/log.h>
+#include <ydb/library/actors/protos/services_common.pb.h>
+
+#include <util/generic/map.h>
+#include <util/string/builder.h>
+
+namespace NActors {
+ namespace {
+
+ TStringBuf FormatSubscriberActivityName(ui32 activityIndex) {
+ return activityIndex == Max<ui32>() ? TStringBuf("manual") : GetActivityTypeName(activityIndex);
+ }
+
+ class TSubscriberLivenessChecker
+ : public TActorLivenessChecker
+ {
+ public:
+ TSubscriberLivenessChecker(
+ const TActorId& subscriptionOwner,
+ TVector<TActorLivenessCheckTarget> subscribers)
+ : TActorLivenessChecker(std::move(subscribers))
+ , SubscriptionOwner(subscriptionOwner)
+ {
+ }
+
+ private:
+ void OnDead(const TActorLivenessCheckTarget& target) override {
+ ++LeakedSubscribersByActivity[static_cast<ui32>(target.Cookie)];
+ // The IC session identifies the subscription by event sender.
+ TActivationContext::Send(new IEventHandle(
+ SubscriptionOwner,
+ target.ActorId,
+ new TEvents::TEvUnsubscribe));
+ }
+
+ void OnFinish() override {
+ LogLeakedSubscribers();
+ }
+
+ void LogLeakedSubscribers() const {
+ if (LeakedSubscribersByActivity.empty()) {
+ return;
+ }
+
+ TStringBuilder details;
+ bool first = true;
+ for (const auto& [activityIndex, count] : LeakedSubscribersByActivity) {
+ if (!first) {
+ details << ", ";
+ }
+ first = false;
+ details << "{activity# " << FormatSubscriberActivityName(activityIndex)
+ << " actors# " << count << '}';
+ }
+ LOG_WARN_S(*TlsActivationContext, NActorsServices::INTERCONNECT_SESSION,
+ "Subscriber liveness check found leaked subscriptions: " << details);
+ }
+
+ private:
+ const TActorId SubscriptionOwner;
+ TMap<ui32, ui64> LeakedSubscribersByActivity;
+ };
+
+ }
+
+ IActor* CreateSubscriberLivenessChecker(
+ const TActorId& subscriptionOwner,
+ TVector<TActorLivenessCheckTarget> subscribers) {
+ return new TSubscriberLivenessChecker(subscriptionOwner, std::move(subscribers));
+ }
+
+}
diff --git a/ydb/library/actors/interconnect/subscriber_liveness_checker.h b/ydb/library/actors/interconnect/subscriber_liveness_checker.h
new file mode 100644
index 00000000000..9d987297a83
--- /dev/null
+++ b/ydb/library/actors/interconnect/subscriber_liveness_checker.h
@@ -0,0 +1,36 @@
+#pragma once
+
+#include <ydb/library/actors/helpers/actor_liveness_checker.h>
+
+#include <util/generic/vector.h>
+
+#include <utility>
+
+namespace NActors {
+
+ IActor* CreateSubscriberLivenessChecker(
+ const TActorId& subscriptionOwner,
+ TVector<TActorLivenessCheckTarget> subscribers);
+
+ template <typename TSubscribers>
+ void RegisterSubscriberLivenessChecker(
+ const TActorId& subscriptionOwner,
+ const TSubscribers& subscribers) {
+ if (subscribers.empty()) {
+ return;
+ }
+
+ TVector<TActorLivenessCheckTarget> subscriberInfos;
+ subscriberInfos.reserve(subscribers.size());
+ for (const auto& item : subscribers) {
+ subscriberInfos.push_back({
+ .ActorId = item.first,
+ .Cookie = item.second.ActivityIndex,
+ });
+ }
+ TActivationContext::Register(
+ CreateSubscriberLivenessChecker(subscriptionOwner, std::move(subscriberInfos)),
+ subscriptionOwner);
+ }
+
+}
diff --git a/ydb/library/actors/interconnect/ut/interconnect_ut.cpp b/ydb/library/actors/interconnect/ut/interconnect_ut.cpp
index d82809d519f..50e0f53bea7 100644
--- a/ydb/library/actors/interconnect/ut/interconnect_ut.cpp
+++ b/ydb/library/actors/interconnect/ut/interconnect_ut.cpp
@@ -162,6 +162,37 @@ private:
std::atomic<size_t> Received = 0;
};
+class TConnectionSubscriberActor : public TActorBootstrapped<TConnectionSubscriberActor> {
+public:
+ explicit TConnectionSubscriberActor(ui32 peerNodeId)
+ : PeerNodeId(peerNodeId)
+ {}
+
+ void Bootstrap() {
+ Become(&TThis::StateFunc);
+ Send(TActivationContext::InterconnectProxy(PeerNodeId), new TEvents::TEvSubscribe);
+ }
+
+ bool IsConnected() const {
+ return Connected.load(std::memory_order_acquire);
+ }
+
+private:
+ void Handle(TEvInterconnect::TEvNodeConnected::TPtr&) {
+ Connected.store(true, std::memory_order_release);
+ }
+
+ STRICT_STFUNC(StateFunc,
+ hFunc(TEvInterconnect::TEvNodeConnected, Handle)
+ cFunc(TEvInterconnect::TEvNodeDisconnected::EventType, PassAway)
+ cFunc(TEvents::TSystem::Poison, PassAway)
+ )
+
+private:
+ const ui32 PeerNodeId;
+ std::atomic<bool> Connected = false;
+};
+
class TBurstSenderActor : public TActorBootstrapped<TBurstSenderActor> {
public:
TBurstSenderActor(TActorId recipient, size_t messages, size_t payloadSize)
@@ -525,6 +556,36 @@ private:
std::shared_ptr<THandshakeFailureLogCounters> OutgoingHandshakeFailures;
};
+struct TSubscriberLivenessLogState {
+ std::atomic<ui32> Warnings = 0;
+ TMutex Mutex;
+ TString LastWarning;
+};
+
+class TSubscriberLivenessLogBackend : public TLogBackend {
+public:
+ explicit TSubscriberLivenessLogBackend(std::shared_ptr<TSubscriberLivenessLogState> state)
+ : State(std::move(state))
+ {}
+
+ void WriteData(const TLogRecord& rec) override {
+ const TStringBuf line(rec.Data, rec.Len);
+ if (rec.Priority == TLOG_WARNING &&
+ line.Contains("Subscriber liveness check found leaked subscriptions")) {
+ with_lock (State->Mutex) {
+ State->LastWarning = line;
+ }
+ State->Warnings.fetch_add(1, std::memory_order_release);
+ }
+ }
+
+ void ReopenLog() override {
+ }
+
+private:
+ std::shared_ptr<TSubscriberLivenessLogState> State;
+};
+
} // namespace
class TSenderActor : public TActorBootstrapped<TSenderActor> {
@@ -1081,6 +1142,79 @@ void RunKernelLivenessReconnectLocalFallbackNotApplied(bool withRdma) {
UNIT_ASSERT_VALUES_EQUAL(WaitForSessionCounter(cluster, 2, 1, "Params.UseKernelLiveness"), 0ULL);
}
+void RunSubscriberLivenessCheck(bool useSessionV2, TDuration checkInterval) {
+ if (useSessionV2 && !TUringContext::IsAvailable()) {
+ Cerr << "io_uring not available; skipping" << Endl;
+ return;
+ }
+
+ auto settingsCustomizer = [=](ui32, TInterconnectSettings& settings) {
+ settings.SubscriberLivenessCheckInterval = checkInterval;
+ settings.V2.Enable = useSessionV2;
+ };
+ auto logState = std::make_shared<TSubscriberLivenessLogState>();
+ auto loggerSettings = MakeIntrusive<NLog::TSettings>(
+ TActorId(0, "logger"),
+ static_cast<NLog::EComponent>(NActorsServices::LOGGER),
+ NLog::PRI_DEBUG,
+ NLog::PRI_DEBUG,
+ 0U);
+ loggerSettings->Append(
+ NActorsServices::EServiceCommon_MIN,
+ NActorsServices::EServiceCommon_MAX,
+ NActorsServices::EServiceCommon_Name);
+ loggerSettings->SetAllowDrop(false);
+ loggerSettings->SetThrottleDelay(TDuration::Zero());
+ auto logBackendFactory = [logState] {
+ return TAutoPtr<TLogBackend>(new TSubscriberLivenessLogBackend(logState));
+ };
+ TTestICCluster cluster(2, TChannelsConfig(), nullptr, loggerSettings,
+ useSessionV2 ? TTestICCluster::EMPTY : TTestICCluster::DISABLE_RDMA,
+ {}, TDuration::Seconds(2), TNode::DefaultInflight(), settingsCustomizer, logBackendFactory);
+
+ auto* subscriber = new TConnectionSubscriberActor(1);
+ const TActorId subscriberId = cluster.RegisterActor(subscriber, 2);
+
+ WaitForCondition(TDuration::Seconds(10), [&] {
+ return subscriber->IsConnected();
+ }, "subscriber connected");
+ WaitForCondition(TDuration::Seconds(10), [&] {
+ try {
+ return GetSessionCounter(cluster, 2, 1, "Subscribers.size()") == 1;
+ } catch (const TPatternNotFound&) {
+ return false;
+ }
+ }, "live subscriber registered");
+
+ if (checkInterval != TDuration::Zero()) {
+ Sleep(3 * checkInterval);
+ UNIT_ASSERT_VALUES_EQUAL(GetSessionCounter(cluster, 2, 1, "Subscribers.size()"), 1);
+ UNIT_ASSERT_VALUES_EQUAL(logState->Warnings.load(std::memory_order_acquire), 0);
+ }
+
+ cluster.KillActor(2, subscriberId);
+ if (checkInterval != TDuration::Zero()) {
+ WaitForCondition(TDuration::Seconds(10), [&] {
+ try {
+ return GetSessionCounter(cluster, 2, 1, "Subscribers.size()") == 0;
+ } catch (const TPatternNotFound&) {
+ return false;
+ }
+ }, "dead subscriber removed");
+ WaitForCondition(TDuration::Seconds(10), [&] {
+ return logState->Warnings.load(std::memory_order_acquire) == 1;
+ }, "leaked subscriber warning");
+ with_lock (logState->Mutex) {
+ UNIT_ASSERT_STRING_CONTAINS(logState->LastWarning, "activity# manual");
+ UNIT_ASSERT_STRING_CONTAINS(logState->LastWarning, "actors# 1");
+ }
+ } else {
+ Sleep(TDuration::MilliSeconds(300));
+ UNIT_ASSERT_VALUES_EQUAL(GetSessionCounter(cluster, 2, 1, "Subscribers.size()"), 1);
+ UNIT_ASSERT_VALUES_EQUAL(logState->Warnings.load(std::memory_order_acquire), 0);
+ }
+}
+
} // namespace
Y_UNIT_TEST_SUITE(Interconnect) {
@@ -1091,6 +1225,22 @@ Y_UNIT_TEST_SUITE(Interconnect) {
RunScopeClassCounterRebindTest(TScopeId(2, 42), "other_tenant");
}
+ Y_UNIT_TEST(SubscriberLivenessCheck) {
+ RunSubscriberLivenessCheck(false, TDuration::MilliSeconds(100));
+ }
+
+ Y_UNIT_TEST(SubscriberLivenessCheckV2) {
+ RunSubscriberLivenessCheck(true, TDuration::MilliSeconds(100));
+ }
+
+ Y_UNIT_TEST(SubscriberLivenessCheckDisabled) {
+ RunSubscriberLivenessCheck(false, TDuration::Zero());
+ }
+
+ Y_UNIT_TEST(SubscriberLivenessCheckDisabledV2) {
+ RunSubscriberLivenessCheck(true, TDuration::Zero());
+ }
+
Y_UNIT_TEST(RdmaRetryWatchdogPendingSessionsAggregated) {
TTestActorRuntimeBase runtime;
runtime.Initialize();
diff --git a/ydb/library/actors/interconnect/ya.make b/ydb/library/actors/interconnect/ya.make
index 278b15da8d6..36f75e47534 100644
--- a/ydb/library/actors/interconnect/ya.make
+++ b/ydb/library/actors/interconnect/ya.make
@@ -57,6 +57,8 @@ SRCS(
profiler.h
rdma_sync_actor.cpp
slowpoke_actor.h
+ subscriber_liveness_checker.cpp
+ subscriber_liveness_checker.h
subscription_manager.cpp
subscription_manager.h
types.cpp
diff --git a/ydb/library/services/services.proto b/ydb/library/services/services.proto
index 250af7fd223..370b7b0761f 100644
--- a/ydb/library/services/services.proto
+++ b/ydb/library/services/services.proto
@@ -480,6 +480,8 @@ enum EServiceKikimr {
// BlobChecker
BLOB_CHECKER_ORCHESTRATOR = 4300;
BLOB_CHECKER_WORKER = 4301;
+
+ DBS_CONTROLLER = 4400;
};
message TActivity {
diff --git a/ydb/library/workload/vector/vector_command_index.cpp b/ydb/library/workload/vector/vector_command_index.cpp
index 61d39c4b2c3..fd8607d60f1 100644
--- a/ydb/library/workload/vector/vector_command_index.cpp
+++ b/ydb/library/workload/vector/vector_command_index.cpp
@@ -54,9 +54,14 @@ int TWorkloadCommandBuildIndex::DoRun() {
ddlQuery << "GLOBAL USING vector_kmeans_tree\n";
ddlQuery << "ON (embedding)\n";
ddlQuery << "WITH (\n";
- ddlQuery << " " << Params.GetDistanceDDL() << ",\n";
- ddlQuery << " vector_type=" << Params.VectorOpts.VectorType << ",\n";
- ddlQuery << " vector_dimension=" << Params.VectorOpts.VectorDimension;
+ ddlQuery << " " << Params.GetDistanceDDL();
+ // When VectorDimension is 0, omit vector_type/vector_dimension so the server
+ // autodetects them from the table data. This is used for datasets imported
+ // from an external source (e.g. S3) whose dimension is not known in advance.
+ if (Params.VectorOpts.VectorDimension) {
+ ddlQuery << ",\n vector_type=" << Params.VectorOpts.VectorType;
+ ddlQuery << ",\n vector_dimension=" << Params.VectorOpts.VectorDimension;
+ }
if (Params.KmeansTreeLevels) {
ddlQuery << ",\n levels=" << Params.KmeansTreeLevels;
}
diff --git a/ydb/library/workload/vector/vector_recall_evaluator.cpp b/ydb/library/workload/vector/vector_recall_evaluator.cpp
index 7733700813e..8b6f501a1b2 100644
--- a/ydb/library/workload/vector/vector_recall_evaluator.cpp
+++ b/ydb/library/workload/vector/vector_recall_evaluator.cpp
@@ -39,7 +39,7 @@ void TVectorRecallEvaluator::SelectReferenceResults(const TVectorSampler& sample
<< "SELECT s.id AS id"
<< ", " << (isAscending ? "BOTTOM_BY" : "TOP_BY") << "(" << MakeKeyExpression(Params, "m.") <<
", Knn::" << functionName << "(m." << Params.EmbeddingColumn << ", s.embedding), " << Params.Limit << ") result_ids"
- << " FROM " << Params.TableOpts.Name << " m"
+ << " FROM `" << Params.TableOpts.Name << "` m"
<< (Params.PrefixColumn ? " INNER JOIN " : " CROSS JOIN ") << "AS_TABLE($Samples) AS s";
if (Params.PrefixColumn) {
refQueryBuilder << " ON s.prefix = m." << *Params.PrefixColumn;
diff --git a/ydb/library/workload/vector/vector_sampler.cpp b/ydb/library/workload/vector/vector_sampler.cpp
index 74e1cbc5a08..f5795a00942 100644
--- a/ydb/library/workload/vector/vector_sampler.cpp
+++ b/ydb/library/workload/vector/vector_sampler.cpp
@@ -19,7 +19,7 @@ TVectorSampler::TVectorSampler(const TVectorWorkloadParams& params)
ui64 TVectorSampler::SelectOneId(bool min) {
std::string query = std::format(R"_(--!syntax_v1
- SELECT Unwrap(CAST({1} AS uint64)) AS id FROM {0} ORDER BY {1} {2} LIMIT 1;
+ SELECT Unwrap(CAST({1} AS uint64)) AS id FROM `{0}` ORDER BY {1} {2} LIMIT 1;
)_", Params.TableOpts.Name.c_str(), Params.KeyColumns[0].c_str(), min ? "" : "DESC");
// Execute the query
@@ -50,7 +50,7 @@ void TVectorSampler::SelectPredefinedVectors() {
if (Params.PrefixColumn) {
vectorQueryBuilder << ", Unwrap(" << Params.PrefixColumn << ") as prefix_value";
}
- vectorQueryBuilder << " FROM " << Params.QueryTableName
+ vectorQueryBuilder << " FROM `" << Params.QueryTableName << "`"
<< " ORDER BY ";
for (size_t i = 0; i < Params.QueryTableKeyColumns.size(); i++) {
if (i > 0) {
@@ -146,13 +146,13 @@ void TVectorSampler::SampleExistingVectors() {
if (Params.PrefixColumn) {
vectorQuery = std::format(R"_(--!syntax_v1
SELECT Unwrap(CAST({0} as uint64)) as id, Unwrap({1}) as embedding, Unwrap({2}) as prefix_value
- FROM {3}
+ FROM `{3}`
WHERE {0} = {4};
)_", Params.KeyColumns[0].c_str(), Params.EmbeddingColumn.c_str(), Params.PrefixColumn->c_str(), Params.TableOpts.Name.c_str(), randomId);
} else {
vectorQuery = std::format(R"_(--!syntax_v1
SELECT Unwrap(CAST({0} as uint64)) as id, Unwrap({1}) as embedding
- FROM {2}
+ FROM `{2}`
WHERE {0} = {3};
)_", Params.KeyColumns[0].c_str(), Params.EmbeddingColumn.c_str(), Params.TableOpts.Name.c_str(), randomId);
}
diff --git a/ydb/library/workload/vector/vector_sql.cpp b/ydb/library/workload/vector/vector_sql.cpp
index e406a2520f6..2e2658c6efa 100644
--- a/ydb/library/workload/vector/vector_sql.cpp
+++ b/ydb/library/workload/vector/vector_sql.cpp
@@ -64,7 +64,7 @@ std::string MakeSelect(const TVectorWorkloadParams& params, const TString& index
if (params.PrefixColumn)
ret << "DECLARE $PrefixValue as " << params.PrefixType << ";" << "\n";
ret << "pragma ydb.KMeansTreeSearchTopSize=\"" << params.KmeansTreeSearchClusters << "\";" << "\n";
- ret << "SELECT " << MakeKeyExpression(params, "") << " AS id FROM " << params.TableOpts.Name << "\n";
+ ret << "SELECT " << MakeKeyExpression(params, "") << " AS id FROM `" << params.TableOpts.Name << "`\n";
if (!indexName.empty())
ret << "VIEW " << indexName << "\n";
if (params.PrefixColumn)
diff --git a/ydb/library/yql/providers/dq/task_runner/tasks_runner_pipe.cpp b/ydb/library/yql/providers/dq/task_runner/tasks_runner_pipe.cpp
index f0b49e35fc0..fa8323abdf6 100644
--- a/ydb/library/yql/providers/dq/task_runner/tasks_runner_pipe.cpp
+++ b/ydb/library/yql/providers/dq/task_runner/tasks_runner_pipe.cpp
@@ -1927,6 +1927,14 @@ public:
// Stats.CodeGenFinalizeTime = f.GetCodeGenFinalizeTime();
// Stats.CodeGenModulePassTime = f.GetCodeGenModulePassTime();
+ Stats.MkqlStats.clear();
+ for (const auto& stat : protoStats.GetMkqlStats()) {
+ Stats.MkqlStats.emplace_back(TMkqlStat{
+ TStatKey(stat.GetName(), stat.GetDeriv()),
+ stat.GetValue()
+ });
+ }
+
for (const auto& input : protoStats.GetInputChannels()) {
InputChannels[input.GetChannelId()]->FromProto(input);
}
diff --git a/ydb/library/yql/providers/generic/actors/ut/yql_generic_lookup_actor_ut.cpp b/ydb/library/yql/providers/generic/actors/ut/yql_generic_lookup_actor_ut.cpp
index 9b529c720e6..d75f8e2a161 100644
--- a/ydb/library/yql/providers/generic/actors/ut/yql_generic_lookup_actor_ut.cpp
+++ b/ydb/library/yql/providers/generic/actors/ut/yql_generic_lookup_actor_ut.cpp
@@ -51,7 +51,21 @@ Y_UNIT_TEST_SUITE(GenericProviderLookupActor) {
, Edge(edge)
, FullscanLimit(fullscanLimit)
{
+ }
+ TCallLookupActor(
+ std::shared_ptr<NKikimr::NMiniKQL::TScopedAlloc>&& alloc,
+ TLookupActorFactory&& lookupActorFactory,
+ NYql::NDqProto::StatusIds::StatusCode expectedError,
+ const NActors::TActorId& edge,
+ size_t fullscanLimit = 0)
+ : Alloc(std::move(alloc))
+ , TypeEnv(std::make_shared<NKikimr::NMiniKQL::TTypeEnvironment>(*Alloc))
+ , LookupActorFactory(std::move(lookupActorFactory))
+ , ExpectedError(expectedError)
+ , Edge(edge)
+ , FullscanLimit(fullscanLimit)
+ {
}
void Bootstrap() {
@@ -69,10 +83,12 @@ Y_UNIT_TEST_SUITE(GenericProviderLookupActor) {
STFUNC(StateFunc) {
switch (ev->GetTypeRewrite()) {
hFunc(NYql::NDq::IDqAsyncLookupSource::TEvLookupResult, Handle);
+ hFunc(NYql::NDq::IDqComputeActorAsyncInput::TEvAsyncInputError, Handle);
}
}
void Handle(NYql::NDq::IDqAsyncLookupSource::TEvLookupResult::TPtr& ev) {
+ UNIT_ASSERT(!ExpectedError);
Callback(Alloc, ev);
{
auto guard = Guard(*Alloc);
@@ -83,6 +99,12 @@ Y_UNIT_TEST_SUITE(GenericProviderLookupActor) {
Send(Edge, new NActors::TEvents::TEvWakeup());
}
+ void Handle(NYql::NDq::IDqComputeActorAsyncInput::TEvAsyncInputError::TPtr& ev) {
+ UNIT_ASSERT(ExpectedError);
+ UNIT_ASSERT_EQUAL_C(ev->Get()->FatalCode, *ExpectedError, static_cast<ui64>(ev->Get()->FatalCode) << " != " << static_cast<ui64>(*ExpectedError) << ", issues = " << ev->Get()->Issues.ToOneLineString());
+ Send(Edge, new NActors::TEvents::TEvWakeup());
+ }
+
void Free() {
if (Alloc) {
auto guard = Guard(*Alloc);
@@ -112,6 +134,7 @@ Y_UNIT_TEST_SUITE(GenericProviderLookupActor) {
NActors::TActorId LookupActor;
TLookupActorFactory LookupActorFactory;
TCallback Callback;
+ std::optional<NYql::NDqProto::StatusIds::StatusCode> ExpectedError;
NActors::TActorId Edge;
size_t FullscanLimit;
};
@@ -507,6 +530,159 @@ Y_UNIT_TEST_SUITE(GenericProviderLookupActor) {
runtime.GrabEdgeEventRethrow<NActors::TEvents::TEvWakeup>(edge);
}
+ Y_UNIT_TEST_TWIN(LookupWithFatalAuthErrors, ListSplitsOrReadError) {
+ auto alloc = std::make_shared<NKikimr::NMiniKQL::TScopedAlloc>(__LOCATION__, NKikimr::TAlignedPagePoolCounters(), true, false);
+ NKikimr::NMiniKQL::TMemoryUsageInfo memUsage("TestMemUsage");
+ NKikimr::NMiniKQL::THolderFactory holderFactory(alloc->Ref(), memUsage);
+
+ auto loggerConfig = NYql::NProto::TLoggingConfig();
+ loggerConfig.set_allcomponentslevel(::NYql::NProto::TLoggingConfig_ELevel::TLoggingConfig_ELevel_TRACE);
+ NYql::NLog::InitLogger(loggerConfig, false);
+
+ TTestActorRuntimeBase runtime(1, 1, true);
+ runtime.Initialize();
+ auto edge = runtime.AllocateEdgeActor();
+
+ NYql::TGenericDataSourceInstance dsi;
+ dsi.Setkind(NYql::EGenericDataSourceKind::YDB);
+ dsi.mutable_endpoint()->Sethost("some_host");
+ dsi.mutable_endpoint()->Setport(2135);
+ dsi.Setdatabase("some_db");
+ dsi.Setuse_tls(true);
+ dsi.set_protocol(::NYql::EGenericProtocol::NATIVE);
+ auto token = dsi.mutable_credentials()->mutable_token();
+ token->Settype("IAM");
+ token->Setvalue("token_value");
+
+ auto connectorMock = std::make_shared<NYql::NConnector::NTest::TConnectorClientMock>();
+
+ // clang-format off
+ // step 1: ListSplits
+ {
+ ::testing::InSequence seq;
+ {
+ auto listBuilder = connectorMock->ExpectListSplits();
+ listBuilder
+ .Select()
+ .DataSourceInstance(dsi)
+ .What()
+ .Column("id", Ydb::Type::UINT64)
+ .NullableColumn("optional_id", Ydb::Type::UINT64)
+ .NullableColumn("string_value", Ydb::Type::STRING)
+ .Done()
+ .Table("lookup_test")
+ .Where()
+ .Filter()
+ .Disjunction()
+ .Operand()
+ .Conjunction()
+ .Operand().Equal().Column("id").Value<ui64>(2).Done().Done()
+ .Operand().Equal().Column("optional_id").OptionalValue<ui64>(102).Done().Done()
+ .Done()
+ .Done()
+ .Operand()
+ .Conjunction()
+ .Operand().Equal().Column("id").Value<ui64>(1).Done().Done()
+ .Operand().Equal().Column("optional_id").OptionalValue<ui64>(101).Done().Done()
+ .Done()
+ .Done()
+ .Operand()
+ .Conjunction()
+ .Operand().Equal().Column("id").Value<ui64>(0).Done().Done()
+ .Operand().Equal().Column("optional_id").OptionalValue<ui64>(100).Done().Done()
+ .Done()
+ .Done()
+ .Operand()
+ .Conjunction()
+ .Operand().Equal().Column("id").Value<ui64>(2).Done().Done()
+ .Operand().Equal().Column("optional_id").OptionalValue<ui64>(102).Done().Done()
+ .Done()
+ .Done()
+ .Done()
+ .Done()
+ .Done()
+ .Done()
+ .MaxSplitCount(1)
+ ;
+ if (ListSplitsOrReadError) {
+ listBuilder
+ .Status(NYdbGrpc::TGrpcStatus(grpc::StatusCode::UNAUTHENTICATED, "Mocked Error"))
+ ;
+ } else {
+ listBuilder
+ .Result()
+ .AddResponse(NewSuccess())
+ .Description("Actual split info is not important")
+ ;
+ }
+ }
+
+ if (!ListSplitsOrReadError) {
+ auto readBuilder = connectorMock->ExpectReadSplits();
+ readBuilder
+ .DataSourceInstance(dsi)
+ .Filtering(NYql::NConnector::NApi::TReadSplitsRequest::FILTERING_MANDATORY)
+ .Split()
+ .Description("Actual split info is not important")
+ .Done()
+ .Status(NYdbGrpc::TGrpcStatus(grpc::StatusCode::UNAUTHENTICATED, "Mocked Error"))
+ ;
+ }
+ }
+ // clang-format on
+
+ NYql::Generic::TLookupSource lookupSourceSettings;
+ *lookupSourceSettings.mutable_data_source_instance() = dsi;
+ lookupSourceSettings.Settable("lookup_test");
+ lookupSourceSettings.SetTokenName("test_token");
+
+ google::protobuf::Any packedLookupSource;
+ Y_ABORT_UNLESS(packedLookupSource.PackFrom(lookupSourceSettings));
+
+ auto lookupActorFactory = [&holderFactory, connectorMock = std::move(connectorMock), lookupSourceSettings = std::move(lookupSourceSettings)](const NActors::TActorId& caller, std::shared_ptr<NKikimr::NMiniKQL::TScopedAlloc>& alloc, NKikimr::NMiniKQL::TTypeEnvironment& typeEnv) mutable {
+ NKikimr::NMiniKQL::TTypeBuilder typeBuilder(typeEnv);
+
+ NKikimr::NMiniKQL::TStructTypeBuilder keyTypeBuilder{typeEnv};
+ keyTypeBuilder.Add("id", typeBuilder.NewDataType(NYql::NUdf::EDataSlot::Uint64, false));
+ keyTypeBuilder.Add("optional_id", typeBuilder.NewDataType(NYql::NUdf::EDataSlot::Uint64, true));
+ NKikimr::NMiniKQL::TStructTypeBuilder outputypeBuilder{typeEnv};
+ outputypeBuilder.Add("string_value", typeBuilder.NewDataType(NYql::NUdf::EDataSlot::String, true));
+
+ auto guard = Guard(*alloc.get());
+ auto keyTypeHelper = std::make_shared<NYql::NDq::IDqAsyncLookupSource::TKeyTypeHelper>(keyTypeBuilder.Build());
+
+ auto [lookupSource, actor] = NYql::NDq::CreateGenericLookupActor(
+ std::move(connectorMock),
+ NKikimr::NKqp::NFederatedQueryTest::CreateCredentialsFactory("token_value"),
+ caller,
+ nullptr,
+ alloc,
+ keyTypeHelper,
+ std::move(lookupSourceSettings),
+ keyTypeBuilder.Build(),
+ outputypeBuilder.Build(),
+ typeEnv,
+ holderFactory,
+ 1'000'000,
+ {{"test_token", "{\"token\": \"token_value\"}"}});
+
+ auto request = std::make_shared<NYql::NDq::IDqAsyncLookupSource::TUnboxedValueMap>(3, keyTypeHelper->GetValueHash(), keyTypeHelper->GetValueEqual());
+ for (size_t i = 0; i != 3; ++i) {
+ NYql::NUdf::TUnboxedValue* keyItems;
+ auto key = holderFactory.CreateDirectArrayHolder(2, keyItems);
+ keyItems[0] = NYql::NUdf::TUnboxedValuePod(ui64(i));
+ keyItems[1] = NYql::NUdf::TUnboxedValuePod(ui64(100 + i));
+ request->emplace(std::move(key), NYql::NUdf::TUnboxedValue{});
+ }
+
+ return std::pair { actor, std::move(request) };
+ };
+
+ auto callLookupActor = new TCallLookupActor(std::move(alloc), std::move(lookupActorFactory), NYql::NDqProto::StatusIds::UNAUTHORIZED, edge);
+ runtime.Register(callLookupActor);
+ runtime.GrabEdgeEventRethrow<NActors::TEvents::TEvWakeup>(edge);
+ }
+
class TMockStructuredTokenCredentialsFactory : public NYql::IStructuredTokenCredentialsFactory {
public:
TMockStructuredTokenCredentialsFactory(const std::string yqlToken, const TVector<bool>& pattern)
diff --git a/ydb/library/yql/providers/generic/actors/yql_generic_lookup_actor.cpp b/ydb/library/yql/providers/generic/actors/yql_generic_lookup_actor.cpp
index bac77c67a02..ae01bfe738f 100644
--- a/ydb/library/yql/providers/generic/actors/yql_generic_lookup_actor.cpp
+++ b/ydb/library/yql/providers/generic/actors/yql_generic_lookup_actor.cpp
@@ -281,17 +281,12 @@ namespace NYql::NDq {
}
void Handle(TEvError::TPtr ev) {
- const auto error = ev->Get()->Error;
- auto issues = NConnector::ErrorToIssues(error);
- auto fatalCode = NDqProto::StatusIds::INTERNAL_ERROR;
-
- try {
- fatalCode = NConnector::ErrorToDqStatus(error);
- } catch (const std::exception& e) {
- issues.AddIssue(TStringBuilder() << "Failed to convert YDB status code: " << e.what());
- }
-
- Send(ParentId, new IDqComputeActorAsyncInput::TEvAsyncInputError(-1, std::move(issues), fatalCode));
+ const auto& error = ev->Get()->Error;
+ Send(ParentId,
+ new IDqComputeActorAsyncInput::TEvAsyncInputError(
+ /*inputId=*/-1, /* will be filled by LookupTransform actor */
+ NConnector::ErrorToIssues(error),
+ NConnector::ErrorToDqStatus(error)));
}
void Handle(TEvLookupRetry::TPtr ev) {
diff --git a/ydb/library/yql/providers/generic/connector/libcpp/error.cpp b/ydb/library/yql/providers/generic/connector/libcpp/error.cpp
index 2ce7d8b5df0..c8695985d41 100644
--- a/ydb/library/yql/providers/generic/connector/libcpp/error.cpp
+++ b/ydb/library/yql/providers/generic/connector/libcpp/error.cpp
@@ -1,14 +1,15 @@
#include "error.h"
#include <grpcpp/impl/codegen/status_code_enum.h>
+#include <ydb/core/grpc_services/local_rpc/local_rpc.h> // for NKikimr::NRpcService::GrpcStatusToYdbStatus
+#include <ydb/library/yql/dq/actors/dq.h>
#include <yql/essentials/public/issue/yql_issue_message.h>
#include <yql/essentials/utils/yql_panic.h>
-#include <ydb/public/api/protos/ydb_status_codes.pb.h>
namespace NYql::NConnector {
NApi::TError NewSuccess() {
NApi::TError error;
- error.set_status(Ydb::StatusIds_StatusCode::StatusIds_StatusCode_SUCCESS);
+ error.set_status(Ydb::StatusIds::SUCCESS);
return error;
}
@@ -28,30 +29,16 @@ namespace NYql::NConnector {
}
NDqProto::StatusIds::StatusCode ErrorToDqStatus(const NApi::TError& error) {
- switch (error.status()) {
- case ::Ydb::StatusIds::StatusCode::StatusIds_StatusCode_BAD_REQUEST:
- return NDqProto::StatusIds::StatusCode::StatusIds_StatusCode_BAD_REQUEST;
- case ::Ydb::StatusIds::StatusCode::StatusIds_StatusCode_INTERNAL_ERROR:
- return NDqProto::StatusIds::StatusCode::StatusIds_StatusCode_INTERNAL_ERROR;
- case ::Ydb::StatusIds::StatusCode::StatusIds_StatusCode_UNSUPPORTED:
- return NDqProto::StatusIds::StatusCode::StatusIds_StatusCode_UNSUPPORTED;
- case ::Ydb::StatusIds::StatusCode::StatusIds_StatusCode_NOT_FOUND:
- return NDqProto::StatusIds::StatusCode::StatusIds_StatusCode_BAD_REQUEST;
- case ::Ydb::StatusIds::StatusCode::StatusIds_StatusCode_SCHEME_ERROR:
- return NDqProto::StatusIds::StatusCode::StatusIds_StatusCode_SCHEME_ERROR;
- default:
- ythrow yexception() << "Unexpected YDB status code: " << ::Ydb::StatusIds::StatusCode_Name(error.status());
- }
+ return NYql::NDq::YdbStatusToDqStatus(error.status(), NYql::NDq::EStatusCompatibilityLevel::WithUnauthorized);
}
NApi::TError ErrorFromGRPCStatus(const NYdbGrpc::TGrpcStatus& status) {
NApi::TError result;
- if (status.GRpcStatusCode == grpc::OK) {
- result.set_status(Ydb::StatusIds_StatusCode::StatusIds_StatusCode_SUCCESS);
+ if (status.Ok()) {
+ result.set_status(Ydb::StatusIds::SUCCESS);
} else {
- // FIXME: more appropriate error code for network error
- result.set_status(Ydb::StatusIds_StatusCode::StatusIds_StatusCode_INTERNAL_ERROR);
+ result.set_status(status.InternalError ? Ydb::StatusIds::INTERNAL_ERROR : NKikimr::NRpcService::GrpcStatusToYdbStatus(static_cast<grpc::StatusCode>(status.GRpcStatusCode)));
result.set_message(TString{status.Msg});
}
diff --git a/ydb/library/yql/providers/generic/connector/libcpp/ya.make b/ydb/library/yql/providers/generic/connector/libcpp/ya.make
index d7b7fd6fdf8..201bd912f0e 100644
--- a/ydb/library/yql/providers/generic/connector/libcpp/ya.make
+++ b/ydb/library/yql/providers/generic/connector/libcpp/ya.make
@@ -10,7 +10,9 @@ PEERDIR(
contrib/libs/apache/arrow
contrib/libs/grpc
ydb/core/formats/arrow/serializer
+ ydb/core/grpc_services/local_rpc
ydb/public/sdk/cpp/src/library/grpc/client
+ ydb/library/yql/dq/actors
ydb/library/yql/dq/actors/protos
yql/essentials/providers/common/proto
yql/essentials/providers/common/proto
diff --git a/ydb/library/yql/providers/generic/connector/tests/utils/ya.make b/ydb/library/yql/providers/generic/connector/tests/utils/ya.make
index 0c2159ee704..5bbd0a6d5b6 100644
--- a/ydb/library/yql/providers/generic/connector/tests/utils/ya.make
+++ b/ydb/library/yql/providers/generic/connector/tests/utils/ya.make
@@ -22,6 +22,7 @@ ENDIF()
PEERDIR(
contrib/python/PyYAML
yql/essentials/providers/common/proto
+ ydb/library/yql/providers/generic/connector/api/service
ydb/library/yql/providers/generic/connector/tests/utils/types
ydb/public/api/protos
)
diff --git a/ydb/mvp/meta/support_links/grafana_dashboard_common.cpp b/ydb/mvp/meta/support_links/grafana_dashboard_common.cpp
index 2ee68d32324..65da15f2704 100644
--- a/ydb/mvp/meta/support_links/grafana_dashboard_common.cpp
+++ b/ydb/mvp/meta/support_links/grafana_dashboard_common.cpp
@@ -29,25 +29,8 @@ void InsertOrReplaceDashboardVar(TCgiParameters& queryParameters, TStringBuf lab
queryParameters.InsertUnescaped(varName, value);
}
-void ApplyGrafanaDashboardBindingPolicy(
- TCgiParameters& queryParameters,
- const THashMap<TString, TString>& clusterInfo,
- const TCgiParameters& requestQueryParameters,
- const TResolvedParamBindings& paramBindings)
-{
- for (const auto& [name, value] : BuildNonIdentityRequestParamValues(requestQueryParameters)) {
- InsertOrReplaceDashboardVar(queryParameters, name, value);
- }
-
- for (const auto& [label, value] : BuildClusterInfoParamValues(clusterInfo, paramBindings.ClusterInfoMappings)) {
- InsertOrReplaceDashboardVar(queryParameters, label, value);
- }
-
- for (const auto& [label, value] : BuildStaticParamValues(paramBindings.StaticMappings)) {
- InsertOrReplaceDashboardVar(queryParameters, label, value);
- }
-
- for (const auto& [label, value] : BuildRequestParamValues(requestQueryParameters, paramBindings.RequestMappings)) {
+void ApplyGrafanaDashboardBindingPolicy(TCgiParameters& queryParameters, const TVector<std::pair<TString, TString>>& parametersToAdd) {
+ for (const auto& [label, value] : parametersToAdd) {
InsertOrReplaceDashboardVar(queryParameters, label, value);
}
}
@@ -78,7 +61,7 @@ TString BuildGrafanaDashboardUrl(
const TResolvedParamBindings& paramBindings)
{
auto [path, queryParameters] = BuildGrafanaDashboardUrlParts(grafanaEndpoint, url);
- ApplyGrafanaDashboardBindingPolicy(queryParameters, clusterInfo, requestQueryParameters, paramBindings);
+ ApplyGrafanaDashboardBindingPolicy(queryParameters, BuildParametersToAdd(requestQueryParameters, clusterInfo, paramBindings));
return queryParameters.empty()
? path
diff --git a/ydb/mvp/meta/support_links/grafana_logging_source.cpp b/ydb/mvp/meta/support_links/grafana_logging_source.cpp
index 55c07029719..c4cc7c6a05b 100644
--- a/ydb/mvp/meta/support_links/grafana_logging_source.cpp
+++ b/ydb/mvp/meta/support_links/grafana_logging_source.cpp
@@ -91,25 +91,15 @@ bool TryBuildGrafanaLoggingUrl(
}
const TCgiParameters forwardedParameters = BuildForwardedParameters(input.Identity, input.AdditionalRequestParams);
- TVector<std::pair<TString, TString>> bindings = BuildNonIdentityRequestParamValues(forwardedParameters);
- for (auto& binding : BuildRequestParamValues(forwardedParameters, paramBindings.RequestMappings)) {
- bindings.push_back(std::move(binding));
- }
- for (auto& binding : BuildClusterInfoParamValues(input.ClusterInfo, paramBindings.ClusterInfoMappings)) {
- bindings.push_back(std::move(binding));
- }
- for (auto& binding : BuildStaticParamValues(paramBindings.StaticMappings)) {
- bindings.push_back(std::move(binding));
- }
-
- TVector<std::pair<TString, TString>> resolvedBindings;
- resolvedBindings.reserve(bindings.size());
- for (auto& binding : bindings) {
- if (binding.first == "datasource") {
- datasource = std::move(binding.second);
+ auto parametersToAdd = BuildParametersToAdd(forwardedParameters, input.ClusterInfo, paramBindings);
+ TVector<std::pair<TString, TString>> bindings;
+ bindings.reserve(parametersToAdd.size());
+ for (auto& paramValue : parametersToAdd) {
+ if (paramValue.first == "datasource") {
+ datasource = std::move(paramValue.second);
continue;
}
- resolvedBindings.push_back(std::move(binding));
+ bindings.push_back(std::move(paramValue));
}
if (datasource.empty()) {
@@ -120,7 +110,7 @@ bool TryBuildGrafanaLoggingUrl(
TCgiParameters queryParameters;
queryParameters.InsertUnescaped("schemaVersion", "1");
queryParameters.InsertUnescaped("panes", NJson::WriteJson(
- BuildGrafanaLoggingPanesJson(datasource, resolvedBindings),
+ BuildGrafanaLoggingPanesJson(datasource, bindings),
false));
queryParameters.InsertUnescaped("orgId", "1");
diff --git a/ydb/mvp/meta/support_links/param_bindings.cpp b/ydb/mvp/meta/support_links/param_bindings.cpp
index 80f1cc39e83..c566247672a 100644
--- a/ydb/mvp/meta/support_links/param_bindings.cpp
+++ b/ydb/mvp/meta/support_links/param_bindings.cpp
@@ -84,6 +84,28 @@ void ValidateParamsAreUnique(const TResolvedParamBindings& paramBindings, const
}
}
+TVector<std::pair<TString, TString>> BuildParametersToAdd(
+ const TCgiParameters& requestParameters,
+ const THashMap<TString, TString>& clusterInfo,
+ const TResolvedParamBindings& paramBindings)
+{
+ TVector<std::pair<TString, TString>> parametersToAdd = BuildNonIdentityRequestParamValues(requestParameters);
+
+ for (auto& paramValue : BuildRequestParamValues(requestParameters, paramBindings.RequestMappings)) {
+ parametersToAdd.push_back(std::move(paramValue));
+ }
+
+ for (auto& paramValue : BuildClusterInfoParamValues(clusterInfo, paramBindings.ClusterInfoMappings)) {
+ parametersToAdd.push_back(std::move(paramValue));
+ }
+
+ for (auto& paramValue : BuildStaticParamValues(paramBindings.StaticMappings)) {
+ parametersToAdd.push_back(std::move(paramValue));
+ }
+
+ return parametersToAdd;
+}
+
TVector<std::pair<TString, TString>> BuildRequestParamValues(
const TCgiParameters& requestParameters,
const TVector<std::pair<TString, TString>>& requestMappings)
diff --git a/ydb/mvp/meta/support_links/param_bindings.h b/ydb/mvp/meta/support_links/param_bindings.h
index 293c709671f..b354f322dc5 100644
--- a/ydb/mvp/meta/support_links/param_bindings.h
+++ b/ydb/mvp/meta/support_links/param_bindings.h
@@ -16,6 +16,10 @@ struct TResolvedParamBindings {
TResolvedParamBindings ResolveParamBindings(const TSupportLinkEntryConfig& config, const TResolvedParamBindings& defaultParamBindings);
void ValidateParamsAreUnique(const TResolvedParamBindings& paramBindings, const TSupportLinkEntryConfig& config);
+TVector<std::pair<TString, TString>> BuildParametersToAdd(
+ const TCgiParameters& requestParameters,
+ const THashMap<TString, TString>& clusterInfo,
+ const TResolvedParamBindings& paramBindings);
TVector<std::pair<TString, TString>> BuildRequestParamValues(
const TCgiParameters& requestParameters,
const TVector<std::pair<TString, TString>>& requestMappings);
diff --git a/ydb/public/sdk/cpp/src/client/impl/internal/db_driver_state/state.cpp b/ydb/public/sdk/cpp/src/client/impl/internal/db_driver_state/state.cpp
index b44b0b99e8d..1061f739f29 100644
--- a/ydb/public/sdk/cpp/src/client/impl/internal/db_driver_state/state.cpp
+++ b/ydb/public/sdk/cpp/src/client/impl/internal/db_driver_state/state.cpp
@@ -331,7 +331,12 @@ NThreading::TFuture<void> TDbDriverStateTracker::SendNotification(
}
}
if (results.empty()) {
- return NThreading::MakeFuture();
+ // MakeFuture<void>() uses a process-wide singleton that may already be
+ // destroyed when driver shutdown is triggered by another singleton.
+ auto promise = NThreading::NewPromise<void>();
+ auto future = promise.GetFuture();
+ promise.SetValue();
+ return future;
}
return NThreading::WaitExceptionOrAll(results);
}
diff --git a/ydb/public/sdk/cpp/src/client/topic/ut/basic_usage_ut.cpp b/ydb/public/sdk/cpp/src/client/topic/ut/basic_usage_ut.cpp
index b97d12c47e7..f566d66b5ff 100644
--- a/ydb/public/sdk/cpp/src/client/topic/ut/basic_usage_ut.cpp
+++ b/ydb/public/sdk/cpp/src/client/topic/ut/basic_usage_ut.cpp
@@ -28,6 +28,7 @@
#include <library/cpp/string_utils/base64/base64.h>
#include <atomic>
+#include <mutex>
#include <util/digest/murmur.h>
#include <util/stream/zlib.h>
@@ -81,9 +82,19 @@ void ReadMessagesAndAssertOrderedBySeqNo(TTopicClient& client,
ui64 SeqNo;
std::string Data;
};
- std::vector<TMessageInfo> messages;
- messages.reserve(expectedCount);
- NThreading::TPromise<void> donePromise = NThreading::NewPromise<void>();
+ // Handlers may still run after Close(); keep shared ownership so late callbacks are safe.
+ struct TState {
+ std::mutex Lock;
+ std::vector<TMessageInfo> Messages;
+ NThreading::TPromise<void> DonePromise = NThreading::NewPromise<void>();
+
+ std::vector<TMessageInfo> CopyMessages() {
+ std::lock_guard guard(Lock);
+ return Messages;
+ }
+ };
+ auto state = std::make_shared<TState>();
+ state->Messages.reserve(expectedCount);
TTopicReadSettings topicSettings(topicPath);
topicSettings.ReadFromTimestamp(TInstant::Zero());
@@ -93,25 +104,31 @@ void ReadMessagesAndAssertOrderedBySeqNo(TTopicClient& client,
.AutoPartitioningSupport(true)
.AppendTopics(topicSettings);
- readSettings.EventHandlers_.SimpleDataHandlers([&](TReadSessionEvent::TDataReceivedEvent& ev) {
+ readSettings.EventHandlers_.SimpleDataHandlers([state, expectedCount](TReadSessionEvent::TDataReceivedEvent& ev) {
+ std::lock_guard guard(state->Lock);
for (auto& msg : ev.GetMessages()) {
- messages.push_back(TMessageInfo{
+ if (state->Messages.size() >= expectedCount) {
+ break;
+ }
+ state->Messages.push_back(TMessageInfo{
msg.GetPartitionSession()->GetPartitionId(),
TString(msg.GetProducerId()),
msg.GetSeqNo(),
TString(msg.GetData()),
});
}
- if (messages.size() >= expectedCount) {
- donePromise.SetValue();
+ if (state->Messages.size() >= expectedCount) {
+ state->DonePromise.TrySetValue();
}
}, true);
auto readSession = client.CreateReadSession(readSettings);
- UNIT_ASSERT_C(donePromise.GetFuture().Wait(timeout),
- "Expected to read " << expectedCount << " messages within " << timeout << ", got " << messages.size());
+ UNIT_ASSERT_C(state->DonePromise.GetFuture().Wait(timeout),
+ "Expected to read " << expectedCount << " messages within " << timeout);
readSession->Close(TDuration::Seconds(5));
+ const auto messages = state->CopyMessages();
+
UNIT_ASSERT_VALUES_EQUAL_C(messages.size(), expectedCount,
"Read message count mismatch: got " << messages.size() << ", expected " << expectedCount);
diff --git a/ydb/public/sdk/cpp/src/library/grpc/client/grpc_common.cpp b/ydb/public/sdk/cpp/src/library/grpc/client/grpc_common.cpp
index 0a89f90fd68..73ef789a627 100644
--- a/ydb/public/sdk/cpp/src/library/grpc/client/grpc_common.cpp
+++ b/ydb/public/sdk/cpp/src/library/grpc/client/grpc_common.cpp
@@ -8,7 +8,6 @@
#include <openssl/pem.h>
#include <openssl/ssl.h>
#include <openssl/x509.h>
-#include <openssl/x509v3.h>
#include <memory>
#include <string>
@@ -55,36 +54,26 @@ bool ValidateRootCertificates(const std::string& pemRootCerts, std::string& erro
size_t certsParsed = 0;
while (true) {
std::unique_ptr<X509, decltype(&X509_free)> cert(
- PEM_read_bio_X509(rootCertsBio, nullptr, nullptr, nullptr),
+ PEM_read_bio_X509_AUX(rootCertsBio, nullptr, nullptr, nullptr),
&X509_free);
if (!cert) {
- const unsigned long errorCode = ERR_peek_last_error();
- if (errorCode == 0 || ERR_GET_REASON(errorCode) == PEM_R_NO_START_LINE) {
- ERR_clear_error();
- break;
+ if (certsParsed == 0) {
+ errorMessage = "root CA PEM: failed to parse certificate #1: " + DrainOpenSslErrors();
+ return false;
}
- errorMessage = "root CA PEM: " + DrainOpenSslErrors();
- return false;
- }
- std::unique_ptr<BASIC_CONSTRAINTS, decltype(&BASIC_CONSTRAINTS_free)> basicConstraints(
- static_cast<BASIC_CONSTRAINTS*>(X509_get_ext_d2i(cert.get(), NID_basic_constraints, nullptr, nullptr)),
- &BASIC_CONSTRAINTS_free);
- const auto isCaCert = basicConstraints && basicConstraints->ca;
-
- if (!isCaCert) {
+ // gRPC treats a read error as the end of the bundle and uses all
+ // certificates parsed before it.
ERR_clear_error();
- errorMessage = "root CA PEM: certificate is not a CA (BasicConstraints)";
- return false;
+ break;
}
+
+ // Match gRPC trust store semantics: with X509_V_FLAG_PARTIAL_CHAIN an
+ // explicitly trusted non-CA certificate may also be a trust anchor.
ERR_clear_error();
++certsParsed;
}
- if (certsParsed == 0) {
- errorMessage = "root CA PEM: no certificates parsed";
- return false;
- }
return true;
}
diff --git a/ydb/public/sdk/cpp/tests/unit/client/driver/driver_ut.cpp b/ydb/public/sdk/cpp/tests/unit/client/driver/driver_ut.cpp
index d69edfdb76d..149b794b021 100644
--- a/ydb/public/sdk/cpp/tests/unit/client/driver/driver_ut.cpp
+++ b/ydb/public/sdk/cpp/tests/unit/client/driver/driver_ut.cpp
@@ -4,6 +4,7 @@
#include <ydb/public/sdk/cpp/include/ydb-cpp-sdk/client/types/exceptions/exceptions.h>
#include <ydb/public/sdk/cpp/include/ydb-cpp-sdk/type_switcher.h>
#include <ydb/public/sdk/cpp/src/client/impl/observability/constants.h>
+#include <ydb/public/sdk/cpp/src/library/grpc/client/grpc_common.h>
#include <ydb/public/sdk/cpp/tests/common/fake_metric_registry.h>
#include <ydb/public/sdk/cpp/tests/common/fake_trace_provider.h>
@@ -29,6 +30,17 @@ using namespace NYdb::NTable;
namespace {
+ constexpr const char LegacyV1Certificate[] = R"(-----BEGIN CERTIFICATE-----
+MIIBbTCCARMCFBthJdWIg/H6ITeelffnCYoK8fDFMAoGCCqGSM49BAMCMDkxCzAJ
+BgNVBAYTAlJVMQwwCgYDVQQKDANZREIxHDAaBgNVBAMME0xlZ2FjeSBUZXN0IFJv
+b3QgQ0EwHhcNMjYwNzI3MDk0NDU2WhcNMzYwNzI0MDk0NDU2WjA5MQswCQYDVQQG
+EwJSVTEMMAoGA1UECgwDWURCMRwwGgYDVQQDDBNMZWdhY3kgVGVzdCBSb290IENB
+MFkwEwYHKoZIzj0CAQYIKoZIzj0DAQcDQgAE4zlS2ha5hOd20QJEh17FP/mjkzsO
+PmwF7iY9zJ0HILwBjqxJSCGnNMMdT+A2d+Nry6de3WC6RkR72HTe6gffuTAKBggq
+hkjOPQQDAgNIADBFAiEA/0rBKAconmtFcliTZ0i9HzIkQeG+E/zVMiUvlhwpylYC
+IGfPhGBVwOMnr+uhwtpj4PAOIrlOQD/fBsaRtYuBRdg2
+-----END CERTIFICATE-----)";
+
std::string ReadBuildInfo(grpc::ServerContext* context) {
const auto& metadata = context->client_metadata();
const auto it = metadata.find(YDB_SDK_BUILD_INFO_HEADER);
@@ -237,6 +249,35 @@ Y_UNIT_TEST_SUITE(CppGrpcClientSimpleTest) {
UNIT_ASSERT_EQUAL(result.GetStatus(), EStatus::TRANSPORT_UNAVAILABLE);
UNIT_ASSERT_STRING_CONTAINS(result.GetIssues().ToString(), "Client TLS credentials validation failed");
UNIT_ASSERT_STRING_CONTAINS(result.GetIssues().ToString(), "root CA PEM:");
+ UNIT_ASSERT_STRING_CONTAINS(result.GetIssues().ToString(), "failed to parse certificate #1");
+ }
+
+ Y_UNIT_TEST(LegacyV1TrustAnchorPassesValidation) {
+ auto driver = TDriver(
+ TDriverConfig()
+ .SetEndpoint("localhost:100")
+ .UseSecureConnection(LegacyV1Certificate));
+ auto client = NTable::TTableClient(driver);
+
+ auto result = client.CreateSession().GetValueSync();
+ auto issues = result.GetIssues().ToString();
+
+ UNIT_ASSERT_EQUAL(result.GetStatus(), EStatus::TRANSPORT_UNAVAILABLE);
+ UNIT_ASSERT(issues.find("Client TLS credentials validation failed") == std::string::npos);
+ }
+
+ Y_UNIT_TEST(MalformedCertificateAfterValidRootPassesValidation) {
+ const std::string rootBundle = std::string(LegacyV1Certificate) + R"(
+-----BEGIN CERTIFICATE-----
+not-base64
+-----END CERTIFICATE-----)";
+ grpc::SslCredentialsOptions sslOptions{
+ .pem_root_certs = NYdb::TStringType{rootBundle},
+ };
+ std::string validationDetail;
+
+ UNIT_ASSERT(NYdbGrpc::ValidateTlsCredentials(sslOptions, validationDetail));
+ UNIT_ASSERT(validationDetail.empty());
}
Y_UNIT_TEST(EmptyRootCertificateWithoutClientCredentialsKeepsBehavior) {
diff --git a/ydb/services/scheme_secret/resolver.h b/ydb/services/scheme_secret/resolver.h
index 509d1edc6cf..99c55a7ea9d 100644
--- a/ydb/services/scheme_secret/resolver.h
+++ b/ydb/services/scheme_secret/resolver.h
@@ -29,6 +29,8 @@ NThreading::TFuture<NKqp::TEvDescribeSecretsResponse::TDescription> DescribeSecr
TDescribeSecretSettings settings = {}
);
+bool IsSchemeSecret(const TString& secretName);
+
bool UseSchemaSecrets(const NKikimr::TFeatureFlags& flags, const TVector<TString>& secretNames);
bool UseSchemaSecrets(const NKikimr::TFeatureFlags& flags, const TString& secretName);
diff --git a/ydb/services/scheme_secret/service.cpp b/ydb/services/scheme_secret/service.cpp
index c48eb1ca178..8b4903c480b 100644
--- a/ydb/services/scheme_secret/service.cpp
+++ b/ydb/services/scheme_secret/service.cpp
@@ -666,6 +666,17 @@ NThreading::TFuture<NKqp::TEvDescribeSecretsResponse::TDescription> DescribeSecr
return promise.GetFuture();
}
+ if (AppData()->FeatureFlags.GetDisableOldSecrets()) {
+ // Just in case - when we disable old secrets, we'll make sure they are not needed any more
+ promise.SetValue(
+ NKqp::TEvDescribeSecretsResponse::TDescription(
+ Ydb::StatusIds::BAD_REQUEST,
+ { NYql::TIssue("Usage of old secrets is disabled now. Please use new secrets") }
+ )
+ );
+ return promise.GetFuture();
+ }
+
actorSystem->Register(CreateDescribeSecretsActor(userToken ? userToken->GetUserSID() : "", secretNames, promise));
return promise.GetFuture();
}
@@ -674,13 +685,17 @@ IActor* TDescribeSchemaSecretsServiceFactory::CreateService() {
return new TDescribeSchemaSecretsService();
}
+bool IsSchemeSecret(const TString& secretName) {
+ return secretName.StartsWith('/');
+}
+
bool UseSchemaSecrets(const NKikimr::TFeatureFlags& flags, const TVector<TString>& secretNames) {
if (!flags.GetEnableSchemaSecrets()) {
return false;
}
for (const auto& secretName : secretNames) {
- if (!secretName.StartsWith('/')) {
+ if (!IsSchemeSecret(secretName)) {
return false;
}
}
@@ -689,7 +704,7 @@ bool UseSchemaSecrets(const NKikimr::TFeatureFlags& flags, const TVector<TString
}
bool UseSchemaSecrets(const NKikimr::TFeatureFlags& flags, const TString& secretName) {
- return flags.GetEnableSchemaSecrets() && secretName.StartsWith('/');
+ return flags.GetEnableSchemaSecrets() && IsSchemeSecret(secretName);
}
} // namespace NKikimr::NSecret
diff --git a/ydb/services/sqs_topic/create_queue.cpp b/ydb/services/sqs_topic/create_queue.cpp
index 91a92178771..8d88c8e777d 100644
--- a/ydb/services/sqs_topic/create_queue.cpp
+++ b/ydb/services/sqs_topic/create_queue.cpp
@@ -169,9 +169,9 @@ namespace NKikimr::NSqsTopic::V1 {
topicRequest.mutable_supported_codecs()->add_codecs(Ydb::Topic::CODEC_RAW);
topicRequest.set_content_based_deduplication(QueueAttributes.ContentBasedDeduplication.GetOrElse(false));
- if (QueueAttributes.ContentBasedDeduplication.GetOrElse(false)) {
- topicRequest.set_partition_write_speed_messages_per_second(NPQ::CONTENT_BASED_DEDUPLICATION_MESSAGE_LIMIT);
- topicRequest.set_partition_write_burst_messages(NPQ::CONTENT_BASED_DEDUPLICATION_MESSAGE_BURST);
+ if (QueueAttributes.FifoQueue) {
+ topicRequest.set_partition_write_speed_messages_per_second(NPQ::FIFO_PARTITION_WRITE_SPEED_MESSAGES_PER_SECOND);
+ topicRequest.set_partition_write_burst_messages(NPQ::FIFO_PARTITION_WRITE_BURST_MESSAGES);
}
AddConsumerToRequest(topicRequest.add_consumers());
diff --git a/ydb/services/sqs_topic/send_message.cpp b/ydb/services/sqs_topic/send_message.cpp
index 18ec9a1d736..3ab08bd8933 100644
--- a/ydb/services/sqs_topic/send_message.cpp
+++ b/ydb/services/sqs_topic/send_message.cpp
@@ -45,7 +45,9 @@
#include <ydb/services/sqs_topic/statuses.h>
#include <library/cpp/digest/md5/md5.h>
+#include <library/cpp/openssl/crypto/sha.h>
#include <util/generic/guid.h>
+#include <util/string/hex.h>
using namespace NActors;
using namespace NKikimrClient;
@@ -135,7 +137,10 @@ namespace NKikimr::NSqsTopic::V1 {
.Index = item.BatchIndex,
.MessageBody = std::move(item.MessageBody),
.MessageGroupId = toOptional(std::move(item.MessageGroupId)),
- .MessageDeduplicationId = toOptional(std::move(item.MessageDeduplicationId)),
+ // MessageDeduplicationId is supported for FIFO queues only.
+ .MessageDeduplicationId = QueueUrl_->Fifo
+ ? toOptional(std::move(item.MessageDeduplicationId))
+ : std::nullopt,
.Attributes = std::move(item.Attributes),
.Delay = TDuration::Seconds(item.DelaySeconds),
});
@@ -225,7 +230,7 @@ namespace NKikimr::NSqsTopic::V1 {
void Handle(TEvents::TEvWakeup::TPtr& ev) {
switch (static_cast<EWakeupTag>(ev->Get()->Tag)) {
case EWakeupTag::RlAllowed:
- CreateWriter();
+ CreateWriterOrReply();
return;
case EWakeupTag::RlNoResource:
return this->ReplyWithError(MakeError(NSQS::NErrors::THROTTLING_EXCEPTION, "Request was throttled by the rate limiter"));
@@ -274,6 +279,40 @@ namespace NKikimr::NSqsTopic::V1 {
return true;
}
+ void ApplyContentBasedDeduplication(bool enabled) {
+ ContentBasedDeduplication_ = enabled;
+ if (!Fifo_ || !WriterSettings_) {
+ return;
+ }
+
+ TVector<NPQ::NMLP::TWriterSettings::TMessage> validMessages(Reserve(WriterSettings_->Messages.size()));
+ for (auto& message : WriterSettings_->Messages) {
+ if (!message.MessageDeduplicationId.has_value()) {
+ if (enabled) {
+ const auto digest = NOpenSsl::NSha256::Calc(message.MessageBody);
+ message.MessageDeduplicationId = HexEncode(TStringBuf(
+ reinterpret_cast<const char*>(digest.data()),
+ digest.size()));
+ } else {
+ Items[message.Index].ValidationError = MakeError(
+ NSQS::NErrors::MISSING_PARAMETER,
+ "No MessageDeduplicationId parameter.");
+ continue;
+ }
+ }
+ validMessages.push_back(std::move(message));
+ }
+ WriterSettings_->Messages = std::move(validMessages);
+ }
+
+ void CreateWriterOrReply() {
+ if (!WriterSettings_ || WriterSettings_->Messages.empty()) {
+ static_cast<TDerived*>(this)->ReplyAndDie(TlsActivationContext->AsActorContext());
+ return;
+ }
+ CreateWriter();
+ }
+
void HandleCacheNavigateResponse(TEvTxProxySchemeCache::TEvNavigateKeySetResult::TPtr& ev) {
// Second navigate: resolve the rate-limiter path from the database
// serverless attributes, then proceed with the write.
@@ -281,12 +320,17 @@ namespace NKikimr::NSqsTopic::V1 {
if (auto rlContext = this->ExtractRlContext(ev)) {
SetRlContext(*rlContext);
if (IsQuotaRequired()) {
- const ui64 ru = NBilling::CalcRu(CalcRuConsumption(PayloadSize_), NBilling::WRITE_BASE_COST, NBilling::WRITE_COST_PER_BLOCK, Fifo_, false);
+ const ui64 ru = NBilling::CalcRu(
+ CalcRuConsumption(PayloadSize_),
+ NBilling::WRITE_BASE_COST,
+ NBilling::WRITE_COST_PER_BLOCK,
+ Fifo_,
+ ContentBasedDeduplication_);
Y_ABORT_UNLESS(MaybeRequestQuota(ru, EWakeupTag::RlAllowed, TlsActivationContext->AsActorContext()));
return;
}
}
- CreateWriter();
+ CreateWriterOrReply();
return;
}
@@ -308,17 +352,21 @@ namespace NKikimr::NSqsTopic::V1 {
TStringBuilder() << "Failed to describe topic: " << response.Status));
}
+ Y_ABORT_UNLESS(response.PQGroupInfo);
+ Fifo_ = QueueUrl_->Fifo;
+ ApplyContentBasedDeduplication(
+ Fifo_ && response.PQGroupInfo->Description.GetPQTabletConfig().GetContentBasedDeduplication());
+
if (ShouldBeCharged_) {
// Always put in request units metering mode
SetMeteringMode(NKikimrPQ::TPQTabletConfig::METERING_MODE_REQUEST_UNITS);
- Fifo_ = QueueUrl_->Fifo;
// RU-metered topics need the rate-limiter path, which is not carried
// by DoLocalRpc requests. Resolve it from the database attributes
// before writing so the charge in Handle(TEvWriteResponse) can fire.
this->SendRlPathNavigate();
} else {
- CreateWriter();
+ CreateWriterOrReply();
}
}
@@ -369,6 +417,7 @@ namespace NKikimr::NSqsTopic::V1 {
TActorId WriterActor_;
ui64 PayloadSize_{};
bool Fifo_{};
+ bool ContentBasedDeduplication_ = false;
TMaybe<NPQ::NMLP::TWriterSettings> WriterSettings_;
};
diff --git a/ydb/services/sqs_topic/set_queue_attributes.cpp b/ydb/services/sqs_topic/set_queue_attributes.cpp
index faede31ff72..8c5a429093f 100644
--- a/ydb/services/sqs_topic/set_queue_attributes.cpp
+++ b/ydb/services/sqs_topic/set_queue_attributes.cpp
@@ -6,7 +6,6 @@
#include "utils.h"
#include <ydb/core/http_proxy/events.h>
-#include <ydb/core/persqueue/public/constants.h>
#include <ydb/core/protos/grpc_pq_old.pb.h>
#include <ydb/core/ymq/base/limits.h>
#include <ydb/core/ymq/error/error.h>
@@ -188,13 +187,6 @@ namespace NKikimr::NSqsTopic::V1 {
if (NewQueueAttributes.ContentBasedDeduplication.Defined()) {
topicRequest.set_set_content_based_deduplication(*NewQueueAttributes.ContentBasedDeduplication);
- if (*NewQueueAttributes.ContentBasedDeduplication) {
- topicRequest.set_set_partition_write_speed_messages_per_second(NPQ::CONTENT_BASED_DEDUPLICATION_MESSAGE_LIMIT);
- topicRequest.set_set_partition_write_burst_messages(NPQ::CONTENT_BASED_DEDUPLICATION_MESSAGE_BURST);
- } else {
- topicRequest.set_set_partition_write_speed_messages_per_second(NPQ::DEFAULT_PARTITION_WRITE_SPEED_MESSAGES_PER_SECOND);
- topicRequest.set_set_partition_write_burst_messages(NPQ::DEFAULT_PARTITION_WRITE_SPEED_MESSAGES_PER_SECOND);
- }
}
auto* consumer = topicRequest.add_alter_consumers();
diff --git a/ydb/services/sqs_topic/ya.make b/ydb/services/sqs_topic/ya.make
index d25a6329ce9..c3f3fd3f643 100644
--- a/ydb/services/sqs_topic/ya.make
+++ b/ydb/services/sqs_topic/ya.make
@@ -51,6 +51,7 @@ PEERDIR(
ydb/core/ymq/base
ydb/core/ymq/error
library/cpp/json
+ library/cpp/openssl/crypto
)
END()
diff --git a/ydb/services/workload_manager/metadata_subscription/resource_pool_classifier/manager.cpp b/ydb/services/workload_manager/metadata_subscription/resource_pool_classifier/manager.cpp
index 4385b72fba8..602deaa82ca 100644
--- a/ydb/services/workload_manager/metadata_subscription/resource_pool_classifier/manager.cpp
+++ b/ydb/services/workload_manager/metadata_subscription/resource_pool_classifier/manager.cpp
@@ -59,6 +59,9 @@ NMetadata::NModifications::TOperationParsingResult TResourcePoolClassifierManage
if (property == "resource_pool") {
return TConclusionStatus::Fail("Cannot reset required property resource_pool");
}
+ if (property == "rank") {
+ return TConclusionStatus::Fail("Cannot reset property rank");
+ }
} else {
continue;
}
diff --git a/ydb/services/workload_manager/ut/workload_service_ut.cpp b/ydb/services/workload_manager/ut/workload_service_ut.cpp
index d48bc6a6fad..1bb372f97f6 100644
--- a/ydb/services/workload_manager/ut/workload_service_ut.cpp
+++ b/ydb/services/workload_manager/ut/workload_service_ut.cpp
@@ -997,9 +997,7 @@ Y_UNIT_TEST_SUITE(ResourcePoolClassifiersDdl) {
ALTER RESOURCE POOL CLASSIFIER )" << classifierId << R"( RESET (
RANK
);
- )");
-
- WaitForFail(ydb, settings, poolId);
+ )", NYdb::EStatus::GENERIC_ERROR, "Cannot reset property rank");
}
Y_UNIT_TEST(TestExplicitPoolId) {
diff --git a/ydb/tests/fq/generic/utils/settings.py b/ydb/tests/fq/generic/utils/settings.py
index 94603e213d2..18508fd587e 100644
--- a/ydb/tests/fq/generic/utils/settings.py
+++ b/ydb/tests/fq/generic/utils/settings.py
@@ -24,7 +24,7 @@ class Settings:
endpoint: str
hmac_secret_file: str
- token_accessor_mock: TokenAccessorMock
+ token_accessor_mock: Optional[TokenAccessorMock] = None
@dataclass
class MdbMock:
@@ -88,11 +88,12 @@ class Settings:
grpc_host='localhost',
grpc_port=endpoint_determiner.get_port('fq-connector-go', 2130),
),
- token_accessor_mock=cls.TokenAccessorMock(
+ )
+ if "TOKEN_ACCESSOR_MOCK_ENDPOINT" in environ.keys():
+ s.token_accessor_mock = cls.TokenAccessorMock(
endpoint=environ['TOKEN_ACCESSOR_MOCK_ENDPOINT'],
hmac_secret_file=environ['TOKEN_ACCESSOR_HMAC_SECRET_FILE'],
- ),
- )
+ )
if 'MDB_MOCK_ENDPOINT' in environ.keys():
s.mdb_mock = cls.MdbMock(
diff --git a/ydb/tests/fq/streaming/generic/conftest.py b/ydb/tests/fq/streaming/generic/conftest.py
new file mode 100644
index 00000000000..ee576b2f33b
--- /dev/null
+++ b/ydb/tests/fq/streaming/generic/conftest.py
@@ -0,0 +1,34 @@
+import pytest
+import random
+import string
+from typing import Final
+
+from ydb.tests.fq.streaming_common.common import Kikimr
+from ydb.tests.fq.streaming_common.common import get_ydb_config
+from ydb.tests.fq.streaming_common.common import set_test_env
+from ydb.tests.fq.generic.utils.settings import Settings
+
+docker_compose_file_path: Final = "ydb/tests/fq/streaming/generic/docker-compose.yml"
+
+
+def settings() -> Settings:
+ return Settings.from_env(docker_compose_file_path=docker_compose_file_path)
+
+
+def kikimr(request, settings: Settings):
+ set_test_env(request)
+ kikimr = Kikimr(get_ydb_config(request, enable_fq_connector=settings))
+ yield kikimr
+ kikimr.stop()
+
+
+def entity_name(request):
+ suffix = ''.join(random.choices(string.ascii_letters + string.digits, k=8))
+
+ def entity_name_wrapper(name: str) -> str:
+ return f"{name}_{suffix}"
+
+ return entity_name_wrapper
diff --git a/ydb/tests/fq/streaming/generic/connector/Dockerfile b/ydb/tests/fq/streaming/generic/connector/Dockerfile
new file mode 100644
index 00000000000..278ab0c7280
--- /dev/null
+++ b/ydb/tests/fq/streaming/generic/connector/Dockerfile
@@ -0,0 +1,12 @@
+ARG BASE_IMAGE=ghcr.io/ydb-platform/fq-connector-go:v0.11.0-rc.1@sha256:58a1e12de21ef3b403450b176b6220fd545d2802ce16dbc3b05ec4c91e075cdc
+FROM $BASE_IMAGE
+
+COPY ./fq-connector-go.yaml /fq-connector-go.yaml
+COPY ./entrypoint.sh /entrypoint.sh
+
+ENTRYPOINT ["/entrypoint.sh"]
+
+HEALTHCHECK --interval=10s --timeout=300s --retries=30 \
+ CMD cat /var/log/log.txt | grep 'server/service_connector.go:182\tstarting GRPC server' || exit 1
+
+
diff --git a/ydb/tests/fq/streaming/generic/connector/entrypoint.sh b/ydb/tests/fq/streaming/generic/connector/entrypoint.sh
new file mode 100755
index 00000000000..9c93d884edf
--- /dev/null
+++ b/ydb/tests/fq/streaming/generic/connector/entrypoint.sh
@@ -0,0 +1,8 @@
+#!/bin/sh
+
+# echo "$(dig tests-fq-streaming-generic-ydb +short) tests-fq-generic-streaming-ydb" >> /etc/hosts
+sed '/\(127.0.0.1\|::1\).*localhost/s/^/# /' /etc/hosts >/tmp/hosts
+cat /tmp/hosts > /etc/hosts
+cat /etc/hosts
+
+/opt/ydb/bin/fq-connector-go server -c /fq-connector-go.yaml 2>&1 | tee /var/log/log.txt
diff --git a/ydb/tests/fq/streaming/generic/connector/fq-connector-go.yaml b/ydb/tests/fq/streaming/generic/connector/fq-connector-go.yaml
new file mode 100644
index 00000000000..851da84b60b
--- /dev/null
+++ b/ydb/tests/fq/streaming/generic/connector/fq-connector-go.yaml
@@ -0,0 +1,60 @@
+connector_server:
+ endpoint:
+ host: "0.0.0.0"
+ port: 2130
+
+logger:
+ log_level: DEBUG
+ enable_sql_query_logging: true
+
+metrics_server:
+ endpoint:
+ host: "0.0.0.0"
+ port: 8766
+
+pprof_server:
+ endpoint:
+ host: "0.0.0.0"
+ port: 6060
+
+paging:
+ bytes_per_page: 4194304
+ prefetch_queue_capacity: 2
+
+conversion:
+ use_unsafe_converters: true
+
+data_source_default: &data_source_default_var
+ open_connection_timeout: 5s
+ ping_connection_timeout: 5s
+ exponential_backoff:
+ initial_interval: 100ms
+ randomization_factor: 0.5
+ multiplier: 1.5
+ max_interval: 10s
+ max_elapsed_time: 10m
+
+datasources:
+ clickhouse:
+ <<: *data_source_default_var
+
+ greenplum:
+ <<: *data_source_default_var
+
+ ms_sql_server:
+ <<: *data_source_default_var
+
+ mysql:
+ <<: *data_source_default_var
+ result_chan_capacity: 1024
+
+ postgresql:
+ <<: *data_source_default_var
+
+ oracle:
+ <<: *data_source_default_var
+
+ ydb:
+ <<: *data_source_default_var
+ use_underlay_network_for_dedicated_databases: true
+ mode: MODE_QUERY_SERVICE_NATIVE
diff --git a/ydb/tests/fq/streaming/generic/docker-compose.yml b/ydb/tests/fq/streaming/generic/docker-compose.yml
new file mode 100644
index 00000000000..6aab5bcdc86
--- /dev/null
+++ b/ydb/tests/fq/streaming/generic/docker-compose.yml
@@ -0,0 +1,10 @@
+services:
+ fq-connector-go:
+ build: ./connector
+ container_name: tests-fq-streaming-generic-fq-connector-go
+ ports:
+ - 2130
+ extra_hosts:
+ - "localhost:host-gateway"
+
+version: "2.4"
diff --git a/ydb/tests/fq/streaming/generic/test_iam_generic.py b/ydb/tests/fq/streaming/generic/test_iam_generic.py
new file mode 100644
index 00000000000..f92a97f1de1
--- /dev/null
+++ b/ydb/tests/fq/streaming/generic/test_iam_generic.py
@@ -0,0 +1,244 @@
+import json
+import pytest
+import logging
+import time
+import datetime
+from typing import Callable
+
+from ydb.tests.fq.streaming_common.common import Kikimr, StreamingTestBase
+from ydb.tests.tools.datastreams_helpers.control_plane import Endpoint
+import ydb.issues
+
+logger = logging.getLogger(__name__)
+
+# Random but stable service-account-id placeholder; real validation is done server-side.
+FAKE_SERVICE_ACCOUNT_ID = "aje00000000000000000"
+
+# The token that the IAM emulator returns by default.
+USER_TOKEN = "root@builtin"
+
+
+class TestIamAuthGeneric(StreamingTestBase):
+ """Verify that a generic source works with AUTH_METHOD=IAM works end-to-end with connector."""
+
+ def create_iam_secret(self, kikimr: Kikimr, secret_name: str) -> None:
+ kikimr.ydb_client.query(f"""
+ CREATE SECRET `{secret_name}` WITH (value="{USER_TOKEN}");
+ """)
+
+ def set_cloud_id(self, kikimr: Kikimr, cloud_id: str = "test-cloud-id") -> None:
+ """Set cloud_id user attribute on the database root path (/Root).
+
+ DescribeResourceId reads GetAttributes() of the database root, so we must
+ use ESchemeOpAlterUserAttributes rather than ALTER TABLE which only supports
+ table-level settings.
+ """
+ kikimr.cluster.client.add_attr("/", "Root", {"cloud_id": cloud_id}, token="root@builtin")
+
+ def create_iam_source(
+ self,
+ kikimr: Kikimr,
+ source_name: str,
+ secret_path: str,
+ endpoint: Endpoint,
+ shared_reading: bool = False,
+ service_account_id: str = FAKE_SERVICE_ACCOUNT_ID
+ ) -> None:
+ """Create an External Data Source that authenticates via IAM."""
+ kikimr.ydb_client.query(f"""
+ CREATE EXTERNAL DATA SOURCE `{source_name}` WITH (
+ SOURCE_TYPE = "Ydb",
+ LOCATION = "{endpoint.endpoint}",
+ DATABASE_NAME = "{endpoint.database}",
+ USE_TLS = "FALSE",
+ AUTH_METHOD = "IAM",
+ INITIAL_TOKEN_SECRET_PATH = "{secret_path}",
+ SERVICE_ACCOUNT_ID = "{service_account_id}",
+ SHARED_READING="{shared_reading}"
+ );
+ """)
+
+ def create_table(
+ self,
+ kikimr: Kikimr,
+ table_name: str,
+ ) -> None:
+ kikimr.ydb_client.query(f"CREATE TABLE `{table_name}` (a INT, b STRING, c Bool, d Timestamp, e Interval, PRIMARY KEY(a, b))")
+ kikimr.ydb_client.query(f"UPSERT INTO `{table_name}` (a, b, c, d, e) VALUES (1, 'abc', false, Timestamp('2025-08-21T11:22:33.456789Z'), Interval('PT1M'))")
+ kikimr.ydb_client.query(f"UPSERT INTO `{table_name}` (a, b, c, d, e) VALUES (2, 'abcdefghijklmnoprstuvwxyz', true, Timestamp('2025-08-21T22:33:44.567Z'), Interval('PT10S'))")
+
+ @pytest.mark.parametrize(
+ "service_account_id", [FAKE_SERVICE_ACCOUNT_ID, "bad", "bad-token", "bad-skip-1", "bad-token-skip-1"]
+ )
+ def test_generic_read_iam_auth(
+ self,
+ kikimr: Kikimr,
+ service_account_id: str,
+ entity_name: Callable[[str], str],
+ ) -> None:
+ """Creates and populates local table, and read it from query
+ via an IAM-auth external data source."""
+
+ endpoint = self.get_endpoint(kikimr, local_topics=True)
+ source_name = entity_name("iam_source")
+ table_name = entity_name("iam_table")
+
+ # 1. Create the secret and set cloud_id on the database root.
+ secret_name = entity_name("iam_secret")
+ self.create_iam_secret(kikimr, secret_name)
+ self.set_cloud_id(kikimr)
+ time.sleep(1)
+
+ # 2. Create and populate local table
+ self.create_table(kikimr, table_name)
+
+ # 3. Create IAM-auth external data source.
+ self.create_iam_source(kikimr, source_name, secret_name, endpoint, service_account_id=service_account_id)
+
+ tab = f"`{source_name}`.`{table_name}`"
+
+ # 3. Read from IAM-auth source, verify results
+ try:
+ result = kikimr.ydb_client.query(f"SELECT * FROM {tab} ORDER BY a, b")
+ except ydb.issues.Error as ex:
+ assert service_account_id != FAKE_SERVICE_ACCOUNT_ID, ex
+ if service_account_id == 'bad':
+ assert 'Reject bad SA' in ex.message
+ if service_account_id == 'bad-token':
+ assert 'Access denied' in ex.message
+ logger.debug(ex)
+ kikimr.ydb_client.query(f"DROP EXTERNAL DATA SOURCE `{source_name}`;")
+ return
+
+ assert service_account_id == FAKE_SERVICE_ACCOUNT_ID, "Unexpected success"
+
+ assert result
+ rows = result[0].rows
+ logger.debug(rows)
+ # note: Interval type is currently not supported by fq connector and silently ignored
+ expected = [
+ {'a': 1, 'b': b'abc', 'c': False, 'd': datetime.datetime(2025, 8, 21, 11, 22, 33, 456789)},
+ {'a': 2, 'b': b'abcdefghijklmnoprstuvwxyz', 'c': True, 'd': datetime.datetime(2025, 8, 21, 22, 33, 44, 567000)},
+ ]
+ assert rows == expected
+
+ result = kikimr.ydb_client.query(f"SELECT * FROM {tab} WHERE a = 1 ORDER BY a, b")
+ expected = [*filter(lambda x: x["a"] == 1, expected)]
+ assert result
+ rows = result[0].rows
+ logger.debug(rows)
+ assert rows == expected
+
+ kikimr.ydb_client.query(f"DROP TABLE `{table_name}`;")
+ kikimr.ydb_client.query(f"DROP EXTERNAL DATA SOURCE `{source_name}`;")
+
+ @pytest.mark.parametrize(
+ "kikimr",
+ [
+ {"enable_dq_source_stream_lookup_join": True},
+ ],
+ indirect=["kikimr"],
+ )
+ @pytest.mark.parametrize(
+ "service_account_id", [FAKE_SERVICE_ACCOUNT_ID, "bad", "bad-token", "bad-skip-1", "bad-token-skip-1"]
+ )
+ def test_generic_lookup_iam_auth(
+ self,
+ kikimr: Kikimr,
+ service_account_id: str,
+ entity_name: Callable[[str], str],
+ ) -> None:
+ """Creates and populates local table, and read it from query
+ via an IAM-auth external data source."""
+
+ endpoint = self.get_endpoint(kikimr, local_topics=True)
+ source_name = entity_name("iam_source")
+ table_name = entity_name("iam_table")
+ query_name = entity_name("iam_query")
+ ttl = 1
+
+ # 1. Create the secret and set cloud_id on the database root.
+ secret_name = entity_name("iam_secret")
+ self.create_iam_secret(kikimr, secret_name)
+ self.set_cloud_id(kikimr)
+ time.sleep(1)
+ # 2. Create and populate local table
+ self.create_table(kikimr, table_name)
+ # 3. Create topics
+ self.init_topics(source_name, create_output=True, partitions_count=2, endpoint=endpoint)
+ # 4. Create IAM-auth external data source.
+ self.create_iam_source(kikimr, source_name, secret_name, endpoint, service_account_id=service_account_id)
+ tab = f"`{source_name}`.`{table_name}`"
+ inp = f"`{source_name}`.`{self.input_topic}`"
+ out = f"`{source_name}`.`{self.output_topic}`"
+
+ # 5. Join data from topic and lookup using external source with IAM-auth
+ if service_account_id != FAKE_SERVICE_ACCOUNT_ID:
+ try:
+ messages = ['1', '2', '1', '2', '3']
+ self.write_stream(messages, endpoint=endpoint)
+ kikimr.ydb_client.query(Rf"""
+ SELECT Data, a, b, c, CAST(d AS String) AS d
+ FROM {inp} AS i
+ LEFT JOIN /*+streamlookup(TTL {ttl} FullscanLimit 0)*/ ANY {tab} AS db
+ ON CAST(i.Data AS Int) = db.a
+ LIMIT 1
+ """)
+ except ydb.issues.Error as ex:
+ if service_account_id == 'bad':
+ assert 'Reject bad SA' in ex.message
+ if service_account_id == 'bad-token':
+ assert 'Access denied' in ex.message
+ logger.info(ex)
+ logger.info(type(ex))
+ logger.info(ex.args)
+ kikimr.ydb_client.query(f"DROP TABLE `{table_name}`;")
+ kikimr.ydb_client.query(f"DROP EXTERNAL DATA SOURCE `{source_name}`;")
+ return
+ assert False, "Unexpected success"
+
+ kikimr.ydb_client.query(Rf"""
+ CREATE STREAMING QUERY `{query_name}` AS
+ DO BEGIN
+ INSERT INTO {out} SELECT UNWRAP(Yson2::SerializeJson(Yson2::From(TableRow())))
+ FROM (
+ SELECT Data, a, b, c, CAST(d AS String) AS d
+ FROM {inp} AS i
+ LEFT JOIN /*+streamlookup(TTL {ttl} FullscanLimit 0)*/ ANY {tab} AS db
+ ON CAST(i.Data AS Int) = db.a
+ )
+ END DO
+ """)
+
+ path = f"/Root/{query_name}"
+ self.wait_completed_checkpoints(kikimr, path)
+ messages = ['1', '2', '1', '2', '3']
+ self.write_stream(messages, endpoint=endpoint)
+ expected = [
+ '{"Data":"1","a":1,"b":"abc","c":false,"d":"2025-08-21T11:22:33.456789Z"}',
+ '{"Data":"2","a":2,"b":"abcdefghijklmnoprstuvwxyz","c":true,"d":"2025-08-21T22:33:44.567000Z"}',
+ ] * 2 + [
+ '{"Data":"3","a":null,"b":null,"c":null,"d":null}',
+ ]
+ result = self.read_stream(len(expected), topic_path=self.output_topic, endpoint=endpoint)
+ logger.debug([*map(json.loads, sorted(result))])
+ assert sorted(result) == sorted(expected)
+
+ time.sleep(ttl) # at least TTL
+
+ kikimr.ydb_client.query(f"UPDATE `{table_name}` SET c = not c")
+ expected = [
+ '{"Data":"1","a":1,"b":"abc","c":true,"d":"2025-08-21T11:22:33.456789Z"}',
+ '{"Data":"2","a":2,"b":"abcdefghijklmnoprstuvwxyz","c":false,"d":"2025-08-21T22:33:44.567000Z"}',
+ ] * 2 + [
+ '{"Data":"3","a":null,"b":null,"c":null,"d":null}',
+ ]
+ self.write_stream(messages, endpoint=endpoint)
+ result = self.read_stream(len(expected), topic_path=self.output_topic, endpoint=endpoint)
+ logger.debug([*map(json.loads, sorted(result))])
+ assert sorted(result) == sorted(expected)
+ logger.debug(kikimr.ydb_client.query("SELECT * FROM `.sys/streaming_queries`")[0].rows)
+
+ kikimr.ydb_client.query(f"DROP STREAMING QUERY `{query_name}`;")
+ kikimr.ydb_client.query(f"DROP TABLE `{table_name}`;")
+ kikimr.ydb_client.query(f"DROP EXTERNAL DATA SOURCE `{source_name}`;")
diff --git a/ydb/tests/fq/streaming/generic/ya.make b/ydb/tests/fq/streaming/generic/ya.make
new file mode 100644
index 00000000000..8b49475e90e
--- /dev/null
+++ b/ydb/tests/fq/streaming/generic/ya.make
@@ -0,0 +1,51 @@
+PY3TEST()
+
+INCLUDE(${ARCADIA_ROOT}/ydb/tests/tools/fq_runner/ydb_runner_with_datastreams.inc)
+INCLUDE(${ARCADIA_ROOT}/ydb/tests/fq/streaming_common/vm_metadata_emulator/recipe/recipe.inc)
+INCLUDE(${ARCADIA_ROOT}/ydb/tests/fq/streaming_common/iam_grpc_emulator/recipe/recipe.inc)
+
+DATA(arcadia/ydb/library/yql/providers/generic/connector/tests/fq-connector-go)
+ENV(COMPOSE_HTTP_TIMEOUT=1200) # during parallel tests execution there could be huge disk io, which triggers timeouts in docker-compose
+INCLUDE(${ARCADIA_ROOT}/library/recipes/docker_compose/recipe.inc)
+
+TEST_SRCS(
+ test_iam_generic.py
+)
+
+PY_SRCS(
+ conftest.py
+)
+
+IF (SANITIZER_TYPE)
+ SIZE(LARGE)
+ INCLUDE(${ARCADIA_ROOT}/ydb/tests/large.inc)
+ REQUIREMENTS(ram:20)
+ELSE()
+ REQUIREMENTS(ram:12)
+ENDIF()
+
+PEERDIR(
+ ydb/tests/library
+ ydb/tests/library/test_meta
+ ydb/public/sdk/python
+ ydb/public/sdk/python/enable_v3_new_behavior
+ library/recipes/common
+ ydb/tests/olap/common
+ ydb/tests/tools/datastreams_helpers
+ ydb/tests/fq/streaming_common
+ yql/essentials/providers/common/proto
+ ydb/library/yql/providers/generic/connector/tests/utils
+ ydb/tests/fq/generic/utils
+ library/python/testing/recipe
+ library/python/testing/yatest_common
+ library/recipes/common
+ ydb/public/api/protos
+ contrib/python/pytest
+)
+
+DEPENDS(
+ ydb/apps/ydb
+ ydb/tests/tools/pq_read
+)
+
+END()
diff --git a/ydb/tests/fq/streaming/ya.make b/ydb/tests/fq/streaming/ya.make
index 3e2b1c6d5d2..2e75ce30fcc 100644
--- a/ydb/tests/fq/streaming/ya.make
+++ b/ydb/tests/fq/streaming/ya.make
@@ -51,3 +51,8 @@ DEPENDS(
)
END()
+
+RECURSE_FOR_TESTS(
+ streaming_large
+ generic
+)
diff --git a/ydb/tests/fq/streaming_common/common.py b/ydb/tests/fq/streaming_common/common.py
index 0ffa856006d..806d6d4ce11 100644
--- a/ydb/tests/fq/streaming_common/common.py
+++ b/ydb/tests/fq/streaming_common/common.py
@@ -29,7 +29,7 @@ def set_test_env(request):
os.environ["YDB_TEST_ROW_DISPATCHER_REBALANCING_TIMEOUT_MS"] = rebalancing_timeout_ms
-def get_ydb_config(request):
+def get_ydb_config(request, enable_fq_connector=None):
param = getattr(request, "param", {})
enable_watermarks = param.get("enable_watermarks", True)
enable_watermarks_advanced = param.get("enable_watermarks_advanced", True)
@@ -37,6 +37,7 @@ def get_ydb_config(request):
enable_streaming_queries = param.get("enable_streaming_queries", True)
enable_streaming_partition_balancing = param.get("use_partition_balancing", True)
enable_user_attributes_in_topic_query = param.get("enable_user_attributes_in_topic_query", True)
+ enable_dq_source_stream_lookup_join = param.get("enable_dq_source_stream_lookup_join", True)
extra_feature_flags = {
"enable_external_data_sources",
@@ -72,6 +73,7 @@ def get_ydb_config(request):
"enable_streaming_partition_balancing": enable_streaming_partition_balancing,
"enable_compile_cache_warmup": False,
"enable_channel_memory_tracking": False, # Remove after fix https://github.com/ydb-platform/ydb/issues/46891
+ "enable_dq_source_stream_lookup_join": enable_dq_source_stream_lookup_join,
},
replication_config={
"iam_service_control": {
@@ -86,6 +88,17 @@ def get_ydb_config(request):
use_in_memory_pdisks=False,
)
+ if enable_fq_connector:
+ config.yaml_config["query_service_config"]["generic"] = {
+ "connector": {
+ "use_ssl": False,
+ "endpoint": {
+ "host": enable_fq_connector.connector.grpc_host,
+ "port": enable_fq_connector.connector.grpc_port,
+ },
+ },
+ }
+
config.yaml_config["log_config"]["default_level"] = 8
if "auth_config" not in config.yaml_config:
config.yaml_config["auth_config"] = {}
diff --git a/ydb/tests/fq/streaming_common/iam_grpc_emulator/bin/main.py b/ydb/tests/fq/streaming_common/iam_grpc_emulator/bin/main.py
index 24ea510101d..a5dbd08172a 100644
--- a/ydb/tests/fq/streaming_common/iam_grpc_emulator/bin/main.py
+++ b/ydb/tests/fq/streaming_common/iam_grpc_emulator/bin/main.py
@@ -1,6 +1,6 @@
import argparse
import logging
-import random
+# import random
import time
from concurrent import futures
@@ -41,6 +41,8 @@ class IamTokenServicer(iam_token_service_pb2_grpc.IamTokenServiceServicer):
def __init__(self, token, expires_in):
self.token = token
self.expires_in = expires_in
+ self.calls = 0
+ self.token_calls = 0
def _pick_token(self):
# if random.random() < 0.1:
@@ -57,12 +59,45 @@ class IamTokenServicer(iam_token_service_pb2_grpc.IamTokenServiceServicer):
return make_response(self._pick_token(), self.expires_in)
def CreateForService(self, request, context):
- token =self._pick_token()
+ token = self._pick_token()
logger.debug(
"IamTokenService.CreateForService called, service_id=%s microservice_id=%s resource_id=%s target_sa=%s token=%s",
request.service_id, request.microservice_id, request.resource_id, request.target_service_account_id, token
)
- return make_response(token, self.expires_in)
+
+ target_sa = request.target_service_account_id
+ expires_in = self.expires_in
+
+ if target_sa == 'bad':
+ context.set_code(grpc.StatusCode.PERMISSION_DENIED)
+ context.set_details("Reject bad SA")
+ return iam_token_service_pb2.CreateIamTokenResponse()
+
+ if target_sa == 'bad-token':
+ return make_response("badtoken@builtin", expires_in)
+
+ if target_sa.startswith('bad-skip-'):
+ skips = int(target_sa.split('-')[-1])
+ self.calls += 1
+ self.calls %= skips + 1
+ expires_in = 0
+ if self.calls == 0:
+ context.set_code(grpc.StatusCode.PERMISSION_DENIED)
+ context.set_details("Reject bad SA")
+ return iam_token_service_pb2.CreateIamTokenResponse()
+
+ if target_sa.startswith('bad-token-skip-'):
+ skips = int(target_sa.split('-')[-1])
+ self.token_calls += 1
+ self.token_calls %= skips + 1
+ expires_in = 0
+ if self.token_calls == 0:
+ return make_response("badtoken@builtin", expires_in)
+
+ if target_sa == 'bad-token':
+ return make_response("badtoken@builtin", expires_in)
+
+ return make_response(token, expires_in)
def main():
diff --git a/ydb/tests/fq/streaming_common/ya.make b/ydb/tests/fq/streaming_common/ya.make
index e9fb4b5c950..1bcaf647524 100644
--- a/ydb/tests/fq/streaming_common/ya.make
+++ b/ydb/tests/fq/streaming_common/ya.make
@@ -13,3 +13,7 @@ PEERDIR(
END()
+RECURSE(
+ iam_grpc_emulator
+ vm_metadata_emulator
+)
diff --git a/ydb/tests/fq/ya.make b/ydb/tests/fq/ya.make
index cf5aac05187..731a82476f8 100644
--- a/ydb/tests/fq/ya.make
+++ b/ydb/tests/fq/ya.make
@@ -18,5 +18,4 @@ RECURSE_FOR_TESTS(
s3
streaming
yds
- streaming/streaming_large
)
diff --git a/ydb/tests/functional/secrets/test_old_secrets_usage.py b/ydb/tests/functional/secrets/test_old_secrets_usage.py
index e9622c85163..bb7cd703f8b 100644
--- a/ydb/tests/functional/secrets/test_old_secrets_usage.py
+++ b/ydb/tests/functional/secrets/test_old_secrets_usage.py
@@ -1,7 +1,10 @@
# -*- coding: utf-8 -*-
import logging
+import os
+import boto3
import pytest
+import requests
from ydb.tests.library.common.wait_for import wait_for
from ydb.tests.library.harness.kikimr_runner import KiKiMR
@@ -11,6 +14,29 @@ from ydb.tests.oss.ydb_sdk_import import ydb
logger = logging.getLogger(__name__)
DATABASE = "/Root"
+OLD_SECRETS_CREATION_DISABLED_MESSAGE = "Old secrets creation syntax is disabled now. Please use the new one"
+CREATION_WITH_OLD_SECRETS_DISABLED_MESSAGE = (
+ "Old secrets are disabled for creating new objects. Please use new secrets"
+)
+OLD_SECRETS_USAGE_DISABLED_MESSAGE = "Usage of old secrets is disabled now. Please use new secrets"
+
+
+def setup_s3():
+ s3_endpoint = os.getenv("S3_ENDPOINT")
+ s3_access_key = "minio"
+ s3_secret_key = "minio123"
+ s3_bucket = "test_bucket_old_secrets"
+
+ resource = boto3.resource(
+ "s3", endpoint_url=s3_endpoint, aws_access_key_id=s3_access_key, aws_secret_access_key=s3_secret_key
+ )
+
+ bucket = resource.Bucket(s3_bucket)
+ bucket.create()
+ bucket.objects.all().delete()
+ bucket.put_object(Key="file.txt", Body="Hello S3!")
+
+ return s3_endpoint, s3_access_key, s3_secret_key, s3_bucket
class Utils:
@@ -23,14 +49,19 @@ class Utils:
self.driver = None
self.session_pool = None
+ self.query_session_pool = None
self._start_client()
def _start_client(self):
self.driver = ydb.Driver(endpoint=self.cluster.nodes[1].endpoint, database=DATABASE)
self.driver.wait(5, fail_fast=True)
self.session_pool = ydb.SessionPool(self.driver)
+ self.query_session_pool = ydb.QuerySessionPool(self.driver)
def _stop_client(self):
+ if self.query_session_pool is not None:
+ self.query_session_pool.stop()
+ self.query_session_pool = None
if self.session_pool is not None:
self.session_pool.stop()
self.session_pool = None
@@ -42,9 +73,15 @@ class Utils:
self._stop_client()
self.cluster.stop()
- def restart_cluster(self, disable_old_secret_creation, enable_schema_secrets=True):
+ def restart_cluster(
+ self,
+ disable_old_secret_creation=False,
+ disable_old_secrets=False,
+ enable_schema_secrets=True,
+ ):
self._stop_client()
self.config.yaml_config["feature_flags"]["disable_old_secret_creation"] = disable_old_secret_creation
+ self.config.yaml_config["feature_flags"]["disable_old_secrets"] = disable_old_secrets
self.config.yaml_config["feature_flags"]["enable_schema_secrets"] = enable_schema_secrets
self.cluster.update_configurator_and_restart(self.config)
self._start_client_after_restart()
@@ -68,34 +105,167 @@ class Utils:
self.driver = driver_holder["driver"]
self.session_pool = ydb.SessionPool(self.driver)
+ self.query_session_pool = ydb.QuerySessionPool(self.driver)
- def create_old_secret(self, secret_name, value):
+ def execute_scheme(self, query):
with self.session_pool.checkout() as session:
- session.execute_scheme(f"CREATE OBJECT {secret_name} (TYPE SECRET) WITH value='{value}';")
+ session.execute_scheme(query)
+
+ def execute_query(self, query):
+ return self.query_session_pool.execute_with_retries(query)
+
+ def create_old_secret(self, secret_name, value):
+ self.execute_scheme(f"CREATE OBJECT {secret_name} (TYPE SECRET) WITH value='{value}';")
def upsert_old_secret(self, secret_name, value):
- with self.session_pool.checkout() as session:
- session.execute_scheme(f"UPSERT OBJECT {secret_name} (TYPE SECRET) WITH value='{value}';")
+ self.execute_scheme(f"UPSERT OBJECT {secret_name} (TYPE SECRET) WITH value='{value}';")
def alter_old_secret(self, secret_name, value):
- with self.session_pool.checkout() as session:
- session.execute_scheme(f"ALTER OBJECT {secret_name} (TYPE SECRET) SET value='{value}';")
+ self.execute_scheme(f"ALTER OBJECT {secret_name} (TYPE SECRET) SET value='{value}';")
- def create_eds(self, eds_name, secret_name):
- with self.session_pool.checkout() as session:
- query = f"""
- CREATE EXTERNAL DATA SOURCE `{eds_name}` WITH (
- SOURCE_TYPE="ObjectStorage",
- LOCATION="my-bucket",
- AUTH_METHOD="SERVICE_ACCOUNT",
- SERVICE_ACCOUNT_ID="mysa",
- SERVICE_ACCOUNT_SECRET_NAME="{secret_name}"
- );"""
- session.execute_scheme(query)
+ def create_eds(self, eds_name, secret_name, schema_secret=False, create_or_replace=False):
+ secret_setting = "SERVICE_ACCOUNT_SECRET_PATH" if schema_secret else "SERVICE_ACCOUNT_SECRET_NAME"
+ create_type = "CREATE OR REPLACE" if create_or_replace else "CREATE"
+ query = f"""
+ {create_type} EXTERNAL DATA SOURCE `{eds_name}` WITH (
+ SOURCE_TYPE="ObjectStorage",
+ LOCATION="my-bucket",
+ AUTH_METHOD="SERVICE_ACCOUNT",
+ SERVICE_ACCOUNT_ID="mysa",
+ {secret_setting}="{secret_name}"
+ );"""
+ self.execute_scheme(query)
+
+ def create_eds_aws(self, eds_name, access_key_secret, secret_key_secret, s3_location):
+ query = f"""
+ CREATE EXTERNAL DATA SOURCE `{eds_name}` WITH (
+ SOURCE_TYPE="ObjectStorage",
+ LOCATION="{s3_location}",
+ AUTH_METHOD="AWS",
+ AWS_ACCESS_KEY_ID_SECRET_NAME="{access_key_secret}",
+ AWS_SECRET_ACCESS_KEY_SECRET_NAME="{secret_key_secret}",
+ AWS_REGION="ru-central-1"
+ );"""
+ self.execute_scheme(query)
+
+ def alter_password_secret(self, object_type, object_name, secret_name, schema_secret=False):
+ secret_setting = "PASSWORD_SECRET_PATH" if schema_secret else "PASSWORD_SECRET_NAME"
+ self.execute_scheme(f"""
+ ALTER {object_type} `{object_name}` SET (STATE = "Paused");
+ ALTER {object_type} `{object_name}` SET ({secret_setting} = "{secret_name}");
+ ALTER {object_type} `{object_name}` SET (STATE = "StandBy");
+ """)
+
+ def read_from_eds(self, eds_name):
+ return self.execute_query(f"""
+ SELECT * FROM `{eds_name}`.`file.txt` WITH (
+ FORMAT = "raw",
+ SCHEMA = ( Data String )
+ );""")
def create_schema_secret(self, secret_name, value):
- with self.session_pool.checkout() as session:
- session.execute_scheme(f"CREATE SECRET {secret_name} WITH (value='{value}');")
+ self.execute_scheme(f"CREATE SECRET `{secret_name}` WITH (value='{value}');")
+
+ def create_table(self, table_name):
+ self.execute_scheme(f"CREATE TABLE `{table_name}` (Key Uint64, PRIMARY KEY (Key));")
+
+ def create_transfer_table(self, table_name):
+ self.execute_scheme(f"""
+ CREATE TABLE `{table_name}` (
+ partition Uint32 NOT NULL,
+ offset Uint64 NOT NULL,
+ message Utf8,
+ PRIMARY KEY (partition, offset)
+ );""")
+
+ def create_topic(self, topic_name):
+ self.execute_scheme(f"CREATE TOPIC `{topic_name}`;")
+
+ def create_async_replication(self, replication_name, table_name, replica_name, secret_name, schema_secret=False):
+ connection_string = f"grpc://{self.cluster.nodes[1].host}:{self.cluster.nodes[1].port}/?database={DATABASE}"
+ secret_setting = "PASSWORD_SECRET_PATH" if schema_secret else "PASSWORD_SECRET_NAME"
+ query = f"""
+ CREATE ASYNC REPLICATION `{replication_name}` FOR `{table_name}` AS `{replica_name}` WITH (
+ CONNECTION_STRING="{connection_string}",
+ USER = "root",
+ {secret_setting} = "{secret_name}"
+ );"""
+ self.execute_scheme(query)
+
+ def create_transfer(self, transfer_name, topic_name, table_name, secret_name, schema_secret=False):
+ connection_string = f"grpc://{self.cluster.nodes[1].host}:{self.cluster.nodes[1].port}/?database={DATABASE}"
+ secret_setting = "PASSWORD_SECRET_PATH" if schema_secret else "PASSWORD_SECRET_NAME"
+ query = f"""
+ $l = ($x) -> {{
+ return [
+ <|
+ partition:CAST($x._partition AS Uint32),
+ offset:CAST($x._offset AS Uint64),
+ message:CAST($x._data AS Utf8)
+ |>
+ ];
+ }};
+
+ CREATE TRANSFER `{transfer_name}`
+ FROM `{topic_name}` TO `{table_name}` USING $l
+ WITH (
+ CONNECTION_STRING="{connection_string}",
+ FLUSH_INTERVAL = Interval('PT1S'),
+ BATCH_SIZE_BYTES = 10,
+ USER = "root",
+ {secret_setting} = "{secret_name}"
+ );"""
+ self.execute_scheme(query)
+
+ def mon_endpoint(self):
+ return f"http://{self.cluster.nodes[1].host}:{self.cluster.nodes[1].mon_port}"
+
+ def viewer_describe(self, path):
+ response = requests.get(
+ f"{self.mon_endpoint()}/viewer/json/describe",
+ params={"database": DATABASE, "path": path},
+ timeout=10,
+ )
+ response.raise_for_status()
+ return response.json()
+
+ def viewer_describe_replication(self, path):
+ response = requests.get(
+ f"{self.mon_endpoint()}/viewer/json/describe_replication",
+ params={"database": DATABASE, "path": path},
+ timeout=10,
+ )
+ response.raise_for_status()
+ return response.json()
+
+ def viewer_describe_transfer(self, path):
+ response = requests.get(
+ f"{self.mon_endpoint()}/viewer/json/describe_transfer",
+ params={"database": DATABASE, "path": path},
+ timeout=10,
+ )
+ response.raise_for_status()
+ return response.json()
+
+ def assert_path_ready(self, path):
+ description = self.viewer_describe(path)
+ assert description["Status"] == "StatusSuccess", description
+ assert description["PathDescription"]["Self"]["CreateFinished"] is True, description
+ return description
+
+ def wait_object_error_state(self, describe_fn, path, expected_issue_substr):
+ def ready():
+ description = describe_fn(path)
+ error = description.get("error")
+ if not error:
+ return False
+ return expected_issue_substr in str(error)
+
+ if not wait_for(ready, timeout_seconds=60, step_seconds=1):
+ description = describe_fn(path)
+ raise AssertionError(
+ f"Object {path} didn't enter error state with {expected_issue_substr!r}, got: {description}"
+ )
@pytest.fixture
@@ -105,30 +275,272 @@ def old_secrets_utils():
utils.stop()
-def test_create_eds_with_old_secret_after_disabling_old_secret_creation(old_secrets_utils):
- # can create old secrets by default
+def test_old_secret_creation_is_disabled(old_secrets_utils):
+ # create old secret
old_secrets_utils.create_old_secret("OldSecret", value="")
- # can use old secrets by default
- old_secrets_utils.create_eds("eds-before-restart", "OldSecret")
-
+ # restart cluster with disabled old secret creation
old_secrets_utils.restart_cluster(disable_old_secret_creation=True)
- # can create schema secrets with old secrets disabled
- old_secrets_utils.create_schema_secret("NewSecret", value="")
-
- # can use old secrets with old secrets disabled
- old_secrets_utils.create_eds("eds-after-restart", "OldSecret")
-
- # can alter old secrets with old secrets disabled
+ # check that old secret can be altered
old_secrets_utils.alter_old_secret("OldSecret", "NewValue")
- # can not create old secrets with old secrets disabled
+ # check that old secret can't be created
with pytest.raises(Exception) as exc_info:
old_secrets_utils.create_old_secret("NewOldSecret", value="")
- assert "Old secrets creation syntax is disabled now. Please use the new one" in str(exc_info.value)
+ assert OLD_SECRETS_CREATION_DISABLED_MESSAGE in str(exc_info.value)
- # can not upsert old secrets with old secrets disabled
+ # check that old secret can't be upserted
with pytest.raises(Exception) as exc_info:
old_secrets_utils.upsert_old_secret("NewOldSecretUpsert", value="")
- assert "Old secrets creation syntax is disabled now. Please use the new one" in str(exc_info.value)
+ assert OLD_SECRETS_CREATION_DISABLED_MESSAGE in str(exc_info.value)
+
+
+def test_existing_objects_are_ok_if_old_secret_creation_is_disabled(old_secrets_utils):
+ # create all types of objects with old secrets
+ # secret
+ old_secrets_utils.create_old_secret("OldSecret", value="")
+
+ # eds
+ old_secrets_utils.create_eds("eds-old-secret", "OldSecret")
+
+ # replication
+ old_secrets_utils.create_table("repl_src")
+ old_secrets_utils.create_async_replication("replication-old-secret", "repl_src", "repl_dst", "OldSecret")
+
+ # transfer
+ old_secrets_utils.create_topic("transfer_topic")
+ old_secrets_utils.create_transfer_table("transfer_dst")
+ old_secrets_utils.create_transfer("transfer-old-secret", "transfer_topic", "transfer_dst", "OldSecret")
+
+ # restart cluster with disabled old secret creation
+ old_secrets_utils.restart_cluster(disable_old_secret_creation=True)
+
+ # check that old objects are still OK
+ old_secrets_utils.assert_path_ready(f"{DATABASE}/eds-old-secret")
+ old_secrets_utils.assert_path_ready(f"{DATABASE}/replication-old-secret")
+ old_secrets_utils.assert_path_ready(f"{DATABASE}/transfer-old-secret")
+
+ # check that existing objects can't be changed to use old secrets
+ with pytest.raises(Exception) as exc_info:
+ old_secrets_utils.create_eds(
+ "eds-old-secret",
+ "OldSecret",
+ create_or_replace=True,
+ )
+ assert CREATION_WITH_OLD_SECRETS_DISABLED_MESSAGE in str(exc_info.value)
+
+ with pytest.raises(Exception) as exc_info:
+ old_secrets_utils.alter_password_secret(
+ "ASYNC REPLICATION",
+ "replication-old-secret",
+ "OldSecret",
+ )
+ assert CREATION_WITH_OLD_SECRETS_DISABLED_MESSAGE in str(exc_info.value)
+
+ with pytest.raises(Exception) as exc_info:
+ old_secrets_utils.alter_password_secret(
+ "TRANSFER",
+ "transfer-old-secret",
+ "OldSecret",
+ )
+ assert CREATION_WITH_OLD_SECRETS_DISABLED_MESSAGE in str(exc_info.value)
+
+
+def test_existing_objects_cannot_use_old_secrets_if_old_secret_creation_is_disabled(old_secrets_utils):
+ # create all types of objects with old secrets
+ # secret
+ old_secrets_utils.create_old_secret("OldSecret", value="")
+
+ # eds
+ old_secrets_utils.create_eds("eds-old-secret", "OldSecret")
+
+ # replication
+ old_secrets_utils.create_table("repl_src")
+ old_secrets_utils.create_async_replication("replication-old-secret", "repl_src", "repl_dst", "OldSecret")
+
+ # transfer
+ old_secrets_utils.create_topic("transfer_topic")
+ old_secrets_utils.create_transfer_table("transfer_dst")
+ old_secrets_utils.create_transfer("transfer-old-secret", "transfer_topic", "transfer_dst", "OldSecret")
+
+ # restart cluster with disabled old secret creation
+ old_secrets_utils.restart_cluster(disable_old_secret_creation=True)
+
+ # check that existing objects can't be changed to use old secrets
+ with pytest.raises(Exception) as exc_info:
+ old_secrets_utils.create_eds(
+ "eds-old-secret",
+ "OldSecret",
+ create_or_replace=True,
+ )
+ assert CREATION_WITH_OLD_SECRETS_DISABLED_MESSAGE in str(exc_info.value)
+
+ with pytest.raises(Exception) as exc_info:
+ old_secrets_utils.alter_password_secret(
+ "ASYNC REPLICATION",
+ "replication-old-secret",
+ "OldSecret",
+ )
+ assert CREATION_WITH_OLD_SECRETS_DISABLED_MESSAGE in str(exc_info.value)
+
+ with pytest.raises(Exception) as exc_info:
+ old_secrets_utils.alter_password_secret(
+ "TRANSFER",
+ "transfer-old-secret",
+ "OldSecret",
+ )
+ assert CREATION_WITH_OLD_SECRETS_DISABLED_MESSAGE in str(exc_info.value)
+
+
+def test_existing_objects_can_migrate_to_new_secrets_if_old_secret_creation_is_disabled(old_secrets_utils):
+ # create all types of objects with old secrets
+ old_secrets_utils.create_old_secret("OldSecret", value="")
+ old_secrets_utils.create_eds("eds-old-secret", "OldSecret")
+ old_secrets_utils.create_table("repl_src")
+ old_secrets_utils.create_async_replication("replication-old-secret", "repl_src", "repl_dst", "OldSecret")
+ old_secrets_utils.create_topic("transfer_topic")
+ old_secrets_utils.create_transfer_table("transfer_dst")
+ old_secrets_utils.create_transfer("transfer-old-secret", "transfer_topic", "transfer_dst", "OldSecret")
+
+ # restart cluster with disabled old secret creation
+ old_secrets_utils.restart_cluster(disable_old_secret_creation=True)
+
+ # non-secret ALTER (e.g. STATE) must still work for objects that use old secrets
+ old_secrets_utils.execute_scheme("""
+ ALTER ASYNC REPLICATION `replication-old-secret` SET (STATE = "Paused");
+ ALTER ASYNC REPLICATION `replication-old-secret` SET (STATE = "StandBy");
+ """)
+ old_secrets_utils.execute_scheme("""
+ ALTER TRANSFER `transfer-old-secret` SET (STATE = "Paused");
+ ALTER TRANSFER `transfer-old-secret` SET (STATE = "StandBy");
+ """)
+
+ # migrate existing objects to new secrets
+ new_secret = "NewSecret"
+ old_secrets_utils.create_schema_secret(new_secret, value="")
+
+ old_secrets_utils.create_eds(
+ "eds-old-secret",
+ new_secret,
+ schema_secret=True,
+ create_or_replace=True,
+ )
+ old_secrets_utils.alter_password_secret(
+ "ASYNC REPLICATION",
+ "replication-old-secret",
+ new_secret,
+ schema_secret=True,
+ )
+ old_secrets_utils.alter_password_secret(
+ "TRANSFER",
+ "transfer-old-secret",
+ new_secret,
+ schema_secret=True,
+ )
+
+ old_secrets_utils.assert_path_ready(f"{DATABASE}/eds-old-secret")
+ old_secrets_utils.assert_path_ready(f"{DATABASE}/replication-old-secret")
+ old_secrets_utils.assert_path_ready(f"{DATABASE}/transfer-old-secret")
+
+
+def test_disable_old_secret_creation(old_secrets_utils):
+ old_secrets_utils.create_old_secret("OldSecret", value="")
+
+ # restart cluster with disabled old secret creation
+ old_secrets_utils.restart_cluster(disable_old_secret_creation=True)
+
+ # check that old secrets can't be used for eds
+ with pytest.raises(Exception) as exc_info:
+ old_secrets_utils.create_eds("eds-with-old-secret", "OldSecret")
+ assert CREATION_WITH_OLD_SECRETS_DISABLED_MESSAGE in str(exc_info.value)
+
+ # check that old secrets can't be used for replication
+ with pytest.raises(Exception) as exc_info:
+ old_secrets_utils.create_async_replication("repl-with-old-secret", "repl_src", "repl_dst_old", "OldSecret")
+ assert CREATION_WITH_OLD_SECRETS_DISABLED_MESSAGE in str(exc_info.value)
+
+ # check that old secrets can't be used for transfer
+ with pytest.raises(Exception) as exc_info:
+ old_secrets_utils.create_transfer("transfer-with-old-secret", "transfer_topic", "transfer_dst", "OldSecret")
+ assert CREATION_WITH_OLD_SECRETS_DISABLED_MESSAGE in str(exc_info.value)
+
+ # check that new secrets can be still used
+ old_secrets_utils.create_schema_secret("NewSecret", value="")
+
+ # create eds with new secret
+ old_secrets_utils.create_eds(
+ "eds-with-new-secret",
+ "NewSecret",
+ schema_secret=True,
+ )
+
+ # create replication with new secret
+ old_secrets_utils.create_table("repl_src_new")
+ old_secrets_utils.create_async_replication(
+ "repl-with-new-secret", "repl_src_new", "repl_dst_new", "NewSecret", schema_secret=True
+ )
+
+ # create transfer with new secret
+ old_secrets_utils.create_topic("transfer_topic_new")
+ old_secrets_utils.create_transfer_table("transfer_dst_new")
+ old_secrets_utils.create_transfer(
+ "transfer-with-new-secret",
+ "transfer_topic_new",
+ "transfer_dst_new",
+ "NewSecret",
+ schema_secret=True,
+ )
+
+
+def test_disable_old_secrets_dont_crash_existing_objects(old_secrets_utils):
+ s3_endpoint, s3_access_key, s3_secret_key, s3_bucket = setup_s3()
+ s3_location = f"{s3_endpoint}/{s3_bucket}"
+
+ old_secrets_utils.create_old_secret("s3_access_key", value=s3_access_key)
+ old_secrets_utils.create_old_secret("s3_secret_key", value=s3_secret_key)
+ old_secrets_utils.create_old_secret("password", value="")
+
+ # create eds with old secrets
+ eds_name = "eds-old-secret"
+ old_secrets_utils.create_eds_aws(eds_name, "s3_access_key", "s3_secret_key", s3_location)
+ old_secrets_utils.assert_path_ready(f"{DATABASE}/{eds_name}")
+
+ # create replication with old secret
+ old_secrets_utils.create_table("repl_src")
+ old_secrets_utils.create_async_replication(
+ "replication-old-secret",
+ "repl_src",
+ "repl_dst",
+ "password",
+ )
+ old_secrets_utils.assert_path_ready(f"{DATABASE}/replication-old-secret")
+
+ # create transfer with old secret
+ old_secrets_utils.create_topic("transfer_topic")
+ old_secrets_utils.create_transfer_table("transfer_dst")
+ old_secrets_utils.create_transfer(
+ "transfer-old-secret",
+ "transfer_topic",
+ "transfer_dst",
+ "password",
+ )
+ old_secrets_utils.assert_path_ready(f"{DATABASE}/transfer-old-secret")
+
+ old_secrets_utils.restart_cluster(disable_old_secrets=True)
+
+ with pytest.raises(Exception) as exc_info:
+ # After restart eds resolves secret only while reading
+ old_secrets_utils.read_from_eds(eds_name)
+ assert OLD_SECRETS_USAGE_DISABLED_MESSAGE in str(exc_info.value)
+
+ old_secrets_utils.wait_object_error_state(
+ old_secrets_utils.viewer_describe_replication,
+ f"{DATABASE}/replication-old-secret",
+ OLD_SECRETS_USAGE_DISABLED_MESSAGE,
+ )
+ old_secrets_utils.wait_object_error_state(
+ old_secrets_utils.viewer_describe_transfer,
+ f"{DATABASE}/transfer-old-secret",
+ OLD_SECRETS_USAGE_DISABLED_MESSAGE,
+ )
diff --git a/ydb/tests/functional/secrets/ya.make b/ydb/tests/functional/secrets/ya.make
index 9facf4e8846..a7335cae6c9 100644
--- a/ydb/tests/functional/secrets/ya.make
+++ b/ydb/tests/functional/secrets/ya.make
@@ -25,6 +25,7 @@ DEPENDS(
PEERDIR(
contrib/python/boto3
+ contrib/python/requests
ydb/tests/functional/secrets/lib
ydb/tests/library
ydb/tests/library/fixtures
diff --git a/ydb/tests/functional/sqs/messaging/test_generic_messaging.py b/ydb/tests/functional/sqs/messaging/test_generic_messaging.py
index dc5af7f17c3..72cd1802f0b 100644
--- a/ydb/tests/functional/sqs/messaging/test_generic_messaging.py
+++ b/ydb/tests/functional/sqs/messaging/test_generic_messaging.py
@@ -250,6 +250,32 @@ class SqsGenericMessagingTest(KikimrSqsTestBase):
self._wait_for_counter_value(receive_counter_labels, 1, default_value=0)
@pytest.mark.parametrize(**TABLES_FORMAT_PARAMS)
+ def test_standard_queue_ignores_message_deduplication_id(self, tables_format):
+ self._init_with_params(is_fifo=False, tables_format=tables_format)
+ created_queue_url = self._create_queue_and_assert(self.queue_name, is_fifo=False)
+
+ body = 'standard body with ignored deduplication id'
+ message_id = self._send_message_and_assert(
+ created_queue_url, body, seq_no='deduplication-id-1'
+ )
+ read_message_result = self._read_while_not_empty(created_queue_url, 1)
+
+ assert_that(
+ read_message_result, ReadResponseMatcher().with_message_ids([message_id])
+ )
+
+ # MessageDeduplicationId is FIFO-only: ignored for standard queues and not returned on receive.
+ attributes = read_message_result[0].get('Attribute')
+ if attributes is not None:
+ attributes_by_name = {}
+ for a in attributes if isinstance(attributes, list) else [attributes]:
+ attributes_by_name[a['Name']] = a
+ assert_that(
+ attributes_by_name,
+ not_(has_item('MessageDeduplicationId'))
+ )
+
+ @pytest.mark.parametrize(**TABLES_FORMAT_PARAMS)
def test_validates_message_attributes(self, tables_format):
self._init_with_params(tables_format=tables_format)
created_queue_url = self._create_queue_and_assert(self.queue_name)
diff --git a/ydb/tests/functional/sqs_topic/test_change_message_visibility.py b/ydb/tests/functional/sqs_topic/test_change_message_visibility.py
index c61c7626ec9..0cb7004f6b2 100644
--- a/ydb/tests/functional/sqs_topic/test_change_message_visibility.py
+++ b/ydb/tests/functional/sqs_topic/test_change_message_visibility.py
@@ -79,6 +79,7 @@ class TestSqsTopicChangeMessageVisibility(KikimrSqsTopicTestBase):
QueueUrl=self._queue_url,
MessageBody=message_body,
MessageGroupId='message-group-1',
+ MessageDeduplicationId='deduplication-id-1',
)
response = self._boto_client.receive_message(
diff --git a/ydb/tests/functional/sqs_topic/test_change_message_visibility_batch.py b/ydb/tests/functional/sqs_topic/test_change_message_visibility_batch.py
index a1eaf0d55f0..c04206af28c 100644
--- a/ydb/tests/functional/sqs_topic/test_change_message_visibility_batch.py
+++ b/ydb/tests/functional/sqs_topic/test_change_message_visibility_batch.py
@@ -89,6 +89,7 @@ class TestSqsTopicChangeMessageVisibilityBatch(KikimrSqsTopicTestBase):
QueueUrl=self._queue_url,
MessageBody=message_body,
MessageGroupId='message-group-{}'.format(index),
+ MessageDeduplicationId='deduplication-id-{}'.format(index),
)
response = self._boto_client.receive_message(
diff --git a/ydb/tests/functional/sqs_topic/test_delete_message.py b/ydb/tests/functional/sqs_topic/test_delete_message.py
index 2097b1686f2..3e8372519ab 100644
--- a/ydb/tests/functional/sqs_topic/test_delete_message.py
+++ b/ydb/tests/functional/sqs_topic/test_delete_message.py
@@ -164,6 +164,7 @@ class TestSqsTopicDeleteMessage(KikimrSqsTopicTestBase):
QueueUrl=self._queue_url,
MessageBody=message_body,
MessageGroupId='message-group-1',
+ MessageDeduplicationId='deduplication-id-1',
)
assert_that(
@@ -207,6 +208,7 @@ class TestSqsTopicDeleteMessage(KikimrSqsTopicTestBase):
QueueUrl=self._queue_url,
MessageBody=message_body,
MessageGroupId='message-group-{}'.format(index),
+ MessageDeduplicationId='deduplication-id-{}'.format(index),
)
assert_that(
diff --git a/ydb/tests/functional/sqs_topic/test_purge_queue.py b/ydb/tests/functional/sqs_topic/test_purge_queue.py
index f44ed42c6b0..5f1f4099b62 100644
--- a/ydb/tests/functional/sqs_topic/test_purge_queue.py
+++ b/ydb/tests/functional/sqs_topic/test_purge_queue.py
@@ -42,6 +42,7 @@ class TestSqsTopicPurgeQueue(KikimrSqsTopicTestBase):
QueueUrl=self._queue_url,
MessageBody='hello from fifo sqs',
MessageGroupId='message-group-1',
+ MessageDeduplicationId='deduplication-id-1',
)
assert_that(
diff --git a/ydb/tests/functional/sqs_topic/test_receive_message.py b/ydb/tests/functional/sqs_topic/test_receive_message.py
index 8cecdab6793..ac8692b8733 100644
--- a/ydb/tests/functional/sqs_topic/test_receive_message.py
+++ b/ydb/tests/functional/sqs_topic/test_receive_message.py
@@ -1,7 +1,7 @@
#!/usr/bin/env python
# -*- coding: utf-8 -*-
-from hamcrest import assert_that, equal_to, has_length, not_none, not_
+from hamcrest import assert_that, equal_to, has_key, has_length, is_not, not_none, not_
from ydb.tests.library.sqs_topic.test_base import KikimrSqsTopicTestBase
@@ -36,6 +36,7 @@ class TestSqsTopicReceiveMessage(KikimrSqsTopicTestBase):
QueueUrl=self._queue_url,
MessageBody=message_body,
MessageGroupId='message-group-1',
+ MessageDeduplicationId='deduplication-id-1',
)
response = self._boto_client.receive_message(
@@ -102,6 +103,33 @@ class TestSqsTopicReceiveMessage(KikimrSqsTopicTestBase):
assert_that(messages[0]['Body'], equal_to(message_body))
assert_that(messages[0]['Attributes']['MessageGroupId'], equal_to(message_group_id))
+ def test_receive_message_standard_ignores_message_deduplication_id(self):
+ queue_name = self._make_queue_name('receive_message_standard_ignores_message_deduplication_id')
+ self._queue_url = self._boto_client.create_queue(QueueName=queue_name)['QueueUrl']
+
+ message_body = 'hello from std sqs'
+ deduplication_id = 'deduplication-id-1'
+ self._boto_client.send_message(
+ QueueUrl=self._queue_url,
+ MessageBody=message_body,
+ MessageDeduplicationId=deduplication_id,
+ )
+
+ response = self._boto_client.receive_message(
+ QueueUrl=self._queue_url,
+ WaitTimeSeconds=20,
+ MaxNumberOfMessages=1,
+ AttributeNames=['All'],
+ )
+
+ messages = response.get('Messages')
+ assert_that(messages, not_none())
+ assert_that(messages, has_length(1))
+ assert_that(messages[0]['Body'], equal_to(message_body))
+ # MessageDeduplicationId is FIFO-only: ignored for standard queues and not returned on receive.
+ attributes = messages[0].get('Attributes', {})
+ assert_that(attributes, is_not(has_key('MessageDeduplicationId')))
+
def test_receive_message_fifo_with_message_deduplication_id(self):
self._create_fifo_queue('receive_message_fifo_with_message_deduplication_id')
diff --git a/ydb/tests/functional/sqs_topic/test_send_message.py b/ydb/tests/functional/sqs_topic/test_send_message.py
index 1500fba1b58..463ade7ff85 100644
--- a/ydb/tests/functional/sqs_topic/test_send_message.py
+++ b/ydb/tests/functional/sqs_topic/test_send_message.py
@@ -1,7 +1,9 @@
#!/usr/bin/env python
# -*- coding: utf-8 -*-
-from hamcrest import assert_that, equal_to, not_none
+import botocore
+
+from hamcrest import assert_that, equal_to, not_, not_none, raises
from ydb.tests.library.sqs_topic.test_base import KikimrSqsTopicTestBase
@@ -30,6 +32,7 @@ class TestSqsTopicSendMessage(KikimrSqsTopicTestBase):
QueueUrl=self._queue_url,
MessageBody=message_body,
MessageGroupId='message-group-1',
+ MessageDeduplicationId='deduplication-id-1',
)
assert_that(response['MessageId'], not_none())
@@ -85,3 +88,76 @@ class TestSqsTopicSendMessage(KikimrSqsTopicTestBase):
)
assert_that(duplicate_response['MessageId'], equal_to(response['MessageId']))
+
+ def test_send_message_fifo_with_content_based_deduplication(self):
+ queue_name = self._make_fifo_queue_name('send_message_fifo_with_content_based_deduplication')
+ self._queue_url = self._boto_client.create_queue(
+ QueueName=queue_name,
+ Attributes={
+ 'FifoQueue': 'true',
+ 'ContentBasedDeduplication': 'true',
+ },
+ )['QueueUrl']
+
+ message_body = 'hello from fifo sqs'
+ response = self._boto_client.send_message(
+ QueueUrl=self._queue_url,
+ MessageBody=message_body,
+ MessageGroupId='message-group-1',
+ )
+
+ assert_that(response['MessageId'], not_none())
+ assert_that(response['SequenceNumber'], not_none())
+
+ # Same body without MessageDeduplicationId: SHA-256 of body is used, so duplicate is dropped.
+ duplicate_response = self._boto_client.send_message(
+ QueueUrl=self._queue_url,
+ MessageBody=message_body,
+ MessageGroupId='message-group-1',
+ )
+
+ assert_that(duplicate_response['MessageId'], equal_to(response['MessageId']))
+
+ # Different body produces a different content hash and is not deduplicated.
+ other_response = self._boto_client.send_message(
+ QueueUrl=self._queue_url,
+ MessageBody='other body',
+ MessageGroupId='message-group-1',
+ )
+
+ assert_that(other_response['MessageId'], not_(equal_to(response['MessageId'])))
+
+ # Explicit MessageDeduplicationId overrides content-based hash.
+ explicit_response = self._boto_client.send_message(
+ QueueUrl=self._queue_url,
+ MessageBody='yet another body',
+ MessageGroupId='message-group-1',
+ MessageDeduplicationId='explicit-deduplication-id',
+ )
+ explicit_duplicate_response = self._boto_client.send_message(
+ QueueUrl=self._queue_url,
+ MessageBody='completely different body',
+ MessageGroupId='message-group-1',
+ MessageDeduplicationId='explicit-deduplication-id',
+ )
+
+ assert_that(explicit_duplicate_response['MessageId'], equal_to(explicit_response['MessageId']))
+
+ def test_send_message_fifo_without_content_based_deduplication(self):
+ self._create_fifo_queue('send_message_fifo_without_content_based_deduplication')
+
+ def send_without_message_deduplication_id():
+ self._boto_client.send_message(
+ QueueUrl=self._queue_url,
+ MessageBody='hello from fifo sqs',
+ MessageGroupId='message-group-1',
+ )
+
+ # Without ContentBasedDeduplication, MessageDeduplicationId is required.
+ assert_that(
+ send_without_message_deduplication_id,
+ raises(
+ botocore.exceptions.ClientError,
+ pattern='MissingParameter',
+ ),
+ )
diff --git a/ydb/tests/functional/sqs_topic/test_send_message_batch.py b/ydb/tests/functional/sqs_topic/test_send_message_batch.py
index 5097816fbe5..4a1a664b425 100644
--- a/ydb/tests/functional/sqs_topic/test_send_message_batch.py
+++ b/ydb/tests/functional/sqs_topic/test_send_message_batch.py
@@ -76,6 +76,7 @@ class TestSqsTopicSendMessageBatch(KikimrSqsTopicTestBase):
'Id': str(index),
'MessageBody': message_body,
'MessageGroupId': 'message-group-1',
+ 'MessageDeduplicationId': 'deduplication-id-{}'.format(index),
}
for index, message_body in enumerate(message_bodies)
],
diff --git a/ydb/tests/functional/tenants/test_remove_storage_groups.py b/ydb/tests/functional/tenants/test_remove_storage_groups.py
index f5a996ec2c5..195c1854821 100644
--- a/ydb/tests/functional/tenants/test_remove_storage_groups.py
+++ b/ydb/tests/functional/tenants/test_remove_storage_groups.py
@@ -78,8 +78,6 @@ def test_remove_storage_group(ydb_cluster, ydb_root, ydb_safe_test_name, ydb_cli
storage_units_to_remove={'hdd': 2},
)
- ydb_cluster.restart_slots()
-
def get_storage_units():
status = ydb_cluster.get_database_status(database)
units = sum([unit.count for unit in status.allocated_resources.storage_units])
@@ -95,5 +93,13 @@ def test_remove_storage_group(ydb_cluster, ydb_root, ydb_safe_test_name, ydb_cli
status = ydb_cluster.get_database_status(database)
assert_that(status.state, equal_to(cms_pb.GetDatabaseStatusResult.RUNNING))
- ydb_cluster.unregister_and_stop_slots(database_nodes)
+ # Check data integrity
+ with ydb.QuerySessionPool(driver, size=1) as pool:
+ result_sets = pool.execute_with_retries("SELECT * FROM " + table_name)
+ assert len(result_sets) == 1
+ assert len(result_sets[0].rows) == 1
+ assert result_sets[0].rows[0].key == 1
+ assert result_sets[0].rows[0].value == b"value1"
+
ydb_cluster.remove_database(database)
+ ydb_cluster.unregister_and_stop_slots(database_nodes)
diff --git a/ydb/tests/functional/udf_store/upload_udf/ya.make b/ydb/tests/functional/udf_store/upload_udf/ya.make
index 7bf85be0153..2012e9e4d81 100644
--- a/ydb/tests/functional/udf_store/upload_udf/ya.make
+++ b/ydb/tests/functional/udf_store/upload_udf/ya.make
@@ -9,6 +9,10 @@ PEERDIR(
ydb/tests/oss/ydb_sdk_import
)
+NO_CHECK_IMPORTS(
+ ydb.tests.oss.canonical.*
+)
+
DEPENDS(
ydb/tests/stress/kv_volume_tool
)
diff --git a/ydb/tests/library/sqs/requests_client.py b/ydb/tests/library/sqs/requests_client.py
index c7cf0f2c240..904f958b28d 100644
--- a/ydb/tests/library/sqs/requests_client.py
+++ b/ydb/tests/library/sqs/requests_client.py
@@ -371,8 +371,9 @@ class SqsHttpApi(object):
delay_seconds=None, attributes=None, deduplication_id=None, group_id=None
):
if not to_bytes(queue_url).endswith(to_bytes('.fifo')):
- if deduplication_id is not None or group_id is not None:
- raise ValueError("Deduplication id and Group id parameters may be set for FIFO queues only")
+ if group_id is not None:
+ raise ValueError("Group id parameter may be set for FIFO queues only")
+ # MessageDeduplicationId is accepted for standard queues but ignored by the server.
params = {
'QueueUrl': queue_url,
'MessageBody': message_body,
diff --git a/ydb/tests/stress/compare_index_performance/tests/README.md b/ydb/tests/stress/compare_index_performance/tests/README.md
index 87e54f4ffdf..ef54b76ec28 100644
--- a/ydb/tests/stress/compare_index_performance/tests/README.md
+++ b/ydb/tests/stress/compare_index_performance/tests/README.md
@@ -73,6 +73,13 @@ sanity check) or two different refs — **without a local build**.
| `compare_flamegraph` | `` | `1`/`true` → collect CPU flamegraphs (see below) |
| `compare_perf_sudo` | `` | `1`/`true` → run `perf` under `sudo` |
| `compare_perf_freq` | `50` | `perf record -F` sampling frequency |
+| `compare_dataset_source` | `generate` | `generate` (random data) or `s3` (import fixed dataset from S3) |
+| `compare_s3_endpoint` | `` | S3 endpoint URL (required when `dataset_source=s3`) |
+| `compare_s3_bucket` | `` | S3 bucket name (required when `dataset_source=s3`) |
+| `compare_s3_query_destination` | `` | Database destination path for the queries table (optional, defaults to `{database}/vector_query_table`) |
+| `compare_s3_destination` | `` | Database destination path for import (required when `dataset_source=s3`) |
+| `compare_s3_query_source` | `` | S3 object key prefix for a pre-computed queries table (optional, skips dynamic creation) |
+| `compare_s3_query_destination` | `` | Database destination path for the queries table (optional, defaults to `{database}/vector_query_table`)
`table_service_config` values like `enable_vector_index_read=true` are parsed
into booleans; everything else is kept as a string. Feature flags are appended
@@ -131,6 +138,31 @@ the toolkit in `contrib/tools/flame-graph` (shipped to the sandbox via `DATA`):
--test-param compare_current_table_service_config=enable_vector_index_read=true \
--test-param compare_flamegraph=1 \
--test-param compare_perf_sudo=1
+
+# Import data from S3 instead of auto-generating (uses a fixed dataset).
+./ya make --build relwithdebinfo -tA \
+ ydb/tests/stress/compare_index_performance/tests \
+ --test-param compare_dataset_source=s3 \
+ --test-param compare_s3_endpoint=https://storage.yandexcloud.net \
+ --test-param compare_s3_bucket=vector-index \
+ --test-param compare_s3_source=wikipedia \
+ --test-param compare_s3_destination=/Root/testdb/wikipedia \
+ --test-param compare_iterations=3 \
+ --test-param compare_duration=60
+
+# Import both dataset and pre-computed queries table from S3 (exact query
+# vectors are reused across iterations and across sides for perfect reproducibility).
+./ya make --build relwithdebinfo -tA \
+ ydb/tests/stress/compare_index_performance/tests \
+ --test-param compare_dataset_source=s3 \
+ --test-param compare_s3_endpoint=https://storage.yandexcloud.net \
+ --test-param compare_s3_bucket=vector-index \
+ --test-param compare_s3_source=wikipedia \
+ --test-param compare_s3_destination=/Root/testdb/wikipedia \
+ --test-param compare_s3_query_source=wikipedia_queries \
+ --test-param compare_s3_query_destination=/Root/vector_query_table \
+ --test-param compare_iterations=3 \
+ --test-param compare_duration=60
```
## Results
diff --git a/ydb/tests/stress/compare_index_performance/tests/test_compare.py b/ydb/tests/stress/compare_index_performance/tests/test_compare.py
index 9a0db0506c7..71432d4c2b5 100644
--- a/ydb/tests/stress/compare_index_performance/tests/test_compare.py
+++ b/ydb/tests/stress/compare_index_performance/tests/test_compare.py
@@ -33,10 +33,14 @@
# if perf cannot profile, the test fails. perf adds overhead, so measured
# Txs/Sec are perturbed when on.
+import contextlib
import os
import signal
import statistics
import subprocess
+import threading
+import time
+import traceback
import urllib.request
import pytest
@@ -201,6 +205,11 @@ class TestCompareIndexPerformance:
self.rows = yatest.common.get_param('compare_rows', default='10000')
self.threads = yatest.common.get_param('compare_threads', default='10')
self.targets = yatest.common.get_param('compare_targets', default='1000')
+ # Vector index structure params; None → server auto-detects
+ _clusters = yatest.common.get_param('compare_vector_clusters', default='')
+ _levels = yatest.common.get_param('compare_vector_levels', default='')
+ self.vector_clusters = _clusters if _clusters else None
+ self.vector_levels = _levels if _levels else None
# Per-cluster config
self.baseline_flags = _parse_feature_flags(
@@ -212,7 +221,31 @@ class TestCompareIndexPerformance:
self.current_tsc = _parse_table_service_config(
yatest.common.get_param('compare_current_table_service_config', default=''))
- self.workload = yatest.common.get_param('compare_workload', default='all')
+ self.workload = yatest.common.get_param('compare_workload', default='vector')
+ if self.workload not in ('vector', 'fulltext'):
+ print(f"WARNING: unrecognized compare_workload={self.workload!r}; "
+ "all workloads will be skipped (expected 'vector' or 'fulltext')")
+
+ # Dataset source: 'generate' (default) or 's3'
+ self.dataset_source = yatest.common.get_param('compare_dataset_source', default='generate')
+ # S3 import params (used when dataset_source == 's3')
+ self.s3_endpoint = yatest.common.get_param('compare_s3_endpoint', default='')
+ self.s3_bucket = yatest.common.get_param('compare_s3_bucket', default='')
+ self.s3_source = yatest.common.get_param('compare_s3_source', default='')
+ self.s3_destination = yatest.common.get_param('compare_s3_destination', default='')
+ # Optional S3 queries table (when dataset_source == 's3' and queries pre-computed)
+ self.s3_query_source = yatest.common.get_param('compare_s3_query_source', default='')
+ self.s3_query_destination = yatest.common.get_param('compare_s3_query_destination', default='')
+
+ if self.dataset_source == 's3':
+ missing = [name for name, val in [
+ ('compare_s3_endpoint', self.s3_endpoint),
+ ('compare_s3_bucket', self.s3_bucket),
+ ('compare_s3_source', self.s3_source),
+ ('compare_s3_destination', self.s3_destination),
+ ] if not val]
+ if missing:
+ pytest.fail(f"compare_dataset_source=s3 requires: {', '.join(missing)}")
# Flamegraph collection (off by default). When on, each workload run is
# profiled with `perf` and a CPU flamegraph SVG is produced per side per
@@ -268,44 +301,113 @@ class TestCompareIndexPerformance:
)
cluster = KiKiMR(config)
cluster.start()
+ # A stray SIGINT keeps aborting the flamegraph run during teardown even
+ # though this process no longer sends any signal (perf is stopped via a
+ # sentinel file). _sigint_debug logs where it comes from and SWALLOWS it
+ # (which, if signal masking works here at all, is also the fix).
+ shield = self._sigint_debug(label) if self.flamegraph else contextlib.nullcontext()
+ with shield:
+ try:
+ endpoint = "grpc://localhost:%s" % cluster.nodes[1].port
+ log_file = yatest.common.output_path(log_name)
+ err_file = yatest.common.output_path(log_name + ".err")
+ print(f"--- Running {label} workload against {ydbd_path} ---")
+ # stdout and stderr go to separate files: the Txs/Sec line is parsed
+ # from stdout, so stderr noise must not interleave into it.
+ with open(log_file, "w") as out, open(err_file, "w") as err:
+ perf = self._perf_start(cluster.nodes[1].pid, svg_name) if self.flamegraph else None
+ workload_ok = False
+ try:
+ run_workload(endpoint, out, err)
+ workload_ok = True
+ finally:
+ if perf is not None:
+ # Convert/validate the profile only if the workload
+ # itself succeeded; otherwise just stop perf so its
+ # error does not mask the original workload failure.
+ self._perf_finish(perf, svg_name, validate=workload_ok)
+ return extract_total_txs_sec(log_file)
+ finally:
+ cluster.stop()
+
+ @contextlib.contextmanager
+ def _sigint_debug(self, label):
+ # Diagnose (and, if possible, suppress) the stray SIGINT that aborts the
+ # flamegraph run. Writes findings to a published sigint_diag.log:
+ # - whether we are on the main thread (signal.signal only works there),
+ # - whether installing a SIGINT handler succeeds,
+ # - for every SIGINT: a timestamp and the full Python stack (which frame
+ # / subprocess call it interrupted).
+ # The handler SWALLOWS the signal (returns without raising), so if signal
+ # masking is effective in this py3test the run survives — making this both
+ # the probe and the fix. If signal.signal() raises (not the main thread),
+ # that is logged and tells us in-process masking is impossible here.
+ diag = yatest.common.output_path("sigint_diag.log")
+
+ def log(msg):
+ with open(diag, "a") as f:
+ f.write("[%.3f][%s] %s\n" % (time.time(), label, msg))
+
+ on_main = threading.current_thread() is threading.main_thread()
+ log("enter: on_main_thread=%s pid=%s" % (on_main, os.getpid()))
+ hits = []
+
+ def handler(signum, frame):
+ hits.append(signum)
+ log("SIGINT #%d (signum=%d) SWALLOWED; interrupted stack:\n%s"
+ % (len(hits), signum, "".join(traceback.format_stack(frame))))
+
+ prev, installed = None, False
try:
- endpoint = "grpc://localhost:%s" % cluster.nodes[1].port
- log_file = yatest.common.output_path(log_name)
- err_file = yatest.common.output_path(log_name + ".err")
- print(f"--- Running {label} workload against {ydbd_path} ---")
- # stdout and stderr go to separate files: the Txs/Sec line is parsed
- # from stdout, so stderr noise must not interleave into it.
- with open(log_file, "w") as out, open(err_file, "w") as err:
- perf = self._perf_start(cluster.nodes[1].pid, svg_name) if self.flamegraph else None
- workload_ok = False
- try:
- run_workload(endpoint, out, err)
- workload_ok = True
- finally:
- if perf is not None:
- # Convert/validate the profile only if the workload
- # itself succeeded; otherwise just stop perf so its error
- # does not mask the original workload failure.
- self._perf_finish(perf, svg_name, validate=workload_ok)
- return extract_total_txs_sec(log_file)
+ prev = signal.signal(signal.SIGINT, handler)
+ installed = True
+ log("signal.signal(SIGINT, handler) OK")
+ except (ValueError, OSError) as exc:
+ log("signal.signal FAILED (%r) -> cannot mask SIGINT in-process" % (exc,))
+ try:
+ yield
finally:
- cluster.stop()
+ if installed:
+ try:
+ signal.signal(signal.SIGINT, prev)
+ except (ValueError, OSError):
+ pass
+ log("exit: sigint_hits=%d" % len(hits))
# --- flamegraph collection (perf record -> stackcollapse -> flamegraph) ---
def _perf_start(self, pid, svg_name):
- # Profile the live ydbd process for the whole workload run (until SIGINT).
- # Run perf in its own session/process group so we can interrupt it
- # reliably, including when wrapped in `sudo` (which would otherwise not
- # forward our SIGINT to the perf child).
- # perf.data is an intermediate; keep it in the work dir (not the
- # published output dir). Under sudo it is written owned by root, which
- # would make ya's output collection fail with PermissionError if it lived
- # under output_path — it is chowned back to us in _perf_stop.
+ # Profile the live ydbd process for the whole workload run.
+ #
+ # `perf record` finalizes perf.data only when it receives SIGINT/SIGTERM,
+ # so it MUST be signaled to stop. We deliberately do NOT signal it from
+ # this (pytest) process: under sudo on the CI runner that SIGINT leaked
+ # back into pytest and aborted the session as a KeyboardInterrupt, and
+ # in-process signal masking proved ineffective here (signal.signal() is a
+ # no-op off the main thread / gets reset around yatest.common.execute).
+ #
+ # Instead perf runs inside a small shell wrapper that waits for a sentinel
+ # file and then sends `kill -INT` to perf ITSELF. _perf_stop just creates
+ # that file — pytest never sends a signal, so none can leak to it. The
+ # wrapper runs in its own session (start_new_session), so the internal
+ # SIGINT stays fully contained. perf.data is an intermediate; keep it in
+ # the work dir (not the published output dir). Under sudo it is written
+ # owned by root, which would make ya's output collection fail with
+ # PermissionError under output_path — it is chowned back in _perf_stop.
perf_data = yatest.common.work_path(svg_name + ".perf.data")
perf_log = yatest.common.output_path(svg_name + ".perf.log")
+ stop_file = yatest.common.work_path(svg_name + ".perf.stop")
+ if os.path.exists(stop_file):
+ os.remove(stop_file)
+ # Positional args ($1..$4) avoid any dependency on sudo env passing.
+ wrapper = (
+ 'perf record -F "$1" --call-graph dwarf -g --proc-map-timeout=10000 '
+ '--pid "$2" -o "$3" & p=$!; '
+ 'while [ ! -e "$4" ]; do sleep 0.2; done; '
+ 'kill -INT "$p"; wait "$p"'
+ )
cmd = (["sudo"] if self.perf_sudo else []) + [
- "perf", "record", "-F", str(self.perf_freq), "--call-graph", "dwarf",
- "-g", "--proc-map-timeout=10000", "--pid", str(pid), "-o", perf_data,
+ "sh", "-c", wrapper, "perfwrap",
+ str(self.perf_freq), str(pid), perf_data, stop_file,
]
log = open(perf_log, "w")
try:
@@ -315,24 +417,29 @@ class TestCompareIndexPerformance:
log.close()
pytest.fail(f"failed to start perf for flamegraph {svg_name}: {exc}; "
f"perf log: {perf_log}")
- return {"proc": proc, "log": log, "data": perf_data, "perf_log": perf_log}
+ return {"proc": proc, "log": log, "data": perf_data,
+ "perf_log": perf_log, "stop": stop_file}
def _perf_stop(self, perf):
- # Send SIGINT to perf's process group (perf flushes perf.data on SIGINT).
- # With sudo, perf is not our direct child, so signal the whole group via
- # `sudo kill` to reach the privileged perf process.
+ # Ask perf to stop by creating the sentinel file the wrapper is waiting
+ # on (see _perf_start); the wrapper then sends perf its SIGINT internally.
+ # pytest itself sends NO signal, so nothing can leak back into it.
proc = perf["proc"]
try:
- if self.perf_sudo:
- subprocess.call(["sudo", "kill", "-INT", f"-{proc.pid}"])
- else:
- os.killpg(proc.pid, signal.SIGINT)
+ open(perf["stop"], "w").close()
except OSError:
- proc.send_signal(signal.SIGINT)
+ pass
try:
proc.wait(timeout=120)
except subprocess.TimeoutExpired:
- proc.kill()
+ # Wrapper never noticed the sentinel; kill the whole perf session
+ # group. It is isolated (start_new_session) so a negative-pgid kill
+ # cannot reach pytest. Under sudo the group is root-owned.
+ pgid = os.getpgid(proc.pid)
+ if self.perf_sudo:
+ subprocess.call(["sudo", "kill", "-KILL", "--", f"-{pgid}"])
+ else:
+ os.killpg(pgid, signal.SIGKILL)
proc.wait()
perf["log"].close()
# perf ran as root under sudo, so perf.data is root-owned. Chown it back
@@ -531,50 +638,96 @@ class TestCompareIndexPerformance:
# --- tests ---
def test_vector(self):
- if self.workload not in ("all", "vector"):
+ if self.workload != "vector":
pytest.skip(f"compare_workload={self.workload} excludes vector")
baseline_ydbd = self._baseline_ydbd()
current_ydbd = self._current_ydbd()
- data_dir = yatest.common.output_path("vector_data")
main_values = []
current_values = []
- # First iteration: baseline runs in generate mode (creates + dumps the
- # query table). Subsequent baseline runs and all current runs use load
- # mode, reusing the dumped query table.
- for i in range(1, self.iterations + 1):
- print(f"=== Vector iteration {i}/{self.iterations} ===")
- baseline_mode = "generate" if i == 1 else "load"
+ vector_index_args = []
+ if self.vector_clusters is not None:
+ vector_index_args += ["--clusters", self.vector_clusters]
+ if self.vector_levels is not None:
+ vector_index_args += ["--levels", self.vector_levels]
- def baseline_workload(endpoint, out, err, mode=baseline_mode):
- self._exec_workload("YDB_VECTOR_WORKLOAD_PATH", endpoint, out, err, [
- "--mode", mode, "--data-dir", data_dir,
- "--targets", self.targets, "--warmup", self.warmup,
- "--rows", self.rows, "--threads", self.threads,
- ])
+ if self.dataset_source == "s3":
+ # S3 mode: import the same fixed dataset from S3 on every iteration
+ # instead of generating random data.
+ s3_args = [
+ "--s3-endpoint", self.s3_endpoint,
+ "--s3-bucket", self.s3_bucket,
+ "--s3-source", self.s3_source,
+ "--s3-destination", self.s3_destination,
+ ]
+ if self.s3_query_source:
+ s3_args += ["--s3-query-source", self.s3_query_source]
+ if self.s3_query_destination:
+ s3_args += ["--s3-query-destination", self.s3_query_destination]
+ for i in range(1, self.iterations + 1):
+ print(f"=== Vector iteration {i}/{self.iterations} (dataset_source=s3) ===")
- def current_workload(endpoint, out, err):
- self._exec_workload("YDB_VECTOR_WORKLOAD_PATH", endpoint, out, err, [
- "--mode", "load", "--data-dir", data_dir,
- "--targets", self.targets, "--warmup", self.warmup,
- "--rows", self.rows, "--threads", self.threads,
- ])
+ def s3_baseline_workload(endpoint, out, err):
+ self._exec_workload("YDB_VECTOR_WORKLOAD_PATH", endpoint, out, err, [
+ "--mode", "s3",
+ "--targets", self.targets, "--warmup", self.warmup,
+ "--rows", self.rows, "--threads", self.threads,
+ ] + vector_index_args + s3_args)
- collect_value(main_values, self._run_one(
- self.ref, baseline_ydbd, self.baseline_flags, self.baseline_tsc,
- baseline_workload, f"vector_main_{i}.log", f"vector_main_{i}.svg"))
- collect_value(current_values, self._run_one(
- "current", current_ydbd, self.current_flags, self.current_tsc,
- current_workload, f"vector_current_{i}.log", f"vector_current_{i}.svg"))
- if self.flamegraph:
- self._flamegraph_diff("vector", i)
+ def s3_current_workload(endpoint, out, err):
+ self._exec_workload("YDB_VECTOR_WORKLOAD_PATH", endpoint, out, err, [
+ "--mode", "s3",
+ "--targets", self.targets, "--warmup", self.warmup,
+ "--rows", self.rows, "--threads", self.threads,
+ ] + vector_index_args + s3_args)
+
+ collect_value(main_values, self._run_one(
+ self.ref, baseline_ydbd, self.baseline_flags, self.baseline_tsc,
+ s3_baseline_workload, f"vector_main_{i}.log", f"vector_main_{i}.svg"))
+ collect_value(current_values, self._run_one(
+ "current", current_ydbd, self.current_flags, self.current_tsc,
+ s3_current_workload, f"vector_current_{i}.log", f"vector_current_{i}.svg"))
+ if self.flamegraph:
+ self._flamegraph_diff("vector", i)
+ else:
+ # Generate mode (default): first iteration generates data + dumps query table,
+ # subsequent iterations reuse the dumped query table.
+ data_dir = yatest.common.output_path("vector_data")
+
+ for i in range(1, self.iterations + 1):
+ print(f"=== Vector iteration {i}/{self.iterations} (dataset_source=generate) ===")
+
+ baseline_mode = "generate" if i == 1 else "load"
+
+ def baseline_workload(endpoint, out, err, mode=baseline_mode):
+ self._exec_workload("YDB_VECTOR_WORKLOAD_PATH", endpoint, out, err, [
+ "--mode", mode, "--data-dir", data_dir,
+ "--targets", self.targets, "--warmup", self.warmup,
+ "--rows", self.rows, "--threads", self.threads,
+ ] + vector_index_args)
+
+ def current_workload(endpoint, out, err):
+ self._exec_workload("YDB_VECTOR_WORKLOAD_PATH", endpoint, out, err, [
+ "--mode", "load", "--data-dir", data_dir,
+ "--targets", self.targets, "--warmup", self.warmup,
+ "--rows", self.rows, "--threads", self.threads,
+ ] + vector_index_args)
+
+ collect_value(main_values, self._run_one(
+ self.ref, baseline_ydbd, self.baseline_flags, self.baseline_tsc,
+ baseline_workload, f"vector_main_{i}.log", f"vector_main_{i}.svg"))
+ collect_value(current_values, self._run_one(
+ "current", current_ydbd, self.current_flags, self.current_tsc,
+ current_workload, f"vector_current_{i}.log", f"vector_current_{i}.svg"))
+ if self.flamegraph:
+ self._flamegraph_diff("vector", i)
self._report("vector", "vector select", self._summarize(main_values, current_values))
def test_fulltext(self):
- if self.workload not in ("all", "fulltext"):
+ if self.workload != "fulltext":
pytest.skip(f"compare_workload={self.workload} excludes fulltext")
baseline_ydbd = self._baseline_ydbd()
diff --git a/ydb/tests/stress/compare_index_performance/tests/ya.make b/ydb/tests/stress/compare_index_performance/tests/ya.make
index 72ef6366f92..a497d09c9ac 100644
--- a/ydb/tests/stress/compare_index_performance/tests/ya.make
+++ b/ydb/tests/stress/compare_index_performance/tests/ya.make
@@ -9,7 +9,7 @@ TEST_SRCS(
test_compare.py
)
-REQUIREMENTS(ram:32 cpu:4)
+REQUIREMENTS(ram:64 cpu:4)
TAG(ya:external)
SIZE(LARGE)
diff --git a/ydb/tests/stress/vector_workload/__main__.py b/ydb/tests/stress/vector_workload/__main__.py
index bdff9c08bbf..84dd6e6afd0 100644
--- a/ydb/tests/stress/vector_workload/__main__.py
+++ b/ydb/tests/stress/vector_workload/__main__.py
@@ -10,19 +10,30 @@ if __name__ == '__main__':
parser.add_argument('--endpoint', default='grpc://localhost:2135', help="YDB endpoint")
parser.add_argument('--database', default=None, required=True, help='A database to connect')
parser.add_argument('--duration', default=120, type=int, help='A duration of workload in seconds')
- parser.add_argument('--mode', default='standalone', choices=['standalone', 'generate', 'load'],
- help='Mode: standalone (default), generate (generate + dump), load (restore + run)')
+ parser.add_argument('--mode', default='standalone', choices=['standalone', 'generate', 'load', 's3'],
+ help='Mode: standalone (default), generate (generate + dump), load (restore + run), s3 (import from S3)')
parser.add_argument('--data-dir', default=None, help='Directory for dump/restore data (required for generate/load modes)')
parser.add_argument('--targets', default=1000, type=int, help='Number of query vectors for run select (default: 1000)')
parser.add_argument('--warmup', default=0, type=int, help='Warmup duration in seconds before measured run (default: 0, disabled)')
parser.add_argument('--rows', default=10000, type=int, help='Number of rows in generated database (default: 10000)')
parser.add_argument('--threads', default=10, type=int, help='Number of threads for load testing (default: 10)')
+ parser.add_argument('--clusters', default=None, type=int, help='Number of clusters in kmeans tree (default: server auto-detect)')
+ parser.add_argument('--levels', default=None, type=int, help='Number of levels in kmeans tree (default: server auto-detect)')
+ parser.add_argument('--s3-endpoint', default=None, help='S3 endpoint for dataset import (required for --mode=s3)')
+ parser.add_argument('--s3-bucket', default=None, help='S3 bucket for dataset import (required for --mode=s3)')
+ parser.add_argument('--s3-source', default=None, help='S3 object key prefix for dataset import (required for --mode=s3)')
+ parser.add_argument('--s3-destination', default=None, help='Database path for dataset import (required for --mode=s3)')
+ parser.add_argument('--s3-query-source', default=None, help='S3 object key prefix for query table import (optional for --mode=s3)')
+ parser.add_argument('--s3-query-destination', default=None, help='Database path for query table import (optional for --mode=s3)')
parser.add_argument('--log_file', default=None, help='Append log into specified file')
args = parser.parse_args()
if args.mode in ('generate', 'load') and not args.data_dir:
parser.error(f"--data-dir is required for --mode={args.mode}")
+ if args.mode == 's3':
+ if not all([args.s3_endpoint, args.s3_bucket, args.s3_source, args.s3_destination]):
+ parser.error("--s3-endpoint, --s3-bucket, --s3-source, and --s3-destination are required for --mode=s3")
if args.log_file:
logging.basicConfig(
@@ -35,6 +46,11 @@ if __name__ == '__main__':
workload = YdbVectorWorkload(args.endpoint, args.database, duration=args.duration,
mode=args.mode, data_dir=args.data_dir, targets=args.targets,
- warmup=args.warmup, rows=args.rows, threads=args.threads)
+ warmup=args.warmup, rows=args.rows, threads=args.threads,
+ clusters=args.clusters, levels=args.levels,
+ s3_endpoint=args.s3_endpoint, s3_bucket=args.s3_bucket,
+ s3_source=args.s3_source, s3_destination=args.s3_destination,
+ s3_query_source=args.s3_query_source,
+ s3_query_destination=args.s3_query_destination)
workload.start()
workload.join()
diff --git a/ydb/tests/stress/vector_workload/workload/__init__.py b/ydb/tests/stress/vector_workload/workload/__init__.py
index 1be27e0dd2c..d32fe1bf50e 100644
--- a/ydb/tests/stress/vector_workload/workload/__init__.py
+++ b/ydb/tests/stress/vector_workload/workload/__init__.py
@@ -1,3 +1,4 @@
+import json
import logging
import os
import shutil
@@ -15,7 +16,8 @@ QUERY_TABLE_NAME = "vector_query_table"
class YdbVectorWorkload(WorkloadBase):
- def __init__(self, endpoint, database, duration, mode="standalone", data_dir=None, targets=1000, warmup=0, rows=10000, threads=10):
+ def __init__(self, endpoint, database, duration, mode="standalone", data_dir=None, targets=1000, warmup=0, rows=10000, threads=10, clusters=None, levels=None,
+ s3_endpoint=None, s3_bucket=None, s3_source=None, s3_destination=None, s3_query_source=None, s3_query_destination=None):
super().__init__(None, '', 'vector_workload', None)
self.endpoint = endpoint
self.database = database
@@ -26,9 +28,37 @@ class YdbVectorWorkload(WorkloadBase):
self.warmup = str(warmup)
self.rows = str(rows)
self.threads = str(threads)
+ self.clusters = str(clusters) if clusters is not None else None
+ self.levels = str(levels) if levels is not None else None
+ self.s3_endpoint = s3_endpoint
+ self.s3_bucket = s3_bucket
+ self.s3_source = s3_source
+ self.s3_destination = s3_destination
+ self.s3_query_source = s3_query_source
+ self.s3_query_destination = s3_query_destination
+ # Table the workload operates on. In s3 mode the data is imported to a
+ # user-provided destination path, otherwise the default workload table.
+ self.table_name = "vector_index_workload"
+ self.query_table_name = QUERY_TABLE_NAME
+ if self.mode == "s3":
+ if self.s3_destination:
+ self.table_name = self._rel_to_database(self.s3_destination)
+ if self.s3_query_source:
+ query_dest = self.s3_query_destination or f"{self.database}/{QUERY_TABLE_NAME}"
+ self.query_table_name = self._rel_to_database(query_dest)
self.tempdir = None
self._unpack_resource('ydb_cli')
+ def _rel_to_database(self, path):
+ """Convert an absolute database path to a name relative to the database."""
+ prefix = self.database.rstrip('/') + '/'
+ if path.startswith(prefix):
+ return path[len(prefix):]
+ if path.rstrip('/') == self.database.rstrip('/'):
+ raise ValueError(
+ f"path must be under the database, not the database itself: {path}")
+ return path.lstrip('/')
+
def __del__(self):
if self.tempdir is not None:
self.tempdir.cleanup()
@@ -78,18 +108,77 @@ class YdbVectorWorkload(WorkloadBase):
print(f"Attempt {attempt}/{retries} failed, retrying in {delay}s...")
time.sleep(delay)
+ def _wait_for_table_stats(self, timeout=90, interval=3):
+ """Poll until the table's row-count statistic is non-zero.
+
+ After an index build the row-count estimate is computed asynchronously.
+ The vector workload's Init() reads it via DescribeTable().GetTableRows()
+ and aborts ("statistics is not calculated yet") if it starts too early.
+ We poll the same underlying datashard stat through the .sys/partition_stats
+ system view and return as soon as it lands — usually far quicker than a
+ fixed sleep. This is best-effort: if the view is not queryable we fall
+ back to a short fixed wait, and on timeout we proceed anyway (the select's
+ own cmd_run_with_retry is the backstop)."""
+ full_path = f"{self.database.rstrip('/')}/{self.table_name}"
+ query = (
+ "SELECT COALESCE(SUM(RowCount), 0u) AS rows "
+ f"FROM `.sys/partition_stats` WHERE Path = '{full_path}';"
+ )
+ cmd = self.get_cli_prefix() + ['yql', '-s', query, '--format', 'json-unicode']
+ print(f"Waiting for table statistics on {full_path}...")
+ deadline = time.time() + timeout
+ query_ok = False
+ while time.time() < deadline:
+ rows = None
+ try:
+ proc = subprocess.run(cmd, check=True, text=True, capture_output=True)
+ query_ok = True
+ for line in proc.stdout.splitlines():
+ line = line.strip()
+ if line:
+ val = json.loads(line).get('rows')
+ if val is not None:
+ rows = int(val)
+ except (subprocess.CalledProcessError, ValueError) as e:
+ if not query_ok:
+ # System view not queryable here: don't spin, fall back to a
+ # brief fixed wait and let the select's retry cover the rest.
+ print(f"Cannot query table statistics ({e}); falling back to fixed wait")
+ time.sleep(30)
+ return
+ print("Stats poll query failed, retrying...")
+ if rows:
+ print(f"Table statistics ready: {full_path} has ~{rows} rows")
+ return
+ time.sleep(interval)
+ print(f"Timed out after {timeout}s waiting for statistics on {full_path}; proceeding")
+
+ def _build_index_subcmds(self):
+ """Subcommands to build the index on the workload table."""
+ subcmds = ['build-index', '--distance', 'cosine', '--table', self.table_name]
+ if self.mode == "s3":
+ # An imported dataset has a fixed, externally-defined vector dimension
+ # and type. Pass 0 so the CLI omits them from the DDL and the server
+ # autodetects both from the data.
+ subcmds += ['--vector-dimension', '0']
+ if self.clusters is not None:
+ subcmds += ['--kmeans-tree-clusters', self.clusters]
+ if self.levels is not None:
+ subcmds += ['--kmeans-tree-levels', self.levels]
+ return subcmds
+
def _create_query_table(self):
"""Create a query table with N rows from the main table."""
- print(f"Creating query table {QUERY_TABLE_NAME} with {self.targets} rows...")
+ print(f"Creating query table {self.query_table_name} with {self.targets} rows...")
create_sql = (
- f"CREATE TABLE `{QUERY_TABLE_NAME}` "
+ f"CREATE TABLE `{self.query_table_name}` "
f"(id Uint64 NOT NULL, embedding String, PRIMARY KEY(id));"
)
self.cmd_run(self.get_cli_prefix() + ['yql', '-s', create_sql])
populate_sql = (
- f"UPSERT INTO `{QUERY_TABLE_NAME}` "
- f"SELECT id, embedding FROM `vector_index_workload` "
+ f"UPSERT INTO `{self.query_table_name}` "
+ f"SELECT id, embedding FROM `{self.table_name}` "
f"ORDER BY id LIMIT {self.targets};"
)
self.cmd_run(self.get_cli_prefix() + ['yql', '-s', populate_sql])
@@ -102,7 +191,7 @@ class YdbVectorWorkload(WorkloadBase):
self.cmd_run(
self.get_tools_prefix(subcmds=[
'dump',
- '-p', self.database + '/' + QUERY_TABLE_NAME,
+ '-p', self.database + '/' + self.query_table_name,
'-o', self.data_dir,
])
)
@@ -134,14 +223,11 @@ class YdbVectorWorkload(WorkloadBase):
)
# Build index explicitly and wait for completion
self.cmd_run_with_retry(
- self.get_command_prefix(subcmds=[
- 'build-index',
- '--distance', 'cosine',
- ])
+ self.get_command_prefix(subcmds=self._build_index_subcmds())
)
- # Wait for table statistics to be calculated after index build
- print("Waiting for table statistics to be calculated...")
- time.sleep(30)
+ # Wait until table statistics (row-count estimate) are computed after the
+ # index build; the select workload aborts if it starts before they land.
+ self._wait_for_table_stats()
def _get_select_subcmds(self, seconds):
subcmds = [
@@ -149,9 +235,10 @@ class YdbVectorWorkload(WorkloadBase):
'--seconds', str(seconds),
'--threads', self.threads,
'--targets', self.targets,
+ '--table', self.table_name,
]
- if self.mode in ('generate', 'load'):
- subcmds.extend(['--query-table', QUERY_TABLE_NAME])
+ if self.mode in ('generate', 'load', 's3'):
+ subcmds.extend(['--query-table', self.query_table_name])
return subcmds
def _run_select(self):
@@ -192,10 +279,45 @@ class YdbVectorWorkload(WorkloadBase):
self.get_command_prefix(subcmds=['clean'])
)
+ def _import_s3_data(self):
+ """Import data from S3 using ydb import s3, optionally including the queries table."""
+ item_main = f"Source={self.s3_source},Destination={self.s3_destination}"
+ cmd = self.get_cli_prefix() + [
+ "import", "s3",
+ "--s3-endpoint", self.s3_endpoint,
+ "--bucket", self.s3_bucket,
+ "--item", item_main,
+ ]
+ if self.s3_query_source:
+ query_dest = self.s3_query_destination or f"{self.database}/{QUERY_TABLE_NAME}"
+ item_query = f"Source={self.s3_query_source},Destination={query_dest}"
+ cmd += ["--item", item_query]
+ print(f"Importing queries table from S3: {self.s3_query_source} -> {query_dest}")
+ self.cmd_run(cmd)
+ # Build index explicitly and wait for completion
+ self.cmd_run_with_retry(
+ self.get_command_prefix(subcmds=self._build_index_subcmds())
+ )
+ # Wait until table statistics (row-count estimate) are computed after the
+ # index build; the select workload aborts if it starts before they land.
+ self._wait_for_table_stats()
+
+ def __loop_s3(self):
+ """S3 mode: import data from S3, optionally import queries table, build index, run select, clean."""
+ self._import_s3_data()
+ if not self.s3_query_source:
+ self._create_query_table()
+ self._run_select()
+ self.cmd_run(
+ self.get_command_prefix(subcmds=['clean'])
+ )
+
def get_workload_thread_funcs(self):
if self.mode == "generate":
return [self.__loop_generate]
elif self.mode == "load":
return [self.__loop_load]
+ elif self.mode == "s3":
+ return [self.__loop_s3]
else:
return [self.__loop_standalone]