diff --git a/.github/scripts/ctest-wrapper/ctest b/.github/scripts/ctest-wrapper/ctest new file mode 100755 index 00000000000..bb0affdcdbd --- /dev/null +++ b/.github/scripts/ctest-wrapper/ctest @@ -0,0 +1,27 @@ +#!/bin/bash + +set -u + +if [ -z "${FLB_CTEST_REAL_PATH:-}" ]; then + echo "FLB_CTEST_REAL_PATH must name the real ctest executable" >&2 + exit 2 +fi + +wrapper_path="$(cd "$(dirname "$0")" && pwd -P)/$(basename "$0")" +ctest_path="$(cd "$(dirname "$FLB_CTEST_REAL_PATH")" && pwd -P)/$(basename "$FLB_CTEST_REAL_PATH")" + +if [ "$ctest_path" = "$wrapper_path" ]; then + echo "FLB_CTEST_REAL_PATH resolves to the ctest timeout wrapper" >&2 + exit 2 +fi + +timeout_seconds="${FLB_CTEST_DEFAULT_TIMEOUT_SECONDS:-300}" + +case "$timeout_seconds" in + ''|*[!0-9]*|0) + echo "FLB_CTEST_DEFAULT_TIMEOUT_SECONDS must be a positive integer" >&2 + exit 2 + ;; +esac + +exec "$FLB_CTEST_REAL_PATH" "$@" --timeout "$timeout_seconds" diff --git a/.github/scripts/tests/test_ctest_timeout_wrapper.sh b/.github/scripts/tests/test_ctest_timeout_wrapper.sh new file mode 100755 index 00000000000..a3560062e67 --- /dev/null +++ b/.github/scripts/tests/test_ctest_timeout_wrapper.sh @@ -0,0 +1,91 @@ +#!/bin/bash + +set -eu + +script_dir="$(cd "$(dirname "$0")" && pwd -P)" +wrapper="$script_dir/../ctest-wrapper/ctest" +temporary_dir="$(mktemp -d "${TMPDIR:-/tmp}/flb-ctest-wrapper.XXXXXX")" + +cleanup() +{ + rm -rf "$temporary_dir" +} + +trap cleanup EXIT + +fake_ctest="$temporary_dir/fake ctest" +arguments_file="$temporary_dir/arguments" + +cat > "$fake_ctest" <<'EOF' +#!/bin/bash +printf '%s\n' "$@" > "$FLB_CTEST_ARGUMENTS_FILE" +exit "${FLB_CTEST_FAKE_EXIT_CODE:-0}" +EOF +chmod +x "$fake_ctest" + +FLB_CTEST_REAL_PATH="$fake_ctest" \ +FLB_CTEST_ARGUMENTS_FILE="$arguments_file" \ +FLB_CTEST_DEFAULT_TIMEOUT_SECONDS=17 \ + "$wrapper" --test-dir "directory with spaces" --output-on-failure + +expected_arguments="$temporary_dir/expected-arguments" +cat > "$expected_arguments" <<'EOF' +--test-dir +directory with spaces +--output-on-failure +--timeout +17 +EOF +cmp "$expected_arguments" "$arguments_file" + +set +e +FLB_CTEST_REAL_PATH="$fake_ctest" \ +FLB_CTEST_ARGUMENTS_FILE="$arguments_file" \ +FLB_CTEST_FAKE_EXIT_CODE=42 \ + "$wrapper" +wrapper_status=$? +set -e + +if [ "$wrapper_status" -ne 42 ]; then + echo "wrapper returned $wrapper_status instead of the ctest status 42" >&2 + exit 1 +fi + +timeout_project="$temporary_dir/default-timeout" +mkdir "$timeout_project" +cat > "$timeout_project/CTestTestfile.cmake" <<'EOF' +add_test(slow-test /bin/sleep 2) +EOF + +set +e +FLB_CTEST_REAL_PATH="$(command -v ctest)" \ +FLB_CTEST_DEFAULT_TIMEOUT_SECONDS=1 \ + "$wrapper" --test-dir "$timeout_project" --output-on-failure +timeout_status=$? +set -e + +if [ "$timeout_status" -eq 0 ]; then + echo "ctest unexpectedly passed a test longer than the default timeout" >&2 + exit 1 +fi + +property_project="$temporary_dir/explicit-timeout" +mkdir "$property_project" +cat > "$property_project/CTestTestfile.cmake" <<'EOF' +add_test(slow-test /bin/sleep 2) +set_tests_properties(slow-test PROPERTIES TIMEOUT 3) +EOF + +FLB_CTEST_REAL_PATH="$(command -v ctest)" \ +FLB_CTEST_DEFAULT_TIMEOUT_SECONDS=1 \ + "$wrapper" --test-dir "$property_project" --output-on-failure + +set +e +FLB_CTEST_REAL_PATH="$wrapper" "$wrapper" +recursion_status=$? +set -e + +if [ "$recursion_status" -ne 2 ]; then + echo "wrapper did not reject itself as the real ctest executable" >&2 + exit 1 +fi diff --git a/.github/workflows/unit-tests.yaml b/.github/workflows/unit-tests.yaml index ba463c447e7..281658ac7e8 100644 --- a/.github/workflows/unit-tests.yaml +++ b/.github/workflows/unit-tests.yaml @@ -189,12 +189,24 @@ jobs: echo "CC = $CC, CXX = $CXX, FLB_OPT = $FLB_OPT" brew update brew install bison flex openssl || true + export FLB_CTEST_REAL_PATH="$(command -v ctest)" + export PATH="$GITHUB_WORKSPACE/.github/scripts/ctest-wrapper:$PATH" ci/scripts/run-unit-tests.sh env: CC: gcc CXX: g++ FLB_OPT: ${{ matrix.flb_option }} + - name: Upload CTest failure diagnostics + if: ${{ failure() || cancelled() }} + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: macos-ctest-logs-${{ matrix.flb_option }} + path: | + build/Testing/Temporary/LastTest.log + build/Testing/Temporary/LastTest.log.tmp + if-no-files-found: warn + run-aarch64-unit-tests: runs-on: ${{(github.repository == 'fluent/fluent-bit') && 'ubuntu-24.04-arm' || 'ubuntu-latest' }} permissions: diff --git a/benchmarks/CMakeLists.txt b/benchmarks/CMakeLists.txt index 1be77cf208d..d3750e78a76 100644 --- a/benchmarks/CMakeLists.txt +++ b/benchmarks/CMakeLists.txt @@ -17,3 +17,9 @@ target_link_libraries(flb-bench-processor_sampling fluent-bit-static ${CMAKE_THREAD_LIBS_INIT} ) + +add_executable(flb-bench-output_throttle flb-bench-output_throttle.c) +target_link_libraries(flb-bench-output_throttle + fluent-bit-static + ${CMAKE_THREAD_LIBS_INIT} +) diff --git a/benchmarks/flb-bench-output_throttle.c b/benchmarks/flb-bench-output_throttle.c new file mode 100644 index 00000000000..8db7318c9b5 --- /dev/null +++ b/benchmarks/flb-bench-output_throttle.c @@ -0,0 +1,169 @@ +/* -*- Mode: C; tab-width: 4; indent-tabs-mode: nil; c-basic-offset: 4 -*- */ + +#include +#include +#include +#include +#include + +#include +#include + +#define DEFAULT_ITERATIONS 1000000 +#define CONTENTION_THREADS 4 + +struct benchmark_worker { + uint64_t iterations; + uint64_t admitted; + struct flb_output_throttle *gate; +}; + +static void *run_admissions(void *data) +{ + uint64_t index; + uint64_t generation; + struct benchmark_worker *worker; + + worker = data; + for (index = 0; index < worker->iterations; index++) { + worker->admitted += flb_output_throttle_admit( + worker->gate, + flb_output_throttle_now_ms(), + &generation); + } + + return NULL; +} + +static double elapsed_seconds(uint64_t start, uint64_t end) +{ + return (double) (end - start) / 1000000000.0; +} + +static void report(const char *name, uint64_t operations, double seconds) +{ + printf("%-30s %12" PRIu64 " ops %10.3f Mops/s\n", + name, operations, (double) operations / seconds / 1000000.0); +} + +static double run_parallel(struct flb_output_throttle *gate, + uint64_t iterations) +{ + int index; + int created; + uint64_t start; + uint64_t end; + pthread_t threads[CONTENTION_THREADS]; + struct benchmark_worker workers[CONTENTION_THREADS]; + + created = 0; + start = cfl_time_now(); + for (index = 0; index < CONTENTION_THREADS; index++) { + workers[index].iterations = iterations; + workers[index].admitted = 0; + workers[index].gate = gate; + if (pthread_create(&threads[index], NULL, + run_admissions, &workers[index]) != 0) { + break; + } + created++; + } + for (index = 0; index < created; index++) { + pthread_join(threads[index], NULL); + } + end = cfl_time_now(); + + if (created != CONTENTION_THREADS) { + return -1.0; + } + return elapsed_seconds(start, end); +} + +static void report_overhead(const char *name, double baseline, double candidate) +{ + printf("%-30s %10.2f%%\n", name, + (candidate / baseline - 1.0) * 100.0); +} + +int main(int argc, char **argv) +{ + uint64_t iterations; + uint64_t start; + uint64_t end; + uint64_t generation; + double disabled_seconds; + double enabled_seconds; + double disabled_parallel_seconds; + double enabled_parallel_seconds; + volatile uint64_t baseline; + struct benchmark_worker workers[CONTENTION_THREADS]; + struct flb_output_throttle disabled_gate; + struct flb_output_throttle enabled_gate; + + iterations = DEFAULT_ITERATIONS; + if (argc > 1) { + iterations = strtoull(argv[1], NULL, 10); + } + if (iterations == 0) { + fprintf(stderr, "iterations must be greater than zero\n"); + return EXIT_FAILURE; + } + + if (flb_output_throttle_init(&disabled_gate, + FLB_FALSE, 1000, 60000) != 0) { + fprintf(stderr, "could not initialize throttle gates\n"); + return EXIT_FAILURE; + } + if (flb_output_throttle_init(&enabled_gate, + FLB_TRUE, 1000, 60000) != 0) { + fprintf(stderr, "could not initialize throttle gates\n"); + flb_output_throttle_destroy(&disabled_gate); + return EXIT_FAILURE; + } + + baseline = 0; + start = cfl_time_now(); + for (generation = 0; generation < iterations; generation++) { + baseline += flb_output_throttle_now_ms() != UINT64_MAX; + } + end = cfl_time_now(); + report("clock-only baseline", iterations, elapsed_seconds(start, end)); + + workers[0].iterations = iterations; + workers[0].admitted = 0; + workers[0].gate = &disabled_gate; + start = cfl_time_now(); + run_admissions(&workers[0]); + end = cfl_time_now(); + disabled_seconds = elapsed_seconds(start, end); + report("disabled gate", iterations, disabled_seconds); + + workers[0].admitted = 0; + workers[0].gate = &enabled_gate; + start = cfl_time_now(); + run_admissions(&workers[0]); + end = cfl_time_now(); + enabled_seconds = elapsed_seconds(start, end); + report("enabled ready gate", iterations, enabled_seconds); + report_overhead("enabled vs disabled overhead", + disabled_seconds, enabled_seconds); + + disabled_parallel_seconds = run_parallel(&disabled_gate, iterations); + enabled_parallel_seconds = run_parallel(&enabled_gate, iterations); + if (disabled_parallel_seconds < 0 || enabled_parallel_seconds < 0) { + fprintf(stderr, "could not create benchmark threads\n"); + flb_output_throttle_destroy(&enabled_gate); + flb_output_throttle_destroy(&disabled_gate); + return EXIT_FAILURE; + } + report("disabled gate, 4 threads", iterations * CONTENTION_THREADS, + disabled_parallel_seconds); + report("enabled ready gate, 4 threads", + iterations * CONTENTION_THREADS, enabled_parallel_seconds); + report_overhead("4-thread enabled overhead", + disabled_parallel_seconds, enabled_parallel_seconds); + + flb_output_throttle_destroy(&enabled_gate); + flb_output_throttle_destroy(&disabled_gate); + return baseline == 0; +} diff --git a/include/fluent-bit/flb_http_retry_after.h b/include/fluent-bit/flb_http_retry_after.h new file mode 100644 index 00000000000..67a8c6dda82 --- /dev/null +++ b/include/fluent-bit/flb_http_retry_after.h @@ -0,0 +1,46 @@ +/* -*- Mode: C; tab-width: 4; indent-tabs-mode: nil; c-basic-offset: 4 -*- */ + +/* Fluent Bit + * ========== + * Copyright (C) 2015-2026 The Fluent Bit Authors + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#ifndef FLB_HTTP_RETRY_AFTER_H +#define FLB_HTTP_RETRY_AFTER_H + +#include + +#include +#include + +enum flb_retry_after_status { + FLB_RETRY_AFTER_ABSENT = 0, + FLB_RETRY_AFTER_VALID, + FLB_RETRY_AFTER_INVALID, + FLB_RETRY_AFTER_SATURATED +}; + +FLB_EXPORT int flb_http_retry_after_parse(const char *value, + size_t length, + int64_t received_wall_time_ms, + uint64_t *delay_ms); + +FLB_EXPORT int flb_http_retry_after_parse_headers(const char *headers, + size_t length, + int64_t received_wall_time_ms, + uint64_t *delay_ms, + size_t *invalid_count); + +#endif diff --git a/include/fluent-bit/flb_macros.h b/include/fluent-bit/flb_macros.h index 5780d7077b8..5cf760b6518 100644 --- a/include/fluent-bit/flb_macros.h +++ b/include/fluent-bit/flb_macros.h @@ -30,6 +30,7 @@ #define FLB_ERROR 0 #define FLB_OK 1 #define FLB_RETRY 2 +#define FLB_THROTTLE 3 /* ala-printf format check */ #if defined(__GNUC__) || defined(__clang__) diff --git a/include/fluent-bit/flb_output.h b/include/fluent-bit/flb_output.h index 919130b469f..02682384e9d 100644 --- a/include/fluent-bit/flb_output.h +++ b/include/fluent-bit/flb_output.h @@ -44,6 +44,7 @@ #include #include #include +#include #include #include #include @@ -109,6 +110,20 @@ int flb_chunk_trace_output(struct flb_chunk_trace *trace, struct flb_output_inst struct flb_output_flush; struct flb_http_server_config; +struct flb_sched_timer; + +#define FLB_OUTPUT_DISPATCH_MAGIC 0x4f445350u +#define FLB_OUTPUT_DISPATCH_TASK 1 + +struct flb_output_dispatch { + uint32_t magic; + uint32_t type; + int result; + struct flb_task *task; + struct flb_output_instance *out; + struct flb_config *config; + struct mk_list _head; +}; /* * Tests callbacks @@ -371,6 +386,13 @@ struct flb_output_instance { /* Plugin properties */ int retry_limit; /* max of retries allowed */ int retry_limit_is_set; /* explicitly set by user? */ + struct flb_output_throttle throttle; /* destination cooldown gate */ + struct mk_list throttle_deferred_routes; /* engine-owned route index */ + size_t throttle_deferred_count; + size_t dispatches_inflight; /* queued, active, or completing */ + struct flb_sched_timer *throttle_wakeup; + int throttle_wakeup_pending; + uint64_t throttle_duration_accounted_ms; int use_tls; /* bool, try to use TLS for I/O */ char *match; /* match rule for tag/routing */ #ifdef FLB_HAVE_REGEX @@ -497,6 +519,11 @@ struct flb_output_instance { struct cmt_histogram *cmt_latency; /* m: output_backpressure_wait_seconds */ struct cmt_histogram *cmt_backpressure_wait; + struct cmt_counter *cmt_throttle_events; + struct cmt_gauge *cmt_throttle_active; + struct cmt_gauge *cmt_throttle_remaining; + struct cmt_gauge *cmt_throttle_deferred_routes; + struct cmt_counter *cmt_throttle_duration; /* OLD Metrics API */ #ifdef FLB_HAVE_METRICS @@ -589,6 +616,9 @@ struct flb_output_flush { struct flb_config *config; /* FLB context */ struct flb_output_instance *o_ins; /* output instance */ struct flb_coro *coro; /* parent coro addr */ + uint64_t admission_generation; /* throttle permit generation */ + uint64_t retry_after_ms; /* flush-local destination hint */ + int retry_after_present; /* * if the original event_chunk has been processed, a new @@ -621,6 +651,13 @@ static FLB_INLINE int flb_output_set_successful_route_data( return 0; } +static FLB_INLINE void flb_output_set_retry_after(struct flb_output_flush *out_flush, + uint64_t delay_ms) +{ + out_flush->retry_after_ms = delay_ms; + out_flush->retry_after_present = FLB_TRUE; +} + static FLB_INLINE void *flb_output_get_retry_context( struct flb_output_flush *out_flush, int *records, @@ -1295,6 +1332,20 @@ struct flb_output_flush *flb_output_flush_create(struct flb_task *task, return out_flush; } +struct flb_output_dispatch *flb_output_dispatch_create( + struct flb_task *task, + struct flb_output_instance *out, + struct flb_config *config); +void flb_output_dispatch_destroy(struct flb_output_dispatch *dispatch); +int flb_output_dispatch_post_result(struct flb_output_dispatch *dispatch, + int result, flb_pipefd_t pipe_fd); +int flb_output_throttle_complete(struct flb_output_flush *out_flush, int result); +int flb_output_throttle_wakeup_schedule(struct flb_output_instance *ins); +void flb_output_throttle_wakeup_cancel(struct flb_output_instance *ins); +void flb_output_throttle_wakeup_scan(struct flb_config *config); +void flb_output_throttle_metrics_update(struct flb_output_instance *ins, + uint64_t now_ms); + /* * This function is used by the output plugins to return. It's mandatory * as it will take care to signal the event loop letting know the flush @@ -1319,6 +1370,7 @@ static inline void flb_output_return(int ret, struct flb_coro *co) { out_flush = (struct flb_output_flush *) co->data; o_ins = out_flush->o_ins; task = out_flush->task; + ret = flb_output_throttle_complete(out_flush, ret); if (out_flush->processed_event_chunk) { counted_event_chunk = out_flush->processed_event_chunk; @@ -1346,6 +1398,7 @@ static inline void flb_output_return(int ret, struct flb_coro *co) { } flb_task_set_route_data(task, o_ins, records, bytes); flb_task_deactivate_route(task, o_ins); + flb_task_route_complete(task, o_ins); flb_task_release_lock(task); #ifdef FLB_HAVE_CHUNK_TRACE @@ -1494,6 +1547,7 @@ int flb_output_task_singleplex_enqueue(struct flb_task_queue *queue, struct flb_task *task, struct flb_output_instance *out_ins, struct flb_config *config); +void flb_output_task_singleplex_complete(struct flb_task_queue *queue); int flb_output_task_singleplex_flush_next(struct flb_task_queue *queue); struct flb_output_instance *flb_output_new(struct flb_config *config, const char *output, void *data, diff --git a/include/fluent-bit/flb_output_thread.h b/include/fluent-bit/flb_output_thread.h index ad0aa4ac442..91aa99d3d53 100644 --- a/include/fluent-bit/flb_output_thread.h +++ b/include/fluent-bit/flb_output_thread.h @@ -24,6 +24,8 @@ #include #include +struct flb_output_dispatch; + /* * For every 'upstream' registered in the output plugin initialization, we create * a local entry so we can manage the connections queues locally, on this way we @@ -69,6 +71,10 @@ struct flb_out_thread_instance { struct flb_output_instance *ins; /* output plugin instance */ struct flb_config *config; struct flb_tp_thread *th; + uint64_t shutdown_requested; + int dispatch_shutdown; + pthread_mutex_t dispatch_mutex; + struct mk_list dispatch_queue; struct mk_list _head; /* @@ -100,9 +106,15 @@ int flb_output_thread_pool_create(struct flb_config *config, int flb_output_thread_pool_coros_size(struct flb_output_instance *ins); void flb_output_thread_pool_destroy(struct flb_output_instance *ins); int flb_output_thread_pool_start(struct flb_output_instance *ins); -int flb_output_thread_pool_flush(struct flb_task *task, - struct flb_output_instance *out_ins, - struct flb_config *config); +int flb_output_thread_pool_flush(struct flb_output_dispatch *dispatch); +int flb_output_thread_post_dispatch_result( + struct flb_out_thread_instance *th_ins, + struct flb_output_dispatch *dispatch, + int result); +struct flb_output_dispatch *flb_output_thread_result_fallback_pop( + struct flb_config *config); +void flb_output_thread_result_fallback_remove( + struct flb_output_instance *ins); void flb_output_thread_instance_init(); diff --git a/include/fluent-bit/flb_output_throttle.h b/include/fluent-bit/flb_output_throttle.h new file mode 100644 index 00000000000..95be26e205f --- /dev/null +++ b/include/fluent-bit/flb_output_throttle.h @@ -0,0 +1,72 @@ +/* -*- Mode: C; tab-width: 4; indent-tabs-mode: nil; c-basic-offset: 4 -*- */ + +/* Fluent Bit + * ========== + * Copyright (C) 2015-2026 The Fluent Bit Authors + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#ifndef FLB_OUTPUT_THROTTLE_H +#define FLB_OUTPUT_THROTTLE_H + +#include + +#include +#include +#include +#include + +#define FLB_OUTPUT_THROTTLE_READY 0 +#define FLB_OUTPUT_THROTTLE_COOLDOWN 1 +#define FLB_OUTPUT_THROTTLE_STOPPING 2 + +struct flb_output_throttle { + int enabled; + int state; + uint64_t base_ms; + uint64_t cap_ms; + uint64_t started_ms; + uint64_t until_ms; + uint64_t generation; + uint32_t consecutive_rounds; + uint64_t events; + pthread_mutex_t lock; +}; + +struct flb_output_throttle_snapshot { + int enabled; + int state; + uint64_t started_ms; + uint64_t until_ms; + uint64_t generation; + uint32_t consecutive_rounds; + uint64_t events; +}; + +int flb_output_throttle_init(struct flb_output_throttle *throttle, + int enabled, uint64_t base_ms, uint64_t cap_ms); +void flb_output_throttle_destroy(struct flb_output_throttle *throttle); +int flb_output_throttle_admit(struct flb_output_throttle *throttle, + uint64_t now_ms, uint64_t *generation); +uint64_t flb_output_throttle_publish(struct flb_output_throttle *throttle, + uint64_t now_ms, int hint_present, + uint64_t hint_ms, uint64_t random_value); +void flb_output_throttle_success(struct flb_output_throttle *throttle, + uint64_t now_ms, uint64_t generation); +void flb_output_throttle_stop(struct flb_output_throttle *throttle); +void flb_output_throttle_snapshot(struct flb_output_throttle *throttle, + struct flb_output_throttle_snapshot *snapshot); +uint64_t flb_output_throttle_now_ms(void); + +#endif diff --git a/include/fluent-bit/flb_search_bulk.h b/include/fluent-bit/flb_search_bulk.h index 502ba32b21d..2cf52b7507d 100644 --- a/include/fluent-bit/flb_search_bulk.h +++ b/include/fluent-bit/flb_search_bulk.h @@ -62,6 +62,7 @@ int flb_search_bulk_process_response(const char *response, int acknowledge_all_conflicts, int drop_unrecoverable_records, struct flb_search_bulk_stats *stats, + int *throttled, struct flb_search_bulk_retry **retry); void flb_search_bulk_retry_destroy(void *data); diff --git a/include/fluent-bit/flb_task.h b/include/fluent-bit/flb_task.h index 4ef8b7cd00e..a21d07f7098 100644 --- a/include/fluent-bit/flb_task.h +++ b/include/fluent-bit/flb_task.h @@ -28,6 +28,9 @@ #define FLB_TASK_NEW 0 #define FLB_TASK_RUNNING 1 +/* Core-only result: no output callback or processor was invoked. */ +#define FLB_OUTPUT_DEFERRED 4 + /* * Macro helpers to determinate return value, task_id and coro_id. When an * output plugin returns, it must call FLB_OUTPUT_RETURN(val) where val is @@ -55,8 +58,17 @@ #define FLB_TASK_ROUTE_ACTIVE 1 #define FLB_TASK_ROUTE_DROPPED 2 +/* Engine-owned route scheduling state (independent from delivery status). */ +#define FLB_TASK_ROUTE_DISPATCH_UNQUEUED 0 +#define FLB_TASK_ROUTE_DISPATCH_QUEUED 1 +#define FLB_TASK_ROUTE_DISPATCH_DEFERRED 2 +#define FLB_TASK_ROUTE_DISPATCH_COMPLETING 3 + +struct flb_task; + struct flb_task_route { int status; + int dispatch_state; int records; size_t bytes; void *retry_context; @@ -64,7 +76,9 @@ struct flb_task_route { int retry_records; size_t retry_bytes; struct flb_output_instance *out; + struct flb_task *task; struct mk_list _head; + struct mk_list _deferred_head; }; /* @@ -88,6 +102,7 @@ struct flb_task { uint64_t ref_id; /* external reference id */ uint8_t status; /* new task or running ? */ int users; /* number of users (threads) */ + int deferred_routes; /* routes retained by throttle scheduling */ struct flb_event_chunk *event_chunk; /* event chunk context */ void *ic; /* input chunk context */ #ifdef FLB_HAVE_METRICS @@ -152,12 +167,30 @@ struct flb_task_queue* flb_task_queue_create(); void flb_task_queue_destroy(struct flb_task_queue *queue); struct flb_task_retry *flb_task_retry_create(struct flb_task *task, struct flb_output_instance *ins); +struct flb_task_retry *flb_task_retry_get(struct flb_task *task, + struct flb_output_instance *ins); void flb_task_retry_destroy(struct flb_task_retry *retry); int flb_task_retry_reschedule(struct flb_task_retry *retry, struct flb_config *config); int flb_task_from_fs_storage(struct flb_task *task); int flb_task_retry_count(struct flb_task *task, void *data); int flb_task_retry_clean(struct flb_task *task, struct flb_output_instance *ins); +struct flb_task_route *flb_task_route_get(struct flb_task *task, + struct flb_output_instance *ins); +int flb_task_route_defer(struct flb_task *task, + struct flb_output_instance *ins, + int transfer_queued_owner); +int flb_task_route_resume(struct flb_task *task, + struct flb_output_instance *ins); +int flb_task_route_cancel_deferred(struct flb_task *task, + struct flb_output_instance *ins); +int flb_task_route_queue(struct flb_task *task, + struct flb_output_instance *ins); +int flb_task_route_complete(struct flb_task *task, + struct flb_output_instance *ins); +int flb_task_route_unqueue(struct flb_task *task, + struct flb_output_instance *ins); +void flb_output_deferred_cancel_all(struct flb_output_instance *ins); struct flb_task *flb_task_chunk_create(uint64_t ref_id, @@ -168,9 +201,15 @@ struct flb_task *flb_task_chunk_create(uint64_t ref_id, const char *tag_buf, int tag_len, struct flb_config *config); +static inline int flb_task_is_releasable(struct flb_task *task) +{ + return task->users == 0 && task->deferred_routes == 0 && + mk_list_size(&task->retries) == 0; +} + static inline void flb_task_users_release(struct flb_task *task) { - if (task->users == 0 && mk_list_size(&task->retries) == 0) { + if (flb_task_is_releasable(task) == FLB_TRUE) { flb_task_destroy(task, FLB_TRUE); } } diff --git a/include/fluent-bit/flb_utils.h b/include/fluent-bit/flb_utils.h index 0b499acaa4f..2f3c82b29a7 100644 --- a/include/fluent-bit/flb_utils.h +++ b/include/fluent-bit/flb_utils.h @@ -52,6 +52,7 @@ int64_t flb_utils_size_to_bytes(const char *size); int64_t flb_utils_size_to_binary_bytes(const char *size); int64_t flb_utils_hex2int(char *hex, int len); int flb_utils_time_to_seconds(const char *time); +int flb_utils_time_to_seconds_strict(const char *time, int *seconds); int flb_utils_pipe_byte_consume(flb_pipefd_t fd); int flb_utils_bool(const char *val); void flb_utils_bytes_to_human_readable_size(size_t bytes, diff --git a/plugins/out_es/THROTTLE_AUDIT.yaml b/plugins/out_es/THROTTLE_AUDIT.yaml new file mode 100644 index 00000000000..e7ebc6429d7 --- /dev/null +++ b/plugins/out_es/THROTTLE_AUDIT.yaml @@ -0,0 +1,35 @@ +plugin: es +baseline_commit: b0f5266d4f9d9271bf532a5c0c6ecbe4598eb6f4 +callback_and_send_functions: + - cb_es_flush + - flb_http_do +execution_model: direct_flush +success_means: mixed +throttle_signals: + - status_or_error_code: HTTP 429 response + scope: output_instance + primary_source: https://www.elastic.co/docs/troubleshoot/elasticsearch/rejected-requests + - status_or_error_code: bulk item status 429 + scope: output_instance + primary_source: https://www.elastic.co/guide/en/elasticsearch/reference/current/docs-bulk.html +retry_after_sources: + - HTTP Retry-After response header or trailer on a top-level 429 +local_retries_or_sleeps: [] +partial_acceptance_model: known_subset +retry_payload_owner: route_retry_context +cleanup_paths: + - success destroys request buffers and clears the retry context + - retry and throttle destroy request buffers while preserving the route retry context + - retry-context replacement failure destroys the newly allocated selective payload +changed_classification_cases: + - top-level HTTP 429 returns FLB_THROTTLE + - a valid bulk response containing an unacknowledged 429 item returns FLB_THROTTLE +unchanged_classification_cases: + - non-429 top-level failures return FLB_RETRY + - non-429 failed bulk items remain in the selective retry payload + - successful and acknowledged conflict items are excluded from retry payloads +new_tests: + - mixed bulk success, conflict, and 429 retains only the 429 item + - top-level 429 honors Retry-After and retries the complete bulk payload +unsupported_cases: + - item-level responses have no per-item Retry-After field in the Bulk API contract diff --git a/plugins/out_es/es.c b/plugins/out_es/es.c index f0d6ca31ae7..85cbccf0f31 100644 --- a/plugins/out_es/es.c +++ b/plugins/out_es/es.c @@ -33,6 +33,7 @@ #include #include #include +#include #include #include @@ -47,6 +48,65 @@ struct flb_output_plugin out_es_plugin; +static void es_apply_retry_after(struct flb_elasticsearch *ctx, + struct flb_http_client *client, + struct flb_output_flush *out_flush) +{ + int status; + int trailer_status; + time_t wall_time; + int64_t wall_time_ms; + uint64_t delay_ms; + uint64_t trailer_delay_ms; + size_t invalid_count; + size_t trailer_invalid_count; + + status = FLB_RETRY_AFTER_ABSENT; + delay_ms = 0; + invalid_count = 0; + wall_time = time(NULL); + if (wall_time < 0 || (uint64_t) wall_time > (uint64_t) INT64_MAX / 1000) { + wall_time_ms = 0; + } + else { + wall_time_ms = (int64_t) wall_time * 1000; + } + + if (client->resp.data != NULL && client->resp.headers_end != NULL) { + status = flb_http_retry_after_parse_headers( + client->resp.data, + (size_t) (client->resp.headers_end - client->resp.data), + wall_time_ms, &delay_ms, &invalid_count); + } + + if (client->resp.trailer_buf != NULL && client->resp.trailer_size > 0) { + trailer_delay_ms = 0; + trailer_invalid_count = 0; + trailer_status = flb_http_retry_after_parse_headers( + client->resp.trailer_buf, + client->resp.trailer_size, + wall_time_ms, &trailer_delay_ms, + &trailer_invalid_count); + invalid_count += trailer_invalid_count; + if ((trailer_status == FLB_RETRY_AFTER_VALID || + trailer_status == FLB_RETRY_AFTER_SATURATED) && + ((status != FLB_RETRY_AFTER_VALID && + status != FLB_RETRY_AFTER_SATURATED) || + trailer_delay_ms > delay_ms)) { + status = trailer_status; + delay_ms = trailer_delay_ms; + } + } + + if (invalid_count > 0) { + flb_plg_debug(ctx->ins, "ignored %zu malformed Retry-After field(s)", + invalid_count); + } + if (status == FLB_RETRY_AFTER_VALID || status == FLB_RETRY_AFTER_SATURATED) { + flb_output_set_retry_after(out_flush, delay_ms); + } +} + static int es_pack_array_content(msgpack_packer *tmp_pck, msgpack_object array, int replace_dots); @@ -1348,6 +1408,7 @@ static void cb_es_flush(struct flb_event_chunk *event_chunk, struct flb_search_bulk_retry *retry_payload; struct flb_search_bulk_retry *next_retry_payload; int compress_gzip; + int throttle_detected; size_t buffer_size; flb_sds_t header_line = NULL; flb_sds_t tmp_sds = NULL; @@ -1373,6 +1434,7 @@ static void cb_es_flush(struct flb_event_chunk *event_chunk, final_payload_buf = NULL; final_payload_size = 0; next_retry_payload = NULL; + throttle_detected = FLB_FALSE; node_ctx = NULL; if (ctx->ha_mode == FLB_TRUE) { @@ -1568,6 +1630,10 @@ static void cb_es_flush(struct flb_event_chunk *event_chunk, flb_plg_error(ctx->ins, "HTTP status=%i URI=%s", c->resp.status, uri); } + if (c->resp.status == 429) { + es_apply_retry_after(ctx, c, out_flush); + goto throttle; + } goto retry; } @@ -1579,6 +1645,7 @@ static void cb_es_flush(struct flb_event_chunk *event_chunk, FLB_SEARCH_BULK_ACK_CREATE_CONFLICTS, ctx->drop_unrecoverable_records, &bulk_stats, + &throttle_detected, &next_retry_payload); retry_records = 0; if (next_retry_payload != NULL) { @@ -1679,6 +1746,10 @@ static void cb_es_flush(struct flb_event_chunk *event_chunk, log_invalid_bulk_response(ctx, c->resp.payload, c->resp.payload_size); } + if (throttle_detected == FLB_TRUE) { + es_apply_retry_after(ctx, c, out_flush); + goto throttle; + } goto retry; } } @@ -1702,8 +1773,15 @@ static void cb_es_flush(struct flb_event_chunk *event_chunk, flb_sds_destroy(uri); FLB_OUTPUT_RETURN(FLB_OK); + throttle: + ret = FLB_THROTTLE; + goto failure; + /* Issue a retry */ retry: + ret = FLB_RETRY; + + failure: if (c != NULL) { flb_http_client_destroy(c); } @@ -1717,7 +1795,7 @@ static void cb_es_flush(struct flb_event_chunk *event_chunk, flb_sds_destroy(uri); flb_upstream_conn_release(u_conn); - FLB_OUTPUT_RETURN(FLB_RETRY); + FLB_OUTPUT_RETURN(ret); } static int elasticsearch_response_test(struct flb_config *config, diff --git a/plugins/out_http/http.c b/plugins/out_http/http.c index 79f7d9dfa1a..da7f71b0898 100644 --- a/plugins/out_http/http.c +++ b/plugins/out_http/http.c @@ -20,6 +20,7 @@ #include #include #include +#include #include #include #include @@ -112,12 +113,20 @@ static void append_headers(struct flb_http_client *c, static int http_request(struct flb_out_http *ctx, const void *body, size_t body_len, const char *tag, int tag_len, - char **headers) + char **headers, + struct flb_output_flush *out_flush) { int ret = 0; int out_ret = FLB_OK; int compressed = FLB_FALSE; + int retry_after_status; size_t b_sent; + size_t invalid_retry_after_count; + size_t parsed_invalid_count; + uint64_t retry_after_ms; + uint64_t parsed_retry_after_ms; + int64_t received_wall_time_ms; + time_t received_wall_time; void *payload_buf = NULL; size_t payload_size = 0; struct flb_upstream *u; @@ -303,6 +312,14 @@ static int http_request(struct flb_out_http *ctx, #endif ret = flb_http_do_with_oauth2(c, &b_sent, ctx->oauth2_ctx); + received_wall_time = time(NULL); + if (received_wall_time < 0 || + (uint64_t) received_wall_time > (uint64_t) INT64_MAX / 1000) { + received_wall_time_ms = 0; + } + else { + received_wall_time_ms = (int64_t) received_wall_time * 1000; + } if (ret == 0) { /* * Only allow the following HTTP status: @@ -326,7 +343,56 @@ static int http_request(struct flb_out_http *ctx, flb_plg_error(ctx->ins, "%s:%i, HTTP status=%i", ctx->host, ctx->port, c->resp.status); } - if (c->resp.status >= 400 && c->resp.status < 500 && + retry_after_status = FLB_RETRY_AFTER_ABSENT; + invalid_retry_after_count = 0; + retry_after_ms = 0; + + if ((c->resp.status == 429 || c->resp.status == 503) && + c->resp.data != NULL && c->resp.headers_end != NULL) { + retry_after_status = flb_http_retry_after_parse_headers( + c->resp.data, + (size_t) (c->resp.headers_end - c->resp.data), + received_wall_time_ms, + &retry_after_ms, + &invalid_retry_after_count); + + if (c->resp.trailer_buf != NULL && c->resp.trailer_size > 0) { + parsed_invalid_count = 0; + parsed_retry_after_ms = 0; + ret = flb_http_retry_after_parse_headers( + c->resp.trailer_buf, c->resp.trailer_size, + received_wall_time_ms, &parsed_retry_after_ms, + &parsed_invalid_count); + invalid_retry_after_count += parsed_invalid_count; + if ((ret == FLB_RETRY_AFTER_VALID || + ret == FLB_RETRY_AFTER_SATURATED) && + ((retry_after_status != FLB_RETRY_AFTER_VALID && + retry_after_status != FLB_RETRY_AFTER_SATURATED) || + parsed_retry_after_ms > retry_after_ms)) { + retry_after_status = ret; + retry_after_ms = parsed_retry_after_ms; + } + } + + if (invalid_retry_after_count > 0) { + flb_plg_debug(ctx->ins, + "ignored %zu malformed Retry-After field(s)", + invalid_retry_after_count); + } + } + + if (retry_after_status == FLB_RETRY_AFTER_VALID || + retry_after_status == FLB_RETRY_AFTER_SATURATED) { + flb_output_set_retry_after(out_flush, retry_after_ms); + } + + if (c->resp.status == 429 || + (c->resp.status == 503 && + (retry_after_status == FLB_RETRY_AFTER_VALID || + retry_after_status == FLB_RETRY_AFTER_SATURATED))) { + out_ret = FLB_THROTTLE; + } + else if (c->resp.status >= 400 && c->resp.status < 500 && c->resp.status != 429 && c->resp.status != 408) { flb_plg_warn(ctx->ins, "could not flush records to %s:%i (http_do=%i), " "chunk will not be retried", @@ -487,6 +553,7 @@ static int compose_payload(struct flb_out_http *ctx, static char **extract_headers(msgpack_object *obj) { size_t i; + size_t header_index; char **headers = NULL; size_t str_count; msgpack_object_map map; @@ -505,6 +572,7 @@ static char **extract_headers(msgpack_object *obj) { goto err; } + header_index = 0; for (i = 0; i < map.size; i++) { if (map.ptr[i].key.type != MSGPACK_OBJECT_STR || map.ptr[i].val.type != MSGPACK_OBJECT_STR) { @@ -514,17 +582,18 @@ static char **extract_headers(msgpack_object *obj) { k = map.ptr[i].key.via.str; v = map.ptr[i].val.via.str; - headers[i * 2] = strndup(k.ptr, k.size); + headers[header_index] = strndup(k.ptr, k.size); - if (!headers[i]) { + if (!headers[header_index]) { goto err; } - headers[i * 2 + 1] = strndup(v.ptr, v.size); + headers[header_index + 1] = strndup(v.ptr, v.size); - if (!headers[i]) { + if (!headers[header_index + 1]) { goto err; } + header_index += 2; } return headers; @@ -545,7 +614,8 @@ static int send_all_requests(struct flb_out_http *ctx, const char *data, size_t size, flb_sds_t body_key, flb_sds_t headers_key, - struct flb_event_chunk *event_chunk) + struct flb_event_chunk *event_chunk, + struct flb_output_flush *out_flush) { msgpack_object map; msgpack_object *k; @@ -614,7 +684,7 @@ static int send_all_requests(struct flb_out_http *ctx, record_count++, ctx->http_method == FLB_HTTP_POST ? "POST" : "PUT"); ret = http_request(ctx, body, body_size, event_chunk->tag, - flb_sds_len(event_chunk->tag), headers); + flb_sds_len(event_chunk->tag), headers, out_flush); } else { flb_plg_warn(ctx->ins, @@ -625,6 +695,10 @@ static int send_all_requests(struct flb_out_http *ctx, } flb_free(headers); + + if (ret == FLB_THROTTLE) { + break; + } } flb_log_event_decoder_destroy(&log_decoder); @@ -646,7 +720,8 @@ static void cb_http_flush(struct flb_event_chunk *event_chunk, if (ctx->body_key) { ret = send_all_requests(ctx, event_chunk->data, event_chunk->size, - ctx->body_key, ctx->headers_key, event_chunk); + ctx->body_key, ctx->headers_key, event_chunk, + out_flush); if (ret < 0) { flb_plg_error(ctx->ins, "failed to send requests using body key \"%s\"", ctx->body_key); @@ -664,14 +739,16 @@ static void cb_http_flush(struct flb_event_chunk *event_chunk, (ctx->out_format == FLB_PACK_JSON_FORMAT_LINES) || (ctx->out_format == FLB_HTTP_OUT_GELF)) { ret = http_request(ctx, out_body, out_size, - event_chunk->tag, flb_sds_len(event_chunk->tag), NULL); + event_chunk->tag, flb_sds_len(event_chunk->tag), + NULL, out_flush); flb_sds_destroy(out_body); } else { /* msgpack */ ret = http_request(ctx, event_chunk->data, event_chunk->size, - event_chunk->tag, flb_sds_len(event_chunk->tag), NULL); + event_chunk->tag, flb_sds_len(event_chunk->tag), + NULL, out_flush); } } diff --git a/plugins/out_opensearch/THROTTLE_AUDIT.yaml b/plugins/out_opensearch/THROTTLE_AUDIT.yaml new file mode 100644 index 00000000000..694ce520c62 --- /dev/null +++ b/plugins/out_opensearch/THROTTLE_AUDIT.yaml @@ -0,0 +1,35 @@ +plugin: opensearch +baseline_commit: b0f5266d4f9d9271bf532a5c0c6ecbe4598eb6f4 +callback_and_send_functions: + - cb_opensearch_flush + - flb_http_do +execution_model: direct_flush +success_means: mixed +throttle_signals: + - status_or_error_code: HTTP 429 response + scope: output_instance + primary_source: https://docs.opensearch.org/latest/api-reference/document-apis/bulk/ + - status_or_error_code: bulk item status 429 + scope: output_instance + primary_source: https://docs.opensearch.org/latest/api-reference/document-apis/bulk/ +retry_after_sources: + - HTTP Retry-After response header or trailer on a top-level 429 +local_retries_or_sleeps: [] +partial_acceptance_model: known_subset +retry_payload_owner: route_retry_context +cleanup_paths: + - success destroys request buffers and clears the retry context + - retry and throttle destroy request buffers while preserving the route retry context + - retry-context replacement failure destroys the newly allocated selective payload +changed_classification_cases: + - top-level HTTP 429 returns FLB_THROTTLE + - a valid bulk response containing an unacknowledged 429 item returns FLB_THROTTLE +unchanged_classification_cases: + - non-429 top-level failures return FLB_RETRY + - non-429 failed bulk items remain in the selective retry payload + - successful and acknowledged conflict items are excluded from retry payloads +new_tests: + - mixed bulk success, conflict, and 429 retains only the 429 item + - top-level 429 honors Retry-After and retries the complete bulk payload +unsupported_cases: + - item-level responses have no per-item Retry-After field in the Bulk API contract diff --git a/plugins/out_opensearch/opensearch.c b/plugins/out_opensearch/opensearch.c index 3ae23eb2f18..59ae18f845d 100644 --- a/plugins/out_opensearch/opensearch.c +++ b/plugins/out_opensearch/opensearch.c @@ -31,9 +31,12 @@ #include #include #include +#include #include #include +#include +#include #include "opensearch.h" #include "os_conf.h" @@ -167,6 +170,65 @@ static void log_bulk_failure_summary(struct flb_opensearch *ctx, } } +static void opensearch_apply_retry_after(struct flb_opensearch *ctx, + struct flb_http_client *client, + struct flb_output_flush *out_flush) +{ + int status; + int trailer_status; + time_t wall_time; + int64_t wall_time_ms; + uint64_t delay_ms; + uint64_t trailer_delay_ms; + size_t invalid_count; + size_t trailer_invalid_count; + + status = FLB_RETRY_AFTER_ABSENT; + delay_ms = 0; + invalid_count = 0; + wall_time = time(NULL); + if (wall_time < 0 || (uint64_t) wall_time > (uint64_t) INT64_MAX / 1000) { + wall_time_ms = 0; + } + else { + wall_time_ms = (int64_t) wall_time * 1000; + } + + if (client->resp.data != NULL && client->resp.headers_end != NULL) { + status = flb_http_retry_after_parse_headers( + client->resp.data, + (size_t) (client->resp.headers_end - client->resp.data), + wall_time_ms, &delay_ms, &invalid_count); + } + + if (client->resp.trailer_buf != NULL && client->resp.trailer_size > 0) { + trailer_delay_ms = 0; + trailer_invalid_count = 0; + trailer_status = flb_http_retry_after_parse_headers( + client->resp.trailer_buf, + client->resp.trailer_size, + wall_time_ms, &trailer_delay_ms, + &trailer_invalid_count); + invalid_count += trailer_invalid_count; + if ((trailer_status == FLB_RETRY_AFTER_VALID || + trailer_status == FLB_RETRY_AFTER_SATURATED) && + ((status != FLB_RETRY_AFTER_VALID && + status != FLB_RETRY_AFTER_SATURATED) || + trailer_delay_ms > delay_ms)) { + status = trailer_status; + delay_ms = trailer_delay_ms; + } + } + + if (invalid_count > 0) { + flb_plg_debug(ctx->ins, "ignored %zu malformed Retry-After field(s)", + invalid_count); + } + if (status == FLB_RETRY_AFTER_VALID || status == FLB_RETRY_AFTER_SATURATED) { + flb_output_set_retry_after(out_flush, delay_ms); + } +} + #ifdef FLB_HAVE_AWS static flb_sds_t add_aws_auth(struct flb_http_client *c, struct flb_opensearch *ctx) @@ -1080,6 +1142,7 @@ static void cb_opensearch_flush(struct flb_event_chunk *event_chunk, size_t successful_bytes; int successful_records; int retry_context_result; + int throttle_detected; /* Get upstream connection */ u_conn = flb_upstream_conn_get(ctx->u); @@ -1088,6 +1151,7 @@ static void cb_opensearch_flush(struct flb_event_chunk *event_chunk, } retry_payload = flb_output_get_retry_context(out_flush, NULL, NULL); + throttle_detected = FLB_FALSE; if (retry_payload != NULL) { pack = flb_sds_create_len(retry_payload->payload, retry_payload->size); @@ -1218,6 +1282,10 @@ static void cb_opensearch_flush(struct flb_event_chunk *event_chunk, flb_sds_destroy(signature); signature = NULL; } + if (c->resp.status == 429) { + opensearch_apply_retry_after(ctx, c, out_flush); + goto throttle; + } goto retry; } @@ -1231,6 +1299,7 @@ static void cb_opensearch_flush(struct flb_event_chunk *event_chunk, FLB_SEARCH_BULK_ACK_ALL_CONFLICTS, ctx->drop_unrecoverable_records, &bulk_stats, + &throttle_detected, &next_retry_payload); } else if (opensearch_error_check(ctx, c) == FLB_TRUE) { @@ -1324,6 +1393,10 @@ static void cb_opensearch_flush(struct flb_event_chunk *event_chunk, flb_sds_destroy(signature); signature = NULL; } + if (throttle_detected == FLB_TRUE) { + opensearch_apply_retry_after(ctx, c, out_flush); + goto throttle; + } goto retry; } else { @@ -1370,8 +1443,15 @@ static void cb_opensearch_flush(struct flb_event_chunk *event_chunk, } FLB_OUTPUT_RETURN(FLB_OK); + throttle: + ret = FLB_THROTTLE; + goto failure; + /* Issue a retry */ retry: + ret = FLB_RETRY; + + failure: if (c != NULL) { flb_http_client_destroy(c); } @@ -1382,7 +1462,7 @@ static void cb_opensearch_flush(struct flb_event_chunk *event_chunk, } flb_upstream_conn_release(u_conn); - FLB_OUTPUT_RETURN(FLB_RETRY); + FLB_OUTPUT_RETURN(ret); } static int cb_opensearch_exit(void *data, struct flb_config *config) diff --git a/plugins/out_opentelemetry/THROTTLE_AUDIT.yaml b/plugins/out_opentelemetry/THROTTLE_AUDIT.yaml new file mode 100644 index 00000000000..7581a57e58d --- /dev/null +++ b/plugins/out_opentelemetry/THROTTLE_AUDIT.yaml @@ -0,0 +1,55 @@ +plugin: opentelemetry +baseline_commit: b0f5266d4f9d9271bf532a5c0c6ecbe4598eb6f4 +callback_and_send_functions: + - cb_opentelemetry_flush + - otel_process_logs + - process_metrics + - process_traces + - process_profiles + - opentelemetry_legacy_post + - opentelemetry_post +execution_model: multi_request_flush +success_means: mixed +throttle_signals: + - status_or_error_code: OTLP/HTTP 429 + scope: output_instance + primary_source: https://opentelemetry.io/docs/specs/otlp/#otlphttp-throttling + - status_or_error_code: OTLP/HTTP 503 with valid Retry-After + scope: output_instance + primary_source: https://opentelemetry.io/docs/specs/otlp/#otlphttp-throttling + - status_or_error_code: OTLP/gRPC UNAVAILABLE with valid google.rpc.RetryInfo + scope: output_instance + primary_source: https://opentelemetry.io/docs/specs/otlp/#failures-1 + - status_or_error_code: OTLP/gRPC RESOURCE_EXHAUSTED with valid google.rpc.RetryInfo + scope: output_instance + primary_source: https://opentelemetry.io/docs/specs/otlp/#failures-1 +retry_after_sources: + - HTTP Retry-After response header or trailer + - google.rpc.RetryInfo in grpc-status-details-bin metadata +local_retries_or_sleeps: + - split metric batches stop after the first failed batch +partial_acceptance_model: known_subset +retry_payload_owner: core_chunk +cleanup_paths: + - legacy HTTP destroys the HTTP client and releases the upstream connection + - unified HTTP and gRPC destroy the request and its response stream + - split metric batches destroy every encoded batch after success, retry, or throttle +changed_classification_cases: + - HTTP 429 returns FLB_THROTTLE + - HTTP 503 with valid Retry-After returns FLB_THROTTLE + - gRPC UNAVAILABLE with valid RetryInfo returns FLB_THROTTLE + - gRPC RESOURCE_EXHAUSTED is retryable only with valid RetryInfo +unchanged_classification_cases: + - HTTP 502 and 504 return FLB_RETRY without closing the throttle gate + - HTTP 503 without a valid hint returns FLB_RETRY + - populated OTLP partial-success responses are acknowledged without retry + - a failure after an accepted split batch is not retried to avoid duplication +new_tests: + - HTTP/1 and HTTP/2 429 with Retry-After + - HTTP 502, 504, and 503 with invalid Retry-After remain ordinary retries + - gRPC UNAVAILABLE with valid RetryInfo + - gRPC RESOURCE_EXHAUSTED with absent and invalid RetryInfo + - populated log partial-success response is not retried +unsupported_cases: + - profiles are implemented but the integration fixture has no profiles service + - a throttle after an accepted split metric batch is acknowledged to preserve the no-replay contract diff --git a/plugins/out_opentelemetry/opentelemetry.c b/plugins/out_opentelemetry/opentelemetry.c index c670979b05b..d1d011f04c9 100644 --- a/plugins/out_opentelemetry/opentelemetry.c +++ b/plugins/out_opentelemetry/opentelemetry.c @@ -34,6 +34,8 @@ #include #include #include +#include +#include #include #include @@ -54,6 +56,16 @@ #include "opentelemetry_conf.h" #include "opentelemetry_utils.h" +#define OTLP_GRPC_STATUS_CANCELLED 1 +#define OTLP_GRPC_STATUS_DEADLINE_EXCEEDED 4 +#define OTLP_GRPC_STATUS_RESOURCE_EXHAUSTED 8 +#define OTLP_GRPC_STATUS_ABORTED 10 +#define OTLP_GRPC_STATUS_OUT_OF_RANGE 11 +#define OTLP_GRPC_STATUS_UNAVAILABLE 14 +#define OTLP_GRPC_STATUS_DATA_LOSS 15 +#define OTLP_GRPC_STATUS_UNAUTHENTICATED 16 +#define OTLP_GRPC_STATUS_DETAILS_MAX 65536 + static int is_http_status_code_retrayable(int http_code) { /* @@ -77,18 +89,297 @@ static int is_http_status_code_retrayable(int http_code) static int opentelemetry_is_grpc_status_retryable(int status_code) { - if (status_code == 1 || /* CANCELLED */ - status_code == 4 || /* DEADLINE_EXCEEDED */ - status_code == 8 || /* RESOURCE_EXHAUSTED */ - status_code == 10 || /* ABORTED */ - status_code == 13 || /* INTERNAL */ - status_code == 14) { /* UNAVAILABLE */ + if (status_code == OTLP_GRPC_STATUS_CANCELLED || + status_code == OTLP_GRPC_STATUS_DEADLINE_EXCEEDED || + status_code == OTLP_GRPC_STATUS_ABORTED || + status_code == OTLP_GRPC_STATUS_OUT_OF_RANGE || + status_code == OTLP_GRPC_STATUS_UNAVAILABLE || + status_code == OTLP_GRPC_STATUS_DATA_LOSS) { return FLB_TRUE; } return FLB_FALSE; } +static int protobuf_read_varint(const unsigned char *buffer, size_t size, + size_t *offset, uint64_t *value) +{ + int shift; + unsigned char byte; + uint64_t result; + + shift = 0; + result = 0; + while (*offset < size && shift < 64) { + byte = buffer[*offset]; + (*offset)++; + if (shift == 63 && (byte & 0xfe) != 0) { + return -1; + } + result |= ((uint64_t) (byte & 0x7f)) << shift; + if ((byte & 0x80) == 0) { + *value = result; + return 0; + } + shift += 7; + } + + return -1; +} + +static int protobuf_next_field(const unsigned char *buffer, size_t size, + size_t *offset, uint64_t *field_number, + uint64_t *wire_type, const unsigned char **value, + size_t *value_size, uint64_t *varint_value) +{ + uint64_t key; + uint64_t length; + + *value = NULL; + *value_size = 0; + *varint_value = 0; + if (protobuf_read_varint(buffer, size, offset, &key) != 0) { + return -1; + } + *field_number = key >> 3; + *wire_type = key & 7; + if (*field_number == 0) { + return -1; + } + + if (*wire_type == 0) { + return protobuf_read_varint(buffer, size, offset, varint_value); + } + if (*wire_type == 1) { + if (size - *offset < 8) { + return -1; + } + *offset += 8; + return 0; + } + if (*wire_type == 2) { + if (protobuf_read_varint(buffer, size, offset, &length) != 0 || + length > size - *offset) { + return -1; + } + *value = buffer + *offset; + *value_size = (size_t) length; + *offset += (size_t) length; + return 0; + } + if (*wire_type == 5) { + if (size - *offset < 4) { + return -1; + } + *offset += 4; + return 0; + } + + return -1; +} + +static int protobuf_retry_duration(const unsigned char *buffer, size_t size, + uint64_t *delay_ms) +{ + int seconds_found; + int nanos_found; + size_t offset; + size_t value_size; + uint64_t field_number; + uint64_t wire_type; + uint64_t value; + uint64_t seconds; + uint64_t nanos; + const unsigned char *bytes; + + offset = 0; + seconds = 0; + nanos = 0; + seconds_found = FLB_FALSE; + nanos_found = FLB_FALSE; + while (offset < size) { + if (protobuf_next_field(buffer, size, &offset, &field_number, + &wire_type, &bytes, &value_size, &value) != 0) { + return FLB_RETRY_AFTER_INVALID; + } + if (field_number == 1 && wire_type == 0) { + seconds = value; + seconds_found = FLB_TRUE; + } + else if (field_number == 2 && wire_type == 0) { + nanos = value; + nanos_found = FLB_TRUE; + } + } + + if (seconds_found == FLB_FALSE && nanos_found == FLB_FALSE) { + *delay_ms = 0; + return FLB_RETRY_AFTER_VALID; + } + if (seconds > INT64_MAX || nanos > 999999999) { + return FLB_RETRY_AFTER_INVALID; + } + if (seconds > (UINT64_MAX - 999) / 1000) { + *delay_ms = UINT64_MAX; + return FLB_RETRY_AFTER_SATURATED; + } + + *delay_ms = seconds * 1000 + (nanos + 999999) / 1000000; + return FLB_RETRY_AFTER_VALID; +} + +static int protobuf_retry_info(const unsigned char *buffer, size_t size, + uint64_t *delay_ms) +{ + size_t offset; + size_t value_size; + uint64_t field_number; + uint64_t wire_type; + uint64_t value; + const unsigned char *bytes; + + offset = 0; + while (offset < size) { + if (protobuf_next_field(buffer, size, &offset, &field_number, + &wire_type, &bytes, &value_size, &value) != 0) { + return FLB_RETRY_AFTER_INVALID; + } + if (field_number == 1 && wire_type == 2) { + return protobuf_retry_duration(bytes, value_size, delay_ms); + } + } + + return FLB_RETRY_AFTER_INVALID; +} + +static int protobuf_type_is_retry_info(const unsigned char *value, size_t size) +{ + static const char suffix[] = "/google.rpc.RetryInfo"; + + if (size < sizeof(suffix) - 1) { + return FLB_FALSE; + } + + return memcmp(value + size - (sizeof(suffix) - 1), + suffix, sizeof(suffix) - 1) == 0; +} + +static int protobuf_any_retry_info(const unsigned char *buffer, size_t size, + uint64_t *delay_ms) +{ + int retry_info_type; + size_t offset; + size_t value_size; + size_t retry_info_size; + uint64_t field_number; + uint64_t wire_type; + uint64_t value; + const unsigned char *bytes; + const unsigned char *retry_info; + + offset = 0; + retry_info = NULL; + retry_info_size = 0; + retry_info_type = FLB_FALSE; + while (offset < size) { + if (protobuf_next_field(buffer, size, &offset, &field_number, + &wire_type, &bytes, &value_size, &value) != 0) { + return FLB_RETRY_AFTER_INVALID; + } + if (field_number == 1 && wire_type == 2) { + retry_info_type = protobuf_type_is_retry_info(bytes, value_size); + } + else if (field_number == 2 && wire_type == 2) { + retry_info = bytes; + retry_info_size = value_size; + } + } + + if (retry_info_type == FLB_TRUE && retry_info != NULL) { + return protobuf_retry_info(retry_info, retry_info_size, delay_ms); + } + + return FLB_RETRY_AFTER_ABSENT; +} + +static int protobuf_status_retry_info(const unsigned char *buffer, size_t size, + uint64_t *delay_ms) +{ + int status; + size_t offset; + size_t value_size; + uint64_t field_number; + uint64_t wire_type; + uint64_t value; + const unsigned char *bytes; + + status = FLB_RETRY_AFTER_ABSENT; + offset = 0; + while (offset < size) { + if (protobuf_next_field(buffer, size, &offset, &field_number, + &wire_type, &bytes, &value_size, &value) != 0) { + return FLB_RETRY_AFTER_INVALID; + } + if (field_number == 3 && wire_type == 2) { + status = protobuf_any_retry_info(bytes, value_size, delay_ms); + if (status != FLB_RETRY_AFTER_ABSENT) { + return status; + } + } + } + + return status; +} + +static int opentelemetry_parse_grpc_retry_info(cfl_sds_t encoded, + uint64_t *delay_ms) +{ + int result; + size_t encoded_size; + size_t padded_size; + size_t padding_size; + size_t decoded_size; + unsigned char *padded; + unsigned char *decoded; + + if (encoded == NULL) { + return FLB_RETRY_AFTER_INVALID; + } + + encoded_size = cfl_sds_len(encoded); + if (encoded_size == 0 || encoded_size > OTLP_GRPC_STATUS_DETAILS_MAX || + encoded_size % 4 == 1) { + return FLB_RETRY_AFTER_INVALID; + } + + padding_size = (4 - encoded_size % 4) % 4; + padded_size = encoded_size + padding_size; + padded = flb_malloc(padded_size); + if (padded == NULL) { + return FLB_RETRY_AFTER_INVALID; + } + memcpy(padded, encoded, encoded_size); + memset(padded + encoded_size, '=', padding_size); + + decoded = flb_malloc(padded_size); + if (decoded == NULL) { + flb_free(padded); + return FLB_RETRY_AFTER_INVALID; + } + result = flb_base64_decode(decoded, padded_size, &decoded_size, + padded, padded_size); + flb_free(padded); + if (result == 0) { + result = protobuf_status_retry_info(decoded, decoded_size, delay_ms); + } + else { + result = FLB_RETRY_AFTER_INVALID; + } + flb_free(decoded); + + return result; +} + static int opentelemetry_lookup_header_value(struct flb_hash_table *table, const char *name, cfl_sds_t *out_value) @@ -120,18 +411,172 @@ static int opentelemetry_lookup_header_value(struct flb_hash_table *table, return FLB_TRUE; } +static int64_t opentelemetry_wall_time_ms(void) +{ + time_t wall_time; + + wall_time = time(NULL); + if (wall_time < 0 || (uint64_t) wall_time > (uint64_t) INT64_MAX / 1000) { + return 0; + } + + return (int64_t) wall_time * 1000; +} + +static int opentelemetry_parse_retry_after_value(cfl_sds_t value, + int64_t wall_time_ms, + uint64_t *delay_ms) +{ + if (value == NULL) { + return FLB_RETRY_AFTER_ABSENT; + } + + return flb_http_retry_after_parse(value, cfl_sds_len(value), + wall_time_ms, delay_ms); +} + +static int opentelemetry_apply_retry_after_response( + struct flb_http_response *response, + struct flb_output_flush *out_flush) +{ + int status; + int trailer_status; + int64_t wall_time_ms; + uint64_t delay_ms; + uint64_t trailer_delay_ms; + cfl_sds_t value; + + wall_time_ms = opentelemetry_wall_time_ms(); + delay_ms = 0; + trailer_delay_ms = 0; + value = NULL; + status = FLB_RETRY_AFTER_ABSENT; + if (opentelemetry_lookup_header_value(response->headers, + "retry-after", &value) == FLB_TRUE) { + status = opentelemetry_parse_retry_after_value(value, wall_time_ms, + &delay_ms); + cfl_sds_destroy(value); + } + + value = NULL; + trailer_status = FLB_RETRY_AFTER_ABSENT; + if (opentelemetry_lookup_header_value(response->trailer_headers, + "retry-after", &value) == FLB_TRUE) { + trailer_status = opentelemetry_parse_retry_after_value( + value, wall_time_ms, &trailer_delay_ms); + cfl_sds_destroy(value); + } + if ((trailer_status == FLB_RETRY_AFTER_VALID || + trailer_status == FLB_RETRY_AFTER_SATURATED) && + ((status != FLB_RETRY_AFTER_VALID && + status != FLB_RETRY_AFTER_SATURATED) || + trailer_delay_ms > delay_ms)) { + status = trailer_status; + delay_ms = trailer_delay_ms; + } + + if (status == FLB_RETRY_AFTER_VALID || status == FLB_RETRY_AFTER_SATURATED) { + flb_output_set_retry_after(out_flush, delay_ms); + } + + return status; +} + +static int opentelemetry_apply_legacy_retry_after( + struct flb_http_client *client, + struct flb_output_flush *out_flush) +{ + int status; + int trailer_status; + uint64_t delay_ms; + uint64_t trailer_delay_ms; + size_t invalid_count; + size_t trailer_invalid_count; + int64_t wall_time_ms; + + status = FLB_RETRY_AFTER_ABSENT; + delay_ms = 0; + invalid_count = 0; + wall_time_ms = opentelemetry_wall_time_ms(); + if (client->resp.data != NULL && client->resp.headers_end != NULL) { + status = flb_http_retry_after_parse_headers( + client->resp.data, + (size_t) (client->resp.headers_end - client->resp.data), + wall_time_ms, &delay_ms, &invalid_count); + } + + if (client->resp.trailer_buf != NULL && client->resp.trailer_size > 0) { + trailer_delay_ms = 0; + trailer_invalid_count = 0; + trailer_status = flb_http_retry_after_parse_headers( + client->resp.trailer_buf, + client->resp.trailer_size, + wall_time_ms, &trailer_delay_ms, + &trailer_invalid_count); + if ((trailer_status == FLB_RETRY_AFTER_VALID || + trailer_status == FLB_RETRY_AFTER_SATURATED) && + ((status != FLB_RETRY_AFTER_VALID && + status != FLB_RETRY_AFTER_SATURATED) || + trailer_delay_ms > delay_ms)) { + status = trailer_status; + delay_ms = trailer_delay_ms; + } + } + + if (status == FLB_RETRY_AFTER_VALID || status == FLB_RETRY_AFTER_SATURATED) { + flb_output_set_retry_after(out_flush, delay_ms); + } + + return status; +} + +static int opentelemetry_apply_grpc_retry_info( + struct flb_http_response *response, + struct flb_output_flush *out_flush) +{ + int status; + uint64_t delay_ms; + cfl_sds_t value; + + value = NULL; + if (opentelemetry_lookup_header_value(response->trailer_headers, + "grpc-status-details-bin", + &value) == FLB_FALSE) { + opentelemetry_lookup_header_value(response->headers, + "grpc-status-details-bin", + &value); + } + if (value == NULL) { + return FLB_RETRY_AFTER_ABSENT; + } + + delay_ms = 0; + status = opentelemetry_parse_grpc_retry_info(value, &delay_ms); + cfl_sds_destroy(value); + if (status == FLB_RETRY_AFTER_VALID || status == FLB_RETRY_AFTER_SATURATED) { + flb_output_set_retry_after(out_flush, delay_ms); + } + + return status; +} + static int opentelemetry_check_grpc_status(struct opentelemetry_context *ctx, - struct flb_http_response *response) + struct flb_http_response *response, + struct flb_output_flush *out_flush) { cfl_sds_t grpc_message; cfl_sds_t grpc_status_text; + char *grpc_status_end; + long parsed_grpc_status; int grpc_status; int result; + int retry_info_status; grpc_message = NULL; grpc_status_text = NULL; grpc_status = 0; result = FLB_OK; + retry_info_status = FLB_RETRY_AFTER_ABSENT; /* ref: https://grpc.io/docs/guides/status-codes/ */ if (opentelemetry_lookup_header_value(response->trailer_headers, @@ -144,7 +589,14 @@ static int opentelemetry_check_grpc_status(struct opentelemetry_context *ctx, return FLB_OK; } - grpc_status = strtol(grpc_status_text, NULL, 10); + grpc_status_end = NULL; + parsed_grpc_status = strtol(grpc_status_text, &grpc_status_end, 10); + if (grpc_status_end == grpc_status_text || *grpc_status_end != '\0' || + parsed_grpc_status < 0 || parsed_grpc_status > 16) { + cfl_sds_destroy(grpc_status_text); + return FLB_ERROR; + } + grpc_status = (int) parsed_grpc_status; if (opentelemetry_lookup_header_value(response->trailer_headers, "grpc-message", @@ -165,10 +617,26 @@ static int opentelemetry_check_grpc_status(struct opentelemetry_context *ctx, flb_plg_error(ctx->ins, "grpc-status=%d", grpc_status); } - if (grpc_status == 16 && ctx->oauth2_ctx != NULL) { + if (grpc_status == OTLP_GRPC_STATUS_UNAUTHENTICATED && + ctx->oauth2_ctx != NULL) { flb_oauth2_invalidate_token(ctx->oauth2_ctx); result = FLB_RETRY; } + else if (grpc_status == OTLP_GRPC_STATUS_RESOURCE_EXHAUSTED || + grpc_status == OTLP_GRPC_STATUS_UNAVAILABLE) { + retry_info_status = opentelemetry_apply_grpc_retry_info(response, + out_flush); + if (retry_info_status == FLB_RETRY_AFTER_VALID || + retry_info_status == FLB_RETRY_AFTER_SATURATED) { + result = FLB_THROTTLE; + } + else if (grpc_status == OTLP_GRPC_STATUS_RESOURCE_EXHAUSTED) { + result = FLB_ERROR; + } + else { + result = FLB_RETRY; + } + } else if (opentelemetry_is_grpc_status_retryable(grpc_status)) { result = FLB_RETRY; } @@ -191,7 +659,8 @@ static int opentelemetry_check_grpc_status(struct opentelemetry_context *ctx, int opentelemetry_legacy_post(struct opentelemetry_context *ctx, const void *body, size_t body_len, const char *tag, int tag_len, - const char *uri) + const char *uri, + struct flb_output_flush *out_flush) { size_t final_body_len; void *final_body; @@ -206,6 +675,7 @@ int opentelemetry_legacy_post(struct opentelemetry_context *ctx, struct flb_config_map_val *mv; struct flb_http_client *c; flb_sds_t signature = NULL; + int retry_after_status; compressed = FLB_FALSE; @@ -368,7 +838,18 @@ int opentelemetry_legacy_post(struct opentelemetry_context *ctx, } /* Retryable status codes according to OTLP spec */ - if (is_http_status_code_retrayable(c->resp.status) == FLB_TRUE) { + retry_after_status = FLB_RETRY_AFTER_ABSENT; + if (c->resp.status == 429 || c->resp.status == 503) { + retry_after_status = opentelemetry_apply_legacy_retry_after( + c, out_flush); + } + if (c->resp.status == 429 || + (c->resp.status == 503 && + (retry_after_status == FLB_RETRY_AFTER_VALID || + retry_after_status == FLB_RETRY_AFTER_SATURATED))) { + out_ret = FLB_THROTTLE; + } + else if (is_http_status_code_retrayable(c->resp.status) == FLB_TRUE) { out_ret = FLB_RETRY; } else { @@ -417,7 +898,8 @@ int opentelemetry_post(struct opentelemetry_context *ctx, const void *body, size_t body_len, const char *tag, int tag_len, const char *http_uri, - const char *grpc_uri) + const char *grpc_uri, + struct flb_output_flush *out_flush) { flb_sds_t oauth2_token; const char *compression_algorithm; @@ -429,6 +911,7 @@ int opentelemetry_post(struct opentelemetry_context *ctx, struct flb_http_request *request; int out_ret = FLB_RETRY; int result; + int retry_after_status; oauth2_token = NULL; @@ -436,7 +919,7 @@ int opentelemetry_post(struct opentelemetry_context *ctx, return opentelemetry_legacy_post(ctx, body, body_len, tag, tag_len, - http_uri); + http_uri, out_flush); } compression_algorithm = NULL; @@ -649,14 +1132,27 @@ int opentelemetry_post(struct opentelemetry_context *ctx, response->status); } - if (out_ret == FLB_RETRY) { + if (ctx->oauth2_ctx != NULL && response->status == 401) { /* OAuth2-authenticated 401s should be retried with a fresh token. */ } - else if (is_http_status_code_retrayable(response->status) == FLB_TRUE) { - out_ret = FLB_RETRY; - } else { - out_ret = FLB_ERROR; + retry_after_status = FLB_RETRY_AFTER_ABSENT; + if (response->status == 429 || response->status == 503) { + retry_after_status = opentelemetry_apply_retry_after_response( + response, out_flush); + } + if (response->status == 429 || + (response->status == 503 && + (retry_after_status == FLB_RETRY_AFTER_VALID || + retry_after_status == FLB_RETRY_AFTER_SATURATED))) { + out_ret = FLB_THROTTLE; + } + else if (is_http_status_code_retrayable(response->status) == FLB_TRUE) { + out_ret = FLB_RETRY; + } + else { + out_ret = FLB_ERROR; + } } } else { @@ -677,7 +1173,7 @@ int opentelemetry_post(struct opentelemetry_context *ctx, } if (ctx->enable_grpc_flag && request->protocol_version == HTTP_PROTOCOL_VERSION_20 && out_ret == FLB_OK) { - result = opentelemetry_check_grpc_status(ctx, response); + result = opentelemetry_check_grpc_status(ctx, response, out_flush); if (result != FLB_OK) { out_ret = result; } @@ -730,7 +1226,8 @@ static int opentelemetry_format_test(struct flb_config *config, static int post_metrics_payload(struct opentelemetry_context *ctx, struct flb_event_chunk *event_chunk, - flb_sds_t payload) + flb_sds_t payload, + struct flb_output_flush *out_flush) { int result; int split_result; @@ -744,7 +1241,8 @@ static int post_metrics_payload(struct opentelemetry_context *ctx, event_chunk->tag, flb_sds_len(event_chunk->tag), ctx->metrics_uri_sanitized, - ctx->grpc_metrics_uri); + ctx->grpc_metrics_uri, + out_flush); } batches = cmt_encode_opentelemetry_split_payload( @@ -770,7 +1268,8 @@ static int post_metrics_payload(struct opentelemetry_context *ctx, event_chunk->tag, flb_sds_len(event_chunk->tag), ctx->metrics_uri_sanitized, - ctx->grpc_metrics_uri); + ctx->grpc_metrics_uri, + out_flush); if (result != FLB_OK) { if (result == FLB_RETRY && index > 0) { flb_plg_warn(ctx->ins, @@ -855,7 +1354,7 @@ static int process_metrics(struct flb_event_chunk *event_chunk, flb_plg_debug(ctx->ins, "final payload size: %lu", flb_sds_len(buf)); if (buf && flb_sds_len(buf) > 0) { /* Send HTTP request */ - result = post_metrics_payload(ctx, event_chunk, buf); + result = post_metrics_payload(ctx, event_chunk, buf, out1_flush); /* Debug http_post() result statuses */ if (result == FLB_OK) { @@ -946,7 +1445,8 @@ static int process_traces(struct flb_event_chunk *event_chunk, event_chunk->tag, flb_sds_len(event_chunk->tag), ctx->traces_uri_sanitized, - ctx->grpc_traces_uri); + ctx->grpc_traces_uri, + out_flush); /* Debug http_post() result statuses */ if (result == FLB_OK) { @@ -1028,7 +1528,8 @@ static int process_profiles(struct flb_event_chunk *event_chunk, event_chunk->tag, flb_sds_len(event_chunk->tag), ctx->profiles_uri_sanitized, - ctx->grpc_profiles_uri); + ctx->grpc_profiles_uri, + out_flush); /* Debug http_post() result statuses */ if (result == FLB_OK) { diff --git a/plugins/out_opentelemetry/opentelemetry.h b/plugins/out_opentelemetry/opentelemetry.h index bfec5824d54..53663d394ac 100644 --- a/plugins/out_opentelemetry/opentelemetry.h +++ b/plugins/out_opentelemetry/opentelemetry.h @@ -221,5 +221,6 @@ int opentelemetry_post(struct opentelemetry_context *ctx, const void *body, size_t body_len, const char *tag, int tag_len, const char *http_uri, - const char *grpc_uri); + const char *grpc_uri, + struct flb_output_flush *out_flush); #endif diff --git a/plugins/out_opentelemetry/opentelemetry_logs.c b/plugins/out_opentelemetry/opentelemetry_logs.c index 2185ed4fa2a..8b320c446e7 100644 --- a/plugins/out_opentelemetry/opentelemetry_logs.c +++ b/plugins/out_opentelemetry/opentelemetry_logs.c @@ -916,7 +916,9 @@ static void free_resource_logs(Opentelemetry__Proto__Logs__V1__ResourceLogs **re flb_free(resource_logs); } -static int logs_flush_to_otel(struct opentelemetry_context *ctx, struct flb_event_chunk *event_chunk, +static int logs_flush_to_otel(struct opentelemetry_context *ctx, + struct flb_event_chunk *event_chunk, + struct flb_output_flush *out_flush, Opentelemetry__Proto__Collector__Logs__V1__ExportLogsServiceRequest *export_logs) { int ret; @@ -941,7 +943,8 @@ static int logs_flush_to_otel(struct opentelemetry_context *ctx, struct flb_even event_chunk->tag, flb_sds_len(event_chunk->tag), ctx->logs_uri_sanitized, - ctx->grpc_logs_uri); + ctx->grpc_logs_uri, + out_flush); flb_free(body); return ret; @@ -1568,7 +1571,7 @@ int otel_process_logs(struct flb_event_chunk *event_chunk, scope_log->n_log_records = log_record_count; if (log_record_count >= ctx->batch_size) { - ret = logs_flush_to_otel(ctx, event_chunk, &export_logs); + ret = logs_flush_to_otel(ctx, event_chunk, out_flush, &export_logs); free_log_records(log_records, log_record_count); log_record_count = 0; scope_log->n_log_records = 0; @@ -1583,7 +1586,7 @@ int otel_process_logs(struct flb_event_chunk *event_chunk, flb_log_event_decoder_destroy(decoder); if (log_record_count > 0 && ret == FLB_OK) { - ret = logs_flush_to_otel(ctx, event_chunk, &export_logs); + ret = logs_flush_to_otel(ctx, event_chunk, out_flush, &export_logs); } /* release all protobuf resources */ diff --git a/src/CMakeLists.txt b/src/CMakeLists.txt index 6e11193d78c..de6af508a2d 100644 --- a/src/CMakeLists.txt +++ b/src/CMakeLists.txt @@ -36,6 +36,7 @@ set(src flb_input_thread.c flb_filter.c flb_output.c + flb_output_throttle.c flb_output_thread.c flb_config.c flb_fips.c @@ -79,6 +80,7 @@ set(src flb_http_client_http1.c flb_http_client_http2.c flb_http_client.c + flb_http_retry_after.c flb_callback.c flb_strptime.c flb_fstore.c diff --git a/src/flb_engine.c b/src/flb_engine.c index 09ca590756e..b995319ff18 100644 --- a/src/flb_engine.c +++ b/src/flb_engine.c @@ -19,6 +19,7 @@ #include #include +#include #include #include #include @@ -147,6 +148,7 @@ void flb_engine_reschedule_retries(struct flb_config *config) struct mk_list *tmp_task; struct mk_list *tmp_retry_task; struct flb_task *task; + struct flb_task_route *route; struct flb_input_instance *ins; struct flb_task_retry *retry; @@ -166,6 +168,12 @@ void flb_engine_reschedule_retries(struct flb_config *config) mk_list_foreach_safe(rt_head, tmp_retry_task, &task->retries) { retry = mk_list_entry(rt_head, struct flb_task_retry, _head); + route = flb_task_route_get(task, retry->o_ins); + if (route != NULL && + route->dispatch_state == FLB_TASK_ROUTE_DISPATCH_DEFERRED) { + flb_output_throttle_wakeup_schedule(retry->o_ins); + continue; + } flb_sched_request_invalidate(config, retry); ret = flb_sched_retry_now(config, retry); if (ret == -1) { @@ -552,6 +560,8 @@ static inline int handle_output_event(uint64_t ts, int effective_records = 0; int retries; int retry_seconds; + uint64_t now_ms; + uint64_t remaining_ms; uint32_t type; uint32_t key; double latency_seconds; @@ -561,6 +571,7 @@ static inline int handle_output_event(uint64_t ts, struct flb_task *task; struct flb_task_retry *retry; struct flb_output_instance *ins; + struct flb_output_throttle_snapshot throttle_snapshot; /* Get type and key */ type = FLB_BITS_U64_HIGH(val); @@ -592,6 +603,12 @@ static inline int handle_output_event(uint64_t ts, else if (ret == FLB_RETRY) { trace_st = "RETRY"; } + else if (ret == FLB_THROTTLE) { + trace_st = "THROTTLE"; + } + else if (ret == FLB_OUTPUT_DEFERRED) { + trace_st = "DEFERRED"; + } flb_trace("%s[engine] [task event]%s task_id=%i out_id=%i return=%s", ANSI_YELLOW, ANSI_RESET, @@ -600,11 +617,34 @@ static inline int handle_output_event(uint64_t ts, task = config->task_map[task_id].task; ins = flb_output_get_instance(config, out_id); + if (task == NULL || ins == NULL) { + flb_error("[engine] stale output event task_id=%i out_id=%i", + task_id, out_id); + return -1; + } + + if (ret == FLB_OUTPUT_DEFERRED) { + if (flb_task_route_defer(task, ins, FLB_TRUE) == -1) { + flb_error("[engine] could not transfer task_id=%i output=%s " + "to deferred ownership", task_id, flb_output_name(ins)); + return -1; + } + + if (ins->flags & FLB_OUTPUT_SYNCHRONOUS) { + flb_output_task_singleplex_complete(ins->singleplex_queue); + flb_output_task_singleplex_flush_next(ins->singleplex_queue); + } + flb_output_throttle_wakeup_schedule(ins); + flb_output_throttle_metrics_update(ins, flb_output_throttle_now_ms()); + return 0; + } + if (flb_output_is_threaded(ins) == FLB_FALSE) { flb_output_flush_finished(config, out_id); } in_name = (char *) flb_input_name(task->i_ins); out_name = (char *) flb_output_name(ins); + flb_output_throttle_metrics_update(ins, flb_output_throttle_now_ms()); flb_task_acquire_lock(task); if (flb_task_get_route_data(task, ins, &effective_records, @@ -614,9 +654,23 @@ static inline int handle_output_event(uint64_t ts, } flb_task_release_lock(task); + flb_task_acquire_lock(task); + if (flb_task_route_unqueue(task, ins) == -1) { + flb_warn("[engine] task_id=%i output=%s completed from an unexpected " + "dispatch state", task_id, flb_output_name(ins)); + } + flb_task_release_lock(task); + + if ((ins->flags & FLB_OUTPUT_NO_MULTIPLEX) && + ins->throttle_deferred_count > 0) { + flb_output_throttle_wakeup_schedule(ins); + } + /* If we are in synchronous mode, flush the next waiting task */ if (ins->flags & FLB_OUTPUT_SYNCHRONOUS) { - if (ret == FLB_OK || ret == FLB_RETRY || ret == FLB_ERROR) { + if (ret == FLB_OK || ret == FLB_RETRY || ret == FLB_ERROR || + ret == FLB_THROTTLE) { + flb_output_task_singleplex_complete(ins->singleplex_queue); flb_output_task_singleplex_flush_next(ins->singleplex_queue); } } @@ -686,7 +740,7 @@ static inline int handle_output_event(uint64_t ts, flb_task_retry_clean(task, ins); flb_task_users_dec(task, FLB_TRUE); } - else if (ret == FLB_RETRY) { + else if (ret == FLB_RETRY || ret == FLB_THROTTLE) { if (ins->retry_limit == FLB_OUT_RETRY_NONE) { handle_dlq_if_available(config, task, ins, 0); @@ -726,6 +780,42 @@ static inline int handle_output_event(uint64_t ts, return 0; } + if (ret == FLB_THROTTLE) { + if (flb_task_route_defer(task, ins, FLB_FALSE) == -1) { + flb_task_users_dec(task, FLB_TRUE); + return -1; + } + + /* Transfer callback ownership to the output deadline queue. */ + flb_task_users_dec(task, FLB_FALSE); + flb_output_throttle_snapshot(&ins->throttle, + &throttle_snapshot); + now_ms = flb_output_throttle_now_ms(); + if (throttle_snapshot.until_ms > now_ms) { + remaining_ms = throttle_snapshot.until_ms - now_ms; + } + else { + remaining_ms = 1; + } + + if (remaining_ms > (uint64_t) INT_MAX * 1000) { + retry_seconds = INT_MAX; + } + else { + retry_seconds = (int) ((remaining_ms + 999) / 1000); + } + flb_output_throttle_wakeup_schedule(ins); + + flb_warn("[engine] throttled flush for chunk '%s', resume in %i seconds: " + "task_id=%i, input=%s > output=%s (out_id=%i)", + flb_input_chunk_get_name(task->ic), + retry_seconds, + task->id, + flb_input_name(task->i_ins), + flb_output_name(ins), out_id); + return 0; + } + /* Create a Task-Retry */ retry = flb_task_retry_create(task, ins); if (!retry) { @@ -867,6 +957,34 @@ static inline int handle_output_event(uint64_t ts, return 0; } +static inline void handle_dispatch_result_fallback( + struct flb_output_dispatch *dispatch, + int result, + struct flb_config *config) +{ + uint32_t set; + uint64_t value; + + set = FLB_TASK_SET(result, dispatch->task->id, dispatch->out->id); + value = FLB_BITS_U64_SET(FLB_ENGINE_TASK, set); + flb_output_dispatch_destroy(dispatch); + handle_output_event(cfl_time_now(), config, value); +} + +static inline void handle_dispatch_result_fallbacks(struct flb_config *config) +{ + struct flb_output_dispatch *dispatch; + + while (1) { + dispatch = flb_output_thread_result_fallback_pop(config); + if (dispatch == NULL) { + break; + } + + handle_dispatch_result_fallback(dispatch, dispatch->result, config); + } +} + static inline int handle_output_events(flb_pipefd_t fd, struct flb_config *config) { @@ -879,7 +997,11 @@ static inline int handle_output_events(flb_pipefd_t fd, memset(&values, 0, sizeof(values)); +#ifdef _WIN32 + bytes = flb_pipe_read_all(fd, &values[0], sizeof(values[0])); +#else bytes = flb_pipe_r(fd, &values, sizeof(values)); +#endif if (bytes == -1) { flb_pipe_error(); @@ -1538,15 +1660,84 @@ int flb_engine_start(struct flb_config *config) flb_sched_event_handler(config, event); } else if (event->type == FLB_ENGINE_EV_THREAD_ENGINE) { + uint64_t generation; + size_t route_status; struct flb_output_flush *output_flush; + struct flb_output_dispatch *dispatch; - /* Read the coroutine reference */ - ret = flb_pipe_r(event->fd, &output_flush, sizeof(struct flb_output_flush *)); - if (ret <= 0 || output_flush == 0) { + ret = flb_pipe_read_all(event->fd, &dispatch, + sizeof(struct flb_output_dispatch *)); + if (ret != sizeof(struct flb_output_dispatch *) || + dispatch == NULL) { flb_pipe_error(); continue; } + if (dispatch->magic != FLB_OUTPUT_DISPATCH_MAGIC || + dispatch->config != config) { + flb_error("[engine] invalid output dispatch envelope"); + flb_output_dispatch_destroy(dispatch); + continue; + } + + if (dispatch->type != FLB_OUTPUT_DISPATCH_TASK) { + flb_error("[engine] invalid output dispatch envelope type"); + flb_output_dispatch_destroy(dispatch); + continue; + } + + flb_task_acquire_lock(dispatch->task); + route_status = flb_task_get_route_status(dispatch->task, + dispatch->out); + flb_task_release_lock(dispatch->task); + if (route_status == FLB_TASK_ROUTE_DROPPED) { + ret = flb_output_dispatch_post_result( + dispatch, FLB_ERROR, + dispatch->out->ch_events[1]); + if (ret == -1) { + handle_dispatch_result_fallback( + dispatch, FLB_ERROR, config); + continue; + } + flb_output_dispatch_destroy(dispatch); + continue; + } + + if (flb_output_throttle_admit(&dispatch->out->throttle, + flb_output_throttle_now_ms(), + &generation) == FLB_FALSE) { + ret = flb_output_dispatch_post_result( + dispatch, FLB_OUTPUT_DEFERRED, + dispatch->out->ch_events[1]); + if (ret == -1) { + handle_dispatch_result_fallback( + dispatch, FLB_OUTPUT_DEFERRED, config); + continue; + } + flb_output_dispatch_destroy(dispatch); + continue; + } + + output_flush = flb_output_flush_create(dispatch->task, + dispatch->task->i_ins, + dispatch->out, + dispatch->config); + if (output_flush == NULL) { + ret = flb_output_dispatch_post_result( + dispatch, FLB_ERROR, + dispatch->out->ch_events[1]); + if (ret == -1) { + handle_dispatch_result_fallback( + dispatch, FLB_ERROR, config); + continue; + } + flb_output_dispatch_destroy(dispatch); + continue; + } + + output_flush->admission_generation = generation; + flb_output_dispatch_destroy(dispatch); + /* Init coroutine */ flb_coro_resume(output_flush->coro); } @@ -1605,8 +1796,11 @@ int flb_engine_start(struct flb_config *config) flb_input_chunk_ring_buffer_collector(config, NULL); } + handle_dispatch_result_fallbacks(config); + /* Cleanup functions associated to events and timers */ if (config->is_running == FLB_TRUE) { + flb_output_throttle_wakeup_scan(config); flb_net_dns_lookup_context_cleanup(&dns_ctx); flb_sched_timer_cleanup(config->sched); flb_upstream_conn_pending_destroy_list(&config->upstreams); diff --git a/src/flb_engine_dispatch.c b/src/flb_engine_dispatch.c index dee466ea012..1259cc1d566 100644 --- a/src/flb_engine_dispatch.c +++ b/src/flb_engine_dispatch.c @@ -87,12 +87,35 @@ int flb_engine_dispatch_retry(struct flb_task_retry *retry, struct flb_config *config) { int ret; + int transfer_queued_owner; + uint64_t generation; char *buf_data; size_t buf_size; struct flb_task *task; + struct flb_task_route *route; struct flb_output_instance *ins; task = retry->parent; + ins = retry->o_ins; + route = flb_task_route_get(task, ins); + if (route == NULL) { + return -1; + } + + transfer_queued_owner = + route->dispatch_state == FLB_TASK_ROUTE_DISPATCH_QUEUED; + + /* A due ordinary retry must still pass the output deadline gate. */ + if (flb_output_throttle_admit(&ins->throttle, + flb_output_throttle_now_ms(), + &generation) == FLB_FALSE) { + ret = flb_task_route_defer(task, ins, transfer_queued_owner); + if (ret == -1) { + return -1; + } + flb_output_throttle_wakeup_schedule(ins); + return 0; + } /* Set file up/down based on restrictions */ ret = flb_input_chunk_set_up(task->ic); @@ -105,6 +128,10 @@ int flb_engine_dispatch_retry(struct flb_task_retry *retry, * enough like errors on delivering data. So if we cannot put the chunk in memory * it cannot be retried. */ + if (transfer_queued_owner == FLB_TRUE) { + flb_task_route_unqueue(task, ins); + flb_task_users_dec(task, FLB_FALSE); + } ret = flb_task_retry_reschedule(retry, config); if (ret == -1) { return -1; @@ -125,13 +152,15 @@ int flb_engine_dispatch_retry(struct flb_task_retry *retry, * Destroying the retry without releasing the task would leave the task * with no users and no retries, a state nothing reaps. */ - ins = retry->o_ins; - if (retry->attempts >= ins->retry_limit && ins->retry_limit >= 0) { flb_error("[engine_dispatch] could not retrieve chunk content, " "task_id=%i reached retry-attempts limit %i/%i, dropping", task->id, retry->attempts, ins->retry_limit); record_retry_failure_metrics(task, ins, config); + if (transfer_queued_owner == FLB_TRUE) { + flb_task_route_unqueue(task, ins); + flb_task_users_dec(task, FLB_FALSE); + } flb_task_retry_destroy(retry); flb_task_users_release(task); return -1; @@ -142,6 +171,10 @@ int flb_engine_dispatch_retry(struct flb_task_retry *retry, "re-scheduling task_id=%i attempts=%i", task->id, retry->attempts); + if (transfer_queued_owner == FLB_TRUE) { + flb_task_route_unqueue(task, ins); + flb_task_users_dec(task, FLB_FALSE); + } ret = flb_task_retry_reschedule(retry, config); if (ret == -1) { return -1; @@ -163,6 +196,12 @@ int flb_engine_dispatch_retry(struct flb_task_retry *retry, ret = flb_output_task_singleplex_enqueue(retry->o_ins->singleplex_queue, retry, task, retry->o_ins, config); if (ret == -1) { + if (transfer_queued_owner == FLB_TRUE && + route->dispatch_state == FLB_TASK_ROUTE_DISPATCH_QUEUED) { + flb_task_route_unqueue(task, ins); + flb_task_users_dec(task, FLB_FALSE); + } + flb_task_users_release(task); return -1; } } @@ -170,6 +209,7 @@ int flb_engine_dispatch_retry(struct flb_task_retry *retry, ret = flb_output_task_flush(task, retry->o_ins, config); if (ret == -1) { flb_task_retry_destroy(retry); + flb_task_users_release(task); return -1; } } @@ -224,6 +264,7 @@ static void test_run_formatter(struct flb_config *config, static int tasks_start(struct flb_input_instance *in, struct flb_config *config) { + int ret; int hits = 0; int retry = 0; struct mk_list *tmp; @@ -274,7 +315,19 @@ static int tasks_start(struct flb_input_instance *in, * running something. */ if (out->flags & FLB_OUTPUT_NO_MULTIPLEX) { - if (flb_output_coros_size(route->out) > 0 || retry > 0) { + if (out->throttle_deferred_count > 0) { + /* Preserve deferred FIFO priority and task ownership. */ + ret = flb_task_route_defer(task, out, FLB_FALSE); + if (ret == 0) { + hits++; + flb_output_throttle_wakeup_schedule(out); + flb_output_throttle_metrics_update( + out, flb_output_throttle_now_ms()); + } + continue; + } + + if (out->dispatches_inflight > 0 || retry > 0) { continue; } } diff --git a/src/flb_http_retry_after.c b/src/flb_http_retry_after.c new file mode 100644 index 00000000000..1f526c22e3c --- /dev/null +++ b/src/flb_http_retry_after.c @@ -0,0 +1,446 @@ +/* -*- Mode: C; tab-width: 4; indent-tabs-mode: nil; c-basic-offset: 4 -*- */ + +/* Fluent Bit + * ========== + * Copyright (C) 2015-2026 The Fluent Bit Authors + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#include + +#include +#include +#include + +#define FLB_RETRY_AFTER_NAME "Retry-After" +#define FLB_RETRY_AFTER_NAME_LEN 11 + +static int ascii_equal_ci(const char *value, const char *expected, size_t length) +{ + size_t index; + unsigned char left; + unsigned char right; + + for (index = 0; index < length; index++) { + left = (unsigned char) value[index]; + right = (unsigned char) expected[index]; + + if (left >= 'A' && left <= 'Z') { + left = (unsigned char) (left + ('a' - 'A')); + } + if (right >= 'A' && right <= 'Z') { + right = (unsigned char) (right + ('a' - 'A')); + } + if (left != right) { + return FLB_FALSE; + } + } + + return FLB_TRUE; +} + +static int parse_digits(const char *value, size_t length, int *result) +{ + size_t index; + int number; + + number = 0; + for (index = 0; index < length; index++) { + if (value[index] < '0' || value[index] > '9') { + return -1; + } + number = number * 10 + value[index] - '0'; + } + + *result = number; + return 0; +} + +static int parse_month(const char *value) +{ + static const char months[][4] = { + "Jan", "Feb", "Mar", "Apr", "May", "Jun", + "Jul", "Aug", "Sep", "Oct", "Nov", "Dec" + }; + int index; + + for (index = 0; index < 12; index++) { + if (ascii_equal_ci(value, months[index], 3)) { + return index + 1; + } + } + + return -1; +} + +static int parse_weekday(const char *value, size_t length) +{ + static const char short_names[][4] = { + "Sun", "Mon", "Tue", "Wed", "Thu", "Fri", "Sat" + }; + static const char *long_names[] = { + "Sunday", "Monday", "Tuesday", "Wednesday", + "Thursday", "Friday", "Saturday" + }; + size_t expected_length; + int index; + + for (index = 0; index < 7; index++) { + if (length == 3 && ascii_equal_ci(value, short_names[index], 3)) { + return index; + } + + expected_length = strlen(long_names[index]); + if (length == expected_length && + ascii_equal_ci(value, long_names[index], length)) { + return index; + } + } + + return -1; +} + +static int is_leap_year(int year) +{ + return year % 4 == 0 && (year % 100 != 0 || year % 400 == 0); +} + +static int days_in_month(int year, int month) +{ + static const int days[] = { + 31, 28, 31, 30, 31, 30, 31, 31, 30, 31, 30, 31 + }; + + if (month == 2 && is_leap_year(year)) { + return 29; + } + + return days[month - 1]; +} + +/* Return the number of days since 1970-01-01. */ +static int64_t days_from_civil(int year, int month, int day) +{ + int era; + unsigned int year_of_era; + unsigned int day_of_year; + unsigned int day_of_era; + + year -= month <= 2; + era = (year >= 0 ? year : year - 399) / 400; + year_of_era = (unsigned int) (year - era * 400); + day_of_year = (153U * (unsigned int) (month + (month > 2 ? -3 : 9)) + 2U) / 5U + + (unsigned int) day - 1U; + day_of_era = year_of_era * 365U + year_of_era / 4U - year_of_era / 100U + + day_of_year; + + return (int64_t) era * 146097 + (int64_t) day_of_era - 719468; +} + +static int year_from_days(int64_t days) +{ + int era; + int year; + unsigned int day_of_era; + unsigned int year_of_era; + + days += 719468; + era = (int) ((days >= 0 ? days : days - 146096) / 146097); + day_of_era = (unsigned int) (days - (int64_t) era * 146097); + year_of_era = (day_of_era - day_of_era / 1460U + day_of_era / 36524U - + day_of_era / 146096U) / 365U; + year = (int) year_of_era + era * 400; + + return year + (day_of_era - (365U * year_of_era + year_of_era / 4U - + year_of_era / 100U) < 306U); +} + +static int validate_date(int year, int month, int day, int hour, int minute, + int second, int weekday) +{ + int calculated_weekday; + int64_t days; + + if (year < 0 || year > 9999 || month < 1 || month > 12 || day < 1 || + day > days_in_month(year, month) || hour < 0 || hour > 23 || + minute < 0 || minute > 59 || second < 0 || second > 59) { + return -1; + } + + days = days_from_civil(year, month, day); + calculated_weekday = (int) ((days + 4) % 7); + if (calculated_weekday < 0) { + calculated_weekday += 7; + } + if (calculated_weekday != weekday) { + return -1; + } + + return 0; +} + +static int parse_time(const char *value, int *hour, int *minute, int *second) +{ + if (value[2] != ':' || value[5] != ':' || + parse_digits(value, 2, hour) != 0 || + parse_digits(value + 3, 2, minute) != 0 || + parse_digits(value + 6, 2, second) != 0) { + return -1; + } + + return 0; +} + +static int parse_http_date(const char *value, size_t length, + int64_t received_wall_time_ms, int64_t *date_ms) +{ + const char *comma; + size_t weekday_length; + int weekday; + int day; + int month; + int year; + int hour; + int minute; + int second; + int current_year; + int64_t days; + + day = -1; + month = -1; + year = -1; + hour = -1; + minute = -1; + second = -1; + + if (length == 29 && value[3] == ',' && value[4] == ' ' && + value[7] == ' ' && value[11] == ' ' && value[16] == ' ' && + value[25] == ' ' && ascii_equal_ci(value + 26, "GMT", 3)) { + weekday = parse_weekday(value, 3); + month = parse_month(value + 8); + if (weekday < 0 || month < 0 || parse_digits(value + 5, 2, &day) != 0 || + parse_digits(value + 12, 4, &year) != 0 || + parse_time(value + 17, &hour, &minute, &second) != 0) { + return -1; + } + } + else if (length >= 30 && length <= 33) { + comma = memchr(value, ',', length); + if (comma == NULL) { + return -1; + } + weekday_length = (size_t) (comma - value); + if (weekday_length < 6 || weekday_length > 9 || + length != weekday_length + 24 || comma[1] != ' ' || comma[4] != '-' || + comma[8] != '-' || comma[11] != ' ' || comma[20] != ' ' || + !ascii_equal_ci(comma + 21, "GMT", 3)) { + return -1; + } + weekday = parse_weekday(value, weekday_length); + month = parse_month(comma + 5); + if (weekday < 0 || month < 0 || parse_digits(comma + 2, 2, &day) != 0 || + parse_digits(comma + 9, 2, &year) != 0 || + parse_time(comma + 12, &hour, &minute, &second) != 0) { + return -1; + } + + current_year = year_from_days(received_wall_time_ms / 1000 / 86400); + year += (current_year / 100) * 100; + if (year > current_year + 50) { + year -= 100; + } + } + else if (length == 24 && value[3] == ' ' && value[7] == ' ' && + value[10] == ' ' && value[19] == ' ') { + weekday = parse_weekday(value, 3); + month = parse_month(value + 4); + if (value[8] == ' ' && value[9] >= '0' && value[9] <= '9') { + day = value[9] - '0'; + } + else if (parse_digits(value + 8, 2, &day) != 0) { + return -1; + } + if (weekday < 0 || month < 0 || + parse_time(value + 11, &hour, &minute, &second) != 0 || + parse_digits(value + 20, 4, &year) != 0) { + return -1; + } + } + else { + return -1; + } + + if (validate_date(year, month, day, hour, minute, second, weekday) != 0) { + return -1; + } + + days = days_from_civil(year, month, day); + *date_ms = (days * 86400 + hour * 3600 + minute * 60 + second) * 1000; + + return 0; +} + +int flb_http_retry_after_parse(const char *value, size_t length, + int64_t received_wall_time_ms, uint64_t *delay_ms) +{ + size_t start; + size_t end; + size_t index; + uint64_t seconds; + int64_t date_ms; + int saturated; + + if (value == NULL || delay_ms == NULL || length == 0) { + return FLB_RETRY_AFTER_INVALID; + } + + start = 0; + end = length; + while (start < end && (value[start] == ' ' || value[start] == '\t')) { + start++; + } + while (end > start && (value[end - 1] == ' ' || value[end - 1] == '\t')) { + end--; + } + if (start == end || memchr(value + start, '\0', end - start) != NULL) { + return FLB_RETRY_AFTER_INVALID; + } + + seconds = 0; + saturated = FLB_FALSE; + for (index = start; index < end; index++) { + if (value[index] < '0' || value[index] > '9') { + break; + } + if (saturated == FLB_FALSE) { + if (seconds > (UINT64_MAX - (uint64_t) (value[index] - '0')) / 10) { + saturated = FLB_TRUE; + } + else { + seconds = seconds * 10 + (uint64_t) (value[index] - '0'); + } + } + } + if (index == end) { + if (saturated == FLB_TRUE || seconds > UINT64_MAX / 1000) { + *delay_ms = UINT64_MAX; + return FLB_RETRY_AFTER_SATURATED; + } + *delay_ms = seconds * 1000; + return FLB_RETRY_AFTER_VALID; + } + + if (parse_http_date(value + start, end - start, + received_wall_time_ms, &date_ms) != 0) { + return FLB_RETRY_AFTER_INVALID; + } + if (date_ms <= received_wall_time_ms) { + *delay_ms = 0; + } + else if (received_wall_time_ms >= 0) { + *delay_ms = (uint64_t) date_ms - (uint64_t) received_wall_time_ms; + } + else if (date_ms < 0) { + *delay_ms = (uint64_t) (date_ms - (received_wall_time_ms + 1)) + 1; + } + else { + *delay_ms = (uint64_t) date_ms + + (uint64_t) (-(received_wall_time_ms + 1)) + 1; + } + + return FLB_RETRY_AFTER_VALID; +} + +int flb_http_retry_after_parse_headers(const char *headers, size_t length, + int64_t received_wall_time_ms, + uint64_t *delay_ms, size_t *invalid_count) +{ + size_t offset; + size_t line_end; + size_t colon; + size_t invalid; + uint64_t parsed_delay; + uint64_t largest_delay; + int parsed_status; + int result; + int found; + + if (invalid_count != NULL) { + *invalid_count = 0; + } + if (headers == NULL || delay_ms == NULL) { + return FLB_RETRY_AFTER_INVALID; + } + + offset = 0; + invalid = 0; + largest_delay = 0; + result = FLB_RETRY_AFTER_ABSENT; + found = FLB_FALSE; + + while (offset < length) { + for (line_end = offset; line_end + 1 < length; line_end++) { + if (headers[line_end] == '\r' && headers[line_end + 1] == '\n') { + break; + } + } + if (line_end + 1 >= length) { + break; + } + if (line_end == offset) { + break; + } + + colon = offset; + while (colon < line_end && headers[colon] != ':') { + colon++; + } + if (colon < line_end && + colon - offset == FLB_RETRY_AFTER_NAME_LEN && + ascii_equal_ci(headers + offset, FLB_RETRY_AFTER_NAME, + FLB_RETRY_AFTER_NAME_LEN)) { + found = FLB_TRUE; + parsed_status = flb_http_retry_after_parse(headers + colon + 1, + line_end - colon - 1, + received_wall_time_ms, + &parsed_delay); + if (parsed_status == FLB_RETRY_AFTER_VALID || + parsed_status == FLB_RETRY_AFTER_SATURATED) { + if (result == FLB_RETRY_AFTER_ABSENT || + result == FLB_RETRY_AFTER_INVALID || parsed_delay > largest_delay) { + largest_delay = parsed_delay; + result = parsed_status; + } + } + else { + invalid++; + } + } + + offset = line_end + 2; + } + + if (invalid_count != NULL) { + *invalid_count = invalid; + } + if (result == FLB_RETRY_AFTER_VALID || result == FLB_RETRY_AFTER_SATURATED) { + *delay_ms = largest_delay; + return result; + } + if (found) { + return FLB_RETRY_AFTER_INVALID; + } + + return FLB_RETRY_AFTER_ABSENT; +} diff --git a/src/flb_lib.c b/src/flb_lib.c index 1d2698a02f0..f3faafb2768 100644 --- a/src/flb_lib.c +++ b/src/flb_lib.c @@ -993,9 +993,6 @@ int static do_start(flb_ctx_t *ctx) } else if (val == FLB_ENGINE_FAILED) { flb_debug("[lib] backend failed"); -#if defined(FLB_SYSTEM_MACOS) - pthread_cancel(tid); -#endif pthread_join(tid, NULL); ctx->status = FLB_LIB_ERROR; return -1; @@ -1052,9 +1049,6 @@ int flb_stop(flb_ctx_t *ctx) * the service exited for some reason (plugin action). Always * wait and double check that the child thread is not running. */ -#if defined(FLB_SYSTEM_MACOS) - pthread_cancel(tid); -#endif pthread_join(tid, NULL); return 0; } @@ -1071,9 +1065,6 @@ int flb_stop(flb_ctx_t *ctx) flb_debug("[lib] sending STOP signal to the engine"); flb_engine_exit(ctx->config); -#if defined(FLB_SYSTEM_MACOS) - pthread_cancel(tid); -#endif ret = pthread_join(tid, NULL); if (ret != 0) { flb_errno(); diff --git a/src/flb_output.c b/src/flb_output.c index 42a3f2b04d3..dd4c163c5a3 100644 --- a/src/flb_output.c +++ b/src/flb_output.c @@ -20,6 +20,7 @@ #include #include #include +#include #include #include @@ -39,12 +40,244 @@ #include #include #include +#include +#include +#include #include #include #include FLB_TLS_DEFINE(struct flb_out_flush_params, out_flush_params); +#define FLB_OUTPUT_THROTTLE_RESUME_BATCH 32 + +int flb_output_throttle_complete(struct flb_output_flush *out_flush, int result) +{ + uint64_t now_ms; + uint64_t random_value; + struct flb_output_instance *ins; + struct flb_output_throttle_snapshot snapshot; + + ins = out_flush->o_ins; + now_ms = flb_output_throttle_now_ms(); + + if (result == FLB_THROTTLE) { + flb_output_throttle_snapshot(&ins->throttle, &snapshot); + if (snapshot.enabled == FLB_FALSE) { + return FLB_RETRY; + } + + random_value = 0; + flb_random_bytes((unsigned char *) &random_value, + sizeof(random_value)); + flb_output_throttle_publish(&ins->throttle, now_ms, + out_flush->retry_after_present, + out_flush->retry_after_ms, + random_value); + } + else if (result == FLB_OK) { + flb_output_throttle_success(&ins->throttle, now_ms, + out_flush->admission_generation); + } + + return result; +} + +static int output_throttle_has_pending(struct flb_output_instance *ins) +{ + if (ins->throttle_deferred_count > 0) { + return FLB_TRUE; + } + + if ((ins->flags & FLB_OUTPUT_SYNCHRONOUS) && + mk_list_is_empty(&ins->singleplex_queue->pending) != 0 && + mk_list_is_empty(&ins->singleplex_queue->in_progress) == 0) { + return FLB_TRUE; + } + + return FLB_FALSE; +} + +static void output_throttle_wakeup_callback(struct flb_config *config, void *data) +{ + int count; + int ret; + size_t pending_count; + uint64_t generation; + struct mk_list *head; + struct mk_list *tmp; + struct flb_task *task; + struct flb_task_retry *retry; + struct flb_task_route *route; + struct flb_output_instance *ins; + + ins = data; + ins->throttle_wakeup = NULL; + ins->throttle_wakeup_pending = FLB_FALSE; + flb_output_throttle_metrics_update(ins, flb_output_throttle_now_ms()); + + if (flb_output_throttle_admit(&ins->throttle, + flb_output_throttle_now_ms(), + &generation) == FLB_FALSE) { + flb_output_throttle_wakeup_schedule(ins); + return; + } + + if ((ins->flags & FLB_OUTPUT_SYNCHRONOUS) && + mk_list_is_empty(&ins->singleplex_queue->in_progress) == 0) { + flb_output_task_singleplex_flush_next(ins->singleplex_queue); + } + + count = 0; + pending_count = ins->throttle_deferred_count; + mk_list_foreach_safe(head, tmp, &ins->throttle_deferred_routes) { + if (count >= FLB_OUTPUT_THROTTLE_RESUME_BATCH || + count >= pending_count) { + break; + } + + if ((ins->flags & FLB_OUTPUT_NO_MULTIPLEX) && + ins->dispatches_inflight > 0) { + /* The completing dispatch will resume the next deferred route. */ + return; + } + + route = mk_list_entry(head, struct flb_task_route, _deferred_head); + task = route->task; + retry = flb_task_retry_get(task, ins); + + ret = flb_task_route_resume(task, ins); + if (ret == -1) { + continue; + } + + if (retry != NULL) { + ret = flb_engine_dispatch_retry(retry, config); + } + else { + if (ins->flags & FLB_OUTPUT_SYNCHRONOUS) { + ret = flb_output_task_singleplex_enqueue(ins->singleplex_queue, + NULL, task, ins, config); + } + else { + ret = flb_output_task_flush(task, ins, config); + } + + if (ret == -1) { + if (route->dispatch_state == FLB_TASK_ROUTE_DISPATCH_QUEUED) { + flb_task_route_unqueue(task, ins); + flb_task_users_dec(task, FLB_FALSE); + } + flb_task_users_release(task); + } + } + + count++; + } + + if (output_throttle_has_pending(ins) == FLB_TRUE) { + flb_output_throttle_wakeup_schedule(ins); + } +} + +int flb_output_throttle_wakeup_schedule(struct flb_output_instance *ins) +{ + int delay; + int ret; + uint64_t now_ms; + uint64_t remaining; + struct flb_output_throttle_snapshot snapshot; + + if (ins == NULL || ins->throttle_wakeup != NULL) { + return 0; + } + + if (ins->config == NULL || ins->config->sched == NULL) { + ins->throttle_wakeup_pending = FLB_TRUE; + return -1; + } + + flb_output_throttle_snapshot(&ins->throttle, &snapshot); + if (snapshot.state == FLB_OUTPUT_THROTTLE_STOPPING) { + ins->throttle_wakeup_pending = FLB_FALSE; + return -1; + } + + now_ms = flb_output_throttle_now_ms(); + if (snapshot.until_ms > now_ms) { + remaining = snapshot.until_ms - now_ms; + } + else { + remaining = 1; + } + + if (remaining > INT_MAX) { + delay = INT_MAX; + } + else { + delay = (int) remaining; + } + + ret = flb_sched_timer_cb_create(ins->config->sched, + FLB_SCHED_TIMER_CB_ONESHOT, + delay, + output_throttle_wakeup_callback, + ins, &ins->throttle_wakeup); + if (ret == -1) { + ins->throttle_wakeup = NULL; + ins->throttle_wakeup_pending = FLB_TRUE; + return -1; + } + + ins->throttle_wakeup_pending = FLB_FALSE; + return 0; +} + +void flb_output_throttle_wakeup_cancel(struct flb_output_instance *ins) +{ + struct flb_sched_timer *timer; + + if (ins == NULL || ins->throttle_wakeup == NULL) { + if (ins != NULL) { + ins->throttle_wakeup_pending = FLB_FALSE; + } + return; + } + + timer = ins->throttle_wakeup; + ins->throttle_wakeup = NULL; + ins->throttle_wakeup_pending = FLB_FALSE; + flb_sched_timer_cb_destroy(timer); +} + +void flb_output_throttle_wakeup_scan(struct flb_config *config) +{ + uint64_t now_ms; + struct mk_list *head; + struct flb_output_instance *ins; + struct flb_output_throttle_snapshot snapshot; + + now_ms = flb_output_throttle_now_ms(); + mk_list_foreach(head, &config->outputs) { + ins = mk_list_entry(head, struct flb_output_instance, _head); + flb_output_throttle_metrics_update(ins, now_ms); + if (ins->throttle_wakeup_pending == FLB_FALSE) { + continue; + } + + flb_output_throttle_snapshot(&ins->throttle, &snapshot); + if (snapshot.state == FLB_OUTPUT_THROTTLE_STOPPING) { + ins->throttle_wakeup_pending = FLB_FALSE; + } + else if (snapshot.until_ms <= now_ms) { + output_throttle_wakeup_callback(config, ins); + } + else { + flb_output_throttle_wakeup_schedule(ins); + } + } +} + /* Histogram buckets for output latency in seconds */ static const double output_latency_buckets[] = { 0.5, 1.0, 1.5, 2.5, 5.0, 10.0, 20.0, 30.0 @@ -54,6 +287,61 @@ static const double output_backpressure_wait_buckets[] = { 0.010, 0.050, 0.100, 0.250, 0.500, 1.0, 2.0, 5.0, 15.0, 30.0, 60.0 }; +void flb_output_throttle_metrics_update(struct flb_output_instance *ins, + uint64_t now_ms) +{ + uint64_t end_ms; + uint64_t timestamp; + double active; + double remaining_seconds; + char *name; + struct flb_output_throttle_snapshot snapshot; + + if (ins == NULL || ins->cmt_throttle_active == NULL) { + return; + } + + flb_output_throttle_snapshot(&ins->throttle, &snapshot); + timestamp = cfl_time_now(); + name = (char *) flb_output_name(ins); + active = 0; + remaining_seconds = 0; + + if (snapshot.enabled == FLB_TRUE && + snapshot.state == FLB_OUTPUT_THROTTLE_COOLDOWN && + snapshot.until_ms > now_ms) { + active = 1; + remaining_seconds = (double) (snapshot.until_ms - now_ms) / 1000.0; + } + + if (snapshot.started_ms > ins->throttle_duration_accounted_ms) { + ins->throttle_duration_accounted_ms = snapshot.started_ms; + } + + end_ms = now_ms; + if (end_ms > snapshot.until_ms) { + end_ms = snapshot.until_ms; + } + + if (end_ms > ins->throttle_duration_accounted_ms) { + cmt_counter_add(ins->cmt_throttle_duration, timestamp, + (double) (end_ms - ins->throttle_duration_accounted_ms) / + 1000.0, + 1, (char *[]) {name}); + ins->throttle_duration_accounted_ms = end_ms; + } + + cmt_gauge_set(ins->cmt_throttle_active, timestamp, active, + 1, (char *[]) {name}); + cmt_counter_set(ins->cmt_throttle_events, timestamp, + (double) snapshot.events, 1, (char *[]) {name}); + cmt_gauge_set(ins->cmt_throttle_remaining, timestamp, remaining_seconds, + 1, (char *[]) {name}); + cmt_gauge_set(ins->cmt_throttle_deferred_routes, timestamp, + (double) ins->throttle_deferred_count, + 1, (char *[]) {name}); +} + struct flb_config_map output_global_properties[] = { { FLB_CONFIG_MAP_STR, "match", NULL, @@ -95,6 +383,21 @@ struct flb_config_map output_global_properties[] = { "Accepted values: a positive integer, 'no_limits', 'false', or 'off' to disable retry limits, " "or 'no_retries' to disable retries entirely." }, + { + FLB_CONFIG_MAP_BOOL, "throttle", "false", + 0, FLB_FALSE, 0, + "Enable destination-driven cooldown for this output instance." + }, + { + FLB_CONFIG_MAP_INT, "throttle.base", "1", + 0, FLB_FALSE, 0, + "Minimum local throttle cooldown in seconds." + }, + { + FLB_CONFIG_MAP_INT, "throttle.cap", "60", + 0, FLB_FALSE, 0, + "Maximum local exponential throttle cooldown in seconds." + }, { FLB_CONFIG_MAP_STR, "tls.windows.certstore_name", NULL, 0, FLB_FALSE, 0, @@ -308,25 +611,29 @@ static int flb_output_task_queue_enqueue(struct flb_task_queue *queue, */ static int flb_output_task_queue_flush_one(struct flb_task_queue *queue) { + uint64_t generation; struct flb_task_enqueued *queued_task; int ret; int is_empty; is_empty = mk_list_is_empty(&queue->pending) == 0; if (is_empty) { - flb_error("Attempting to flush task from an empty in_progress queue"); + flb_error("Attempting to flush task from an empty pending queue"); return -1; } queued_task = mk_list_entry_first(&queue->pending, struct flb_task_enqueued, _head); + + /* Keep the singleplex owner in place while its output gate is closed. */ + if (flb_output_throttle_admit(&queued_task->out_instance->throttle, + flb_output_throttle_now_ms(), + &generation) == FLB_FALSE) { + flb_output_throttle_wakeup_schedule(queued_task->out_instance); + return 0; + } + mk_list_del(&queued_task->_head); mk_list_add(&queued_task->_head, &queue->in_progress); - - /* - * Remove temporary user now that task is out of singleplex queue. - * Flush will add back the user representing queued_task->out_instance if it succeeds. - */ - flb_task_users_dec(queued_task->task, FLB_FALSE); ret = flb_output_task_flush(queued_task->task, queued_task->out_instance, queued_task->config); @@ -337,10 +644,17 @@ static int flb_output_task_queue_flush_one(struct flb_task_queue *queue) flb_task_retry_destroy(queued_task->retry); } /* Flush the next task */ + flb_output_task_singleplex_complete(queue); flb_output_task_singleplex_flush_next(queue); return -1; } + /* The route now has deferred ownership; release this queue node. */ + if (ret == FLB_OUTPUT_DEFERRED) { + flb_output_task_singleplex_complete(queue); + flb_output_task_singleplex_flush_next(queue); + } + return ret; } @@ -354,8 +668,10 @@ int flb_output_task_singleplex_enqueue(struct flb_task_queue *queue, struct flb_output_instance *out_ins, struct flb_config *config) { + int acquired_owner; int ret; int is_empty; + struct flb_task_route *route; /* * Add temporary user to preserve task while in singleplex queue. @@ -365,11 +681,39 @@ int flb_output_task_singleplex_enqueue(struct flb_task_queue *queue, * deleted if the task's users go to 0 while we are waiting in the * queue. */ - flb_task_users_inc(task); + route = flb_task_route_get(task, out_ins); + if (route == NULL) { + if (retry != NULL) { + flb_task_retry_destroy(retry); + } + return -1; + } + + acquired_owner = FLB_FALSE; + if (route->dispatch_state == FLB_TASK_ROUTE_DISPATCH_UNQUEUED) { + ret = flb_task_route_queue(task, out_ins); + if (ret == -1) { + if (retry != NULL) { + flb_task_retry_destroy(retry); + } + return -1; + } + acquired_owner = FLB_TRUE; + } + else if (route->dispatch_state != FLB_TASK_ROUTE_DISPATCH_QUEUED) { + if (retry != NULL) { + flb_task_retry_destroy(retry); + } + return -1; + } /* Enqueue task */ ret = flb_output_task_queue_enqueue(queue, retry, task, out_ins, config); if (ret == -1) { + if (acquired_owner == FLB_TRUE) { + flb_task_route_unqueue(task, out_ins); + flb_task_users_dec(task, FLB_FALSE); + } return -1; } @@ -382,11 +726,8 @@ int flb_output_task_singleplex_enqueue(struct flb_task_queue *queue, return 0; } -/* - * Clear in progress task and flush a single queued task if exists - * Deletes retry context on next flush if flush fails - */ -int flb_output_task_singleplex_flush_next(struct flb_task_queue *queue) +/* Retire the current synchronous queue node without dispatching another. */ +void flb_output_task_singleplex_complete(struct flb_task_queue *queue) { int is_empty; struct flb_task_enqueued *ended_task; @@ -399,11 +740,26 @@ int flb_output_task_singleplex_flush_next(struct flb_task_queue *queue) mk_list_del(&ended_task->_head); flb_free(ended_task); } +} + +/* Flush one pending task. The dequeue path rechecks the output gate. */ +int flb_output_task_singleplex_flush_next(struct flb_task_queue *queue) +{ + int ret; + int is_empty; + struct flb_task *task; + struct flb_task_enqueued *queued_task; - /* Flush if there is a pending task queued */ is_empty = mk_list_is_empty(&queue->pending) == 0; if (!is_empty) { - return flb_output_task_queue_flush_one(queue); + queued_task = mk_list_entry_first(&queue->pending, + struct flb_task_enqueued, _head); + task = queued_task->task; + ret = flb_output_task_queue_flush_one(queue); + if (ret == -1) { + flb_task_users_release(task); + } + return ret; } return 0; } @@ -417,45 +773,65 @@ int flb_output_task_flush(struct flb_task *task, struct flb_config *config) { int ret; - struct flb_output_flush *out_flush; + uint64_t generation; + struct flb_task_route *route; + struct flb_output_dispatch *dispatch; - if (flb_output_is_threaded(out_ins) == FLB_TRUE) { - flb_task_users_inc(task); + route = flb_task_route_get(task, out_ins); + if (route == NULL) { + return -1; + } - /* Dispatch the task to the thread pool */ - ret = flb_output_thread_pool_flush(task, out_ins, config); + /* Advisory check. The receiver repeats this check authoritatively. */ + if (flb_output_throttle_admit(&out_ins->throttle, + flb_output_throttle_now_ms(), + &generation) == FLB_FALSE) { + ret = flb_task_route_defer(task, out_ins, + route->dispatch_state == + FLB_TASK_ROUTE_DISPATCH_QUEUED); if (ret == -1) { - flb_task_users_dec(task, FLB_FALSE); - - /* If we are in synchronous mode, flush one waiting task */ - if (out_ins->flags & FLB_OUTPUT_SYNCHRONOUS) { - flb_output_task_singleplex_flush_next(out_ins->singleplex_queue); - } + return -1; } + flb_output_throttle_wakeup_schedule(out_ins); + return FLB_OUTPUT_DEFERRED; } - else { - /* Queue co-routine handling */ - out_flush = flb_output_flush_create(task, - task->i_ins, - out_ins, - config); - if (!out_flush) { + + if (route->dispatch_state == FLB_TASK_ROUTE_DISPATCH_UNQUEUED) { + ret = flb_task_route_queue(task, out_ins); + if (ret == -1) { return -1; } + } + else if (route->dispatch_state != FLB_TASK_ROUTE_DISPATCH_QUEUED) { + return -1; + } + + dispatch = flb_output_dispatch_create(task, out_ins, config); + if (dispatch == NULL) { + flb_task_route_unqueue(task, out_ins); + flb_task_users_dec(task, FLB_FALSE); + return -1; + } - flb_task_users_inc(task); - ret = flb_pipe_w(config->ch_self_events[1], &out_flush, - sizeof(struct flb_output_flush*)); + if (flb_output_is_threaded(out_ins) == FLB_TRUE) { + /* Dispatch the task to the thread pool */ + ret = flb_output_thread_pool_flush(dispatch); if (ret == -1) { + flb_output_dispatch_destroy(dispatch); + flb_task_route_unqueue(task, out_ins); + flb_task_users_dec(task, FLB_FALSE); + return -1; + } + } + else { + ret = flb_pipe_write_all(config->ch_self_events[1], &dispatch, + sizeof(struct flb_output_dispatch *)); + if (ret != sizeof(struct flb_output_dispatch *)) { flb_pipe_error(); - flb_output_flush_destroy(out_flush); + flb_output_dispatch_destroy(dispatch); + flb_task_route_unqueue(task, out_ins); flb_task_users_dec(task, FLB_FALSE); - /* If we are in synchronous mode, flush one waiting task */ - if (out_ins->flags & FLB_OUTPUT_SYNCHRONOUS) { - flb_output_task_singleplex_flush_next(out_ins->singleplex_queue); - } - return -1; } } @@ -463,8 +839,59 @@ int flb_output_task_flush(struct flb_task *task, return 0; } +struct flb_output_dispatch *flb_output_dispatch_create( + struct flb_task *task, + struct flb_output_instance *out, + struct flb_config *config) +{ + struct flb_output_dispatch *dispatch; + + dispatch = flb_malloc(sizeof(struct flb_output_dispatch)); + if (dispatch == NULL) { + flb_errno(); + return NULL; + } + + dispatch->magic = FLB_OUTPUT_DISPATCH_MAGIC; + dispatch->type = FLB_OUTPUT_DISPATCH_TASK; + dispatch->result = 0; + dispatch->task = task; + dispatch->out = out; + dispatch->config = config; + mk_list_init(&dispatch->_head); + return dispatch; +} + +void flb_output_dispatch_destroy(struct flb_output_dispatch *dispatch) +{ + dispatch->magic = 0; + flb_free(dispatch); +} + +int flb_output_dispatch_post_result(struct flb_output_dispatch *dispatch, + int result, flb_pipefd_t pipe_fd) +{ + ssize_t written; + uint32_t set; + uint64_t value; + + set = FLB_TASK_SET(result, dispatch->task->id, dispatch->out->id); + value = FLB_BITS_U64_SET(FLB_ENGINE_TASK, set); + written = flb_pipe_write_all(pipe_fd, &value, sizeof(value)); + if (written != sizeof(value)) { + flb_pipe_error(); + return -1; + } + + return 0; +} + int flb_output_instance_destroy(struct flb_output_instance *ins) { + flb_output_throttle_metrics_update(ins, flb_output_throttle_now_ms()); + flb_output_throttle_stop(&ins->throttle); + flb_output_throttle_wakeup_cancel(ins); + flb_output_deferred_cancel_all(ins); if (ins->alias) { flb_sds_destroy(ins->alias); } @@ -557,6 +984,7 @@ int flb_output_instance_destroy(struct flb_output_instance *ins) flb_processor_destroy(ins->processor); } + flb_output_throttle_destroy(&ins->throttle); flb_free(ins); return 0; @@ -907,6 +1335,24 @@ struct flb_output_instance *flb_output_new(struct flb_config *config, mk_list_init(&instance->flush_list); mk_list_init(&instance->flush_list_destroy); + mk_list_init(&instance->throttle_deferred_routes); + instance->throttle_deferred_count = 0; + instance->dispatches_inflight = 0; + instance->throttle_wakeup = NULL; + instance->throttle_wakeup_pending = FLB_FALSE; + instance->throttle_duration_accounted_ms = 0; + + ret = flb_output_throttle_init(&instance->throttle, FLB_FALSE, 1000, 60000); + if (ret != 0) { + if (instance->singleplex_queue != NULL) { + flb_task_queue_destroy(instance->singleplex_queue); + } + flb_callback_destroy(instance->callback); + flb_free(instance->http_server_config); + flb_free(instance); + return NULL; + } + mk_list_add(&instance->_head, &config->outputs); /* processor instance */ @@ -938,6 +1384,7 @@ int flb_output_set_property(struct flb_output_instance *ins, { int len; int ret; + int seconds; ssize_t limit; flb_sds_t tmp; struct flb_kv *kv; @@ -1023,6 +1470,30 @@ int flb_output_set_property(struct flb_output_instance *ins, ins->retry_limit = 1; } } + else if (prop_key_check("throttle", k, len) == 0 && tmp) { + ret = flb_utils_bool(tmp); + flb_sds_destroy(tmp); + if (ret == -1) { + return -1; + } + ins->throttle.enabled = ret; + } + else if (prop_key_check("throttle.base", k, len) == 0 && tmp) { + ret = flb_utils_time_to_seconds_strict(tmp, &seconds); + flb_sds_destroy(tmp); + if (ret != 0) { + return -1; + } + ins->throttle.base_ms = (uint64_t) seconds * 1000; + } + else if (prop_key_check("throttle.cap", k, len) == 0 && tmp) { + ret = flb_utils_time_to_seconds_strict(tmp, &seconds); + flb_sds_destroy(tmp); + if (ret != 0) { + return -1; + } + ins->throttle.cap_ms = (uint64_t) seconds * 1000; + } else if (strncasecmp("net.", k, 4) == 0 && tmp) { kv = flb_kv_item_create(&ins->net_properties, (char *) k, NULL); if (!kv) { @@ -1454,6 +1925,11 @@ int flb_output_init_all(struct flb_config *config) /* Retrieve the plugin reference */ mk_list_foreach_safe(head, tmp, &config->outputs) { ins = mk_list_entry(head, struct flb_output_instance, _head); + if (ins->throttle.cap_ms < ins->throttle.base_ms) { + flb_error("[output %s] throttle.cap must be greater than or equal to " + "throttle.base", ins->name); + return -1; + } if (ins->log_level == -1) { ins->log_level = config->log->level; } @@ -1623,6 +2099,49 @@ int flb_output_init_all(struct flb_config *config) buckets, 1, (char *[]) {"output"}); + ins->cmt_throttle_events = cmt_counter_create(ins->cmt, "fluentbit", + "output", "throttle_events_total", + "Number of destination throttle reports.", + 1, (char *[]) {"name"}); + ins->cmt_throttle_active = cmt_gauge_create(ins->cmt, "fluentbit", + "output", "throttle_active", + "Whether the output throttle gate is closed.", + 1, (char *[]) {"name"}); + ins->cmt_throttle_remaining = cmt_gauge_create(ins->cmt, "fluentbit", + "output", + "throttle_remaining_seconds", + "Remaining output throttle cooldown.", + 1, (char *[]) {"name"}); + ins->cmt_throttle_deferred_routes = cmt_gauge_create(ins->cmt, + "fluentbit", "output", + "throttle_deferred_routes", + "Routes awaiting throttle admission.", + 1, (char *[]) {"name"}); + ins->cmt_throttle_duration = cmt_counter_create(ins->cmt, "fluentbit", + "output", + "throttle_duration_seconds_total", + "Union of elapsed closed-gate time.", + 1, (char *[]) {"name"}); + if (ins->cmt_throttle_events == NULL || + ins->cmt_throttle_active == NULL || + ins->cmt_throttle_remaining == NULL || + ins->cmt_throttle_deferred_routes == NULL || + ins->cmt_throttle_duration == NULL) { + flb_error("could not create throttle metrics for %s", name); + return -1; + } + + cmt_counter_set(ins->cmt_throttle_events, ts, 0, + 1, (char *[]) {name}); + cmt_gauge_set(ins->cmt_throttle_active, ts, 0, + 1, (char *[]) {name}); + cmt_gauge_set(ins->cmt_throttle_remaining, ts, 0, + 1, (char *[]) {name}); + cmt_gauge_set(ins->cmt_throttle_deferred_routes, ts, 0, + 1, (char *[]) {name}); + cmt_counter_set(ins->cmt_throttle_duration, ts, 0, + 1, (char *[]) {name}); + /* old API */ ins->metrics = flb_metrics_create(name); if (ins->metrics) { diff --git a/src/flb_output_thread.c b/src/flb_output_thread.c index bedcbcb0eab..82615493db2 100644 --- a/src/flb_output_thread.c +++ b/src/flb_output_thread.c @@ -26,13 +26,103 @@ #include #include #include +#include static pthread_once_t local_thread_instance_init = PTHREAD_ONCE_INIT; FLB_TLS_DEFINE(struct flb_out_thread_instance, local_thread_instance); +static pthread_mutex_t result_fallback_mutex; +static struct mk_list result_fallbacks; + +int flb_output_thread_post_dispatch_result( + struct flb_out_thread_instance *th_ins, + struct flb_output_dispatch *dispatch, + int result) +{ + int ret; + + pthread_once(&local_thread_instance_init, flb_output_thread_instance_init); + + ret = flb_output_dispatch_post_result(dispatch, result, + th_ins->ch_thread_events[1]); + if (ret == 0) { + return 0; + } + + flb_plg_warn(th_ins->ins, "could not post dispatch result through worker pipe; " + "using parent pipe"); + ret = flb_output_dispatch_post_result(dispatch, result, + th_ins->ins->ch_events[1]); + if (ret == 0) { + return 0; + } + + flb_plg_warn(th_ins->ins, "could not post dispatch result through parent pipe; " + "queueing engine-thread recovery"); + dispatch->result = result; + pthread_mutex_lock(&result_fallback_mutex); + mk_list_add(&dispatch->_head, &result_fallbacks); + pthread_mutex_unlock(&result_fallback_mutex); + return 1; +} void flb_output_thread_instance_init() { FLB_TLS_INIT(local_thread_instance); + pthread_mutex_init(&result_fallback_mutex, NULL); + mk_list_init(&result_fallbacks); +} + +struct flb_output_dispatch *flb_output_thread_result_fallback_pop( + struct flb_config *config) +{ + struct mk_list *head; + struct flb_output_dispatch *dispatch; + + pthread_once(&local_thread_instance_init, flb_output_thread_instance_init); + dispatch = NULL; + + pthread_mutex_lock(&result_fallback_mutex); + mk_list_foreach(head, &result_fallbacks) { + dispatch = mk_list_entry(head, struct flb_output_dispatch, _head); + if (dispatch->config == config) { + mk_list_del(&dispatch->_head); + break; + } + dispatch = NULL; + } + pthread_mutex_unlock(&result_fallback_mutex); + + return dispatch; +} + +static void output_thread_dispatch_release(struct flb_output_dispatch *dispatch) +{ + struct flb_task *task; + + task = dispatch->task; + if (flb_task_route_unqueue(task, dispatch->out) == 0) { + flb_task_users_dec(task, FLB_FALSE); + } + flb_output_dispatch_destroy(dispatch); +} + +void flb_output_thread_result_fallback_remove(struct flb_output_instance *ins) +{ + struct mk_list *head; + struct mk_list *tmp; + struct flb_output_dispatch *dispatch; + + pthread_once(&local_thread_instance_init, flb_output_thread_instance_init); + + pthread_mutex_lock(&result_fallback_mutex); + mk_list_foreach_safe(head, tmp, &result_fallbacks) { + dispatch = mk_list_entry(head, struct flb_output_dispatch, _head); + if (dispatch->out == ins) { + mk_list_del(&dispatch->_head); + output_thread_dispatch_release(dispatch); + } + } + pthread_mutex_unlock(&result_fallback_mutex); } struct flb_out_thread_instance *flb_output_thread_instance_get() @@ -69,8 +159,8 @@ static inline int handle_output_event(struct flb_config *config, uint32_t key; uint64_t val; - bytes = flb_pipe_r(fd, &val, sizeof(val)); - if (bytes == -1) { + bytes = flb_pipe_read_all(fd, &val, sizeof(val)); + if (bytes != sizeof(val)) { flb_pipe_error(); return -1; } @@ -88,15 +178,17 @@ static inline int handle_output_event(struct flb_config *config, ret = FLB_TASK_RET(key); out_id = FLB_TASK_OUT(key); - /* Destroy the output co-routine context */ - flb_output_flush_finished(config, out_id); + /* DEFERRED has no flush/coroutine context to destroy. */ + if (ret != FLB_OUTPUT_DEFERRED) { + flb_output_flush_finished(config, out_id); + } /* * Notify the parent event loop the return status, just forward the same * 64 bits value. */ - ret = flb_pipe_w(ch_parent, &val, sizeof(val)); - if (ret == -1) { + ret = flb_pipe_write_all(ch_parent, &val, sizeof(val)); + if (ret != sizeof(val)) { flb_pipe_error(); return -1; } @@ -158,6 +250,155 @@ static void upstream_thread_destroy(struct flb_out_thread_instance *th_ins) } } +static void output_thread_wakeup_consume(flb_pipefd_t fd) +{ + int bytes; + char buffer[64]; + + bytes = flb_pipe_r(fd, buffer, sizeof(buffer)); + if (bytes == -1 && !FLB_PIPE_WOULDBLOCK()) { + flb_pipe_error(); + } +} + +static struct flb_output_dispatch *output_thread_dispatch_pop( + struct flb_out_thread_instance *th_ins) +{ + struct flb_output_dispatch *dispatch; + + pthread_mutex_lock(&th_ins->dispatch_mutex); + if (mk_list_is_empty(&th_ins->dispatch_queue) == 0) { + dispatch = NULL; + } + else { + dispatch = mk_list_entry_first(&th_ins->dispatch_queue, + struct flb_output_dispatch, _head); + mk_list_del(&dispatch->_head); + mk_list_init(&dispatch->_head); + } + pthread_mutex_unlock(&th_ins->dispatch_mutex); + + return dispatch; +} + +static int output_thread_dispatch_queue_is_empty( + struct flb_out_thread_instance *th_ins) +{ + int result; + + pthread_mutex_lock(&th_ins->dispatch_mutex); + result = mk_list_is_empty(&th_ins->dispatch_queue) == 0; + pthread_mutex_unlock(&th_ins->dispatch_mutex); + + return result; +} + +static void output_thread_dispatch_process(struct flb_out_thread_instance *th_ins, + struct flb_output_dispatch *dispatch) +{ + int ret; + uint64_t generation; + size_t route_status; + struct flb_output_flush *out_flush; + + if (dispatch->magic != FLB_OUTPUT_DISPATCH_MAGIC || + dispatch->type != FLB_OUTPUT_DISPATCH_TASK || + dispatch->out != th_ins->ins) { + flb_plg_error(th_ins->ins, "invalid output dispatch envelope"); + flb_output_dispatch_destroy(dispatch); + return; + } + + flb_task_acquire_lock(dispatch->task); + route_status = flb_task_get_route_status(dispatch->task, dispatch->out); + flb_task_release_lock(dispatch->task); + if (route_status == FLB_TASK_ROUTE_DROPPED) { + ret = flb_output_thread_post_dispatch_result(th_ins, dispatch, FLB_ERROR); + if (ret == 0) { + flb_output_dispatch_destroy(dispatch); + } + return; + } + + if (flb_output_throttle_admit(&dispatch->out->throttle, + flb_output_throttle_now_ms(), + &generation) == FLB_FALSE) { + ret = flb_output_thread_post_dispatch_result(th_ins, dispatch, + FLB_OUTPUT_DEFERRED); + if (ret == 0) { + flb_output_dispatch_destroy(dispatch); + } + return; + } + + out_flush = flb_output_flush_create(dispatch->task, + dispatch->task->i_ins, + dispatch->out, + dispatch->config); + if (!out_flush) { + ret = flb_output_thread_post_dispatch_result(th_ins, dispatch, FLB_ERROR); + if (ret == 0) { + flb_output_dispatch_destroy(dispatch); + } + return; + } + out_flush->admission_generation = generation; + flb_output_dispatch_destroy(dispatch); + flb_coro_resume(out_flush->coro); +} + +static void output_thread_dispatch_drain(struct flb_out_thread_instance *th_ins) +{ + int count; + struct flb_output_dispatch *dispatch; + + for (count = 0; count < FLB_ENGINE_LOOP_MAX_ITER; count++) { + dispatch = output_thread_dispatch_pop(th_ins); + if (dispatch == NULL) { + break; + } + output_thread_dispatch_process(th_ins, dispatch); + } +} + +static void output_thread_dispatch_discard_all( + struct flb_out_thread_instance *th_ins) +{ + struct flb_output_dispatch *dispatch; + + while (FLB_TRUE) { + dispatch = output_thread_dispatch_pop(th_ins); + if (dispatch == NULL) { + break; + } + output_thread_dispatch_release(dispatch); + } +} + +static void output_thread_unstarted_destroy(struct flb_out_thread_instance *th_ins) +{ + if (th_ins->notification_channels_initialized == FLB_TRUE) { + mk_event_channel_destroy(th_ins->evl, + th_ins->notification_channels[0], + th_ins->notification_channels[1], + &th_ins->notification_event); + th_ins->notification_channels_initialized = FLB_FALSE; + } + + pthread_mutex_lock(&th_ins->dispatch_mutex); + th_ins->dispatch_shutdown = FLB_TRUE; + mk_event_channel_destroy(th_ins->evl, + th_ins->ch_parent_events[0], + th_ins->ch_parent_events[1], + th_ins); + th_ins->ch_parent_events[0] = FLB_INVALID_SOCKET; + th_ins->ch_parent_events[1] = FLB_INVALID_SOCKET; + pthread_mutex_unlock(&th_ins->dispatch_mutex); + upstream_thread_destroy(th_ins); + mk_event_loop_destroy(th_ins->evl); + flb_bucket_queue_destroy(th_ins->evl_bktq); +} + /* * This is the worker function that creates an event loop and synchronize * messages from the engine like 'flush' requests. Note that the running @@ -167,19 +408,18 @@ static void upstream_thread_destroy(struct flb_out_thread_instance *th_ins) */ static void output_thread(void *data) { - int n; int ret; int running = FLB_TRUE; int stopping = FLB_FALSE; + int thread_channel_initialized = FLB_FALSE; + int worker_initialized = FLB_FALSE; int thread_id; char tmp[64]; struct mk_event event_local; struct mk_event *event; struct flb_sched *sched; - struct flb_task *task; struct flb_connection *u_conn; struct flb_output_instance *ins; - struct flb_output_flush *out_flush; struct flb_out_thread_instance *th_ins = data; struct flb_out_flush_params *params; struct flb_sched_timer_coro_cb_params *sched_params; @@ -213,7 +453,7 @@ static void output_thread(void *data) sched = flb_sched_create(ins->config, th_ins->evl); if (!sched) { flb_plg_error(ins, "could not create thread scheduler"); - return; + goto cleanup; } flb_sched_ctx_set(sched); @@ -226,7 +466,7 @@ static void output_thread(void *data) 1500, cb_thread_sched_timer, ins, NULL); if (ret == -1) { flb_plg_error(ins, "could not schedule permanent callback"); - return; + goto cleanup; } snprintf(tmp, sizeof(tmp) - 1, "flb-out-%s-w%i", ins->name, thread_id); @@ -241,19 +481,24 @@ static void output_thread(void *data) &event_local); if (ret == -1) { flb_plg_error(th_ins->ins, "could not create thread channel"); - flb_engine_evl_set(NULL); - return; + goto cleanup; } + thread_channel_initialized = FLB_TRUE; event_local.type = FLB_ENGINE_EV_OUTPUT; if (ins->p->cb_worker_init) { ret = ins->p->cb_worker_init(ins->context, ins->config); } + worker_initialized = FLB_TRUE; flb_plg_info(th_ins->ins, "worker #%i started", thread_id); /* Thread event loop */ while (running) { + if (cfl_atomic_load(&th_ins->shutdown_requested) == FLB_TRUE) { + stopping = FLB_TRUE; + } + mk_event_wait(th_ins->evl); flb_event_priority_live_foreach(event, th_ins->evl_bktq, th_ins->evl, FLB_ENGINE_LOOP_MAX_ITER) { @@ -286,33 +531,8 @@ static void output_thread(void *data) } else if (event->type == FLB_ENGINE_EV_THREAD_OUTPUT) { - /* Read the task reference */ - n = flb_pipe_r(event->fd, &task, sizeof(struct flb_task *)); - if (n <= 0) { - flb_pipe_error(); - continue; - } - /* - * If the address receives 0xdeadbeef, means the thread must - * be terminated. - */ - if (task == (struct flb_task *) 0xdeadbeef) { - stopping = FLB_TRUE; - flb_plg_info(th_ins->ins, "thread worker #%i stopping...", - thread_id); - continue; - } - else { - /* Start the co-routine with the flush callback */ - out_flush = flb_output_flush_create(task, - task->i_ins, - th_ins->ins, - th_ins->config); - if (!out_flush) { - continue; - } - flb_coro_resume(out_flush->coro); - } + output_thread_wakeup_consume(event->fd); + output_thread_dispatch_drain(th_ins); } else if (event->type == FLB_ENGINE_EV_CUSTOM) { event->handler(event); @@ -353,6 +573,9 @@ static void output_thread(void *data) } } + /* Recover queued dispatches when their advisory wake could not be sent. */ + output_thread_dispatch_drain(th_ins); + flb_net_dns_lookup_context_cleanup(&dns_ctx); /* Destroy upstream connections from the 'pending destroy list' */ @@ -360,8 +583,14 @@ static void output_thread(void *data) flb_sched_timer_cleanup(sched); + if (cfl_atomic_load(&th_ins->shutdown_requested) == FLB_TRUE) { + stopping = FLB_TRUE; + } + /* Check if we should stop the event loop */ - if (stopping == FLB_TRUE && mk_list_size(&th_ins->flush_list) == 0) { + if (stopping == FLB_TRUE && + output_thread_dispatch_queue_is_empty(th_ins) == FLB_TRUE && + mk_list_size(&th_ins->flush_list) == 0) { /* * If there are no busy network connections (and no coroutines) its * safe to stop it. @@ -372,14 +601,17 @@ static void output_thread(void *data) } } - if (ins->p->cb_worker_exit) { +cleanup: + if (worker_initialized == FLB_TRUE && ins->p->cb_worker_exit) { ret = ins->p->cb_worker_exit(ins->context, ins->config); } - mk_event_channel_destroy(th_ins->evl, - th_ins->ch_thread_events[0], - th_ins->ch_thread_events[1], - &event_local); + if (thread_channel_initialized == FLB_TRUE) { + mk_event_channel_destroy(th_ins->evl, + th_ins->ch_thread_events[0], + th_ins->ch_thread_events[1], + &event_local); + } /* * Final cleanup, destroy all resources associated with: * @@ -394,7 +626,9 @@ static void output_thread(void *data) flb_upstream_conn_active_destroy_list(&th_ins->upstreams); flb_upstream_conn_pending_destroy_list(&th_ins->upstreams); - flb_sched_destroy(sched); + if (sched != NULL) { + flb_sched_destroy(sched); + } params = FLB_TLS_GET(out_flush_params); if (params) { flb_free(params); @@ -406,12 +640,15 @@ static void output_thread(void *data) flb_free(sched_params); FLB_TLS_SET(sched_timer_coro_cb_params, NULL); } - - + pthread_mutex_lock(&th_ins->dispatch_mutex); + th_ins->dispatch_shutdown = FLB_TRUE; mk_event_channel_destroy(th_ins->evl, th_ins->ch_parent_events[0], th_ins->ch_parent_events[1], th_ins); + th_ins->ch_parent_events[0] = FLB_INVALID_SOCKET; + th_ins->ch_parent_events[1] = FLB_INVALID_SOCKET; + pthread_mutex_unlock(&th_ins->dispatch_mutex); if (th_ins->notification_channels_initialized == FLB_TRUE) { mk_event_channel_destroy(th_ins->evl, @@ -424,34 +661,42 @@ static void output_thread(void *data) mk_event_loop_destroy(th_ins->evl); flb_bucket_queue_destroy(th_ins->evl_bktq); + flb_engine_evl_set(NULL); flb_plg_info(ins, "thread worker #%i stopped", thread_id); } -int flb_output_thread_pool_flush(struct flb_task *task, - struct flb_output_instance *out_ins, - struct flb_config *config) +int flb_output_thread_pool_flush(struct flb_output_dispatch *dispatch) { int n; + char wakeup; struct flb_tp_thread *th; struct flb_out_thread_instance *th_ins; /* Choose the worker that will handle the Task (round-robin) */ - th = flb_tp_thread_get_rr(out_ins->tp); + th = flb_tp_thread_get_rr(dispatch->out->tp); if (!th) { return -1; } th_ins = th->params.data; - flb_plg_debug(out_ins, "task_id=%i assigned to thread #%i", - task->id, th->id); + flb_plg_debug(dispatch->out, "task_id=%i assigned to thread #%i", + dispatch->task->id, th->id); - n = flb_pipe_w(th_ins->ch_parent_events[1], &task, sizeof(struct flb_task*)); + wakeup = 1; + pthread_mutex_lock(&th_ins->dispatch_mutex); + if (th_ins->dispatch_shutdown == FLB_TRUE) { + pthread_mutex_unlock(&th_ins->dispatch_mutex); + return -1; + } + mk_list_add(&dispatch->_head, &th_ins->dispatch_queue); + n = flb_pipe_w(th_ins->ch_parent_events[1], &wakeup, sizeof(wakeup)); + pthread_mutex_unlock(&th_ins->dispatch_mutex); - if (n == -1) { + if (n == -1 && !FLB_PIPE_WOULDBLOCK()) { flb_pipe_error(); - return -1; + flb_plg_warn(th_ins->ins, "could not wake worker thread for dispatch"); } return 0; @@ -497,6 +742,8 @@ int flb_output_thread_pool_create(struct flb_config *config, mk_list_init(&th_ins->flush_list); mk_list_init(&th_ins->flush_list_destroy); pthread_mutex_init(&th_ins->flush_mutex, NULL); + pthread_mutex_init(&th_ins->dispatch_mutex, NULL); + mk_list_init(&th_ins->dispatch_queue); mk_list_init(&th_ins->upstreams); upstream_thread_create(th_ins, ins); @@ -505,13 +752,19 @@ int flb_output_thread_pool_create(struct flb_config *config, evl = mk_event_loop_create(64); if (!evl) { flb_plg_error(ins, "could not create thread event loop"); + upstream_thread_destroy(th_ins); + pthread_mutex_destroy(&th_ins->dispatch_mutex); + pthread_mutex_destroy(&th_ins->flush_mutex); flb_free(th_ins); continue; } evl_bktq = flb_bucket_queue_create(FLB_ENGINE_PRIORITY_COUNT); if (!evl_bktq) { flb_plg_error(ins, "could not create thread event loop bucket queue"); - flb_free(evl); + mk_event_loop_destroy(evl); + upstream_thread_destroy(th_ins); + pthread_mutex_destroy(&th_ins->dispatch_mutex); + pthread_mutex_destroy(&th_ins->flush_mutex); flb_free(th_ins); continue; } @@ -535,6 +788,39 @@ int flb_output_thread_pool_create(struct flb_config *config, flb_plg_error(th_ins->ins, "could not create thread channel"); mk_event_loop_destroy(th_ins->evl); flb_bucket_queue_destroy(th_ins->evl_bktq); + upstream_thread_destroy(th_ins); + pthread_mutex_destroy(&th_ins->dispatch_mutex); + pthread_mutex_destroy(&th_ins->flush_mutex); + flb_free(th_ins); + continue; + } + ret = flb_pipe_set_nonblocking(th_ins->ch_parent_events[0]); + if (ret == -1) { + flb_plg_error(th_ins->ins, "could not configure thread channel"); + mk_event_channel_destroy(th_ins->evl, + th_ins->ch_parent_events[0], + th_ins->ch_parent_events[1], + th_ins); + mk_event_loop_destroy(th_ins->evl); + flb_bucket_queue_destroy(th_ins->evl_bktq); + upstream_thread_destroy(th_ins); + pthread_mutex_destroy(&th_ins->dispatch_mutex); + pthread_mutex_destroy(&th_ins->flush_mutex); + flb_free(th_ins); + continue; + } + ret = flb_pipe_set_nonblocking(th_ins->ch_parent_events[1]); + if (ret == -1) { + flb_plg_error(th_ins->ins, "could not configure thread wakeup channel"); + mk_event_channel_destroy(th_ins->evl, + th_ins->ch_parent_events[0], + th_ins->ch_parent_events[1], + th_ins); + mk_event_loop_destroy(th_ins->evl); + flb_bucket_queue_destroy(th_ins->evl_bktq); + upstream_thread_destroy(th_ins); + pthread_mutex_destroy(&th_ins->dispatch_mutex); + pthread_mutex_destroy(&th_ins->flush_mutex); flb_free(th_ins); continue; } @@ -557,6 +843,9 @@ int flb_output_thread_pool_create(struct flb_config *config, mk_event_loop_destroy(th_ins->evl); flb_bucket_queue_destroy(th_ins->evl_bktq); + upstream_thread_destroy(th_ins); + pthread_mutex_destroy(&th_ins->dispatch_mutex); + pthread_mutex_destroy(&th_ins->flush_mutex); flb_free(th_ins); continue; @@ -572,6 +861,22 @@ int flb_output_thread_pool_create(struct flb_config *config, th = flb_tp_thread_create(tp, output_thread, th_ins, config); if (!th) { flb_plg_error(ins, "could not register worker thread #%i", i); + if (th_ins->notification_channels_initialized == FLB_TRUE) { + mk_event_channel_destroy(th_ins->evl, + th_ins->notification_channels[0], + th_ins->notification_channels[1], + &th_ins->notification_event); + } + mk_event_channel_destroy(th_ins->evl, + th_ins->ch_parent_events[0], + th_ins->ch_parent_events[1], + th_ins); + mk_event_loop_destroy(th_ins->evl); + flb_bucket_queue_destroy(th_ins->evl_bktq); + upstream_thread_destroy(th_ins); + pthread_mutex_destroy(&th_ins->dispatch_mutex); + pthread_mutex_destroy(&th_ins->flush_mutex); + flb_free(th_ins); continue; } th_ins->th = th; @@ -611,7 +916,7 @@ int flb_output_thread_pool_coros_size(struct flb_output_instance *ins) void flb_output_thread_pool_destroy(struct flb_output_instance *ins) { int n; - struct flb_task *stop = (struct flb_task *) 0xdeadbeef; + char wakeup; struct flb_tp *tp = ins->tp; struct mk_list *head; struct flb_out_thread_instance *th_ins; @@ -621,25 +926,52 @@ void flb_output_thread_pool_destroy(struct flb_output_instance *ins) return; } - /* Signal each worker thread that needs to stop doing work */ + wakeup = 1; + + /* Close dispatch queues and wake every worker before joining. */ mk_list_foreach(head, &tp->list_threads) { th = mk_list_entry(head, struct flb_tp_thread, _head); - if (th->status != FLB_THREAD_POOL_RUNNING) { - continue; + th_ins = th->params.data; + pthread_mutex_lock(&th_ins->dispatch_mutex); + if (th_ins->dispatch_shutdown == FLB_FALSE && + th->status == FLB_THREAD_POOL_RUNNING) { + th_ins->dispatch_shutdown = FLB_TRUE; + cfl_atomic_store(&th_ins->shutdown_requested, FLB_TRUE); + n = flb_pipe_w(th_ins->ch_parent_events[1], &wakeup, sizeof(wakeup)); + } + else { + th_ins->dispatch_shutdown = FLB_TRUE; + n = sizeof(wakeup); + } + pthread_mutex_unlock(&th_ins->dispatch_mutex); + if (n == -1 && !FLB_PIPE_WOULDBLOCK()) { + flb_pipe_error(); + flb_plg_warn(th_ins->ins, "could not wake worker thread during shutdown"); } + } + /* Workers also observe shutdown from their periodic scheduler wakeup. */ + mk_list_foreach(head, &tp->list_threads) { + th = mk_list_entry(head, struct flb_tp_thread, _head); th_ins = th->params.data; - n = flb_pipe_w(th_ins->ch_parent_events[1], &stop, sizeof(stop)); - if (n < 0) { - flb_pipe_error(); - flb_plg_error(th_ins->ins, "could not signal worker thread"); - flb_free(th_ins); - continue; + if (th->status == FLB_THREAD_POOL_RUNNING) { + pthread_join(th->tid, NULL); + th->status = FLB_THREAD_POOL_STOPPED; + } + else { + output_thread_unstarted_destroy(th_ins); } - pthread_join(th->tid, NULL); + + output_thread_dispatch_discard_all(th_ins); + pthread_mutex_destroy(&th_ins->dispatch_mutex); + pthread_mutex_destroy(&th_ins->flush_mutex); flb_free(th_ins); + th->params.data = NULL; } + /* Release fallback dispatch ownership before the output instance is freed. */ + flb_output_thread_result_fallback_remove(ins); + flb_tp_destroy(ins->tp); ins->tp = NULL; } diff --git a/src/flb_output_throttle.c b/src/flb_output_throttle.c new file mode 100644 index 00000000000..ee838940b86 --- /dev/null +++ b/src/flb_output_throttle.c @@ -0,0 +1,233 @@ +/* -*- Mode: C; tab-width: 4; indent-tabs-mode: nil; c-basic-offset: 4 -*- */ + +#include +#include +#include + +#include + +uint64_t flb_output_throttle_now_ms(void) +{ +#ifdef FLB_SYSTEM_WINDOWS + return (uint64_t) GetTickCount64(); +#else + struct timespec now; + uint64_t seconds; + + if (clock_gettime(CLOCK_MONOTONIC, &now) != 0 || now.tv_sec < 0) { + return 0; + } + + seconds = (uint64_t) now.tv_sec; + if (seconds > UINT64_MAX / 1000) { + return UINT64_MAX; + } + + return seconds * 1000 + (uint64_t) now.tv_nsec / 1000000; +#endif +} + +static uint64_t saturating_add(uint64_t left, uint64_t right) +{ + if (UINT64_MAX - left < right) { + return UINT64_MAX; + } + + return left + right; +} + +static uint64_t local_ceiling(struct flb_output_throttle *throttle) +{ + uint32_t shifts; + uint64_t value; + + value = throttle->base_ms; + shifts = throttle->consecutive_rounds - 1; + + while (shifts > 0 && value < throttle->cap_ms) { + if (value > throttle->cap_ms / 2) { + value = throttle->cap_ms; + } + else { + value *= 2; + } + shifts--; + } + + return value; +} + +static void expire_locked(struct flb_output_throttle *throttle, uint64_t now_ms) +{ + if (throttle->state == FLB_OUTPUT_THROTTLE_COOLDOWN && + now_ms >= throttle->until_ms) { + throttle->state = FLB_OUTPUT_THROTTLE_READY; + } +} + +int flb_output_throttle_init(struct flb_output_throttle *throttle, + int enabled, uint64_t base_ms, uint64_t cap_ms) +{ + int result; + + if (throttle == NULL || base_ms == 0 || cap_ms < base_ms) { + return -1; + } + + memset(throttle, 0, sizeof(struct flb_output_throttle)); + throttle->enabled = enabled; + throttle->state = FLB_OUTPUT_THROTTLE_READY; + throttle->base_ms = base_ms; + throttle->cap_ms = cap_ms; + throttle->started_ms = 0; + + result = pthread_mutex_init(&throttle->lock, NULL); + if (result != 0) { + return -1; + } + + return 0; +} + +void flb_output_throttle_destroy(struct flb_output_throttle *throttle) +{ + pthread_mutex_destroy(&throttle->lock); +} + +int flb_output_throttle_admit(struct flb_output_throttle *throttle, + uint64_t now_ms, uint64_t *generation) +{ + int result; + + if (throttle == NULL || generation == NULL) { + return FLB_FALSE; + } + + pthread_mutex_lock(&throttle->lock); + + if (throttle->state == FLB_OUTPUT_THROTTLE_STOPPING) { + result = FLB_FALSE; + } + else if (throttle->enabled == FLB_FALSE) { + *generation = 0; + result = FLB_TRUE; + } + else { + expire_locked(throttle, now_ms); + if (throttle->state == FLB_OUTPUT_THROTTLE_COOLDOWN) { + result = FLB_FALSE; + } + else { + *generation = throttle->generation; + result = FLB_TRUE; + } + } + + pthread_mutex_unlock(&throttle->lock); + return result; +} + +uint64_t flb_output_throttle_publish(struct flb_output_throttle *throttle, + uint64_t now_ms, int hint_present, + uint64_t hint_ms, uint64_t random_value) +{ + uint64_t floor; + uint64_t span; + uint64_t delay; + uint64_t deadline; + uint64_t ceiling; + uint64_t old_deadline; + + pthread_mutex_lock(&throttle->lock); + + if (throttle->enabled == FLB_FALSE || + throttle->state == FLB_OUTPUT_THROTTLE_STOPPING) { + deadline = throttle->until_ms; + pthread_mutex_unlock(&throttle->lock); + return deadline; + } + + expire_locked(throttle, now_ms); + old_deadline = 0; + if (throttle->state == FLB_OUTPUT_THROTTLE_COOLDOWN) { + old_deadline = throttle->until_ms; + } + else { + throttle->started_ms = now_ms; + if (throttle->consecutive_rounds < UINT32_MAX) { + throttle->consecutive_rounds++; + } + } + + ceiling = local_ceiling(throttle); + floor = ceiling / 2 + ceiling % 2; + if (floor < throttle->base_ms) { + floor = throttle->base_ms; + } + span = ceiling - floor; + if (span == UINT64_MAX) { + delay = random_value; + } + else { + delay = floor + random_value % (span + 1); + } + if (hint_present == FLB_TRUE && hint_ms > delay) { + delay = hint_ms; + } + + deadline = saturating_add(now_ms, delay); + if (deadline < old_deadline) { + deadline = old_deadline; + } + + throttle->until_ms = deadline; + if (throttle->generation == UINT64_MAX) { + throttle->state = FLB_OUTPUT_THROTTLE_STOPPING; + pthread_mutex_unlock(&throttle->lock); + return deadline; + } + throttle->generation++; + throttle->events++; + throttle->state = FLB_OUTPUT_THROTTLE_COOLDOWN; + + pthread_mutex_unlock(&throttle->lock); + return deadline; +} + +void flb_output_throttle_success(struct flb_output_throttle *throttle, + uint64_t now_ms, uint64_t generation) +{ + pthread_mutex_lock(&throttle->lock); + + if (throttle->enabled == FLB_TRUE && + throttle->state != FLB_OUTPUT_THROTTLE_STOPPING) { + expire_locked(throttle, now_ms); + if (throttle->state == FLB_OUTPUT_THROTTLE_READY && + generation == throttle->generation) { + throttle->consecutive_rounds = 0; + } + } + + pthread_mutex_unlock(&throttle->lock); +} + +void flb_output_throttle_stop(struct flb_output_throttle *throttle) +{ + pthread_mutex_lock(&throttle->lock); + throttle->state = FLB_OUTPUT_THROTTLE_STOPPING; + pthread_mutex_unlock(&throttle->lock); +} + +void flb_output_throttle_snapshot(struct flb_output_throttle *throttle, + struct flb_output_throttle_snapshot *snapshot) +{ + pthread_mutex_lock(&throttle->lock); + snapshot->enabled = throttle->enabled; + snapshot->state = throttle->state; + snapshot->started_ms = throttle->started_ms; + snapshot->until_ms = throttle->until_ms; + snapshot->generation = throttle->generation; + snapshot->consecutive_rounds = throttle->consecutive_rounds; + snapshot->events = throttle->events; + pthread_mutex_unlock(&throttle->lock); +} diff --git a/src/flb_search_bulk.c b/src/flb_search_bulk.c index 01157daf917..629b95c6bcf 100644 --- a/src/flb_search_bulk.c +++ b/src/flb_search_bulk.c @@ -505,6 +505,7 @@ int flb_search_bulk_process_response(const char *response, int acknowledge_all_conflicts, int drop_unrecoverable_records, struct flb_search_bulk_stats *stats, + int *out_throttled, struct flb_search_bulk_retry **out_retry) { int index; @@ -532,6 +533,9 @@ int flb_search_bulk_process_response(const char *response, if (stats != NULL) { memset(stats, 0, sizeof(struct flb_search_bulk_stats)); } + if (out_throttled != NULL) { + *out_throttled = FLB_FALSE; + } packed_response = NULL; retry = NULL; items.type = MSGPACK_OBJECT_NIL; @@ -653,6 +657,9 @@ int flb_search_bulk_process_response(const char *response, if (item_result == FLB_SEARCH_BULK_ITEM_RETRYABLE || (item_result == FLB_SEARCH_BULK_ITEM_UNRECOVERABLE && drop_unrecoverable_records == FLB_FALSE)) { + if (status == 429 && out_throttled != NULL) { + *out_throttled = FLB_TRUE; + } memcpy(retry->payload + retry->size, payload + entry_start, entry_size); retry->size += entry_size; @@ -674,6 +681,9 @@ int flb_search_bulk_process_response(const char *response, result = FLB_SEARCH_BULK_RETRY; done: + if (result != FLB_SEARCH_BULK_RETRY && out_throttled != NULL) { + *out_throttled = FLB_FALSE; + } flb_search_bulk_retry_destroy(retry); msgpack_unpacked_destroy(&unpacked); flb_free(packed_response); diff --git a/src/flb_task.c b/src/flb_task.c index dcefdc66a89..f9d34f82ef2 100644 --- a/src/flb_task.c +++ b/src/flb_task.c @@ -37,6 +37,20 @@ #endif #include +static void task_route_init(struct flb_task_route *route, + struct flb_task *task, + struct flb_output_instance *out, + int records, size_t bytes) +{ + route->status = FLB_TASK_ROUTE_INACTIVE; + route->dispatch_state = FLB_TASK_ROUTE_DISPATCH_UNQUEUED; + route->records = records; + route->bytes = bytes; + route->out = out; + route->task = task; + mk_list_init(&route->_deferred_head); +} + /* * Every task created must have an unique ID, this function lookup the * lowest number available in the task_map. @@ -301,7 +315,7 @@ int flb_task_retry_reschedule(struct flb_task_retry *retry, struct flb_config *c * resides only in memory, it will be lost. */ flb_warn("[task] retry for task %i could not be re-scheduled", task->id); flb_task_retry_destroy(retry); - if (task->users == 0 && mk_list_size(&task->retries) == 0) { + if (flb_task_is_releasable(task) == FLB_TRUE) { flb_task_destroy(task, FLB_TRUE); } return -1; @@ -383,6 +397,22 @@ struct flb_task_retry *flb_task_retry_create(struct flb_task *task, return retry; } +struct flb_task_retry *flb_task_retry_get(struct flb_task *task, + struct flb_output_instance *ins) +{ + struct mk_list *head; + struct flb_task_retry *retry; + + mk_list_foreach(head, &task->retries) { + retry = mk_list_entry(head, struct flb_task_retry, _head); + if (retry->o_ins == ins) { + return retry; + } + } + + return NULL; +} + /* * Return FLB_TRUE or FLB_FALSE if the chunk pointed by the task was * created on this running instance or it comes from a chunk in the @@ -465,6 +495,7 @@ struct flb_task *task_alloc(struct flb_config *config) task->config = config; task->status = FLB_TASK_NEW; task->users = 0; + task->deferred_routes = 0; mk_list_init(&task->routes); mk_list_init(&task->retries); @@ -486,7 +517,8 @@ int flb_task_running_count(struct flb_config *config) ins = mk_list_entry(head, struct flb_input_instance, _head); mk_list_foreach(t_head, &ins->tasks) { task = mk_list_entry(t_head, struct flb_task, _head); - if (task->users > 0 || mk_list_size(&task->retries) > 0) { + if (task->users > 0 || task->deferred_routes > 0 || + mk_list_size(&task->retries) > 0) { count++; } } @@ -711,10 +743,9 @@ struct flb_task *flb_task_create(uint64_t ref_id, break; } - route->status = FLB_TASK_ROUTE_INACTIVE; - route->records = evc->total_events; - route->bytes = evc->size; - route->out = stored_matches[stored_match_index]; + task_route_init(route, task, + stored_matches[stored_match_index], + evc->total_events, evc->size); mk_list_add(&route->_head, &task->routes); direct_count++; } @@ -815,10 +846,8 @@ struct flb_task *flb_task_create(uint64_t ref_id, return NULL; } - route->status = FLB_TASK_ROUTE_INACTIVE; - route->records = evc->total_events; - route->bytes = evc->size; - route->out = o_ins; + task_route_init(route, task, o_ins, + evc->total_events, evc->size); mk_list_add(&route->_head, &task->routes); direct_count++; } @@ -863,10 +892,8 @@ struct flb_task *flb_task_create(uint64_t ref_id, continue; } - route->status = FLB_TASK_ROUTE_INACTIVE; - route->records = evc->total_events; - route->bytes = evc->size; - route->out = o_ins; + task_route_init(route, task, o_ins, + evc->total_events, evc->size); mk_list_add(&route->_head, &task->routes); count++; } @@ -896,8 +923,25 @@ struct flb_task *flb_task_create(uint64_t ref_id, return task; } +static int task_output_is_registered(struct flb_task *task, + struct flb_output_instance *output) +{ + struct mk_list *head; + struct flb_output_instance *instance; + + mk_list_foreach(head, &task->config->outputs) { + instance = mk_list_entry(head, struct flb_output_instance, _head); + if (instance == output) { + return FLB_TRUE; + } + } + + return FLB_FALSE; +} + void flb_task_destroy(struct flb_task *task, int del) { + int output_is_registered; struct mk_list *tmp; struct mk_list *head; struct flb_task_route *route; @@ -911,6 +955,21 @@ void flb_task_destroy(struct flb_task *task, int del) /* Remove routes */ mk_list_foreach_safe(head, tmp, &task->routes) { route = mk_list_entry(head, struct flb_task_route, _head); + if (route->dispatch_state == FLB_TASK_ROUTE_DISPATCH_DEFERRED) { + output_is_registered = task_output_is_registered(task, route->out); + if (output_is_registered == FLB_TRUE) { + mk_list_del(&route->_deferred_head); + route->out->throttle_deferred_count--; + } + task->deferred_routes--; + } + else if (route->dispatch_state == FLB_TASK_ROUTE_DISPATCH_QUEUED || + route->dispatch_state == FLB_TASK_ROUTE_DISPATCH_COMPLETING) { + output_is_registered = task_output_is_registered(task, route->out); + if (output_is_registered == FLB_TRUE) { + route->out->dispatches_inflight--; + } + } if (route->retry_context != NULL && route->retry_context_destroy != NULL) { route->retry_context_destroy(route->retry_context); @@ -947,6 +1006,168 @@ void flb_task_destroy(struct flb_task *task, int del) flb_free(task); } +struct flb_task_route *flb_task_route_get(struct flb_task *task, + struct flb_output_instance *ins) +{ + struct mk_list *head; + struct flb_task_route *route; + + mk_list_foreach(head, &task->routes) { + route = mk_list_entry(head, struct flb_task_route, _head); + if (route->out == ins) { + return route; + } + } + + return NULL; +} + +int flb_task_route_defer(struct flb_task *task, + struct flb_output_instance *ins, + int transfer_queued_owner) +{ + struct flb_task_route *route; + + route = flb_task_route_get(task, ins); + if (route == NULL || route->status == FLB_TASK_ROUTE_DROPPED) { + return -1; + } + + if (route->dispatch_state == FLB_TASK_ROUTE_DISPATCH_DEFERRED) { + return transfer_queued_owner == FLB_FALSE ? 0 : -1; + } + + if (transfer_queued_owner == FLB_TRUE) { + if (route->dispatch_state != FLB_TASK_ROUTE_DISPATCH_QUEUED || + task->users <= 0) { + return -1; + } + } + else if (route->dispatch_state != FLB_TASK_ROUTE_DISPATCH_UNQUEUED) { + return -1; + } + + /* Acquire deferred ownership before releasing a queued execution owner. */ + route->dispatch_state = FLB_TASK_ROUTE_DISPATCH_DEFERRED; + mk_list_add(&route->_deferred_head, &ins->throttle_deferred_routes); + task->deferred_routes++; + ins->throttle_deferred_count++; + + if (transfer_queued_owner == FLB_TRUE) { + ins->dispatches_inflight--; + flb_task_users_dec(task, FLB_FALSE); + } + + return 0; +} + +int flb_task_route_resume(struct flb_task *task, + struct flb_output_instance *ins) +{ + struct flb_task_route *route; + + route = flb_task_route_get(task, ins); + if (route == NULL || + route->dispatch_state != FLB_TASK_ROUTE_DISPATCH_DEFERRED) { + return -1; + } + + /* Acquire queued ownership before releasing deferred ownership. */ + flb_task_users_inc(task); + ins->dispatches_inflight++; + mk_list_del(&route->_deferred_head); + mk_list_init(&route->_deferred_head); + task->deferred_routes--; + ins->throttle_deferred_count--; + route->dispatch_state = FLB_TASK_ROUTE_DISPATCH_QUEUED; + + return 0; +} + +int flb_task_route_cancel_deferred(struct flb_task *task, + struct flb_output_instance *ins) +{ + struct flb_task_route *route; + + route = flb_task_route_get(task, ins); + if (route == NULL || + route->dispatch_state != FLB_TASK_ROUTE_DISPATCH_DEFERRED) { + return -1; + } + + mk_list_del(&route->_deferred_head); + mk_list_init(&route->_deferred_head); + task->deferred_routes--; + ins->throttle_deferred_count--; + route->dispatch_state = FLB_TASK_ROUTE_DISPATCH_UNQUEUED; + + return 0; +} + +int flb_task_route_queue(struct flb_task *task, + struct flb_output_instance *ins) +{ + struct flb_task_route *route; + + route = flb_task_route_get(task, ins); + if (route == NULL || route->status == FLB_TASK_ROUTE_DROPPED || + route->dispatch_state != FLB_TASK_ROUTE_DISPATCH_UNQUEUED) { + return -1; + } + + flb_task_users_inc(task); + ins->dispatches_inflight++; + route->dispatch_state = FLB_TASK_ROUTE_DISPATCH_QUEUED; + return 0; +} + +int flb_task_route_complete(struct flb_task *task, + struct flb_output_instance *ins) +{ + struct flb_task_route *route; + + route = flb_task_route_get(task, ins); + if (route == NULL || + route->dispatch_state != FLB_TASK_ROUTE_DISPATCH_QUEUED) { + return -1; + } + + route->dispatch_state = FLB_TASK_ROUTE_DISPATCH_COMPLETING; + return 0; +} + +int flb_task_route_unqueue(struct flb_task *task, + struct flb_output_instance *ins) +{ + struct flb_task_route *route; + + route = flb_task_route_get(task, ins); + if (route == NULL || + (route->dispatch_state != FLB_TASK_ROUTE_DISPATCH_QUEUED && + route->dispatch_state != FLB_TASK_ROUTE_DISPATCH_COMPLETING)) { + return -1; + } + + route->dispatch_state = FLB_TASK_ROUTE_DISPATCH_UNQUEUED; + ins->dispatches_inflight--; + return 0; +} + +void flb_output_deferred_cancel_all(struct flb_output_instance *ins) +{ + struct mk_list *head; + struct mk_list *tmp; + struct flb_task *task; + struct flb_task_route *route; + + mk_list_foreach_safe(head, tmp, &ins->throttle_deferred_routes) { + route = mk_list_entry(head, struct flb_task_route, _deferred_head); + task = route->task; + flb_task_route_cancel_deferred(task, ins); + flb_task_users_release(task); + } +} + struct flb_task_queue* flb_task_queue_create() { struct flb_task_queue *tq; tq = flb_malloc(sizeof(struct flb_task_queue)); @@ -959,7 +1180,8 @@ struct flb_task_queue* flb_task_queue_create() { return tq; } -void flb_task_queue_destroy(struct flb_task_queue *queue) { +void flb_task_queue_destroy(struct flb_task_queue *queue) +{ struct flb_task_enqueued *queued_task; struct mk_list *tmp; struct mk_list *head; @@ -967,12 +1189,26 @@ void flb_task_queue_destroy(struct flb_task_queue *queue) { mk_list_foreach_safe(head, tmp, &queue->pending) { queued_task = mk_list_entry(head, struct flb_task_enqueued, _head); mk_list_del(&queued_task->_head); + if (flb_task_route_unqueue(queued_task->task, + queued_task->out_instance) == 0) { + flb_task_users_dec(queued_task->task, FLB_FALSE); + } + if (queued_task->retry != NULL) { + flb_task_retry_destroy(queued_task->retry); + } flb_free(queued_task); } mk_list_foreach_safe(head, tmp, &queue->in_progress) { queued_task = mk_list_entry(head, struct flb_task_enqueued, _head); mk_list_del(&queued_task->_head); + if (flb_task_route_unqueue(queued_task->task, + queued_task->out_instance) == 0) { + flb_task_users_dec(queued_task->task, FLB_FALSE); + } + if (queued_task->retry != NULL) { + flb_task_retry_destroy(queued_task->retry); + } flb_free(queued_task); } diff --git a/src/flb_utils.c b/src/flb_utils.c index 5f536a5c53f..10332f917a0 100644 --- a/src/flb_utils.c +++ b/src/flb_utils.c @@ -22,6 +22,8 @@ #include #include #include +#include +#include #include #include @@ -754,6 +756,41 @@ int flb_utils_time_to_seconds(const char *time) return val; } +int flb_utils_time_to_seconds_strict(const char *time, int *seconds) +{ + int result; + char *end; + long checked_value; + const unsigned char *cursor; + + if (time == NULL || seconds == NULL || time[0] == '\0') { + return -1; + } + + cursor = (const unsigned char *) time; + while (*cursor != '\0') { + if (isdigit(*cursor) == 0) { + return -1; + } + cursor++; + } + + errno = 0; + checked_value = strtol(time, &end, 10); + if (errno == ERANGE || end == time || *end != '\0' || + checked_value <= 0 || checked_value > INT_MAX) { + return -1; + } + + result = flb_utils_time_to_seconds(time); + if (result <= 0 || result != checked_value) { + return -1; + } + + *seconds = result; + return 0; +} + int flb_utils_bool(const char *val) { if (strcasecmp(val, "true") == 0 || diff --git a/tests/integration/scenarios/out_es/config/out_es_partial_bulk_retry.yaml b/tests/integration/scenarios/out_es/config/out_es_partial_bulk_retry.yaml index c2493c886f6..653b110a180 100644 --- a/tests/integration/scenarios/out_es/config/out_es_partial_bulk_retry.yaml +++ b/tests/integration/scenarios/out_es/config/out_es_partial_bulk_retry.yaml @@ -23,3 +23,6 @@ pipeline: index: fluent-bit suppress_type_name: on retry_limit: 2 + throttle: on + throttle.base: 1 + throttle.cap: 1 diff --git a/tests/integration/scenarios/out_es/config/out_opensearch_partial_bulk_retry.yaml b/tests/integration/scenarios/out_es/config/out_opensearch_partial_bulk_retry.yaml index 59c567a5ae8..f8e509c5049 100644 --- a/tests/integration/scenarios/out_es/config/out_opensearch_partial_bulk_retry.yaml +++ b/tests/integration/scenarios/out_es/config/out_opensearch_partial_bulk_retry.yaml @@ -23,3 +23,6 @@ pipeline: index: fluent-bit suppress_type_name: on retry_limit: 2 + throttle: on + throttle.base: 1 + throttle.cap: 1 diff --git a/tests/integration/scenarios/out_es/tests/test_out_es_ndjson_action_line_001.py b/tests/integration/scenarios/out_es/tests/test_out_es_ndjson_action_line_001.py index fbf3017e782..498bdc8f3e1 100644 --- a/tests/integration/scenarios/out_es/tests/test_out_es_ndjson_action_line_001.py +++ b/tests/integration/scenarios/out_es/tests/test_out_es_ndjson_action_line_001.py @@ -31,20 +31,54 @@ def do_POST(self): "path": self.path, "headers": dict(self.headers), "body": body.decode("utf-8", errors="replace"), + "received_at": time.monotonic(), } ) if self.server.response_factory is None: response = b'{"errors":false,"items":[{"create":{"status":201}}]}' else: - response = self.server.response_factory( + response_result = self.server.response_factory( len(self.server.requests), self.server.requests[-1] ) - self.send_response(200) + if isinstance(response_result, tuple): + if len(response_result) == 4: + ( + status_code, + response, + response_headers, + response_trailers, + ) = response_result + else: + status_code, response, response_headers = response_result + response_trailers = [] + else: + status_code = 200 + response = response_result + response_headers = [] + response_trailers = [] + if self.server.response_factory is None: + status_code = 200 + response_headers = [] + response_trailers = [] + self.send_response(status_code) self.send_header("Content-Type", "application/json") - self.send_header("Content-Length", str(len(response))) + for name, value in response_headers: + self.send_header(name, value) + if response_trailers: + self.send_header("Transfer-Encoding", "chunked") + self.send_header("Trailer", ", ".join(name for name, _ in response_trailers)) + else: + self.send_header("Content-Length", str(len(response))) self.end_headers() - self.wfile.write(response) + if response_trailers: + self.wfile.write(f"{len(response):x}\r\n".encode("ascii")) + self.wfile.write(response + b"\r\n0\r\n") + for name, value in response_trailers: + self.wfile.write(f"{name}: {value}\r\n".encode("ascii")) + self.wfile.write(b"\r\n") + else: + self.wfile.write(response) class _BulkCaptureServer(ThreadingHTTPServer): @@ -278,6 +312,39 @@ def _unrecoverable_bulk_response(request_number, request): return json.dumps({"errors": True, "items": [item] * action_count}).encode() +def _top_level_throttle_response(request_number, request): + if request_number == 1: + return 429, b'{"status":429}', [("Retry-After", "2")] + + action_count = len(_bulk_action_lines(request["body"])) + items = b",".join( + b'{"create":{"status":201}}' for _ in range(action_count) + ) + return b'{"errors":false,"items":[' + items + b"]}" + + +def _item_throttle_response(retry_after_source): + def _response(request_number, request): + action_count = len(_bulk_action_lines(request["body"])) + + if request_number == 1: + assert action_count == 3 + response = ( + b'{"errors":true,"items":[' + b'{"create":{"status":201}},' + b'{"create":{"status":429}},' + b'{"create":{"status":409}}]}' + ) + if retry_after_source == "header": + return 200, response, [("Retry-After", "2")] + return 200, response, [], [("Retry-After", "2")] + + assert action_count == 1 + return b'{"errors":false,"items":[{"create":{"status":201}}]}' + + return _response + + @pytest.mark.parametrize( "config_file", [ @@ -350,6 +417,7 @@ def test_partial_bulk_retry_sends_only_unresolved_records(config_file): assert "bulk response reported errors: 1/3 items failed" in log_text assert "status=429 type='es_rejected_execution_exception'" in log_text assert "reason='bulk queue is full'; retrying 1 record(s)" in log_text + assert requests_seen[1]["received_at"] - requests_seen[0]["received_at"] >= 0.8 @pytest.mark.parametrize( @@ -396,3 +464,49 @@ def test_unrecoverable_bulk_errors_are_logged_and_not_retried(config_file): assert accounting["routed_bytes"] == 0.0 assert accounting["dropped_route_records"] == 1.0 assert accounting["dropped_route_bytes"] > 4000 + + +@pytest.mark.parametrize( + "config_file", + [ + "out_es_partial_bulk_retry.yaml", + "out_opensearch_partial_bulk_retry.yaml", + ], +) +def test_top_level_throttle_retries_complete_bulk_after_hint(config_file): + service = Service(config_file, response_factory=_top_level_throttle_response) + + try: + service.start() + requests_seen = service.wait_for_requests(2) + finally: + service.stop() + + assert len(_bulk_action_lines(requests_seen[0]["body"])) == 3 + assert len(_bulk_action_lines(requests_seen[1]["body"])) == 3 + assert requests_seen[1]["received_at"] - requests_seen[0]["received_at"] >= 1.8 + + +@pytest.mark.parametrize( + "config_file", + [ + "out_es_partial_bulk_retry.yaml", + "out_opensearch_partial_bulk_retry.yaml", + ], +) +@pytest.mark.parametrize("retry_after_source", ["header", "trailer"]) +def test_item_throttle_honors_retry_after(config_file, retry_after_source): + service = Service( + config_file, + response_factory=_item_throttle_response(retry_after_source), + ) + + try: + service.start() + requests_seen = service.wait_for_requests(2) + finally: + service.stop() + + assert len(_bulk_action_lines(requests_seen[0]["body"])) == 3 + assert len(_bulk_action_lines(requests_seen[1]["body"])) == 1 + assert requests_seen[1]["received_at"] - requests_seen[0]["received_at"] >= 1.8 diff --git a/tests/integration/scenarios/out_http/config/out_http_throttle.yaml b/tests/integration/scenarios/out_http/config/out_http_throttle.yaml new file mode 100644 index 00000000000..451de3ed7d0 --- /dev/null +++ b/tests/integration/scenarios/out_http/config/out_http_throttle.yaml @@ -0,0 +1,27 @@ +service: + flush: 0.2 + scheduler.base: 1 + scheduler.cap: 1 + log_level: debug + http_server: on + http_port: ${FLUENT_BIT_HTTP_MONITORING_PORT} + +pipeline: + inputs: + - name: dummy + tag: out_http + dummy: '{"message":"throttle me","source":"dummy"}' + samples: 1 + + outputs: + - name: http + match: out_http + host: 127.0.0.1 + port: ${TEST_SUITE_HTTP_PORT} + uri: /data + format: json + json_date_key: false + throttle: true + throttle.base: 1 + throttle.cap: 1 + retry_limit: 1 diff --git a/tests/integration/scenarios/out_http/config/out_http_throttle_body_key.yaml b/tests/integration/scenarios/out_http/config/out_http_throttle_body_key.yaml new file mode 100644 index 00000000000..c6c21910425 --- /dev/null +++ b/tests/integration/scenarios/out_http/config/out_http_throttle_body_key.yaml @@ -0,0 +1,26 @@ +service: + flush: 0.2 + log_level: info + http_server: on + http_port: ${FLUENT_BIT_HTTP_MONITORING_PORT} + +pipeline: + inputs: + - name: dummy + tag: out_http + dummy: '{"body":"first","headers":{}}' + samples: 1 + copies: 2 + + outputs: + - name: http + match: out_http + host: 127.0.0.1 + port: ${TEST_SUITE_HTTP_PORT} + uri: /data + body_key: $body + headers_key: $headers + throttle: true + throttle.base: 1 + throttle.cap: 1 + retry_limit: 1 diff --git a/tests/integration/scenarios/out_http/config/out_http_throttle_disabled.yaml b/tests/integration/scenarios/out_http/config/out_http_throttle_disabled.yaml new file mode 100644 index 00000000000..88eddc3f261 --- /dev/null +++ b/tests/integration/scenarios/out_http/config/out_http_throttle_disabled.yaml @@ -0,0 +1,23 @@ +service: + flush: 0.2 + log_level: info + http_server: on + http_port: ${FLUENT_BIT_HTTP_MONITORING_PORT} + +pipeline: + inputs: + - name: dummy + tag: out_http + dummy: '{"message":"retry me","source":"dummy"}' + samples: 1 + + outputs: + - name: http + match: out_http + host: 127.0.0.1 + port: ${TEST_SUITE_HTTP_PORT} + uri: /data + format: json + json_date_key: false + throttle: false + retry_limit: 1 diff --git a/tests/integration/scenarios/out_http/tests/test_out_http_001.py b/tests/integration/scenarios/out_http/tests/test_out_http_001.py index 48e23fe0f87..b19e063a1e0 100644 --- a/tests/integration/scenarios/out_http/tests/test_out_http_001.py +++ b/tests/integration/scenarios/out_http/tests/test_out_http_001.py @@ -192,6 +192,152 @@ def test_out_http_receiver_error_is_observable(): assert len(requests_seen) >= 1 +def test_out_http_429_uses_largest_retry_after_field(): + service = Service( + "out_http_throttle.yaml", + response_setup=lambda: configure_http_response( + status_code=429, + headers=[ + ("Retry-After", "1"), + ("retry-after", "invalid"), + ("RETRY-AFTER", "2"), + ], + ), + ) + service.start() + + requests_seen = service.wait_for_requests(2, timeout=10) + log_text = service.wait_for_log_message("ignored 1 malformed Retry-After field(s)") + service.stop() + + assert requests_seen[1]["received_at"] - requests_seen[0]["received_at"] >= 1.8 + assert "ignored 1 malformed Retry-After field(s)" in log_text + + +def test_out_http_429_without_hint_uses_local_throttle_delay(): + service = Service( + "out_http_throttle.yaml", + response_setup=lambda: configure_http_response(status_code=429), + ) + service.start() + + requests_seen = service.wait_for_requests(2, timeout=8) + service.stop() + + assert requests_seen[1]["received_at"] - requests_seen[0]["received_at"] >= 0.8 + + +def test_out_http_503_with_retry_after_throttles(): + service = Service( + "out_http_throttle.yaml", + response_setup=lambda: configure_http_response( + status_code=503, + headers=[("Retry-After", "2")], + ), + ) + service.start() + + requests_seen = service.wait_for_requests(2, timeout=10) + service.stop() + + assert requests_seen[1]["received_at"] - requests_seen[0]["received_at"] >= 1.8 + + +def test_out_http_503_without_valid_hint_preserves_retry_behavior(): + service = Service( + "out_http_throttle.yaml", + response_setup=lambda: configure_http_response( + status_code=503, + headers=[("Retry-After", "not-a-delay")], + ), + ) + service.start() + + requests_seen = service.wait_for_requests(2, timeout=8) + service.stop() + + assert len(requests_seen) == 2 + + +def test_out_http_408_preserves_retry_behavior(): + service = Service( + "out_http_throttle.yaml", + response_setup=lambda: configure_http_response(status_code=408), + ) + service.start() + + requests_seen = service.wait_for_requests(2, timeout=15) + service.stop() + + assert len(requests_seen) == 2 + + +def test_out_http_permanent_4xx_is_not_retried(): + service = Service( + "out_http_throttle.yaml", + response_setup=lambda: configure_http_response(status_code=400), + ) + service.start() + + service.wait_for_requests(1) + time.sleep(2) + requests_seen = list(data_storage["requests"]) + service.stop() + + assert len(requests_seen) == 1 + + +def test_out_http_accepts_205_without_retry(): + service = Service( + "out_http_throttle.yaml", + response_setup=lambda: configure_http_response(status_code=205), + ) + service.start() + + service.wait_for_requests(1) + time.sleep(2) + requests_seen = list(data_storage["requests"]) + service.stop() + + assert len(requests_seen) == 1 + + +def test_out_http_disabled_throttle_ignores_retry_after_cooldown(): + service = Service( + "out_http_throttle_disabled.yaml", + response_setup=lambda: configure_http_response( + status_code=429, + headers=[("Retry-After", "15")], + ), + ) + service.start() + + requests_seen = service.wait_for_requests(2, timeout=12) + service.stop() + + if not memory_check_enabled(): + assert requests_seen[1]["received_at"] - requests_seen[0]["received_at"] < 12 + + +def test_out_http_body_key_stops_each_attempt_after_throttle(): + service = Service( + "out_http_throttle_body_key.yaml", + response_setup=lambda: configure_http_response( + status_code=429, + headers=[("Retry-After", "4")], + ), + ) + service.start() + + try: + requests_seen = service.wait_for_requests(3, timeout=18) + finally: + service.stop() + + request_times = [request["received_at"] for request in requests_seen[:3]] + assert all(later - earlier >= 3.8 for earlier, later in zip(request_times, request_times[1:])) + + def test_out_http_oauth2_basic_adds_bearer_token(): service = Service("out_http_oauth2_basic.yaml") service.start() diff --git a/tests/integration/scenarios/out_opentelemetry/config/out_otel_grpc_logs_throttle.yaml b/tests/integration/scenarios/out_opentelemetry/config/out_otel_grpc_logs_throttle.yaml new file mode 100644 index 00000000000..292c7c1248e --- /dev/null +++ b/tests/integration/scenarios/out_opentelemetry/config/out_otel_grpc_logs_throttle.yaml @@ -0,0 +1,25 @@ +service: + flush: 1 + grace: 1 + scheduler.base: 1 + scheduler.cap: 1 + http_server: on + http_port: ${FLUENT_BIT_HTTP_MONITORING_PORT} + +pipeline: + inputs: + - name: dummy + samples: 1 + dummy: '{"message":"gRPC throttle"}' + + outputs: + - name: opentelemetry + match: "*" + host: 127.0.0.1 + port: ${TEST_SUITE_HTTP_PORT} + http2: on + grpc: on + retry_limit: 2 + throttle: on + throttle.base: 1 + throttle.cap: 1 diff --git a/tests/integration/scenarios/out_opentelemetry/config/out_otel_grpc_metrics_max_datapoints.conf b/tests/integration/scenarios/out_opentelemetry/config/out_otel_grpc_metrics_max_datapoints.conf index 9e476823877..260a8aa2176 100644 --- a/tests/integration/scenarios/out_opentelemetry/config/out_otel_grpc_metrics_max_datapoints.conf +++ b/tests/integration/scenarios/out_opentelemetry/config/out_otel_grpc_metrics_max_datapoints.conf @@ -13,6 +13,7 @@ match * host 127.0.0.1 port ${TEST_SUITE_HTTP_PORT} + throttle on http2 on grpc on grpc_metrics_uri /batched.metrics.v1.Metrics/Export diff --git a/tests/integration/scenarios/out_opentelemetry/config/out_otel_http2_ipv6_throttle.yaml b/tests/integration/scenarios/out_opentelemetry/config/out_otel_http2_ipv6_throttle.yaml new file mode 100644 index 00000000000..39eecfb50af --- /dev/null +++ b/tests/integration/scenarios/out_opentelemetry/config/out_otel_http2_ipv6_throttle.yaml @@ -0,0 +1,25 @@ +service: + flush: 1 + grace: 1 + scheduler.base: 1 + scheduler.cap: 1 + http_server: on + http_port: ${FLUENT_BIT_HTTP_MONITORING_PORT} + +pipeline: + inputs: + - name: dummy + tag: ipv6_http2_throttle + samples: 1 + dummy: '{"message":"HTTP/2 throttle"}' + + outputs: + - name: opentelemetry + match: ipv6_http2_throttle + host: "::1" + port: ${TEST_SUITE_HTTP_PORT} + http2: on + retry_limit: 2 + throttle: on + throttle.base: 1 + throttle.cap: 1 diff --git a/tests/integration/scenarios/out_opentelemetry/config/out_otel_http_logs_throttle.yaml b/tests/integration/scenarios/out_opentelemetry/config/out_otel_http_logs_throttle.yaml new file mode 100644 index 00000000000..9dff4b3eb84 --- /dev/null +++ b/tests/integration/scenarios/out_opentelemetry/config/out_otel_http_logs_throttle.yaml @@ -0,0 +1,24 @@ +service: + flush: 1 + grace: 1 + scheduler.base: 1 + scheduler.cap: 1 + http_server: on + http_port: ${FLUENT_BIT_HTTP_MONITORING_PORT} + +pipeline: + inputs: + - name: dummy + samples: 1 + dummy: '{"message":"HTTP throttle"}' + + outputs: + - name: opentelemetry + match: "*" + host: 127.0.0.1 + port: ${TEST_SUITE_HTTP_PORT} + logs_uri: /v1/logs + retry_limit: 2 + throttle: on + throttle.base: 1 + throttle.cap: 1 diff --git a/tests/integration/scenarios/out_opentelemetry/config/out_otel_http_logs_throttle_long_base.yaml b/tests/integration/scenarios/out_opentelemetry/config/out_otel_http_logs_throttle_long_base.yaml new file mode 100644 index 00000000000..686c48bbed6 --- /dev/null +++ b/tests/integration/scenarios/out_opentelemetry/config/out_otel_http_logs_throttle_long_base.yaml @@ -0,0 +1,24 @@ +service: + flush: 1 + grace: 1 + scheduler.base: 1 + scheduler.cap: 1 + http_server: on + http_port: ${FLUENT_BIT_HTTP_MONITORING_PORT} + +pipeline: + inputs: + - name: dummy + samples: 1 + dummy: '{"message":"HTTP retry"}' + + outputs: + - name: opentelemetry + match: "*" + host: 127.0.0.1 + port: ${TEST_SUITE_HTTP_PORT} + logs_uri: /v1/logs + retry_limit: 2 + throttle: on + throttle.base: 5 + throttle.cap: 5 diff --git a/tests/integration/scenarios/out_opentelemetry/config/out_otel_http_metrics_max_datapoints.conf b/tests/integration/scenarios/out_opentelemetry/config/out_otel_http_metrics_max_datapoints.conf index 4094c2b39b7..ce8a71c8960 100644 --- a/tests/integration/scenarios/out_opentelemetry/config/out_otel_http_metrics_max_datapoints.conf +++ b/tests/integration/scenarios/out_opentelemetry/config/out_otel_http_metrics_max_datapoints.conf @@ -13,5 +13,6 @@ match * host 127.0.0.1 port ${TEST_SUITE_HTTP_PORT} + throttle on metrics_uri /batched/metrics metrics_max_datapoints 4 diff --git a/tests/integration/scenarios/out_opentelemetry/tests/test_out_opentelemetry_001.py b/tests/integration/scenarios/out_opentelemetry/tests/test_out_opentelemetry_001.py index bb25d0ef150..fb6ff55d69e 100644 --- a/tests/integration/scenarios/out_opentelemetry/tests/test_out_opentelemetry_001.py +++ b/tests/integration/scenarios/out_opentelemetry/tests/test_out_opentelemetry_001.py @@ -7,14 +7,23 @@ import socket import threading from collections import Counter +import time +import grpc import requests import pytest from google.protobuf import json_format +from google.protobuf.any_pb2 import Any +from google.protobuf.duration_pb2 import Duration +from google.rpc.error_details_pb2 import RetryInfo +from google.rpc.status_pb2 import Status from h2.config import H2Configuration from h2.connection import H2Connection from h2.events import DataReceived, RequestReceived, StreamEnded -from opentelemetry.proto.collector.logs.v1.logs_service_pb2 import ExportLogsServiceRequest +from opentelemetry.proto.collector.logs.v1.logs_service_pb2 import ( + ExportLogsServiceRequest, + ExportLogsServiceResponse, +) from opentelemetry.proto.collector.metrics.v1.metrics_service_pb2 import ExportMetricsServiceRequest from opentelemetry.proto.collector.trace.v1.trace_service_pb2 import ExportTraceServiceRequest @@ -25,6 +34,7 @@ ) from server.otlp_server import ( configure_otlp_grpc_methods, + configure_otlp_grpc_responses, configure_otlp_response, data_storage, otlp_server_run, @@ -36,6 +46,36 @@ logger = logging.getLogger(__name__) +def _configure_http_throttle(status_code, retry_after): + headers = [] + if retry_after is not None: + headers.append(("Retry-After", retry_after)) + configure_otlp_response( + status_codes=[status_code, 200], + headers=headers, + ) + + +def _configure_grpc_throttle(status_code, retry_delay): + configure_otlp_grpc_responses( + [ + {"status": status_code, "retry_delay": retry_delay}, + ] + ) + + +def _grpc_retry_status_details(status_code, retry_delay, message): + retry_info = RetryInfo(retry_delay=Duration(seconds=retry_delay)) + detail = Any() + detail.Pack(retry_info) + status = Status( + code=status_code, + message=message, + details=[detail], + ) + return base64.b64encode(status.SerializeToString()).decode("ascii") + + def _repo_relative(*parts): return os.path.abspath(os.path.join(os.path.dirname(__file__), *parts)) @@ -84,6 +124,7 @@ def __init__( use_tls=False, grpc_methods=None, use_oauth_server=False, + response_setup=None, ): self.config_file = _repo_relative("../config", config_file) cert_dir = _repo_relative("../../in_splunk/certificate") @@ -93,6 +134,7 @@ def __init__( self.use_tls = use_tls self.grpc_methods = grpc_methods or {} self.use_oauth_server = use_oauth_server + self.response_setup = response_setup self.oauth_server_port = None self.service = FluentBitTestService( self.config_file, @@ -137,6 +179,8 @@ def _start_receiver(self, service): tls_key_file=self.tls_key_file, use_grpc=self.receiver_mode == "grpc", ) + if self.response_setup is not None: + self.response_setup() if self.receiver_mode == "grpc": self._wait_for_tcp_port(service.test_suite_http_port) @@ -188,6 +232,24 @@ def wait_for_requests(self, minimum_count, timeout=10): description=f"{minimum_count} OTLP requests", ) + def assert_no_additional_requests(self, expected_count, timeout=2): + deadline = time.monotonic() + timeout + + while time.monotonic() < deadline: + actual_count = len(data_storage["requests"]) + if actual_count != expected_count: + raise AssertionError( + f"expected {expected_count} OTLP requests, " + f"saw {actual_count}" + ) + time.sleep(0.1) + + actual_count = len(data_storage["requests"]) + if actual_count != expected_count: + raise AssertionError( + f"expected {expected_count} OTLP requests, saw {actual_count}" + ) + def wait_for_oauth_requests(self, minimum_count, timeout=10): return self.service.wait_for_condition( lambda: http_data_storage["requests"] if len(http_data_storage["requests"]) >= minimum_count else None, @@ -267,17 +329,20 @@ def _build_signal_payload_from_dict(self, payload_dict, signal_type): class IPv6Http2OtlpReceiver: - def __init__(self, port): + def __init__(self, port, responses=None, host="::1"): self.port = port + self.host = host self.server_socket = None self.thread = None self.requests = [] + self.responses = list(responses or []) self.stop_event = threading.Event() def start(self): - self.server_socket = socket.socket(socket.AF_INET6, socket.SOCK_STREAM) + family = socket.AF_INET6 if ":" in self.host else socket.AF_INET + self.server_socket = socket.socket(family, socket.SOCK_STREAM) self.server_socket.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1) - self.server_socket.bind(("::1", self.port)) + self.server_socket.bind((self.host, self.port)) self.server_socket.listen(5) self.server_socket.settimeout(0.5) self.thread = threading.Thread(target=self._serve, daemon=True) @@ -340,6 +405,7 @@ def _handle_connection(self, client_socket): event.stream_id, ) elif isinstance(event, StreamEnded): + request["received_at"] = time.monotonic() self.requests.append(request) self._send_response(connection, event.stream_id) client_socket.sendall(connection.data_to_send()) @@ -351,23 +417,42 @@ def _handle_connection(self, client_socket): def _send_response(self, connection, stream_id): body = b"{}" - - connection.send_headers( - stream_id, - [ - (":status", "200"), - ("content-type", "application/json"), - ("content-length", str(len(body))), - ], - ) - connection.send_data(stream_id, body, end_stream=True) + status = "200" + headers = [] + trailers = [] + content_type = "application/json" + + if self.responses: + response = self.responses.pop(0) + status = str(response["status"]) + headers = response.get("headers", []) + trailers = response.get("trailers", []) + body = response.get("body", body) + content_type = response.get("content_type", content_type) + + response_headers = [ + (":status", status), + ("content-type", content_type), + ] + headers + if not trailers: + response_headers.append(("content-length", str(len(body)))) + connection.send_headers(stream_id, response_headers) + if body: + connection.send_data(stream_id, body) + if trailers: + connection.send_headers(stream_id, trailers, end_stream=True) + else: + connection.end_stream(stream_id) class Http2IPv6Service: - def __init__(self): - self.config_file = _repo_relative("../config", "out_otel_http2_ipv6_logs.yaml") + def __init__(self, config_file="out_otel_http2_ipv6_logs.yaml", responses=None, + host="::1"): + self.config_file = _repo_relative("../config", config_file) self.receiver = None self.test_suite_http_port = None + self.responses = responses + self.host = host self.service = FluentBitTestService( self.config_file, pre_start=self._start_receiver, @@ -376,7 +461,9 @@ def __init__(self): def _start_receiver(self, service): self.test_suite_http_port = service.test_suite_http_port - self.receiver = IPv6Http2OtlpReceiver(service.test_suite_http_port) + self.receiver = IPv6Http2OtlpReceiver( + service.test_suite_http_port, self.responses, self.host + ) self.receiver.start() def _stop_receiver(self, service): @@ -397,7 +484,7 @@ def wait_for_requests(self, minimum_count, timeout=10): else None, timeout=timeout, interval=0.5, - description=f"{minimum_count} IPv6 HTTP/2 OTLP requests", + description=f"{minimum_count} HTTP/2 OTLP requests", ) @@ -607,6 +694,18 @@ def resource(service_name, metric): } +def _build_single_resource_batched_metrics_payload(): + payload = _build_batched_metrics_payload() + resource_metrics = payload["resource_metrics"] + gauge_points = resource_metrics[0]["scope_metrics"][0]["metrics"][0]["gauge"] + gauge_points["data_points"].extend( + resource_metrics[1]["scope_metrics"][0]["metrics"][0]["sum"]["data_points"] + ) + payload["resource_metrics"] = [resource_metrics[0]] + + return payload + + def iter_metric_points_with_resource(output): data_keys = ("gauge", "sum", "histogram", "exponentialHistogram", "summary") @@ -901,6 +1000,26 @@ def test_out_opentelemetry_http2_ipv6_authority_header(): assert request_seen["headers"][":authority"] == f"[::1]:{service.test_suite_http_port}" +def test_out_opentelemetry_http2_429_honors_retry_after(): + if not ipv6_loopback_available(): + pytest.skip("IPv6 loopback is not available") + + service = Http2IPv6Service( + "out_otel_http2_ipv6_throttle.yaml", + responses=[ + {"status": 429, "headers": [("retry-after", "2")]}, + {"status": 200}, + ], + ) + try: + service.start() + requests_seen = service.wait_for_requests(2, timeout=30) + finally: + service.stop() + + assert requests_seen[1]["received_at"] - requests_seen[0]["received_at"] >= 1.8 + + @pytest.mark.parametrize( "config_file,receiver_mode", [ @@ -1037,13 +1156,7 @@ def test_out_opentelemetry_metrics_max_datapoints( def test_out_opentelemetry_metrics_partial_success_is_not_retried(): - payload = _build_batched_metrics_payload() - resource_metrics = payload["resource_metrics"] - gauge_points = resource_metrics[0]["scope_metrics"][0]["metrics"][0]["gauge"] - gauge_points["data_points"].extend( - resource_metrics[1]["scope_metrics"][0]["metrics"][0]["sum"]["data_points"] - ) - payload["resource_metrics"] = [resource_metrics[0]] + payload = _build_single_resource_batched_metrics_payload() service = Service("out_otel_http_metrics_max_datapoints.conf") service.start() @@ -1058,6 +1171,7 @@ def test_out_opentelemetry_metrics_partial_success_is_not_retried(): assert len(metrics_seen) == 2 assert len(requests_seen) == 2 + assert len(data_storage["requests"]) == 2 batch_series = [] for export_request in metrics_seen: @@ -1076,6 +1190,60 @@ def test_out_opentelemetry_metrics_partial_success_is_not_retried(): assert {8, 9, 10}.isdisjoint(set().union(*batch_series)) +@pytest.mark.parametrize( + "receiver_mode,status_code", + [ + ("http", 429), + ("http", 503), + ("grpc", grpc.StatusCode.RESOURCE_EXHAUSTED), + ("grpc", grpc.StatusCode.UNAVAILABLE), + ], + ids=["http-429", "http-503", "grpc-resource-exhausted", "grpc-unavailable"], +) +def test_out_opentelemetry_later_metrics_batch_throttle_is_retried( + receiver_mode, + status_code, +): + if receiver_mode == "grpc": + config_file = "out_otel_grpc_metrics_max_datapoints.conf" + request_path = "/batched.metrics.v1.Metrics/Export" + grpc_methods = {"metrics": request_path} + else: + config_file = "out_otel_http_metrics_max_datapoints.conf" + request_path = "/batched/metrics" + grpc_methods = None + + service = Service( + config_file, + receiver_mode=receiver_mode, + grpc_methods=grpc_methods, + ) + service.start() + try: + if receiver_mode == "grpc": + configure_otlp_grpc_responses( + [ + None, + {"status": status_code, "retry_delay": 2}, + ] + ) + else: + configure_otlp_response( + status_codes=[200, status_code, 200], + headers=[("Retry-After", "2")], + ) + + service.send_payload_dict(_build_single_resource_batched_metrics_payload(), "metrics") + requests_seen = service.wait_for_requests(5, timeout=15) + _wait_for_log_message(service, "throttled flush") + finally: + service.stop() + + assert len(data_storage["requests"]) == 5 + assert {request["path"] for request in requests_seen} == {request_path} + assert requests_seen[2]["received_at"] - requests_seen[1]["received_at"] >= 1.8 + + def test_out_opentelemetry_traces_uri(): service = Service("out_otel_http_traces.yaml") service.start() @@ -1391,3 +1559,175 @@ def test_out_opentelemetry_logs_max_scopes_enforcement(): output = json.loads(json_format.MessageToJson(logs_seen[0])) assert len(output["resourceLogs"]) == 4 assert all(len(resource_log["scopeLogs"]) == 1 for resource_log in output["resourceLogs"]) + + +def test_out_opentelemetry_http_429_honors_retry_after(): + service = Service( + "out_otel_http_logs_throttle.yaml", + response_setup=lambda: _configure_http_throttle(429, "2"), + ) + + try: + service.start() + requests_seen = service.wait_for_requests(2, timeout=10) + finally: + service.stop() + + assert requests_seen[1]["received_at"] - requests_seen[0]["received_at"] >= 1.8 + + +@pytest.mark.parametrize("status_code", [502, 504]) +def test_out_opentelemetry_gateway_retry_is_not_throttle(status_code): + service = Service( + "out_otel_http_logs_throttle_long_base.yaml", + response_setup=lambda: _configure_http_throttle(status_code, "5"), + ) + + try: + service.start() + requests_seen = service.wait_for_requests(2, timeout=8) + finally: + service.stop() + + assert requests_seen[1]["received_at"] - requests_seen[0]["received_at"] < 3 + + +def test_out_opentelemetry_http_503_with_invalid_hint_is_not_throttle(): + service = Service( + "out_otel_http_logs_throttle_long_base.yaml", + response_setup=lambda: _configure_http_throttle(503, "invalid"), + ) + + try: + service.start() + requests_seen = service.wait_for_requests(2, timeout=8) + finally: + service.stop() + + assert requests_seen[1]["received_at"] - requests_seen[0]["received_at"] < 3 + + +def test_out_opentelemetry_grpc_unavailable_honors_retry_info(): + service = Service( + "out_otel_grpc_logs_throttle.yaml", + receiver_mode="grpc", + response_setup=lambda: _configure_grpc_throttle( + grpc.StatusCode.UNAVAILABLE, 2 + ), + ) + + try: + service.start() + requests_seen = service.wait_for_requests(2, timeout=10) + finally: + service.stop() + + assert requests_seen[1]["received_at"] - requests_seen[0]["received_at"] >= 1.8 + + +@pytest.mark.parametrize("grpc_status", [8, 14]) +@pytest.mark.parametrize( + "message,expected_remainder", + [ + ("scripted OTLP response", 2), + ("scripted OTLP response!", 3), + ], +) +def test_out_opentelemetry_grpc_honors_unpadded_retry_info( + grpc_status, message, expected_remainder +): + status_details = _grpc_retry_status_details(grpc_status, 2, message).rstrip("=") + assert len(status_details) % 4 == expected_remainder + + service = Http2IPv6Service( + "out_otel_grpc_logs_throttle.yaml", + host="127.0.0.1", + responses=[ + { + "status": 200, + "content_type": "application/grpc", + "body": b"", + "trailers": [ + ("grpc-status", str(grpc_status)), + ("grpc-status-details-bin", status_details), + ], + }, + { + "status": 200, + "content_type": "application/grpc", + "body": b"", + "trailers": [("grpc-status", "0")], + }, + ], + ) + + try: + service.start() + requests_seen = service.wait_for_requests(2, timeout=10) + finally: + service.stop() + + assert requests_seen[1]["received_at"] - requests_seen[0]["received_at"] >= 1.8 + + +@pytest.mark.parametrize("retry_delay", [None, "invalid"]) +def test_out_opentelemetry_grpc_resource_exhausted_requires_retry_info(retry_delay): + service = Service( + "out_otel_grpc_logs_throttle.yaml", + receiver_mode="grpc", + response_setup=lambda: _configure_grpc_throttle( + grpc.StatusCode.RESOURCE_EXHAUSTED, retry_delay + ), + ) + + try: + service.start() + service.wait_for_requests(1) + service.assert_no_additional_requests(1) + finally: + service.stop() + + assert len(data_storage["requests"]) == 1 + + +def test_out_opentelemetry_populated_partial_success_is_not_retried(): + response = ExportLogsServiceResponse() + response.partial_success.rejected_log_records = 1 + response.partial_success.error_message = "scripted partial acceptance" + service = Service( + "out_otel_http_logs_throttle.yaml", + response_setup=lambda: configure_otlp_response( + status_code=200, + body=response.SerializeToString(), + content_type="application/x-protobuf", + ), + ) + + try: + service.start() + service.wait_for_requests(1) + service.assert_no_additional_requests(1) + finally: + service.stop() + + assert len(data_storage["requests"]) == 1 + + +@pytest.mark.parametrize("late_request", [False, True]) +def test_no_additional_requests_checks_deadline(monkeypatch, late_request): + clock = iter([0.0, 0.0, 1.0]) + requests_seen = [] + monkeypatch.setitem(data_storage, "requests", requests_seen) + monkeypatch.setattr(time, "monotonic", lambda: next(clock)) + + def finish_interval(_seconds): + if late_request: + requests_seen.append({"received_at": 0.99}) + + monkeypatch.setattr(time, "sleep", finish_interval) + service = Service.__new__(Service) + if late_request: + with pytest.raises(AssertionError, match="expected 0 OTLP requests, saw 1"): + service.assert_no_additional_requests(0, timeout=1) + else: + service.assert_no_additional_requests(0, timeout=1) diff --git a/tests/integration/src/server/http_server.py b/tests/integration/src/server/http_server.py index f06fe909a06..4a667b3ea7f 100644 --- a/tests/integration/src/server/http_server.py +++ b/tests/integration/src/server/http_server.py @@ -44,6 +44,7 @@ "fragment_delay_seconds": 0, "hang_before_response": False, "hang_after_fragment_index": None, + "headers": [], } oauth_token_response = { "status_code": 200, @@ -96,6 +97,7 @@ def reset_http_server_state(): "fragment_delay_seconds": 0, "hang_before_response": False, "hang_after_fragment_index": None, + "headers": [], } ) oauth_token_response.update( @@ -120,7 +122,7 @@ def configure_http_response(*, status_code=UNSET, body=UNSET, content_type=UNSET delay_seconds=UNSET, stream_fragments=UNSET, fragment_delay_seconds=UNSET, hang_before_response=UNSET, - hang_after_fragment_index=UNSET): + hang_after_fragment_index=UNSET, headers=UNSET): if status_code is not UNSET: response_config["status_code"] = status_code if body is not UNSET: @@ -137,6 +139,8 @@ def configure_http_response(*, status_code=UNSET, body=UNSET, content_type=UNSET response_config["hang_before_response"] = hang_before_response if hang_after_fragment_index is not UNSET: response_config["hang_after_fragment_index"] = hang_after_fragment_index + if headers is not UNSET: + response_config["headers"] = list(headers) def configure_oauth_token_response(*, status_code=UNSET, body=UNSET, @@ -185,12 +189,16 @@ def _stream_fragments(config): def _build_streaming_response(config): - return Response( + response = Response( _stream_fragments(config), status=config["status_code"], content_type=config["content_type"], direct_passthrough=True, ) + for name, value in config.get("headers", []): + response.headers.add(name, value) + + return response def _build_response(): @@ -206,13 +214,19 @@ def _build_response(): body = response_config["body"] if isinstance(body, (dict, list)): - return jsonify(body), response_config["status_code"] + response = jsonify(body) + response.status_code = response_config["status_code"] + else: + response = Response( + body, + status=response_config["status_code"], + content_type=response_config["content_type"], + ) - return Response( - body, - status=response_config["status_code"], - content_type=response_config["content_type"], - ) + for name, value in response_config.get("headers", []): + response.headers.add(name, value) + + return response def _record_request(): @@ -225,6 +239,7 @@ def _record_request(): data_storage["payloads"].append(data) data_storage["requests"].append( { + "received_at": time.monotonic(), "path": request.path, "query_string": request.query_string.decode("utf-8", errors="replace"), "method": request.method, diff --git a/tests/integration/src/server/otlp_server.py b/tests/integration/src/server/otlp_server.py index e4a755ac832..8cadfb13bd7 100644 --- a/tests/integration/src/server/otlp_server.py +++ b/tests/integration/src/server/otlp_server.py @@ -23,7 +23,11 @@ import grpc from flask import Flask, Response, jsonify, request +from google.protobuf.any_pb2 import Any +from google.protobuf.duration_pb2 import Duration from google.protobuf.message import DecodeError +from google.rpc.error_details_pb2 import RetryInfo +from google.rpc.status_pb2 import Status from opentelemetry.proto.collector.logs.v1.logs_service_pb2 import ( ExportLogsServiceRequest, ExportLogsServiceResponse, @@ -46,7 +50,9 @@ "body": {"status": "received"}, "content_type": "application/json", "delay_seconds": 0, + "headers": [], } +grpc_response_config = {"responses": []} grpc_method_paths = { "logs": "/opentelemetry.proto.collector.logs.v1.LogsService/Export", "metrics": "/opentelemetry.proto.collector.metrics.v1.MetricsService/Export", @@ -70,8 +76,10 @@ def reset_otlp_server_state(): "body": {"status": "received"}, "content_type": "application/json", "delay_seconds": 0, + "headers": [], } ) + grpc_response_config["responses"] = [] grpc_method_paths.update( { "logs": "/opentelemetry.proto.collector.logs.v1.LogsService/Export", @@ -89,6 +97,7 @@ def configure_otlp_response( body=None, content_type=None, delay_seconds=None, + headers=None, ): if status_code is not None: response_config["status_code"] = status_code @@ -100,6 +109,12 @@ def configure_otlp_response( response_config["content_type"] = content_type if delay_seconds is not None: response_config["delay_seconds"] = delay_seconds + if headers is not None: + response_config["headers"] = list(headers) + + +def configure_otlp_grpc_responses(responses): + grpc_response_config["responses"] = list(responses) def configure_otlp_grpc_methods(*, logs=None, metrics=None, traces=None): @@ -122,13 +137,18 @@ def _build_response(): body = response_config["body"] if isinstance(body, (dict, list)): - return jsonify(body), status_code + response = jsonify(body) + response.status_code = status_code + else: + response = Response( + body, + status=status_code, + content_type=response_config["content_type"], + ) + for name, value in response_config["headers"]: + response.headers.add(name, value) - return Response( - body, - status=status_code, - content_type=response_config["content_type"], - ) + return response def _record_request(*, path, headers, raw_payload, transport): @@ -139,6 +159,7 @@ def _record_request(*, path, headers, raw_payload, transport): "raw_size": len(raw_payload), "raw_payload": raw_payload, "transport": transport, + "received_at": time.monotonic(), } ) @@ -236,6 +257,8 @@ def run_server(port=4317, *, use_tls=False, tls_crt_file=None, tls_key_file=None def _build_grpc_handler(signal_name, message_type, response_type): def _handler(request_message, context): + response_spec = None + data_storage[signal_name].append(request_message) _record_request( path=context._rpc_event.call_details.method.decode(), @@ -243,6 +266,30 @@ def _handler(request_message, context): raw_payload=request_message.SerializeToString(), transport="grpc", ) + if grpc_response_config["responses"]: + response_spec = grpc_response_config["responses"].pop(0) + if response_spec is not None: + status_code = response_spec["status"] + retry_delay = response_spec.get("retry_delay") + if retry_delay == "invalid": + context.set_trailing_metadata( + (("grpc-status-details-bin", b"invalid-status-details"),) + ) + elif retry_delay is not None: + retry_info = RetryInfo( + retry_delay=Duration(seconds=retry_delay) + ) + detail = Any() + detail.Pack(retry_info) + status = Status( + code=status_code.value[0], + message="scripted OTLP response", + details=[detail], + ) + context.set_trailing_metadata( + (("grpc-status-details-bin", status.SerializeToString()),) + ) + context.abort(status_code, "scripted OTLP response") return response_type() return grpc.unary_unary_rpc_method_handler( diff --git a/tests/integration/src/utils/fluent_bit_manager.py b/tests/integration/src/utils/fluent_bit_manager.py index 55a90bf18ca..3f89430545b 100644 --- a/tests/integration/src/utils/fluent_bit_manager.py +++ b/tests/integration/src/utils/fluent_bit_manager.py @@ -173,9 +173,10 @@ def __init__(self, config_path=None, binary_path=None, *, shutdown_timeout=None) self.output_handle = None def set_http_monitoring_port(self, env_var_name, starting_port=0): - port = find_available_port(starting_port) - os.environ[env_var_name] = str(port) - self.http_monitoring_port = str(port) + if self.http_monitoring_port is None: + self.http_monitoring_port = str(find_available_port(starting_port)) + + os.environ[env_var_name] = self.http_monitoring_port def start(self): if not self.config_path or not os.path.exists(self.config_path): diff --git a/tests/integration/src/utils/test_service.py b/tests/integration/src/utils/test_service.py index 841bb459bb6..585f0f0fa44 100644 --- a/tests/integration/src/utils/test_service.py +++ b/tests/integration/src/utils/test_service.py @@ -81,6 +81,7 @@ def start(self): self.flb = FluentBitManager(self.config_path, shutdown_timeout=self.shutdown_timeout) self.flb_listener_port = self._allocate_port() self.test_suite_http_port = self._allocate_port() + self.flb.http_monitoring_port = str(self._allocate_port()) self._set_env("FLUENT_BIT_TEST_LISTENER_PORT", str(self.flb_listener_port)) self._set_env("TEST_SUITE_HTTP_PORT", str(self.test_suite_http_port)) diff --git a/tests/integration/src/utils/valgrind.py b/tests/integration/src/utils/valgrind.py index 506f4fe2620..60cccf774c0 100644 --- a/tests/integration/src/utils/valgrind.py +++ b/tests/integration/src/utils/valgrind.py @@ -3,6 +3,60 @@ from pathlib import Path +_ERROR_CONTEXT_HEADER = re.compile( + r"^==\d+==\s+(?:" + r"Invalid (?:read|write|free|alignment)|" + r"Use of uninitialised|" + r"Conditional jump or move depends on uninitialised|" + r"Syscall param|" + r"Mismatched free|" + r"Source and destination overlap|" + r"Jump to invalid address|" + r"Bad permissions for mapped region|" + r"Argument .* has a fishy|" + r"Realloc size zero" + r")" +) +_VALGRIND_BLANK_LINE = re.compile(r"^==\d+==\s*$") +_VALGRIND_SUMMARY_HEADER = re.compile( + r"^==\d+==\s+(?:HEAP SUMMARY|LEAK SUMMARY|ERROR SUMMARY):" +) + + +def _extract_error_contexts(text, *, max_contexts=3, max_lines=20, max_chars=6000): + lines = text.splitlines() + contexts = [] + index = 0 + + while index < len(lines) and len(contexts) < max_contexts: + if not _ERROR_CONTEXT_HEADER.match(lines[index]): + index += 1 + continue + + context = [] + while index < len(lines) and len(context) < max_lines: + line = lines[index] + if context and (_VALGRIND_BLANK_LINE.match(line) or + _VALGRIND_SUMMARY_HEADER.match(line)): + break + context.append(line) + index += 1 + + if ( + len(context) == max_lines + and index < len(lines) + and not _VALGRIND_BLANK_LINE.match(lines[index]) + ): + context.append("... (Valgrind context truncated)") + contexts.append("\n".join(context)) + + excerpt = "\n\n".join(contexts) + if len(excerpt) > max_chars: + excerpt = excerpt[:max_chars].rstrip() + "\n... (Valgrind excerpt truncated)" + + return excerpt + + @dataclass class ValgrindSummary: definitely_lost: int = 0 @@ -77,8 +131,11 @@ def assert_valgrind_clean(log_path, *, allow_definitely_lost=0, allow_error_coun problems.append(f"errors={summary.error_count}") if problems: - raise AssertionError( - f"Valgrind issues found in {log_path}: " + ", ".join(problems) - ) + message = f"Valgrind issues found in {log_path}: " + ", ".join(problems) + text = Path(log_path).read_text(encoding="utf-8") + error_contexts = _extract_error_contexts(text) + if error_contexts: + message += "\nValgrind error context:\n" + error_contexts + raise AssertionError(message) return summary diff --git a/tests/integration/test_valgrind_utils.py b/tests/integration/test_valgrind_utils.py index 833d04b388d..5fc81bb8e5f 100644 --- a/tests/integration/test_valgrind_utils.py +++ b/tests/integration/test_valgrind_utils.py @@ -45,3 +45,33 @@ def test_assert_valgrind_clean_rejects_leaks(tmp_path): with pytest.raises(AssertionError): assert_valgrind_clean(log_path) + +def test_assert_valgrind_clean_reports_bounded_error_context(tmp_path): + log_path = tmp_path / "valgrind.log" + stack_lines = [f"==42== by 0x{i:08X}: frame_{i} (worker.c:{i})" for i in range(30)] + log_path.write_text( + "\n".join( + [ + "==42== Invalid read of size 8", + "==42== at 0x00123456: flush_complete (output.c:123)", + *stack_lines, + "==42==", + "==42== HEAP SUMMARY:", + "==42== in use at exit: 0 bytes in 0 blocks", + "==42== ERROR SUMMARY: 2 errors from 1 contexts (suppressed: 0 from 0)", + ] + ), + encoding="utf-8", + ) + + with pytest.raises(AssertionError) as raised: + assert_valgrind_clean(log_path) + + message = str(raised.value) + assert "errors=2" in message + assert "Invalid read of size 8" in message + assert "flush_complete (output.c:123)" in message + assert "frame_17" in message + assert "frame_18" not in message + assert "Valgrind context truncated" in message + assert "HEAP SUMMARY" not in message diff --git a/tests/integration/tests/test_test_service.py b/tests/integration/tests/test_test_service.py new file mode 100644 index 00000000000..ecbfb908432 --- /dev/null +++ b/tests/integration/tests/test_test_service.py @@ -0,0 +1,52 @@ +# Fluent Bit +# ========== +# Copyright (C) 2015-2026 The Fluent Bit Authors +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +from utils import fluent_bit_manager as manager_module +from utils import test_service as service_module +from utils.fluent_bit_manager import ENV_FLB_HTTP_MONITORING_PORT +from utils.fluent_bit_manager import FluentBitManager +from utils.test_service import FluentBitTestService + + +def test_service_reserves_distinct_monitoring_port(monkeypatch): + allocated_ports = iter([31001, 31002, 31001, 31003]) + + monkeypatch.delenv(ENV_FLB_HTTP_MONITORING_PORT, raising=False) + monkeypatch.setattr( + service_module, + "find_available_port", + lambda starting_port=0: next(allocated_ports), + ) + monkeypatch.setattr(manager_module, "find_available_port", lambda starting_port=0: 31001) + monkeypatch.setattr( + FluentBitManager, + "start", + lambda manager: manager.set_http_monitoring_port(ENV_FLB_HTTP_MONITORING_PORT), + ) + + service = FluentBitTestService("/tmp/fluent-bit.yaml") + + try: + service.start() + + assert service.flb_listener_port == 31001 + assert service.test_suite_http_port == 31002 + assert int(service.flb.http_monitoring_port) not in { + service.flb_listener_port, + service.test_suite_http_port, + } + finally: + service.stop() diff --git a/tests/internal/CMakeLists.txt b/tests/internal/CMakeLists.txt index 2f888009415..66b7c2d9f44 100644 --- a/tests/internal/CMakeLists.txt +++ b/tests/internal/CMakeLists.txt @@ -19,6 +19,7 @@ set(UNIT_TESTS_FILES unit_sizes.c hashtable.c http_client.c + http_retry_after.c utils.c gzip.c zstd.c @@ -66,6 +67,8 @@ set(UNIT_TESTS_FILES storage_dlq.c engine_adaptive_flush.c engine_dispatch.c + output_thread.c + output_throttle.c search_bulk.c ) @@ -289,6 +292,10 @@ endfunction(prepare_unit_tests) prepare_unit_tests(flb-it- "${UNIT_TESTS_FILES}") +if(TARGET flb-it-input_chunk) + set_tests_properties(flb-it-input_chunk PROPERTIES TIMEOUT 120) +endif() + if(TARGET flb-it-engine_dispatch) target_include_directories(flb-it-engine_dispatch PRIVATE ${PROJECT_SOURCE_DIR}/lib/chunkio/deps) diff --git a/tests/internal/engine_dispatch.c b/tests/internal/engine_dispatch.c index ecb340179f6..b76088fc0d0 100644 --- a/tests/internal/engine_dispatch.c +++ b/tests/internal/engine_dispatch.c @@ -135,8 +135,14 @@ static int test_output_init(struct flb_output_instance *output, const char *name memset(output, 0, sizeof(struct flb_output_instance)); strncpy(output->name, name, sizeof(output->name) - 1); + if (flb_output_throttle_init(&output->throttle, + FLB_FALSE, 1000, 60000) != 0) { + return -1; + } + output->cmt = cmt_create(); if (output->cmt == NULL) { + flb_output_throttle_destroy(&output->throttle); return -1; } @@ -158,6 +164,7 @@ static int test_output_init(struct flb_output_instance *output, const char *name output->cmt_dropped_records == NULL) { cmt_destroy(output->cmt); output->cmt = NULL; + flb_output_throttle_destroy(&output->throttle); return -1; } @@ -174,6 +181,7 @@ static int test_output_init(struct flb_output_instance *output, const char *name } cmt_destroy(output->cmt); output->cmt = NULL; + flb_output_throttle_destroy(&output->throttle); return -1; } #endif @@ -183,6 +191,8 @@ static int test_output_init(struct flb_output_instance *output, const char *name static void test_output_destroy(struct flb_output_instance *output) { + flb_output_throttle_destroy(&output->throttle); + #ifdef FLB_HAVE_METRICS if (output->metrics != NULL) { flb_metrics_destroy(output->metrics); @@ -291,9 +301,11 @@ static struct flb_task_retry *create_retry_dispatch_task( return NULL; } route->out = output; + route->task = task; route->status = FLB_TASK_ROUTE_INACTIVE; route->records = TEST_ROUTE_RECORDS; route->bytes = TEST_ROUTE_BYTES; + mk_list_init(&route->_deferred_head); mk_list_add(&route->_head, &task->routes); retry = flb_calloc(1, sizeof(struct flb_task_retry)); @@ -482,7 +494,9 @@ static void test_retry_flush_failure_preserves_pending_retry(void) } route->out = &output_b; + route->task = task; route->status = FLB_TASK_ROUTE_INACTIVE; + mk_list_init(&route->_deferred_head); mk_list_add(&route->_head, &task->routes); remaining_retry->attempts = 1; @@ -590,6 +604,162 @@ static void test_retry_flush_failure_reschedules_within_retry_limit(void) test_ctx_destroy(ctx); } +static void test_delayed_singleplex_failure_releases_task(void) +{ + int ret; + int task_id; + char *chunk_buffer; + struct cio_memfs *memfs; + struct test_ctx *ctx; + struct flb_input_chunk *chunk; + struct flb_task *task; + struct flb_task_enqueued blocker; + struct flb_task_queue queue; + struct flb_task_retry *retry; + struct flb_output_instance output; + + ctx = test_ctx_create(); + TEST_CHECK(ctx != NULL); + if (ctx == NULL) { + return; + } + + ret = test_output_init(&output, "output_a"); + TEST_CHECK(ret == 0); + if (ret != 0) { + test_ctx_destroy(ctx); + return; + } + + retry = create_retry_dispatch_task(ctx, &output, &task_id, &chunk_buffer); + TEST_CHECK(retry != NULL); + if (retry == NULL) { + test_output_destroy(&output); + test_ctx_destroy(ctx); + return; + } + task = retry->parent; + chunk = task->ic; + memfs = ((struct cio_chunk *) chunk->chunk)->backend; + memfs->buf_data = chunk_buffer; + flb_task_retry_destroy(retry); + + mk_list_init(&queue.pending); + mk_list_init(&queue.in_progress); + memset(&blocker, 0, sizeof(blocker)); + mk_list_add(&blocker._head, &queue.in_progress); + output.flags = FLB_OUTPUT_SYNCHRONOUS; + output.singleplex_queue = &queue; + + ret = flb_output_task_singleplex_enqueue(&queue, NULL, task, &output, + ctx->config); + TEST_CHECK(ret == 0); + TEST_CHECK(task->users == 1); + TEST_CHECK(mk_list_size(&queue.pending) == 1); + + mk_list_del(&blocker._head); + ctx->config->ch_self_events[1] = -1; + ret = flb_output_task_singleplex_flush_next(&queue); + TEST_CHECK(ret == -1); + TEST_CHECK(ctx->config->task_map[task_id].task == NULL); + TEST_CHECK(mk_list_size(&ctx->input->tasks) == 0); + TEST_CHECK(mk_list_size(&ctx->input->chunks) == 0); + TEST_CHECK(mk_list_size(&queue.pending) == 0); + TEST_CHECK(mk_list_size(&queue.in_progress) == 0); + + test_output_destroy(&output); + test_ctx_destroy(ctx); +} + +static void test_task_destroy_after_output_destroy(void) +{ + int ret; + int task_id; + char *chunk_buffer; + struct cio_memfs *memfs; + struct test_ctx *ctx; + struct flb_input_chunk *chunk; + struct flb_task *task; + struct flb_task_retry *retry; + struct flb_output_instance *output; + + ctx = test_ctx_create(); + TEST_CHECK(ctx != NULL); + if (ctx == NULL) { + return; + } + + output = flb_calloc(1, sizeof(struct flb_output_instance)); + TEST_CHECK(output != NULL); + if (output == NULL) { + test_ctx_destroy(ctx); + return; + } + + ret = test_output_init(output, "output_a"); + TEST_CHECK(ret == 0); + if (ret != 0) { + flb_free(output); + test_ctx_destroy(ctx); + return; + } + mk_list_add(&output->_head, &ctx->config->outputs); + + retry = create_retry_dispatch_task(ctx, output, &task_id, &chunk_buffer); + TEST_CHECK(retry != NULL); + if (retry == NULL) { + mk_list_del(&output->_head); + test_output_destroy(output); + flb_free(output); + test_ctx_destroy(ctx); + return; + } + + task = retry->parent; + chunk = task->ic; + memfs = ((struct cio_chunk *) chunk->chunk)->backend; + memfs->buf_data = chunk_buffer; + flb_task_retry_destroy(retry); + + TEST_CHECK(flb_task_route_queue(task, output) == 0); + TEST_CHECK(output->dispatches_inflight == 1); + + /* A live output still receives normal route accounting. */ + flb_task_destroy(task, FLB_TRUE); + TEST_CHECK(ctx->config->task_map[task_id].task == NULL); + TEST_CHECK(output->dispatches_inflight == 0); + + retry = create_retry_dispatch_task(ctx, output, &task_id, &chunk_buffer); + TEST_CHECK(retry != NULL); + if (retry == NULL) { + mk_list_del(&output->_head); + test_output_destroy(output); + flb_free(output); + test_ctx_destroy(ctx); + return; + } + + task = retry->parent; + chunk = task->ic; + memfs = ((struct cio_chunk *) chunk->chunk)->backend; + memfs->buf_data = chunk_buffer; + flb_task_retry_destroy(retry); + + TEST_CHECK(flb_task_route_queue(task, output) == 0); + TEST_CHECK(output->dispatches_inflight == 1); + + /* Shutdown destroys output instances before their input-owned tasks. */ + mk_list_del(&output->_head); + test_output_destroy(output); + flb_free(output); + + flb_task_destroy(task, FLB_TRUE); + TEST_CHECK(ctx->config->task_map[task_id].task == NULL); + TEST_CHECK(mk_list_size(&ctx->input->tasks) == 0); + + test_ctx_destroy(ctx); +} + TEST_LIST = { { "retry_flush_failure_releases_last_task_owner", test_retry_flush_failure_releases_last_task_owner }, @@ -599,5 +769,9 @@ TEST_LIST = { test_retry_flush_failure_preserves_active_task_owner }, { "retry_flush_failure_preserves_pending_retry", test_retry_flush_failure_preserves_pending_retry }, + { "delayed_singleplex_failure_releases_task", + test_delayed_singleplex_failure_releases_task }, + { "task_destroy_after_output_destroy", + test_task_destroy_after_output_destroy }, { 0 } }; diff --git a/tests/internal/fuzzers/CMakeLists.txt b/tests/internal/fuzzers/CMakeLists.txt index cb9bcdf0a77..aa2857b8991 100644 --- a/tests/internal/fuzzers/CMakeLists.txt +++ b/tests/internal/fuzzers/CMakeLists.txt @@ -24,6 +24,7 @@ set(UNIT_TESTS_FILES multiline_fuzzer.c pack_json_state_fuzzer.c http_fuzzer.c + http_retry_after_fuzzer.c strp_fuzzer.c utils_fuzzer.c config_map_fuzzer.c diff --git a/tests/internal/fuzzers/http_retry_after_fuzzer.c b/tests/internal/fuzzers/http_retry_after_fuzzer.c new file mode 100644 index 00000000000..7cb3288c3cd --- /dev/null +++ b/tests/internal/fuzzers/http_retry_after_fuzzer.c @@ -0,0 +1,16 @@ +#include + +#include +#include + +int LLVMFuzzerTestOneInput(const uint8_t *data, size_t size) +{ + uint64_t delay_ms; + size_t invalid_count; + + flb_http_retry_after_parse((const char *) data, size, 0, &delay_ms); + flb_http_retry_after_parse_headers((const char *) data, size, 0, + &delay_ms, &invalid_count); + + return 0; +} diff --git a/tests/internal/http_retry_after.c b/tests/internal/http_retry_after.c new file mode 100644 index 00000000000..85b35894b69 --- /dev/null +++ b/tests/internal/http_retry_after.c @@ -0,0 +1,171 @@ +/* -*- Mode: C; tab-width: 4; indent-tabs-mode: nil; c-basic-offset: 4 -*- */ + +/* Fluent Bit + * ========== + * Copyright (C) 2015-2026 The Fluent Bit Authors + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#include + +#include "flb_tests_internal.h" + +#include +#include + +struct retry_after_case { + const char *value; + size_t length; + int64_t received_wall_time_ms; + int expected_status; + uint64_t expected_delay_ms; +}; + +static void check_case(const struct retry_after_case *test_case) +{ + uint64_t delay_ms; + int status; + + delay_ms = 123; + status = flb_http_retry_after_parse(test_case->value, test_case->length, + test_case->received_wall_time_ms, + &delay_ms); + TEST_CHECK(status == test_case->expected_status); + if (status == FLB_RETRY_AFTER_VALID || status == FLB_RETRY_AFTER_SATURATED) { + TEST_CHECK(delay_ms == test_case->expected_delay_ms); + } +} + +static void test_delta_seconds(void) +{ + static const char embedded_nul[] = {'1', '0', '\0', '0'}; + static const struct retry_after_case cases[] = { + {"0", 1, 0, FLB_RETRY_AFTER_VALID, 0}, + {" 12\t", 5, 0, FLB_RETRY_AFTER_VALID, 12000}, + {"18446744073709551", 17, 0, FLB_RETRY_AFTER_VALID, + UINT64_C(18446744073709551000)}, + {"18446744073709551615", 20, 0, FLB_RETRY_AFTER_SATURATED, UINT64_MAX}, + {"999999999999999999999999", 24, 0, FLB_RETRY_AFTER_SATURATED, UINT64_MAX}, + {"999999999999999999999999x", 25, 0, FLB_RETRY_AFTER_INVALID, 0}, + {"", 0, 0, FLB_RETRY_AFTER_INVALID, 0}, + {" \t ", 3, 0, FLB_RETRY_AFTER_INVALID, 0}, + {"+1", 2, 0, FLB_RETRY_AFTER_INVALID, 0}, + {"-1", 2, 0, FLB_RETRY_AFTER_INVALID, 0}, + {"1.5", 3, 0, FLB_RETRY_AFTER_INVALID, 0}, + {"1 2", 3, 0, FLB_RETRY_AFTER_INVALID, 0}, + {embedded_nul, sizeof(embedded_nul), 0, FLB_RETRY_AFTER_INVALID, 0} + }; + size_t index; + + for (index = 0; index < sizeof(cases) / sizeof(cases[0]); index++) { + check_case(&cases[index]); + } +} + +static void test_http_dates(void) +{ + static const struct retry_after_case cases[] = { + {"Sun, 06 Nov 1994 08:49:37 GMT", 29, INT64_C(784111772000), + FLB_RETRY_AFTER_VALID, 5000}, + {"Sunday, 06-Nov-94 08:49:37 GMT", 30, INT64_C(784111772000), + FLB_RETRY_AFTER_VALID, 5000}, + {"Sun Nov 6 08:49:37 1994", 24, INT64_C(784111772000), + FLB_RETRY_AFTER_VALID, 5000}, + {"\tWed, 21 Oct 2015 07:28:00 GMT ", 31, INT64_C(1445412470000), + FLB_RETRY_AFTER_VALID, 10000}, + {"Sat, 29 Feb 2020 00:00:00 GMT", 29, INT64_C(1582934395000), + FLB_RETRY_AFTER_VALID, 5000}, + {"Sun, 06 Nov 1994 08:49:37 GMT", 29, INT64_C(784111778000), + FLB_RETRY_AFTER_VALID, 0}, + {"Thu, 29 Feb 2019 00:00:00 GMT", 29, 0, + FLB_RETRY_AFTER_INVALID, 0}, + {"Mon, 06 Nov 1994 08:49:37 GMT", 29, 0, + FLB_RETRY_AFTER_INVALID, 0}, + {"Sun, 06 Nov 1994 08:49:37 UTC", 29, 0, + FLB_RETRY_AFTER_INVALID, 0}, + {"Sun, 06 Nov 1994 08:49", 22, 0, FLB_RETRY_AFTER_INVALID, 0} + }; + size_t index; + + for (index = 0; index < sizeof(cases) / sizeof(cases[0]); index++) { + check_case(&cases[index]); + } +} + +static void test_header_block(void) +{ + const char headers[] = + "HTTP/1.1 429 Too Many Requests\r\n" + "retry-after: 5\r\n" + "X-Test: value\r\n" + "Retry-After: invalid\r\n" + "RETRY-AFTER:\t12 \r\n" + "\r\nbody"; + const char date_header[] = + "Retry-After: Sun, 06 Nov 1994 08:49:37 GMT\r\n\r\n"; + const char combined_header[] = "Retry-After: 5, 12\r\n\r\n"; + const char absent_header[] = "X-Retry-After: 12\r\n\r\n"; + const char colonless_header[] = "Retry-After\r\n\r\n"; + const char colonless_then_valid_header[] = + "Retry-After\r\n" + "Retry-After: 7\r\n\r\n"; + uint64_t delay_ms; + size_t invalid_count; + int status; + + status = flb_http_retry_after_parse_headers(headers, sizeof(headers) - 1, 0, + &delay_ms, &invalid_count); + TEST_CHECK(status == FLB_RETRY_AFTER_VALID); + TEST_CHECK(delay_ms == 12000); + TEST_CHECK(invalid_count == 1); + + status = flb_http_retry_after_parse_headers(date_header, sizeof(date_header) - 1, + INT64_C(784111772000), &delay_ms, + &invalid_count); + TEST_CHECK(status == FLB_RETRY_AFTER_VALID); + TEST_CHECK(delay_ms == 5000); + TEST_CHECK(invalid_count == 0); + + status = flb_http_retry_after_parse_headers(combined_header, + sizeof(combined_header) - 1, 0, + &delay_ms, &invalid_count); + TEST_CHECK(status == FLB_RETRY_AFTER_INVALID); + TEST_CHECK(invalid_count == 1); + + status = flb_http_retry_after_parse_headers(absent_header, + sizeof(absent_header) - 1, 0, + &delay_ms, &invalid_count); + TEST_CHECK(status == FLB_RETRY_AFTER_ABSENT); + TEST_CHECK(invalid_count == 0); + + status = flb_http_retry_after_parse_headers(colonless_header, + sizeof(colonless_header) - 1, 0, + &delay_ms, &invalid_count); + TEST_CHECK(status == FLB_RETRY_AFTER_ABSENT); + TEST_CHECK(invalid_count == 0); + + status = flb_http_retry_after_parse_headers(colonless_then_valid_header, + sizeof(colonless_then_valid_header) - 1, + 0, &delay_ms, &invalid_count); + TEST_CHECK(status == FLB_RETRY_AFTER_VALID); + TEST_CHECK(delay_ms == 7000); + TEST_CHECK(invalid_count == 0); +} + +TEST_LIST = { + {"delta_seconds", test_delta_seconds}, + {"http_dates", test_http_dates}, + {"header_block", test_header_block}, + {0} +}; diff --git a/tests/internal/input_chunk.c b/tests/internal/input_chunk.c index 0aa8b407453..78e3071604c 100644 --- a/tests/internal/input_chunk.c +++ b/tests/internal/input_chunk.c @@ -7,6 +7,7 @@ #include #include #include +#include #include #include #include @@ -18,6 +19,7 @@ #define DPATH FLB_TESTS_DATA_PATH "data/input_chunk/" #define MAX_LINES 32 +#define STOP_TIMEOUT_SEC 10 int64_t result_time; struct tail_test_result { @@ -36,6 +38,28 @@ static inline int64_t set_result(int64_t v) return old; } +static int stop_engine(flb_ctx_t *ctx) +{ +#if defined(FLB_SYSTEM_MACOS) + int ret; + double deadline; + + ret = flb_engine_exit(ctx->config); + TEST_CHECK_(ret >= 0, "requesting graceful engine shutdown"); + + deadline = flb_time_now() + STOP_TIMEOUT_SEC; + while (ctx->status == FLB_LIB_OK && flb_time_now() < deadline) { + flb_time_msleep(10); + } + TEST_CHECK_(ctx->status != FLB_LIB_OK, + "engine did not stop within %d seconds", STOP_TIMEOUT_SEC); + + return flb_stop(ctx); +#else + return flb_stop(ctx); +#endif +} + static int file_to_buf(const char *path, char **out_buf, size_t *out_size) { int ret; @@ -274,7 +298,7 @@ void do_test(char *system, const char *target, ...) sleep(1); - ret = flb_stop(ctx); + ret = stop_engine(ctx); TEST_CHECK_(ret == 0, "stopping engine"); if (ctx) { @@ -368,7 +392,7 @@ void flb_test_input_chunk_dropping_chunks() } flb_time_msleep(2100); - flb_stop(ctx); + stop_engine(ctx); flb_destroy(ctx); flb_free(storage_path); } @@ -793,7 +817,7 @@ void flb_test_input_chunk_grouped_auto_records(void) _head); TEST_CHECK(i_ins != NULL); if (!i_ins) { - flb_stop(ctx); + stop_engine(ctx); flb_destroy(ctx); flb_free(payload); return; @@ -821,7 +845,7 @@ void flb_test_input_chunk_grouped_auto_records(void) TEST_CHECK(ic->total_records == 1); } - flb_stop(ctx); + stop_engine(ctx); flb_destroy(ctx); flb_free(payload); } @@ -908,7 +932,7 @@ void flb_test_input_chunk_grouped_release_space_drop_counters(void) _head); TEST_CHECK(i_ins != NULL); if (!i_ins) { - flb_stop(ctx); + stop_engine(ctx); flb_destroy(ctx); flb_free(payload); flb_free(storage_path); @@ -920,7 +944,7 @@ void flb_test_input_chunk_grouped_release_space_drop_counters(void) _head); TEST_CHECK(o_ins != NULL); if (!o_ins) { - flb_stop(ctx); + stop_engine(ctx); flb_destroy(ctx); flb_free(payload); flb_free(storage_path); @@ -960,7 +984,7 @@ void flb_test_input_chunk_grouped_release_space_drop_counters(void) TEST_CHECK(router_dropped_records <= append_count); TEST_CHECK(output_dropped_records == router_dropped_records); - flb_stop(ctx); + stop_engine(ctx); flb_destroy(ctx); flb_free(payload); flb_free(storage_path); diff --git a/tests/internal/output_thread.c b/tests/internal/output_thread.c new file mode 100644 index 00000000000..955a2c2c4a5 --- /dev/null +++ b/tests/internal/output_thread.c @@ -0,0 +1,270 @@ +/* -*- Mode: C; tab-width: 4; indent-tabs-mode: nil; c-basic-offset: 4 -*- */ + +/* Fluent Bit + * ========== + * Copyright (C) 2015-2026 The Fluent Bit Authors + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#include +#include + +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "flb_tests_internal.h" + +struct test_worker_context { + uint64_t exit_count; + flb_pipefd_t result_writer; +}; + +static int test_worker_init(void *context, struct flb_config *config) +{ + struct test_worker_context *test_context; + struct flb_out_thread_instance *thread_instance; + + (void) config; + + test_context = context; + thread_instance = flb_output_thread_instance_get(); + test_context->result_writer = thread_instance->ch_thread_events[1]; + thread_instance->ch_thread_events[1] = FLB_INVALID_SOCKET; + + return 0; +} + +static int test_worker_exit(void *context, struct flb_config *config) +{ + struct test_worker_context *test_context; + + (void) config; + + test_context = context; + cfl_atomic_store(&test_context->exit_count, + cfl_atomic_load(&test_context->exit_count) + 1); + + return 0; +} + +static void init_dropped_route(struct flb_task *task, + struct flb_task_route *route, + struct flb_output_instance *output, + int task_id) +{ + memset(task, 0, sizeof(*task)); + memset(route, 0, sizeof(*route)); + pthread_mutex_init(&task->lock, NULL); + mk_list_init(&task->routes); + mk_list_init(&task->retries); + task->id = task_id; + route->task = task; + route->out = output; + route->status = FLB_TASK_ROUTE_ACTIVE; + route->dispatch_state = FLB_TASK_ROUTE_DISPATCH_UNQUEUED; + mk_list_init(&route->_deferred_head); + mk_list_add(&route->_head, &task->routes); +} + +static void test_failed_wakeup_preserves_dispatch_ownership(void) +{ + int ret; + char wakeups[3]; + flb_pipefd_t writer; + struct flb_config *config; + struct flb_log log; + struct flb_output_instance output; + struct flb_output_plugin plugin; + struct flb_output_dispatch *dispatch_a; + struct flb_output_dispatch *dispatch_b; + struct flb_task task_a; + struct flb_task task_b; + struct flb_task_route route_a; + struct flb_task_route route_b; + struct flb_tp_thread *thread; + struct flb_out_thread_instance *thread_instance; + struct test_worker_context test_context; +#ifdef _WIN32 + WSADATA wsa_data; +#endif + + memset(wakeups, 0xa5, sizeof(wakeups)); + memset(&output, 0, sizeof(output)); + memset(&plugin, 0, sizeof(plugin)); + memset(&log, 0, sizeof(log)); + memset(&test_context, 0, sizeof(test_context)); + test_context.result_writer = FLB_INVALID_SOCKET; + +#ifdef _WIN32 + ret = WSAStartup(MAKEWORD(2, 2), &wsa_data); + TEST_CHECK(ret == 0); + if (ret != 0) { + return; + } +#endif + + flb_init_env(); + flb_sched_ctx_init(); + + config = flb_config_init(); + TEST_CHECK(config != NULL); + if (config == NULL) { +#ifdef _WIN32 + WSACleanup(); +#endif + return; + } + log.level = FLB_LOG_OFF; + config->log = &log; + + plugin.name = "output_thread_test"; + plugin.cb_worker_init = test_worker_init; + plugin.cb_worker_exit = test_worker_exit; + output.config = config; + output.p = &plugin; + output.context = &test_context; + output.tp_workers = 1; + output.log_level = FLB_LOG_OFF; + output.ch_events[1] = FLB_INVALID_SOCKET; + memcpy(output.name, plugin.name, strlen(plugin.name) + 1); + mk_list_init(&output.upstreams); + + ret = flb_output_thread_pool_create(config, &output); + TEST_CHECK(ret == 0); + if (ret != 0) { + config->log = NULL; + flb_config_exit(config); +#ifdef _WIN32 + WSACleanup(); +#endif + return; + } + TEST_CHECK(mk_list_size(&output.tp->list_threads) == 1); + if (mk_list_is_empty(&output.tp->list_threads) == 0) { + flb_output_thread_pool_destroy(&output); + config->log = NULL; + flb_config_exit(config); +#ifdef _WIN32 + WSACleanup(); +#endif + return; + } + + thread = mk_list_entry_first(&output.tp->list_threads, + struct flb_tp_thread, _head); + thread_instance = thread->params.data; + writer = thread_instance->ch_parent_events[1]; + + /* A closed queue leaves ownership with the caller. */ + init_dropped_route(&task_a, &route_a, &output, 1); + dispatch_a = flb_output_dispatch_create(&task_a, &output, config); + TEST_CHECK(dispatch_a != NULL); + if (dispatch_a == NULL) { + flb_output_thread_pool_start(&output); + flb_output_thread_pool_destroy(&output); + flb_pipe_close(test_context.result_writer); + pthread_mutex_destroy(&task_a.lock); + config->log = NULL; + flb_config_exit(config); +#ifdef _WIN32 + WSACleanup(); +#endif + return; + } + thread_instance->dispatch_shutdown = FLB_TRUE; + TEST_CHECK(flb_output_thread_pool_flush(dispatch_a) == -1); + TEST_CHECK(dispatch_a->_head.next == &dispatch_a->_head); + TEST_CHECK(dispatch_a->_head.prev == &dispatch_a->_head); + thread_instance->dispatch_shutdown = FLB_FALSE; + flb_output_dispatch_destroy(dispatch_a); + + /* Wake bytes carry no pointer data and a failed wake leaves queued ownership intact. */ + ret = flb_pipe_w(writer, wakeups, sizeof(wakeups)); + TEST_CHECK(ret == sizeof(wakeups)); + thread_instance->ch_parent_events[1] = FLB_INVALID_SOCKET; + + init_dropped_route(&task_b, &route_b, &output, 2); + TEST_CHECK(flb_task_route_queue(&task_a, &output) == 0); + TEST_CHECK(flb_task_route_queue(&task_b, &output) == 0); + route_a.status = FLB_TASK_ROUTE_DROPPED; + route_b.status = FLB_TASK_ROUTE_DROPPED; + dispatch_a = flb_output_dispatch_create(&task_a, &output, config); + dispatch_b = flb_output_dispatch_create(&task_b, &output, config); + TEST_CHECK(dispatch_a != NULL); + TEST_CHECK(dispatch_b != NULL); + if (dispatch_a == NULL || dispatch_b == NULL) { + if (dispatch_a != NULL) { + flb_output_dispatch_destroy(dispatch_a); + } + if (dispatch_b != NULL) { + flb_output_dispatch_destroy(dispatch_b); + } + flb_task_route_unqueue(&task_a, &output); + flb_task_users_dec(&task_a, FLB_FALSE); + flb_task_route_unqueue(&task_b, &output); + flb_task_users_dec(&task_b, FLB_FALSE); + flb_output_thread_pool_start(&output); + flb_output_thread_pool_destroy(&output); + flb_pipe_close(writer); + flb_pipe_close(test_context.result_writer); + pthread_mutex_destroy(&task_a.lock); + pthread_mutex_destroy(&task_b.lock); + config->log = NULL; + flb_config_exit(config); +#ifdef _WIN32 + WSACleanup(); +#endif + return; + } + TEST_CHECK(flb_output_thread_pool_flush(dispatch_a) == 0); + TEST_CHECK(flb_output_thread_pool_flush(dispatch_b) == 0); + TEST_CHECK(mk_list_size(&thread_instance->dispatch_queue) == 2); + + flb_output_thread_pool_start(&output); + TEST_CHECK(thread->status == FLB_THREAD_POOL_RUNNING); + flb_output_thread_pool_destroy(&output); + + TEST_CHECK(output.tp == NULL); + TEST_CHECK(cfl_atomic_load(&test_context.exit_count) == 1); + TEST_CHECK(task_a.users == 0); + TEST_CHECK(task_b.users == 0); + TEST_CHECK(output.dispatches_inflight == 0); + TEST_CHECK(route_a.dispatch_state == FLB_TASK_ROUTE_DISPATCH_UNQUEUED); + TEST_CHECK(route_b.dispatch_state == FLB_TASK_ROUTE_DISPATCH_UNQUEUED); + + flb_pipe_close(writer); + flb_pipe_close(test_context.result_writer); + pthread_mutex_destroy(&task_a.lock); + pthread_mutex_destroy(&task_b.lock); + config->log = NULL; + flb_config_exit(config); +#ifdef _WIN32 + WSACleanup(); +#endif +} + +TEST_LIST = { + {"failed_wakeup_preserves_dispatch_ownership", + test_failed_wakeup_preserves_dispatch_ownership}, + {0} +}; diff --git a/tests/internal/output_throttle.c b/tests/internal/output_throttle.c new file mode 100644 index 00000000000..75706e20e31 --- /dev/null +++ b/tests/internal/output_throttle.c @@ -0,0 +1,485 @@ +/* -*- Mode: C; tab-width: 4; indent-tabs-mode: nil; c-basic-offset: 4 -*- */ + +#include + +#include +#include +#include +#include +#include +#include + +#include "flb_tests_internal.h" + +static void test_deadline_and_generation(void) +{ + uint64_t token; + struct flb_output_throttle gate; + struct flb_output_throttle_snapshot snapshot; + + TEST_CHECK(flb_output_throttle_init(&gate, FLB_TRUE, 1000, 60000) == 0); + TEST_CHECK(flb_output_throttle_admit(&gate, 100, &token) == FLB_TRUE); + TEST_CHECK(token == 0); + TEST_CHECK(flb_output_throttle_publish(&gate, 100, FLB_TRUE, 5000, 0) == 5100); + TEST_CHECK(flb_output_throttle_admit(&gate, 5099, &token) == FLB_FALSE); + TEST_CHECK(flb_output_throttle_admit(&gate, 5100, &token) == FLB_TRUE); + TEST_CHECK(token == 1); + + flb_output_throttle_success(&gate, 5101, token); + flb_output_throttle_snapshot(&gate, &snapshot); + TEST_CHECK(snapshot.consecutive_rounds == 0); + flb_output_throttle_destroy(&gate); +} + +static void test_publication_rules(void) +{ + uint64_t token; + struct flb_output_throttle gate; + struct flb_output_throttle_snapshot snapshot; + + TEST_CHECK(flb_output_throttle_init(&gate, FLB_TRUE, 1000, 2000) == 0); + TEST_CHECK(flb_output_throttle_publish(&gate, 0, FLB_TRUE, 600000, 0) == 600000); + TEST_CHECK(flb_output_throttle_publish(&gate, 100, FLB_TRUE, 10, 0) == 600000); + flb_output_throttle_snapshot(&gate, &snapshot); + TEST_CHECK(snapshot.consecutive_rounds == 1); + TEST_CHECK(snapshot.events == 2); + + TEST_CHECK(flb_output_throttle_admit(&gate, 600000, &token) == FLB_TRUE); + TEST_CHECK(flb_output_throttle_publish(&gate, 600000, FLB_FALSE, 0, 1000) == 602000); + flb_output_throttle_snapshot(&gate, &snapshot); + TEST_CHECK(snapshot.consecutive_rounds == 2); + flb_output_throttle_destroy(&gate); +} + +static void test_stale_success_and_saturation(void) +{ + uint64_t old_token; + struct flb_output_throttle gate; + struct flb_output_throttle_snapshot snapshot; + + TEST_CHECK(flb_output_throttle_init(&gate, FLB_TRUE, 1000, 60000) == 0); + TEST_CHECK(flb_output_throttle_admit(&gate, 0, &old_token) == FLB_TRUE); + TEST_CHECK(flb_output_throttle_publish(&gate, 1, FLB_TRUE, UINT64_MAX, 0) == UINT64_MAX); + flb_output_throttle_success(&gate, 2000, old_token); + flb_output_throttle_snapshot(&gate, &snapshot); + TEST_CHECK(snapshot.state == FLB_OUTPUT_THROTTLE_COOLDOWN); + TEST_CHECK(snapshot.consecutive_rounds == 1); + flb_output_throttle_destroy(&gate); +} + +static void test_disabled_and_stopping(void) +{ + uint64_t token; + struct flb_output_throttle gate; + + TEST_CHECK(flb_output_throttle_init(&gate, FLB_FALSE, 1000, 60000) == 0); + TEST_CHECK(flb_output_throttle_publish(&gate, 0, FLB_TRUE, 5000, 0) == 0); + TEST_CHECK(flb_output_throttle_admit(&gate, 1, &token) == FLB_TRUE); + flb_output_throttle_stop(&gate); + TEST_CHECK(flb_output_throttle_admit(&gate, UINT64_MAX, &token) == FLB_FALSE); + flb_output_throttle_destroy(&gate); + + TEST_CHECK(flb_output_throttle_init(&gate, FLB_TRUE, 0, 1) == -1); + TEST_CHECK(flb_output_throttle_init(&gate, FLB_TRUE, 2, 1) == -1); +} + +static void test_result_encoding(void) +{ + uint32_t encoded; + + encoded = FLB_TASK_SET(FLB_THROTTLE, 0x3fff, 0x3fff); + TEST_CHECK(FLB_TASK_RET(encoded) == FLB_THROTTLE); + TEST_CHECK(FLB_TASK_ID(encoded) == 0x3fff); + TEST_CHECK(FLB_TASK_OUT(encoded) == 0x3fff); + TEST_CHECK(FLB_ERROR == 0); + TEST_CHECK(FLB_OK == 1); + TEST_CHECK(FLB_RETRY == 2); +} + +static void test_output_properties(void) +{ + struct flb_config *config; + struct flb_output_instance output; + + config = flb_config_init(); + TEST_CHECK(config != NULL); + if (config == NULL) { + return; + } + + memset(&output, 0, sizeof(struct flb_output_instance)); + output.config = config; + TEST_CHECK(flb_output_throttle_init(&output.throttle, + FLB_FALSE, 1000, 60000) == 0); + + TEST_CHECK(flb_output_set_property(&output, "throttle", "true") == 0); + TEST_CHECK(output.throttle.enabled == FLB_TRUE); + TEST_CHECK(flb_output_set_property(&output, "throttle.base", "120") == 0); + TEST_CHECK(flb_output_set_property(&output, "throttle.cap", "180") == 0); + TEST_CHECK(output.throttle.base_ms == 120000); + TEST_CHECK(output.throttle.cap_ms == 180000); + + TEST_CHECK(flb_output_set_property(&output, "throttle.base", "0") == -1); + TEST_CHECK(flb_output_set_property(&output, "throttle.base", "1s") == -1); + TEST_CHECK(flb_output_set_property(&output, "throttle.cap", "-1") == -1); + + flb_output_throttle_destroy(&output.throttle); + flb_config_exit(config); +} + +static void init_route_fixture(struct flb_task *task, + struct flb_task_route *route, + struct flb_output_instance *output) +{ + memset(task, 0, sizeof(struct flb_task)); + memset(route, 0, sizeof(struct flb_task_route)); + memset(output, 0, sizeof(struct flb_output_instance)); + + mk_list_init(&task->routes); + mk_list_init(&task->retries); + mk_list_init(&output->throttle_deferred_routes); + route->task = task; + route->out = output; + route->dispatch_state = FLB_TASK_ROUTE_DISPATCH_UNQUEUED; + mk_list_init(&route->_deferred_head); + mk_list_add(&route->_head, &task->routes); +} + +static void test_deferred_route_ownership(void) +{ + struct flb_task task; + struct flb_task_route route; + struct flb_output_instance output; + + init_route_fixture(&task, &route, &output); + + TEST_CHECK(flb_task_route_defer(&task, &output, FLB_FALSE) == 0); + TEST_CHECK(task.users == 0); + TEST_CHECK(task.deferred_routes == 1); + TEST_CHECK(output.throttle_deferred_count == 1); + TEST_CHECK(route.dispatch_state == FLB_TASK_ROUTE_DISPATCH_DEFERRED); + TEST_CHECK(flb_task_is_releasable(&task) == FLB_FALSE); + + /* A fresh duplicate is idempotent and does not gain another owner. */ + TEST_CHECK(flb_task_route_defer(&task, &output, FLB_FALSE) == 0); + TEST_CHECK(task.deferred_routes == 1); + TEST_CHECK(output.throttle_deferred_count == 1); + + TEST_CHECK(flb_task_route_resume(&task, &output) == 0); + TEST_CHECK(task.users == 1); + TEST_CHECK(task.deferred_routes == 0); + TEST_CHECK(output.throttle_deferred_count == 0); + TEST_CHECK(route.dispatch_state == FLB_TASK_ROUTE_DISPATCH_QUEUED); + TEST_CHECK(flb_task_is_releasable(&task) == FLB_FALSE); + + /* Queued -> deferred transfers, rather than duplicates, its owner. */ + TEST_CHECK(flb_task_route_defer(&task, &output, FLB_TRUE) == 0); + TEST_CHECK(task.users == 0); + TEST_CHECK(task.deferred_routes == 1); + TEST_CHECK(flb_task_route_defer(&task, &output, FLB_TRUE) == -1); + TEST_CHECK(task.users == 0); + TEST_CHECK(task.deferred_routes == 1); + + TEST_CHECK(flb_task_route_cancel_deferred(&task, &output) == 0); + TEST_CHECK(task.deferred_routes == 0); + TEST_CHECK(output.throttle_deferred_count == 0); + TEST_CHECK(route.dispatch_state == FLB_TASK_ROUTE_DISPATCH_UNQUEUED); + TEST_CHECK(flb_task_is_releasable(&task) == FLB_TRUE); +} + +static void test_dispatch_state_and_envelope(void) +{ + uint64_t before; + uint64_t after; + struct flb_task task; + struct flb_task_route route; + struct flb_output_instance output; + struct flb_output_dispatch *dispatch; + + init_route_fixture(&task, &route, &output); + + TEST_CHECK(flb_task_route_queue(&task, &output) == 0); + TEST_CHECK(task.users == 1); + TEST_CHECK(output.dispatches_inflight == 1); + TEST_CHECK(route.dispatch_state == FLB_TASK_ROUTE_DISPATCH_QUEUED); + TEST_CHECK(flb_task_route_queue(&task, &output) == -1); + + dispatch = flb_output_dispatch_create(&task, &output, NULL); + TEST_CHECK(dispatch != NULL); + if (dispatch != NULL) { + TEST_CHECK(dispatch->magic == FLB_OUTPUT_DISPATCH_MAGIC); + TEST_CHECK(dispatch->type == FLB_OUTPUT_DISPATCH_TASK); + TEST_CHECK(dispatch->task == &task); + TEST_CHECK(dispatch->out == &output); + flb_output_dispatch_destroy(dispatch); + } + + TEST_CHECK(flb_task_route_complete(&task, &output) == 0); + TEST_CHECK(route.dispatch_state == FLB_TASK_ROUTE_DISPATCH_COMPLETING); + TEST_CHECK(flb_task_route_unqueue(&task, &output) == 0); + TEST_CHECK(route.dispatch_state == FLB_TASK_ROUTE_DISPATCH_UNQUEUED); + TEST_CHECK(output.dispatches_inflight == 0); + flb_task_users_dec(&task, FLB_FALSE); + TEST_CHECK(flb_task_is_releasable(&task) == FLB_TRUE); + + before = flb_output_throttle_now_ms(); + after = flb_output_throttle_now_ms(); + TEST_CHECK(after >= before); +} + +static void test_dispatch_result_fallback_queue(void) +{ + struct flb_config config; + struct flb_config other_config; + struct flb_task task; + struct flb_task_route route; + struct flb_output_instance output; + struct flb_out_thread_instance thread; + struct flb_output_dispatch *dispatch; + struct flb_output_dispatch *popped; + + memset(&config, 0, sizeof(struct flb_config)); + memset(&other_config, 0, sizeof(struct flb_config)); + memset(&thread, 0, sizeof(struct flb_out_thread_instance)); + init_route_fixture(&task, &route, &output); + output.log_level = FLB_LOG_OFF; + output.ch_events[1] = -1; + thread.ins = &output; + thread.ch_thread_events[1] = -1; + + dispatch = flb_output_dispatch_create(&task, &output, &config); + TEST_CHECK(dispatch != NULL); + if (dispatch == NULL) { + return; + } + + TEST_CHECK(flb_output_thread_post_dispatch_result( + &thread, dispatch, FLB_OUTPUT_DEFERRED) == 1); + TEST_CHECK(flb_output_thread_result_fallback_pop(&other_config) == NULL); + popped = flb_output_thread_result_fallback_pop(&config); + TEST_CHECK(popped == dispatch); + if (popped != NULL) { + TEST_CHECK(popped->result == FLB_OUTPUT_DEFERRED); + flb_output_dispatch_destroy(popped); + } + + dispatch = flb_output_dispatch_create(&task, &output, &config); + TEST_CHECK(dispatch != NULL); + if (dispatch == NULL) { + return; + } + TEST_CHECK(flb_task_route_queue(&task, &output) == 0); + TEST_CHECK(task.users == 1); + TEST_CHECK(output.dispatches_inflight == 1); + TEST_CHECK(flb_output_thread_post_dispatch_result( + &thread, dispatch, FLB_ERROR) == 1); + flb_output_thread_result_fallback_remove(&output); + TEST_CHECK(flb_output_thread_result_fallback_pop(&config) == NULL); + TEST_CHECK(task.users == 0); + TEST_CHECK(output.dispatches_inflight == 0); + TEST_CHECK(route.dispatch_state == FLB_TASK_ROUTE_DISPATCH_UNQUEUED); +} + +static void test_advisory_dispatch_deferral(void) +{ + struct flb_task task; + struct flb_task_route route; + struct flb_output_instance output; + + init_route_fixture(&task, &route, &output); + TEST_CHECK(flb_output_throttle_init(&output.throttle, + FLB_TRUE, 1000, 60000) == 0); + flb_output_throttle_publish(&output.throttle, + flb_output_throttle_now_ms(), + FLB_TRUE, 60000, 0); + + TEST_CHECK(flb_output_task_flush(&task, &output, NULL) == + FLB_OUTPUT_DEFERRED); + TEST_CHECK(task.users == 0); + TEST_CHECK(task.deferred_routes == 1); + TEST_CHECK(route.dispatch_state == FLB_TASK_ROUTE_DISPATCH_DEFERRED); + + TEST_CHECK(flb_task_route_cancel_deferred(&task, &output) == 0); + flb_output_throttle_destroy(&output.throttle); +} + +static void test_flush_completion_publication(void) +{ + int ret; + struct flb_output_flush flush; + struct flb_output_instance output; + struct flb_output_throttle_snapshot snapshot; + + memset(&flush, 0, sizeof(struct flb_output_flush)); + memset(&output, 0, sizeof(struct flb_output_instance)); + flush.o_ins = &output; + + TEST_CHECK(flb_output_throttle_init(&output.throttle, + FLB_FALSE, 1000, 60000) == 0); + ret = flb_output_throttle_complete(&flush, FLB_THROTTLE); + TEST_CHECK(ret == FLB_RETRY); + flb_output_throttle_snapshot(&output.throttle, &snapshot); + TEST_CHECK(snapshot.events == 0); + flb_output_throttle_destroy(&output.throttle); + + TEST_CHECK(flb_output_throttle_init(&output.throttle, + FLB_TRUE, 1000, 60000) == 0); + flush.retry_after_present = FLB_TRUE; + flush.retry_after_ms = 120000; + ret = flb_output_throttle_complete(&flush, FLB_THROTTLE); + TEST_CHECK(ret == FLB_THROTTLE); + flb_output_throttle_snapshot(&output.throttle, &snapshot); + TEST_CHECK(snapshot.state == FLB_OUTPUT_THROTTLE_COOLDOWN); + TEST_CHECK(snapshot.events == 1); + TEST_CHECK(snapshot.until_ms >= flb_output_throttle_now_ms() + 119000); + + flush.admission_generation = snapshot.generation; + ret = flb_output_throttle_complete(&flush, FLB_OK); + TEST_CHECK(ret == FLB_OK); + flb_output_throttle_snapshot(&output.throttle, &snapshot); + TEST_CHECK(snapshot.consecutive_rounds == 1); + flb_output_throttle_destroy(&output.throttle); +} + +static void test_due_retry_defers_without_attempt(void) +{ + struct flb_task task; + struct flb_task_retry retry; + struct flb_task_route route; + struct flb_output_instance output; + + init_route_fixture(&task, &route, &output); + memset(&retry, 0, sizeof(struct flb_task_retry)); + retry.attempts = 3; + retry.parent = &task; + retry.o_ins = &output; + + TEST_CHECK(flb_output_throttle_init(&output.throttle, + FLB_TRUE, 1000, 60000) == 0); + flb_output_throttle_publish(&output.throttle, + flb_output_throttle_now_ms(), + FLB_TRUE, 60000, 0); + + TEST_CHECK(flb_engine_dispatch_retry(&retry, NULL) == 0); + TEST_CHECK(retry.attempts == 3); + TEST_CHECK(route.dispatch_state == FLB_TASK_ROUTE_DISPATCH_DEFERRED); + TEST_CHECK(task.deferred_routes == 1); + TEST_CHECK(output.throttle_wakeup_pending == FLB_TRUE); + + TEST_CHECK(flb_task_route_cancel_deferred(&task, &output) == 0); + flb_output_throttle_wakeup_cancel(&output); + + TEST_CHECK(flb_task_route_queue(&task, &output) == 0); + TEST_CHECK(flb_engine_dispatch_retry(&retry, NULL) == 0); + TEST_CHECK(retry.attempts == 3); + TEST_CHECK(task.users == 0); + TEST_CHECK(route.dispatch_state == FLB_TASK_ROUTE_DISPATCH_DEFERRED); + TEST_CHECK(output.dispatches_inflight == 0); + + TEST_CHECK(flb_task_route_cancel_deferred(&task, &output) == 0); + flb_output_throttle_wakeup_cancel(&output); + flb_output_throttle_destroy(&output.throttle); +} + +static void test_metrics_union_duration(void) +{ + int ret; + double value; + char *labels[] = {"test.0"}; + struct flb_output_instance output; + + memset(&output, 0, sizeof(struct flb_output_instance)); + snprintf(output.name, sizeof(output.name), "test.0"); + output.cmt = cmt_create(); + TEST_CHECK(output.cmt != NULL); + if (output.cmt == NULL) { + return; + } + + output.cmt_throttle_events = cmt_counter_create(output.cmt, "fluentbit", + "output", + "throttle_events_total", + "test", 1, + (char *[]) {"name"}); + output.cmt_throttle_active = cmt_gauge_create(output.cmt, "fluentbit", + "output", "throttle_active", + "test", 1, + (char *[]) {"name"}); + output.cmt_throttle_remaining = cmt_gauge_create(output.cmt, "fluentbit", + "output", + "throttle_remaining_seconds", + "test", 1, + (char *[]) {"name"}); + output.cmt_throttle_deferred_routes = cmt_gauge_create(output.cmt, + "fluentbit", "output", + "throttle_deferred_routes", + "test", 1, + (char *[]) {"name"}); + output.cmt_throttle_duration = cmt_counter_create(output.cmt, "fluentbit", + "output", + "throttle_duration_seconds_total", + "test", 1, + (char *[]) {"name"}); + TEST_CHECK(output.cmt_throttle_events != NULL); + TEST_CHECK(output.cmt_throttle_active != NULL); + TEST_CHECK(output.cmt_throttle_remaining != NULL); + TEST_CHECK(output.cmt_throttle_deferred_routes != NULL); + TEST_CHECK(output.cmt_throttle_duration != NULL); + if (output.cmt_throttle_events == NULL || + output.cmt_throttle_active == NULL || + output.cmt_throttle_remaining == NULL || + output.cmt_throttle_deferred_routes == NULL || + output.cmt_throttle_duration == NULL) { + cmt_destroy(output.cmt); + return; + } + + TEST_CHECK(flb_output_throttle_init(&output.throttle, + FLB_TRUE, 1000, 60000) == 0); + flb_output_throttle_publish(&output.throttle, 1000, + FLB_TRUE, 5000, 0); + flb_output_throttle_metrics_update(&output, 2000); + + ret = cmt_gauge_get_val(output.cmt_throttle_active, 1, labels, &value); + TEST_CHECK(ret == 0); + TEST_CHECK(value == 1.0); + ret = cmt_gauge_get_val(output.cmt_throttle_remaining, 1, labels, &value); + TEST_CHECK(ret == 0); + TEST_CHECK(value == 4.0); + + /* The extension adds only newly elapsed closed-gate time. */ + flb_output_throttle_publish(&output.throttle, 2000, + FLB_TRUE, 10000, 0); + flb_output_throttle_metrics_update(&output, 3000); + flb_output_throttle_metrics_update(&output, 12000); + ret = cmt_counter_get_val(output.cmt_throttle_duration, 1, labels, &value); + TEST_CHECK(ret == 0); + TEST_CHECK(value == 11.0); + + ret = cmt_gauge_get_val(output.cmt_throttle_active, 1, labels, &value); + TEST_CHECK(ret == 0); + TEST_CHECK(value == 0.0); + ret = cmt_gauge_get_val(output.cmt_throttle_remaining, 1, labels, &value); + TEST_CHECK(ret == 0); + TEST_CHECK(value == 0.0); + + flb_output_throttle_destroy(&output.throttle); + cmt_destroy(output.cmt); +} + +TEST_LIST = { + {"deadline_and_generation", test_deadline_and_generation}, + {"publication_rules", test_publication_rules}, + {"stale_success_and_saturation", test_stale_success_and_saturation}, + {"disabled_and_stopping", test_disabled_and_stopping}, + {"result_encoding", test_result_encoding}, + {"output_properties", test_output_properties}, + {"deferred_route_ownership", test_deferred_route_ownership}, + {"dispatch_state_and_envelope", test_dispatch_state_and_envelope}, + {"dispatch_result_fallback_queue", test_dispatch_result_fallback_queue}, + {"advisory_dispatch_deferral", test_advisory_dispatch_deferral}, + {"flush_completion_publication", test_flush_completion_publication}, + {"due_retry_defers_without_attempt", test_due_retry_defers_without_attempt}, + {"metrics_union_duration", test_metrics_union_duration}, + {0} +}; diff --git a/tests/internal/search_bulk.c b/tests/internal/search_bulk.c index ae29abfab0a..162e4fbec7a 100644 --- a/tests/internal/search_bulk.c +++ b/tests/internal/search_bulk.c @@ -33,9 +33,14 @@ "{\"update\":{\"_index\":\"logs\",\"_id\":\"two\"}}\n" \ "{\"doc_as_upsert\":true,\"doc\":{\"message\":\"two\"}}\n" +#define THIRD_ENTRY \ + "{\"create\":{\"_index\":\"logs\",\"_id\":\"three\"}}\n" \ + "{\"message\":\"three\"}\n" + static void test_mixed_response_keeps_only_unresolved(void) { int result; + int throttled; const char *response; struct flb_search_bulk_retry *retry; @@ -48,9 +53,10 @@ static void test_mixed_response_keeps_only_unresolved(void) BULK_PAYLOAD, strlen(BULK_PAYLOAD), FLB_SEARCH_BULK_ACK_CREATE_CONFLICTS, - FLB_FALSE, NULL, + FLB_FALSE, NULL, &throttled, &retry); TEST_CHECK(result == FLB_SEARCH_BULK_RETRY); + TEST_CHECK(throttled == FLB_TRUE); TEST_CHECK(retry != NULL); TEST_CHECK(retry->records == 1); TEST_CHECK(retry->size == strlen(SECOND_ENTRY)); @@ -61,6 +67,7 @@ static void test_mixed_response_keeps_only_unresolved(void) static void test_create_conflicts_are_complete(void) { int result; + int throttled; const char *response; struct flb_search_bulk_retry *retry; @@ -73,15 +80,47 @@ static void test_create_conflicts_are_complete(void) BULK_PAYLOAD, strlen(BULK_PAYLOAD), FLB_SEARCH_BULK_ACK_CREATE_CONFLICTS, - FLB_FALSE, NULL, + FLB_FALSE, NULL, &throttled, &retry); TEST_CHECK(result == FLB_SEARCH_BULK_COMPLETE); + TEST_CHECK(throttled == FLB_FALSE); TEST_CHECK(retry == NULL); } +static void test_mixed_failures_preserve_existing_retry_subset(void) +{ + int result; + int throttled; + const char *response; + struct flb_search_bulk_retry *retry; + + response = "{\"errors\":true,\"items\":[" + "{\"create\":{\"status\":201}}," + "{\"create\":{\"status\":400}}," + "{\"create\":{\"status\":503}}]}"; + + result = flb_search_bulk_process_response(response, strlen(response), + BULK_PAYLOAD, + strlen(BULK_PAYLOAD), + FLB_SEARCH_BULK_ACK_CREATE_CONFLICTS, + FLB_FALSE, NULL, + &throttled, + &retry); + TEST_CHECK(result == FLB_SEARCH_BULK_RETRY); + TEST_CHECK(throttled == FLB_FALSE); + TEST_CHECK(retry != NULL); + TEST_CHECK(retry->records == 2); + TEST_CHECK(retry->size == strlen(SECOND_ENTRY) + strlen(THIRD_ENTRY)); + TEST_CHECK(memcmp(retry->payload, SECOND_ENTRY, strlen(SECOND_ENTRY)) == 0); + TEST_CHECK(memcmp(retry->payload + strlen(SECOND_ENTRY), + THIRD_ENTRY, strlen(THIRD_ENTRY)) == 0); + flb_search_bulk_retry_destroy(retry); +} + static void test_update_conflict_is_retried(void) { int result; + int throttled; const char *payload; const char *response; struct flb_search_bulk_retry *retry; @@ -94,9 +133,10 @@ static void test_update_conflict_is_retried(void) result = flb_search_bulk_process_response(response, strlen(response), payload, strlen(payload), FLB_SEARCH_BULK_ACK_CREATE_CONFLICTS, - FLB_FALSE, NULL, + FLB_FALSE, NULL, &throttled, &retry); TEST_CHECK(result == FLB_SEARCH_BULK_RETRY); + TEST_CHECK(throttled == FLB_FALSE); TEST_CHECK(retry != NULL); TEST_CHECK(retry->records == 1); TEST_CHECK(retry->size == strlen(payload)); @@ -106,6 +146,7 @@ static void test_update_conflict_is_retried(void) static void test_update_conflict_is_complete_when_all_conflicts_are_acknowledged(void) { int result; + int throttled; const char *payload; const char *response; struct flb_search_bulk_retry *retry; @@ -118,15 +159,17 @@ static void test_update_conflict_is_complete_when_all_conflicts_are_acknowledged result = flb_search_bulk_process_response(response, strlen(response), payload, strlen(payload), FLB_SEARCH_BULK_ACK_ALL_CONFLICTS, - FLB_FALSE, NULL, + FLB_FALSE, NULL, &throttled, &retry); TEST_CHECK(result == FLB_SEARCH_BULK_COMPLETE); + TEST_CHECK(throttled == FLB_FALSE); TEST_CHECK(retry == NULL); } static void test_truncated_success_response_is_complete(void) { int result; + int throttled; const char *response; struct flb_search_bulk_retry *retry; @@ -137,15 +180,17 @@ static void test_truncated_success_response_is_complete(void) BULK_PAYLOAD, strlen(BULK_PAYLOAD), FLB_SEARCH_BULK_ACK_CREATE_CONFLICTS, - FLB_FALSE, NULL, + FLB_FALSE, NULL, &throttled, &retry); TEST_CHECK(result == FLB_SEARCH_BULK_COMPLETE); + TEST_CHECK(throttled == FLB_FALSE); TEST_CHECK(retry == NULL); } static void test_nested_success_marker_with_top_level_errors_is_invalid(void) { int result; + int throttled; const char *response; struct flb_search_bulk_retry *retry; @@ -156,15 +201,17 @@ static void test_nested_success_marker_with_top_level_errors_is_invalid(void) BULK_PAYLOAD, strlen(BULK_PAYLOAD), FLB_SEARCH_BULK_ACK_CREATE_CONFLICTS, - FLB_FALSE, NULL, + FLB_FALSE, NULL, &throttled, &retry); TEST_CHECK(result == FLB_SEARCH_BULK_INVALID); + TEST_CHECK(throttled == FLB_FALSE); TEST_CHECK(retry == NULL); } static void test_item_count_mismatch_is_invalid(void) { int result; + int throttled; const char *response; struct flb_search_bulk_retry *retry; @@ -175,9 +222,10 @@ static void test_item_count_mismatch_is_invalid(void) BULK_PAYLOAD, strlen(BULK_PAYLOAD), FLB_SEARCH_BULK_ACK_CREATE_CONFLICTS, - FLB_FALSE, NULL, + FLB_FALSE, NULL, &throttled, &retry); TEST_CHECK(result == FLB_SEARCH_BULK_INVALID); + TEST_CHECK(throttled == FLB_FALSE); TEST_CHECK(retry == NULL); } @@ -202,7 +250,7 @@ static void test_mixed_response_populates_stats(void) FOUR_ENTRY_BULK_PAYLOAD, strlen(FOUR_ENTRY_BULK_PAYLOAD), FLB_SEARCH_BULK_ACK_CREATE_CONFLICTS, - FLB_FALSE, &stats, &retry); + FLB_FALSE, &stats, NULL, &retry); TEST_CHECK(result == FLB_SEARCH_BULK_RETRY); TEST_CHECK(retry != NULL); TEST_CHECK(retry->records == 2); @@ -236,7 +284,7 @@ static void test_drop_unrecoverable_upsert_records(void) UPSERT_PAYLOAD, strlen(UPSERT_PAYLOAD), FLB_SEARCH_BULK_ACK_CREATE_CONFLICTS, - FLB_TRUE, &stats, &retry); + FLB_TRUE, &stats, NULL, &retry); TEST_CHECK(result == FLB_SEARCH_BULK_RETRY); TEST_CHECK(retry != NULL); TEST_CHECK(retry->records == 1); @@ -250,7 +298,7 @@ static void test_drop_unrecoverable_upsert_records(void) UPSERT_PAYLOAD, strlen(UPSERT_PAYLOAD), FLB_SEARCH_BULK_ACK_CREATE_CONFLICTS, - FLB_FALSE, &stats, &retry); + FLB_FALSE, &stats, NULL, &retry); TEST_CHECK(result == FLB_SEARCH_BULK_RETRY); TEST_CHECK(retry != NULL); TEST_CHECK(retry->records == 2); @@ -274,7 +322,7 @@ static void test_only_unrecoverable_records_complete_when_dropped(void) UPSERT_PAYLOAD, strlen(UPSERT_PAYLOAD), FLB_SEARCH_BULK_ACK_CREATE_CONFLICTS, - FLB_TRUE, &stats, &retry); + FLB_TRUE, &stats, NULL, &retry); TEST_CHECK(result == FLB_SEARCH_BULK_COMPLETE); TEST_CHECK(retry == NULL); TEST_CHECK(stats.failed_items == 2); @@ -287,6 +335,8 @@ static void test_only_unrecoverable_records_complete_when_dropped(void) TEST_LIST = { {"mixed_response_keeps_only_unresolved", test_mixed_response_keeps_only_unresolved}, + {"mixed_failures_preserve_existing_retry_subset", + test_mixed_failures_preserve_existing_retry_subset}, {"create_conflicts_are_complete", test_create_conflicts_are_complete}, {"update_conflict_is_retried", test_update_conflict_is_retried}, {"update_conflict_is_complete_when_all_conflicts_are_acknowledged", diff --git a/tests/internal/utils.c b/tests/internal/utils.c index 2dbb67812bb..6c41ee06b96 100644 --- a/tests/internal/utils.c +++ b/tests/internal/utils.c @@ -1080,6 +1080,25 @@ void test_size_to_binary_bytes() } } +void test_time_to_seconds_strict() +{ + int seconds; + + TEST_CHECK(flb_utils_time_to_seconds_strict("1", &seconds) == 0); + TEST_CHECK(seconds == 1); + TEST_CHECK(flb_utils_time_to_seconds_strict("60", &seconds) == 0); + TEST_CHECK(seconds == 60); + TEST_CHECK(flb_utils_time_to_seconds_strict("", &seconds) == -1); + TEST_CHECK(flb_utils_time_to_seconds_strict("0", &seconds) == -1); + TEST_CHECK(flb_utils_time_to_seconds_strict("-1", &seconds) == -1); + TEST_CHECK(flb_utils_time_to_seconds_strict("+1", &seconds) == -1); + TEST_CHECK(flb_utils_time_to_seconds_strict("1s", &seconds) == -1); + TEST_CHECK(flb_utils_time_to_seconds_strict("1.5", &seconds) == -1); + TEST_CHECK(flb_utils_time_to_seconds_strict("1foo", &seconds) == -1); + TEST_CHECK(flb_utils_time_to_seconds_strict("999999999999999999999", &seconds) == -1); + TEST_CHECK(flb_utils_time_to_seconds_strict("1", NULL) == -1); +} + TEST_LIST = { /* JSON maps iteration */ { "url_split", test_url_split }, @@ -1101,5 +1120,6 @@ TEST_LIST = { { "test_flb_utils_get_machine_id", test_flb_utils_get_machine_id }, { "test_size_to_bytes", test_size_to_bytes }, { "test_size_to_bianry_bytes", test_size_to_binary_bytes }, + { "test_time_to_seconds_strict", test_time_to_seconds_strict }, { 0 } }; diff --git a/tests/runtime/CMakeLists.txt b/tests/runtime/CMakeLists.txt index 895886433c7..4e876db5e44 100644 --- a/tests/runtime/CMakeLists.txt +++ b/tests/runtime/CMakeLists.txt @@ -254,6 +254,7 @@ endif() # Output Plugins if(FLB_IN_LIB) + FLB_RT_TEST(1 "output_throttle_runtime.c") FLB_RT_TEST(FLB_OUT_LIB "core_engine.c") FLB_RT_TEST(FLB_OUT_LIB "core_log.c") FLB_RT_TEST(FLB_OUT_LIB "core_routes.c") @@ -395,6 +396,10 @@ foreach(source_file ${CHECK_PROGRAMS}) endif() endforeach() +if(TEST flb-rt-output_throttle_runtime) + set_tests_properties(flb-rt-output_throttle_runtime PROPERTIES TIMEOUT 120) +endif() + function(flb_runtime_lock_tests resource_name) foreach(test_name ${ARGN}) if (TEST ${test_name}) diff --git a/tests/runtime/core_shutdown_spin.c b/tests/runtime/core_shutdown_spin.c index 961fb941366..c7e92e6d6f3 100644 --- a/tests/runtime/core_shutdown_spin.c +++ b/tests/runtime/core_shutdown_spin.c @@ -19,6 +19,8 @@ #include #include +#include +#include #include #include #include @@ -29,12 +31,98 @@ #define SHUTDOWN_TIME_LIMIT_SEC 5 /* grace=2 + safety margin */ #define SHUTDOWN_WATCHDOG_SEC 10 +struct shutdown_observer { + int worker_exit_called; + int exit_called; + int worker_exit_on_pipeline_thread; + int exit_after_worker_exit; + int exit_on_worker_thread; + pthread_t caller_thread; + pthread_t worker_thread; +}; + +static int shutdown_observer_init(struct flb_output_instance *ins, + struct flb_config *config, void *data) +{ + (void) config; + + flb_output_set_context(ins, data); + return 0; +} + +static void shutdown_observer_flush(struct flb_event_chunk *event_chunk, + struct flb_output_flush *out_flush, + struct flb_input_instance *i_ins, + void *out_context, + struct flb_config *config) +{ + (void) event_chunk; + (void) i_ins; + (void) out_context; + (void) config; + + FLB_OUTPUT_RETURN(FLB_OK); +} + +static int shutdown_observer_worker_exit(void *data, struct flb_config *config) +{ + struct shutdown_observer *observer = data; + + (void) config; + + observer->worker_exit_called++; + observer->worker_thread = pthread_self(); + observer->worker_exit_on_pipeline_thread = + pthread_equal(observer->worker_thread, observer->caller_thread) == 0; + return 0; +} + +static int shutdown_observer_exit(void *data, struct flb_config *config) +{ + struct shutdown_observer *observer = data; + + (void) config; + + observer->exit_called++; + observer->exit_after_worker_exit = observer->worker_exit_called == 1; + if (observer->exit_after_worker_exit) { + observer->exit_on_worker_thread = + pthread_equal(pthread_self(), observer->worker_thread) != 0; + } + return 0; +} + +static struct flb_output_plugin shutdown_observer_plugin = { + .name = "shutdown_observer", + .description = "Observe library-mode engine shutdown", + .cb_init = shutdown_observer_init, + .cb_flush = shutdown_observer_flush, + .cb_exit = shutdown_observer_exit, + .cb_worker_exit = shutdown_observer_worker_exit, + .flags = 0 +}; + +static int register_shutdown_observer(flb_ctx_t *ctx) +{ + struct flb_output_plugin *plugin; + + plugin = flb_malloc(sizeof(struct flb_output_plugin)); + if (plugin == NULL) { + return -1; + } + + memcpy(plugin, &shutdown_observer_plugin, + sizeof(struct flb_output_plugin)); + mk_list_add(&plugin->_head, &ctx->config->out_plugins); + return 0; +} + /* Async-signal-safe abort used when flb_stop() hangs on a regression. */ static void timeout_abort(int sig) { static const char msg[] = - "\nFAIL: flb_test_duplicate_stop_no_spin timed out; " - "shutdown spin regression likely present.\n"; + "\nFAIL: core shutdown test timed out; " + "shutdown regression likely present.\n"; (void) sig; (void) write(STDERR_FILENO, msg, sizeof(msg) - 1); _exit(1); @@ -103,8 +191,70 @@ void flb_test_duplicate_stop_no_spin(void) } } +/* Regression: flb_stop() must wait for pipeline-owned cleanup to finish. */ +void flb_test_stop_waits_for_cleanup(void) +{ + int in_ffd; + int out_ffd; + int ret; + flb_ctx_t *ctx; + struct sigaction sa; + struct shutdown_observer observer; + + memset(&observer, 0, sizeof(observer)); + observer.caller_thread = pthread_self(); + + ctx = flb_create(); + TEST_CHECK(ctx != NULL); + if (ctx == NULL) { + return; + } + + TEST_CHECK(register_shutdown_observer(ctx) == 0); + TEST_CHECK(flb_service_set(ctx, + "Flush", "1", + "Grace", "1", + "Log_Level", "error", + NULL) == 0); + + in_ffd = flb_input(ctx, (char *) "lib", NULL); + TEST_CHECK(in_ffd >= 0); + TEST_CHECK(flb_input_set(ctx, in_ffd, "tag", "test", NULL) == 0); + + out_ffd = flb_output(ctx, (char *) "shutdown_observer", + (struct flb_lib_out_cb *) &observer); + TEST_CHECK(out_ffd >= 0); + TEST_CHECK(flb_output_set(ctx, out_ffd, "match", "*", NULL) == 0); + + ret = flb_start(ctx); + TEST_CHECK_(ret == 0, "starting engine"); + if (ret != 0) { + flb_destroy(ctx); + return; + } + + memset(&sa, 0, sizeof(sa)); + sa.sa_handler = timeout_abort; + sigaction(SIGALRM, &sa, NULL); + alarm(SHUTDOWN_WATCHDOG_SEC); + + ret = flb_stop(ctx); + + alarm(0); + TEST_CHECK_(ret == 0, "flb_stop returned %d", ret); + TEST_CHECK(observer.worker_exit_called == 1); + TEST_CHECK(observer.exit_called == 1); + TEST_CHECK(observer.worker_exit_on_pipeline_thread != 0); + TEST_CHECK(observer.exit_after_worker_exit != 0); + TEST_CHECK(observer.exit_on_worker_thread != 0); + TEST_CHECK(ctx->config->is_running == FLB_FALSE); + + flb_destroy(ctx); +} + /* Test list */ TEST_LIST = { {"duplicate_stop_no_spin", flb_test_duplicate_stop_no_spin}, + {"stop_waits_for_cleanup", flb_test_stop_waits_for_cleanup}, {NULL, NULL} }; diff --git a/tests/runtime/in_simple_systems.c b/tests/runtime/in_simple_systems.c index f20dd477eb9..120c340690e 100644 --- a/tests/runtime/in_simple_systems.c +++ b/tests/runtime/in_simple_systems.c @@ -26,6 +26,8 @@ #endif #include "flb_tests_runtime.h" +#define TEST_STOP_WAIT_STEP_MS 10 +#define TEST_STOP_TIMEOUT_MS 10000 int64_t result_time; static inline int64_t set_result(int64_t v) @@ -55,6 +57,31 @@ static inline int64_t time_in_ms() return flb_time_to_millisec(&s); } +#if defined(FLB_SYSTEM_MACOS) +static int stop_engine(flb_ctx_t *ctx) +{ + int ret; + int trys; + + ret = flb_engine_exit(ctx->config); + TEST_CHECK_(ret >= 0, "requesting graceful engine shutdown"); + + for (trys = 0; + trys < TEST_STOP_TIMEOUT_MS / TEST_STOP_WAIT_STEP_MS && + ctx->status == FLB_LIB_OK; + trys++) { + flb_time_msleep(TEST_STOP_WAIT_STEP_MS); + } + + TEST_CHECK_(ctx->status != FLB_LIB_OK, "engine did not stop within %d ms", + TEST_STOP_TIMEOUT_MS); + + return flb_stop(ctx); +} + +#define flb_stop stop_engine +#endif + int callback_test(void* data, size_t size, void* cb_data) { if (size > 0) { @@ -71,16 +98,41 @@ struct callback_record { }; struct callback_records { + pthread_mutex_t mutex; int num_records; struct callback_record *records; }; +static int callback_records_count(struct callback_records *records) +{ + int count; + + pthread_mutex_lock(&records->mutex); + count = records->num_records; + pthread_mutex_unlock(&records->mutex); + + return count; +} + +static void callback_records_wait(struct callback_records *records, + int minimum_records, int wait_seconds) +{ + int trys; + + for (trys = 0; + trys < wait_seconds && callback_records_count(records) < minimum_records; + trys++) { + flb_time_msleep(1000); + } +} + int callback_add_record(void* data, size_t size, void* cb_data) { struct callback_records *ctx = (struct callback_records *)cb_data; if (size > 0) { flb_info("[test] flush record"); + pthread_mutex_lock(&ctx->mutex); if (ctx->records == NULL) { ctx->records = (struct callback_record *) flb_calloc(1, sizeof(struct callback_record)); @@ -90,11 +142,13 @@ int callback_add_record(void* data, size_t size, void* cb_data) (ctx->num_records+1)*sizeof(struct callback_record)); } if (ctx->records == NULL) { + pthread_mutex_unlock(&ctx->mutex); return -1; } ctx->records[ctx->num_records].size = size; ctx->records[ctx->num_records].data = data; ctx->num_records++; + pthread_mutex_unlock(&ctx->mutex); } return 0; } @@ -183,11 +237,11 @@ void do_test_records(char *system, void (*records_cb)(struct callback_records *) char *key; char *value; int idx; - int trys; struct flb_lib_out_cb cb; struct callback_records *records; records = flb_calloc(1, sizeof(struct callback_records)); + pthread_mutex_init(&records->mutex, NULL); records->num_records = 0; records->records = NULL; cb.cb = callback_add_record; @@ -221,9 +275,7 @@ void do_test_records(char *system, void (*records_cb)(struct callback_records *) /* Start test */ TEST_CHECK(flb_start(ctx) == 0); - for (trys = 0; trys < 5 && records->num_records <= 0; trys++) { - flb_time_msleep(1000); - } + callback_records_wait(records, 1, 5); flb_stop(ctx); @@ -233,12 +285,14 @@ void do_test_records(char *system, void (*records_cb)(struct callback_records *) flb_lib_free(records->records[idx].data); } flb_free(records->records); + pthread_mutex_destroy(&records->mutex); flb_free(records); flb_destroy(ctx); } -void do_test_records_single(char *system, void (*records_cb)(struct callback_records *), ...) +void do_test_records_single(char *system, int minimum_records, int wait_seconds, + void (*records_cb)(struct callback_records *), ...) { flb_ctx_t *ctx = NULL; int in_ffd; @@ -252,6 +306,7 @@ void do_test_records_single(char *system, void (*records_cb)(struct callback_rec struct callback_records *records; records = flb_calloc(1, sizeof(struct callback_records)); + pthread_mutex_init(&records->mutex, NULL); records->num_records = 0; records->records = NULL; cb.cb = callback_add_record; @@ -290,8 +345,7 @@ void do_test_records_single(char *system, void (*records_cb)(struct callback_rec /* Start test */ TEST_CHECK(flb_start(ctx) == 0); - /* 4 sec passed. It must have flushed */ - flb_time_msleep(5000); + callback_records_wait(records, minimum_records, wait_seconds); flb_stop(ctx); @@ -301,12 +355,14 @@ void do_test_records_single(char *system, void (*records_cb)(struct callback_rec flb_lib_free(records->records[i].data); } flb_free(records->records); + pthread_mutex_destroy(&records->mutex); flb_free(records); flb_destroy(ctx); } -void do_test_records_wait_time(char *system, int wait_time, void (*records_cb)(struct callback_records *), ...) +void do_test_records_wait_time(char *system, int minimum_records, int wait_seconds, + void (*records_cb)(struct callback_records *), ...) { flb_ctx_t *ctx = NULL; int in_ffd; @@ -319,6 +375,7 @@ void do_test_records_wait_time(char *system, int wait_time, void (*records_cb)(s struct callback_records *records; records = flb_calloc(1, sizeof(struct callback_records)); + pthread_mutex_init(&records->mutex, NULL); records->num_records = 0; records->records = NULL; cb.cb = callback_add_record; @@ -352,8 +409,7 @@ void do_test_records_wait_time(char *system, int wait_time, void (*records_cb)(s /* Start test */ TEST_CHECK(flb_start(ctx) == 0); - /* Set wait_time plus 2 sec passed. It must have flushed */ - flb_time_msleep((wait_time + 2) * 1000); + callback_records_wait(records, minimum_records, wait_seconds); flb_stop(ctx); @@ -363,6 +419,7 @@ void do_test_records_wait_time(char *system, int wait_time, void (*records_cb)(s flb_lib_free(records->records[i].data); } flb_free(records->records); + pthread_mutex_destroy(&records->mutex); flb_free(records); flb_destroy(ctx); @@ -537,31 +594,16 @@ void flb_test_dummy_records_message_copies_1(struct callback_records *records) void flb_test_dummy_records_message_copies_5(struct callback_records *records) { - int trys; - - for (trys = 0; trys < 5 && records->num_records < 5; trys++) { - flb_time_msleep(1000); - } TEST_CHECK(records->num_records >= 5); } void flb_test_dummy_records_message_copies_100(struct callback_records *records) { - int trys; - - for (trys = 0; trys < 100 && records->num_records < 100; trys++) { - flb_time_msleep(1000); - } TEST_CHECK(records->num_records >= 100); } void flb_test_dummy_records_message_rate(struct callback_records *records) { - int trys; - - for (trys = 0; trys < 20 && records->num_records < 20; trys++) { - flb_time_msleep(1000); - } TEST_CHECK(records->num_records >= 20); } @@ -601,27 +643,27 @@ void flb_test_in_dummy_flush(void) "start_time_nsec", "1999", "fixed_timestamp", "on", NULL); - do_test_records_single("dummy", flb_test_dummy_records_message_copies_1, + do_test_records_single("dummy", 1, 5, flb_test_dummy_records_message_copies_1, "copies", "1", NULL); - do_test_records_single("dummy", flb_test_dummy_records_message_copies_5, + do_test_records_single("dummy", 5, 5, flb_test_dummy_records_message_copies_5, "copies", "5", NULL); - do_test_records_single("dummy", flb_test_dummy_records_message_copies_100, + do_test_records_single("dummy", 100, 100, flb_test_dummy_records_message_copies_100, "copies", "100", NULL); - do_test_records_wait_time("dummy", 1, flb_test_dummy_records_message_rate, + do_test_records_wait_time("dummy", 20, 20, flb_test_dummy_records_message_rate, "rate", "20", NULL); - do_test_records_wait_time("dummy", 2, flb_test_dummy_records_message_interval_sec, + do_test_records_wait_time("dummy", 1, 4, flb_test_dummy_records_message_interval_sec, "interval_sec", "2", "interval_nsec", "0", NULL); - do_test_records_wait_time("dummy", 1, flb_test_dummy_records_message_interval_nsec, + do_test_records_wait_time("dummy", 1, 3, flb_test_dummy_records_message_interval_nsec, "interval_sec", "0", "interval_nsec", "700000000", NULL); - do_test_records_wait_time("dummy", 5, flb_test_dummy_records_message_flush_on_startup, + do_test_records_wait_time("dummy", 2, 7, flb_test_dummy_records_message_flush_on_startup, "interval_sec", "5", "interval_nsec", "0", "flush_on_startup", "true", diff --git a/tests/runtime/out_stackdriver.c b/tests/runtime/out_stackdriver.c index 7ea57cfd5fd..a4a73986d6c 100644 --- a/tests/runtime/out_stackdriver.c +++ b/tests/runtime/out_stackdriver.c @@ -50,6 +50,7 @@ #define STACKDRIVER_TEST_WAIT_STEP_MS 10 #define STACKDRIVER_TEST_TIMEOUT_MS 2000 +#define STACKDRIVER_TEST_STOP_TIMEOUT_MS 10000 typedef void (*stackdriver_test_callback)(void *, int, int, void *, size_t, void *); @@ -121,7 +122,33 @@ static void stackdriver_wait_for_formatter(void) } } +#if defined(FLB_SYSTEM_MACOS) +static int stackdriver_stop_engine(flb_ctx_t *ctx) +{ + int ret; + int trys; + + ret = flb_engine_exit(ctx->config); + TEST_CHECK_(ret >= 0, "requesting graceful engine shutdown"); + + for (trys = 0; + trys < STACKDRIVER_TEST_STOP_TIMEOUT_MS / STACKDRIVER_TEST_WAIT_STEP_MS && + ctx->status == FLB_LIB_OK; + trys++) { + flb_time_msleep(STACKDRIVER_TEST_WAIT_STEP_MS); + } + + TEST_CHECK_(ctx->status != FLB_LIB_OK, "engine did not stop within %d ms", + STACKDRIVER_TEST_STOP_TIMEOUT_MS); + + return flb_stop(ctx); +} +#endif + #define flb_output_set_test stackdriver_output_set_test +#if defined(FLB_SYSTEM_MACOS) +#define flb_stop stackdriver_stop_engine +#endif /* * Fluent Bit Stackdriver plugin, always set as payload a JSON strings contained in a diff --git a/tests/runtime/output_throttle_runtime.c b/tests/runtime/output_throttle_runtime.c new file mode 100644 index 00000000000..aade06e6706 --- /dev/null +++ b/tests/runtime/output_throttle_runtime.c @@ -0,0 +1,1017 @@ +/* -*- Mode: C; tab-width: 4; indent-tabs-mode: nil; c-basic-offset: 4 -*- */ + +#include +#include + +#include +#include +#include +#include +#include + +#include "flb_tests_runtime.h" +#include "../include/flb_tests_tmpdir.h" + +#define SCRIPT_CAPACITY 64 +#define TEST_TIMEOUT_MS 5000 +#define TEST_COOLDOWN_MS 1200 + +static int stop_engine(flb_ctx_t *ctx) +{ + uint64_t deadline; + int ret; + + /* On macOS flb_stop() cancels an active pipeline before its TLS cleanup. */ + ret = flb_engine_exit(ctx->config); + TEST_CHECK_(ret >= 0, "requesting graceful engine shutdown"); + + deadline = flb_output_throttle_now_ms() + TEST_TIMEOUT_MS; + while (ctx->status == FLB_LIB_OK && flb_output_throttle_now_ms() < deadline) { + flb_time_msleep(10); + } + TEST_CHECK_(ctx->status != FLB_LIB_OK, + "engine did not stop within %d ms", TEST_TIMEOUT_MS); + + return flb_stop(ctx); +} + +struct scripted_output { + pthread_mutex_t lock; + pthread_cond_t condition; + uint64_t blocked_calls; + uint64_t released_calls; + int outcomes[3]; + uint64_t hints[3]; + size_t outcome_count; + size_t calls; + uint64_t timestamps[SCRIPT_CAPACITY]; + uint64_t generations[SCRIPT_CAPACITY]; + char tag_markers[SCRIPT_CAPACITY]; +}; + +static void scripted_output_init(struct scripted_output *script, + int first_outcome) +{ + memset(script, 0, sizeof(struct scripted_output)); + pthread_mutex_init(&script->lock, NULL); + pthread_cond_init(&script->condition, NULL); + script->outcomes[0] = first_outcome; + script->outcomes[1] = FLB_OK; + script->outcomes[2] = FLB_OK; + script->hints[0] = TEST_COOLDOWN_MS; + script->outcome_count = 3; +} + +static void scripted_output_destroy(struct scripted_output *script) +{ + pthread_cond_destroy(&script->condition); + pthread_mutex_destroy(&script->lock); +} + +static int scripted_init(struct flb_output_instance *ins, + struct flb_config *config, void *data) +{ + (void) config; + flb_output_set_context(ins, data); + return 0; +} + +static void scripted_flush(struct flb_event_chunk *event_chunk, + struct flb_output_flush *out_flush, + struct flb_input_instance *i_ins, + void *out_context, + struct flb_config *config) +{ + int outcome; + size_t index; + struct scripted_output *script; + + (void) event_chunk; + (void) i_ins; + (void) config; + + script = out_context; + pthread_mutex_lock(&script->lock); + index = script->calls; + if (index < SCRIPT_CAPACITY) { + script->timestamps[index] = flb_output_throttle_now_ms(); + script->generations[index] = out_flush->admission_generation; + script->tag_markers[index] = event_chunk->tag[0]; + } + script->calls++; + pthread_cond_broadcast(&script->condition); + + while (index < 64 && + (script->blocked_calls & (UINT64_C(1) << index)) != 0 && + (script->released_calls & (UINT64_C(1) << index)) == 0) { + pthread_cond_wait(&script->condition, &script->lock); + } + + if (index < script->outcome_count) { + outcome = script->outcomes[index]; + } + else { + outcome = FLB_OK; + } + pthread_mutex_unlock(&script->lock); + + if (outcome == FLB_THROTTLE) { + flb_output_set_retry_after(out_flush, script->hints[index]); + } + FLB_OUTPUT_RETURN(outcome); +} + +static struct flb_output_plugin scripted_plugin = { + .name = "m6_scripted", + .description = "M6 deterministic throttle output", + .cb_init = scripted_init, + .cb_flush = scripted_flush, + .flags = 0 +}; + +static int register_scripted_plugin(flb_ctx_t *ctx, const char *name, int flags) +{ + struct flb_output_plugin *plugin; + + plugin = flb_malloc(sizeof(struct flb_output_plugin)); + if (plugin == NULL) { + return -1; + } + + memcpy(plugin, &scripted_plugin, sizeof(struct flb_output_plugin)); + plugin->name = (char *) name; + plugin->flags = flags; + mk_list_add(&plugin->_head, &ctx->config->out_plugins); + return 0; +} + +static size_t scripted_calls(struct scripted_output *script) +{ + size_t calls; + + pthread_mutex_lock(&script->lock); + calls = script->calls; + pthread_mutex_unlock(&script->lock); + return calls; +} + +static uint64_t scripted_timestamp(struct scripted_output *script, size_t index) +{ + uint64_t value; + + pthread_mutex_lock(&script->lock); + value = script->timestamps[index]; + pthread_mutex_unlock(&script->lock); + return value; +} + +static uint64_t scripted_generation(struct scripted_output *script, size_t index) +{ + uint64_t value; + + pthread_mutex_lock(&script->lock); + value = script->generations[index]; + pthread_mutex_unlock(&script->lock); + return value; +} + +static char scripted_tag_marker(struct scripted_output *script, size_t index) +{ + char value; + + pthread_mutex_lock(&script->lock); + value = script->tag_markers[index]; + pthread_mutex_unlock(&script->lock); + return value; +} + +static int wait_for_calls(struct scripted_output *script, size_t expected, + int timeout_ms) +{ + uint64_t deadline; + + deadline = flb_output_throttle_now_ms() + timeout_ms; + while (flb_output_throttle_now_ms() < deadline) { + if (scripted_calls(script) >= expected) { + return 0; + } + flb_time_msleep(10); + } + + return -1; +} + +static int wait_for_gate_events(struct flb_output_instance *output, + uint64_t expected, int timeout_ms) +{ + uint64_t deadline; + struct flb_output_throttle_snapshot snapshot; + + deadline = flb_output_throttle_now_ms() + timeout_ms; + while (flb_output_throttle_now_ms() < deadline) { + flb_output_throttle_snapshot(&output->throttle, &snapshot); + if (snapshot.events >= expected) { + return 0; + } + flb_time_msleep(10); + } + + return -1; +} + +static void release_call(struct scripted_output *script, size_t index) +{ + pthread_mutex_lock(&script->lock); + script->released_calls |= UINT64_C(1) << index; + pthread_cond_broadcast(&script->condition); + pthread_mutex_unlock(&script->lock); +} + +static double counter_value(struct cmt_counter *counter, const char *name) +{ + double value; + char *labels[] = {(char *) name}; + + if (cmt_counter_get_val(counter, 1, labels, &value) != 0) { + return -1; + } + return value; +} + +static double gauge_value(struct cmt_gauge *gauge, const char *name) +{ + double value; + char *labels[] = {(char *) name}; + + if (cmt_gauge_get_val(gauge, 1, labels, &value) != 0) { + return -1; + } + return value; +} + +static int wait_for_deferred_routes(struct flb_output_instance *output, + double expected, int timeout_ms) +{ + uint64_t deadline; + + deadline = flb_output_throttle_now_ms() + timeout_ms; + while (flb_output_throttle_now_ms() < deadline) { + if (gauge_value(output->cmt_throttle_deferred_routes, + flb_output_name(output)) >= expected) { + return 0; + } + flb_time_msleep(10); + } + + return -1; +} + +static void run_fanout_case(int workers, int flags) +{ + int input_id; + int output_a_id; + int output_b_id; + int ret; + uint64_t first_admission; + char workers_buffer[16]; + struct flb_output_instance *output_a; + struct scripted_output output_a_script; + struct scripted_output output_b_script; + flb_ctx_t *ctx; + const char *record = "[1, {\"m6\":true}]"; + + scripted_output_init(&output_a_script, FLB_THROTTLE); + scripted_output_init(&output_b_script, FLB_OK); + + ctx = flb_create(); + TEST_CHECK(ctx != NULL); + if (ctx == NULL) { + scripted_output_destroy(&output_a_script); + scripted_output_destroy(&output_b_script); + return; + } + + TEST_CHECK(register_scripted_plugin(ctx, "m6_scripted", flags) == 0); + TEST_CHECK(register_scripted_plugin(ctx, "m6_healthy", 0) == 0); + input_id = flb_input(ctx, "lib", NULL); + TEST_CHECK(input_id >= 0); + flb_input_set(ctx, input_id, "tag", "m6", NULL); + + output_a_id = flb_output(ctx, "m6_scripted", + (struct flb_lib_out_cb *) &output_a_script); + output_b_id = flb_output(ctx, "m6_healthy", + (struct flb_lib_out_cb *) &output_b_script); + TEST_CHECK(output_a_id >= 0); + TEST_CHECK(output_b_id >= 0); + flb_output_set(ctx, output_a_id, "match", "m6", "throttle", "true", + "throttle.base", "1", "throttle.cap", "1", + "retry_limit", "3", NULL); + flb_output_set(ctx, output_b_id, "match", "m6", NULL); + if (workers > 0) { + snprintf(workers_buffer, sizeof(workers_buffer), "%d", workers); + flb_output_set(ctx, output_a_id, "workers", workers_buffer, NULL); + } + flb_service_set(ctx, "Flush", "0.1", "Grace", "1", + "Log_Level", "error", NULL); + + ret = flb_start(ctx); + TEST_CHECK(ret == 0); + if (ret != 0) { + flb_destroy(ctx); + scripted_output_destroy(&output_a_script); + scripted_output_destroy(&output_b_script); + return; + } + + TEST_CHECK(flb_lib_push(ctx, input_id, record, strlen(record)) > 0); + TEST_CHECK(wait_for_calls(&output_a_script, 1, TEST_TIMEOUT_MS) == 0); + TEST_CHECK(wait_for_calls(&output_b_script, 1, TEST_TIMEOUT_MS) == 0); + first_admission = scripted_timestamp(&output_a_script, 0); + + TEST_CHECK(flb_lib_push(ctx, input_id, record, strlen(record)) > 0); + ret = wait_for_calls(&output_b_script, 2, 1000); + if (ret != 0) { + fprintf(stderr, "healthy output stalled: workers=%d flags=%d calls=%zu\n", + workers, flags, scripted_calls(&output_b_script)); + } + TEST_CHECK(ret == 0); + flb_time_msleep(100); + TEST_CHECK(scripted_calls(&output_a_script) == 1); + + output_a = flb_output_get_instance(ctx->config, output_a_id); + TEST_CHECK(output_a != NULL); + if (output_a != NULL) { + TEST_CHECK(counter_value(output_a->cmt_throttle_events, + flb_output_name(output_a)) == 1.0); + TEST_CHECK(gauge_value(output_a->cmt_throttle_active, + flb_output_name(output_a)) == 1.0); + TEST_CHECK(gauge_value(output_a->cmt_throttle_deferred_routes, + flb_output_name(output_a)) >= 1.0); + TEST_CHECK(counter_value(output_a->cmt_retries, + flb_output_name(output_a)) == 0.0); + } + + TEST_CHECK(wait_for_calls(&output_a_script, 2, TEST_TIMEOUT_MS) == 0); + TEST_CHECK(scripted_timestamp(&output_a_script, 1) >= + first_admission + TEST_COOLDOWN_MS); + TEST_CHECK(scripted_generation(&output_a_script, 1) > + scripted_generation(&output_a_script, 0)); + + if (output_a != NULL) { + flb_time_msleep(50); + TEST_CHECK(gauge_value(output_a->cmt_throttle_active, + flb_output_name(output_a)) == 0.0); + TEST_CHECK(gauge_value(output_a->cmt_throttle_remaining, + flb_output_name(output_a)) == 0.0); + TEST_CHECK(counter_value(output_a->cmt_throttle_duration, + flb_output_name(output_a)) >= 1.0); + TEST_CHECK(counter_value(output_a->cmt_retries, + flb_output_name(output_a)) == 0.0); + } + + stop_engine(ctx); + flb_destroy(ctx); + scripted_output_destroy(&output_a_script); + scripted_output_destroy(&output_b_script); +} + +static void test_fanout_and_modes(void) +{ + run_fanout_case(0, 0); + run_fanout_case(1, 0); + run_fanout_case(4, 0); + run_fanout_case(0, FLB_OUTPUT_SYNCHRONOUS); + run_fanout_case(0, FLB_OUTPUT_NO_MULTIPLEX); +} + +static void test_inflight_before_publication(void) +{ + int input_id; + int output_id; + int ret; + struct scripted_output script; + flb_ctx_t *ctx; + const char *record = "[1, {\"barrier\":true}]"; + + scripted_output_init(&script, FLB_THROTTLE); + script.blocked_calls = UINT64_C(1); + ctx = flb_create(); + TEST_CHECK(ctx != NULL); + if (ctx == NULL) { + scripted_output_destroy(&script); + return; + } + + TEST_CHECK(register_scripted_plugin(ctx, "m6_scripted", 0) == 0); + input_id = flb_input(ctx, "lib", NULL); + flb_input_set(ctx, input_id, "tag", "barrier", NULL); + output_id = flb_output(ctx, "m6_scripted", + (struct flb_lib_out_cb *) &script); + flb_output_set(ctx, output_id, "match", "barrier", "workers", "4", + "throttle", "true", "throttle.base", "1", + "throttle.cap", "1", "retry_limit", "3", NULL); + flb_service_set(ctx, "Flush", "0.1", "Grace", "1", + "Log_Level", "error", NULL); + ret = flb_start(ctx); + TEST_CHECK(ret == 0); + + TEST_CHECK(flb_lib_push(ctx, input_id, record, strlen(record)) > 0); + TEST_CHECK(wait_for_calls(&script, 1, TEST_TIMEOUT_MS) == 0); + TEST_CHECK(flb_lib_push(ctx, input_id, record, strlen(record)) > 0); + TEST_CHECK(wait_for_calls(&script, 2, TEST_TIMEOUT_MS) == 0); + TEST_CHECK(scripted_generation(&script, 0) == scripted_generation(&script, 1)); + + release_call(&script, 0); + + TEST_CHECK(flb_lib_push(ctx, input_id, record, strlen(record)) > 0); + flb_time_msleep(200); + TEST_CHECK(scripted_calls(&script) == 2); + TEST_CHECK(wait_for_calls(&script, 3, TEST_TIMEOUT_MS) == 0); + TEST_CHECK(scripted_timestamp(&script, 2) >= + scripted_timestamp(&script, 0) + TEST_COOLDOWN_MS); + + stop_engine(ctx); + flb_destroy(ctx); + scripted_output_destroy(&script); +} + +static void test_extended_deadline_rechecks_old_timer(void) +{ + int input_id; + int output_id; + int ret; + uint64_t first_release; + uint64_t second_release; + struct flb_output_instance *output; + struct scripted_output script; + flb_ctx_t *ctx; + const char *record = "[1, {\"extension\":true}]"; + + scripted_output_init(&script, FLB_THROTTLE); + script.outcomes[1] = FLB_THROTTLE; + script.hints[0] = TEST_COOLDOWN_MS; + script.hints[1] = 2000; + script.blocked_calls = UINT64_C(3); + ctx = flb_create(); + TEST_CHECK(ctx != NULL); + if (ctx == NULL) { + scripted_output_destroy(&script); + return; + } + + TEST_CHECK(register_scripted_plugin(ctx, "m6_scripted", 0) == 0); + input_id = flb_input(ctx, "lib", NULL); + flb_input_set(ctx, input_id, "tag", "extension", NULL); + output_id = flb_output(ctx, "m6_scripted", + (struct flb_lib_out_cb *) &script); + flb_output_set(ctx, output_id, "match", "extension", "workers", "4", + "throttle", "true", "throttle.base", "1", + "throttle.cap", "1", "retry_limit", "3", NULL); + flb_service_set(ctx, "Flush", "0.1", "Grace", "1", + "Log_Level", "error", NULL); + ret = flb_start(ctx); + TEST_CHECK(ret == 0); + + TEST_CHECK(flb_lib_push(ctx, input_id, record, strlen(record)) > 0); + TEST_CHECK(wait_for_calls(&script, 1, TEST_TIMEOUT_MS) == 0); + TEST_CHECK(flb_lib_push(ctx, input_id, record, strlen(record)) > 0); + TEST_CHECK(wait_for_calls(&script, 2, TEST_TIMEOUT_MS) == 0); + + output = flb_output_get_instance(ctx->config, output_id); + TEST_CHECK(output != NULL); + + first_release = flb_output_throttle_now_ms(); + release_call(&script, 0); + if (output != NULL) { + TEST_CHECK(wait_for_gate_events(output, 1, TEST_TIMEOUT_MS) == 0); + } + flb_time_msleep(300); + second_release = flb_output_throttle_now_ms(); + release_call(&script, 1); + TEST_CHECK(flb_lib_push(ctx, input_id, record, strlen(record)) > 0); + + while (flb_output_throttle_now_ms() < first_release + 1400) { + flb_time_msleep(10); + } + TEST_CHECK(scripted_calls(&script) == 2); + TEST_CHECK(wait_for_calls(&script, 3, TEST_TIMEOUT_MS) == 0); + TEST_CHECK(scripted_timestamp(&script, 2) >= second_release + script.hints[1]); + + stop_engine(ctx); + flb_destroy(ctx); + scripted_output_destroy(&script); +} + +static void test_repeated_throttle_does_not_spend_retry_limit(void) +{ + int input_id; + int output_id; + int ret; + struct flb_output_instance *output; + struct scripted_output script; + flb_ctx_t *ctx; + const char *record = "[1, {\"repeated_throttle\":true}]"; + + scripted_output_init(&script, FLB_THROTTLE); + script.outcomes[1] = FLB_THROTTLE; + script.outcomes[2] = FLB_OK; + script.hints[1] = TEST_COOLDOWN_MS; + + ctx = flb_create(); + TEST_CHECK(ctx != NULL); + if (ctx == NULL) { + scripted_output_destroy(&script); + return; + } + + TEST_CHECK(register_scripted_plugin(ctx, "m6_scripted", 0) == 0); + input_id = flb_input(ctx, "lib", NULL); + TEST_CHECK(input_id >= 0); + flb_input_set(ctx, input_id, "tag", "repeated_throttle", NULL); + output_id = flb_output(ctx, "m6_scripted", + (struct flb_lib_out_cb *) &script); + TEST_CHECK(output_id >= 0); + flb_output_set(ctx, output_id, "match", "repeated_throttle", + "workers", "4", "throttle", "true", + "throttle.base", "1", "throttle.cap", "1", + "retry_limit", "1", NULL); + flb_service_set(ctx, "Flush", "0.1", "Grace", "1", + "Log_Level", "error", NULL); + ret = flb_start(ctx); + TEST_CHECK(ret == 0); + if (ret != 0) { + flb_destroy(ctx); + scripted_output_destroy(&script); + return; + } + + TEST_CHECK(flb_lib_push(ctx, input_id, record, strlen(record)) > 0); + TEST_CHECK(wait_for_calls(&script, 3, TEST_TIMEOUT_MS) == 0); + TEST_CHECK(scripted_calls(&script) == 3); + TEST_CHECK(scripted_timestamp(&script, 1) >= + scripted_timestamp(&script, 0) + TEST_COOLDOWN_MS); + TEST_CHECK(scripted_timestamp(&script, 2) >= + scripted_timestamp(&script, 1) + TEST_COOLDOWN_MS); + + output = flb_output_get_instance(ctx->config, output_id); + TEST_CHECK(output != NULL); + if (output != NULL) { + TEST_CHECK(counter_value(output->cmt_retries, + flb_output_name(output)) == 0.0); + TEST_CHECK(counter_value(output->cmt_retries_failed, + flb_output_name(output)) == 0.0); + } + + stop_engine(ctx); + flb_destroy(ctx); + scripted_output_destroy(&script); +} + +static void test_throttle_preserves_existing_retry_attempt(void) +{ + int input_id; + int output_id; + int ret; + struct flb_output_instance *output; + struct scripted_output script; + flb_ctx_t *ctx; + const char *record = "[1, {\"retry_then_throttle\":true}]"; + + scripted_output_init(&script, FLB_RETRY); + script.outcomes[1] = FLB_THROTTLE; + script.outcomes[2] = FLB_OK; + script.hints[1] = TEST_COOLDOWN_MS; + + ctx = flb_create(); + TEST_CHECK(ctx != NULL); + if (ctx == NULL) { + scripted_output_destroy(&script); + return; + } + + TEST_CHECK(register_scripted_plugin(ctx, "m6_scripted", 0) == 0); + input_id = flb_input(ctx, "lib", NULL); + TEST_CHECK(input_id >= 0); + flb_input_set(ctx, input_id, "tag", "retry_then_throttle", NULL); + output_id = flb_output(ctx, "m6_scripted", + (struct flb_lib_out_cb *) &script); + TEST_CHECK(output_id >= 0); + flb_output_set(ctx, output_id, "match", "retry_then_throttle", + "workers", "4", "throttle", "true", + "throttle.base", "1", "throttle.cap", "1", + "retry_limit", "1", NULL); + flb_service_set(ctx, "Flush", "0.1", "Grace", "1", + "Log_Level", "error", "scheduler.base", "1", + "scheduler.cap", "1", NULL); + ret = flb_start(ctx); + TEST_CHECK(ret == 0); + if (ret != 0) { + flb_destroy(ctx); + scripted_output_destroy(&script); + return; + } + + TEST_CHECK(flb_lib_push(ctx, input_id, record, strlen(record)) > 0); + TEST_CHECK(wait_for_calls(&script, 3, TEST_TIMEOUT_MS) == 0); + TEST_CHECK(scripted_calls(&script) == 3); + TEST_CHECK(scripted_timestamp(&script, 2) >= + scripted_timestamp(&script, 1) + TEST_COOLDOWN_MS); + + output = flb_output_get_instance(ctx->config, output_id); + TEST_CHECK(output != NULL); + if (output != NULL) { + TEST_CHECK(counter_value(output->cmt_retries, + flb_output_name(output)) == 1.0); + TEST_CHECK(counter_value(output->cmt_retries_failed, + flb_output_name(output)) == 0.0); + } + + stop_engine(ctx); + flb_destroy(ctx); + scripted_output_destroy(&script); +} + +static void run_deferred_wakeup_serialization_case(int flags) +{ + int input_id; + int output_id; + int ret; + struct flb_output_instance *output; + struct scripted_output script; + flb_ctx_t *ctx; + const char *record = "[1, {\"serialized_wakeup\":true}]"; + + scripted_output_init(&script, FLB_THROTTLE); + script.blocked_calls = UINT64_C(1) << 1; + ctx = flb_create(); + TEST_CHECK(ctx != NULL); + if (ctx == NULL) { + scripted_output_destroy(&script); + return; + } + + TEST_CHECK(register_scripted_plugin(ctx, "m6_serialized", flags) == 0); + input_id = flb_input(ctx, "lib", NULL); + TEST_CHECK(input_id >= 0); + flb_input_set(ctx, input_id, "tag", "serialized_wakeup", NULL); + output_id = flb_output(ctx, "m6_serialized", + (struct flb_lib_out_cb *) &script); + TEST_CHECK(output_id >= 0); + flb_output_set(ctx, output_id, "match", "serialized_wakeup", + "throttle", "true", "throttle.base", "1", + "throttle.cap", "1", "retry_limit", "3", NULL); + if ((flags & FLB_OUTPUT_SYNCHRONOUS) == 0) { + flb_output_set(ctx, output_id, "workers", "4", NULL); + } + flb_service_set(ctx, "Flush", "0.1", "Grace", "1", + "Log_Level", "error", NULL); + ret = flb_start(ctx); + TEST_CHECK(ret == 0); + if (ret != 0) { + flb_destroy(ctx); + scripted_output_destroy(&script); + return; + } + + output = flb_output_get_instance(ctx->config, output_id); + TEST_CHECK(output != NULL); + TEST_CHECK(flb_lib_push(ctx, input_id, record, strlen(record)) > 0); + TEST_CHECK(wait_for_calls(&script, 1, TEST_TIMEOUT_MS) == 0); + if (output != NULL) { + TEST_CHECK(wait_for_gate_events(output, 1, TEST_TIMEOUT_MS) == 0); + } + + /* Create separate tasks while the output gate is closed. */ + TEST_CHECK(flb_lib_push(ctx, input_id, record, strlen(record)) > 0); + flb_time_msleep(200); + TEST_CHECK(flb_lib_push(ctx, input_id, record, strlen(record)) > 0); + flb_time_msleep(200); + TEST_CHECK(scripted_calls(&script) == 1); + + TEST_CHECK(wait_for_calls(&script, 2, TEST_TIMEOUT_MS) == 0); + flb_time_msleep(300); + TEST_CHECK(scripted_calls(&script) == 2); + + release_call(&script, 1); + TEST_CHECK(wait_for_calls(&script, 3, TEST_TIMEOUT_MS) == 0); + + stop_engine(ctx); + flb_destroy(ctx); + scripted_output_destroy(&script); +} + +static void test_deferred_wakeups_preserve_serialization(void) +{ + run_deferred_wakeup_serialization_case(FLB_OUTPUT_NO_MULTIPLEX); + run_deferred_wakeup_serialization_case(FLB_OUTPUT_SYNCHRONOUS); +} + +static void test_no_multiplex_prioritizes_deferred_routes(void) +{ + int old_input_id; + int new_input_id; + int output_id; + int healthy_output_id; + int ret; + struct flb_output_instance *output; + struct scripted_output script; + struct scripted_output healthy_script; + flb_ctx_t *ctx; + const char *record = "[1, {\"deferred_priority\":true}]"; + + scripted_output_init(&script, FLB_THROTTLE); + script.hints[0] = 60000; + scripted_output_init(&healthy_script, FLB_OK); + ctx = flb_create(); + TEST_CHECK(ctx != NULL); + if (ctx == NULL) { + scripted_output_destroy(&script); + scripted_output_destroy(&healthy_script); + return; + } + + TEST_CHECK(register_scripted_plugin(ctx, "m6_priority", + FLB_OUTPUT_NO_MULTIPLEX) == 0); + TEST_CHECK(register_scripted_plugin(ctx, "m6_priority_healthy", 0) == 0); + old_input_id = flb_input(ctx, "lib", NULL); + new_input_id = flb_input(ctx, "lib", NULL); + TEST_CHECK(old_input_id >= 0); + TEST_CHECK(new_input_id >= 0); + flb_input_set(ctx, old_input_id, "tag", "old", NULL); + flb_input_set(ctx, new_input_id, "tag", "new", NULL); + output_id = flb_output(ctx, "m6_priority", + (struct flb_lib_out_cb *) &script); + TEST_CHECK(output_id >= 0); + flb_output_set(ctx, output_id, "match", "*", "workers", "4", + "throttle", "true", "throttle.base", "1", + "throttle.cap", "1", "retry_limit", "3", NULL); + healthy_output_id = flb_output(ctx, "m6_priority_healthy", + (struct flb_lib_out_cb *) &healthy_script); + TEST_CHECK(healthy_output_id >= 0); + flb_output_set(ctx, healthy_output_id, "match", "new", NULL); + flb_service_set(ctx, "Flush", "0.1", "Grace", "1", + "Log_Level", "error", NULL); + + ret = flb_start(ctx); + TEST_CHECK(ret == 0); + if (ret != 0) { + flb_destroy(ctx); + scripted_output_destroy(&script); + scripted_output_destroy(&healthy_script); + return; + } + + output = flb_output_get_instance(ctx->config, output_id); + TEST_CHECK(output != NULL); + TEST_CHECK(flb_lib_push(ctx, old_input_id, record, strlen(record)) > 0); + TEST_CHECK(wait_for_calls(&script, 1, TEST_TIMEOUT_MS) == 0); + TEST_CHECK(scripted_tag_marker(&script, 0) == 'o'); + if (output != NULL) { + TEST_CHECK(wait_for_gate_events(output, 1, TEST_TIMEOUT_MS) == 0); + TEST_CHECK(wait_for_deferred_routes(output, 1, TEST_TIMEOUT_MS) == 0); + + /* Open the gate while its old deferred-route wakeup remains far away. */ + pthread_mutex_lock(&output->throttle.lock); + output->throttle.until_ms = flb_output_throttle_now_ms(); + pthread_mutex_unlock(&output->throttle.lock); + } + + TEST_CHECK(flb_lib_push(ctx, new_input_id, record, strlen(record)) > 0); + TEST_CHECK(wait_for_calls(&healthy_script, 1, TEST_TIMEOUT_MS) == 0); + flb_time_msleep(300); + TEST_CHECK(scripted_calls(&script) == 1); + if (output != NULL) { + TEST_CHECK(wait_for_deferred_routes(output, 2, TEST_TIMEOUT_MS) == 0); + } + + stop_engine(ctx); + flb_destroy(ctx); + scripted_output_destroy(&script); + scripted_output_destroy(&healthy_script); +} + +static void run_no_retry_case(const char *storage_type) +{ + int input_id; + int output_id; + int ret; + uint64_t first_admission; + char *storage_path; + struct flb_output_instance *output; + struct scripted_output script; + flb_ctx_t *ctx; + const char *record = "[1, {\"no_retry\":true}]"; + + storage_path = NULL; + scripted_output_init(&script, FLB_THROTTLE); + ctx = flb_create(); + TEST_CHECK(ctx != NULL); + if (ctx == NULL) { + scripted_output_destroy(&script); + return; + } + + if (storage_type != NULL && strcmp(storage_type, "filesystem") == 0) { + storage_path = flb_test_tmpdir_cat("/flb-output-throttle-XXXXXX"); + TEST_CHECK(storage_path != NULL); + if (storage_path == NULL || mkdtemp(storage_path) == NULL) { + flb_free(storage_path); + flb_destroy(ctx); + scripted_output_destroy(&script); + return; + } + flb_service_set(ctx, "storage.path", storage_path, NULL); + } + + TEST_CHECK(register_scripted_plugin(ctx, "m6_scripted", 0) == 0); + input_id = flb_input(ctx, "lib", NULL); + TEST_CHECK(input_id >= 0); + flb_input_set(ctx, input_id, "tag", "no_retry", NULL); + if (storage_type != NULL) { + flb_input_set(ctx, input_id, "storage.type", storage_type, NULL); + } + output_id = flb_output(ctx, "m6_scripted", + (struct flb_lib_out_cb *) &script); + TEST_CHECK(output_id >= 0); + flb_output_set(ctx, output_id, "match", "no_retry", "workers", "4", + "throttle", "true", "throttle.base", "1", + "throttle.cap", "1", "retry_limit", "no_retries", NULL); + flb_service_set(ctx, "Flush", "0.1", "Grace", "1", + "Log_Level", "error", NULL); + ret = flb_start(ctx); + TEST_CHECK(ret == 0); + if (ret != 0) { + flb_destroy(ctx); + if (storage_path != NULL) { + cio_utils_recursive_delete(storage_path); + flb_free(storage_path); + } + scripted_output_destroy(&script); + return; + } + + output = flb_output_get_instance(ctx->config, output_id); + TEST_CHECK(output != NULL); + TEST_CHECK(flb_lib_push(ctx, input_id, record, strlen(record)) > 0); + TEST_CHECK(wait_for_calls(&script, 1, TEST_TIMEOUT_MS) == 0); + first_admission = scripted_timestamp(&script, 0); + if (output != NULL) { + TEST_CHECK(wait_for_gate_events(output, 1, TEST_TIMEOUT_MS) == 0); + } + + TEST_CHECK(flb_lib_push(ctx, input_id, record, strlen(record)) > 0); + flb_time_msleep(200); + TEST_CHECK(scripted_calls(&script) == 1); + TEST_CHECK(wait_for_calls(&script, 2, TEST_TIMEOUT_MS) == 0); + TEST_CHECK(scripted_timestamp(&script, 1) >= first_admission + TEST_COOLDOWN_MS); + if (output != NULL) { + TEST_CHECK(counter_value(output->cmt_throttle_events, + flb_output_name(output)) == 1.0); + TEST_CHECK(counter_value(output->cmt_retries, + flb_output_name(output)) == 0.0); + } + + stop_engine(ctx); + flb_destroy(ctx); + if (storage_path != NULL) { + cio_utils_recursive_delete(storage_path); + flb_free(storage_path); + } + scripted_output_destroy(&script); +} + +static void test_no_retry_gate_storage_modes(void) +{ + run_no_retry_case(NULL); + run_no_retry_case("filesystem"); + run_no_retry_case("memrb"); +} + +static void test_shutdown_cancels_long_cooldown(void) +{ + int input_id; + int output_id; + int ret; + uint64_t stop_started; + struct flb_output_instance *output; + struct scripted_output script; + flb_ctx_t *ctx; + const char *record = "[1, {\"shutdown\":true}]"; + + scripted_output_init(&script, FLB_THROTTLE); + script.hints[0] = 60000; + ctx = flb_create(); + TEST_CHECK(ctx != NULL); + if (ctx == NULL) { + scripted_output_destroy(&script); + return; + } + + TEST_CHECK(register_scripted_plugin(ctx, "m6_scripted", 0) == 0); + input_id = flb_input(ctx, "lib", NULL); + TEST_CHECK(input_id >= 0); + flb_input_set(ctx, input_id, "tag", "shutdown", NULL); + output_id = flb_output(ctx, "m6_scripted", + (struct flb_lib_out_cb *) &script); + TEST_CHECK(output_id >= 0); + flb_output_set(ctx, output_id, "match", "shutdown", "workers", "4", + "throttle", "true", "throttle.base", "1", + "throttle.cap", "1", "retry_limit", "3", NULL); + flb_service_set(ctx, "Flush", "0.1", "Grace", "1", + "Log_Level", "error", NULL); + ret = flb_start(ctx); + TEST_CHECK(ret == 0); + if (ret != 0) { + flb_destroy(ctx); + scripted_output_destroy(&script); + return; + } + + output = flb_output_get_instance(ctx->config, output_id); + TEST_CHECK(output != NULL); + TEST_CHECK(flb_lib_push(ctx, input_id, record, strlen(record)) > 0); + TEST_CHECK(wait_for_calls(&script, 1, TEST_TIMEOUT_MS) == 0); + if (output != NULL) { + TEST_CHECK(wait_for_gate_events(output, 1, TEST_TIMEOUT_MS) == 0); + } + + stop_started = flb_output_throttle_now_ms(); + stop_engine(ctx); + TEST_CHECK(flb_output_throttle_now_ms() - stop_started < 5000); + TEST_CHECK(scripted_calls(&script) == 1); + flb_destroy(ctx); + scripted_output_destroy(&script); +} + +static void test_disabled_mode_is_not_gated(void) +{ + int input_id; + int output_id; + int ret; + struct flb_output_instance *output; + struct scripted_output script; + flb_ctx_t *ctx; + const char *record = "[1, {\"disabled\":true}]"; + + scripted_output_init(&script, FLB_THROTTLE); + ctx = flb_create(); + TEST_CHECK(ctx != NULL); + if (ctx == NULL) { + scripted_output_destroy(&script); + return; + } + + TEST_CHECK(register_scripted_plugin(ctx, "m6_scripted", 0) == 0); + input_id = flb_input(ctx, "lib", NULL); + flb_input_set(ctx, input_id, "tag", "disabled", NULL); + output_id = flb_output(ctx, "m6_scripted", + (struct flb_lib_out_cb *) &script); + flb_output_set(ctx, output_id, "match", "disabled", "workers", "4", + "retry_limit", "3", NULL); + flb_service_set(ctx, "Flush", "0.1", "Grace", "1", + "Log_Level", "error", NULL); + ret = flb_start(ctx); + TEST_CHECK(ret == 0); + + TEST_CHECK(flb_lib_push(ctx, input_id, record, strlen(record)) > 0); + TEST_CHECK(wait_for_calls(&script, 1, TEST_TIMEOUT_MS) == 0); + TEST_CHECK(flb_lib_push(ctx, input_id, record, strlen(record)) > 0); + TEST_CHECK(wait_for_calls(&script, 2, 1000) == 0); + TEST_CHECK(scripted_generation(&script, 0) == 0); + TEST_CHECK(scripted_generation(&script, 1) == 0); + + output = flb_output_get_instance(ctx->config, output_id); + TEST_CHECK(output != NULL); + if (output != NULL) { + TEST_CHECK(counter_value(output->cmt_throttle_events, + flb_output_name(output)) == 0.0); + TEST_CHECK(gauge_value(output->cmt_throttle_active, + flb_output_name(output)) == 0.0); + TEST_CHECK(gauge_value(output->cmt_throttle_deferred_routes, + flb_output_name(output)) == 0.0); + } + + stop_engine(ctx); + flb_destroy(ctx); + scripted_output_destroy(&script); +} + +TEST_LIST = { + {"fanout_and_modes", test_fanout_and_modes}, + {"inflight_before_publication", test_inflight_before_publication}, + {"extended_deadline_rechecks_old_timer", + test_extended_deadline_rechecks_old_timer}, + {"repeated_throttle_does_not_spend_retry_limit", + test_repeated_throttle_does_not_spend_retry_limit}, + {"throttle_preserves_existing_retry_attempt", + test_throttle_preserves_existing_retry_attempt}, + {"deferred_wakeups_preserve_serialization", + test_deferred_wakeups_preserve_serialization}, + {"no_multiplex_prioritizes_deferred_routes", + test_no_multiplex_prioritizes_deferred_routes}, + {"no_retry_gate_storage_modes", test_no_retry_gate_storage_modes}, + {"shutdown_cancels_long_cooldown", + test_shutdown_cancels_long_cooldown}, + {"disabled_mode_is_not_gated", test_disabled_mode_is_not_gated}, + {NULL, NULL} +};