Created
April 16, 2026 10:36
-
-
Save andreypfau/02d0987cf1bc1ebff447f0c6b3cf5f8b to your computer and use it in GitHub Desktop.
Standalone repro for cocoon AnswerPostprocessor SSE bug (issue TelegramMessenger/cocoon#49)
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| #include "runners/helpers/ValidateRequest.h" | |
| #include "td/utils/Slice.h" | |
| #include "auto/tl/cocoon_api.h" | |
| #include <cstdio> | |
| #include <string> | |
| #include <vector> | |
| /* | |
| * Standalone repro and regression test for AnswerPostprocessor. | |
| * | |
| * Three scenarios in one binary: | |
| * 1. Whole SSE events delivered as one slice each. Exercises the streaming | |
| * fix end-to-end: before the fix, output was empty; after, SSE events | |
| * flow through and usage is aggregated from the final chunk. | |
| * 2. Byte-split SSE: the same stream fed one byte at a time. Verifies the | |
| * buffering across slice boundaries works (most realistic network case). | |
| * 3. Raw JSON (non-streaming) response as a single slice. Verifies backward | |
| * compat with pre-fix callers that pass whole JSON documents. | |
| * | |
| * Multipliers mirror production-root values so adjust_tokens() passes counts | |
| * through (count * coef * 0.001 * mult * 0.0001 = count at coef=1000, mult=10000). | |
| */ | |
| using namespace cocoon; | |
| static AnswerPostprocessor make_pp() { | |
| td::Bits256 zero = td::Bits256::zero(); | |
| return AnswerPostprocessor(/*coef*/ 1000, | |
| /*prompt_mult*/ 10000, | |
| /*cached_mult*/ 10000, | |
| /*completion_mult*/ 10000, | |
| /*reasoning_mult*/ 10000, | |
| /*price_per_token*/ 1, | |
| /*sender_priv*/ zero, | |
| /*receiver_pub*/ zero); | |
| } | |
| static const std::vector<std::string> kSseChunks = { | |
| "data: {\"id\":\"c1\",\"object\":\"chat.completion.chunk\"," | |
| "\"choices\":[{\"delta\":{\"content\":\"hi\"},\"index\":0}]}\n\n", | |
| "data: {\"id\":\"c1\",\"object\":\"chat.completion.chunk\"," | |
| "\"choices\":[{\"delta\":{\"content\":\" there\"},\"index\":0}]}\n\n", | |
| "data: {\"id\":\"c1\",\"object\":\"chat.completion.chunk\"," | |
| "\"choices\":[{\"delta\":{},\"finish_reason\":\"stop\",\"index\":0}]," | |
| "\"usage\":{\"prompt_tokens\":14," | |
| "\"prompt_tokens_details\":{\"cached_tokens\":3}," | |
| "\"completion_tokens\":7," | |
| "\"completion_tokens_details\":{\"reasoning_tokens\":0}}}\n\n", | |
| "data: [DONE]\n\n", | |
| }; | |
| static const std::string kRawJson = | |
| "{\"id\":\"c1\",\"object\":\"chat.completion\"," | |
| "\"choices\":[{\"message\":{\"role\":\"assistant\",\"content\":\"hi there\"}," | |
| "\"finish_reason\":\"stop\",\"index\":0}]," | |
| "\"usage\":{\"prompt_tokens\":14," | |
| "\"prompt_tokens_details\":{\"cached_tokens\":3}," | |
| "\"completion_tokens\":7," | |
| "\"completion_tokens_details\":{\"reasoning_tokens\":0}}}"; | |
| struct Result { | |
| std::string forwarded; | |
| td::int64 prompt, cached, completion, reasoning, total; | |
| }; | |
| static Result run_scenario(const std::vector<std::string> &slices) { | |
| auto pp = make_pp(); | |
| std::string forwarded; | |
| for (const auto &s : slices) { | |
| forwarded += pp.add_next_answer_slice(s); | |
| } | |
| forwarded += pp.finalize(); | |
| auto u = pp.usage(); | |
| return Result{std::move(forwarded), | |
| u->prompt_tokens_used_, u->cached_tokens_used_, | |
| u->completion_tokens_used_, u->reasoning_tokens_used_, | |
| u->total_tokens_used_}; | |
| } | |
| static bool check_result(const char *name, const Result &r, size_t min_bytes, | |
| td::int64 expect_completion) { | |
| std::fprintf(stdout, "\n=== %s: forwarded %zu bytes ===\n%s=== end ===\n", | |
| name, r.forwarded.size(), r.forwarded.c_str()); | |
| std::fprintf(stdout, "usage: prompt=%lld cached=%lld completion=%lld " | |
| "reasoning=%lld total=%lld\n", | |
| (long long)r.prompt, (long long)r.cached, (long long)r.completion, | |
| (long long)r.reasoning, (long long)r.total); | |
| if (r.forwarded.size() < min_bytes || r.completion != expect_completion) { | |
| std::fprintf(stderr, "[FAIL] %s: bytes=%zu (min %zu), completion=%lld " | |
| "(want %lld)\n", | |
| name, r.forwarded.size(), min_bytes, (long long)r.completion, | |
| (long long)expect_completion); | |
| return false; | |
| } | |
| std::fprintf(stderr, "[OK] %s\n", name); | |
| return true; | |
| } | |
| int main() { | |
| bool ok = true; | |
| // Scenario 1: whole SSE events. | |
| ok &= check_result("sse_whole_events", run_scenario(kSseChunks), /*min*/ 500, | |
| /*completion*/ 7); | |
| // Scenario 2: byte-by-byte SSE. Catches buffering bugs across slice boundaries. | |
| std::string concat; | |
| for (const auto &c : kSseChunks) concat += c; | |
| std::vector<std::string> byte_slices; | |
| byte_slices.reserve(concat.size()); | |
| for (char c : concat) byte_slices.emplace_back(1, c); | |
| ok &= check_result("sse_byte_split", run_scenario(byte_slices), /*min*/ 500, | |
| /*completion*/ 7); | |
| // Scenario 3: raw non-stream JSON as a single slice. | |
| ok &= check_result("raw_json_non_stream", run_scenario({kRawJson}), | |
| /*min*/ 200, /*completion*/ 7); | |
| return ok ? 0 : 1; | |
| } |
Sign up for free
to join this conversation on GitHub.
Already have an account?
Sign in to comment