Skip to content

Instantly share code, notes, and snippets.

@andreypfau
Created April 16, 2026 10:36
Show Gist options
  • Select an option

  • Save andreypfau/02d0987cf1bc1ebff447f0c6b3cf5f8b to your computer and use it in GitHub Desktop.

Select an option

Save andreypfau/02d0987cf1bc1ebff447f0c6b3cf5f8b to your computer and use it in GitHub Desktop.
Standalone repro for cocoon AnswerPostprocessor SSE bug (issue TelegramMessenger/cocoon#49)
#include "runners/helpers/ValidateRequest.h"
#include "td/utils/Slice.h"
#include "auto/tl/cocoon_api.h"
#include <cstdio>
#include <string>
#include <vector>
/*
* Standalone repro and regression test for AnswerPostprocessor.
*
* Three scenarios in one binary:
* 1. Whole SSE events delivered as one slice each. Exercises the streaming
* fix end-to-end: before the fix, output was empty; after, SSE events
* flow through and usage is aggregated from the final chunk.
* 2. Byte-split SSE: the same stream fed one byte at a time. Verifies the
* buffering across slice boundaries works (most realistic network case).
* 3. Raw JSON (non-streaming) response as a single slice. Verifies backward
* compat with pre-fix callers that pass whole JSON documents.
*
* Multipliers mirror production-root values so adjust_tokens() passes counts
* through (count * coef * 0.001 * mult * 0.0001 = count at coef=1000, mult=10000).
*/
using namespace cocoon;
static AnswerPostprocessor make_pp() {
td::Bits256 zero = td::Bits256::zero();
return AnswerPostprocessor(/*coef*/ 1000,
/*prompt_mult*/ 10000,
/*cached_mult*/ 10000,
/*completion_mult*/ 10000,
/*reasoning_mult*/ 10000,
/*price_per_token*/ 1,
/*sender_priv*/ zero,
/*receiver_pub*/ zero);
}
static const std::vector<std::string> kSseChunks = {
"data: {\"id\":\"c1\",\"object\":\"chat.completion.chunk\","
"\"choices\":[{\"delta\":{\"content\":\"hi\"},\"index\":0}]}\n\n",
"data: {\"id\":\"c1\",\"object\":\"chat.completion.chunk\","
"\"choices\":[{\"delta\":{\"content\":\" there\"},\"index\":0}]}\n\n",
"data: {\"id\":\"c1\",\"object\":\"chat.completion.chunk\","
"\"choices\":[{\"delta\":{},\"finish_reason\":\"stop\",\"index\":0}],"
"\"usage\":{\"prompt_tokens\":14,"
"\"prompt_tokens_details\":{\"cached_tokens\":3},"
"\"completion_tokens\":7,"
"\"completion_tokens_details\":{\"reasoning_tokens\":0}}}\n\n",
"data: [DONE]\n\n",
};
static const std::string kRawJson =
"{\"id\":\"c1\",\"object\":\"chat.completion\","
"\"choices\":[{\"message\":{\"role\":\"assistant\",\"content\":\"hi there\"},"
"\"finish_reason\":\"stop\",\"index\":0}],"
"\"usage\":{\"prompt_tokens\":14,"
"\"prompt_tokens_details\":{\"cached_tokens\":3},"
"\"completion_tokens\":7,"
"\"completion_tokens_details\":{\"reasoning_tokens\":0}}}";
struct Result {
std::string forwarded;
td::int64 prompt, cached, completion, reasoning, total;
};
static Result run_scenario(const std::vector<std::string> &slices) {
auto pp = make_pp();
std::string forwarded;
for (const auto &s : slices) {
forwarded += pp.add_next_answer_slice(s);
}
forwarded += pp.finalize();
auto u = pp.usage();
return Result{std::move(forwarded),
u->prompt_tokens_used_, u->cached_tokens_used_,
u->completion_tokens_used_, u->reasoning_tokens_used_,
u->total_tokens_used_};
}
static bool check_result(const char *name, const Result &r, size_t min_bytes,
td::int64 expect_completion) {
std::fprintf(stdout, "\n=== %s: forwarded %zu bytes ===\n%s=== end ===\n",
name, r.forwarded.size(), r.forwarded.c_str());
std::fprintf(stdout, "usage: prompt=%lld cached=%lld completion=%lld "
"reasoning=%lld total=%lld\n",
(long long)r.prompt, (long long)r.cached, (long long)r.completion,
(long long)r.reasoning, (long long)r.total);
if (r.forwarded.size() < min_bytes || r.completion != expect_completion) {
std::fprintf(stderr, "[FAIL] %s: bytes=%zu (min %zu), completion=%lld "
"(want %lld)\n",
name, r.forwarded.size(), min_bytes, (long long)r.completion,
(long long)expect_completion);
return false;
}
std::fprintf(stderr, "[OK] %s\n", name);
return true;
}
int main() {
bool ok = true;
// Scenario 1: whole SSE events.
ok &= check_result("sse_whole_events", run_scenario(kSseChunks), /*min*/ 500,
/*completion*/ 7);
// Scenario 2: byte-by-byte SSE. Catches buffering bugs across slice boundaries.
std::string concat;
for (const auto &c : kSseChunks) concat += c;
std::vector<std::string> byte_slices;
byte_slices.reserve(concat.size());
for (char c : concat) byte_slices.emplace_back(1, c);
ok &= check_result("sse_byte_split", run_scenario(byte_slices), /*min*/ 500,
/*completion*/ 7);
// Scenario 3: raw non-stream JSON as a single slice.
ok &= check_result("raw_json_non_stream", run_scenario({kRawJson}),
/*min*/ 200, /*completion*/ 7);
return ok ? 0 : 1;
}
Sign up for free to join this conversation on GitHub. Already have an account? Sign in to comment