@@ -131,7 +131,19 @@ struct AttentionActivations {
131131 inv_timescale_global(CreateInvTimescale(
132132 allocator, max_qkv_dim,
133133 layer_config.post_qk == PostQKType::HalfRope,
134- config.global_rope_theta, config.partial_rotary_factor)) {
134+ config.global_rope_theta, config.partial_rotary_factor)),
135+ s_att_q(config.is_encoder_decoder ? config.decoder_num_layers
136+ : config.num_layers,
137+ max_workers),
138+ s_att_k(config.is_encoder_decoder ? config.decoder_num_layers
139+ : config.num_layers,
140+ max_workers),
141+ s_att_v(config.is_encoder_decoder ? config.decoder_num_layers
142+ : config.num_layers,
143+ max_workers),
144+ s_att_out(config.is_encoder_decoder ? config.decoder_num_layers
145+ : config.num_layers,
146+ max_workers) {
135147 // Batch size can be 0 in experimental code so do not assert.
136148 if (batch_size == 0 ) {
137149 static std::atomic_flag warned = ATOMIC_FLAG_INIT ;
@@ -246,6 +258,13 @@ struct AttentionActivations {
246258 // Rope
247259 MatStorageT<float > inv_timescale;
248260 MatStorageT<float > inv_timescale_global;
261+
262+ // Only active when GCPP_TENSOR_STATS.
263+ TensorStats s_att_q;
264+ TensorStats s_att_k;
265+ TensorStats s_att_v;
266+ TensorStats s_att_out;
267+
249268 // Replication factor to help evenly share work over threads.
250269 static constexpr size_t kThreadReplicationFactor = 4 ;
251270};
@@ -301,6 +320,10 @@ struct AttentionActivationsPtrs {
301320 int8_queries = &activations.int8_queries ;
302321 float_queries = &activations.float_queries ;
303322 q_scales = &activations.q_scales ;
323+ s_att_q = &activations.s_att_q ;
324+ s_att_k = &activations.s_att_k ;
325+ s_att_v = &activations.s_att_v ;
326+ s_att_out = &activations.s_att_out ;
304327 }
305328
306329 void SetBatchSize (size_t batch_size) {
@@ -383,6 +406,12 @@ struct AttentionActivationsPtrs {
383406 hwy::Divisor div_heads;
384407 // Query scaling factor for attention computation.
385408 float query_scale;
409+
410+ // Only active when GCPP_TENSOR_STATS.
411+ TensorStats* s_att_q = nullptr ;
412+ TensorStats* s_att_k = nullptr ;
413+ TensorStats* s_att_v = nullptr ;
414+ TensorStats* s_att_out = nullptr ;
386415};
387416
388417static inline size_t MoEBatchSize (const LayerConfig& layer_config,
@@ -646,6 +675,10 @@ struct Activations {
646675 }
647676
648677 ~Activations () {
678+ attention_storage.s_att_q .ReduceAndPrint (" att_q" );
679+ attention_storage.s_att_k .ReduceAndPrint (" att_k" );
680+ attention_storage.s_att_v .ReduceAndPrint (" att_v" );
681+ attention_storage.s_att_out .ReduceAndPrint (" att_out" );
649682 s_ffw_in.ReduceAndPrint (" ffw_in" );
650683 s_ffw_hidden.ReduceAndPrint (" ffw_hidden" );
651684 s_ffw_out.ReduceAndPrint (" ffw_out" );
0 commit comments