{"record":{"id":"6c4a5f5d00bf9cff","repo":"tensorflow/models","slug":"the-input-size-d-is-not-a-multiple-of-the-numbe-6c4a5f","errorCode":null,"errorMessage":"The input size (%d) is not a multiple of the number of attention heads (%d)","messagePattern":"The input size \\((.+?)\\) is not a multiple of the number of attention heads \\((.+?)\\)","errorType":"exception","errorClass":"ValueError","httpStatus":null,"severity":"error","filePath":"official/projects/qat/nlp/modeling/layers/transformer_encoder_block.py","lineNumber":160,"sourceCode":"    else:\n      self._attention_initializer = self._kernel_initializer\n    self._attention_axes = attention_axes\n\n  def build(self, input_shape):\n    if isinstance(input_shape, tf.TensorShape):\n      input_tensor_shape = input_shape\n    elif isinstance(input_shape, (list, tuple)):\n      input_tensor_shape = tf.TensorShape(input_shape[0])\n    else:\n      raise ValueError(\n          \"The type of input shape argument is not supported, got: %s\" %\n          type(input_shape))\n    if len(input_tensor_shape.as_list()) != 3:\n      raise ValueError(\"TransformerEncoderBlock expects a three-dimensional \"\n                       \"input of shape [batch, sequence, width].\")\n    hidden_size = input_tensor_shape[-1]\n    if hidden_size % self._num_heads != 0:\n      raise ValueError(\n          \"The input size (%d) is not a multiple of the number of attention \"\n          \"heads (%d)\" % (hidden_size, self._num_heads))\n    self._attention_head_size = int(hidden_size // self._num_heads)\n    common_kwargs = dict(\n        bias_initializer=self._bias_initializer,\n        kernel_regularizer=self._kernel_regularizer,\n        bias_regularizer=self._bias_regularizer,\n        activity_regularizer=self._activity_regularizer,\n        kernel_constraint=self._kernel_constraint,\n        bias_constraint=self._bias_constraint)\n    self._attention_layer = _quantized_multi_head_attention(\n        num_heads=self._num_heads,\n        key_dim=self._attention_head_size,\n        dropout=self._attention_dropout,\n        use_bias=self._use_bias,\n        kernel_initializer=self._attention_initializer,\n        attention_axes=self._attention_axes,\n        name=\"self_attention\",","sourceCodeStart":142,"sourceCodeEnd":178,"githubUrl":"https://github.com/tensorflow/models/blob/e006f5f0d534913e49c1f1dae87364039fa607e2/official/projects/qat/nlp/modeling/layers/transformer_encoder_block.py#L142-L178","documentation":"Error \"The input size (%d) is not a multiple of the number of attention heads (%d)\" thrown in tensorflow/models.","triggerScenarios":"Thrown at official/projects/qat/nlp/modeling/layers/transformer_encoder_block.py:160 when the library encounters an invalid state.","commonSituations":"See trigger scenarios.","solutions":[],"exampleFix":null,"handlingStrategy":null,"validationCode":null,"typeGuard":null,"tryCatchPattern":null,"preventionTips":[],"tags":[],"backgroundTag":null,"analyzedSha":"e006f5f0d534913e49c1f1dae87364039fa607e2","analyzedAt":"2026-08-24T14:09:15.576Z","schemaVersion":2},"datasetVersion":"2026-08-24T17:17:21.512Z"}