Ë
    GêñiÆ9 ã                   ó  — d dl Z d dlZd dlZd dlZd dlZd dlmZ d dlmZ d dl	m
Z
 d dlmZmZmZmZ d dlZd dlmZ d dlmZ ddlmZmZmZmZmZ dd	lmZmZmZmZ dd
l m!Z! ddl"m#Z# ddl$m%Z% ddl&m'Z' ddl(m)Z)m*Z*m+Z+m,Z, ddl-m.Z. ddl/m0Z0m1Z1m2Z2m3Z3m4Z4m5Z5m6Z6m7Z7m8Z8m9Z9 ddl:m;Z;m<Z<m=Z=m>Z>m?Z? ddl@mAZA ddlBmCZCmDZDmEZEmFZFmGZGmHZHmIZImJZJmKZKmLZLmMZMmNZNmOZOmPZPmQZQmRZRmSZSmTZTmUZUmVZVmWZWmXZXmYZYmZZZm[Z[m\Z\ ddl]m^Z^m_Z_m`Z`maZambZbmcZcmdZd erddlemfZf ddlgmhZh ddlimjZj ddlkmlZl  e,jÚ                  en«      Zo e+«       rd dlpmqZqmrZr g d¢Zse?jè                  de?jê                  de?jì                  de?jî                  de?jð                  de?jò                  de?jô                  d e?jö                  d!e?jø                  d"i	Z}e
 G d#„ d$e)«      «       Z~e
 G d%„ d&e)«      «       Ze
 G d'„ d(e)«      «       Z€e
 G d)„ d*e)«      «       Z�e~ez  Z‚e€e�z  Zƒe‚eƒz  Z„ G d+„ d,eA«      Z…d-„ Z†d/d.„Z‡y)0é    N)ÚCallable)Úcontextmanager)Ú	dataclass)ÚTYPE_CHECKINGÚAnyÚOptionalÚcast)Únné   )ÚCacheÚDynamicCacheÚEncoderDecoderCacheÚQuantizedCacheÚStaticCache)Úcheck_python_requirementsÚget_cached_module_fileÚget_class_in_moduleÚresolve_trust_remote_code)Úis_deepspeed_zero3_enabled)Úis_fsdp_managed_module)Úcreate_masks_for_generate)ÚExtensionsTrie)ÚModelOutputÚTransformersKwargsÚis_accelerate_availableÚlogging)Úis_flash_attention_requestedé   )
ÚAssistantVocabTranslatorCacheÚAssistedCandidateGeneratorÚ-AssistedCandidateGeneratorDifferentTokenizersÚCandidateGeneratorÚEarlyExitCandidateGeneratorÚPromptLookupCandidateGeneratorÚ%UniversalSpeculativeDecodingGeneratorÚ_prepare_attention_maskÚ_prepare_position_idsÚ_prepare_token_type_ids)Ú ALL_STATIC_CACHE_IMPLEMENTATIONSÚ'DEPRECATED_STATIC_CACHE_IMPLEMENTATIONSÚSTATIC_CACHE_IMPLEMENTATIONSÚGenerationConfigÚGenerationMode)ÚContinuousMixin)Ú#EncoderNoRepeatNGramLogitsProcessorÚ'EncoderRepetitionPenaltyLogitsProcessorÚEpsilonLogitsWarperÚEtaLogitsWarperÚExponentialDecayLengthPenaltyÚForcedBOSTokenLogitsProcessorÚForcedEOSTokenLogitsProcessorÚInfNanRemoveLogitsProcessorÚLogitNormalizationÚLogitsProcessorListÚMinLengthLogitsProcessorÚ!MinNewTokensLengthLogitsProcessorÚMinPLogitsWarperÚNoBadWordsLogitsProcessorÚNoRepeatNGramLogitsProcessorÚ PrefixConstrainedLogitsProcessorÚ RepetitionPenaltyLogitsProcessorÚSequenceBiasLogitsProcessorÚ$SuppressTokensAtBeginLogitsProcessorÚSuppressTokensLogitsProcessorÚTemperatureLogitsWarperÚTopHLogitsWarperÚTopKLogitsWarperÚTopPLogitsWarperÚTypicalLogitsWarperÚ.UnbatchedClassifierFreeGuidanceLogitsProcessor)ÚConfidenceCriteriaÚEosTokenCriteriaÚMaxLengthCriteriaÚMaxTimeCriteriaÚStoppingCriteriaÚStoppingCriteriaListÚStopStringCriteria)ÚGenerativePreTrainedModel)ÚPreTrainedModel)ÚPreTrainedTokenizerBase)ÚBaseStreamer)ÚAlignDevicesHookÚadd_hook_to_module)Úpast_key_valuesÚcache_paramsÚstateÚmemsÚpast_buckets_statesÚ_sampleÚ_beam_searchÚ_assisted_decodingztransformers-community/dolaz)transformers-community/contrastive-searchz(transformers-community/group-beam-searchz.transformers-community/constrained-beam-searchc                   ó  — e Zd ZU dZej
                  ed<   dZeej                     dz  ed<   dZ
eej                     dz  ed<   dZeeej                        dz  ed<   dZeeej                        dz  ed<   dZedz  ed<   y)	ÚGenerateDecoderOnlyOutputa\  
    Outputs of decoder-only generation models, when using non-beam methods.

    Args:
        sequences (`torch.LongTensor` of shape `(batch_size, sequence_length)`):
            The generated sequences. The second dimension (sequence_length) is either equal to `max_length` or shorter
            if all batches finished early due to the `eos_token_id`.
        scores (`tuple(torch.FloatTensor)` *optional*, returned when `output_scores=True`):
            Processed prediction scores of the language modeling head (scores for each vocabulary token before SoftMax)
            at each generation step. Tuple of `torch.FloatTensor` with up to `max_new_tokens` elements (one element for
            each generated token), with each tensor of shape `(batch_size, config.vocab_size)`.
        logits (`tuple(torch.FloatTensor)` *optional*, returned when `output_logits=True`):
            Unprocessed prediction scores of the language modeling head (scores for each vocabulary token before SoftMax)
            at each generation step. Tuple of `torch.FloatTensor` with up to `max_new_tokens` elements (one element for
            each generated token), with each tensor of shape `(batch_size, config.vocab_size)`.
        attentions (`tuple(tuple(torch.FloatTensor))`, *optional*, returned when `output_attentions=True`):
            Tuple (one element for each generated token) of tuples (one element for each layer of the decoder) of
            `torch.FloatTensor` of shape `(batch_size, num_heads, generated_length, sequence_length)`.
        hidden_states (`tuple(tuple(torch.FloatTensor))`, *optional*, returned when `output_hidden_states=True`):
            Tuple (one element for each generated token) of tuples (one element for each layer of the decoder) of
            `torch.FloatTensor` of shape `(batch_size, generated_length, hidden_size)`.
        past_key_values (`Cache`, *optional*, returned when `use_cache=True`):
            Returns the model cache, used to speed up decoding. Different models have a different cache format, check
            the model's documentation. Usually, a [`~cache_utils.Cache`] instance.
    Ú	sequencesNÚscoresÚlogitsÚ
attentionsÚhidden_statesrV   )Ú__name__Ú
__module__Ú__qualname__Ú__doc__ÚtorchÚ
LongTensorÚ__annotations__ra   ÚtupleÚFloatTensorrb   rc   rd   rV   r   © ó    ú_/var/www/pod-logistic/pod-ai/venv/lib/python3.12/site-packages/transformers/generation/utils.pyr_   r_   “   s•   … ñð4 ×ÑÓØ.2€FˆE�%×#Ñ#Ñ$ tÑ+Ó2Ø.2€FˆE�%×#Ñ#Ñ$ tÑ+Ó2Ø9=€J��e˜E×-Ñ-Ñ.Ñ/°$Ñ6Ó=Ø<@€M�5˜˜u×0Ñ0Ñ1Ñ2°TÑ9Ó@Ø$(€O�U˜T‘\Ô(ro   r_   c                   ó˜  — e Zd ZU dZej
                  ed<   dZeej                     dz  ed<   dZ
eej                     dz  ed<   dZeej                     dz  ed<   dZeej                     dz  ed<   dZeeej                        dz  ed<   dZeeej                        dz  ed	<   dZeeej                        dz  ed
<   dZedz  ed<   y)ÚGenerateEncoderDecoderOutputa  
    Outputs of encoder-decoder generation models, when using non-beam methods.

    Args:
        sequences (`torch.LongTensor` of shape `(batch_size*num_return_sequences, sequence_length)`):
            The generated sequences. The second dimension (sequence_length) is either equal to `max_length` or shorter
            if all batches finished early due to the `eos_token_id`.
        scores (`tuple(torch.FloatTensor)` *optional*, returned when `output_scores=True`):
            Processed prediction scores of the language modeling head (scores for each vocabulary token before SoftMax)
            at each generation step. Tuple of `torch.FloatTensor` with up to `max_new_tokens` elements (one element for
            each generated token), with each tensor of shape `(batch_size, config.vocab_size)`.
        logits (`tuple(torch.FloatTensor)` *optional*, returned when `output_logits=True`):
            Unprocessed prediction scores of the language modeling head (scores for each vocabulary token before SoftMax)
            at each generation step. Tuple of `torch.FloatTensor` with up to `max_new_tokens` elements (one element for
            each generated token), with each tensor of shape `(batch_size, config.vocab_size)`.
        encoder_attentions (`tuple(torch.FloatTensor)`, *optional*, returned when `output_attentions=True`):
            Tuple of `torch.FloatTensor` (one for each layer of the decoder) of shape `(batch_size, num_heads,
            sequence_length, sequence_length)`.
        encoder_hidden_states (`tuple(torch.FloatTensor)`, *optional*, returned when `output_hidden_states=True`):
            Tuple of `torch.FloatTensor` (one for the output of the embeddings + one for the output of each layer) of
            shape `(batch_size, sequence_length, hidden_size)`.
        decoder_attentions (`tuple(tuple(torch.FloatTensor))`, *optional*, returned when `output_attentions=True`):
            Tuple (one element for each generated token) of tuples (one element for each layer of the decoder) of
            `torch.FloatTensor` of shape `(batch_size, num_heads, generated_length, sequence_length)`.
        cross_attentions (`tuple(tuple(torch.FloatTensor))`, *optional*, returned when `output_attentions=True`):
            Tuple (one element for each generated token) of tuples (one element for each layer of the decoder) of
            `torch.FloatTensor` of shape `(batch_size, num_heads, generated_length, sequence_length)`.
        decoder_hidden_states (`tuple(tuple(torch.FloatTensor))`, *optional*, returned when `output_hidden_states=True`):
            Tuple (one element for each generated token) of tuples (one element for each layer of the decoder) of
            `torch.FloatTensor` of shape `(batch_size, generated_length, hidden_size)`.
        past_key_values (`Cache`, *optional*, returned when `use_cache=True`):
            Returns the model cache, used to speed up decoding. Different models have a different cache format, check
            the model's documentation. Usually, a [`~cache_utils.Cache`] instance.
    r`   Nra   rb   Úencoder_attentionsÚencoder_hidden_statesÚdecoder_attentionsÚcross_attentionsÚdecoder_hidden_statesrV   )re   rf   rg   rh   ri   rj   rk   ra   rl   rm   rb   rs   rt   ru   rv   rw   rV   r   rn   ro   rp   rr   rr   ·   sî   … ñ!ðF ×ÑÓØ.2€FˆE�%×#Ñ#Ñ$ tÑ+Ó2Ø.2€FˆE�%×#Ñ#Ñ$ tÑ+Ó2Ø:>Ð˜˜e×/Ñ/Ñ0°4Ñ7Ó>Ø=AÐ˜5 ×!2Ñ!2Ñ3°dÑ:ÓAØAEÐ˜˜e E×$5Ñ$5Ñ6Ñ7¸$Ñ>ÓEØ?CÐ�e˜E %×"3Ñ"3Ñ4Ñ5¸Ñ<ÓCØDHÐ˜5  u×'8Ñ'8Ñ!9Ñ:¸TÑAÓHØ$(€O�U˜T‘\Ô(ro   rr   c                   óX  — e Zd ZU dZej
                  ed<   dZej                  dz  ed<   dZ	e
ej                     dz  ed<   dZe
ej                     dz  ed<   dZej
                  dz  ed<   dZe
e
ej                        dz  ed<   dZe
e
ej                        dz  ed	<   dZedz  ed
<   y)ÚGenerateBeamDecoderOnlyOutputaÎ
  
    Outputs of decoder-only generation models, when using beam methods.

    Args:
        sequences (`torch.LongTensor` of shape `(batch_size*num_return_sequences, sequence_length)`):
            The generated sequences. The second dimension (sequence_length) is either equal to `max_length` or shorter
            if all batches finished early due to the `eos_token_id`.
        sequences_scores (`torch.FloatTensor` of shape `(batch_size*num_return_sequences)`, *optional*, returned when `output_scores=True`):
            Final beam scores of the generated `sequences`.
        scores (`tuple(torch.FloatTensor)` *optional*, returned when `output_scores=True`):
            Beam transition scores for each vocabulary token at each generation step. Beam transition scores consisting
            of log probabilities of tokens conditioned on log softmax of previously generated tokens in this beam.
            Tuple of `torch.FloatTensor` with up to `max_new_tokens` elements (one element for each generated token),
            with each tensor of shape `(batch_size*num_beams, config.vocab_size)`.
        logits (`tuple(torch.FloatTensor)` *optional*, returned when `output_logits=True`):
            Unprocessed prediction scores of the language modeling head (scores for each vocabulary token before SoftMax)
            at each generation step. Tuple of `torch.FloatTensor` with up to `max_new_tokens` elements (one element for
            each generated token), with each tensor of shape `(batch_size*num_beams, config.vocab_size)`.
        beam_indices (`torch.LongTensor`, *optional*, returned when `output_scores=True`):
            Beam indices of generated token id at each generation step. `torch.LongTensor` of shape
            `(batch_size*num_return_sequences, sequence_length)`.
        attentions (`tuple(tuple(torch.FloatTensor))`, *optional*, returned when `output_attentions=True`):
            Tuple (one element for each generated token) of tuples (one element for each layer of the decoder) of
            `torch.FloatTensor` of shape `(batch_size*num_beams, num_heads, generated_length, sequence_length)`.
        hidden_states (`tuple(tuple(torch.FloatTensor))`, *optional*, returned when `output_hidden_states=True`):
            Tuple (one element for each generated token) of tuples (one element for each layer of the decoder) of
            `torch.FloatTensor` of shape `(batch_size*num_beams*num_return_sequences, generated_length, hidden_size)`.
        past_key_values (`Cache`, *optional*, returned when `use_cache=True`):
            Returns the model cache, used to speed up decoding. Different models have a different cache format, check
            the model's documentation. Usually, a [`~cache_utils.Cache`] instance.
    r`   NÚsequences_scoresra   rb   Úbeam_indicesrc   rd   rV   )re   rf   rg   rh   ri   rj   rk   rz   rm   ra   rl   rb   r{   rc   rd   rV   r   rn   ro   rp   ry   ry   ç   sÁ   … ñð@ ×ÑÓØ15Ð�e×'Ñ'¨$Ñ.Ó5Ø.2€FˆE�%×#Ñ#Ñ$ tÑ+Ó2Ø.2€FˆE�%×#Ñ#Ñ$ tÑ+Ó2Ø,0€L�%×"Ñ" TÑ)Ó0Ø9=€J��e˜E×-Ñ-Ñ.Ñ/°$Ñ6Ó=Ø<@€M�5˜˜u×0Ñ0Ñ1Ñ2°TÑ9Ó@Ø$(€O�U˜T‘\Ô(ro   ry   c                   óè  — e Zd ZU dZej
                  ed<   dZej                  dz  ed<   dZ	e
ej                     dz  ed<   dZe
ej                     dz  ed<   dZej
                  dz  ed<   dZe
ej                     dz  ed<   dZe
ej                     dz  ed	<   dZe
e
ej                        dz  ed
<   dZe
e
ej                        dz  ed<   dZe
e
ej                        dz  ed<   dZedz  ed<   y)Ú GenerateBeamEncoderDecoderOutputa¡  
    Outputs of encoder-decoder generation models, when using beam methods.

    Args:
        sequences (`torch.LongTensor` of shape `(batch_size*num_return_sequences, sequence_length)`):
            The generated sequences. The second dimension (sequence_length) is either equal to `max_length` or shorter
            if all batches finished early due to the `eos_token_id`.
        sequences_scores (`torch.FloatTensor` of shape `(batch_size*num_return_sequences)`, *optional*, returned when `output_scores=True`):
            Final beam scores of the generated `sequences`.
        scores (`tuple(torch.FloatTensor)` *optional*, returned when `output_scores=True`):
            Beam transition scores for each vocabulary token at each generation step. Beam transition scores consisting
            of log probabilities of tokens conditioned on log softmax of previously generated tokens in this beam.
            Tuple of `torch.FloatTensor` with up to `max_new_tokens` elements (one element for each generated token),
            with each tensor of shape `(batch_size*num_beams, config.vocab_size)`.
        logits (`tuple(torch.FloatTensor)` *optional*, returned when `output_logits=True`):
            Unprocessed prediction scores of the language modeling head (scores for each vocabulary token before SoftMax)
            at each generation step. Tuple of `torch.FloatTensor` with up to `max_new_tokens` elements (one element for
            each generated token), with each tensor of shape `(batch_size*num_beams, config.vocab_size)`.
        beam_indices (`torch.LongTensor`, *optional*, returned when `output_scores=True`):
            Beam indices of generated token id at each generation step. `torch.LongTensor` of shape
            `(batch_size*num_return_sequences, sequence_length)`.
        encoder_attentions (`tuple(torch.FloatTensor)`, *optional*, returned when `output_attentions=True`):
            Tuple of `torch.FloatTensor` (one for each layer of the decoder) of shape `(batch_size, num_heads,
            sequence_length, sequence_length)`.
        encoder_hidden_states (`tuple(torch.FloatTensor)`, *optional*, returned when `output_hidden_states=True`):
            Tuple of `torch.FloatTensor` (one for the output of the embeddings + one for the output of each layer) of
            shape `(batch_size*num_beams*num_return_sequences, sequence_length, hidden_size)`.
        decoder_attentions (`tuple(tuple(torch.FloatTensor))`, *optional*, returned when `output_attentions=True`):
            Tuple (one element for each generated token) of tuples (one element for each layer of the decoder) of
            `torch.FloatTensor` of shape `(batch_size*num_beams*num_return_sequences, num_heads, generated_length,
            sequence_length)`.
        cross_attentions (`tuple(tuple(torch.FloatTensor))`, *optional*, returned when `output_attentions=True`):
            Tuple (one element for each generated token) of tuples (one element for each layer of the decoder) of
            `torch.FloatTensor` of shape `(batch_size, num_heads, generated_length, sequence_length)`.
        decoder_hidden_states (`tuple(tuple(torch.FloatTensor))`, *optional*, returned when `output_hidden_states=True`):
            Tuple (one element for each generated token) of tuples (one element for each layer of the decoder) of
            `torch.FloatTensor` of shape `(batch_size*num_beams*num_return_sequences, generated_length, hidden_size)`.
        past_key_values (`Cache`, *optional*, returned when `use_cache=True`):
            Returns the model cache, used to speed up decoding. Different models have a different cache format, check
            the model's documentation. Usually, a [`~cache_utils.Cache`] instance.
    r`   Nrz   ra   rb   r{   rs   rt   ru   rv   rw   rV   )re   rf   rg   rh   ri   rj   rk   rz   rm   ra   rl   rb   r{   rs   rt   ru   rv   rw   rV   r   rn   ro   rp   r}   r}     s  … ñ(ðT ×ÑÓØ15Ð�e×'Ñ'¨$Ñ.Ó5Ø.2€FˆE�%×#Ñ#Ñ$ tÑ+Ó2Ø.2€FˆE�%×#Ñ#Ñ$ tÑ+Ó2Ø,0€L�%×"Ñ" TÑ)Ó0Ø:>Ð˜˜e×/Ñ/Ñ0°4Ñ7Ó>Ø=AÐ˜5 ×!2Ñ!2Ñ3°dÑ:ÓAØAEÐ˜˜e E×$5Ñ$5Ñ6Ñ7¸$Ñ>ÓEØ?CÐ�e˜E %×"3Ñ"3Ñ4Ñ5¸Ñ<ÓCØDHÐ˜5  u×'8Ñ'8Ñ!9Ñ:¸TÑAÓHØ$(€O�U˜T‘\Ô(ro   r}   c            $       óÀ  — e Zd ZdZdZ	 	 d~d„Z	 	 ddeej                  z  dz  de	dz  d	e
fd
„Z	 	 	 	 	 d€dddej                  dedz  dedz  dej                  dz  dej                   dz  de	dz  fd„Zdddej$                  dz  dej$                  dz  deeej$                  f   d	eej$                  edz  eeej$                  f   f   f
d„Zdddej$                  dz  dej$                  dz  deeej$                  f   d	ej                  f
d„Zd„ Zdej$                  dedeeef   d	ej                  fd„Zdddej$                  dedz  ded	eeef   f
d„Z	 d�dddededeeej$                  f   dej$                  dej8                  dz  d	eej                  eeej$                  f   f   fd „Ze	 	 	 d‚d!ed"e	dej                  dz  d	eej                  eeef   f   fd#„«       Z	 	 dƒd$e deeef   d"e	d%ed	eeef   f
d&„Z!	 	 	 d„dddedej                  dej$                  d'e"deeef   d(e#d)   d*e#d+   d,e#d+   d	e$fd-„Z%	 	 	 	 	 	 	 	 d…ddded.edz  d/ej                  dz  d0e
eej$                  ge&e   f   dz  d'e"dz  dedz  deeef   dz  d1ej$                  dz  d2ej$                  dz  d	e"fd3„Z'	 d�ddded4e(dz  d5e#d+   d	e(f
d6„Z)d7e"e(z  d8e"e(z  d	e"e(z  fd9„Z*	 	 d†ddd:ej$                  d;eej$                     d<ej$                  dz  d=e	d	ej$                  fd>„Z+	 	 d~d?„Z,dddeeef   fd@„Z-	 	 d~dA„Z.	 	 d~dB„Z/dddedz  dCed	eeef   fdD„Z0dddEededFed	ef
dG„Z1e2dHe3d   d	e	fdI„«       Z4dddededJe5dedKed	e	fdL„Z6ddd	e	fdM„Z7	 	 ddddedNe	dz  dej8                  ez  dz  fdO„Z8dddeeef   ded	e	fdP„Z9e:d~dQ„«       Z;	 d�dJe5de	dRedz  d	edz  fdS„Z<d	eeef   fdT„Z= ej|                  «       	 	 	 	 	 	 	 	 	 	 	 d‡dddej$                  dz  dedz  d'e"dz  d4e(dz  d0e
eej$                  ge&e   f   dz  dUe	dz  d(e#d)   dVe#dW   d1ej$                  dz  d2ej$                  dz  dRee
z  dz  d	e?ej                  z  fdX„«       Z@dYe	dUe	dej8                  d	e	fdZ„ZA	 d�dej                  d5e#d+   d	ej                  fd[„ZB	 	 dˆdddej                  d'e"d4e(dedUe	dVe#dW   d	eCej                  z  fd\„ZDed]ej$                  d	ej$                  fd^„«       ZEed]ej$                  ded_ed	ej$                  fd`„«       ZFed]ej$                  d<ej$                  d	ej$                  fda„«       ZGedbej$                  dcej$                  ddej$                  deej$                  dfedgedhedie	ez  djeHfdk„«       ZIedbej$                  deej$                  dlej$                  die	ez  fdm„«       ZJdnej$                  doej$                  dpej$                  dfedhedqe	dred_edseded	eej$                  ej$                  ej$                  f   fdt„ZKduej$                  dvej$                  dwej$                  dlej$                  d_ed	eej$                  ej$                  ej$                  f   fdx„ZLd:ej$                  dvej$                  ddej$                  duej$                  d<ej$                  dwej$                  dbej$                  deej$                  dlej$                  dyej$                  d_edfedhedjeHdie	ez  d	eej$                  ej$                  ej$                  ej$                  f   f dz„ZM	 d‰dddej                  d'e"d4e(dedUe	d	eNej                  z  fd{„ZO	 	 	 	 	 	 dŠdddej                  d'e"d4e(dedUe	dVe#dW   dej                   dz  d(e#d)   d,e#d+   d5e#d+   d	eCej                  z  fd|„ZP	 d‹dddej                  dedede	f
d}„ZQy)ŒÚGenerationMixinaž  
    A class containing all functions for auto-regressive text generation, to be used as a mixin in model classes.
    Inheriting from this class causes the model to have special generation-related behavior, such as loading a
    `GenerationConfig` at initialization time or ensuring `generate`-related tests are run in `transformers` CI.

    A model class should inherit from `GenerationMixin` to enable calling methods like `generate`, or when it
    has defined a custom `generate` method that relies on `GenerationMixin`, directly or indirectly, which
    approximately shares the same interface to public methods like `generate`. Three examples:
        - `LlamaForCausalLM` should inherit from `GenerationMixin` to enable calling `generate` and other public
            methods in the mixin;
        - `BlipForQuestionAnswering` has a custom `generate` method that approximately shares the same interface as
           `GenerationMixin.generate` (it has a few extra arguments, and the same output). That function also calls
           `GenerationMixin.generate` indirectly, through an inner model. As such, `BlipForQuestionAnswering` should
           inherit from `GenerationMixin` to benefit from all generation-related automation in our codebase;
        - `BarkModel` has a custom `generate` method and one of its inner models calls `GenerationMixin.generate`.
            However, its `generate` does not share the same interface as `GenerationMixin.generate`. In this case,
            `BarkModel` should NOT inherit from `GenerationMixin`, as it breaks the `generate` interface.

    The class exposes [`~generation.GenerationMixin.generate`], which can be used for:
        - *greedy decoding* if `num_beams=1` and `do_sample=False`
        - *multinomial sampling* if `num_beams=1` and `do_sample=True`
        - *beam-search decoding* if `num_beams>1` and `do_sample=False`
        - *beam-search multinomial sampling* if `num_beams>1` and `do_sample=True`
        - *assisted decoding* if `assistant_model` or `prompt_lookup_num_tokens` is passed to `.generate()`

    To learn more about decoding strategies refer to the [text generation strategies guide](../generation_strategies).
    )ÚtextÚselfrP   c           	      ó.  — | j                  «       r1|�/| j                  j                  |j                  «       «      | _        y | j                  «       rq|�n|||||	|
|dœ|¥}	 t	        j
                  |f||dœ|¤Ž| _        t        | d«      r6|r3	  | j                  |fd|i|¤Ž}t        j                  || ¬	«      | _        y y y y y # t        $ r8 t        j                  d«       t	        j
                  |fd||ddœ|¤Ž| _        Y Œ…w xY w# t        $ r Y y w xY w)
N)Ú	cache_dirÚforce_downloadÚproxiesÚlocal_files_onlyÚtokenÚrevisionÚ	subfolder)Ú
_from_autoÚ_from_pipelinezZGeneration config file not found, using a generation config created from the model config.zconfig.jsonT)Úconfig_file_namerŠ   r‹   Ú_from_model_configÚload_custom_generateÚtrust_remote_code)Úmodel)Úcan_generateÚgeneration_configÚ	from_dictÚto_dictr,   Úfrom_pretrainedÚOSErrorÚloggerÚinfoÚhasattrrŽ   Ú	functoolsÚpartialÚgenerate)r�   r’   Úfrom_auto_classÚfrom_pipelineÚpretrained_model_name_or_pathrƒ   r„   r…   r†   r‡   rˆ   r‰   r�   ÚkwargsÚrepo_loading_kwargsÚcustom_generates                   rp   Úadjust_generation_fnz$GenerationMixin.adjust_generation_fnr  sm  € ð  ×ÑÔÐ#4Ð#@Ø%)×%;Ñ%;×%EÑ%EÐFW×F_ÑF_ÓFaÓ%bˆDÕ"Ø×ÑÔ Ð%BÐ%Nà&Ø"0Ø"Ø$4ØØ$Ø&ñ	#ð ð	#ÐðÜ)9×)IÑ)IØ1ð*à.Ø#0ñ*ð *ñ	*�Ô&ô. �tÐ3Ô4Ñ9JðØ&? d×&?Ñ&?Ø5ñ'ØIZð'Ø^qñ'�Oô %.×$5Ñ$5°oÈTÔ$R�D•Mð :KÐ4ðI &OÐ øô& ò ô —‘Øpôô *:×)IÑ)IØ1ð*à%2Ø.Ø#0Ø'+ñ*ð *ñ*�Ö&ðûô. ò Ùðús$   Á!C Â1D Ã>DÄDÄ	DÄDNrŸ   r�   Úreturnc                 óü   — 	 t        |fddi|¤Ž}t        j                  j	                  |«      }d|› d�}t        |||| |¬«       t        |fdd	i|¤Ž t        d
|«      }|S # t        $ r t        d|› d�«      ‚w xY w)at  
        Loads and returns a custom generate function, given a model repo.

        Args:
            pretrained_model_name_or_path (`str` or `os.PathLike`):
                 Can be either:
                    - A string, the *model id* of a pretrained model hosted inside a model repo on huggingface.co.
                    - A path to a *directory* containing model weights saved using
                      [`~PreTrainedModel.save_pretrained`], e.g., `./my_model_directory/`.
            trust_remote_code (`bool`, *optional*):
                Whether or not to allow for custom models defined on the Hub in their own modeling files. This option
                should only be set to `True` for repositories you trust and in which you have read the code, as it will
                execute code present on the Hub on your local machine.
            **kwargs:
                Additional keyword arguments for remote code loading.

        Raises:
            OSError: If `pretrained_model_name_or_path` does not contain a `custom_generate` subdirectory.

        Returns:
            A callable that can be used to generate text.
        Úmodule_filezcustom_generate/generate.pyú`zw` does not contain a `custom_generate` subdirectory with a `generate.py` file, can't load the custom generate function.zThe repository `zS` contains custom generation code that will override the default `generate` method.)Úhas_local_codeÚhas_remote_codeÚerror_messageÚrequirements_filez custom_generate/requirements.txtrœ   )r   r–   ÚosÚpathÚexistsr   r   r   )r�   rŸ   r�   r    ÚmoduleÚis_local_coderª   Úcustom_generate_functions           rp   rŽ   z$GenerationMixin.load_custom_generate±  sÒ   € ð<	Ü+Ø-ñØ;XðØ\bñˆFô Ÿ™Ÿ™Ð'DÓEˆàÐ<Ð=ð >-ð -ð 	ô 	"ØØ)Ø(Ø -Ð-Ø'õ	
ô 	"Ø)ñ	
Ø=_ð	
Øciò	
ô $7°zÀ6Ó#JÐ Ø'Ð'øô3 ò 	ÜØÐ1Ð2ð 3Oð Oóð ð	ús   ‚A" Á"A;Ú	input_idsÚnext_sequence_lengthrV   Úattention_maskÚinputs_embedsÚis_first_iterationc                 óÒ  — i }| j                   j                  rdnd}	| j                   j                  sR|�P|rNd||	<   |�|dd…| d…dd…f   n|}
|
j                  t        j                  ¬«      |d<   |
j
                  dd \  }}nE|�|dd…| d…f   n|}|j                  t        j                  ¬«      ||	<   |j
                  dd \  }}|�||d<   | j                   j                  rdnd	}|j                  |d«      x}�|||<   |j                  d
d«      x}�||d
<   |d
dfD ]V  }|j                  |«      }|€Œ|j
                  d   |k7  sŒ*|d| d…f   j                  t        j                  ¬«      }|||<   ŒX | j                   j                  r|nd}| j                   j                  rdnd}| j                   j                  r|j                  dd«      n|}t        |t        «      r±|j                  r¥|�£|j                  dk(  r”t        | dt        «      } || j                   t        j                  ||df| j                  |j                   ¬«      ||j                  d«      |j                  |«      |j                  d
«      |j                  d«      |¬«      }|�|||<   |�||d<   d}|j#                  «       D ]  \  }}||vsŒ||vsŒ|||<   Œ | j%                  «       r†dt'        t)        j*                  | j,                  «      j.                  «      v rRt0        j3                  d«       |�|j5                  «       nd}t        j6                  ||j                   ¬«      |z   }||d<   |S )a_  
        Prepare the model inputs for generation. Notable steps include selecting the correct input key and cloning when appropriate,
        creating position_ids from the attention_mask when missing, slicing inputs and converting 2D attention masks to 4D for
        compilable caches, and finally forwarding all additional keyword arguments unchanged to the model's forward pass.

        See the forward pass in the model documentation for expected arguments (different models might have different
        requirements for e.g. `past_key_values`). This function should work as is for most LLMs.
        Údecoder_input_idsr²   N)Úmemory_formatrµ   r   rV   Údecoder_position_idsÚposition_idsÚtoken_type_idsÚmm_token_type_idséÿÿÿÿ.Údecoder_attention_maskr´   r   r   ©ÚdtypeÚdevice)Úconfigrµ   r´   rV   r»   r¼   r½   r¶   )Úlabelsr³   Úcache_positiona9  The remote code model you are currently using seems to expect `cache_position`. This arg has been removed from the Transformers library, and will stop being created in `generate` even for remote code models in a future release. Please open a PR on the remote code hub repo to remove any usage of `cache_position`.©rÂ   )rÃ   Úis_encoder_decoderÚcloneri   Úcontiguous_formatÚshapeÚpopÚgetÚ
isinstancer   Úis_compileableÚndimÚgetattrr   ÚemptyrÁ   rÂ   ÚitemsÚis_remote_codeÚsetÚinspectÚ	signatureÚforwardÚ
parametersr—   Úwarning_onceÚget_seq_lengthÚarange)r�   r²   r³   rV   r´   rµ   r¶   r    Úmodel_inputsÚinput_ids_keyÚprompt_embedsÚ
batch_sizeÚsequence_lengthÚposition_ids_keyr»   r¼   Úmodel_input_nameÚmodel_inputÚencoder_attention_maskÚattention_mask_keyÚcausal_mask_creation_functionÚkwargs_to_avoid_forwardingÚkeyÚvalueÚpast_seen_tokensrÅ   s                             rp   Úprepare_inputs_for_generationz-GenerationMixin.prepare_inputs_for_generationî  sÁ  € ð& ˆð 04¯{©{×/MÒ/MÑ+ÐS^ˆà�{‰{×-Ò-°-Ð2KÑPbØ*.ˆL˜Ñ'à?SÐ?_�šaÐ"6Ð!6Ñ!7ºÐ:Ò;Ðerð ð -:×,?Ñ,?Ìe×NeÑNeÐ,?Ó,fˆL˜Ñ)Ø*7×*=Ñ*=¸b¸qÐ*AÑ'ˆJ™ð AUÐ@`˜	¢!Ð&:Ð%:Ñ%;Ð";Ò<ÐfoˆIØ*3¯/©/Ì×H_ÑH_¨/Ó*`ˆL˜Ñ'Ø*3¯/©/¸"¸1Ð*=Ñ'ˆJ˜ð Ð&Ø.=ˆLÐ*Ñ+Ø59·[±[×5SÒ5SÑ1ÐYgÐØ"ŸJ™JÐ'7¸Ó>Ð>ˆLÐKØ-9ˆLÐ)Ñ*Ø$Ÿj™jÐ)9¸4Ó@Ð@ˆNÐMØ-;ˆLÐ)Ñ*ð "2Ð3CÐEXÐ Yò 	=ÐØ&×*Ñ*Ð+;Ó<ˆKØÑ&¨;×+<Ñ+<¸RÑ+@ÀOÓ+Sà)¨#°Ð/?Ñ/@Ð*@ÑA×GÑGÔV[×VmÑVmÐGÓn�Ø1<�Ð-Ò.ð	=ð 48·;±;×3QÒ3Q¡ÐW[ÐØ9=¿¹×9WÒ9WÑ5Ð]mÐà:>¿+¹+×:XÒ:XˆF�J‰JÐ/°Ô6Ð^lð 	ô �¬Ô.Ø×.Ò.ØÐ*Ø×#Ñ# qÒ(ô -4°DÐ:UÔWpÓ,qÐ)Ù:Ø—{‘{ä#Ÿk™k¨:°ÈÐ*JÐRV×R\ÑR\Ðen×euÑeuÔvØ-Ø ,× 0Ñ 0Ð1BÓ CØ)×-Ñ-Ð.>Ó?à+×/Ñ/Ð0@ÓAØ".×"2Ñ"2Ð3FÓ"GØ#5ôˆNð Ð%Ø/=ˆLÐ+Ñ,à!Ð-Ø-CˆLÐ)Ñ*ð &HÐ"Ø Ÿ,™,›.ò 	*‰JˆC�Ø˜,Ò&¨3Ð6PÒ+PØ$)�˜SÒ!ð	*ð ×ÑÔ Ð%5¼¼W×=NÑ=NÈtÏ|É|Ó=\×=gÑ=gÓ9hÑ%hÜ×Ñð}ôð
 DSÐC^˜×=Ñ=Ô?ÐdeÐÜ"Ÿ\™\¨/À)×BRÑBRÔSÐVfÑfˆNØ-;ˆLÐ)Ñ*àÐro   ÚinputsÚbos_token_idÚmodel_kwargsc                 ó  — | j                   j                  rFt        | d«      r:| j                  j                  | j                  k7  r| j                  j                  }n| j                  }|j                  |d«      }|�|�t        d|› d|› d|› d|› d�	«      ‚|�|}|dk(  rËd	|v rÇ|d	   €|j                  d	«       n°| j                   j                  s†d	t        t        j                  | j                  «      j                  j                  «       «      v }|s#t        d
| j                  j                  › d�«      ‚| j                  |||¬«      |d<   |d	   d	}}n|�t        d«      ‚|d	   d	}}| j                  |||«      }|||fS )zT
        This function extracts the model-specific `inputs` for generation.
        ÚencoderNz
`inputs`: z` were passed alongside z0 which is not allowed. Make sure to either pass z or z=...r²   rµ   zAYou passed `inputs_embeds` to `.generate()`, but the model class z² doesn't have its forwarding implemented. See the GPT2 implementation for an example (https://github.com/huggingface/transformers/pull/21405), and feel free to open a PR with it!)rî   zMYou passed `inputs_embeds` and `input_ids` to `.generate()`. Please pick one.)rÃ   rÇ   r™   rð   Úmain_input_namerË   Ú
ValueErrorrÔ   rÕ   rÖ   rë   rØ   ÚkeysÚ	__class__re   Ú*_maybe_initialize_input_ids_for_generation)r�   rì   rí   rî   Ú
input_nameÚinputs_kwargÚhas_inputs_embeds_forwardings          rp   Ú_prepare_model_inputsz%GenerationMixin._prepare_model_inputsZ  sÆ  € ð �K‰K×*Ò*Ü˜˜iÔ(Ø—‘×,Ñ,°×0DÑ0DÒDàŸ™×5Ñ5‰Jà×-Ñ-ˆJð $×'Ñ'¨
°DÓ9ˆØÐ#¨Ð(:ÜØ˜V˜HÐ$<¸Z¸Lð I,Ø,2¨8°4¸
°|À4ðIóð ð Ð%Ø!ˆFð ˜Ò$¨¸LÑ)HØ˜OÑ,Ð4Ø× Ñ  Õ1Ø—[‘[×3Ò3Ø/>Ä#Ü×%Ñ% d×&HÑ&HÓI×TÑT×YÑYÓ[óCð 0Ð,ñ 4Ü$Ø[Ð\`×\jÑ\j×\sÑ\sÐ[tð uxð xóð ð -1×,[Ñ,[Ø˜L°|ð -\ó -�˜[Ñ)ð &2°/Ñ%BÀO˜
‘àÐ%Ü$Ð%tÓuÐuØ%1°/Ñ%BÀO˜
�ð ×@Ñ@ÀÈÐWcÓdˆØ�z <Ð/Ð/ro   c                 óÐ  — |�|S |j                  d«      }t        |dd«      }| j                  j                  rH|�F|j	                  «       dd }t        j                  |t
        j                  | j                  ¬«      dz  S d}|j                  «       D ]-  }t        |t
        j                  «      sŒ|j                  d   } n d	|v r_t        j                  |dft
        j                  | j                  j                  d
k7  r| j                  ¬«      S |d	   j                  ¬«      S |€t        d«      ‚t        j                  |dft
        j                  | j                  ¬«      |z  S )z3Initializes input ids for generation, if necessary.NÚencoder_outputsÚlast_hidden_stater¾   rÀ   iœÿÿÿr   r   rµ   ÚmetazB`bos_token_id` has to be defined when no `input_ids` are provided.)rÌ   rÐ   rÃ   rÇ   Úsizeri   ÚonesÚlongrÂ   ÚvaluesrÍ   ÚTensorrÊ   Útyperò   )	r�   rì   rí   rî   rû   rü   rÊ   rß   ré   s	            rp   rõ   z:GenerationMixin._maybe_initialize_input_ids_for_generation›  sN  € ð ÐØˆMà&×*Ñ*Ð+<Ó=ˆÜ# OÐ5HÈ$ÓOÐØ�;‰;×)Ò)Ð.?Ð.Kà%×*Ñ*Ó,¨S¨bÐ1ˆEÜ—:‘:˜e¬5¯:©:¸d¿k¹kÔJÈTÑQÐQð ˆ
Ø!×(Ñ(Ó*ò 	ˆEÜ˜%¤§¡Õ.Ø"Ÿ[™[¨™^�
Ùð	ð
 ˜lÑ*Ü—:‘:Ø˜Q�Ü—j‘jð '+§k¡k×&6Ñ&6¸&Ò&@�t—{‘{ôð ð GSÐSbÑFc×FjÑFjôð ð ÐÜÐaÓbÐbä�z‰z˜: q˜/´·±ÀDÇKÁKÔPÐS_Ñ_Ð_ro   c                 óÊ  — d|v r|d   j                   d   dkD  r|d   }|j                   d   }|j                  d«      x}�9|j                  «       j                  d«      dz
  }|j	                  |dk(  d«      }|S d}|j                  d«      x}�|j                  «       }t        j                  ||z   t        j                  |j                  ¬«      }|j                  d«      }|S )z¤
        Tries to infer position ids given attention mask and past kv cache length. All instances when
        `position_ids=None` should call this method.
        r²   r   r   r´   r¾   rV   rÀ   )
rÊ   rÌ   r   ÚcumsumÚmasked_fillrÚ   ri   rÛ   rÂ   Ú	unsqueeze)r�   Úinputs_tensorrî   Ú
seq_lengthr´   r»   Úpast_lengthÚcaches           rp   Ú$_prepare_position_ids_for_generationz4GenerationMixin._prepare_position_ids_for_generationÃ  sö   € ð ˜,Ñ&¨<¸Ñ+D×+JÑ+JÈ1Ñ+MÐPQÒ+QØ(¨Ñ5ˆMà"×(Ñ(¨Ñ+ˆ
à*×.Ñ.Ð/?Ó@Ð@ˆNÐMØ)×.Ñ.Ó0×7Ñ7¸Ó;¸aÑ?ˆLà'×3Ñ3°NÀaÑ4GÈÓKˆLð Ðð ˆKØ%×)Ñ)Ð*;Ó<Ð<�ÐIØ#×2Ñ2Ó4�ä Ÿ<™<¨
°[Ñ(@ÌÏ
É
Ð[h×[oÑ[oÔpˆLØ'×1Ñ1°!Ó4ˆLØÐro   r  r’   c                 ó’  — |j                   }|j                  }d|v r|d   j                  d   dkD  r|d   }t        j                  |j                  d d t        j
                  |j                  ¬«      }|€|S t        |j                  «      dk(  xr, |j                  t        j                  t        j
                  fv }|s|S |d uxr$ t        j                  ||«      j                  «       }|d u xs% t        j                  ||«      j                  «        }	||	z  }
|j                  |«      j                  «       }||
z  ||
 z  z   }|S )Nr²   r   r   r   rÀ   )Ú_pad_token_tensorÚ_eos_token_tensorrÊ   ri   rÿ   r   rÂ   ÚlenrÁ   ÚintÚisinÚanyÚne)r�   r  r’   rî   Úpad_token_idÚeos_token_idÚdefault_attention_maskÚis_input_idsÚis_pad_token_in_inputsÚ&is_pad_token_not_equal_to_eos_token_idÚcan_infer_attention_maskÚattention_mask_from_paddingr´   s                rp   Ú&_prepare_attention_mask_for_generationz6GenerationMixin._prepare_attention_mask_for_generationÛ  sa  € ð )×:Ñ:ˆØ(×:Ñ:ˆð ˜,Ñ&¨<¸Ñ+D×+JÑ+JÈ1Ñ+MÐPQÒ+QØ(¨Ñ5ˆMô "'§¡¨M×,?Ñ,?ÀÀÐ,CÌ5Ï:É:Ð^k×^rÑ^rÔ!sÐØÐØ)Ð)ä˜=×.Ñ.Ó/°1Ñ4Òg¸×9LÑ9LÔQV×QZÑQZÔ\a×\fÑ\fÐPgÐ9gˆÙØ)Ð)à".°dÐ":Ò!oÄÇÁÈMÐ[gÓAh×AlÑAlÓAnÐØ2>À$Ð2Fò 2
Ü�J‰J�| \Ó2×6Ñ6Ó8ðL
Ð.ð $:Ð<bÑ#bÐ Ø&3×&6Ñ&6°|Ó&D×&IÑ&IÓ&KÐ#ð (Ð*BÑBÐE[Ð_wÐ^wÑEwÑwð 	ð Ðro   râ   c                 óŠ  ‡— | j                  «       }t        | d«      r4t        |d«      rd|j                  _        nt	        |t        d¬«      «       g d¢}|j                  «       D �‡�ci c]  \  Š}t        ˆfd„|D «       «      s‰|“Œ }	}}t        t        j                  |j                  «      j                  «      }
d|
v xs d|
v }|s(|	j                  «       D ��ci c]  \  }}||
v sŒ||“Œ }	}}|j                  |	d	<   |j                  |	d
<   |�|n| j                  }d|	d<   ||	|<    |di |	¤Ž|d<   |S c c}}w c c}}w )NÚhf_device_mapÚ_hf_hookT)Úio_same_device)Údecoder_Ú
cross_attnÚ	use_cacherV   rW   c              3   ó@   •K  — | ]  }‰j                  |«      –— Œ y ­w©N)Ú
startswith)Ú.0ÚpÚarguments     €rp   ú	<genexpr>zQGenerationMixin._prepare_encoder_decoder_kwargs_for_generation.<locals>.<genexpr>  s   øè ø€ ÒI°!�x×*Ñ*¨1×-ÑIùs   ƒr    rî   Úoutput_attentionsÚoutput_hidden_statesÚreturn_dictrû   rn   )Úget_encoderr™   r   r!  rU   rT   rÒ   r  rÔ   rÕ   rÖ   r×   rØ   r,  r-  rñ   )r�   r  rî   râ   r’   rð   Úirrelevant_prefixr*  ré   Úencoder_kwargsÚencoder_signatureÚencoder_accepts_wildcards          `    rp   Ú._prepare_encoder_decoder_kwargs_for_generationz>GenerationMixin._prepare_encoder_decoder_kwargs_for_generationý  sy  ø€ ð ×"Ñ"Ó$ˆô �4˜Ô)Ü�w 
Ô+Ø26�× Ñ Õ/ä" 7Ô,<ÈDÔ,QÔRò gÐð $0×#5Ñ#5Ó#7÷
ð 
á�˜%ÜÓIÐ7HÔIÔIð �e‰Oð
ˆñ 
ô
  ¤× 1Ñ 1°'·/±/Ó B× MÑ MÓNÐØ#+Ð/@Ð#@Ò#gÀNÐVgÐDgÐ Ù'à7E×7KÑ7KÓ7M÷Ù$3 H¨eÐQYÐ]nÒQn�˜%‘ðˆNñ ð /@×.QÑ.QˆÐ*Ñ+Ø1B×1WÑ1WˆÐ-Ñ.ð 0@Ð/KÑ+ÐQU×QeÑQeÐØ(,ˆ�}Ñ%Ø+8ˆÐ'Ñ(Ù7>Ñ7PÀÑ7PˆÐ&Ñ'àÐùó)
ùós   Á*!D9ÃD?Ã,D?rß   Údecoder_start_token_idrÂ   c                 óÔ  — |�d|v r|j                  d«      }nd|v r|dk7  r|j                  d«      }nd}|€| j                  }|j                  dk(  rC|j                  d   |k7  rt	        d|› d|j                  d   › �«      ‚|j                  dd«      }n+t        j                  |dft        j                  |¬	«      |z  }|€|}||fS d
| j                  j                  j                  «       v sI| j                  j                  dk(  r5d
| j                  j                  j                  j                  «       v r	 ||fS | j                  j                  dk(  r	 ||fS |dd…df   |dd…df   k7  j                  «       j!                  «       r\t        j"                  ||gd¬«      }d|v r?|d   }t        j"                  t        j$                  |«      dd…dd…f   |fd¬«      }||d<   ||fS )zGPrepares `decoder_input_ids` for generation with encoder-decoder modelsNr¸   r²   r   r   z1`decoder_start_token_id` expected to have length z	 but got r¾   rÀ   Údonutzvision-encoder-decoderÚwhisper©Údimr¿   )rË   rÂ   rÏ   rÊ   rò   Úviewri   rÿ   r   rô   re   ÚlowerrÃ   Ú
model_typerð   ÚallÚitemÚcatÚ	ones_like)r�   rß   râ   rî   r5  rÂ   r¸   r¿   s           rp   Ú)_prepare_decoder_input_ids_for_generationz9GenerationMixin._prepare_decoder_input_ids_for_generation&  s(  € ð Ð#Ð(;¸|Ñ(KØ ,× 0Ñ 0Ð1DÓ EÑØ˜LÑ(Ð-=ÀÒ-LØ ,× 0Ñ 0°Ó =Ñà $Ðð ˆ>Ø—[‘[ˆFØ!×&Ñ&¨!Ò+Ø%×+Ñ+¨AÑ.°*Ò<Ü ØGÈ
À|ÐS\Ð]s×]yÑ]yÐz{Ñ]|Ð\}Ð~óð ð &<×%@Ñ%@ÀÀQÓ%GÑ"ô —
‘
˜J¨˜?´%·*±*ÀVÔLÐOeÑeð #ð Ð$Ø 6Ðð, ! ,Ð.Ð.ð% ˜Ÿ™×/Ñ/×5Ñ5Ó7Ñ7Ø�K‰K×"Ñ"Ð&>Ò>À7ÈdÏkÉk×NaÑNa×NlÑNl×NrÑNrÓNtÑCtàð ! ,Ð.Ð.ð �[‰[×#Ñ# yÒ0Øð ! ,Ð.Ð.ð  ¢ 1 Ñ%Ð)?ÂÀ1ÀÑ)EÑE×JÑJÓL×QÑQÔSÜ %§	¡	Ð+AÐCTÐ*UÐ[]Ô ^ÐØ'¨<Ñ7Ø)5Ð6NÑ)OÐ&Ü).¯©Ü—_‘_Ð%;Ó<ºQÀÀÀ¸UÑCÐE[Ð\Øô*Ð&ð :P�Ð5Ñ6à  ,Ð.Ð.ro   Úexpand_sizerÇ   c                 óº   ‡ — ‰ dk(  r||fS ˆ fd„}|�|j                  ‰ d¬«      } ||«      }|r*|j                  d«      €t        d«      ‚ ||d   «      |d<   ||fS )zIExpands tensors from [batch_size, ...] to [batch_size * expand_size, ...]r   c                 ó�   •— | D ]?  }| |   €Œ	t        | |   t        j                  «      sŒ'| |   j                  ‰d¬«      | |<   ŒA | S )Nr   r9  )rÍ   ri   r  Úrepeat_interleave)Údict_to_expandrè   rC  s     €rp   Ú_expand_dict_for_generationzRGenerationMixin._expand_inputs_for_generation.<locals>._expand_dict_for_generationn  s^   ø€ Ø%ò d�Ø! #Ñ&Ñ2´zÀ.ÐQTÑBUÔW\×WcÑWcÕ7dØ*8¸Ñ*=×*OÑ*OÐP[ÐabÐ*OÓ*c�N 3Ò'ðdð "Ð!ro   r   r9  rû   zMIf `is_encoder_decoder` is True, make sure that `encoder_outputs` is defined.)rF  rÌ   rò   )rC  rÇ   r²   rî   rH  s   `    rp   Ú_expand_inputs_for_generationz-GenerationMixin._expand_inputs_for_generationa  s†   ø€ ð ˜!ÒØ˜lÐ*Ð*ô	"ð Ð Ø!×3Ñ3°KÀQÐ3ÓGˆIá2°<Ó@ˆáØ×ÑÐ 1Ó2Ð:Ü Ð!pÓqÐqÙ.IÈ,ÐWhÑJiÓ.jˆLÐ*Ñ+à˜,Ð&Ð&ro   ÚoutputsÚnum_new_tokensc                 ó:  — t         D ]   }||v sŒ|dv rd}n|}t        ||«      ||<    n |j                  d«      x}�&t        j                  ||dd…| d…f   gd¬«      |d<   |j                  d«      x}�:t        j                  ||j                  |j                  d   |f«      gd¬«      |d<   |sd	nd
}	|j                  |	«      x}
�dg|
j                  «       dz
  z  dgz   } t        j                  ||
j                  |
j                  ¬«      j                  |Ž |
ddd…f   z   dz   }t        j                  |
|gd¬«      }|||	<   |sdnd}|j                  |«      x}�:t        j                  ||j                  |j                  d   |f«      gd¬«      ||<   |S )am  
        Update the model kwargs to account for the `num_new_tokens` new tokens that were just generated.
        That is, update the `attention_mask`, `position_ids`, and `token_type_ids` to account for the
        new tokens of the total sequence.
        Note that this function never slices inputs, this is performed in `prepare_inputs_for_generation`.
        )rZ   rY   rV   r¼   Nr¾   r9  r½   r   r»   rº   r   rÀ   .r´   r¿   )ÚALL_CACHE_NAMESrÐ   rÌ   ri   r@  Ú	new_zerosrÊ   r:  rÛ   rÁ   rÂ   r;  Únew_ones)r�   rJ  rî   rÇ   rK  Úpossible_cache_nameÚ
cache_namer¼   r½   rá   r»   Úrequired_dimÚnext_position_idsrå   r´   s                  rp   Ú#_update_model_kwargs_for_generationz3GenerationMixin._update_model_kwargs_for_generation€  s  € ô $3ò 	ÐØ" gÒ-à&Ð*IÑIØ!2‘Jà!4�JÜ+2°7Ð<OÓ+P�˜ZÑ(Ùð	ð +×.Ñ.Ð/?Ó@Ð@ˆNÐMÜ-2¯Y©Y¸ÈÒWXÐ[iÐZiÑZjÐWjÑHkÐ7lÐrtÔ-uˆLÐ)Ñ*ð ".×!1Ñ!1Ð2EÓ!FÐFÐÐSÜ05·	±	Ø"Ð$5×$?Ñ$?ÐAR×AXÑAXÐYZÑA[Ð]kÐ@lÓ$mÐnÐtvô1ˆLÐ,Ñ-ñ
 2D™>ÐI_ÐØ(×,Ñ,Ð-=Ó>Ð>ˆLÐKà˜3 ,×"2Ñ"2Ó"4°qÑ"8Ñ9¸R¸DÑ@ˆLàg”—‘˜^°<×3EÑ3EÈl×NaÑNaÔb×gÑgÐiuÐvØ˜s B¡C˜xÑ(ñ)àñð ô
 !&§	¡	¨<Ð9JÐ*KÐQSÔ TÐØ->ˆLÐ)Ñ*ñ 6HÑ-ÐMeÐØ*×.Ñ.Ð/AÓBÐBˆNÐOÜ/4¯y©yØ ×!8Ñ!8¸.×:NÑ:NÈqÑ:QÐSaÐ9bÓ!cÐdÐjlô0ˆLÐ+Ñ,ð Ðro   Úlogits_processorÚassistant_modelrQ   Útarget_tokenizerrR   Úassistant_tokenizerc	                 ó4  — t        d„ |||fD «       «      }	|j                  �t        || ||||¬«      }
|
S |j                  �at	        |j
                  |j                  |j                  xs d|j                  || j                  j                  «       j                  ¬«      }
|
S |	rãt        d|«      }t        d|«      }t        d|«      }|j                  du rct        j                  ||| j                  j                  «       j                  |d¬	«      }d|j                  _        t#        |||||||||¬
«	      }
|
S |j                  du rt%        ||||||||¬«      }
|
S t'        dt)        |j                  «      j*                  › �«      ‚t-        ||||||¬«      }
|
S )zU
        Returns the candidate generator to be used in `assisted_generation`
        c              3   ó$   K  — | ]  }|d u–— Œ
 y ­wr&  rn   )r(  Úvs     rp   r+  z;GenerationMixin._get_candidate_generator.<locals>.<genexpr>Æ  s   è ø€ Ò"s°Q 1¨D¤=Ñ"sùó   ‚N)r²   rV  r’   rî   r  rU  r   )r  Únum_output_tokensÚmax_matching_ngram_sizeÚ
max_lengthrU  Ú
vocab_sizerQ   rR   T)rV  Úassistant_prune_lm_head)	r²   rV  r’   rî   r  rU  rW  rX  Úatm_translatorF)r²   rV  r’   rî   r  rU  rW  rX  z7Invalid value for `do_sample`: expected a boolean, got )r>  Úassistant_early_exitr#   Úprompt_lookup_num_tokensr$   r  r^  r_  rÃ   Úget_text_configr`  r	   Ú	do_sampler   Úget_translatorr’   Úrepetition_penaltyr%   r!   rò   r  re   r    )r�   r’   r²   r  rU  rî   rV  rW  rX  Údifferent_tokenizersÚcandidate_generatorrb  s               rp   Ú_get_candidate_generatorz(GenerationMixin._get_candidate_generator¸  sö  € ô  #Ñ"s¸?ÐL\Ð^qÐ:rÔ"sÓsÐà×1Ñ1Ð=Ü"=Ø#Ø $Ø"3Ø)Ø+Ø!1ô#ÐðD #Ð"ðu ×7Ñ7ÐCÜ"@Ø.×@Ñ@Ø"3×"LÑ"LØ(9×(QÑ(QÒ(VÐUVØ,×7Ñ7Ø!1ØŸ;™;×6Ñ6Ó8×CÑCô#Ððr #Ð"ñc "Ü"Ð#4°oÓFˆOÜ#Ð$=Ð?OÓPÐÜ"&Ð'@ÐBUÓ"VÐØ ×*Ñ*¨dÑ2Ü!>×!MÑ!MØ$Ø'Ø—K‘K×/Ñ/Ó1×<Ñ<Ø$3Ø,0ô"�ð HL�×1Ñ1ÔDÜ&KØ'Ø$3Ø&7Ø!-Ø"/Ø%5Ø%5Ø(;Ø#1ô
'Ð#ðF #Ð"ð1 #×,Ñ,°Ñ5Ü&SØ'Ø$3Ø&7Ø!-Ø"/Ø%5Ø%5Ø(;ô	'Ð#ð. #Ð"ô !ØMÌdÐSd×SnÑSnÓNo×NxÑNxÐMyÐzóð ô #=Ø#Ø /Ø"3Ø)Ø+Ø!1ô#Ðð #Ð"ro   Úinput_ids_seq_lengthÚencoder_input_idsÚprefix_allowed_tokens_fnÚnegative_prompt_idsÚnegative_prompt_attention_maskc
           	      ó  — t        «       }
|€g }|j                  �B|j                  dk7  r3|
j                  t        |j                  | ||	|j                  ¬«      «       |j
                  �%|
j                  t        |j
                  ¬«      «       |j                  �j|j                  dk7  r[|�?t        |j                  «      dk(  r'|
j                  t        |j                  |¬«      «       nt        j                  dt        «       |j                  �4|j                  dk7  r%|
j                  t        |j                  ¬	«      «       |j                   �3|j                   d
kD  r$|
j                  t#        |j                   «      «       |j$                  �i|j$                  d
kD  rZ|�>t        |j                  «      dk(  r&|
j                  t'        |j$                  |«      «       nt        j                  dt        «       |j(                  �/|
j                  t+        |j(                  |j,                  «      «       |j.                  �Mt1        |dd«      �@|j.                  d
kD  r1|
j                  t3        |j.                  |j,                  |¬«      «       |j4                  �Nt1        |dd«      �A|j4                  d
kD  r2|
j                  t7        ||j4                  |j,                  |¬«      «       |�%|
j                  t9        ||j:                  «      «       |j<                  �$|
j                  t?        |j<                  «      «       |j@                  �1|
j                  tC        |jD                  |j@                  |¬«      «       |jF                  du r|
j                  tI        «       «       |jJ                  �0|
j                  tM        |jJ                  |j,                  |«      «       |jN                  �&|
j                  tQ        |jN                  |¬«      «       |jR                  �A|}|dkD  s|j<                  €|n|dz   }|
j                  tU        |jR                  ||¬«      «       | jW                  |
|«      }
|jX                  �rŽ|j:                  �†|j:                  dkD  rwt[        |j,                  t\        «      rt        |j,                  «      dz   }nFt[        |j,                  t^        j`                  «      r|j,                  j                  d
   dz   }nd}nd}|jb                  �3|jb                  dk7  r$|
j                  te        |jb                  «      «       |jf                  �%|
j                  ti        |jf                  ¬«      «       |jj                  �5|jj                  d
k7  r&|
j                  tm        |jj                  |¬«      «       |jn                  �5|jn                  dk  r&|
j                  tq        |jn                  |¬«      «       |jr                  �&|
j                  tu        |jr                  |¬«      «       |jv                  �5|jv                  dk  r&|
j                  ty        |jv                  |¬«      «       |jz                  �>d|jz                  cxk  rdk  r)n n&|
j                  t}        |jz                  |¬«      «       |j~                  �?d|j~                  cxk  rdk  r*n n'|
j                  t�        |j~                  ||¬«      «       |j‚                  �M|
j                  |j‚                  j…                  | j†                  j‰                  «       jŠ                  |«      «       |jŒ                  du r|
j                  t�        «       «       |
S )zÁ
        This class returns a [`LogitsProcessorList`] list object that contains all relevant [`LogitsProcessor`]
        instances used to modify the scores of the language model head.
        Nr   )Úunconditional_idsÚunconditional_attention_maskr$  ©Úsequence_biasç      ð?r   )Úpenaltyrm  zyPassing `encoder_repetition_penalty` requires some form of `input_ids` to be passed to `generate`, ignoring the argument.)rw  r   z{Passing `encoder_no_repeat_ngram_size` requires some form of `input_ids` to be passed to `generate`, ignoring the argument.r  rÆ   T)Útop_h)Útop_kÚmin_tokens_to_keep)Útop_prz  )Úmin_prz  )Úmassrz  ç        )Úepsilonrz  )r  rz  rÂ   )Hr8   Úguidance_scaleÚappendrH   r$  ru  r@   Úencoder_repetition_penaltyr  rÊ   r0   ÚwarningsÚwarnÚUserWarningrh  r?   Úno_repeat_ngram_sizer=   Úencoder_no_repeat_ngram_sizer/   Úbad_words_idsr<   r  Ú
min_lengthrÐ   r9   Úmin_new_tokensr:   r>   Ú	num_beamsÚforced_bos_token_idr4   Úforced_eos_token_idr5   r_  Úremove_invalid_valuesr6   Ú exponential_decay_length_penaltyr3   Úsuppress_tokensrB   Úbegin_suppress_tokensrA   Ú_merge_criteria_processor_listrf  rÍ   Úlistri   r  ÚtemperaturerC   rx  rD   ry  rE   r{  rF   r|  r;   Ú	typical_prG   Úepsilon_cutoffr1   Ú
eta_cutoffr2   Úwatermarking_configÚconstruct_processorrÃ   re  r`  Úrenormalize_logitsr7   )r�   r’   rl  rm  rn  rU  rÂ   rî   ro  rp  Ú
processorsÚbegin_indexrz  s                rp   Ú_get_logits_processorz%GenerationMixin._get_logits_processor  s  € ô" )Ó*ˆ
ØÐ#Ø!Ðà×+Ñ+Ð7Ð<M×<\Ñ<\Ð`aÒ<aØ×ÑÜ>Ø%×4Ñ4ØØ&9Ø1OØ/×9Ñ9ôôð ×*Ñ*Ð6Ø×ÑÔ9ÐHY×HgÑHgÔhÔið ×8Ñ8ÐDØ!×<Ñ<ÀÒCà Ð,´Ð5F×5LÑ5LÓ1MÐQRÒ1RØ×!Ñ!Ü;Ø 1× LÑ LØ*;ôõô —‘ð9äôð
 ×/Ñ/Ð;Ð@Q×@dÑ@dÐhkÒ@kØ×ÑÔ>ÐGX×GkÑGkÔlÔmØ×1Ñ1Ð=ÐBS×BhÑBhÐklÒBlØ×ÑÔ:Ð;L×;aÑ;aÓbÔcà×:Ñ:ÐFØ!×>Ñ>ÀÒBà Ð,´Ð5F×5LÑ5LÓ1MÐQRÒ1RØ×!Ñ!Ü7Ø)×FÑFØ)óõô —‘ð9äôð
 ×*Ñ*Ð6Ø×ÑÜ)Ø%×3Ñ3Ø%×7Ñ7óôð ×(Ñ(Ð4ÜÐ)Ð+>ÀÓEÐQØ!×,Ñ,¨qÒ0à×ÑÜ(Ø%×0Ñ0Ø%×7Ñ7Ø!ôôð ×,Ñ,Ð8ÜÐ)Ð+>ÀÓEÐQØ!×0Ñ0°1Ò4à×ÑÜ1Ø(Ø%×4Ñ4Ø%×7Ñ7Ø!ô	ôð $Ð/Ø×ÑÜ0Ø,Ø%×/Ñ/óôð ×0Ñ0Ð<Ø×ÑÜ-Ø%×9Ñ9óôð
 ×0Ñ0Ð<Ø×ÑÜ-Ø%×0Ñ0Ø%×9Ñ9Ø!ôôð ×2Ñ2°dÑ:Ø×ÑÔ9Ó;Ô<Ø×=Ñ=ÐIØ×ÑÜ-Ø%×FÑFØ%×7Ñ7Ø(óôð ×,Ñ,Ð8Ø×ÑÜ-Ø%×5Ñ5Ø!ôôð ×2Ñ2Ð>Ø.ˆKð )¨1Ò,Ð0A×0UÑ0UÐ0]ñ à  1‘_ð ð
 ×ÑÜ4Ø%×;Ñ;ØØ!ôôð ×8Ñ8¸ÐEUÓVˆ
ð ×&Ó&ð !×*Ñ*Ð6Ð;L×;VÑ;VÐYZÒ;ZÜÐ/×AÑAÄ4ÔHÜ),Ð->×-PÑ-PÓ)QÐTUÑ)UÑ&ÜÐ 1× CÑ CÄUÇ\Á\ÔRØ):×)LÑ)L×)RÑ)RÐSTÑ)UÐXYÑ)YÑ&à)*Ñ&à%&Ð"ð !×,Ñ,Ð8Ð=N×=ZÑ=ZÐ^aÒ=aØ×!Ñ!Ô"9Ð:K×:WÑ:WÓ"XÔYØ ×&Ñ&Ð2Ø×!Ñ!Ô"2Ð9J×9PÑ9PÔ"QÔRØ ×&Ñ&Ð2Ð7H×7NÑ7NÐRSÒ7SØ×!Ñ!Ü$Ð+<×+BÑ+BÐWiÔjôð !×&Ñ&Ð2Ð7H×7NÑ7NÐQTÒ7TØ×!Ñ!Ü$Ð+<×+BÑ+BÐWiÔjôð !×&Ñ&Ð2à×!Ñ!Ü$Ð+<×+BÑ+BÐWiÔjôð !×*Ñ*Ð6Ð;L×;VÑ;VÐY\Ò;\Ø×!Ñ!Ü'Ð->×-HÑ-HÐ]oÔpôð !×/Ñ/Ð;ÀÐFW×FfÑFfÔ@lÐilÕ@lØ×!Ñ!Ü'Ø 1× @Ñ @ÐUgôôð
 !×+Ñ+Ð7¸CÐBS×B^ÑB^Ô<dÐadÕ<dØ×!Ñ!Ü#Ø 1× <Ñ <ÐQcÐlrôôð ×0Ñ0Ð<Ø×ÑØ!×5Ñ5×IÑIØ—K‘K×/Ñ/Ó1×<Ñ<¸fóôð ×/Ñ/°4Ñ7Ø×ÑÔ0Ó2Ô3ØÐro   Ústopping_criteriaÚ	tokenizerc                 óª  — t        «       }|j                  �=t        | j                  dd «      }|j	                  t        |j                  |¬«      «       |j                  �%|j	                  t        |j                  ¬«      «       |j                  �3|€t        d«      ‚|j	                  t        |j                  |¬«      «       |j                  �%|j	                  t        |j                  ¬«      «       |j                  r@|j                  �4|j                  dkD  r%|j	                  t        |j                  ¬«      «       | j!                  ||«      }|S )	NÚmax_position_embeddings)r_  r¡  )Úmax_timea  There are one or more stop strings, either in the arguments to `generate` or in the model's generation config, but we could not locate a tokenizer. When generating with stop strings, you must pass the model's tokenizer to the `tokenizer` argument of `generate`.)Ústop_stringsrŸ  )r  r   )Úassistant_confidence_threshold)rN   r_  rÐ   rÃ   r�  rK   r¢  rL   r£  rò   rO   r  rJ   Úis_assistantr¤  rI   r’  )r�   r’   rž  rŸ  Úcriteriar¡  s         rp   Ú_get_stopping_criteriaz&GenerationMixin._get_stopping_criteriaì  s3  € ô (Ó)ˆØ×'Ñ'Ð3Ü&-¨d¯k©kÐ;TÐVZÓ&[Ð#Ø�O‰OÜ!Ø0×;Ñ;Ø,Côôð ×%Ñ%Ð1Ø�O‰OœOÐ5F×5OÑ5OÔPÔQØ×)Ñ)Ð5ØÐ Ü ðsóð ð
 �O‰OÔ.Ð<M×<ZÑ<ZÐfoÔpÔqØ×.Ñ.Ð:Ø�O‰OÔ,Ð:K×:]Ñ:]Ô^Ô_à×*Ò*Ø!×@Ñ@ÐLØ!×@Ñ@À1ÒDà�O‰OÜ"ÐBS×BrÑBrÔsôð ×6Ñ6°xÐARÓSˆØˆro   Údefault_listÚcustom_listc                 óÀ  — t        |«      dk(  r|S  t        |«      «       }|D ]›  }d}|D ]~  }t        |«      t        |«      u sŒt        |t        «      rdnd}t        j                  d|› dt        |«      › dt        |«      › dt        |«      › d	�	«       |j                  |«       d
} n |rŒ‹|j                  |«       Œ� |D ]  }||vsŒ|j                  |«       Œ |S )a4  
        Merge user-defined processors/criteria with the ones instantiated inside `generate`. In case the same
        processor/criteria is present on both lists, use the user-defined one.

        (Note: up to v4.49.0, this function threw an exception is the same logit processor was found twice.)
        r   Fzstopping criteriazlogits processorz	A custom z	 of type zt has been passed to `.generate()`, but it was also created in `.generate()`, given its parameterization. The custom z5 will take precedence. Please check the docstring of z$ to see related `.generate()` flags.T)r  r  rÍ   rM   r—   rÙ   r�  )r�   r¨  r©  Ú
final_listÚdefaultÚusing_customÚcustomÚobject_types           rp   r’  z.GenerationMixin._merge_criteria_processor_list  s  € ô ˆ{Ó˜qÒ ØÐà'”T˜,Ó'Ó)ˆ
Ø#ò 	+ˆGØ ˆLØ%ò �Ü˜“<¤4¨£=Ò0Ü9CÀFÔL\Ô9]Ñ"5Ðcu�KÜ×'Ñ'Ø# K =°	¼$¸v»,¸ð HeÜeiÐjpÓeqÐdrð sOÜOSÐTZË|Ènð ]/ð/ôð ×%Ñ% fÔ-Ø#'�LÙðò  Ø×!Ñ! 'Õ*ð	+ð" "ò 	*ˆFØ˜ZÒ'Ø×!Ñ! &Õ)ð	*ð Ðro   r`   ra   r{   Únormalize_logitsc                 óì  — |€it        j                  |d   j                  d   «      j                  dd«      j	                  |j
                  «      }|j                  dt        |«      «      }t        j                  |«      j                  t        |«      d«      j                  dd«      }|rŒ|j                  d| j                  j                  «       j                  |j                  d   «      }t         j                  j                  j!                  |d¬«      }|j                  d|j                  d   «      }|dk  }d|j#                  «       z
  j%                  d«      j'                  «       }|j)                  «       dd…d|…f   }|dd…d|…f   }d||<   || j                  j                  «       j                  z  }|j                  d   |z
  }	|dd…|	d…f   |z   }
|j+                  d|
«      }d||<   |S )a‡  
        Computes the transition scores of sequences given the generation scores (and beam indices, if beam search was
        used). This is a convenient method to quickly obtain the scores of the selected tokens at generation time.

        Parameters:
            sequences (`torch.LongTensor`):
                The generated sequences. The second dimension (sequence_length) is either equal to `max_length` or
                shorter if all batches finished early due to the `eos_token_id`.
            scores (`tuple(torch.FloatTensor)`):
                Transition scores for each vocabulary token at each generation step. Beam transition scores consisting
                of log probabilities of tokens conditioned on log softmax of previously generated tokens in this beam.
                Tuple of `torch.FloatTensor` with up to `max_new_tokens` elements (one element for each generated token),
                with each tensor of shape `(batch_size*num_beams, config.vocab_size)`.
            beam_indices (`torch.LongTensor`, *optional*):
                Beam indices of generated token id at each generation step. `torch.LongTensor` of shape
                `(batch_size*num_return_sequences, sequence_length)`. Only required if a `num_beams>1` at
                generate-time.
            normalize_logits (`bool`, *optional*, defaults to `False`):
                Whether to normalize the logits (which, for legacy reasons, may be unnormalized).

        Return:
            `torch.Tensor`: A `torch.Tensor` of shape `(batch_size*num_return_sequences, sequence_length)` containing
                the transition scores (logits)

        Examples:

        ```python
        >>> from transformers import GPT2Tokenizer, AutoModelForCausalLM
        >>> import numpy as np

        >>> tokenizer = GPT2Tokenizer.from_pretrained("gpt2")
        >>> model = AutoModelForCausalLM.from_pretrained("openai-community/gpt2")
        >>> tokenizer.pad_token_id = tokenizer.eos_token_id
        >>> inputs = tokenizer(["Today is"], return_tensors="pt")

        >>> # Example 1: Print the scores for each token generated with Greedy Search
        >>> outputs = model.generate(**inputs, max_new_tokens=5, return_dict_in_generate=True, output_scores=True)
        >>> transition_scores = model.compute_transition_scores(
        ...     outputs.sequences, outputs.scores, normalize_logits=True
        ... )
        >>> # input_length is the length of the input prompt for decoder-only models, like the GPT family, and 1 for
        >>> # encoder-decoder models, like BART or T5.
        >>> input_length = 1 if model.config.is_encoder_decoder else inputs.input_ids.shape[1]
        >>> generated_tokens = outputs.sequences[:, input_length:]
        >>> for tok, score in zip(generated_tokens[0], transition_scores[0]):
        ...     # | token | token string | log probability | probability
        ...     print(f"| {tok:5d} | {tokenizer.decode(tok):8s} | {score.numpy():.3f} | {np.exp(score.numpy()):.2%}")
        |   262 |  the     | -1.414 | 24.33%
        |  1110 |  day     | -2.609 | 7.36%
        |   618 |  when    | -2.010 | 13.40%
        |   356 |  we      | -1.859 | 15.58%
        |   460 |  can     | -2.508 | 8.14%

        >>> # Example 2: Reconstruct the sequence scores from Beam Search
        >>> outputs = model.generate(
        ...     **inputs,
        ...     max_new_tokens=5,
        ...     num_beams=4,
        ...     num_return_sequences=4,
        ...     return_dict_in_generate=True,
        ...     output_scores=True,
        ... )
        >>> transition_scores = model.compute_transition_scores(
        ...     outputs.sequences, outputs.scores, outputs.beam_indices, normalize_logits=False
        ... )
        >>> # If you sum the generated tokens' scores and apply the length penalty, you'll get the sequence scores.
        >>> # Tip 1: recomputing the scores is only guaranteed to match with `normalize_logits=False`. Depending on the
        >>> # use case, you might want to recompute it with `normalize_logits=True`.
        >>> # Tip 2: the output length does NOT include the input length
        >>> output_length = np.sum(transition_scores.numpy() < 0, axis=1)
        >>> length_penalty = model.generation_config.length_penalty
        >>> reconstructed_scores = transition_scores.sum(axis=1) / (output_length**length_penalty)
        >>> print(np.allclose(outputs.sequences_scores, reconstructed_scores))
        True
        ```Nr   r¾   r   r9  )ri   rÛ   rÊ   r;  ÚtorÂ   Úexpandr  ÚstackÚreshapeÚ	transposerÃ   re  r`  r
   Ú
functionalÚlog_softmaxr   ÚsumÚmaxrÈ   Úgather)r�   r`   ra   r{   r°  Ústacked_scoresÚbeam_indices_maskÚmax_beam_lengthÚbeam_sequence_indicesÚcut_idxÚindicesÚtransition_scoress               rp   Úcompute_transition_scoresz)GenerationMixin.compute_transition_scores7  sï  € ðh ÐÜ Ÿ<™<¨¨q©	¯©¸Ñ(:Ó;×@Ñ@ÀÀQÓG×JÑJÈ9×K[ÑK[Ó\ˆLØ'×.Ñ.¨r´3°v³;Ó?ˆLô (-§{¡{°6Ó':×'BÑ'BÄ3ÀvÃ;ÐPRÓ'S×']Ñ']Ð^_ÐabÓ'cˆñ Ø+×3Ñ3Ø�D—K‘K×/Ñ/Ó1×<Ñ<¸n×>RÑ>RÐSUÑ>VóˆNô #ŸX™X×0Ñ0×<Ñ<¸^ÐQRÐ<ÓSˆNØ+×3Ñ3°B¸×8LÑ8LÈRÑ8PÓQˆNð )¨1Ñ,ÐØÐ0×5Ñ5Ó7Ñ7×<Ñ<¸RÓ@×DÑDÓFˆØ#×)Ñ)Ó+ªAÐ/?°Ð/?Ð,?Ñ@ˆØ-ªaÐ1A°/Ð1AÐ.AÑBÐð +,ˆÐ&Ñ'ð !-¨t¯{©{×/JÑ/JÓ/L×/WÑ/WÑ WÐð —/‘/ "Ñ%¨Ñ7ˆØšA˜w™x˜KÑ(Ð+@Ñ@ˆð +×1Ñ1°!°WÓ=Ðð 01ÐÐ+Ñ,à Ð ro   c                 ó.  ‡ ‡— |t         j                  k(  rd|v rt        d«      ‚|t         j                  k(  rV|j                  dkD  rt        d|j                  › d�«      ‚‰ j
                  r"t        d‰ j                  j                  › �«      ‚|j                  d«      xŠ�ð‰ j                  j                  rc‰j                  j                  sMg d¢}t        ‰j                  «      D �cg c]	  }||v sŒ|‘Œ }}t        ˆˆ fd	„|D «       «      }|st        d
«      ‚d}‰ j                  j                  «       j                  ‰j                  j                  «       j                  k(  rd|v rt        d|› d�«      ‚y d|vsd|vrt        d|› d�«      ‚y y c c}w )NÚstreamerzZ`streamer` cannot be used with beam search (yet!). Make sure that `num_beams` is set to 1.r   zFnum_return_sequences has to be 1 when doing assisted generate, but is ú.zCassisted generation is not supported with stateful models, such as rV  )Úencoder_attention_headsÚencoder_ffn_dimÚencoder_layersc              3   óx   •K  — | ]1  }t        ‰j                  |«      t        ‰j                  |«      k(  –— Œ3 y ­wr&  )rÐ   rÃ   )r(  ÚattrrV  r�   s     €€rp   r+  z<GenerationMixin._validate_generation_mode.<locals>.<genexpr>Ì  s4   øè ø€ ò  Ø\`”G˜DŸK™K¨Ó.´'¸/×:PÑ:PÐRVÓ2WÕWñ ùs   ƒ7:zÊThe main model and the assistant don't have compatible encoder-dependent input shapes. Ensure you load the assistant with the correct encoder-decoder class, e.g. `AutoModelForSpeechSeq2Seq` for Whisper.zc(see https://huggingface.co/docs/transformers/en/generation_strategies#universal-assisted-decoding)rX  z•`assistant_tokenizer` is not required when the main and assistant models use the same tokenizer. Please omit `assistant_tokenizer` from `generate()` rŸ  z~The main and assistant models have different tokenizers. Please provide `tokenizer` and `assistant_tokenizer` to `generate()` )r-   ÚBEAM_SEARCHrò   ÚASSISTED_GENERATIONÚnum_return_sequencesÚ_is_statefulrô   re   rÌ   rÃ   rÇ   Údirr>  re  r`  )	r�   Úgeneration_moder’   Úgeneration_mode_kwargsÚattributes_to_checkrË  Ú	are_equalÚdoc_referencerV  s	   `       @rp   Ú_validate_generation_modez)GenerationMixin._validate_generation_mode³  s  ù€ ð œn×8Ñ8Ò8¸ZÐKaÑ=aÜØlóð ð œn×@Ñ@Ò@Ø ×5Ñ5¸Ò9Ü ðØ/×DÑDÐEÀQðHóð ð × Ò ô !ØYÐZ^×ZhÑZh×ZqÑZqÐYrÐsóð ð  6×9Ñ9Ð:KÓLÐLˆOÐYØ�{‰{×-Ò-°o×6LÑ6L×6_Ò6_Ú&fÐ#Ü8;¸O×<RÑ<RÓ8SÖ&s°ÐW[Ð_rÒWr¢tÐ&sÐ#Ð&sÜô  Ødwô ó �	ñ !Ü$ðNóð ð vð ð �{‰{×*Ñ*Ó,×7Ñ7¸?×;QÑ;Q×;aÑ;aÓ;c×;nÑ;nÒnØ(Ð,BÑBÜ$ð pð  q~ð  pð  @ð  Aóð ð Cð
 Ð&<Ñ<Ð@UÐ]sÑ@sÜ$ð Yð  Zgð  Yhð  hið  jóð ð Atð- Zùò 'ts   Ã(	FÃ2Fc                 óÜ  — | j                   j                  rdD ]  }|j                  |d«       Œ g }t        t	        j
                  | j                  «      j                  «      }d|v sd|v r5|t        t	        j
                  | j                  «      j                  «      z  }| j                   j                  rÖt        | | j                  d«      }t        | dd«      }|€|�t        |dd«      }|�7t        t	        j
                  |j                  «      j                  «      }||z  }t        | dd«      }|€|�t        |dd«      }|�Ht        t	        j
                  |j                  «      j                  «      }	||	D �
ch c]  }
d|
› �’Œ	 c}
z  }|j                  «       D ]7  \  }}|€Œ	||vsŒ|t        j                  vsŒ!|dk7  sŒ'|j                  |«       Œ9 |rt        d	|› d
�«      ‚yc c}
w )zXValidates model kwargs for generation. Generate argument typos will also be caught here.)r¸   Nr    rî   rð   Údecoderr"  Údebug_ioz8The following `model_kwargs` are not used by the model: zG (note: typos in the generate arguments will also show up in this list))rÃ   rÇ   rË   rÔ   rÕ   rÖ   rë   rØ   r×   rÐ   Úbase_model_prefixrÒ   r   Ú__optional_keys__r�  rò   )r�   rî   rè   Úunused_model_argsÚ
model_argsÚ
base_modelrð   Úencoder_model_argsrØ  Údecoder_model_argsÚxré   s               rp   Ú_validate_model_kwargsz&GenerationMixin._validate_model_kwargsã  sú  € ð �;‰;×)Ò)Ø,ò ,�Ø× Ñ   dÕ+ð,ð ÐÜœ×*Ñ*¨4×+MÑ+MÓN×YÑYÓZˆ
ð �zÑ! ^°zÑ%AØœ#œg×/Ñ/°·±Ó=×HÑHÓIÑIˆJð �;‰;×)Ò)Ü   t×'=Ñ'=¸tÓDˆJô ˜d I¨tÓ4ˆGð ˆ :Ð#9Ü! *¨i¸Ó>�àÐ"Ü%(¬×):Ñ):¸7¿?¹?Ó)K×)VÑ)VÓ%WÐ"ØÐ0Ñ0�
ô ˜d I¨tÓ4ˆGØˆ :Ð#9Ü! *¨i¸Ó>�àÐ"Ü%(¬×):Ñ):¸7¿?¹?Ó)K×)VÑ)VÓ%WÐ"ØÐ7IÖJ°! ¨!¨š~ÒJÑJ�
ð '×,Ñ,Ó.ò 	.‰JˆC�àÑ!Ø˜zÒ)ØÔ1×CÑCÒCØ˜:Ó%à!×(Ñ(¨Õ-ð	.ñ ÜØJÐK\ÐJ]ð ^Fð Fóð ð ùò Ks   Å<G)c           	      ó¶  — |r4|j                   €(t        j                  d|j                  › d�t        «       ||j                  k\  r9| j
                  j                  rdnd}t        d|› d|› d|j                  › d	�«      ‚d
}|r|d|j                  › d�z  }|j                  �Q|j                  |j                  kD  r8t        j                  d|j                  › d|j                  › d�|z   t        «       |j                  �[|j                  |z   }||j                  kD  r<t        j                  d|j                  › d|› d|j                  › d�|z   t        «       yyy)z=Performs validation related to the resulting generated lengthNz0Using the model-agnostic default `max_length` (=zz) to control the generation length. We recommend setting `max_new_tokens` to control the maximum length of the generation.r¸   r²   zInput length of z is z, but `max_length` is set to z}. This can lead to unexpected behavior. You should consider increasing `max_length` or, better yet, setting `max_new_tokens`.z Generation will stop at the defined maximum length. You should decrease the minimum length and/or increase the maximum length.z" Note that `max_length` is set to z, its default value.z-Unfeasible length constraints: `min_length` (z.) is larger than the maximum possible length (z).z1Unfeasible length constraints: `min_new_tokens` (z$), when added to the prompt length (z/), is larger than the maximum possible length ()
Úmax_new_tokensrƒ  r„  r_  r…  rÃ   rÇ   rò   r‰  rŠ  )r�   r’   Úinput_ids_lengthÚhas_default_max_lengthÚinput_ids_stringÚmin_length_error_suffixr‰  s          rp   Ú_validate_generated_lengthz*GenerationMixin._validate_generated_length  s¼  € ñ
 "Ð&7×&FÑ&FÐ&Nä�M‰MØBÐCT×C_ÑC_ÐB`ð að ô ô	ð Ð0×;Ñ;Ò;Ø6:·k±k×6TÒ6TÑ2ÐZeÐÜØ"Ð#3Ð"4°DÐ9IÐ8Jð KØ%×0Ñ0Ð1ð 2UðUóð ð+ð 	 ñ "Ø#Ø4Ð5F×5QÑ5QÐ4RÐRfÐgñÐ#ð ×'Ñ'Ð3Ð8I×8TÑ8TÐWh×WsÑWsÒ8sÜ�M‰MØ?Ð@Q×@\Ñ@\Ð?]ð ^1Ø1B×1MÑ1MÐ0NÈbðRØTkñläôð
 ×+Ñ+Ð7Ø*×9Ñ9Ð<LÑLˆJØÐ-×8Ñ8Ò8Ü—‘ØGÐHY×HhÑHhÐGið j3Ø3CÐ2Dð E5Ø5F×5QÑ5QÐ4RÐRTðVàXoñpô  õ	ð 9ð 8ro   c                 óT  — |j                   �S|s<|j                  �0t        j                  d|j                   › d|j                  › d�«       |j                   |z   |_        nœ|dk(  rM||j                  d   k7  r;| j
                  j                  s%|s#|xj                  |j                  d   z  c_        nJ|rH|j                  |z   |_        t        | j
                  dd«      }|�t        |j                  |«      |_        |j                  �H|s0t        j                  d|j                  › d	|j                  › d
�«       |j                  |z   |_
        |S |dk(  rS||j                  d   k7  rA| j
                  j                  s+t        |j                  |j                  d   z
  d«      |_
        |S )z]Prepared max and min length in generation configs to avoid clashes between similar attributesNzBoth `max_new_tokens` (=z) and `max_length`(=zÇ) seem to have been set. `max_new_tokens` will take precedence. Please refer to the documentation for more information. (https://huggingface.co/docs/transformers/main/en/main_classes/text_generation)rµ   r   r¡  zBoth `min_new_tokens` (=z) and `min_length`(=zÇ) seem to have been set. `min_new_tokens` will take precedence. Please refer to the documentation for more information. (https://huggingface.co/docs/transformers/main/en/main_classes/text_generation)r   )rä  r_  r—   ÚwarningrÊ   rÃ   rÇ   rÐ   ÚminrŠ  r‰  rº  )r�   r’   ræ  Úhas_default_min_lengthrâ   rå  r  r¡  s           rp   Ú_prepare_generated_lengthz)GenerationMixin._prepare_generated_lengthH  s¿  € ð ×+Ñ+Ð7Ù)Ð.?×.JÑ.JÐ.VÜ—‘Ø.Ð/@×/OÑ/OÐ.PÐPdØ(×3Ñ3Ð4ð 5fðfôð ,=×+KÑ+KÐN^Ñ+^ÐÕ(ð
  Ò/Ø  M×$7Ñ$7¸Ñ$:Ò:Ø—K‘K×2Ò2Ù*à×(Ò(¨M×,?Ñ,?ÀÑ,BÑBÖ(Ù#Ø+<×+GÑ+GÐJZÑ+ZÐÔ(Ü&-¨d¯k©kÐ;TÐVZÓ&[Ð#Ø&Ð2Ü/2Ð3D×3OÑ3OÐQhÓ/iÐ!Ô,ð ×+Ñ+Ð7Ù)Ü—‘Ø.Ð/@×/OÑ/OÐ.PÐPdØ(×3Ñ3Ð4ð 5fðfôð ,=×+KÑ+KÐN^Ñ+^ÐÔ(ð !Ð ð  Ò/Ø  M×$7Ñ$7¸Ñ$:Ò:Ø—K‘K×2Ò2ä+.Ð/@×/KÑ/KÈm×NaÑNaÐbcÑNdÑ/dÐfgÓ+hÐÔ(à Ð ro   r    c                 óŠ  — |du}|€Wt        | j                  j                  «       «      dkD  r't        d| j                  j                  «       › d�«      ‚t	        «       }t        j                  |«      }| j                  j                  «       } |j                  di | j                  j                  «       ¤dddœ¤Ž  |j                  di |¤ddi¤Ž  |j                  di |¤Ž}|j                  dk(  rd|_        |rt        |j                  «       «      t        |j                  «       «      z
  rLt        |j                  «       «      t        |j                  «       «      z
  }t        j                  d	|› d
�«       |j                   }|j"                  }|j                  |rd|ini «       |j                  |rd|ini «       ||fS )zË
        Prepares the base generation config, then applies any generation configuration options from kwargs. This
        function handles retrocompatibility with respect to configuration files.
        Nr   zrYou have modified the pretrained model configuration to control generation We detected the following values set - zÛ. This strategy to control generation is not supported anymore. Please use and modify `model.generation_config` (see https://huggingface.co/docs/transformers/generation_strategies#default-text-generation-configuration )T)Údefaults_onlyÚallow_custom_entriesrð  ÚhybridzHPassing `generation_config` together with generation-related arguments=(zž) is deprecated and will be removed in future versions. Please pass either a `generation_config` object OR all generation parameters explicitly, but not both.r,  r-  rn   )r  rÃ   Ú_get_generation_parametersrò   r,   ÚcopyÚdeepcopyr’   Ú_get_default_generation_paramsÚupdater”   Úcache_implementationrÔ   ró   r—   rÙ   r,  r-  )	r�   r’   r    Úgeneration_config_providedÚglobal_defaultsrî   Úgeneration_kwargsr,  r-  s	            rp   Ú_prepare_generation_configz*GenerationMixin._prepare_generation_config€  sÑ  € ð &7¸dÐ%BÐ"ØÐ$ä�4—;‘;×9Ñ9Ó;Ó<¸qÒ@Ü ð>Ø>B¿k¹k×>dÑ>dÓ>fÐ=gð hBðBóð ô !1Ó 2Ðô !ŸM™MÐ*;Ó<Ðð ×0Ñ0×OÑOÓQˆØ Ð× Ñ Ñs 4×#9Ñ#9×#AÑ#AÓ#CÐsÐSWÐnrÔsØ Ð× Ñ ÑG ?ÑGÀ$ÓGð 0Ð(×/Ñ/Ñ9°&Ñ9ˆð ×1Ñ1°XÒ=Ø59ÐÔ2ñ &¬#¨f¯k©k«mÓ*<¼sÀ<×CTÑCTÓCVÓ?WÒ*WÜ # F§K¡K£MÓ 2´S¸×9JÑ9JÓ9LÓ5MÑ MÐÜ×ÑðØ/Ð0ð 17ð7ôð .×?Ñ?ÐØ0×EÑEÐØ×ÑÑHYÐ0Ð2CÑDÐ_aÔbØ×ÑÑNbÐ3Ð5IÑJÐhjÔkà  ,Ð.Ð.ro   rø  Úmax_cache_lenc                 ó~  — d|v }d}t        | d«      rWt        | j                  t        «      r| j                  j                  }n&t        | j                  t
        «      r| j                  }|du xs1 |j                  |k7  xs  |j                  |k7  xs |j                  |k  }t        | dd«      }t        |t        «      r0|xs, |j                  j                  |d   d   j                  d   k7  }|r©| j                  j                  d¬«      ||d	œ}	t        d
i |	¤Ž| _        | j                  j                  rW| j                  j                  d¬«      |d   d   j                  d   |d	œ}
t        | j                  t        d
i |
¤Ž«      | _        | j                  S | j                  j                  «        | j                  S )zö
        Sets a cache for `generate`, that will persist across calls. A new cache will only be initialized a
        new `generate` call requires a larger cache or uses a different batch size.

        Returns the resulting cache object.
        Ú	offloadedNÚ_cacherû   r   r   T©rØ  )rÃ   rý  Ú
offloadingrn   )r™   rÍ   r   r   Úself_attention_cacher   r  Úmax_batch_sizerý  rÐ   Úcross_attention_cacherÊ   rÃ   re  rÇ   Úreset)r�   rø  rß   rý  rî   Úoffload_cacheÚcache_to_checkÚneed_new_cacheÚencoder_decoder_cacheÚself_attention_cache_kwargsÚcross_attention_cache_kwargss              rp   Ú_prepare_static_cachez%GenerationMixin._prepare_static_cacheÃ  s¹  € ð $Ð';Ð;ˆà-1ˆÜ�4˜Ô"Ü˜$Ÿ+™+Ô':Ô;Ø!%§¡×!AÑ!A‘Ü˜DŸK™K¬Ô5Ø!%§¡�ð ˜dÐ"ò <Ø×(Ñ(¨MÑ9ò<à×,Ñ,°
Ñ:ò<ð ×+Ñ+¨mÑ;ð	 	ô !(¨¨h¸Ó =ÐÜÐ+Ô-@ÔAàò ?Ø(×>Ñ>×LÑLØÐ 1Ñ2°1Ñ5×;Ñ;¸AÑ>ñ?ð ñ àŸ+™+×5Ñ5¸dÐ5ÓCØ!.Ø+ñ+Ð'ô
 &ÑDÐ(CÑDˆDŒKØ�{‰{×-Ò-à"Ÿk™k×9Ñ9À$Ð9ÓGØ%1Ð2CÑ%DÀQÑ%G×%MÑ%MÈaÑ%PØ"/ñ0Ð,ô
 2°$·+±+¼{Ñ?jÐMiÑ?jÓk�”ð �{‰{Ðð �K‰K×ÑÔØ�{‰{Ðro   Úclsc                 ól   ‡ — d}d‰ j                   j                  «       v xs t        ˆ fd„|D «       «      S )z{
        Return `True` if current model can use a `DynamicCache` instance when initializing the `past_key_values`.
        )ÚreformerÚminimaxÚxlnetÚ
olmohybridÚrwkvÚxlstmÚ	minimaxm2c              3   óV   •K  — | ]   }|‰j                   j                  «       v–— Œ" y ­wr&  )re   r<  )r(  Úunsupported_namer  s     €rp   r+  zBGenerationMixin._supports_default_dynamic_cache.<locals>.<genexpr>  s)   øè ø€ ò :
Ø=MÐ C§L¡L×$6Ñ$6Ó$8Ô8ñ:
ùs   ƒ&))re   r<  r>  )r  Úunsupported_model_namess   ` rp   Ú_supports_default_dynamic_cachez/GenerationMixin._supports_default_dynamic_cacheö  s@   ø€ ð#
Ðð ˜cŸl™l×0Ñ0Ó2Ð2ò 
´có :
ØQhô:
ó 7
ð 	
ro   rÑ  Úmax_cache_lengthc                 óh  — d| j                   j                  j                  «       v }|sdnd}|j                  |«      }|�7|j                  �t        d|› d�«      ‚t        |t        «      rt        d«      ‚y|j                  du ry| j                  «       s0|j                  �#t        j                  d	|j                  › d
�«       y|t        j                  t        j                  fv r6|j                  �#t        j                  d|j                  › d�«       d|_        i }	t        d„ t!        | j"                  j%                  d¬«      dg «      xs g D «       «      }
|j                  dk7  s|
r| j"                  j%                  d¬«      |	d<   |j                  dk(  rd|	d<   |j                  t&        v r€|j                  t(        v r*t        j                  d|j                  › dt*        › d�«       | j-                  |j                  t/        |j0                  |j2                  «      |z  ||¬«      ||<   n·|j                  dk(  rš| j"                  j4                  s| j                  «       st        d«      ‚|j6                  �|j6                  ni }|j9                  d| j"                  j%                  d¬«      «       |j;                  dd«      }t=        dd|i|¤Ž||<   nt?        di |	¤Ž||<   | j"                  j4                  r5||v r0t        ||   t@        «      stA        ||   t?        di |	¤Ž«      ||<   yyyy)zÐ
        Prepares the cache for generation (if applicable), given `generate`'s parameterization. If a cache is
        instantiated, writes it to `model_kwargs`, under the name expected by the model.
        ÚmambarV   rW   NzMPassing both `cache_implementation` (used to initialize certain caches) and `zB` (a Cache object) is unsupported. Please use only one of the two.z]Passing a tuple of `past_key_values` is not supported anymore. Please use a `Cache` instance.FzNThis model does not support `Cache` instances. `cache_implementation` (set to z) will be ignored.zRAn assistant model is provided, using a dynamic cache instead of a cache of type='z'.Údynamic_fullc              3   ó$   K  — | ]  }|d v –— Œ
 y­w))r  ÚconvÚlinear_attentionNrn   )r(  rá  s     rp   r+  z@GenerationMixin._prepare_cache_for_generation.<locals>.<genexpr>E  s   è ø€ ò "
àð Ð6Ô6ñ"
ùr\  Tr  Úlayer_typesrÃ   rÿ  r  zUsing `cache_implementation='zE' is deprecated and will be removed in v5.13. Please only use one of z9, and the layer structure will be inferred automatically.)rø  rß   rý  rî   Ú	quantizedz�This model does not support the quantized cache. If you want your model to support quantized cache, please open an issue and tag @zucchini-nlp.ÚbackendÚquantorn   )!rô   re   r<  rÌ   rø  rò   rÍ   rl   r$  r  r—   rÙ   r-   rÍ  ÚCONTRASTIVE_SEARCHr  rÐ   rÃ   re  r)   r*   r+   r  rº  r‹  rÎ  rÇ   Úcache_configÚ
setdefaultrË   r   r   r   )r�   r’   rî   rÑ  rß   r  Úis_linear_attn_cacherQ  Úuser_defined_cacheÚdynamic_cache_kwargsÚis_linear_attentionr'  r$  s                rp   Ú_prepare_cache_for_generationz-GenerationMixin._prepare_cache_for_generation	  s„  € ð  '¨$¯.©.×*AÑ*A×*GÑ*GÓ*IÐIÐÙ.BÑ&Èˆ
ð *×-Ñ-¨jÓ9ÐØÐ)Ø ×5Ñ5ÐAÜ ØcÐdnÐcoð pTð Tóð ô Ð,¬eÔ4Ü Øsóð ð ð ×&Ñ&¨%Ñ/Øð ×3Ñ3Ô5Ø ×5Ñ5ÐAÜ×#Ñ#ØdØ(×=Ñ=Ð>Ð>PðRôð ð œ~×AÑAÄ>×CdÑCdÐeÑeØ ×5Ñ5ÐAÜ×#Ñ#ðØ)×>Ñ>Ð?¸rðCôð 6DÐÔ2à!Ðä!ñ "
ä˜dŸk™k×9Ñ9À$Ð9ÓGÈÐXZÓ[ÒaÐ_aô"
ó 
Ðð ×1Ñ1°^ÒCÑGZØ-1¯[©[×-HÑ-HÐQUÐ-HÓ-VÐ  Ñ*à×1Ñ1°[Ò@Ø15Ð  Ñ.à×1Ñ1Ô5UÑUØ ×5Ñ5Ô9`Ñ`Ü×#Ñ#Ø3Ð4E×4ZÑ4ZÐ3[ð \LÜLhÐKið jNðNôð
 (,×'AÑ'AØ%6×%KÑ%KÜÐ0×:Ñ:Ð<M×<bÑ<bÓcÐfpÑpØ.Ø)ð	 (Bó (ˆL˜Ò$ð ×3Ñ3°{ÒBØ�{‰{×-Ò-°T×5YÑ5YÔ5[Ü ðIóð ð
 >O×=[Ñ=[Ð=gÐ,×9Ò9ÐmoˆLØ×#Ñ# H¨d¯k©k×.IÑ.IÐRVÐ.IÓ.WÔXØ"×&Ñ& y°(Ó;ˆGÜ'5Ñ'V¸gÐ'VÈÑ'VˆL˜Ò$ô (4Ñ'KÐ6JÑ'KˆL˜Ñ$ð �K‰K×*Ò*Ø˜lÑ*Ü˜|¨JÑ7Ô9LÔMä':Ø˜ZÑ(ÜÑ4Ð3Ñ4ó(ˆL˜Ò$ð Nð +ð +ro   c                 ó†   — dt        t        j                  | j                  «      j                  j                  «       «      v S )zË
        Return True if the current model supports the keyword argument `logits_to_keep` in forward()
        to save memory. Checking it in this way allows to avoid using a new model attribute.
        Úlogits_to_keep)rÔ   rÕ   rÖ   r×   rØ   ró   )r�   s    rp   Ú_supports_logits_to_keepz(GenerationMixin._supports_logits_to_keepv  s2   € ð
  ¤3¤w×'8Ñ'8¸¿¹Ó'F×'QÑ'Q×'VÑ'VÓ'XÓ#YÐYÐYro   Úkwargs_has_attention_maskc                 ó&  ‡ — dˆ fd„	} ||j                   |¬«      } ||j                  |¬«      } ||j                  |¬«      } ||j                  |¬«      }‰ j                  j
                  r|�|n|}|� |j                  dk(  r|j                  d«      }|€9|�7|�|st        j                  d«       |d   }t        j                  d|› d�«       ‰ j                  j
                  r|€t        d«      ‚|�=t        j                  ||«      j                  «       r|�|st        j                  d	«       |�At        j                  |«      s|dk  j                  «       rt        j                  d
|› d�«       ||_        ||_        ||_        ||_        y)aó  
        Prepares the special tokens for generation, overwriting the generation config with their processed versions
        converted to tensor.

        Note that `generation_config` is changed in place and stops being serializable after this method is called.
        That is no problem if called within `generate` (`generation_config` is a local copy that doesn't leave the
        function). However, if called outside `generate`, consider creating a copy of `generation_config` first.
        Nc                 óÎ   •— | €| S |�|n‰j                   }t        | t        j                  «      r| j	                  |«      S t        j
                  | |t        j                  ¬«      S )N)rÂ   rÁ   )rÂ   rÍ   ri   r  r²  Útensorr   )r‡   rÂ   r�   s     €rp   Ú_tensor_or_nonez@GenerationMixin._prepare_special_tokens.<locals>._tensor_or_none�  sQ   ø€ Øˆ}Ø�à%Ð1‘V°t·{±{ˆFÜ˜%¤§¡Ô.Ø—x‘x Ó'Ð'Ü—<‘< ¨f¼E¿J¹JÔGÐGro   rÆ   r   z²The attention mask and the pad token id were not set. As a consequence, you may observe unexpected behavior. Please pass your input's `attention_mask` to obtain reliable results.z)Setting `pad_token_id` to `eos_token_id`:z for open-end generation.z\`decoder_start_token_id` or `bos_token_id` has to be defined for encoder-decoder generation.zäThe attention mask is not set and cannot be inferred from input because pad token is same as eos token. As a consequence, you may observe unexpected behavior. Please pass your input's `attention_mask` to obtain reliable results.z;`eos_token_id` should consist of positive integers, but is zq. Your generation will not stop until the maximum length is reached. Depending on other flags, it may even crash.r&  )rí   r  r  r5  rÃ   rÇ   rÏ   r  r—   rë  rò   ri   r  r  rÙ   Úis_floating_pointÚ_bos_token_tensorr  r  Ú_decoder_start_token_tensor)	r�   r’   r1  rÂ   r5  Úbos_token_tensorÚeos_token_tensorÚpad_token_tensorÚdecoder_start_token_tensors	   `        rp   Ú_prepare_special_tokensz'GenerationMixin._prepare_special_tokens}  sÇ  ø€ õ 	Hñ +Ð+<×+IÑ+IÐRXÔYÐÙ*Ð+<×+IÑ+IÐRXÔYÐÙ*Ð+<×+IÑ+IÐRXÔYÐÙ%4Ð5F×5]Ñ5]ÐflÔ%mÐ"ð �;‰;×)Ò)à.HÐ.TÑ*ÐZjð 'ð
 Ð'Ð,<×,AÑ,AÀQÒ,FØ/×9Ñ9¸!Ó<Ðð Ð#Ð(8Ð(DØ(Ð4Ñ=VÜ—‘ðqôð  0°Ñ2ÐÜ�N‰NÐFÐGWÐFXÐXqÐrÔsð �;‰;×)Ò)Ð.HÐ.PÜØnóð ð Ð'¬E¯J©JÐ7GÐIYÓ,Z×,^Ñ,^Ô,`Ø(Ð4Ñ=VÜ×#Ñ#ðCôð
 Ð'Ü×#Ñ#Ð$4Ô5Ð:JÈQÑ:N×9SÑ9SÔ9Uä�N‰NØMÐN^ÐM_ð `rð rôð /?ÐÔ+Ø.>ÐÔ+Ø.>ÐÔ+Ø8RÐÕ5ro   c                 óf  — |j                   ry|j                  d|j                  d«      «      }| j                  j                  dv xs/ t	        |j
                  duxr |j
                  j                  «      }|duxr |j                  xr t        |«      t        u}|xr |}t        | dd«      �|| j                  j                  z  }t        | d«      rGt        | j                  j                  «       «      }d|v xr t        |«      d	kD  }|| z  }d
|v }	||	 z  }|j
                  �|st         j#                  d«       |rpdt$        j&                  d<   t)        | j*                  «      rH|j
                  �<|j
                  j,                  r&t         j#                  d«       d|j
                  _        |S )zp
        Determines whether to trigger auto-compilation of the model's forward pass at generation time.
        FrV   rW   )ÚcudaÚxpuÚneuronNÚhf_quantizerr  Úcpur   ÚdiskzsYou have set `compile_config`, but we are unable to meet the criteria for compilation. Compilation will be skipped.Ú0ÚTOKENIZERS_PARALLELISMz¶When using Flash Attention and a static cache, you cannot use the option `CompileConfig(fullgraph=True)` as FA introduces graph breaks. We overrode the option with `fullgraph=False`.)Údisable_compilerÌ   rÂ   r  ÚboolÚcompile_configÚ_compile_all_devicesrÎ   r   rÐ   rB  r™   rÔ   r  r  r  r—   rÙ   r¬   Úenvironr   rÃ   Ú	fullgraph)
r�   rî   r’   r  Úvalid_hardwareÚusing_compilable_cacheÚcan_compileÚall_model_devicesÚhas_cpu_offloadÚhas_disk_offloads
             rp   Ú_valid_auto_compile_criteriaz,GenerationMixin._valid_auto_compile_criteriaÌ  sÄ  € ð ×,Ò,Øà× Ñ Ð!2°L×4DÑ4DÀ^Ó4TÓUˆð Ÿ™×)Ñ)Ð-FÐFò 
Ì$Ø×,Ñ,°DÐ8ÒrÐ=N×=]Ñ=]×=rÑ=róK
ˆð
 "'¨dÐ!2Ò!o°u×7KÑ7KÒ!oÔPTÐUZÓP[ÔcoÐPoÐØ$Ò?Ð)?ˆô �4˜¨Ó.Ð:Ø˜4×,Ñ,×;Ñ;Ñ;ˆKä�4˜Ô)Ü # D×$6Ñ$6×$=Ñ$=Ó$?Ó @Ðà#Ð'8Ð8ÒW¼SÐARÓ=SÐVWÑ=WˆOØ˜Ð.Ñ.ˆKð  &Ð):Ð:ÐØÐ/Ð/Ñ/ˆKð ×+Ñ+Ð7ÁÜ×Ñð#ôñ
 à36ŒB�J‰JÐ/Ñ0ô ,¨D¯K©KÔ8à$×3Ñ3Ð?ÐDU×DdÑDd×DnÒDnÜ×'Ñ'ðeôð BGÐ%×4Ñ4Ô>àÐro   c              #   ó˜  K  — | j                   j                  }|dk(  r?| j                  j                  dk7  r&t        j                  d«       | j                  d«       	 d –— |dk(  r,| j                  j                  dk7  r| j                  |«       y y y # |dk(  r,| j                  j                  dk7  r| j                  |«       w w w xY w­w)NÚ
grouped_mmrC  zïWe will be switching to 'batched_mm' for the decoding stage as it is much more performant than 'grouped_mm' on smaller inputs. If you experience any issues with this, please open an issue on the Hugging Face Transformers GitHub repository.Ú
batched_mm)rÃ   Ú_experts_implementationrÂ   r  r—   Ú	info_onceÚset_experts_implementation)r�   Úoriginal_experts_implementations     rp   Ú_optimize_model_for_decodez*GenerationMixin._optimize_model_for_decode  sÆ   è ø€ à*.¯+©+×*MÑ*MÐ'ð +¨lÒ:¸t¿{¹{×?OÑ?OÐSXÒ?XÜ×ÑðCôð ×+Ñ+¨LÔ9ð	QÛà.°,Ò>À4Ç;Á;×CSÑCSÐW\ÒC\Ø×/Ñ/Ð0OÕPð D]Ð>øÐ.°,Ò>À4Ç;Á;×CSÑCSÐW\ÒC\Ø×/Ñ/Ð0OÕPð D]Ð>üs   ‚AC
ÁB Á"2C
Â3CÃC
r¢   c                 ó(  — |€dt         |   x}vryt        j                  |j                  j	                  dd«      j                  «       › d|› d|› d�«       |s9t        |j                  j	                  dd«      j                  «       › d|› d	�«      ‚|S )
zP
        Returns the Hub repo for a deprecated generation mode, if any.
        Nú/Ú_ú z6 was moved to a `custom_generate` repo: https://hf.co/zC. To prevent loss of backward compatibility, add `custom_generate='z*'` to your `generate` call before v4.62.0.zY requires `trust_remote_code=True` in your `generate` call, since it loads https://hf.co/rÆ  )ÚGENERATION_MODES_MAPPINGr—   rÙ   ÚnameÚreplaceÚtitlerò   )r�   rÑ  r�   r¢   Úrepos        rp   Ú_get_deprecated_gen_repoz(GenerationMixin._get_deprecated_gen_repo  s½   € ð Ð&¨#Ô>VÐWfÑ>gÐ6g°dÑ*hØä×ÑØ×#Ñ#×+Ñ+¨C°Ó5×;Ñ;Ó=Ð>Ð>tÐuyÐtzð {PØPTÈvð V6ð6ô	
ñ
 !ÜØ"×'Ñ'×/Ñ/°°SÓ9×?Ñ?ÓAÐBð C0Ø04¨v°Qð8óð ð ˆro   c                 ó¾  — |j                  dd«      |j                  dd«      ||dœ}t        j                  «       r(t        j                  «       rt        j                  «       nd}|€t        «       xs t        | «      xr |dkD  n||d<   |j                  «       D ��	ci c]  \  }}	|	€Œ	||	“Œ }}}	t        |t        «      r‘t        j                  t        j                  «      j                  j                  «       }
t        j                  |«      j                  j                  «       }||
z
  }|D �ci c]  }||v sŒ||j                  |«      “Œ }}|S c c}	}w c c}w )zn
        Extracts and returns the generation mode related keyword arguments from the provided kwargs.
        rŸ  NrX  )rŸ  rX  rV  rÅ  r   Úsynced_gpus)rË   ÚdistÚis_availableÚis_initializedÚget_world_sizer   r   rÒ   rÍ   r   rÕ   rÖ   r   r[   rØ   ró   )r�   r¢   r    rg  rV  rÅ  rÒ  Ú
world_sizeÚkr[  Úusual_mode_kwargsÚcustom_generate_kwargsÚnew_custom_keyss                rp   Ú_extract_generation_mode_kwargsz/GenerationMixin._extract_generation_mode_kwargs1  sQ  € ð  Ÿ™ K°Ó6Ø#)§:¡:Ð.CÀTÓ#JØ.Ø ñ	"
Ðô /3×.?Ñ.?Ô.AÄd×FYÑFYÔF[”T×(Ñ(Ô*Ðabˆ
ð Ð"ô (Ó)ÒIÔ-CÀDÓ-IÒ]ÈzÐ\]Ê~àð 	˜}Ñ-ð
 4J×3OÑ3OÓ3Q×!c©4¨1¨aÐUVÑUb ! Q¡$Ð!cÐÑ!cô �o¤xÔ0Ü '× 1Ñ 1´/×2IÑ2IÓ J× UÑ U× ZÑ ZÓ \ÐÜ%,×%6Ñ%6°Ó%G×%RÑ%R×%WÑ%WÓ%YÐ"Ø4Ð7HÑHˆOØ@OÖ%_¸1ÐSTÐX^ÒS^ a¨¯©°A«Ñ&6Ð%_Ð"Ð%_Ø%Ð%ùó "dùò &`s   Â
EÂ)EÄ2	EÄ<Erg  rÅ  rS   c                 ó^  — |j                  dd«      }|�tt        |t        «      rdh d£}t        «       j	                  «       D ��ci c]  \  }}||vsŒ||“Œ }}}|j                  |«        | j                  |fd|i|¤Ž} |d5d| i|¤ŽS |j                  d«      dk(  �rt        j                  d«       |�|n|j                  d«      }|€t        d	«      ‚|j                  «       d
k(  r |j                  d«      j                  «       }n@|j                  «       dk(  r|j                  «       }nt        d|j                  «       ›�«      ‚|�t        d|›�«      ‚|�t        d|›�«      ‚|�t        d|›�«      ‚|�t        d|›�«      ‚|	�t        d|	›�«      ‚|
�t        d|
›�«      ‚|�t        j                  d|›�«       |j                  dd
«      }|d
kD  rt        j                  d|›d�«        | j                  d5| | j                   |fi |¤Žd   dœ|¤Ž}t#        t%        |«      «      D �cg c]'  }|d|› �   j&                  |d|› �   j(                  z   ‘Œ) }}t+        j,                  |t*        j.                  | j0                  ¬«      }|j                  d«      }|S | j3                  |||||«      }|j                  d«      du xr. |du xs |j4                  du xr | j6                  j4                  du }|j                  d«      du xr. |du xs |j8                  du xr | j6                  j8                  du } | j                   |fi |¤Ž\  }}|j;                  |«      }| j=                  |||«      }t        |t>        «      r|}n|€tA        tC        | «      tD        |   «      }| jG                  |jI                  «       «       | jK                  |||«       |�#tM        jN                  | f|||||||	|
||dœ
|¤|¤ŽS |�|n	tQ        «       }|�|n	tS        «       }dtU        tW        jX                  | jZ                  «      j\                  j_                  «       «      v }|j                  dd«      du} | ja                  ||jb                  |«      \  }!}"}dtW        jX                  «      j\                  j_                  «       v r|!|d<   |!jd                  d   }#|!j0                  }$| jg                  || |$¬ «       | jh                  jj                  sÇ|jl                  �»|#d
kD  r¶t%        |!jd                  «      dk(  rž|j                  dd«      }%|%�G|%jd                  |!jd                  k(  r.t+        jn                  |%dd…d!f   dk(  «      jq                  «       }&n,t+        jr                  |!dd…d!f   |jl                  k(  «      dkD  }&|&rt        j                  d"«       | jh                  jj                  s|"d#k(  rd$|_:        | s/| jh                  jj                  s|r| jw                  |!||«      |d<   n-| r+|"dk(  r&t%        |d   jd                  «      dkD  rt        d%«      ‚|j                  d&d«      du}'d&tU        tW        jX                  | jZ                  «      j\                  j_                  «       «      v }(|'s-|(r+| jh                  jj                  s| jy                  |!|«      |d&<   | jh                  jj                  rd'|vr| j{                  |!||"|«      }| jh                  jj                  r.| j}                  |#|"||j~                  |!j0                  ¬(«      \  })}n|"dk(  r|!n|j                  d«      }) | j€                  d5|)tƒ        |j„                  |j†                  «      | jh                  jj                  d)œ|¤Ž\  })}|jˆ                  r!| j‹                  |)|j                  d*«      «      })|�|j�                  |)j�                  «       «       |)jd                  d
   }*| j‘                  ||||"|!|*¬+«      }| j“                  «       r	d,|vrd
|d,<   | j•                  ||*|«       |j4                  d
z
  }+|!jd                  d
   |*k7  r-|"d#k(  r(| jh                  jj                  s|+|!jd                  d
   z  }+| j—                  ||||#|+«       | j0                  jB                  |)j0                  jB                  k7  r`t™        jš                  d-|)j0                  jB                  › d.| j0                  jB                  › d/| j0                  jB                  › d0�tœ        «       | jŸ                  ||*|!|||!j0                  ||	|
¬1«	      },| j¡                  |||j                  d*«      ¬2«      }-|jt                  |d3<    || |)f|,|-|d4œ|¤|¤Ž}.|.S c c}}w c c}w )6aV  

        Generates sequences of token ids for models with a language modeling head.

        <Tip warning={true}>

        Most generation-controlling parameters are set in `generation_config` which, if not passed, will be set to the
        model's default generation configuration. You can override any `generation_config` by passing the corresponding
        parameters to generate(), e.g. `.generate(inputs, num_beams=4, do_sample=True)`.

        For an overview of generation strategies and code examples, check out the [following
        guide](../generation_strategies).

        </Tip>

        Parameters:
            inputs (`torch.Tensor` of varying shape depending on the modality, *optional*):
                The sequence used as a prompt for the generation or as model inputs to the encoder. If `None` the
                method initializes it with `bos_token_id` and a batch size of 1. For decoder-only models `inputs`
                should be in the format of `input_ids`. For encoder-decoder models *inputs* can represent any of
                `input_ids`, `input_values`, `input_features`, or `pixel_values`.
            generation_config ([`~generation.GenerationConfig`], *optional*):
                The generation configuration to be used as base parametrization for the generation call. `**kwargs`
                passed to generate matching the attributes of `generation_config` will override them. If
                `generation_config` is not provided, the default will be used, which has the following loading
                priority: 1) from the `generation_config.json` model file, if it exists; 2) from the model
                configuration. Please note that unspecified parameters will inherit [`~generation.GenerationConfig`]'s
                default values, whose documentation should be checked to parameterize generation.
            logits_processor (`LogitsProcessorList`, *optional*):
                Custom logits processors that complement the default logits processors built from arguments and
                generation config. If a logit processor is passed that is already created with the arguments or a
                generation config an error is thrown. This feature is intended for advanced users.
            stopping_criteria (`StoppingCriteriaList`, *optional*):
                Custom stopping criteria that complements the default stopping criteria built from arguments and a
                generation config. If a stopping criteria is passed that is already created with the arguments or a
                generation config an error is thrown. If your stopping criteria depends on the `scores` input, make
                sure you pass `return_dict_in_generate=True, output_scores=True` to `generate`. This feature is
                intended for advanced users.
            prefix_allowed_tokens_fn (`Callable[[int, torch.Tensor], list[int]]`, *optional*):
                If provided, this function constraints the beam search to allowed tokens only at each step. If not
                provided no constraint is applied. This function takes 2 arguments: the batch ID `batch_id` and
                `input_ids`. It has to return a list with the allowed tokens for the next generation step conditioned
                on the batch ID `batch_id` and the previously generated tokens `inputs_ids`. This argument is useful
                for constrained generation conditioned on the prefix, as described in [Autoregressive Entity
                Retrieval](https://huggingface.co/papers/2010.00904).
            synced_gpus (`bool`, *optional*):
                Whether to continue running the while loop until max_length. Unless overridden, this flag will be set
                to `True` if using `FullyShardedDataParallel` or DeepSpeed ZeRO Stage 3 with multiple GPUs to avoid
                deadlocking if one GPU finishes generating before other GPUs. Otherwise, defaults to `False`.
            assistant_model (`PreTrainedModel`, *optional*):
                An assistant model that can be used to accelerate generation. The assistant model must have the exact
                same tokenizer. The acceleration is achieved when forecasting candidate tokens with the assistant model
                is much faster than running generation with the model you're calling generate from. As such, the
                assistant model should be much smaller.
            streamer (`BaseStreamer`, *optional*):
                Streamer object that will be used to stream the generated sequences. Generated tokens are passed
                through `streamer.put(token_ids)` and the streamer is responsible for any further processing.
            negative_prompt_ids (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
                The negative prompt needed for some processors such as CFG. The batch size must match the input batch
                size. This is an experimental feature, subject to breaking API changes in future versions.
            negative_prompt_attention_mask (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
                Attention_mask for `negative_prompt_ids`.
            custom_generate (`str` or `Callable`, *optional*):
                One of the following:
                - `str` (Hugging Face Hub repository name): runs the custom `generate` function defined at
                  `custom_generate/generate.py` in that repository instead of the standard `generate` method. The
                  repository fully replaces the generation logic, and the return type may differ.
                - `str` (local repository path): same as above but from a local path, `trust_remote_code` not required.
                - `Callable`: `generate` will perform the usual input preparation steps, then call the provided callable to
                  run the decoding loop.
                For more information, see [the docs](../../generation_strategies#custom-generation-methods).
            kwargs (`dict[str, Any]`, *optional*):
                Ad hoc parametrization of `generation_config` and/or additional model-specific kwargs that will be
                forwarded to the `forward` function of the model. If the model is an encoder-decoder model, encoder
                specific kwargs should not be prefixed and decoder specific kwargs should be prefixed with *decoder_*.

        Return:
            [`~utils.ModelOutput`] or `torch.LongTensor`: A [`~utils.ModelOutput`] (if `return_dict_in_generate=True`
            or when `config.return_dict_in_generate=True`) or a `torch.LongTensor`.

                If the model is *not* an encoder-decoder model (`model.config.is_encoder_decoder=False`), the possible
                [`~utils.ModelOutput`] types are:

                    - [`~generation.GenerateDecoderOnlyOutput`],
                    - [`~generation.GenerateBeamDecoderOnlyOutput`]

                If the model is an encoder-decoder model (`model.config.is_encoder_decoder=True`), the possible
                [`~utils.ModelOutput`] types are:

                    - [`~generation.GenerateEncoderDecoderOutput`],
                    - [`~generation.GenerateBeamEncoderDecoderOutput`]
        r�   N>   r�   r    r¢   r�   Úglobal_keys_to_excluder�   rø  Úpagedz�Detected cache_implementation=paged: switching to continuous batching. You should consider using generate_batch directly instead.r²   z7inputs or input_ids must be provided for CB generation.r   r   r   z5inputs must be a 1D or 2D tensor, got inputs.dim() = zTstopping_criteria is not supported for continuous batching. Got stopping_criteria = zbprefix_allowed_tokens_fn is not supported for continuous batching. Got prefix_allowed_tokens_fn = zPassistant_model is not supported for continuous batching. Got assistant_model = zCstreaming is not supported for continuous batching. Got streamer = zXnegative_prompt_ids is not supported for continuous batching. Got negative_prompt_ids = znnegative_prompt_attention_mask is not supported for continuous batching. Got negative_prompt_attention_mask = zFsynced_gpus is not ignored for continuous batching. Got synced_gpus = r‹  zHnum_beams is not supported for continuous batching yet. Got num_beams = z. )rì   r’   Úreq_rÀ   r_  r‰  )
rì   r’   rU  rž  rn  rV  ro  rp  r¢   r�   r´   r  rÆ   r¾   z¬A decoder-only architecture is being used, but right-padding was detected! For correct generation results, please set `padding_side='left'` when initializing the tokenizer.rµ   Tz1`attention_mask` passed to `generate` must be 2D.r»   rû   )rß   râ   rî   r5  rÂ   )r²   rC  rÇ   rŸ  )r’   ræ  rí  râ   r  rå  r/  z~You are calling .generate() with the `input_ids` being on a device type different than your model's device. `input_ids` is on z, whereas the model is on z¶. You may experience unexpected behaviors or slower generation. Please make sure that you have put `input_ids` to the correct device by calling for example input_ids = input_ids.to('z ') before running `.generate()`.)	r’   rl  rm  rn  rU  rÂ   rî   ro  rp  )r’   rž  rŸ  r$  )rU  rž  r’   rn   )QrË   rÍ   ÚstrÚlocalsrÒ   r÷  rŽ   rÌ   r—   rë  rò   r:  r  ÚtolistÚNotImplementedErrorÚgenerate_batchrü  Úranger  Ú
prompt_idsÚgenerated_tokensri   r4  r   rÂ   rq  r_  r’   r‰  Úget_generation_modere  r   rÐ   r  r`  râ  rô  rÖ  r   rœ   r8   rN   rÔ   rÕ   rÖ   r×   rØ   ró   rù   rí   rÊ   r=  rÃ   rÇ   r  r  r?  r¹  r$  r  r  r4  rB  r8  rI  rº  r‹  rÎ  Útoken_healingÚheal_tokensÚputrC  rî  r0  ré  r-  rƒ  r„  r…  r�  r§  )/r�   rì   r’   rU  rž  rn  rg  rV  rÅ  ro  rp  r¢   r    r�   rs  rè   ré   Úgenerate_argumentsr±   r‹  rJ  Úir`   Úsequences_as_tensorrÒ  ræ  rí  rî   rÑ  Údeprecated_mode_repoÚdecoding_methodÚaccepts_attention_maskr1  r  râ   rß   rÂ   r´   Úhas_right_paddingÚkwargs_has_position_idsÚaccepts_position_idsr²   rå  r  Úprepared_logits_processorÚprepared_stopping_criteriaÚresults/                                                  rp   rœ   zGenerationMixin.generateR  sé
  € ðZ #ŸJ™JÐ':¸DÓAÐàÐ&¬:°oÄsÔ+Kò&Ð"ô @F»x¿~¹~Ó?O×!u±°°eÐSVÐ^tÒSt # u¡*Ð!uÐÑ!uØ×%Ñ% fÔ-à'@ t×'@Ñ'@Øñ(Ø3Dð(ØHNñ(Ð$ñ ,ÑM°$ÐMÐ:LÑMÐMð �:‰:Ð,Ó-°Ó8Ü�N‰Nð3ôð  &Ð1‘V°v·z±zÀ+Ó7NˆFØˆ~Ü Ð!ZÓ[Ð[à�z‰z‹|˜qÒ Ø×)Ñ)¨!Ó,×3Ñ3Ó5‘Ø—‘“ Ò"ØŸ™›‘ä Ð#YÈ&Ï*É*Ë,ÐIZÐ![Ó\Ð\ð !Ð,Ü)ØkÐWhÐVlÐmóð ð (Ð3Ü)ØyÐ^vÐ]zÐ{óð ð Ð*Ü)ØgÐUdÐThÐióð ð Ð#Ü)Ð,pÐemÐdqÐ*rÓsÐsØ"Ð.Ü)ØoÐYlÐXpÐqóð ð .Ð9Ü)ð Fð  eCð  dGð  Hóð ð
 Ð&Ü—‘Ð!hÐZeÐYiÐjÔkØŸ
™
 ;°Ó2ˆIØ˜1Š}Ü—‘Ð!jÐ^gÐ]kÐkmÐnÔoð *�d×)Ñ)ð ØØ"A $×"AÑ"AÐBSÑ"^ÐW]Ñ"^Ð_`Ñ"añð ñˆGô `eÔehÐipÓeqÓ_röØZ[�˜$˜q˜c˜
Ñ#×.Ñ.°¸4À¸s¸Ñ1D×1UÑ1UÓUðˆIð ô
 #(§,¡,¨yÄÇ
Á
ÐSW×S^ÑS^Ô"_ÐØ"5×"?Ñ"?ÀÓ"BÐØ&Ð&ð "&×!EÑ!EØØØØØó"
Ðð �J‰J�|Ó$¨Ð,ò :Ø" dÐ*ÒRÐ.?×.JÑ.JÈdÐ.Rò:à×&Ñ&×1Ñ1°TÐ9ð 	ð �J‰J�|Ó$¨Ð,ò :Ø" dÐ*ÒRÐ.?×.JÑ.JÈdÐ.Rò:à×&Ñ&×1Ñ1°TÐ9ð 	ð
 +J¨$×*IÑ*IÐJ[Ñ*fÐ_eÑ*fÑ'Ð˜<à+×?Ñ?ÀÓPˆØ#×<Ñ<¸_ÐN_ÐapÓqÐä�o¤xÔ0Ø-‰OØ!Ð)ä%¤d¨4£jÔ2JÈ?Ñ2[Ó\ˆOà×#Ñ# L×$5Ñ$5Ó$7Ô8Ø×&Ñ& Ð8IÐKaÔbð  Ð+Ü"×+Ñ+ØðàØ"3Ø!1Ø"3Ø)AØ /Ø$7Ø/MØ 4Ø"3ñð )ðð ñð ð" 0@Ð/KÑ+ÔQdÓQfÐØ1BÐ1NÑ-ÔThÓTjÐà!1´S¼×9JÑ9JÈ4Ï<É<Ó9X×9cÑ9c×9hÑ9hÓ9jÓ5kÐ!kÐØ$0×$4Ñ$4Ð5EÀtÓ$LÐTXÐ$XÐ!ð 9=×8RÑ8RØÐ%×2Ñ2°Ló9
Ñ5ˆÐ'¨ð œg×/Ñ/°Ó@×KÑK×PÑPÓRÑRØ6CÐ" ?Ñ3Ø"×(Ñ(¨Ñ+ˆ
à×%Ñ%ˆØ×$Ñ$Ð%6Ð8QÐZ`Ð$Ôað �{‰{×-Ò-ð !×2Ñ2Ð>À:ÐPQÂ>ÔVYÐZg×ZmÑZmÓVnÐrsÒVsð ".×!1Ñ!1Ð2BÀDÓ!I�Ø!Ð-°.×2FÑ2FÈ-×J]ÑJ]Ò2]ä(-¯	©	°.ÂÀBÀÑ2GÈ1Ñ2LÓ(M×(RÑ(RÓ(TÑ%ô ).¯	©	°-ÂÀ2ÀÑ2FÐJ[×JmÑJmÑ2mÓ(nÐqrÑ(rÐ%Ù$Ü—N‘Nðpôð �{‰{×-Ò-Ð2BÀoÒ2UØ*.ÐÔ'á(°·±×1OÒ1OÑTjØ-1×-XÑ-XØÐ0°,ó.ˆLÐ)Ò*ñ 'à ;Ò.´3°|ÐDTÑ7U×7[Ñ7[Ó3\Ð_`Ò3`Ü Ð!TÓUÐUà".×"2Ñ"2°>À4Ó"HÐPTÐ"TÐØ-´´W×5FÑ5FÀtÇ|Á|Ó5T×5_Ñ5_×5dÑ5dÓ5fÓ1gÐgÐÙ&Ñ+?ÈÏÉ×HfÒHfØ+/×+TÑ+TÐUbÐdpÓ+qˆL˜Ñ(à�;‰;×)Ò)Ð.?À|Ñ.Sà×NÑNØ˜|Ð-=Ð?PóˆLð
 �;‰;×)Ò)Ø&*×&TÑ&TØ%Ø!1Ø)Ø'8×'TÑ'TØ$×+Ñ+ð 'Uó 'Ñ#ˆI‘|ð *:¸[Ò)H™Èl×N^ÑN^Ð_jÓNkˆIð #E $×"DÑ"Dð #
ØÜÐ-×7Ñ7Ð9J×9_Ñ9_Ó`Ø#Ÿ{™{×=Ñ=ñ#
ð ñ	#
Ñˆ	�<ð ×*Ò*Ø×(Ñ(¨Ð4J×4NÑ4NÈ{Ó4[Ó\ˆIàÐØ�L‰L˜Ÿ™›Ô)ð %Ÿ?™?¨1Ñ-ÐØ ×:Ñ:Ø/Ø#9Ø#9Ø-Ø'Ø-ð ;ó 
Ðð ×(Ñ(Ô*Ð/?À|Ñ/SØ-.ˆLÐ)Ñ*à×'Ñ'Ð(9Ð;KÐMcÔdð -×7Ñ7¸!Ñ;Ðà×Ñ Ñ"Ð&6Ò6Ø  OÒ3Ø—K‘K×2Ò2à × 3Ñ 3°AÑ 6Ñ6ÐØ×*Ñ*Ø˜|¨_¸jÐJZô	
ð �;‰;×Ñ˜y×/Ñ/×4Ñ4Ò4Ü�M‰Mð@Ø@I×@PÑ@P×@UÑ@UÐ?Vð WØŸ+™+×*Ñ*Ð+ð ,TàTX×T_ÑT_×TdÑTdÐSeð f*ð	*ô ôð %)×$>Ñ$>Ø/Ø!1Ø+Ø%=Ø-Ø ×'Ñ'Ø%Ø 3Ø+Ið %?ó 
%
Ð!ð &*×%@Ñ%@Ø/Ø/Ø,×0Ñ0°Ó=ð &Aó &
Ð"ð %6×$?Ñ$?ˆ�[Ñ!ñ !ØØð
ð 7Ø8Ø/ñ
ð %ð
ð ñ
ˆð ˆùóS	 "vùò@s   Áf$Áf$È1,f*Úthis_peer_finishedc                 óÌ   — |r_t        j                  |rdnd|¬«      }t        j                  |t        j                  j
                  ¬«       |j                  «       dk(  ryy|ryy)z¿
        Returns whether there are still unfinished sequences in the device. The existence of unfinished sequences is
        fed through `this_peer_finished`. ZeRO stage 3-friendly.
        r~  rv  rÆ   )ÚopFT)ri   r4  rh  Ú
all_reduceÚReduceOpÚSUMr?  )r�   rŽ  rg  rÂ   Úthis_peer_finished_flags        rp   Ú_has_unfinished_sequencesz)GenerationMixin._has_unfinished_sequences÷	  s^   € ñ
 ô ',§l¡lÑ:L±3ÐRUÐ^dÔ&eÐ#ä�O‰OÐ3¼¿¹×8IÑ8IÕJà&×+Ñ+Ó-°Ò4Øð ñ  ØØro   c                 ó˜  ‡‡— ‰€t        d«      ‚‰j                  ‰j                  }}t        ‰j	                  «       «      }t        d|¬«      }‰j                  |d¬«      D �cg c]  }|j                  «       ‘Œ }} ‰|dd¬«      j                  j                  |j                  «      }t        j                  ||k(  ||«      }|j                  «       d	k(  r|S |dd…d
f   j                  «       }	‰j                  d«      �0‰j!                  ‰j                  d«      «      d	   Šˆˆfd„|	D «       }
nˆfd„|	D «       }
t#        t%        |	|
«      «      D ]ì  \  }\  }}||   }t        j&                  ||k(  «      j)                  «       rŒ5	 |j+                  |¬«      D �ci c]  }‰j                  |«      fd“Œ }}t-        |«      dk(  rŒu||fxx   dz  cc<   |j/                  |¬«       |dd
 }	 |j                  «       d	k(  rŒ¯t-        |||k7     «      dk(  r||d
<   | j1                  |j3                  d	«      |¬«      ||<   Œî |S c c}w c c}w )a¸  
        Generates sequences of token ids for models with a language modeling head.
        Parameters:
            input_ids (`torch.LongTensor`): The sequence used as a prompt for the generation.
            tokenizer (`PreTrainedTokenizerBase`, *optional*): The tokenizer used to decode the input ids.
        Return:
            `torch.LongTensor` where each sequence has its tail token replaced with its appropriate extension.
        Nzs When generating with token healing, you must pass the model's tokenizer to the `tokenizer` argument of `generate`.r   )rä  r  T)Úskip_special_tokensÚpt)Úreturn_tensorsÚpaddingr   r¾   r_  c              3   ó|   •K  — | ]3  }t        t        ‰j                  |«      «      j                  d ‰«      –— Œ5 y­w)r_  N)r	   rv  Údecoderb  )r(  ÚtÚ	space_tokrŸ  s     €€rp   r+  z.GenerationMixin.heal_tokens.<locals>.<genexpr>4
  s1   øè ø€ ÒbÐTUœœc 9×#3Ñ#3°AÓ#6Ó7×?Ñ?ÀÀY×OÑbùs   ƒ9<c              3   ó\   •K  — | ]#  }t        t        ‰j                  |«      «      –— Œ% y ­wr&  )r	   rv  rœ  )r(  r�  rŸ  s     €rp   r+  z.GenerationMixin.heal_tokens.<locals>.<genexpr>6
  s#   øè ø€ ÒJ¸Aœœc 9×#3Ñ#3°AÓ#6×7ÑJùs   ƒ),)Úprefixg      $@rv  rt  )r’   )rò   rí   r  r   Ú	get_vocabr,   rœ  Ústripr²   r²  rÂ   ri   ÚwhereÚnumelrx  Úconvert_tokens_to_idsÚconvert_ids_to_tokensÚ	enumerateÚzipr>  r?  Ú
extensionsr  r÷  rœ   r  )r�   r²   rŸ  rí   r  Ú
vocab_trier’   r)  ÚpromptsÚtail_idsÚ	tail_toksÚ	batch_idxÚtail_idÚtail_tokÚ	batch_idsÚalt_tokÚseq_biasÚtrimmed_idsrž  s     `               @rp   r€  zGenerationMixin.heal_tokens	
  sˆ  ù€ ð ÐÜð*óð ð
 &/×%;Ñ%;¸Y×=SÑ=S�lˆÜ# I×$7Ñ$7Ó$9Ó:ˆ
Ü,¸AÈLÔYÐð '0×&6Ñ&6°yÐVZÐ&6Ó&[Ö\ �1—7‘7•9Ð\ˆÐ\ÙØØØô
÷ ‰)—B‘B�y×'Ñ'Ó(ð	 	ô —K‘K 	¨\Ñ 9¸<ÈÓSˆ	ð �?‰?Ó Ò!ØÐàšQ ˜UÑ#×*Ñ*Ó,ˆð ×*Ñ*¨3Ó/Ð;Ø!×7Ñ7¸	×8WÑ8WÐX[Ó8\Ó]Ð^_Ñ`ˆIÜbÐYaÔb‰IãJÀÔJˆIä.7¼¸HÀiÓ8PÓ.Qò "	pÑ*ˆIÑ*˜ Ø! )Ñ,ˆIÜ�y‰y˜ lÑ2Ó3×8Ñ8Ô:Øðð
 R\×QfÑQfÐnvÐQfÓQwöØFM�×0Ñ0°Ó9Ð;¸TÑAðˆHð ô �8‹} Ò!Øð �g�ZÓ  CÑ'Ó Ø×$Ñ$°8Ð$Ô<à# C R˜.ˆKðð × Ñ Ó" aÒ'Øô �9˜Y¨,Ñ6Ñ7Ó8¸AÒ=Ø".�˜B‘à#'§=¡=°×1FÑ1FÀqÓ1IÐ]n =Ó#oˆI�iÒ ðE"	pðH Ðùòy ]ùòDs   Á#IÆIc                 óô  ‡— |j                   }|j                  }	|j                  }
|j                  }|j                  }|j
                  }t        d„ |D «       «      }|j                  }|r|rdnd}|r|rdnd}|r|	rdnd}|r|	rdnd}|r|
rdnd}|rF| j                  j                  r0|	r‰d   j                  d«      nd}|
r‰d   j                  d«      nd}|j                  d   }d}t        j                  |t        j                  |j                  ¬	«      }| j!                  ‰|«      r| j#                  |j$                  «      n| j&                  }d}| j)                  ||‰|j*                   ¬
«      }| j-                  |||j                  ¬«      �rR|rC‰d   rdnd} | j.                  |fd|i‰¤Ž}| j1                  «       5   |di |¤ddi¤Ž}ddd«       d}| j3                  ‰| j                  j                  ¬«      Š|r|rŒ“|j4                  dd…ddd…f   j7                  dt        j8                  |j                  ¬«      } |||«      } |r |r|| fz  }|r||fz  }|	rY|| j                  j                  r|j:                  fn|j<                  fz  }| j                  j                  r||j>                  fz  }|
r3|| j                  j                  r|j@                  fn|jB                  fz  }|rHtD        jF                  jI                  | d¬«      }!t        jJ                  |!d¬«      jM                  d«      }"nt        jN                  | d¬«      }"|r|"|z  |d|z
  z  z   }"t        jP                  ||"dd…df   gd¬«      }|�|jS                  |"jU                  «       «       | |||«       z  }|jW                  «       dk(  }~| j-                  |||j                  ¬«      r�ŒR|�|jY                  «        |rrd}#t        ˆfd„tZ        D «       «      rt]        ˆfd„tZ        D «       «      }$‰|$   }#| j                  j                  rt_        |||||||#¬«	      S ta        ||||||#¬«      S |S # 1 sw Y   �ŒŸxY w)aë  
        Generates sequences of token ids for models with a language modeling head using **multinomial sampling** and
        can be used for text-decoder, text-to-text, speech-to-text, and vision-to-text models.

        Parameters:
            input_ids (`torch.LongTensor` of shape `(batch_size, sequence_length)`):
                The sequence used as a prompt for the generation.
            logits_processor (`LogitsProcessorList`):
                An instance of [`LogitsProcessorList`]. List of instances of class derived from [`LogitsProcessor`]
                used to modify the prediction scores of the language modeling head applied at each generation step.
            stopping_criteria (`StoppingCriteriaList`):
                An instance of [`StoppingCriteriaList`]. List of instances of class derived from [`StoppingCriteria`]
                used to tell if the generation loop should stop.
            generation_config ([`~generation.GenerationConfig`]):
                The generation configuration to be used as parametrization of the decoding method.
            synced_gpus (`bool`):
                Whether to continue running the while loop until max_length (needed to avoid deadlocking with
                `FullyShardedDataParallel` and DeepSpeed ZeRO Stage 3).
            streamer (`BaseStreamer`, *optional*):
                Streamer object that will be used to stream the generated sequences. Generated tokens are passed
                through `streamer.put(token_ids)` and the streamer is responsible for any further processing.
            model_kwargs:
                Additional model specific kwargs will be forwarded to the `forward` function of the model. If model is
                an encoder-decoder model the kwargs should include `encoder_outputs`.

        Return:
            [`~generation.GenerateDecoderOnlyOutput`], [`~generation.GenerateEncoderDecoderOutput`] or `torch.LongTensor`:
            A `torch.LongTensor` containing the generated tokens (default behaviour) or a
            [`~generation.GenerateDecoderOnlyOutput`] if `model.config.is_encoder_decoder=False` and
            `return_dict_in_generate=True` or a [`~generation.GenerateEncoderDecoderOutput`] if
            `model.config.is_encoder_decoder=True`.
        c              3   ó4   K  — | ]  }t        |d «      –— Œ y­w)r  N)r™   )r(  r¦  s     rp   r+  z*GenerationMixin._sample.<locals>.<genexpr>�
  s   è ø€ Ò'lÈh¬°¸.×(IÑ'lùs   ‚rn   Nrû   rc   rd   r   FrÀ   ©r¶   rÆ   r$  r   r³   r.  T©rÇ   r¾   ©rô  rÁ   rÂ   r9  ©Únum_samplesc              3   ó&   •K  — | ]  }|‰v –— Œ
 y ­wr&  rn   ©r(  Ú	cache_keyrî   s     €rp   r+  z*GenerationMixin._sample.<locals>.<genexpr>ü
  ó   øè ø€ ÒN°�9 Ô,ÑNùó   ƒc              3   ó,   •K  — | ]  }|‰v sŒ|–— Œ y ­wr&  rn   r½  s     €rp   r+  z*GenerationMixin._sample.<locals>.<genexpr>ý
  ó   øè ø€ Ò i¨yÈyÐ\hÒOh¤Ñ iùó   ƒ	�©	r`   ra   rb   rs   rt   ru   rv   rw   rV   ©r`   ra   rb   rc   rd   rV   )1r  r,  r-  Úoutput_scoresÚoutput_logitsÚreturn_dict_in_generater  rf  rÃ   rÇ   rÌ   rÊ   ri   rÿ   r   rÂ   rS  Úget_compiled_callrI  Ú__call__Ú_prefillr¥  r•  rë   r[  rT  rb   r²  Úfloat32ru   rc   rv   rw   rd   r
   r·  ÚsoftmaxÚmultinomialÚsqueezeÚargmaxr@  r�  rC  rº  ÚendrM  Únextrr   r_   )%r�   r²   rU  rž  r’   rg  rÅ  rî   r  r,  r-  rÆ  rÇ  rÈ  Úhas_eos_stopping_criteriarf  ra   Ú
raw_logitsru   rv   rw   rs   rt   rß   rŽ  Úunfinished_sequencesÚmodel_forwardÚprefill_consumedrJ  r³   rÜ   Únext_token_logitsÚnext_token_scoresÚprobsÚnext_tokensr  r¾  s%          `                             rp   r[   zGenerationMixin._sample^
  sä  ø€ ðV )×:Ñ:ˆØ-×?Ñ?ÐØ0×EÑEÐØ)×7Ñ7ˆØ)×7Ñ7ˆØ"3×"KÑ"KÐÜ$'Ñ'lÐZkÔ'lÓ$lÐ!Ø%×/Ñ/ˆ	ñ 0±M‘ÈˆÙ3¹‘RÈDˆ
Ù$;Ñ@Q™RÐX\ÐÙ"9Ñ>O™2ÐVZÐÙ'>ÑCW¡Ð^bÐñ # t§{¡{×'EÒ'EÙVg Ð.?Ñ!@×!DÑ!DÀ\Ô!RÐmqÐáH\�Ð.Ñ/×3Ñ3°OÔDÐbfð "ð
 —_‘_ QÑ'ˆ
Ø"ÐÜ$Ÿz™z¨*¼E¿J¹JÈy×O_ÑO_Ô`Ðð ×0Ñ0°Ð?PÔQð ×"Ñ"Ð#4×#CÑ#CÔDà—‘ð 	ð !ÐØ—-‘-ØØØØ#4×#AÑ#AÐAð	  ó 
ˆð ×,Ñ,Ð-?ÀÐU^×UeÑUeÐ,ÕfÙØ,8¸Ò,E¡qÈ4Ð$ØA˜t×AÑAØñ Ø4Hð ØLXñ �ð ×4Ñ4Ó6ñ NÙ+ÑM¨lÑMÈÒM�G÷Nà#ÐØ×CÑCØØØ#'§;¡;×#AÑ#Að Dó ˆLñ
 Ñ1Øð !(§¡ªq°"²a¨xÑ 8× ;Ñ ;ÀÌUÏ]É]Ðcl×csÑcsÐ ;Ó tÐñ !1°Ð<MÓ NÐñ 'Ù ØÐ0Ð2Ñ2�FÙ ØÐ#4Ð"6Ñ6�JÙ$Ø&Ø9=¿¹×9WÒ9W˜×3Ñ3Ñ5Ð^e×^pÑ^pÐ]rñÐ&ð —{‘{×5Ò5Ø(¨W×-EÑ-EÐ,GÑGÐ(á'Ø)àŸ;™;×9Ò9ð !×6Ñ6Ñ8à%×3Ñ3Ð5ñÐ)ñ ÜŸ™×-Ñ-Ð.?ÀRÐ-ÓH�ä#×/Ñ/°À1ÔE×MÑMÈaÓP‘ä#Ÿl™lÐ+<À"ÔE�ñ )Ø)Ð,@Ñ@À<ÐSTÐWkÑSkÑClÑl�ô Ÿ	™	 9¨kº!¸T¸'Ñ.BÐ"CÈÔLˆIØÐ#Ø—‘˜[Ÿ_™_Ó.Ô/à#7Ñ;LÈYÐX^Ó;_Ð:_Ñ#_Ð Ø!5×!9Ñ!9Ó!;¸qÑ!@Ðð ðE ×,Ñ,Ð-?ÀÐU^×UeÑUeÐ,ÖfðH ÐØ�L‰LŒNá"ØˆEÜÓN¼oÔNÔNÜ Ó i¼OÔ iÓi�	Ø$ YÑ/�Ø�{‰{×-Ò-Ü3Ø'Ø!Ø%Ø'9Ø*?Ø'9Ø%5Ø*?Ø$)ô
ð 
ô 1Ø'Ø!Ø%Ø1Ø"7Ø$)ôð ð Ð÷yNñ Nús   ÇQ-Ñ-Q7r4  c                 óx   — t        | j                  «      }t        j                  | |d   |d   z  g|dd z   «      S )z=[batch_size, num_beams, ...] -> [batch_size * num_beams, ...]r   r   r   N©r“  rÊ   ri   rµ  )r4  rÊ   s     rp   Ú_flatten_beam_dimz!GenerationMixin._flatten_beam_dim  s>   € ô �V—\‘\Ó"ˆÜ�}‰}˜V e¨A¡h°°q±Ñ&9Ð%:¸UÀ1À2¸YÑ%FÓGÐGro   r‹  c                 óh   — t        | j                  «      }t        j                  | ||g|dd z   «      S )z=[batch_size * num_beams, ...] -> [batch_size, num_beams, ...]r   NrÝ  )r4  rß   r‹  rÊ   s       rp   Ú_unflatten_beam_dimz#GenerationMixin._unflatten_beam_dim  s3   € ô �V—\‘\Ó"ˆÜ�}‰}˜V j°)Ð%<¸uÀQÀR¸yÑ%HÓIÐIro   c                 ó  — t        |j                  «      t        | j                  «      k  r=|j                  d«      }t        |j                  «      t        | j                  «      k  rŒ=t        j                  | |d¬«      }|S )aü  
        Gathers the beam slices indexed by beam_indices into new beam array.

        Args:
            tensor (`torch.Tensor`): A tensor containing data to be gathered. The tensor is a 2D or a 3D tensor
                with the two first dimensions depicting the batch and the beam dimensions.
            beam_indices (`torch.Tensor` of shape `(batch_size, num_beams_to_select)`): The indices of the beams to
                select .

        Returns:
            A tensor with the selected beams
        r¾   r   )ÚinputrÁ  r:  )r  rÊ   r  ri   Útake_along_dim)r4  r{   Úgathered_tensors      rp   Ú_gather_beamszGenerationMixin._gather_beams#  si   € ô �,×$Ñ$Ó%¬¨F¯L©LÓ(9Ò9Ø'×1Ñ1°"Ó5ˆLô �,×$Ñ$Ó%¬¨F¯L©LÓ(9Ó9ä×.Ñ.°VÀ\ÐWXÔYˆØÐro   Ú#is_early_stop_heuristic_unsatisfiedÚrunning_beam_scoresÚbeam_scoresÚis_sent_finishedÚcur_lenr_  Údecoder_prompt_lenÚearly_stoppingÚlength_penaltyc	                 óê   — |dk(  r|dkD  r||z
  }	n||z
  }	|dd…dd…f   |	|z  z  }
t        j                  |t        j                  |dd¬«      d   d«      }| t        j                  |
|kD  d	d¬«      z  S )
uH  
        Determine whether early stopping is possible by checking if the best possible score of running beams
        could still improve upon the finished ones.

        Mechanism:
        - Without a length penalty, beam scores typically decrease as more tokens are generated.
        So, if the *best possible* score from any running beam is already worse than the *worst* finished beam,
        we can safely stop early.
        - With a length penalty, scores may increase with longer sequences. In this case, we use heuristics
        to estimate the best possible score â€” though this estimate may not always be correct â€” and stop
        if no further improvement seems likely.

        We apply different heuristics depending on the value of `early_stopping`:
        1. `early_stopping == False`:
        -> Use a heuristic that assumes the best score comes from the current length minus the decoder prompt length.
        -> See detailed discussion: https://github.com/huggingface/transformers/pull/20901#issuecomment-1369845565

        2. `early_stopping == "never"`:
        -> Estimate the best score using either `max_length` or `cur_len`, depending on the sign of `length_penalty`.
        -> A positive length penalty favors longer sequences, so we use `max_length` in that case.

        NOTE: the canonical beam search implementation can be replicated with `early_stopping="never"` and
        `length_penalty=0.0`, which are NOT the default flags. The default behavior was empirically found to produce
        better sequences (prior to 2022), and changing it is BC breaking.
        Úneverr~  Nr   T)r:  Úkeepdimr   ç    eÍÍÁr¾   )ri   r£  rì  r  )ræ  rç  rè  ré  rê  r_  rë  rì  rí  Úbest_hypothetical_lengthÚbest_possible_running_scoreÚworst_finished_scores               rp   Ú_check_early_stop_heuristicz+GenerationMixin._check_early_stop_heuristic7  s›   € ðJ ˜WÒ$¨¸#Ò)=Ø'1Ð4FÑ'FÑ$à'.Ð1CÑ'CÐ$Ø&9º!¸R¸a¸R¸%Ñ&@ÐD\Ð^lÑDlÑ&mÐ#Ü$Ÿ{™{Ð+;¼U¿Y¹YÀ{ÐXYÐcgÔ=hÐijÑ=kÐmsÓtÐØ2´U·Y±YØ'Ð*>Ñ>ÀBÐPTô6
ñ 
ð 	
ro   Ú!next_token_hits_stopping_criteriac                 óž   — t        j                  | «      }t        j                  |«      |du z   }t        j                  |«       }||z  |z  S )zv
        Beam Search stopping condition -- halts the generation loop if any of these conditions becomes False
        T)ri   r  r>  )ræ  ré  rö  rì  Úimprovement_possibleÚexists_open_beamÚvalid_continuationss          rp   Ú%_beam_search_has_unfinished_sequencesz5GenerationMixin._beam_search_has_unfinished_sequencesf  sZ   € ô  %Ÿy™yÐ)LÓMÐô #ŸY™YÐ'7Ó8¸NÈdÐ<RÑSÐTÐô  %Ÿy™yÐ)JÓKÐKÐà#Ð&6Ñ6Ð9LÑLÐLro   Úaccumulated_log_probsÚrunning_sequencesÚrunning_beam_indicesrf  Úbeams_to_keepr`  c                 óæ  — |rOt        j                  t        j                  j	                  |d¬«      |¬«      }t        j
                  |d|¬«      }nt        j                  ||¬«      \  }}||	z  }| j                  ||«      }| j                  ||«      }||	z  }||dd…dd…|f<   t        j                  |
|j                  ¬«      j                  dd«      |z  }||z   }||dd…dd…||z
  f<   |||fS )	a'  
        Get top-K continuations given the accumulated log probs on the next token.

        A few notes to understand what's going on:
        1. Each item in batch has `num_beams` * `vocab_size` candidate continuations. For each item, get the
        top K [K = (number of EOS tokens + 1) * `num_beams`] candidates with the highest accumulated
        log-probabilities, or sample them without replacement using the accumulated scores
        2. We gather the top K (as opposed to `num_beams`, or any number lower than K) here so that we have at
        least `num_beams` sequences remaining to continue the live beam search.
        3. Note that other stopping criteria might result in impossible to continue beams, i.e. all continuations
        selected in this step hit the stopping criteria.
        r¾   r9  rº  r   )râ  r:  Úindex©rm  NrÆ   )ri   rÎ  r
   r·  rÍ  r»  Útopkrå  rÛ   rÂ   r;  )r�   rü  rý  rþ  rê  rë  rf  rÿ  r‹  r`  rß   Útopk_indicesÚtopk_log_probsÚtopk_current_beam_indicesÚtopk_running_beam_indicesÚtopk_running_sequencesÚtopk_idsÚbatch_offsetÚbatch_modified_indicess                      rp   Ú_get_top_k_continuationsz(GenerationMixin._get_top_k_continuations}  s  € ñ< Ü ×,Ñ,Ü—‘×%Ñ%Ð&;ÀÐ%ÓDÐR_ôˆLô #Ÿ\™\Ð0EÈ1ÐT`Ôa‰Nä+0¯:©:Ð6KÈ}Ô+]Ñ(ˆN˜Lð %1°JÑ$>Ð!Ø$(×$6Ñ$6Ð7KÐMfÓ$gÐ!Ø!%×!3Ñ!3Ð4EÐG`Ó!aÐØ *Ñ,ˆð 19Ðšq¢! W˜}Ñ-ô —|‘| J°x·±ÔG×LÑLÈRÐQRÓSÐV_Ñ_ˆØ!:¸\Ñ!IÐØH^Ð!¢!¢Q¨Ð2DÑ(DÐ"DÑEàÐ5Ð7PÐPÐPro   r  r  r  c                 óö   — ||j                  t        j                  «      dz  z   }t        j                  ||¬«      d   }| j	                  ||«      }| j	                  ||«      }	| j	                  ||«      }
||	|
fS )zÂ
        Given the top-K continuations, their scores, and whether they hit a stopping criteria, select the
        best non-finished beams to continue beam search in the next iteration.
        rñ  r  r   )r²  ri   rÌ  r  rå  )r�   r  r  r  rö  r‹  Útopk_running_log_probsÚnext_topk_indicesrý  rç  rþ  s              rp   Ú%_get_running_beams_for_next_iterationz5GenerationMixin._get_running_beams_for_next_iteration³  s�   € ð "0Ð2S×2VÑ2VÔW\×WdÑWdÓ2eÐhnÑ2nÑ!nÐä!ŸJ™JÐ'=ÀÔKÈAÑNÐØ ×.Ñ.Ð/EÐGXÓYÐØ"×0Ñ0Ð1GÐIZÓ[ÐØ#×1Ñ1Ð2KÐM^Ó_ÐØ Ð"5Ð7KÐKÐKro   Útop_num_beam_maskc                 ó°  — |	|
ddd…f   z  }||dz   |z
  |z  z  }t        j                  |dd¬«      |du z  }||j                  t         j                  «      dz  z  }|| j                  t         j                  «      dz  z  }|| dz  z  }t        j                  ||fd¬«      }t        j                  ||fd¬«      }t        j                  ||fd¬«      }t        j                  ||fd¬«      }t        j
                  ||¬«      d   }| j                  ||«      }| j                  ||«      }| j                  ||«      }| j                  ||«      }||||fS )	z¥
        Updates the finished beams if (and only if) there are new completed sequences that have a higher score than
        the current finished sequences.
        Nr   r¾   T)ÚaxisÚkeepdimsrñ  r9  r  )ri   r>  r²  rÌ  r@  r  rå  )r�   r`   r  rè  r  r{   r  ræ  ré  rö  r  r‹  rê  rë  rí  rì  Údid_top_num_beams_just_finishedÚbeams_in_batch_are_fullÚmerged_sequencesÚmerged_scoresÚmerged_beam_indicesÚmerged_is_sent_finishedÚtopk_merged_indicess                          rp   Ú_update_finished_beamsz&GenerationMixin._update_finished_beamsÉ  s†  € ð2 +LÐN_Ð`dÒfgÐ`gÑNhÑ*hÐ'ð (¨G°a©KÐ:LÑ,LÐQ_Ñ+_Ñ`ˆä"'§)¡)Ð,<À2ÐPTÔ"UÐYgÐkoÐYoÑ"pÐØÐ1×4Ñ4´U·]±]ÓCÀfÑLÑLˆàÐ?Ð?×CÑCÄEÇMÁMÓRÐU[Ñ[Ñ[ˆð 	Ð;Ð;¸vÑEÑEˆô
 !Ÿ9™9 iÐ1GÐ%HÈaÔPÐÜŸ	™	 ;°Ð"?ÀQÔGˆÜ#Ÿi™i¨Ð7PÐ(QÐWXÔYÐÜ"'§)¡)Ð-=Ð?^Ð,_ÐefÔ"gÐÜ#Ÿj™j¨¸)ÔDÀQÑGÐØ×&Ñ&Ð'7Ð9LÓMˆ	Ø×(Ñ(¨Ð8KÓLˆØ×)Ñ)Ð*=Ð?RÓSˆØ×-Ñ-Ð.EÐGZÓ[ÐØ˜+ |Ð5EÐEÐEro   c                 ó  ‡— |j                   }|j                  }|j                  }	|j                  }
|j                  }|j
                  }|j                  }|j                  }|j                  }|j                  }|j                  }|j                  }|j                  }|j                  dd \  }}||z  }| j                  j                  dk(  r| j                   j"                  }nˆ| j                  j                  dk(  r| j%                  «       j&                  }nT| j                  j                  dk(  r| j                   j(                  }n$| j                   j+                  «       j,                  }|}d}|�|j                  d   nd}t/        dd|z   «      |z  }t1        j2                  t1        j4                  |t0        j6                  ¬	«      t1        j8                  ||z
  t0        j6                  ¬	«      fd¬
«      j;                  |j<                  «      }|j>                  }|rtA        d«      ‚|r|rdnd}|r|rdnd}|r|rdnd} |r|	rdnd}!|r|	rdnd}"|r|
rdnd}#|rF| j                   jB                  r0|	r‰d   jE                  d«      nd}$|
r‰d   jE                  d«      nd}%|�	|xs |d   nd}&t1        jF                  |||f|&t0        jH                  |j<                  ¬«      }'| jK                  |||«      |'dd…dd…d|…f<   |'jM                  «       jO                  «       }(t1        j8                  ||ft0        jP                  |j<                  ¬«      })d|)dd…dd…f<   t1        jF                  ||fdt0        jP                  |j<                  ¬«      }*t1        j8                  ||ft0        j6                  |j<                  ¬«      }+t1        j4                  |dft0        j6                  |j<                  ¬«      },t1        j8                  ||ft0        j6                  |j<                  ¬«      }-t1        jF                  ||||z
  fdt0        jR                  |j<                  ¬«      }.|.jM                  «       jO                  «       } |}/d}0| jU                  ||‰|jV                   ¬«      }1| jY                  |||j<                  ¬«      �r/|0rG| j[                  |'dd…dd…d|…f   «      }/‰d   rdnd}2 | j\                  |/fd|2i‰¤Ž}3 | di |3¤ddi¤Ž}1d}0| j_                  1‰| j                   jB                  ¬«      Š|r|rŒ—|1j`                  dd…ddd…f   j;                  dt0        jb                  |j<                  ¬«      }4td        jf                  ji                  |4d¬
«      }5 ||/|5«      }5|r¾|r||4jO                  «       fz  }|r|r||5jO                  «       fz  }|	rY|!| j                   jB                  r|1jj                  fn|1jl                  fz  }!| j                   jB                  r|"|1jn                  fz  }"|
r3|#| j                   jB                  r|1jp                  fn|1jr                  fz  }#~1| jK                  |5||«      }5|5|)dd…dd…df   z   }5t1        jt                  |5|||z  f«      }5| jw                  |5|'|.|||||||¬«
      \  }6}7}8 || j[                  |7dd…dd…d|dz   …f   «      |«      }-| jK                  |-||«      }-| jy                  |6|7|8|-|¬«      \  }'})}.| j{                  |(|7|*|6| |8|,|+|-||||||¬«      \  }(}*} }+‰jE                  d«      �R| j[                  |.d ||z
  f   «      }9t}        | d!«      r| j                  ‰d   |9«      ‰d<   n‰d   j�                  |9«       |dz   }| jƒ                  |,|)|*|+|||||¬"«	      },| j…                  |,|+|-|«       }| jY                  |||j<                  ¬«      r�Œ/| j[                  |(dd…d|…dd…f   «      }(| j[                  |*dd…d|…f   «      }*| j[                  | dd…d|…dd…f   «      } | dz   j7                  «       j‡                  d¬
«      j/                  «       }:||:z   };|(dd…d|;…f   }(| dd…d|:…f   } |rz|sd}*d}<t‰        ˆfd#„tŠ        D «       «      rt�        ˆfd$„tŠ        D «       «      }=‰|=   }<| j                   jB                  rt�        |(|*||| $%|!|"|#|<¬%«      S t‘        |(|*||| |!|#|<¬&«      S |(S )'aº	  
        Generates sequences of token ids for models with a language modeling head using **beam search decoding** and
        can be used for text-decoder, text-to-text, speech-to-text, and vision-to-text models.

        If it's the first time you're diving into Beam Search, we recommend you read the following blog post:
        https://huggingface.co/blog/how-to-generate (especially the beam search section).

        You can recompute the sequence scores from the individual scores using the `compute_transition_scores` function
        (https://huggingface.co/docs/transformers/main_classes/text_generation#transformers.GenerationMixin.compute_transition_scores)

        Parameters:
            input_ids (`torch.LongTensor` of shape `(batch_size*num_beams, sequence_length)`):
                The sequence used as a prompt for the generation.
            logits_processor (`LogitsProcessorList`):
                An instance of [`LogitsProcessorList`]. List of instances of class derived from [`LogitsProcessor`]
                used to modify the prediction scores of the language modeling head applied at each generation step.
            stopping_criteria (`StoppingCriteriaList`:
                An instance of [`StoppingCriteriaList`]. List of instances of class derived from [`StoppingCriteria`]
                used to tell if the generation loop should stop.
            generation_config ([`~generation.GenerationConfig`]):
                The generation configuration to be used as parametrization of the decoding method.
            synced_gpus (`bool`):
                Whether to continue running the while loop until max_length (needed to avoid deadlocking with
                `FullyShardedDataParallel` and DeepSpeed ZeRO Stage 3).
            model_kwargs:
                Additional model specific kwargs will be forwarded to the `forward` function of the model. If model is
                an encoder-decoder model the kwargs should include `encoder_outputs`.

        Return:
            [`generation.GenerateBeamDecoderOnlyOutput`], [`~generation.GenerateBeamEncoderDecoderOutput`] or
            `torch.LongTensor`: A `torch.LongTensor` containing the generated tokens (default behaviour) or a
            [`~generation.GenerateBeamDecoderOnlyOutput`] if `model.config.is_encoder_decoder=False` and
            `return_dict_in_generate=True` or a [`~generation.GenerateBeamEncoderDecoderOutput`] if
            `model.config.is_encoder_decoder=True`.
        Nr   ÚMoshiDepthDecoderÚImageGPTForCausalImageModelingÚBarkSemanticModelFr   r   )rÁ   r9  zÃ`low_memory=True` is not supported after the beam search refactor. Please check the discussion in #35802 *after the PR got merged*, and add a comment there if your questions are not yet answered.rn   rû   rc   rd   r¾   )Ú
fill_valuerÁ   rÂ   rÀ   rñ  r·  rÆ   r$  r³   r.  Tr¸  r¹  )
rü  rý  rþ  rê  rë  rf  rÿ  r‹  r`  rß   )r  r  r  rö  r‹  )r`   r  rè  r  r{   r  ræ  ré  rö  r  r‹  rê  rë  rí  rì  rV   .Ú_reorder_cache)	ræ  rç  rè  ré  rê  r_  rë  rì  rí  c              3   ó&   •K  — | ]  }|‰v –— Œ
 y ­wr&  rn   r½  s     €rp   r+  z/GenerationMixin._beam_search.<locals>.<genexpr>=  r¿  rÀ  c              3   ó,   •K  — | ]  }|‰v sŒ|–— Œ y ­wr&  rn   r½  s     €rp   r+  z/GenerationMixin._beam_search.<locals>.<genexpr>>  rÂ  rÃ  )r`   rz   ra   rb   r{   rs   rt   ru   rv   rw   rV   )r`   rz   ra   rb   r{   rc   rd   rV   )Ir  r  r,  r-  rÆ  rÇ  rÈ  rf  rì  rí  r_  r‹  rÎ  rÊ   rô   re   rÃ   Úaudio_vocab_sizeÚget_output_embeddingsÚout_featuresÚoutput_vocab_sizere  r`  rº  ri   r@  rÿ   rH  Úzerosr²  rÂ   Ú
low_memoryrò   rÇ   rÌ   ÚfullÚint64rà  ÚdetachrÈ   ÚfloatÚint32rË  r¥  r•  rÞ  rë   rT  rb   rÌ  r
   r·  r¸  ru   rc   rv   rw   rd   rµ  r  r  r  r™   r"  Úreorder_cacherõ  rû  r¹  r  rM  rÒ  r}   ry   )>r�   r²   rU  rž  r’   rg  rî   r  r  r,  r-  rÆ  rÇ  rÈ  rf  rì  rí  r_  r‹  rÎ  Úbatch_size_unflattenedrê  rß   r`  rë  rŽ  Ún_eos_tokensrÿ  r  Ú
sequentialÚ
all_scoresrÔ  r{   ru   rv   rw   rs   rt   Úoutput_fill_valuerý  r`   rç  rè  ré  ræ  rö  rþ  Úflat_running_sequencesr×  Úmodel_outputsr³   rÜ   rb   Ú	log_probsr  r  r  Úbeam_idxÚmax_generated_lengthÚoutput_lengthr  r¾  s>         `                                                       rp   r\   zGenerationMixin._beam_search   s
  ø€ ðZ )×:Ñ:ˆØ(×:Ñ:ˆØ-×?Ñ?ÐØ0×EÑEÐØ)×7Ñ7ˆØ)×7Ñ7ˆØ"3×"KÑ"KÐØ%×/Ñ/ˆ	Ø*×9Ñ9ˆØ*×9Ñ9ˆØ&×1Ñ1ˆ
Ø%×/Ñ/ˆ	Ø0×EÑEÐà*3¯/©/¸"¸1Ð*=Ñ'Ð Ø+¨yÑ8ˆ
à�>‰>×"Ñ"Ð&9Ò9ØŸ™×5Ñ5‰JØ�^‰^×$Ñ$Ð(HÒHØ×3Ñ3Ó5×BÑB‰JØ�^‰^×$Ñ$Ð(;Ò;ØŸ™×6Ñ6‰JàŸ™×4Ñ4Ó6×AÑAˆJØ$ÐØ"Ðð 1=Ð0H�|×)Ñ)¨!Ò,ÈaˆÜ˜A˜q <Ñ/Ó0°9Ñ<ˆÜ!ŸI™IÜ�Z‰Z˜¬5¯:©:Ô6¼¿¹À]ÐU^ÑE^Ôgl×gqÑgqÔ8rÐsØô
÷ ‰"ˆY×ÑÓ
ð 	ð '×1Ñ1ˆ
ÙÜðtóð ñ 4¹‘RÈDˆ
Ù3¹‘RÈDˆ
Ù5¹-‘rÈdˆÙ$;Ñ@Q™RÐX\ÐÙ"9Ñ>O™2ÐVZÐÙ'>ÑCW¡Ð^bÐñ # t§{¡{×'EÒ'EÙVg Ð.?Ñ!@×!DÑ!DÀ\Ô!RÐmqÐáH\�Ð.Ñ/×3Ñ3°OÔDÐbfð "ð @LÐ?W˜LÒ;¨L¸ªOÐ]_ÐÜ!ŸJ™JØ˜ JÐ/Ø(Ü—+‘+Ø×#Ñ#ô	
Ðð -1×,DÑ,DÀYÐPZÐ\eÓ,fÐš!šQ   ˜.Ñ)Ø%×,Ñ,Ó.×4Ñ4Ó6ˆ	ô
 $Ÿk™k¨:°yÐ*AÌÏÉÐ]f×]mÑ]mÔnÐØ%)ÐšA˜q™r˜EÑ"Ü—j‘j *¨iÐ!8ÀTÔQV×Q\ÑQ\Ðen×euÑeuÔvˆô !Ÿ;™;¨
°IÐ'>ÄeÇjÁjÐYb×YiÑYiÔjÐô /4¯j©j¸*Àa¸ÔPU×PZÑPZÐcl×csÑcsÔ.tÐ+ô -2¯K©KØ˜Ð#¬5¯:©:¸i×>NÑ>Nô-
Ð)ô
  %Ÿz™zØ˜ J°Ñ$8Ð9ÀbÔPU×P[ÑP[Ðdm×dtÑdtô 
Ðð ,×2Ñ2Ó4×:Ñ:Ó<ˆà!*ÐØ ÐØŸ™ØØØØ#4×#AÑ#AÐAð	 &ó 
ˆð ×,Ñ,Ð-?ÀÐU^×UeÑUeÐ,ÕfÙà)-×)?Ñ)?Ð@QÒRSÒUVÐX`ÐY`ÐX`ÐR`Ñ@aÓ)bÐ&Ø,8¸Ò,E¡qÈ4Ð$ØA˜t×AÑAØ*ñ ØAUð ØYeñ �ñ !%Ñ F |Ñ FÀÒ F�Ø#Ðð  ×CÑCØØØ#'§;¡;×#AÑ#Að Dó ˆLñ
 Ñ1Øð #×)Ñ)ª!¨R²¨(Ñ3×6Ñ6¸DÌÏÉÐ^g×^nÑ^nÐ6ÓoˆFô Ÿ™×1Ñ1°&¸bÐ1ÓAˆIÙ(Ð)?ÀÓKˆIñ 'Ù Ø 6§<¡<£>Ð"3Ñ3�JÙ*©}Ø 9§?¡?Ó#4Ð"6Ñ6�Já$Ø&àŸ;™;×9Ò9ð '×9Ñ9Ñ;à+×6Ñ6Ð8ñÐ&ð
 —{‘{×5Ò5Ø(¨]×-KÑ-KÐ,MÑMÐ(á'Ø)àŸ;™;×9Ò9ð '×<Ñ<Ñ>à+×9Ñ9Ð;ñÐ)ð à×0Ñ0°¸JÈ	ÓRˆIØ!Ð$7ºº1¸d¸
Ñ$CÑCˆIÜŸ™ i°*¸iÈ*Ñ>TÐ1UÓVˆIð QU×PmÑPmØ&/Ø"3Ø%9ØØ#5Ø#Ø+Ø#Ø%Ø%ð Qnó QÑMˆNÐ2Ð4Mñ 1BØ×&Ñ&Ð'=ºaÂÀMÀgÐPQÁkÀMÐ>QÑ'RÓSØó1Ð-ð 15×0HÑ0HØ1°:¸}ó1Ð-ð
 LP×KuÑKuØ-Ø'=Ø*CØ2SØ#ð Lvó LÑHÐÐ2Ð4Hð FJ×E`ÑE`Ø#Ø'=Ø'Ø-Ø)Ø*CØ4WØ!1Ø2SØ"3Ø#ØØ#5Ø-Ø-ð Faó FÑBˆI�{ LÐ2Bð. ×ÑÐ 1Ó2Ð>Ø×1Ñ1Ð2FÀsÈGÐVhÑLhÐGhÑ2iÓj�Ü˜4Ð!1Ô2Ø6:×6IÑ6IÈ,ÐWhÑJiÐksÓ6t�LÐ!2Ò3à Ð!2Ñ3×AÑAÀ(ÔKà ‘kˆGØ26×2RÑ2RØ4WØ$7Ø'Ø!1ØØ%Ø#5Ø-Ø-ð 3Só 
3Ð/ð &*×%OÑ%OØ3Ø Ø1Øó	&ð "ÐðO ×,Ñ,Ð-?ÀÐU^×UeÑUeÐ,Öfð` ×*Ñ*¨9²QÐ8MÐ9MÐ8MÊqÐ5PÑ+QÓRˆ	Ø×,Ñ,¨[ºÐ<QÐ=QÐ<QÐ9QÑ-RÓSˆØ×-Ñ-¨lº1Ð>SÐ?SÐ>SÒUVÐ;VÑ.WÓXˆð ".°Ñ!1× 7Ñ 7Ó 9×>Ñ>À1Ð>ÓE×IÑIÓKÐØ*Ð-AÑAˆØša  - Ð/Ñ0ˆ	Ø#¢AÐ'<Ð(<Ð'<Ð$<Ñ=ˆá"Ù Ø"�àˆEÜÓN¼oÔNÔNÜ Ó i¼OÔ iÓi�	Ø$ YÑ/�à�{‰{×-Ò-Ü7Ø'Ø%0Ø%Ø%Ø!-Ø'9Ø*?Ø'9Ø%5Ø*?Ø$)ôð ô 5Ø'Ø%0Ø%Ø%Ø!-Ø1Ø"7Ø$)ô	ð 	ð Ðro   c                 óf  ‡‡3‡4— ‰d   st        d«      ‚|j                  dv s t        ‰j                  d«      «      t        u rt        d«      ‚| j                  ||||||
|	‰¬«      }|j                  }|j                  }|j                  }|j                  }|j                  }|j                  }|r|rdnd}|r|rdnd}|r|rdnd}|r|rdnd}|r|rdnd}|rF| j                  j                  r0|r‰d	   j                  d
«      nd}|r‰d	   j                  d«      nd}|j                  dd \  }}|dkD  rt        d«      ‚t        j                   |t        j"                  |j$                  ¬«      }d}d}| j'                  |||j$                  ¬«      �rÀ|j                  d   }|j)                  |«      \  }} |j+                  | j$                  «      }| �| j+                  | j$                  «      } |j                  d   |j                  d   z
  }! ||d«      }"t-        j,                  ‰«      }#t/        |#|j                  d   | j                  j                  «      }#t1        |#|j                  d   «      }#|#j                  d«      x}$�8|!dkD  r3|!|$j                  d   z   }%t3        |#|%| j                  j                  «      }#|s|!dz   nd}& | j4                  |f|&|dœ|#¤Ž}'d|'v r|!dz   |'d<    | di |'¤Ž}(|(j6                  dd…|! dz
  d…f   j+                  t        j8                  |j$                  ¬«      Š3‰3j;                  «       Š4t=        |«      dkD  r<t?        |!dz   «      D ]+  }) ||dd…d||)z   …f   ‰3dd…|)dd…f   «      ‰3dd…|)dd…f<   Œ- |r| �tA        || |!‰3|"«      \  }*}+n³|rJ‰3jC                  d¬«      },t        jD                  |,ddd…dd…f   d¬«      jG                  d«      ddd…f   }-n‰3jI                  d¬«      }-|dd…|d…f   }.|.|-dd…dd…f   k(   jK                  d¬«      dk  jM                  «       }+|"r
|+|!k(  r|+dz  }+|-dd…d|+dz   …f   }*t        jN                  ||*fd¬«      }|�|jQ                  |*jS                  «       «       |j                  d   }/|(jT                  jW                  |/dz
  «       |jY                  |‰3|+«       | j[                  |(‰| j                  j                  |+dz   ¬«      Š|r|r�ŒŽ|�r|+dz   }0|r |t]        ˆ3fd„t?        |0«      D «       «      z  }|r |t]        ˆ4fd„t?        |0«      D «       «      z  }|r|/n|0}0|rr| j                  j                  r3t_        ||(j`                  ||0«      }t_        ||(jb                  ||0d¬«      }n)|(jd                  d   �t_        ||(jd                  ||0d¬«      }|rG| j                  j                  rt_        ||(jf                  ||0«      }nt_        ||(jh                  ||0«      }| |||«       z  }|jk                  «       dk(  }d}| j'                  |||j$                  ¬«      r�ŒÀ|�|jm                  «        to        |tp        «      rH|jr                  jt                  jv                  dk(  r%|jx                  |jr                  jt                  _<        |rrd}1t{        ˆfd„t|        D «       «      rt        ˆfd „t|        D «       «      }2‰|2   }1| j                  j                  rt�        |||||||1¬!«	      S tƒ        ||||||1¬"«      S |S )#aª  
        Generates sequences of token ids for models with a language modeling head using **greedy decoding** or
        **sample** (depending on `do_sample`), assisted by candidate sequences. Assisted generation is an example of a
        candidate decoding strategy. Can be used for text-decoder, text-to-text, speech-to-text, and vision-to-text
        models.

        Parameters:
            input_ids (`torch.LongTensor` of shape `(batch_size, sequence_length)`):
                The sequence used as a prompt for the generation.
            logits_processor (`LogitsProcessorList`):
                An instance of [`LogitsProcessorList`]. List of instances of class derived from [`LogitsProcessor`]
                used to modify the prediction scores of the language modeling head applied at each generation step.
            stopping_criteria (`StoppingCriteriaList`):
                An instance of [`StoppingCriteriaList`]. List of instances of class derived from [`StoppingCriteria`]
                used to tell if the generation loop should stop.
            generation_config ([`~generation.GenerationConfig`]):
                The generation configuration to be used as parametrization of the decoding method.
            synced_gpus (`bool`):
                Whether to continue running the while loop until max_length (needed to avoid deadlocking with
                `FullyShardedDataParallel` and DeepSpeed ZeRO Stage 3).
            streamer (`BaseStreamer`, *optional*):
                Streamer object that will be used to stream the generated sequences. Generated tokens are passed
                through `streamer.put(token_ids)` and the streamer is responsible for any further processing.
            inputs_tensor (`torch.FloatTensor`, *optional*):
                The input tensor for generation. For decoder models, usually `input_ids`. For encoder-decoder models,
                the tensor that produced `model_kwargs["encoder_outputs"]`.
            assistant_model (`PreTrainedModel`, *optional*):
                The model used to assist the generation process. If not provided, the main model will be used.
            assistant_tokenizer (`PreTrainedTokenizerBase`, *optional*):
                The tokenizer used for the assistant model. If not provided, the token space is assumed to be the same.
            tokenizer (`PreTrainedTokenizerBase`, *optional*):
                The tokenizer used for the main model. If not provided, the token space is assumed to be the same.
            model_kwargs:
                Additional model specific keyword arguments will be forwarded to the `forward` function of the model.
                If model is an encoder-decoder model the kwargs should include `encoder_outputs`.

        Return:
            [`~generation.GenerateDecoderOnlyOutput`], [`~generation.GenerateEncoderDecoderOutput`] or
            `torch.LongTensor`: A `torch.LongTensor` containing the generated tokens (default behaviour) or a
            [`~generation.GenerateDecoderOnlyOutput`] if `model.config.is_encoder_decoder=False` and
            `return_dict_in_generate=True` or a [`~generation.GenerateEncoderDecoderOutput`] if
            `model.config.is_encoder_decoder=True`.
        r$  z+assisted generate requires `use_cache=True`)Ústaticrò  Úsliding_windowrV   z=assisted generate is not supported with Static cache classes`)r’   r²   r  rV  rU  rW  rX  rî   rn   Nrû   rc   rd   r   r   z6assisted generate is only supported for batch_size = 1rÀ   FTrÆ   r»   r   r¾   ©r³   r¶   r/  r9  rº  )rÇ   rK  c              3   ó6   •K  — | ]  }‰d d …|d d …f   –— Œ y ­wr&  rn   )r(  rƒ  Ú
new_logitss     €rp   r+  z5GenerationMixin._assisted_decoding.<locals>.<genexpr>8  s   øè ø€ Ò#[¸A Jªq°!²Q¨wÕ$7Ñ#[ùó   ƒc              3   ó6   •K  — | ]  }‰d d …|d d …f   –— Œ y ­wr&  rn   )r(  rƒ  rØ  s     €rp   r+  z5GenerationMixin._assisted_decoding.<locals>.<genexpr>:  s   øè ø€ Ò'fÀqÐ(9º!¸QÂ¸'Õ(BÑ'fùrB  )Úis_decoder_attentionÚ	heuristicc              3   ó&   •K  — | ]  }|‰v –— Œ
 y ­wr&  rn   r½  s     €rp   r+  z5GenerationMixin._assisted_decoding.<locals>.<genexpr>l  r¿  rÀ  c              3   ó,   •K  — | ]  }|‰v sŒ|–— Œ y ­wr&  rn   r½  s     €rp   r+  z5GenerationMixin._assisted_decoding.<locals>.<genexpr>m  rÂ  rÃ  rÄ  rÅ  )Brò   rø  r  rÌ   r   rk  rf  r,  r-  rÆ  rÇ  rÈ  rÃ   rÇ   rÊ   ri   rÿ   r   rÂ   r•  Úget_candidatesr²  rô  r&   r(   r'   rë   rb   rÌ  rÈ   r  r{  Ú_speculative_samplingrÍ  rÎ  rÏ  rÐ  r  r¹  r@  r�  rC  rV   ÚcropÚupdate_candidate_strategyrT  rl   Ú_split_model_outputsrv   ru   rc   rw   rd   rº  rÑ  rÍ   r    rV  r’   Únum_assistant_tokens_scheduleÚnum_assistant_tokensr  rM  rÒ  rr   r_   )5r�   r²   rU  rž  r’   rg  rÅ  r  rV  rX  rŸ  rî   rj  rf  r,  r-  rÆ  rÇ  rÈ  ra   rÔ  ru   rv   rw   rs   rt   rß   rê  rÕ  rŽ  r¶   Úcandidate_input_idsÚcandidate_logitsÚcandidate_lengthÚis_done_candidateÚcandidate_kwargsr»   Ú
new_lengthr³   rÜ   rJ  rƒ  Úvalid_tokensÚ	n_matchesrÚ  Úselected_tokensÚcandidate_new_tokensÚnew_cur_lenÚnewly_added_lengthr  r¾  rA  rØ  s5              `                                       @@rp   r]   z"GenerationMixin._assisted_decoding]  s  ú€ ðt ˜KÒ(ÜÐJÓKÐKà×2Ñ2Ð6\Ñ\Ü�L×$Ñ$Ð%6Ó7Ó8¼KÑGäÐ\Ó]Ð]à"×;Ñ;Ø/ØØ'Ø+Ø-Ø&Ø 3Ø%ð <ó 	
Ðð &×/Ñ/ˆ	Ø-×?Ñ?ÐØ0×EÑEÐØ)×7Ñ7ˆØ)×7Ñ7ˆØ"3×"KÑ"KÐñ 0±M‘ÈˆÙ3¹‘RÈDˆ
Ù$;Ñ@Q™RÐX\ÐÙ"9Ñ>O™2ÐVZÐÙ'>ÑCW¡Ð^bÐñ # t§{¡{×'EÒ'EÙVg Ð.?Ñ!@×!DÑ!DÀ\Ô!RÐmqÐáH\�Ð.Ñ/×3Ñ3°OÔDÐbfð "ð
 (Ÿo™o¨b¨qÐ1Ñˆ
�GØ˜Š>ÜÐUÓVÐVÜ$Ÿz™z¨*¼E¿J¹JÈy×O_ÑO_Ô`Ðà"ÐØ!ÐØ×,Ñ,Ð-?ÀÐU^×UeÑUeÐ,ÕfØ—o‘o aÑ(ˆGð 5H×4VÑ4VÐW`Ó4aÑ1ÐÐ!1Ø"5×"8Ñ"8¸¿¹Ó"EÐØÐ+Ø#3×#6Ñ#6°t·{±{Ó#CÐ à2×8Ñ8¸Ñ;¸i¿o¹oÈaÑ>PÑPÐÙ 1Ð2EÀtÓ LÐô  $Ÿy™y¨Ó6ÐÜ6Ø Ð"5×";Ñ";¸AÑ">ÀÇÁ×@^Ñ@^ó Ðô  7Ð7GÐI\×IbÑIbÐcdÑIeÓfÐØ 0× 4Ñ 4°^Ó DÐD�ÐQÐVfÐijÒVjØ-°×0BÑ0BÀ2Ñ0FÑF�
Ü#8Ð9IÈ:ÐW[×WbÑWb×WuÑWuÓ#vÐ á?QÐ#3°aÒ#7ÐW[Ð Ø=˜4×=Ñ=Ø#ðà%9Ø#5ñð #ñ	ˆLð   <Ñ/Ø1AÀAÑ1E�Ð-Ñ.ñ Ñ*˜\Ñ*ˆGð !Ÿ™ªÐ,<Ð+<¸qÑ+@Ñ+BÐ(BÑC×FÑFÜ—m‘m¨I×,<Ñ,<ð Gó ˆJð !+× 0Ñ 0Ó 2ÐÜÐ#Ó$ qÒ(ÜÐ/°!Ñ3Ó4ò w�AÙ*:Ð;NÊqÐR_ÐT[Ð^_ÑT_ÐR_ÐO_Ñ;`ÐblÒmnÐpqÒstÐmtÑbuÓ*v�Jšq !¢Q˜wÒ'ðwñ Ð-Ð9Ü*?Ø'Ø$Ø$ØØ%ó+Ñ'�™iñ Ø&×.Ñ.°2Ð.Ó6�EÜ&+×&7Ñ&7¸¸aÂÂA¸g¹ÐTUÔ&V×&^Ñ&^Ð_`Ó&aÐbfÒhiÐbiÑ&j‘Oà&0×&7Ñ&7¸BÐ&7Ó&?�Oà':º1¸g¹h¸;Ñ'GÐ$Ø 4¸ÊÈ3ÈBÈ3ÈÑ8OÑ OÐP×XÑXÐ]_ÐXÓ`ÐcdÑd×iÑiÓk�	ñ %¨Ð6FÒ)FØ ‘N�IØ.ªq°/°IÀ±M°/Ð/AÑB�ô Ÿ	™	 9¨lÐ";ÀÔDˆIØÐ#Ø—‘˜\×-Ñ-Ó/Ô0Ø#Ÿ/™/¨!Ñ,ˆKð ×#Ñ#×(Ñ(¨°q©Ô9ð  ×9Ñ9¸)ÀZÐQZÔ[ð  ×CÑCØØØ#'§;¡;×#AÑ#AØ(¨1™}ð	 Dó ˆLñ Ñ1Ùò 'Ø%.°¡]Ð"Ù ØœeÓ#[ÄÐGYÓAZÔ#[Ó[Ñ[�FÙ Ø¤%Ó'fÌEÐRdÓLeÔ'fÓ"fÑf�Já4F¡[ÐL^Ð"Ù$Ø—{‘{×5Ò5Ü+?Ø,¨g×.FÑ.FÈÐQcó,Ð(ô .BØ.Ø#×6Ñ6Ø#Ø.Ø15ô.Ñ*ð !×+Ñ+¨AÑ.Ð:Ü-AØ.Ø#×.Ñ.Ø#Ø.Ø15ô.Ð*ñ (Ø—{‘{×5Ò5Ü0DØ1°7×3PÑ3PÐRYÐ[mó1Ñ-ô 1EØ1°7×3HÑ3HÈ'ÐSeó1Ð-ð $8Ñ;LÈYÐX^Ó;_Ð:_Ñ#_Ð Ø!5×!9Ñ!9Ó!;¸qÑ!@ÐØ!&Ððo ×,Ñ,Ð-?ÀÐU^×UeÑUeÐ,Öfðr ÐØ�L‰LŒNô Ð*Ô,FÔGØ#×3Ñ3×EÑE×cÑcÐgrÒrð $×8Ñ8ð  ×/Ñ/×AÑAÔVñ #ØˆEÜÓN¼oÔNÔNÜ Ó i¼OÔ iÓi�	Ø$ YÑ/�Ø�{‰{×-Ò-Ü3Ø'Ø!Ø%Ø'9Ø*?Ø'9Ø%5Ø*?Ø$)ô
ð 
ô 1Ø'Ø!Ø%Ø1Ø"7Ø$)ôð ð Ðro   c                 ó>  — d}|j                  d«      }d}| j                  j                  s|�|rd}|j                  d«      x}�†|j                  «       }	|r|d   j                  d   |	z
  }n^| j                  j                  rdnd}
|j                  |
«      }|�1|j                  d   |j                  d   k(  r|j                  d   |	z
  }|j
                  €" | j                  |f||d	œ|¤Ž} | di |¤d
di¤ŽS dt        t        d«      j                  _	        |j
                  }t        j                  ||d¬«      }d|vrt        d«      ‚| j                  ||«      r| j                  |j                  «      n| j                  }|j!                  dd«      }|j!                  dd«      }d}	|D ]d  }|	|j                  d   z   }|�|dd…d|…f   |d<   |�|dd…|	|…f   |d<    | j                  |fi |¤Ž} |di |¤d
di¤Ž}|j"                  |d<   |}	Œf ||d<   ||d<   S )a§  
        Perform the prefill stage of generation.

        Note that usually, the prefill stage is always the first iteration of a new input batch, and thus multimodal inputs etc
        should be treated as if it's the first iteration. However, for assisted decoding, assistants call `generate`
        several time in a row for a same batch of inputs, so we need to pass `is_first_iteration` here for such cases.
        Nrµ   FTrV   r   r¿   r´   r?  r.  é@   Ú_dynamor¾   r9  z+Cannot use prefill chunking without a cacher»   r   rn   )rÌ   rÃ   rÇ   rÚ   rÊ   Úprefill_chunk_sizerë   rÐ   ri   Úcache_size_limitÚsplitrò   rS  rÉ  rI  rÊ  rË   rV   )r�   r²   r’   rî   r¶   r³   rµ   Úuse_inputs_embedsr  r
  rå   r´   rÜ   Ú
chunk_sizeÚinput_chunksrÖ  r»   Úinput_chunkÚcurrent_lengthrJ  s                       rp   rË  zGenerationMixin._prefillˆ  s•  € ð$  $ÐØ$×(Ñ(¨Ó9ˆØ!ÐØ�{‰{×-Ò-°-Ð2KÑPbØ $ÐØ!×%Ñ%Ð&7Ó8Ð8ˆEÐEØ×.Ñ.Ó0ˆKá Ø'3°OÑ'D×'JÑ'JÈ1Ñ'MÐP[Ñ'[Ñ$àAEÇÁ×A_ÒA_Ñ%=ÐeuÐ"Ø!-×!1Ñ!1Ð2DÓ!E�à!Ð-°)·/±/À!Ñ2DÈ×H\ÑH\Ð]^ÑH_Ò2_à+4¯?©?¸1Ñ+=ÀÑ+KÐ(ð ×/Ñ/Ð7Ø=˜4×=Ñ=Øðà%9Ø#5ñð ñ	ˆLñ Ñ9˜,Ñ9°DÒ9Ð9ð ACŒG”E˜9Ó%×,Ñ,Ô=à*×=Ñ=ˆJÜ Ÿ;™; y°*À"ÔEˆLà ¨Ñ4Ü Ð!NÓOÐOð ×4Ñ4°\ÐCTÔUð ×&Ñ&Ð'8×'GÑ'GÔHà—]‘]ð ð *×-Ñ-Ð.>ÀÓEˆNØ'×+Ñ+¨N¸DÓAˆLØˆKØ+ò -�Ø!,¨{×/@Ñ/@ÀÑ/DÑ!D�Ø!Ð-Ø5CÂAÀÈÀÐDVÑ5W�LÐ!1Ñ2ØÐ+Ø3?ÂÀ;È~ÐC]Ð@]Ñ3^�L Ñ0ØA˜t×AÑAÀ+Ñ^ÐQ]Ñ^�á'ÑI¨,ÑIÀDÒI�à29×2IÑ2I�Ð.Ñ/Ø,‘ð-ð .<ˆLÐ)Ñ*Ø+7ˆL˜Ñ(ð ˆNro   )r�   rP   )NN)NNNNFr&  )r   FN)Fr   )NNN)NNNNNNNN)NF)NNNNNNNNNNN)FN©F)FNNNNN)T)Rre   rf   rg   rh   Úoutput_modalitiesr£   rv  r¬   ÚPathLikerH  r   rŽ   ri   rj   r  r   rm   rë   r  Údictrl   rù   rõ   r  r,   r   r  r4  rÂ   rB  ÚstaticmethodrI  r   rT  r8   r   r"   rk  r“  r�  rN   r§  r’  rÃ  rÖ  râ  ré  rî  rü  r  Úclassmethodr  r  r-   r-  r0  r=  rS  r   r[  re  rq  Úno_gradÚGenerateOutputrœ   r•  r€  ÚGenerateNonBeamOutputr[   rÞ  rà  rå  r.  rõ  rû  r  r  r  ÚGenerateBeamOutputr\   r]   rË  rn   ro   rp   r   r   R  s  „ ñð: "Ðð=Ø)ó=ðB CGØ)-ñ;(à'*¨R¯[©[Ñ'8¸4Ñ'?ð;(ð   $™;ð;(ð
 
ó;(ð@ ,0Ø(,Ø26Ø26Ø*/ñjØ)ðjà×#Ñ#ðjð " D™jðjð  ™ð	jð
 ×(Ñ(¨4Ñ/ðjð ×(Ñ(¨4Ñ/ðjð ! 4™KójðX?0Ø)ð?0à—‘˜tÑ#ð?0ð —l‘l TÑ)ð?0ð ˜3 §¡Ð,Ñ-ð	?0ð
 
ˆu�|‰|˜S 4™Z¨¨c°5·<±<Ð.?Ñ)@Ð@Ñ	Aó?0ðB&`Ø)ð&`à—‘˜tÑ#ð&`ð —l‘l TÑ)ð&`ð ˜3 §¡Ð,Ñ-ð	&`ð
 
×	Ñ	ó&`òPð0 à—|‘|ð ð ,ð ð ˜3 ˜8‘nð	 ð
 
×	Ñ	ó ðD'Ø)ð'à—|‘|ð'ð  ™*ð	'ð
 ,ð'ð 
ˆc�3ˆh‰ó'ð^ '+ñ9/Ø)ð9/àð9/ð ð9/ð ˜3 §¡Ð,Ñ-ð	9/ð
 !&§¡ð9/ð —‘˜tÑ#ð9/ð 
ˆu×Ñ  c¨5¯<©<Ð&7Ñ!8Ð8Ñ	9ó9/ðv àØ#(Ø-1ñ'Øð'à ð'ð ×#Ñ# dÑ*ð'ð
 
ˆu×Ñ  c¨3 h¡Ð/Ñ	0ò'ó ð'ðD $)Øñ6àð6ð ˜3 ˜8‘nð6ð !ð	6ð
 ð6ð 
ˆc�3ˆh‰ó6ð~ 8<Ø@DØCGñS#Ø)ðS#à+ðS#ð ×#Ñ#ðS#ð —|‘|ð	S#ð
 .ðS#ð ˜3 ˜8‘nðS#ð "Ð"3Ñ4ðS#ð #Ð#<Ñ=ðS#ð &Ð&?Ñ@ðS#ð 
óS#ðp ,0Ø59ØTXØ7;Ø!Ø.2Ø37Ø>Bñ]Ø)ð]à+ð]ð " D™jð]ð !×+Ñ+¨dÑ2ð	]ð
 #+¨C°·±Ð+>ÀÀSÁ	Ð+IÑ"JÈTÑ"Qð]ð .°Ñ4ð]ð �d‘
ð]ð ˜3 ˜8‘n tÑ+ð]ð #Ÿ\™\¨DÑ0ð]ð ).¯©°tÑ(;ð]ð 
ó]ðF :>ñ	$Ø)ð$à+ð$ð 0°$Ñ6ð$ð Ð5Ñ6ð	$ð
 
ó$ðL#à)Ð,@Ñ@ð#ð )Ð+?Ñ?ð#ð 
Ð3Ñ	3ó	#ðR -1Ø!&ñz!Ø)ðz!à—<‘<ðz!ð �e—l‘lÑ#ðz!ð —l‘l TÑ)ð	z!ð
 ðz!ð 
�‰óz!ðx.Ø)ó.ð`5Ð%@ð 5ÐPTÐUXÐZ]ÐU]ÑP^ó 5ðn,Ø)ó,ð\6!Ø)ó6!ðpA/Ø)ðA/à+¨dÑ2ðA/ð ðA/ð 
Ð Ð%Ñ	&ó	A/ðF1Ø)ð1ØADð1ØRUð1Øfið1à	ó1ðf ð
¨TÐ2MÑ-Nð 
ÐSWò 
ó ð
ð$kØ)ðkà+ðkð ðkð (ð	kð
 ðkð ðkð 
ókðZZÐ'Bð ZÀtó Zð 26Ø,0ñ	MSØ)ðMSà+ðMSð $(¨$¡;ðMSð —‘˜sÑ" TÑ)ó	MSð^9Ø)ð9Ø9=¸cÀ3¸h¹ð9Ø\lð9à	ó9ðv òQó ðQð* '+ñ	à'ðð  ðð ˜t™ð	ð
 
ˆt‰óð0&ð 
ˆc�3ˆh‰ó&ðB €U‡]�]ƒ_ð '+Ø59Ø7;Ø9=ØTXØ#'Ø7;Ø-1Ø37Ø>BØ15ñbØ)ðbà—‘˜tÑ#ðbð ,¨dÑ2ðbð .°Ñ4ð	bð
 0°$Ñ6ðbð #+¨C°·±Ð+>ÀÀSÁ	Ð+IÑ"JÈTÑ"Qðbð ˜D‘[ðbð "Ð"3Ñ4ðbð ˜>Ñ*ðbð #Ÿ\™\¨DÑ0ðbð ).¯©°tÑ(;ðbð ˜x™¨$Ñ.ðbð 
˜%×*Ñ*Ñ	*òbó ðbðH¸Dð Ètð Ð]b×]iÑ]ið Ðnró ð& ]añSØ×)Ñ)ðSØ6>Ð?XÑ6YðSà	×	Ñ	óSðv "Ø-1ñwØ)ðwà×#Ñ#ðwð .ðwð 0ð	wð
 ,ðwð ðwð ˜>Ñ*ðwð 
 ×!1Ñ!1Ñ	1ówðr ðH %§,¡,ð H°5·<±<ò Hó ðHð
 ðJ E§L¡Lð J¸cð JÈcð JÐV[×VbÑVbò Jó ðJð
 ð˜eŸl™lð ¸%¿,¹,ð È5Ï<É<ò ó ðð& ð,
Ø-2¯\©\ð,
à"Ÿ\™\ð,
ð —\‘\ð,
ð  Ÿ,™,ð	,
ð
 ð,
ð ð,
ð  ð,
ð ˜s™
ð,
ð ò,
ó ð,
ð\ ðMØ-2¯\©\ðMàŸ,™,ðMð ,1¯<©<ðMð ˜s™
ò	Mó ðMð,4Qà$Ÿ|™|ð4Qð !Ÿ<™<ð4Qð $Ÿl™lð	4Qð
 ð4Qð  ð4Qð ð4Qð ð4Qð ð4Qð ð4Qð ð4Qð 
ˆu�|‰|˜UŸ\™\¨5¯<©<Ð7Ñ	8ó4QðlLàŸ™ðLð !&§¡ðLð $)§<¡<ð	Lð
 ,1¯<©<ðLð ðLð 
ˆu�|‰|˜UŸ\™\¨5¯<©<Ð7Ñ	8óLð,3Fà—<‘<ð3Fð !&§¡ð3Fð —\‘\ð	3Fð
 Ÿ™ð3Fð —l‘lð3Fð $)§<¡<ð3Fð .3¯\©\ð3Fð  Ÿ,™,ð3Fð ,1¯<©<ð3Fð !Ÿ<™<ð3Fð ð3Fð ð3Fð  ð3Fð ð3Fð  ˜s™
ð!3Fð" 
ˆu�|‰|˜UŸ\™\¨5¯<©<¸¿¹ÐEÑ	Fó#3Fðz "ñ[Ø)ð[à×#Ñ#ð[ð .ð[ð 0ð	[ð
 ,ð[ð ð[ð 
˜e×.Ñ.Ñ	.ó[ðF "Ø-1Ø26Ø7;ØCGØ9=ñhØ)ðhà×#Ñ#ðhð .ðhð 0ð	hð
 ,ðhð ðhð ˜>Ñ*ðhð ×(Ñ(¨4Ñ/ðhð "Ð"3Ñ4ðhð &Ð&?Ñ@ðhð Ð5Ñ6ðhð 
 ×!1Ñ!1Ñ	1óhð`	 $(ñUØ)ðUà×#Ñ#ðUð ,ðUð ð	Uð
 !ôUro   r   c                 ód  — | dd…| d…f   }|j                  d¬«      }|dd…t        j                  |«      |f   j                  dd«      }|j                  d¬«      }|dd…t        j                  |«      |f   j                  dd«      }	|	|z  }
t        j                  |
«      }||
k  }| j                  d¬«      dk  j                  «       }|r||k(  r|dz  }|dd…d|dz   …f   }||fS |j                  d   }|dd…|dd…f   }||k  rF|dd…|dd…f   }t        j                  ||z
  d¬«      }|j                  |j                  «       «       n|}t        j                  |d¬«      j                  d«      ddd…f   }|dkD  r&t        j                  |dd…d|…f   |fd¬«      }||fS |}||fS )a  
    Applies sampling as in the speculative decoding paper (https://huggingface.co/papers/2211.17192, algorithm 1). Returns
    the selected tokens, as well as the number of candidate matches.

    NOTE: Unless otherwise stated, the variable names match those in the paper.
    Nr¾   r9  r   r   )rì  rº  )rÍ  ri   rÛ   rÏ  Ú	rand_liker  r¹  rÊ   ÚclampÚdiv_rÎ  r@  )rO  rP  rQ  rA  rR  Únew_candidate_input_idsÚqÚq_ir)  Úp_iÚprobability_ratioÚr_iÚis_acceptedrV  rU  ÚgammaÚ
p_n_plus_1Ú
q_n_plus_1Úp_primer�  s                       rp   rI  rI  à  sö  € ð 2²!Ð6FÐ5FÑ5GÐ2GÑHÐð 	× Ñ  RÐ Ó(€AØ
ŠAŒu�|‰|Ð,Ó-Ð/FÐFÑ
G×
OÑ
OÐPQÐSTÓ
U€CØ×Ñ˜rÐÓ"€AØ
ŠAŒu�|‰|Ð,Ó-Ð/FÐFÑ
G×
OÑ
OÐPQÐSTÓ
U€CØ˜c™	Ðô
 �/‰/Ð+Ó
,€CØÐ*Ñ*€KØ�,×&Ñ&¨2Ð&Ó.°Ñ2×7Ñ7Ó9€Iñ ˜YÐ*:Ò:ð 	�Q‰ˆ	Ø.ªq°/°IÀ±M°/Ð/AÑBˆð& ˜Ð"Ð"ð! !×&Ñ& qÑ)ˆØ’q˜)¢Q�Ñ'ˆ
Ø�uÒØš1˜iª˜?Ñ+ˆJÜ—k‘k :°
Ñ#:ÀÔCˆGØ�L‰L˜Ÿ™›Õ'à ˆGÜ×Ñ˜g°1Ô5×=Ñ=¸aÓ@ÀÂqÀÑIˆð �qŠ=Ü Ÿ9™9Ð&=ºaÀÀ)À¸mÑ&LÈaÐ%PÐVXÔYˆLð ˜Ð"Ð"ð ˆLà˜Ð"Ð"ro   c                 ó*  — t        | «      dk(  r<d}|D ]%  }|r|n|j                  d   }||dd|…d|…f   fz  }Œ' | |fz  } |dz  }||z  }t        |«      D ]:  }d}|D ]+  }|r||z   n|j                  d   }||d||dz   …d|…f   fz  }Œ- | |fz  } Œ< | S )z»
    Given the (decoder/cross attentions)/(decoder hidden states) for multiple generated tokens, splits it into a tuple
    where each member corresponds to a single generated token.
    r   rn   r¾   .Nr   )r  rÊ   r{  )	rJ  Únew_outputsrê  Ú	added_lenrD  Ú	new_tupleÚlayerÚlast_dim_sizerƒ  s	            rp   rL  rL    só   € ô ˆ7ƒ|�qÒØˆ	Ø ò 	AˆEÙ';™GÀÇÁÈRÁˆMØ˜%  X g X¨~°¨~Ð =Ñ>Ð@Ñ@‰Ið	Að 	�I�<Ñˆà�1‰ˆØ�WÑˆ	ä�9Óò  ˆØˆ	Ø ò 	BˆEÙ+?˜G ašKÀUÇ[Á[ÐQSÁ_ˆMØ˜%  Q¨¨Q© Y°°°Ð >Ñ?ÐAÑA‰Ið	Bð 	�I�<Ñ‰ð ð €Nro   rf  )ˆrô  rš   rÕ   r¬   rƒ  Úcollections.abcr   Ú
contextlibr   Údataclassesr   Útypingr   r   r   r	   ri   Útorch.distributedÚdistributedrh  r
   Úcache_utilsr   r   r   r   r   Údynamic_module_utilsr   r   r   r   Úintegrations.deepspeedr   Úintegrations.fsdpr   Úmasking_utilsr   Útokenization_pythonr   Úutilsr   r   r   r   Úutils.genericr   rj  r   r    r!   r"   r#   r$   r%   r&   r'   r(   Úconfiguration_utilsr)   r*   r+   r,   r-   Úcontinuous_batchingr.   Úlogits_processr/   r0   r1   r2   r3   r4   r5   r6   r7   r8   r9   r:   r;   r<   r=   r>   r?   r@   rA   rB   rC   rD   rE   rF   rG   rH   rž  rI   rJ   rK   rL   rM   rN   rO   Ú_typingrP   Úmodeling_utilsrQ   Útokenization_utils_baserR   Ú	streamersrS   Ú
get_loggerre   r—   Úaccelerate.hooksrT   rU   rM  ÚSAMPLEÚGREEDY_SEARCHrÌ  ÚBEAM_SAMPLErÍ  ÚDOLA_GENERATIONr&  ÚGROUP_BEAM_SEARCHÚCONSTRAINED_BEAM_SEARCHr`  r_   rr   ry   r}   rn  ro  rm  r   rI  rL  rn   ro   rp   ú<module>r¢     s  ðó Û Û Û 	Û Ý $Ý %Ý !ß 5Ó 5ã Ý  Ý ÷õ ÷ó õ @Ý 6Ý 5Ý 0÷ó õ 9÷÷ ÷ ÷õ õ 1÷÷ ÷ ÷ ÷ ÷ ÷ ÷8÷ ñ ñ Ý3Ý0ÝAÝ'à	ˆ×	Ñ	˜HÓ	%€áÔßEò€ð ×Ñ˜9Ø× Ñ  )Ø×Ñ Ø×Ñ Ø×&Ñ&Ð(<à×"Ñ"Ð$AØ×%Ñ%Ð'RØ×$Ñ$Ð&PØ×*Ñ*Ð,\ðÐ ð ô ) ó  )ó ð )ðF ô,) ;ó ,)ó ð,)ð^ ô() Kó ()ó ð()ðV ô5) {ó 5)ó ð5)ðr 2Ð4PÑPÐ Ø2Ð5UÑUÐ Ø&Ð);Ñ;€ôK6�oô K6ò\l5#ôpro   