§
    ‚Štj%  ã                   ó  — d Z ddlZddlZddlZddlmZ ddlmZm	Z	m
Z
mZmZ ddlmZmZmZmZmZ ddlmZ  e¦   «         rddlZ ej        e¦  «        Zdd	iZd
„ Zd„ Z ed¬¦  «         G d„ de¦  «        ¦   «         ZdgZdS )z!Tokenization class for Pop2Piano.é    Né   )ÚBatchFeature)Ú
AddedTokenÚBatchEncodingÚPaddingStrategyÚPreTrainedTokenizerÚTruncationStrategy)Ú
TensorTypeÚis_pretty_midi_availableÚloggingÚrequires_backendsÚto_numpy)ÚrequiresÚvocabz
vocab.jsonc                 ó4   — || z  }|�t          ||¦  «        }|S ©N)Úmin©ÚnumberÚcutoff_time_idxÚcurrent_idxs      úr/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/transformers/models/pop2piano/tokenization_pop2piano.pyÚtoken_time_to_noter   &   s'   € Ø�6Ñ€KØÐ"Ý˜+ Ñ7Ô7ˆàÐó    c                 ó’   — ||          �9||          }||k     r*|}|                      ||| |g¦  «         |dk    rd n|}||| <   n||| <   |S )Nr   )Úappend)	r   Úcurrent_velocityÚdefault_velocityÚnote_onsets_readyr   ÚnotesÚ	onset_idxÚ
offset_idxÚonsets_readys	            r   Útoken_note_to_noter$   .   sr   € Ø˜Ô Ð,à% fÔ-ˆ	Ø�{Ò"Ð"à$ˆJØ�LŠL˜) Z°Ð9IÐJÑKÔKÐKØ#3°qÒ#8Ð#8˜4˜4¸kˆLØ(4Ð˜fÑ%øà$/Ð˜&Ñ!Ø€Lr   )Úpretty_midiÚtorch)Úbackendsc                   ó¼  ‡ — e Zd ZdZddgZeZ	 	 	 	 	 	 d5ˆ fd
„	Zed„ ¦   «         Z	d„ Z
dedefd„Zd6defd„Zdej        dededefd„Z	 	 	 d7dej        dej        dededef
d„Zd8dej        dededz  fd„Zd9dej        dej        d efd!„Zd8d"ed#edz  dee         fd$„Z	 	 d:dej        eej                 z  d%edz  d&edz  defd'„Z	 	 d:dej        eej                 z  d%edz  d&edz  defd(„Z	 	 	 	 	 	 	 d;dej        eej                 z  eeej                          z  d+eez  e z  d,eez  ez  d&edz  d-edz  d.edz  d/ee!z  dz  d0edefd1„Z"	 d<d2e#d3efd4„Z$ˆ xZ%S )=ÚPop2PianoTokenizeraš  
    Constructs a Pop2Piano tokenizer. This tokenizer does not require training.

    This tokenizer inherits from [`PreTrainedTokenizer`] which contains most of the main methods. Users should refer to
    this superclass for more information regarding those methods.

    Args:
        vocab (`str`):
            Path to the vocab file which contains the vocabulary.
        default_velocity (`int`, *optional*, defaults to 77):
            Determines the default velocity to be used while creating midi Notes.
        num_bars (`int`, *optional*, defaults to 2):
            Determines cutoff_time_idx in for each token.
        unk_token (`str` or `tokenizers.AddedToken`, *optional*, defaults to `"-1"`):
            The unknown token. A token that is not in the vocabulary cannot be converted to an ID and is set to be this
            token instead.
        eos_token (`str` or `tokenizers.AddedToken`, *optional*, defaults to 1):
            The end of sequence token.
        pad_token (`str` or `tokenizers.AddedToken`, *optional*, defaults to 0):
             A special token used to make arrays of tokens the same size for batching purpose. Will then be ignored by
            attention mechanisms or loss computation.
        bos_token (`str` or `tokenizers.AddedToken`, *optional*, defaults to 2):
            The beginning of sequence token that was used during pretraining. Can be used a sequence classifier token.
    Ú	token_idsÚattention_maskéM   é   ú-1Ú1Ú0Ú2c                 óz  •— t          |t          ¦  «        rt          |dd¬¦  «        n|}t          |t          ¦  «        rt          |dd¬¦  «        n|}t          |t          ¦  «        rt          |dd¬¦  «        n|}t          |t          ¦  «        rt          |dd¬¦  «        n|}|| _        || _        t          |d¦  «        5 }	t          j        |	¦  «        | _        d d d ¦  «         n# 1 swxY w Y   d„ | j         	                    ¦   «         D ¦   «         | _
         t          ¦   «         j        d||||dœ|¤Ž d S )NF)ÚlstripÚrstripÚrbc                 ó   — i | ]\  }}||“Œ	S © r7   )Ú.0ÚkÚvs      r   ú
<dictcomp>z/Pop2PianoTokenizer.__init__.<locals>.<dictcomp>s   s   € Ð>Ð>Ð>¡  A˜˜1Ð>Ð>Ð>r   )Ú	unk_tokenÚ	eos_tokenÚ	pad_tokenÚ	bos_tokenr7   )Ú
isinstanceÚstrr   r   Únum_barsÚopenÚjsonÚloadÚencoderÚitemsÚdecoderÚsuperÚ__init__)Úselfr   r   rB   r<   r=   r>   r?   ÚkwargsÚfileÚ	__class__s             €r   rJ   zPop2PianoTokenizer.__init__[   sŸ  ø€ õ JTÐT]Õ_bÑIcÔIcÐr•J˜y°¸uÐEÑEÔEÐEÐirˆ	ÝISÐT]Õ_bÑIcÔIcÐr•J˜y°¸uÐEÑEÔEÐEÐirˆ	ÝISÐT]Õ_bÑIcÔIcÐr•J˜y°¸uÐEÑEÔEÐEÐirˆ	ÝISÐT]Õ_bÑIcÔIcÐr•J˜y°¸uÐEÑEÔEÐEÐirˆ	à 0ˆÔØ ˆŒõ �%˜ÑÔð 	+ $Ýœ9 T™?œ?ˆDŒLð	+ð 	+ð 	+ñ 	+ô 	+ð 	+ð 	+ð 	+ð 	+ð 	+ð 	+øøøð 	+ð 	+ð 	+ð 	+ð ?Ð>¨¬×);Ò);Ñ)=Ô)=Ð>Ñ>Ô>ˆŒà�‰ŒÔð 	
ØØØØð		
ð 	
ð
 ð	
ð 	
ð 	
ð 	
ð 	
s   ÃC*Ã*C.Ã1C.c                 ó*   — t          | j        ¦  «        S )z-Returns the vocabulary size of the tokenizer.)ÚlenrF   ©rK   s    r   Ú
vocab_sizezPop2PianoTokenizer.vocab_size}   s   € õ �4”<Ñ Ô Ð r   c                 ó0   — t          | j        fi | j        ¤ŽS )z(Returns the vocabulary of the tokenizer.)ÚdictrF   Úadded_tokens_encoderrQ   s    r   Ú	get_vocabzPop2PianoTokenizer.get_vocab‚   s   € å�D”LÐ>Ð> DÔ$=Ð>Ð>Ð>r   Útoken_idÚreturnc                 óÞ   — | j                              || j        › d�¦  «        }|                     d¦  «        }d                     |dd…         ¦  «        t          |d         ¦  «        }}||gS )a?  
        Decodes the token ids generated by the transformer into notes.

        Args:
            token_id (`int`):
                This denotes the ids generated by the transformers to be converted to Midi tokens.

        Returns:
            `List`: A list consists of token_type (`str`) and value (`int`).
        Ú_TOKEN_TIMEÚ_é   Nr   )rH   Úgetr<   ÚsplitÚjoinÚint)rK   rW   Útoken_type_valueÚ
token_typeÚvalues        r   Ú_convert_id_to_tokenz'Pop2PianoTokenizer._convert_id_to_token†   st   € ð  œ<×+Ò+¨H¸¼Ð6TÐ6TÐ6TÑUÔUÐØ+×1Ò1°#Ñ6Ô6ÐØŸHšHÐ%5°a°b°bÔ%9Ñ:Ô:½CÐ@PÐQRÔ@SÑ<TÔ<T�Eˆ
à˜EÐ"Ð"r   Ú
TOKEN_TIMEc                 óf   — | j                              |› d|› �t          | j        ¦  «        ¦  «        S )a¹  
        Encodes the Midi tokens to transformer generated token ids.

        Args:
            token (`int`):
                This denotes the token value.
            token_type (`str`):
                This denotes the type of the token. There are four types of midi tokens such as "TOKEN_TIME",
                "TOKEN_VELOCITY", "TOKEN_NOTE" and "TOKEN_SPECIAL".

        Returns:
            `int`: returns the id of the token.
        r[   )rF   r]   r`   r<   )rK   Útokenrb   s      r   Ú_convert_token_to_idz'Pop2PianoTokenizer._convert_token_to_id˜   s4   € ð Œ|×Ò 5Ð 7Ð 7¨:Ð 7Ð 7½¸T¼^Ñ9LÔ9LÑMÔMÐMr   ÚtokensÚbeat_offset_idxÚbars_per_batchr   c                 ó  — d}t          t          |¦  «        ¦  «        D ]c}||         }|||z  dz  z   }||z   }	|                      |||	¬¦  «        }
t          |
¦  «        dk    rŒF|€|
}ŒKt          j        ||
fd¬¦  «        }Œd|€g S |S )a  
        Converts relative tokens to notes which are then used to generate pretty midi object.

        Args:
            tokens (`numpy.ndarray`):
                Tokens to be converted to notes.
            beat_offset_idx (`int`):
                Denotes beat offset index for each note in generated Midi.
            bars_per_batch (`int`):
                A parameter to control the Midi output generation.
            cutoff_time_idx (`int`):
                Denotes the cutoff time index for each note in generated Midi.
        Né   )Ú	start_idxr   r   )Úaxis)ÚrangerP   Úrelative_tokens_ids_to_notesÚnpÚconcatenate)rK   ri   rj   rk   r   r    ÚindexÚ_tokensÚ
_start_idxÚ_cutoff_time_idxÚ_notess              r   Ú"relative_batch_tokens_ids_to_notesz5Pop2PianoTokenizer.relative_batch_tokens_ids_to_notes¨   s¾   € ð* ˆå�3˜v™;œ;Ñ'Ô'ð 	@ð 	@ˆEØ˜U”mˆGØ(¨5°>Ñ+AÀAÑ+EÑEˆJØ.°Ñ;ÐØ×6Ò6ØØ$Ø 0ð 7ñ ô ˆFõ �6‰{Œ{˜aÒÐØØ�Ø��åœ¨¨v ¸QÐ?Ñ?Ô?��àˆ=ØˆIØˆr   r   é   Úbeatstepc                 ó€   — |€dn|}|                       ||||¬¦  «        }|                      ||||         ¬¦  «        }|S )al  
        Converts tokens to Midi. This method calls `relative_batch_tokens_ids_to_notes` method to convert batch tokens
        to notes then uses `notes_to_midi` method to convert them to Midi.

        Args:
            tokens (`numpy.ndarray`):
                Denotes tokens which alongside beatstep will be converted to Midi.
            beatstep (`np.ndarray`):
                We get beatstep from feature extractor which is also used to get Midi.
            beat_offset_idx (`int`, *optional*, defaults to 0):
                Denotes beat offset index for each note in generated Midi.
            bars_per_batch (`int`, *optional*, defaults to 2):
                A parameter to control the Midi output generation.
            cutoff_time_idx (`int`, *optional*, defaults to 12):
                Denotes the cutoff time index for each note in generated Midi.
        Nr   )ri   rj   rk   r   )Ú
offset_sec)ry   Únotes_to_midi)rK   ri   r{   rj   rk   r   r    Úmidis           r   Ú!relative_batch_tokens_ids_to_midiz4Pop2PianoTokenizer.relative_batch_tokens_ids_to_midiÔ   s^   € ð0  /Ð6˜!˜!¸OˆØ×7Ò7ØØ+Ø)Ø+ð	 8ñ 
ô 
ˆð ×!Ò! %¨¸hÀÔ>WÐ!ÑXÔXˆØˆr   Nrn   c           	      óî  ‡ — ˆ fd„|D ¦   «         }|}d}d„ t          t          d„ ‰ j        D ¦   «         ¦  «        dz   ¦  «        D ¦   «         }g }|D ]e\  }	}
|	dk    r	|
dk    r nSŒ|	dk    rt          |
||¬¦  «        }Œ-|	d	k    r|
}Œ6|	d
k    rt	          |
|‰ j        |||¬¦  «        }ŒWt          d¦  «        ‚t          |¦  «        D ]P\  }}|�I|€|dz   }nt          ||dz   ¦  «        }t          ||¦  «        }| 	                    |||‰ j        g¦  «         ŒQt          |¦  «        dk    rg S t          j        |¦  «        }|dd…df         dz  |dd…df         z   }||                     ¦   «                  }|S )a¶  
        Converts relative tokens to notes which will then be used to create Pretty Midi objects.

        Args:
            tokens (`numpy.ndarray`):
                Relative Tokens which will be converted to notes.
            start_idx (`float`):
                A parameter which denotes the starting index.
            cutoff_time_idx (`float`, *optional*):
                A parameter used while converting tokens to notes.
        c                 ó:   •— g | ]}‰                      |¦  «        ‘ŒS r7   )rd   )r8   rg   rK   s     €r   ú
<listcomp>zCPop2PianoTokenizer.relative_tokens_ids_to_notes.<locals>.<listcomp>  s'   ø€ ÐFÐFÐF°e�×*Ò*¨5Ñ1Ô1ÐFÐFÐFr   r   c                 ó   — g | ]}d ‘ŒS r   r7   ©r8   Úis     r   rƒ   zCPop2PianoTokenizer.relative_tokens_ids_to_notes.<locals>.<listcomp>  s   € ÐdÐdÐd a˜TÐdÐdÐdr   c              3   ó@   K  — | ]}|                      d ¦  «        V — ŒdS )ÚNOTEN)Úendswith)r8   r9   s     r   ú	<genexpr>zBPop2PianoTokenizer.relative_tokens_ids_to_notes.<locals>.<genexpr>  s.   è è € Ð4^Ð4^ÈA°Q·Z²ZÀÑ5GÔ5GÐ4^Ð4^Ð4^Ð4^Ð4^Ð4^r   r\   ÚTOKEN_SPECIALre   r   ÚTOKEN_VELOCITYÚ
TOKEN_NOTE)r   r   r   r   r   r    zToken type not understood!Né€   )rp   ÚsumrF   r   r$   r   Ú
ValueErrorÚ	enumerateÚmaxr   rP   rr   ÚarrayÚargsort)rK   ri   rn   r   Úwordsr   r   r   r    rb   r   ÚpitchÚ
note_onsetÚcutoffr"   Ú
note_orders   `               r   rq   z/Pop2PianoTokenizer.relative_tokens_ids_to_notesø   sÿ  ø€ ð GÐFÐFÐF¸vÐFÑFÔFˆàˆØÐØdÐd­5µÐ4^Ð4^ÐQUÔQ]Ð4^Ñ4^Ô4^Ñ1^Ô1^ÐabÑ1bÑ+cÔ+cÐdÑdÔdÐØˆØ"'ð 	?ð 	?ÑˆJ˜Ø˜_Ò,Ð,Ø˜Q’;�;Ø�Eð à˜|Ò+Ð+Ý0Ø!°?ÐP[ðñ ô ��ð Ð/Ò/Ð/Ø#)Ð Ð à˜|Ò+Ð+Ý*Ø!Ø%5Ø%)Ô%:Ø&7Ø +Øðñ ô ��õ !Ð!=Ñ>Ô>Ð>å!*Ð+<Ñ!=Ô!=ð 		Uð 		UÑˆE�:àÐ%Ø"Ð*Ø'¨!™^�F�Få  °*¸q±.ÑAÔA�Få  ¨fÑ5Ô5�
Ø—’˜j¨*°e¸TÔ=RÐSÑTÔTÐTøåˆu‰:Œ:˜Š?ˆ?ØˆIå”H˜U‘O”OˆEØ˜q˜q˜q !˜tœ sÑ*¨U°1°1°1°a°4¬[Ñ8ˆJØ˜*×,Ò,Ñ.Ô.Ô/ˆEØˆLr   ç        r    r}   c                 ó~  — t          | dg¦  «         t          j        dd¬¦  «        }t          j        d¬¦  «        }g }|D ]F\  }}}	}
t          j        |
|	||         |z
  ||         |z
  ¬¦  «        }|                     |¦  «         ŒG||_        |j                             |¦  «         |                     ¦   «          |S )a»  
        Converts notes to Midi.

        Args:
            notes (`numpy.ndarray`):
                This is used to create Pretty Midi objects.
            beatstep (`numpy.ndarray`):
                This is the extrapolated beatstep that we get from feature extractor.
            offset_sec (`int`, *optional*, defaults to 0.0):
                This represents the offset seconds which is used while creating each Pretty Midi Note.
        r%   i€  g      ^@)Ú
resolutionÚinitial_tempor   )Úprogram)Úvelocityr–   ÚstartÚend)	r   r%   Ú
PrettyMIDIÚ
InstrumentÚNoter   r    ÚinstrumentsÚremove_invalid_notes)rK   r    r{   r}   Únew_pmÚnew_instÚ	new_notesr!   r"   r–   rŸ   Únew_notes               r   r~   z Pop2PianoTokenizer.notes_to_midi4  sà   € õ 	˜$  Ñ0Ô0Ð0åÔ'°3ÀeÐLÑLÔLˆÝÔ)°!Ð4Ñ4Ô4ˆØˆ	à6;ð 	'ð 	'Ñ2ˆI�z 5¨(Ý"Ô'Ø!ØØ˜yÔ)¨JÑ6Ø˜ZÔ(¨:Ñ5ð	ñ ô ˆHð ×Ò˜XÑ&Ô&Ð&Ð&Ø"ˆŒØÔ×!Ò! (Ñ+Ô+Ð+Ø×#Ò#Ñ%Ô%Ð%Øˆr   Úsave_directoryÚfilename_prefixc                 ó˜  — t           j                             |¦  «        s t                               d|› d�¦  «         dS t           j                             ||r|dz   ndt          d         z   ¦  «        }t          |d¦  «        5 }|                     t          j
        | j        ¦  «        ¦  «         ddd¦  «         n# 1 swxY w Y   |fS )a}  
        Saves the tokenizer's vocabulary dictionary to the provided save_directory.

        Args:
            save_directory (`str`):
                A path to the directory where to saved. It will be created if it doesn't exist.
            filename_prefix (`Optional[str]`, *optional*):
                A prefix to add to the names of the files saved by the tokenizer.
        zVocabulary path (z) should be a directoryNú-Ú r   Úw)ÚosÚpathÚisdirÚloggerÚerrorr_   ÚVOCAB_FILES_NAMESrC   ÚwriterD   ÚdumpsrF   )rK   r«   r¬   Úout_vocab_filerM   s        r   Úsave_vocabularyz"Pop2PianoTokenizer.save_vocabularyT  s  € õ Œw�}Š}˜^Ñ,Ô,ð 	Ý�LŠLÐT¨^ÐTÐTÐTÑUÔUÐUØˆFõ œŸšØ°oÐM˜_¨sÑ2Ð2È2ÕQbÐcjÔQkÑkñ
ô 
ˆõ �. #Ñ&Ô&ð 	1¨$Ø�JŠJ•t”z $¤,Ñ/Ô/Ñ0Ô0Ð0ð	1ð 	1ð 	1ñ 	1ô 	1ð 	1ð 	1ð 	1ð 	1ð 	1ð 	1øøøð 	1ð 	1ð 	1ð 	1ð Ð Ð s   Â-B>Â>CÃCÚtruncation_strategyÚ
max_lengthc                 ó`  — t          | dg¦  «         t          |d         t          j        ¦  «        r2t	          j        d„ |D ¦   «         ¦  «                             dd¦  «        }t	          j        |¦  «                             t          j	        ¦  «        }|dd…dd…f          
                    ¦   «         }d„ t          |d	z   ¦  «        D ¦   «         }|D ]A\  }}}	}
||                              |	|
g¦  «         ||                              |	dg¦  «         ŒBg }d}t          |¦  «        D ]·\  }}t          |¦  «        dk    rŒ|                     |                      |d
¦  «        ¦  «         |D ]r\  }	}
t!          |
dk    ¦  «        }
||
k    r+|
}|                     |                      |
d¦  «        ¦  «         |                     |                      |	d¦  «        ¦  «         ŒsŒ¸t          |¦  «        }|t"          j        k    r |r||k    r | j        d|||z
  |dœ|¤Ž\  }}}t)          d|i¦  «        S )až  
        This is the `encode_plus` method for `Pop2PianoTokenizer`. It converts the midi notes to the transformer
        generated token ids. It only works on a single batch, to process multiple batches please use
        `batch_encode_plus` or `__call__` method.

        Args:
            notes (`numpy.ndarray` of shape `[sequence_length, 4]` or `list` of `pretty_midi.Note` objects):
                This represents the midi notes. If `notes` is a `numpy.ndarray`:
                    - Each sequence must have 4 values, they are `onset idx`, `offset idx`, `pitch` and `velocity`.
                If `notes` is a `list` containing `pretty_midi.Note` objects:
                    - Each sequence must have 4 attributes, they are `start`, `end`, `pitch` and `velocity`.
            truncation_strategy ([`~tokenization_utils_base.TruncationStrategy`], *optional*):
                Indicates the truncation strategy that is going to be used during truncation.
            max_length (`int`, *optional*):
                Maximum length of the returned list and optionally padding length (see above).

        Returns:
            `BatchEncoding` containing the tokens ids.
        r%   r   c                 óB   — g | ]}|j         |j        |j        |j        g‘ŒS r7   )r    r¡   r–   rŸ   )r8   Ú	each_notes     r   rƒ   z2Pop2PianoTokenizer.encode_plus.<locals>.<listcomp>Œ  s+   € ÐnÐnÐnÐ[d�)”/ 9¤=°)´/À9ÔCUÐVÐnÐnÐnr   éÿÿÿÿrm   Nr-   c                 ó   — g | ]}g ‘ŒS r7   r7   r…   s     r   rƒ   z2Pop2PianoTokenizer.encode_plus.<locals>.<listcomp>“  s   € Ð5Ð5Ð5˜�Ð5Ð5Ð5r   r\   re   rŒ   r�   )ÚidsÚnum_tokens_to_remover»   r*   r7   )r   r@   r%   r¤   rr   r“   ÚreshapeÚroundÚastypeÚint32r’   rp   r   r‘   rP   rh   r`   r	   ÚDO_NOT_TRUNCATEÚtruncate_sequencesr   )rK   r    r»   r¼   rL   Úmax_time_idxÚtimesÚonsetÚoffsetr–   rŸ   ri   r   r†   ÚtimeÚ	total_lenr[   s                    r   Úencode_pluszPop2PianoTokenizer.encode_plusk  sy  € õ6 	˜$  Ñ0Ô0Ð0õ �e˜A”h¥Ô 0Ñ1Ô1ð 	Ý”HØnÐnÐhmÐnÑnÔnñô çŠg�b˜!‰nŒnð õ
 ”˜‘”×&Ò&¥r¤xÑ0Ô0ˆØ˜Q˜Q˜Q   ˜U”|×'Ò'Ñ)Ô)ˆà5Ð5�U <°!Ñ#3Ñ4Ô4Ð5Ñ5Ô5ˆØ.3ð 	-ð 	-Ñ*ˆE�6˜5 (Ø�%ŒL×Ò ¨Ð 1Ñ2Ô2Ð2Ø�&ŒM× Ò  %¨ Ñ,Ô,Ð,Ð,àˆØÐÝ  Ñ'Ô'ð 		Nð 		N‰GˆAˆtÝ�4‰yŒy˜AŠ~ˆ~ØØ�MŠM˜$×3Ò3°A°|ÑDÔDÑEÔEÐEØ#'ð Nð N‘��xÝ˜x¨!š|Ñ,Ô,�Ø# xÒ/Ð/Ø'/Ð$Ø—M’M $×";Ò";¸HÐFVÑ"WÔ"WÑXÔXÐXØ—’˜d×7Ò7¸¸|ÑLÔLÑMÔMÐMÐMðNõ ˜‘K”Kˆ	ð Õ"4Ô"DÒDÐDÈÐDÐXaÐdnÒXnÐXnØ2˜4Ô2ð ØØ%.°Ñ%;Ø$7ðð ð ð	ð ‰LˆF�A�qõ ˜k¨6Ð2Ñ3Ô3Ð3r   c           	      óÆ   — g }t          t          |¦  «        ¦  «        D ]2}|                      | j        ||         f||dœ|¤Žd         ¦  «         Œ3t	          d|i¦  «        S )a†  
        This is the `batch_encode_plus` method for `Pop2PianoTokenizer`. It converts the midi notes to the transformer
        generated token ids. It works on multiple batches by calling `encode_plus` multiple times in a loop.

        Args:
            notes (`numpy.ndarray` of shape `[batch_size, sequence_length, 4]` or `list` of `pretty_midi.Note` objects):
                This represents the midi notes. If `notes` is a `numpy.ndarray`:
                    - Each sequence must have 4 values, they are `onset idx`, `offset idx`, `pitch` and `velocity`.
                If `notes` is a `list` containing `pretty_midi.Note` objects:
                    - Each sequence must have 4 attributes, they are `start`, `end`, `pitch` and `velocity`.
            truncation_strategy ([`~tokenization_utils_base.TruncationStrategy`], *optional*):
                Indicates the truncation strategy that is going to be used during truncation.
            max_length (`int`, *optional*):
                Maximum length of the returned list and optionally padding length (see above).

        Returns:
            `BatchEncoding` containing the tokens ids.
        )r»   r¼   r*   )rp   rP   r   rÐ   r   )rK   r    r»   r¼   rL   Úencoded_batch_token_idsr†   s          r   Úbatch_encode_plusz$Pop2PianoTokenizer.batch_encode_plus²  s—   € ð4 #%ÐÝ•s˜5‘z”zÑ"Ô"ð 	ð 	ˆAØ#×*Ò*Ø �Ô Ø˜!”Hðà(;Ø)ðð ð ð	ð ð
 ôñô ð ð õ ˜kÐ+BÐCÑDÔDÐDr   FTÚpaddingÚ
truncationÚpad_to_multiple_ofÚreturn_attention_maskÚreturn_tensorsÚverbosec	           	      óD  — t          |t          j        ¦  «        r|j        dk    nt          |d         t          ¦  «        }
 | j        d|||||dœ|	¤Ž\  }}}}	|
r|€dn|} | j        d|||dœ|	¤Ž}n | j        d|||dœ|	¤Ž}|                      |||||||¬¦  «        }|S )	ay  
        This is the `__call__` method for `Pop2PianoTokenizer`. It converts the midi notes to the transformer generated
        token ids.

        Args:
            notes (`numpy.ndarray` of shape `[batch_size, max_sequence_length, 4]` or `list` of `pretty_midi.Note` objects):
                This represents the midi notes.

                If `notes` is a `numpy.ndarray`:
                    - Each sequence must have 4 values, they are `onset idx`, `offset idx`, `pitch` and `velocity`.
                If `notes` is a `list` containing `pretty_midi.Note` objects:
                    - Each sequence must have 4 attributes, they are `start`, `end`, `pitch` and `velocity`.
            padding (`bool`, `str` or [`~file_utils.PaddingStrategy`], *optional*, defaults to `False`):
                Activates and controls padding. Accepts the following values:

                - `True` or `'longest'`: Pad to the longest sequence in the batch (or no padding if only a single
                  sequence if provided).
                - `'max_length'`: Pad to a maximum length specified with the argument `max_length` or to the maximum
                  acceptable input length for the model if that argument is not provided.
                - `False` or `'do_not_pad'` (default): No padding (i.e., can output a batch with sequences of different
                  lengths).
            truncation (`bool`, `str` or [`~tokenization_utils_base.TruncationStrategy`], *optional*, defaults to `False`):
                Activates and controls truncation. Accepts the following values:

                - `True` or `'longest_first'`: Truncate to a maximum length specified with the argument `max_length` or
                  to the maximum acceptable input length for the model if that argument is not provided. This will
                  truncate token by token, removing a token from the longest sequence in the pair if a pair of
                  sequences (or a batch of pairs) is provided.
                - `'only_first'`: Truncate to a maximum length specified with the argument `max_length` or to the
                  maximum acceptable input length for the model if that argument is not provided. This will only
                  truncate the first sequence of a pair if a pair of sequences (or a batch of pairs) is provided.
                - `'only_second'`: Truncate to a maximum length specified with the argument `max_length` or to the
                  maximum acceptable input length for the model if that argument is not provided. This will only
                  truncate the second sequence of a pair if a pair of sequences (or a batch of pairs) is provided.
                - `False` or `'do_not_truncate'` (default): No truncation (i.e., can output batch with sequence lengths
                  greater than the model maximum admissible input size).
            max_length (`int`, *optional*):
                Controls the maximum length to use by one of the truncation/padding parameters. If left unset or set to
                `None`, this will use the predefined model maximum length if a maximum length is required by one of the
                truncation/padding parameters. If the model has no specific maximum input length (like XLNet)
                truncation/padding to a maximum length will be deactivated.
            pad_to_multiple_of (`int`, *optional*):
                If set will pad the sequence to a multiple of the provided value. This is especially useful to enable
                the use of Tensor Cores on NVIDIA hardware with compute capability `>= 7.5` (Volta).
            return_attention_mask (`bool`, *optional*):
                Whether to return the attention mask. If left to the default, will return the attention mask according
                to the specific tokenizer's default, defined by the `return_outputs` attribute.

                [What are attention masks?](../glossary#attention-mask)
            return_tensors (`str` or [`~file_utils.TensorType`], *optional*):
                If set, will return tensors instead of list of python integers. Acceptable values are:

                - `'pt'`: Return PyTorch `torch.Tensor` objects.
                - `'np'`: Return Numpy `np.ndarray` objects.
            verbose (`bool`, *optional*, defaults to `True`):
                Whether or not to print more information and warnings.

        Returns:
            `BatchEncoding` containing the token_ids.
        r   r   )rÔ   rÕ   r¼   rÖ   rÙ   NT)r    r»   r¼   )rÔ   r¼   rÖ   r×   rØ   rÙ   r7   )	r@   rr   ÚndarrayÚndimÚlistÚ"_get_padding_truncation_strategiesrÓ   rÐ   Úpad)rK   r    rÔ   rÕ   r¼   rÖ   r×   rØ   rÙ   rL   Ú
is_batchedÚpadding_strategyr»   r*   s                 r   Ú__call__zPop2PianoTokenizer.__call__Ù  s9  € õZ )3°5½"¼*Ñ(EÔ(EÐe�U”Z 1’_�_Í:ÐV[Ð\]ÔV^Õ`dÑKeÔKeˆ
ð ElÀDÔDkð E
ØØ!Ø!Ø1ØðE
ð E
ð ðE
ð E
ÑAÐÐ-¨z¸6ð ð 	à,AÐ,I D DÐOdÐ!Ø.˜Ô.ð ØØ$7Ø%ðð ð ð	ð ˆIˆIð )˜Ô(ð ØØ$7Ø%ðð ð ð	ð ˆIð —H’HØØ$Ø!Ø1Ø"7Ø)Øð ñ 
ô 
ˆ	ð Ðr   Úfeature_extractor_outputÚreturn_midic                 ó8  — t          t          |d¦  «        ot          |d¦  «        ot          |d¦  «        ¦  «        }|s&|d         j        d         dk    rt          d¦  «        ‚|rùt	          |d         dd…df         dk    ¦  «        |d         j        d         k    s(|d         j        d         |d	         j        d         k    rEt          d
|j        d         › d|d         j        d         › d|d	         j        d         › �¦  «        ‚|d         j        d         |j        d         k    r1t          d|d         j        d         › d|j        d         › �¦  «        ‚nf|d         j        d         dk    s|d	         j        d         dk    r8t          d|d         j        d         › d|d	         j        d         › d�¦  «        ‚|r/t          j        |d         dd…df         dk    ¦  «        d         }n|j        d         g}g }g }d}t          |¦  «        D �]Ú\  }	}
|||
…         }|dd…dt          j        t          j        |t          | j
        ¦  «        k    ¦  «        d         ¦  «        dz   …f         }|d         |	         }|d	         |	         }|r’|d         |	         }|d         |	         }|dt          j        t          j        |dk    ¦  «        d         ¦  «        dz   …         }|dt          j        t          j        |dk    ¦  «        d         ¦  «        dz   …         }t          |¦  «        }t          |¦  «        }t          |¦  «        }|                      ||| j        | j        dz   dz  ¬¦  «        }|j        d         j        D ]C}|xj        |d         z  c_        |xj        |d         z  c_        |                     |¦  «         ŒD|                     |¦  «         ||
dz   z  }�ŒÜ|rt'          ||dœ¦  «        S t'          d|i¦  «        S )a;  
        This is the `batch_decode` method for `Pop2PianoTokenizer`. It converts the token_ids generated by the
        transformer to midi_notes and returns them.

        Args:
            token_ids (`Union[np.ndarray, torch.Tensor]`):
                Output token_ids of `Pop2PianoConditionalGeneration` model.
            feature_extractor_output (`BatchFeature`):
                Denotes the output of `Pop2PianoFeatureExtractor.__call__`. It must contain `"beatstep"` and
                `"extrapolated_beatstep"`. Also `"attention_mask_beatsteps"` and
                `"attention_mask_extrapolated_beatstep"`
                 should be present if they were returned by the feature extractor.
            return_midi (`bool`, *optional*, defaults to `True`):
                Whether to return midi object or not.
        Returns:
            If `return_midi` is True:
                - `BatchEncoding` containing both `notes` and `pretty_midi.pretty_midi.PrettyMIDI` objects.
            If `return_midi` is False:
                - `BatchEncoding` containing `notes`.
        r+   Úattention_mask_beatstepsÚ$attention_mask_extrapolated_beatstepÚ	beatstepsr   r\   z—attention_mask, attention_mask_beatsteps and attention_mask_extrapolated_beatstep must be present for batched inputs! But one of them were not present.NÚextrapolated_beatstepzbLength mistamtch between token_ids, beatsteps and extrapolated_beatstep! Found token_ids length - z, beatsteps shape - z$ and extrapolated_beatsteps shape - z!Found attention_mask of length - z but token_ids of length - zœLength mistamtch of beatsteps and extrapolated_beatstep! Since attention_mask is not present the number of examples must be 1, But found beatsteps length - z", extrapolated_beatsteps length - ú.rm   )ri   r{   rk   r   )r    Úpretty_midi_objectsr    )ÚboolÚhasattrÚshaper�   r�   rr   Úwherer‘   r’   r`   r=   r   r€   rB   r¥   r    r    r¡   r   r   )rK   r*   rã   rä   Úattention_masks_presentÚ	batch_idxÚ
notes_listÚpretty_midi_objects_listrn   rt   Úend_idxÚeach_tokens_idsrè   ré   ræ   rç   Úpretty_midi_objectÚnotes                     r   Úbatch_decodezPop2PianoTokenizer.batch_decodeP  sF  € õ8 #'ÝÐ,Ð.>Ñ?Ô?ð ZÝÐ0Ð2LÑMÔMðZåÐ0Ð2XÑYÔYñ#
ô #
Ðð 'ð 	Ð+CÀKÔ+PÔ+VÐWXÔ+YÐ\]Ò+]Ð+]ÝðHñô ð ð #ð 	õ Ð,Ð-=Ô>¸q¸q¸qÀ!¸tÔDÈÒIÑJÔJØ+¨KÔ8Ô>¸qÔAòBð Bà+¨KÔ8Ô>¸qÔAØ+Ð,CÔDÔJÈ1ÔMòNð Nõ !ðwØ*3¬/¸!Ô*<ðwð wØRjÐkvÔRwÔR}Ð~ô  SAðwð wà:RÐSjÔ:kÔ:qÐrsÔ:tðwð wñô ð ð
 (Ð(8Ô9Ô?ÀÔBÀiÄoÐVWÔFXÒXÐXÝ ð ]Ð8PÐQaÔ8bÔ8hÐijÔ8kð  ]ð  ]ð  IRô  IXð  YZô  I[ð  ]ð  ]ñô ð ð Yð )¨Ô5Ô;¸AÔ>À!ÒCÐCØ+Ð,CÔDÔJÈ1ÔMÐQRÒRÐRå ðDØ4LÈ[Ô4YÔ4_Ð`aÔ4bðDð Dð G_ð  `wô  Gxô  G~ð  @ô  GAðDð Dð Dñô ð ð
 #ð 	-åœÐ!9Ð:JÔ!KÈAÈAÈAÈqÈDÔ!QÐUVÒ!VÑWÔWÐXYÔZˆIˆIà"œ¨Ô+Ð,ˆIàˆ
Ø#%Ð Øˆ	Ý'¨	Ñ2Ô2ð #	%ñ #	%‰NˆE�7Ø'¨	°'Ð(9Ô:ˆOà-¨a¨a¨aÐ1rµ2´6½"¼(À?ÕVYÐZ^ÔZhÑViÔViÒCiÑ:jÔ:jÐklÔ:mÑ3nÔ3nÐqrÑ3rÐ1rÐ.rÔsˆOØ0°Ô=¸eÔDˆIØ$<Ð=TÔ$UÐV[Ô$\Ð!ð 'ð Ø+CÐD^Ô+_Ð`eÔ+fÐ(Ø7OØ:ô8àô8Ð4ð &Ð&^­¬­r¬xÐ8PÐTUÒ8UÑ/VÔ/VÐWXÔ/YÑ(ZÔ(ZÐ]^Ñ(^Ð&^Ô_�	Ø(=ØX•b”f�RœXÐ&JÈaÒ&OÑPÔPÐQRÔSÑTÔTÐWXÑXÐXô)Ð%õ ' Ñ7Ô7ˆOÝ  Ñ+Ô+ˆIÝ$,Ð-BÑ$CÔ$CÐ!à!%×!GÒ!GØ&Ø.Ø#œ}Ø!%¤°Ñ!2°aÑ 7ð	 "Hñ "ô "Ðð +Ô6°qÔ9Ô?ð (ð (�Ø�
”
˜i¨œlÑ*�
”
Ø�”˜I aœLÑ(�”Ø×!Ò! $Ñ'Ô'Ð'Ð'à$×+Ò+Ð,>Ñ?Ô?Ð?Ø˜ 1™Ñ$ˆI‰Iàð 	iÝ ¨:ÐNfÐ!gÐ!gÑhÔhÐhå˜g zÐ2Ñ3Ô3Ð3r   )r,   r-   r.   r/   r0   r1   )re   )r   r-   rz   r   )rš   )NN)FNNNNNT)T)&Ú__name__Ú
__module__Ú__qualname__Ú__doc__Úmodel_input_namesr¶   Úvocab_files_namesrJ   ÚpropertyrR   rV   r`   rÝ   rd   rh   rr   rÛ   ry   r€   Úfloatrq   r~   rA   Útuplerº   r%   r¤   r	   r   rÐ   rÓ   rì   r   r
   râ   r   rø   Ú__classcell__)rN   s   @r   r)   r)   =   s  ø€ € € € € ðð ð2 %Ð&6Ð7ÐØ)Ðð
 ØØØØØð 
ð  
ð  
ð  
ð  
ð  
ðD ð!ð !ñ „Xð!ð?ð ?ð ?ð#¨Sð #°Tð #ð #ð #ð #ð$Nð NÀcð Nð Nð Nð Nð *à”
ð*ð ð*ð ð	*ð
 ð*ð *ð *ð *ð`  !ØØ!ð ð  à”
ð ð ”*ð ð ð	 ð
 ð ð ð ð  ð  ð  ðH:ð :°2´:ð :È%ð :ÐbgÐjnÑbnð :ð :ð :ð :ðxð  2¤:ð ¸¼ð ÐQTð ð ð ð ð@!ð !¨cð !ÀCÈ$ÁJð !ÐZ_Ð`cÔZdð !ð !ð !ð !ð4 :>Ø!%ð	E4ð E4àŒz˜D Ô!1Ô2Ñ2ðE4ð 0°$Ñ6ðE4ð ˜$‘Jð	E4ð 
ðE4ð E4ð E4ð E4ðT :>Ø!%ð	%Eð %EàŒz˜D Ô!1Ô2Ñ2ð%Eð 0°$Ñ6ð%Eð ˜$‘Jð	%Eð 
ð%Eð %Eð %Eð %EðT 16Ø6:Ø!%Ø)-Ø-1Ø26Øðuð uàŒz˜D Ô!1Ô2Ñ2°T¸$¸{Ô?OÔ:PÔ5QÑQðuð ˜‘˜oÑ-ðuð ˜3‘JÐ!3Ñ3ð	uð
 ˜$‘Jðuð   $™Jðuð  $ d™{ðuð ˜jÑ(¨4Ñ/ðuð ðuð 
ðuð uð uð uðv !ð	w4ð w4ð #/ðw4ð ð	w4ð w4ð w4ð w4ð w4ð w4ð w4ð w4r   r)   )rü   rD   r±   Únumpyrr   Úfeature_extraction_utilsr   Útokenization_pythonr   r   r   r   r	   Úutilsr
   r   r   r   r   Úutils.import_utilsr   r%   Ú
get_loggerrù   r´   r¶   r   r$   r)   Ú__all__r7   r   r   ú<module>r
     s^  ðð (Ð 'à €€€Ø 	€	€	€	à Ð Ð Ð à 4Ð 4Ð 4Ð 4Ð 4Ð 4Ø vÐ vÐ vÐ vÐ vÐ vÐ vÐ vÐ vÐ vÐ vÐ vÐ vÐ vØ _Ð _Ð _Ð _Ð _Ð _Ð _Ð _Ð _Ð _Ð _Ð _Ð _Ð _Ø *Ð *Ð *Ð *Ð *Ð *ð ÐÑÔð ØÐÐÐà	ˆÔ	˜HÑ	%Ô	%€ð ˆ\ðÐ ð
ð ð ðð ð ð 
€Ð+Ð,Ñ,Ô,ðI
4ð I
4ð I
4ð I
4ð I
4Ð,ñ I
4ô I
4ñ -Ô,ðI
4ðX  Ð
 €€€r   