o
    ØRWjØ‰  ã                   @  s”  U d dl mZ d dlZd dlZd dlmZmZmZmZm	Z	 d dl
mZ ddlmZ G dd„ de	ƒZed	d
d�Zd`dd„Zdadd„Zddgfdbdd„Zdcdddd„Zddgfdbdd„Zd`d d!„Zd`d"d#„Zd`d$d%„Zd`d&d'„Zd`d(d)„Zd`d*d+„Zded/d0„Zddgfdfd3d4„Zd`d5d6„Zdgd8d9„Zdhd=d>„Z did@dA„Z!djdCdD„Z"dkdFdG„Z#dldIdJ„Z$dmdLdM„Z%dndNdO„Z&dodpdSdT„Z'dUZ(dVe)dW< dqdYdZ„Z*drd^d_„Z+dS )sé    )ÚannotationsN)ÚAnyÚTypeVarÚCallableÚOptionalÚ
NamedTuple)Ú	TypeAliasé   )Úpandasc                   @  s^   e Zd ZU ded< dZded< dZded< dZded< dZded	< dZded
< dZ	ded< dS )ÚRemediationÚstrÚnameNzOptional[str]Úimmediate_msgÚnecessary_msgzOptional[Callable[[Any], Any]]Únecessary_fnÚoptional_msgÚoptional_fnÚ	error_msg)
Ú__name__Ú
__module__Ú__qualname__Ú__annotations__r   r   r   r   r   r   © r   r   úe/home/esfera/Documents/content_generation/venv/lib/python3.10/site-packages/openai/lib/_validators.pyr      s   
 r   ÚOptionalDataFrameTzOptional[pd.DataFrame])ÚboundÚdfúpd.DataFrameÚreturnc                 C  s8   d}t | ƒ|kr
dnd}dt | ƒ› d|› �}td|d�S )z’
    This validator will only print out the number of examples and recommend to the user to increase the number of examples if less than 100.
    éd   Ú z§. In general, we recommend having at least a few hundred examples. We've found that performance tends to linearly increase for every doubling of the number of examplesz
- Your file contains z prompt-completion pairsÚnum_examples©r   r   )Úlenr   )r   ÚMIN_EXAMPLESÚoptional_suggestionr   r   r   r   Únum_examples_validator   s   ÿýr&   Únecessary_columnr   c                   s„   ddd„‰ d}d}d}d}ˆ| j vr9ˆd	d
„ | j D ƒv r3d‡ ‡fdd„}|}dˆ› d�}dˆ› d�}ndˆ› d�}td||||d�S )z[
    This validator will ensure that the necessary column is present in the dataframe.
    r   r   Úcolumnr   r   c                   s2   ‡ fdd„| j D ƒ}| j|d ˆ  ¡ idd� | S )Nc                   s    g | ]}t |ƒ ¡ ˆ kr|‘qS r   ©r   Úlower©Ú.0Úc©r(   r   r   Ú
<listcomp>-   s     zInecessary_column_validator.<locals>.lower_case_column.<locals>.<listcomp>r   T)ÚcolumnsÚinplace)r0   Úrenamer*   )r   r(   Úcolsr   r.   r   Úlower_case_column,   s   z5necessary_column_validator.<locals>.lower_case_columnNc                 S  s   g | ]}t |ƒ ¡ ‘qS r   r)   r+   r   r   r   r/   7   ó    z.necessary_column_validator.<locals>.<listcomp>c                   ó
   ˆ | ˆƒS ©Nr   )r   ©r4   r'   r   r   Úlower_case_column_creator9   ó   
z=necessary_column_validator.<locals>.lower_case_column_creatorz
- The `z ` column/key should be lowercasezLower case column name to `ú`z^` column/key is missing. Please make sure you name your columns/keys appropriately, then retryr'   )r   r   r   r   r   )r   r   r(   r   r   r   )r   r   r   r   )r0   r   )r   r'   r   r   r   r   r9   r   r8   r   Únecessary_column_validator'   s&   

ûr<   ÚpromptÚ
completionÚfieldsú	list[str]c                   sª   g }d}d}d}t | jƒdkrM‡fdd„| jD ƒ}d}|D ]‰ ‡ fdd„|D ƒ}t |ƒdkr9|dˆ › d	ˆ › d
�7 }qd|› |› �}d|› �}d‡fdd„}td|||d�S )zK
    This validator will remove additional columns from the dataframe.
    Nr	   c                   s   g | ]}|ˆ vr|‘qS r   r   r+   ©r?   r   r   r/   U   r5   z/additional_column_validator.<locals>.<listcomp>r    c                   s   g | ]}ˆ |v r|‘qS r   r   r+   )Úacr   r   r/   X   r5   r   z9
  WARNING: Some of the additional columns/keys contain `z<` in their name. These will be ignored, and the column/key `z`` will be used instead. This could also result from a duplicate column/key in the provided file.zh
- The input file should contain exactly two columns/keys per row. Additional columns/keys present are: z Remove additional columns/keys: Úxr   r   c                   s   | ˆ  S r7   r   ©rC   rA   r   r   r   ^   s   z1additional_column_validator.<locals>.necessary_fnÚadditional_column©r   r   r   r   ©rC   r   r   r   )r#   r0   r   )r   r?   Úadditional_columnsr   r   r   Úwarn_messageÚdupsr   )rB   r?   r   Úadditional_column_validatorK   s*   €
ürK   Úfieldc                   s¦   d}d}d}| ˆ    dd„ ¡ ¡ s| ˆ   ¡  ¡ rH| ˆ  dk| ˆ   ¡ B }|  ¡ j|  ¡ }dˆ › d|› �}d‡ fd
d„}dt|ƒ› dˆ › d�}tdˆ › �|||d�S )zA
    This validator will ensure that no completion is empty.
    Nc                 S  s   | dkS )Nr    r   rD   r   r   r   Ú<lambda>q   s    z+non_empty_field_validator.<locals>.<lambda>r    z
- `z?` column/key should not contain empty strings. These are rows: rC   r   r   c                   s   | | ˆ  dk j ˆ gd�S )Nr    ©Úsubset)ÚdropnarD   ©rL   r   r   r   v   s   z/non_empty_field_validator.<locals>.necessary_fnúRemove z rows with empty ÚsÚempty_rF   rG   )ÚapplyÚanyÚisnullÚreset_indexÚindexÚtolistr#   r   )r   rL   r   r   r   Ú
empty_rowsÚempty_indexesr   rQ   r   Únon_empty_field_validatori   s   &ür]   c                   s„   | j ˆ d�}|  ¡ j|  ¡ }d}d}d}t|ƒdkr:dt|ƒ› dd ˆ ¡› d|› �}dt|ƒ› d	�}d‡ fdd„}td|||d�S )zY
    This validator will suggest to the user to remove duplicate rows if they exist.
    rN   Nr   ú
- There are z duplicated ú-z sets. These are rows: rR   z duplicate rowsrC   r   r   c                   s   | j ˆ d�S )NrN   )Údrop_duplicatesrD   rA   r   r   r   ‘   ó   z.duplicated_rows_validator.<locals>.optional_fnÚduplicated_rows©r   r   r   r   rG   )Ú
duplicatedrX   rY   rZ   r#   Újoinr   )r   r?   rb   Úduplicated_indexesr   r   r   r   rA   r   Úduplicated_rows_validatorƒ   s    ürg   c                   s€   d}d}d}t | ƒ}|dkr8ddd„‰ ˆ | ƒ‰tˆƒd	kr8d
tˆƒ› dˆ› d�}dtˆƒ› d�}d‡ ‡fdd„}td|||d�S )zW
    This validator will suggest to the user to remove examples that are too long.
    Núopen-ended generationÚdr   r   r   c                 S  s$   | j dd„ dd�}|  ¡ j|  ¡ S )Nc                 S  s   t | jƒt | jƒ dkS )Ni'  )r#   r=   r>   rD   r   r   r   rM   ¨   r5   zClong_examples_validator.<locals>.get_long_indexes.<locals>.<lambda>é   )Úaxis)rU   rX   rY   rZ   )ri   Úlong_examplesr   r   r   Úget_long_indexes§   s   z1long_examples_validator.<locals>.get_long_indexesr   r^   z. examples that are very long. These are rows: zf
For conditional generation, and for classification the examples shouldn't be longer than 2048 tokens.rR   z long examplesrC   c                   s8   ˆ | ƒ}ˆ|krt j dt|ƒ› d|› d�¡ |  |¡S )NzeThe indices of the long examples has changed as a result of a previously applied recommendation.
The z? long examples to be dropped are now at the following indices: Ú
)ÚsysÚstdoutÚwriter#   Údrop)rC   Úlong_indexes_to_drop©rm   Úlong_indexesr   r   r   ±   s   ÿ
z,long_examples_validator.<locals>.optional_fnrl   rc   )ri   r   r   r   rG   )Úinfer_task_typer#   r   )r   r   r   r   Úft_typer   rt   r   Úlong_examples_validatorœ   s"   
ürx   c                   sp  d}d}d}d}d‰g d¢}|D ]}|dkr | j j d¡ ¡ r q| j jj|dd� ¡ r,q|‰ ˆ dd¡}t| ƒ}|d	krBtd
d�S d$dd„‰ t| j dd�}	| j |	k ¡ rad|	› d�}td
|d�S |	dkr›|	 dd¡}
d|
› d�}t	|	ƒdkr|d|› d�7 }| j jdt	|	ƒ … jj|	dd� ¡ rš|d|	› d�7 }nd}|	dkr¯d|› d�}d%‡ ‡fd d!„}td"||||d#�S )&zœ
    This validator will suggest to add a common suffix to the prompt if one doesn't already exist in case of classification or conditional generation.
    Nz


### =>

)ú ->z

###

z

===

z

---

z

===>

z

--->

ry   rn   F©Úregexú\nrh   Úcommon_suffix©r   rC   r   Úsuffixr   c                 S  ó   | d  |7  < | S ©Nr=   r   ©rC   r   r   r   r   Ú
add_suffixâ   ó   z2common_prompt_suffix_validator.<locals>.add_suffix©ÚxfixzAll prompts are identical: `zt`
Consider leaving the prompts blank if you want to do open-ended generation, otherwise ensure prompts are different©r   r   r    z 
- All prompts end with suffix `r;   é
   úR. This suffix seems very long. Consider replacing with a shorter suffix, such as `z5
  WARNING: Some of your prompts contain the suffix `zZ` more than once. We strongly suggest that you review your prompts and add a unique suffixa”  
- Your data does not contain a common separator at the end of your prompts. Having a separator string appended to the end of the prompt makes it clearer to the fine-tuned model where the completion should begin. See https://platform.openai.com/docs/guides/fine-tuning/preparing-your-dataset for more detail and examples. If you intend to do open-ended generation, then you should leave the prompts emptyzAdd a suffix separator `z` to all promptsc                   r6   r7   r   rD   ©rƒ   Úsuggested_suffixr   r   r   ù   r:   z3common_prompt_suffix_validator.<locals>.optional_fnÚcommon_completion_suffix©r   r   r   r   r   ©rC   r   r   r   r   r   rG   )
r=   r   ÚcontainsrV   Úreplacerv   r   Úget_common_xfixÚallr#   )r   r   r   r   r   Úsuffix_optionsÚsuffix_optionÚdisplay_suggested_suffixrw   r}   Úcommon_suffix_new_line_handledr   rŠ   r   Úcommon_prompt_suffix_validatorÁ   sT   

&€ûr—   c                   s¦   d}d}d}d}t | jdd�‰ ˆ dkrtdd�S ddd„‰| jˆ k ¡ r)tdd�S ˆ dkrKdˆ › d�}|tˆ ƒk rK|d7 }dˆ › d�}d‡ ‡fdd„}td|||d�S )zd
    This validator will suggest to remove a common prefix from the prompt if a long one exist.
    é   NÚprefixr…   r    Úcommon_prefixr~   rC   r   r   c                 S  s   | d j t|ƒd … | d< | S r�   ©r   r#   )rC   r™   r   r   r   Úremove_common_prefix  s   z<common_prompt_prefix_validator.<locals>.remove_common_prefixz"
- All prompts start with prefix `r;   zÒ. Fine-tuning doesn't require the instruction specifying the task, or a few-shot example scenario. Most of the time you should only add the input data into the prompt, and the desired output into the completionúRemove prefix `z` from all promptsc                   s
   ˆ| ˆ ƒS r7   r   rD   ©rš   rœ   r   r   r   !  r:   z3common_prompt_prefix_validator.<locals>.optional_fnÚcommon_prompt_prefixrc   )rC   r   r™   r   r   r   rG   )r‘   r=   r   r’   r#   ©r   ÚMAX_PREFIX_LENr   r   r   r   rž   r   Úcommon_prompt_prefix_validator  s,   


ür¢   c                   sœ   d}t | jdd�‰ tˆ ƒdkoˆ d dk‰tˆ ƒ|k r tdd�S ddd„‰| jˆ k ¡ r1tdd�S dˆ › d�}dˆ › d�}d‡ ‡‡fdd„}td|||d�S )zh
    This validator will suggest to remove a common prefix from the completion if a long one exist.
    é   r™   r…   r   ú rš   r~   rC   r   Ú	ws_prefixr   c                 S  s4   | d j t|ƒd … | d< |rd| d › �| d< | S )Nr>   r¤   r›   )rC   r™   r¥   r   r   r   rœ   7  s   z@common_completion_prefix_validator.<locals>.remove_common_prefixz&
- All completions start with prefix `z_`. Most of the time you should only add the output data into the completion, without any prefixr�   z` from all completionsc                   s   ˆ| ˆ ˆƒS r7   r   rD   ©rš   rœ   r¥   r   r   r   E  ra   z7common_completion_prefix_validator.<locals>.optional_fnÚcommon_completion_prefixrc   N)rC   r   r™   r   r¥   r   r   r   rG   )r‘   r>   r#   r   r’   r    r   r¦   r   Ú"common_completion_prefix_validator,  s"   


ür¨   c                   sb  d}d}d}d}t | ƒ}|dks|dkrtdd�S t| jdd�}| j|k ¡ r6d|› d	|› d
�}td|d�S d‰g d¢}|D ]}| jjj|dd� ¡ rLq>|‰ ˆ dd¡}	d$dd„‰ |dkr”| dd¡}
d|
› d
�}t	|ƒdkrx|d|	› d
�7 }| jjdt	|ƒ … jj|dd� ¡ r“|d|› d�7 }nd}|dkr¨d|	› d�}d%‡ ‡fd d!„}td"||||d#�S )&z 
    This validator will suggest to add a common suffix to the completion if one doesn't already exist in case of classification or conditional generation.
    Nrh   Úclassificationr}   r~   r   r…   z All completions are identical: `zJ`
Ensure completions are different, otherwise the model will just repeat `r;   r‡   z [END])	rn   Ú.z ENDz***z+++z&&&z$$$z@@@z%%%Frz   rn   r|   rC   r   r   c                 S  r€   ©Nr>   r   r‚   r   r   r   rƒ   v  r„   z6common_completion_suffix_validator.<locals>.add_suffixr    z$
- All completions end with suffix `rˆ   r‰   z9
  WARNING: Some of your completions contain the suffix `zU` more than once. We suggest that you review your completions and add a unique endingaH  
- Your data does not contain a common ending at the end of your completions. Having a common ending string appended to the end of the completion makes it clearer to the fine-tuned model where the completion should end. See https://platform.openai.com/docs/guides/fine-tuning/preparing-your-dataset for more detail and examples.zAdd a suffix ending `z` to all completionsc                   r6   r7   r   rD   rŠ   r   r   r   ˆ  r:   z7common_completion_suffix_validator.<locals>.optional_fnrŒ   r�   rŽ   rG   )
rv   r   r‘   r>   r’   r   r�   rV   r�   r#   )r   r   r   r   r   rw   r}   r“   r”   r•   r–   r   rŠ   r   Ú"common_completion_suffix_validatorP  sN   

&€ûr¬   c                 C  s^   ddd„}d}d}d}| j jdd…  ¡ dks!| j jd d d	kr'd
}d}|}td|||d�S )zŽ
    This validator will suggest to add a space at the start of the completion if it doesn't already exist. This helps with tokenization.
    rC   r   r   c                 S  s   | d   dd„ ¡| d< | S )Nr>   c                 S  s   |   d¡r	d|  S d|  S )Nr¤   r    )Ú
startswith)rS   r   r   r   rM   š  ó    zLcompletions_space_start_validator.<locals>.add_space_start.<locals>.<lambda>)rU   rD   r   r   r   Úadd_space_start™  s   z:completions_space_start_validator.<locals>.add_space_startNrj   r   r¤   zæ
- The completion should start with a whitespace character (` `). This tends to produce better results due to the tokenization we use. See https://platform.openai.com/docs/guides/fine-tuning/preparing-your-dataset for more detailsz=Add a whitespace character to the beginning of the completionÚcompletion_space_startrc   rG   )r>   r   ÚnuniqueÚvaluesr   )r   r¯   r   r   r   r   r   r   Ú!completions_space_start_validator”  s   
,ür³   r(   r   úRemediation | Nonec                   sp   d‡ fdd„}| ˆ    dd„ ¡ ¡ }| ˆ    dd„ ¡ ¡ }|d	 |kr6td
dˆ › dˆ › d�dˆ › d�|d�S dS )zt
    This validator will suggest to lowercase the column values, if more than a third of letters are uppercase.
    rC   r   r   c                   s   | ˆ  j  ¡ | ˆ < | S r7   r)   rD   r.   r   r   Ú
lower_case²  s   z(lower_case_validator.<locals>.lower_casec                 S  ó   t dd„ | D ƒƒS )Nc                 s  ó$   � | ]}|  ¡ r| ¡ rd V  qdS ©rj   N)ÚisalphaÚisupperr+   r   r   r   Ú	<genexpr>¶  ó   €" ú9lower_case_validator.<locals>.<lambda>.<locals>.<genexpr>©ÚsumrD   r   r   r   rM   ¶  ó    z&lower_case_validator.<locals>.<lambda>c                 S  r¶   )Nc                 s  r·   r¸   )r¹   Úislowerr+   r   r   r   r»   ·  r¼   r½   r¾   rD   r   r   r   rM   ·  rÀ   r	   rµ   z
- More than a third of your `z%` column/key is uppercase. Uppercase z÷s tends to perform worse than a mixture of case encountered in normal language. We recommend to lower case the data if that makes sense in your domain. See https://platform.openai.com/docs/guides/fine-tuning/preparing-your-dataset for more detailsz'Lowercase all your data in column/key `r;   rc   NrG   )rU   r¿   r   )r   r(   rµ   Úcount_upperÚcount_lowerr   r.   r   Úlower_case_validator­  s   
ürÄ   Úfnameú'tuple[pd.DataFrame | None, Remediation]c              
   C  sÆ  d}d}d}d}d}t j | ¡�rQ�z|  ¡  d¡s!|  ¡  d¡rF|  ¡  d¡r*dnd\}}d|› d�}d|› d	�}tj| |td
� d¡}nè|  ¡  d¡rnd}d}t 	| ¡}	|	j
}
t|
ƒdkrc|d7 }tj| td� d¡}nÀ|  ¡  d¡r¦d}d}t| dƒ�}| ¡ }tjdd„ | d¡D ƒ|td� d¡}W d  ƒ n1 s w   Y  nˆ|  ¡  d¡rÏtj| dtd� d¡}t|ƒdkrÍd}d}tj| td� d¡}na	 n_|  ¡  d¡�rz"tj| dtd� d¡}t|ƒdkrôtj| td� d¡}nd }d}W n4 t�y   tj| td� d¡}Y n!w d!}d"| v �r&|d#| › d$|  d"¡d% › d&�7 }n|d#| › d'�7 }W n' ttf�yP   |  d"¡d%  ¡ }d(| › d)|› d*|› d+�}Y nw d,| › d-�}td.|||d/�}||fS )0zÕ
    This function will read a file saved in .csv, .json, .txt, .xlsx or .tsv format using pandas.
     - for .xlsx it will read the first sheet
     - for .txt it will assume completions and split on newline
    Nz.csvz.tsv)ÚCSVú,)ÚTSVú	z=
- Based on your file extension, your file is formatted as a z filezYour format `z` will be converted to `JSONL`)ÚsepÚdtyper    z.xlsxzH
- Based on your file extension, your file is formatted as an Excel filez/Your format `XLSX` will be converted to `JSONL`rj   z¥
- Your Excel file contains more than one sheet. Please either save as csv or ensure all data is present in the first sheet. WARNING: Reading only the first sheet...)rÌ   z.txtz9
- Based on your file extension, you provided a text filez.Your format `TXT` will be converted to `JSONL`Úrc                 S  s   g | ]}d |g‘qS )r    r   )r,   Úliner   r   r   r/   è  s    z#read_any_format.<locals>.<listcomp>rn   )r0   rÌ   ú.jsonlT)ÚlinesrÌ   z^
- Your JSONL file appears to be in a JSON format. Your file will be converted to JSONL formatz/Your format `JSON` will be converted to `JSONL`z.jsonz^
- Your JSON file appears to be in a JSONL format. Your file will be converted to JSONL formatz]Your file must have one of the following extensions: .CSV, .TSV, .XLSX, .TXT, .JSON or .JSONLrª   z Your file `z` ends with the extension `.éÿÿÿÿz` which is not supported.z` is missing a file extension.zYour file `z!` does not appear to be in valid z9 format. Please ensure your file is formatted as a valid z file.zFile z does not exist.Úread_any_format)r   r   r   r   )ÚosÚpathÚisfiler*   ÚendswithÚpdÚread_csvr   ÚfillnaÚ	ExcelFileÚsheet_namesr#   Ú
read_excelÚopenÚreadÚ	DataFrameÚsplitÚ	read_jsonÚ
ValueErrorÚ	TypeErrorÚupperr   )rÅ   r?   Úremediationr   r   r   r   Úfile_extension_strÚ	separatorÚxlsÚsheetsÚfÚcontentr   r   r   rÒ   Ã  sŽ   
ÿ
ýüþ€€þÿ
"€þürÒ   c                 C  s,   t | ƒ}d}|dkrd|› d�}td|d�S )zÓ
    This validator will infer the likely fine-tuning format of the data, and display it to the user if it is classification.
    It will also suggest to use ada and explain train/validation split benefits.
    Nr©   zK
- Based on your data it seems like you're trying to fine-tune a model for zã
- For classification, we recommend you try one of the faster and cheaper models, such as `ada`
- For classification, you can estimate the expected model performance by keeping a held out dataset, which is not used for trainingr!   r"   )rv   r   )r   rw   r   r   r   r   Úformat_inferrer_validator  s
   rì   rå   c                 C  sb   |j durtj d|j› d|j › d�¡ t d¡ |jdur%tj |j¡ |jdur/| | ¡} | S )zs
    This function will apply a necessary remediation to a dataframe, or print an error message if one exists.
    Nz

ERROR in z validator: z

Aborting...rj   )	r   ro   Ústderrrq   r   Úexitr   rp   r   )r   rå   r   r   r   Úapply_necessary_remediation(  s   




rï   Ú
input_textÚauto_acceptÚboolc                 C  s.   t j | ¡ |rt j d¡ dS tƒ  ¡ dkS )NzY
TÚn)ro   rp   rq   Úinputr*   )rð   rñ   r   r   r   Úaccept_suggestion6  s
   rõ   útuple[pd.DataFrame, bool]c                 C  sj   d}d|j › d�}|j dur!t||ƒr!|jdusJ ‚| | ¡} d}|jdur1tj d|j› d�¡ | |fS )zc
    This function will apply an optional remediation to a dataframe, based on the user input.
    Fz- [Recommended] z [Y/n]: NTz- [Necessary] rn   )r   rõ   r   r   ro   rp   rq   )r   rå   rñ   Úoptional_appliedrð   r   r   r   Úapply_optional_remediation>  s   



rø   ÚNonec                 C  sl   t | ƒ}d}|dkrt| ƒ}|d }n| jdd� ¡ }|d }ddd„}||d ƒ}tj d|› d�¡ dS )z?
    Estimate the time it'll take to fine-tune the dataset
    g      ð?r©   g
×£p=
÷?T)rY   g‘í|?5^ª?ÚtimeÚfloatr   r   c                 S  sd   | dk rt | dƒ› d�S | dk rt | d dƒ› d�S | dk r(t | d dƒ› d�S t | d dƒ› d�S )	Né<   r	   z secondsi  z minutesi€Q z hoursz days)Úround)rú   r   r   r   Úformat_time]  s   z.estimate_fine_tuning_time.<locals>.format_timeéŒ   z:Once your model starts training, it'll approximately take z~ to train a `curie` model, and less for `ada` and `babbage`. Queue will approximately take half an hour per job ahead of you.
N)rú   rû   r   r   )rv   r#   Úmemory_usager¿   ro   rp   rq   )r   Ú	ft_formatÚexpected_timer!   Úsizerþ   Útime_stringr   r   r   Úestimate_fine_tuning_timeP  s   



ÿr  rà   c                   sd   |rddgndg}d}	 |dkrd|› d�nd‰‡ ‡fdd	„|D ƒ}t d
d„ |D ƒƒs-|S |d7 }q)NÚ_trainÚ_validr    r   Tz (ú)c                   s,   g | ]}t j ˆ ¡d  › d|› ˆ› d�‘qS )r   Ú	_preparedrÏ   )rÓ   rÔ   Úsplitext)r,   r   ©rÅ   Úindex_suffixr   r   r/   r  s   , z!get_outfnames.<locals>.<listcomp>c                 s  s   � | ]	}t j |¡V  qd S r7   )rÓ   rÔ   rÕ   )r,   rê   r   r   r   r»   s  s   € z get_outfnames.<locals>.<genexpr>rj   )rV   )rÅ   rà   ÚsuffixesÚiÚcandidate_fnamesr   r  r   Úget_outfnamesm  s   ûr  útuple[int, object]c                 C  s.   | j  ¡ }d }|dkr| j  ¡ jd }||fS )Nr	   r   )r>   r±   Úvalue_countsrY   )r   Ú	n_classesÚ	pos_classr   r   r   Úget_classification_hyperparamsx  s
   
r  Úany_remediationsc                 C  s„  t | ƒ}t| jdd�}t| jdd�}d}d}|dkr!t||ƒr!d}d}	| dd	¡}
| dd	¡}t|ƒd
kr;d|› d�nd}d}|s\|s\tj 	d|› d|	› d|
› d|› d�	¡ t
| ƒ dS t||ƒ�r:t||ƒ}|rÚt|ƒdkr{d|d
 v r{d|d v s}J ‚d}tt| ƒ| tt| ƒd ƒƒ}| j|dd�}|  |j¡}|ddg j|d
 ddddd� |ddg j|d ddddd� t| ƒ\}}|	d7 }	|dkrÒ|	d |› d�7 }	n |	d!|› �7 }	nt|ƒdksâJ ‚| ddg j|d
 ddddd� |röd"ndd# d$ |¡ }|�r
d%|d › d�nd}t|
ƒd
k�rdnd&|
› d�}tj 	d'|› d(|d
 › d|› |	› d)|› |› d�¡ t
| ƒ dS tj 	d*¡ dS )+aQ  
    This function will write out a dataframe to a file, if the user would like to proceed, and also offer a fine-tuning command with the newly created file.
    For classification it will optionally ask the user if they would like to split the data into train/valid files, and modify the suggested command to include the valid set.
    r   r…   FzQ- [Recommended] Would you like to split into training and validation set? [Y/n]: r©   Tr    rn   r|   r   z Make sure to include `stop=["z;"]` so that the generated texts ends at the expected place.z@

Your data will be written to a new JSONL file. Proceed [Y/n]: zK
You can use your file for fine-tuning:
> openai api fine_tunes.create -t "ú"ue   

After youâ€™ve fine-tuned a model, remember that your prompt has to end with the indicator string `zX` for the model to start generating completions, rather than continuing with the prompt.r	   ÚtrainÚvalidrj   iè  gš™™™™™é?é*   )ró   Úrandom_stater=   r>   ÚrecordsN)rÐ   ÚorientÚforce_asciiÚindentz! --compute_classification_metricsz" --classification_positive_class "z --classification_n_classes rS   z to `z` and `z -v "uc   After youâ€™ve fine-tuned a model, remember that your prompt has to end with the indicator string `z
Wrote modified filezd`
Feel free to take a look!

Now use that file when fine-tuning:
> openai api fine_tunes.create -t "z

z#Aborting... did not write the file
)rv   r‘   r=   r>   rõ   r�   r#   ro   rp   rq   r  r  ÚmaxÚintÚsamplerr   rY   Úto_jsonr  re   )r   rÅ   r  rñ   r  Úcommon_prompt_suffixrŒ   rà   rð   Úadditional_paramsÚ%common_prompt_suffix_new_line_handledÚ)common_completion_suffix_new_line_handledÚoptional_ending_stringÚfnamesÚMAX_VALID_EXAMPLESÚn_trainÚdf_trainÚdf_validr  r  Úfiles_stringÚvalid_stringÚseparator_reminderr   r   r   Úwrite_out_file€  sn   
ÿýÿ
(ÿÿÿÿ
ý(ÿr1  c                 C  s>   d}t | jj ¡ ƒdkrdS t| j ¡ ƒt| ƒ| k rdS dS )z>
    Infer the likely fine-tuning task type from the data
    é   r   rh   r©   zconditional generation)r¿   r=   r   r#   r>   Úunique)r   ÚCLASSIFICATION_THRESHOLDr   r   r   rv   Ë  s   rv   r   Úseriesr†   c                 C  sn   d}	 |dkr| j t|ƒd  d… n
| j dt|ƒd … }| ¡ dkr'	 |S ||jd kr1	 |S |jd }q)zQ
    Finds the longest common suffix or prefix of all the values in a series
    r    Tr   rj   Nr   )r   r#   r±   r²   )r5  r†   Úcommon_xfixÚcommon_xfixesr   r   r   r‘   Ù  s   4ÿü
ÿ÷r‘   z,Callable[[pd.DataFrame], Remediation | None]r   Ú	Validatorúlist[Validator]c                   C  s2   t dd„ dd„ tttttdd„ dd„ tttt	t
gS )Nc                 S  ó
   t | dƒS r�   ©r<   rD   r   r   r   rM   ñ  ó   
 z get_validators.<locals>.<lambda>c                 S  r:  r«   r;  rD   r   r   r   rM   ò  r<  c                 S  r:  r�   ©rÄ   rD   r   r   r   rM   ø  r<  c                 S  r:  r«   r=  rD   r   r   r   rM   ù  r<  )r&   rK   r]   rì   rg   rx   r—   r¢   r¨   r¬   r³   r   r   r   r   Úget_validatorsî  s    ñr>  Ú
validatorsÚwrite_out_file_funcúCallable[..., Any]c                 C  sÆ   g }|d ur|  |¡ |D ]}|| ƒ}|d ur!|  |¡ t| |ƒ} qtdd„ |D ƒƒ}tdd„ |D ƒƒ}	d}
|rPtj d¡ |D ]}t| ||ƒ\} }|
pM|}
q@ntj d¡ |
pY|	}|| |||ƒ d S )Nc                 S  s$   g | ]}|j d us|jd ur|‘qS r7   )r   r   ©r,   rå   r   r   r   r/     s
    þz$apply_validators.<locals>.<listcomp>c                 S  s   g | ]	}|j d ur|‘qS r7   )r   rB  r   r   r   r/     r®   Fz?

Based on the analysis we will perform the following actions:
z

No remediations found.
)Úappendrï   rV   ro   rp   rq   rø   )r   rÅ   rå   r?  rñ   r@  Úoptional_remediationsÚ	validatorÚ&any_optional_or_necessary_remediationsÚany_necessary_appliedÚany_optional_appliedr÷   Ú!any_optional_or_necessary_appliedr   r   r   Úapply_validators  s6   


€þÿÿ
þrJ  )r   r   r   r   )r   r   r'   r   r   r   )r   r   r?   r@   r   r   )r>   )r   r   rL   r   r   r   )r   r   r(   r   r   r´   )rÅ   r   r?   r@   r   rÆ   )r   r   rå   r   r   r   )rð   r   rñ   rò   r   rò   )r   r   rå   r   rñ   rò   r   rö   )r   r   r   rù   )rÅ   r   rà   rò   r   r@   )r   r   r   r  )
r   r   rÅ   r   r  rò   rñ   rò   r   rù   )r   r   r   r   )r   )r5  r   r†   r   r   r   )r   r9  )r   r   rÅ   r   rå   r´   r?  r9  rñ   rò   r@  rA  r   rù   ),Ú
__future__r   rÓ   ro   Útypingr   r   r   r   r   Útyping_extensionsr   Ú_extrasr
   r×   r   r   r&   r<   rK   r]   rg   rx   r—   r¢   r¨   r¬   r³   rÄ   rÒ   rì   rï   rõ   rø   r  r  r  r1  rv   r‘   r8  r   r>  rJ  r   r   r   r   Ú<module>   sF   


$

%
D
'
$
D
ÿ
Y







K
