Ë
    çÍ:j
  ã                   ót   — d Z ddlmZ ddlmZ ddlmZ d„ Z G d„ de«      Z G d„ d	e«      Z	 G d
„ de«      Z
y)z…Smoothing algorithms for language modeling.

According to Chen & Goodman 1995 these should work with both Backoff and
Interpolation.
é    )Úmethodcaller)Ú	Smoothing)ÚConditionalFreqDistc                 ó„   ‡— t        | t        «      rt        d«      nd„ Št        ˆfd„| j	                  «       D «       «      S )zµCount values that are greater than zero in a distribution.

    Assumes distribution is either a mapping with counts as values or
    an instance of `nltk.ConditionalFreqDist`.
    ÚNc                 ó   — | S ©N© )Úcounts    úf/home/mcse/projects/srt_converter/srt-converter-venv/lib/python3.12/site-packages/nltk/lm/smoothing.pyú<lambda>z'_count_values_gt_zero.<locals>.<lambda>   s   € ˜5€ ó    c              3   ó:   •K  — | ]  } ‰|«      d kD  sŒd–— Œ y­w)r   é   Nr
   )Ú.0Údist_or_countÚas_counts     €r   ú	<genexpr>z(_count_values_gt_zero.<locals>.<genexpr>   s#   øè ø€ ò Ø¹ÀÓ8OÐRSÓ8SŒñùs   ƒ”)Ú
isinstancer   r   ÚsumÚvalues)Údistributionr   s    @r   Ú_count_values_gt_zeror      sG   ø€ ô �lÔ$7Ô8ô 	�SÔá ð ô ó Ø+×2Ñ2Ó4ôó ð r   c                   ó4   ‡ — e Zd ZdZˆ fd„Zd„ Zd„ Zd„ Zˆ xZS )Ú
WittenBellzWitten-Bell smoothing.c                 ó(   •— t        ‰| �  ||fi |¤Ž y r	   )ÚsuperÚ__init__)ÚselfÚ
vocabularyÚcounterÚkwargsÚ	__class__s       €r   r   zWittenBell.__init__'   s   ø€ Ü‰Ñ˜ WÑ7°Ó7r   c                 ót   — | j                   |   j                  |«      }| j                  |«      }d|z
  |z  |fS )Ng      ð?)ÚcountsÚfreqÚ_gamma©r   ÚwordÚcontextÚalphaÚgammas        r   Úalpha_gammazWittenBell.alpha_gamma*   s=   € Ø—‘˜GÑ$×)Ñ)¨$Ó/ˆØ—‘˜GÓ$ˆØ�e‘˜uÑ$ eÐ+Ð+r   c                 óx   — t        | j                  |   «      }||| j                  |   j                  «       z   z  S r	   )r   r%   r   ©r   r*   Ún_pluss      r   r'   zWittenBell._gamma/   s7   € Ü& t§{¡{°7Ñ';Ó<ˆØ˜ $§+¡+¨gÑ"6×"8Ñ"8Ó":Ñ:Ñ;Ð;r   c                 óL   — | j                   j                  j                  |«      S r	   ©r%   Úunigramsr&   ©r   r)   s     r   Úunigram_scorezWittenBell.unigram_score3   ó   € Ø�{‰{×#Ñ#×(Ñ(¨Ó.Ð.r   ©	Ú__name__Ú
__module__Ú__qualname__Ú__doc__r   r-   r'   r5   Ú__classcell__©r#   s   @r   r   r   $   s   ø„ Ù ô8ò,ò
<ö/r   r   c                   ó6   ‡ — e Zd ZdZdˆ fd„	Zd„ Zd„ Zd„ Zˆ xZS )ÚAbsoluteDiscountingz!Smoothing with absolute discount.c                 ó6   •— t        ‰| �  ||fi |¤Ž || _        y r	   )r   r   Údiscount)r   r    r!   rA   r"   r#   s        €r   r   zAbsoluteDiscounting.__init__:   s   ø€ Ü‰Ñ˜ WÑ7°Ò7Ø ˆ�r   c                 óº   — t        | j                  |   |   | j                  z
  d«      | j                  |   j                  «       z  }| j	                  |«      }||fS )Nr   )Úmaxr%   rA   r   r'   r(   s        r   r-   zAbsoluteDiscounting.alpha_gamma>   s\   € ä�—‘˜GÑ$ TÑ*¨T¯]©]Ñ:¸AÓ>Ø�k‰k˜'Ñ"×$Ñ$Ó&ñ'ð 	ð —‘˜GÓ$ˆØ�eˆ|Ðr   c                 óŒ   — t        | j                  |   «      }| j                  |z  | j                  |   j                  «       z  S r	   )r   r%   rA   r   r/   s      r   r'   zAbsoluteDiscounting._gammaF   s;   € Ü& t§{¡{°7Ñ';Ó<ˆØ—‘ Ñ&¨$¯+©+°gÑ*>×*@Ñ*@Ó*BÑBÐBr   c                 óL   — | j                   j                  j                  |«      S r	   r2   r4   s     r   r5   z!AbsoluteDiscounting.unigram_scoreJ   r6   r   )g      è?r7   r=   s   @r   r?   r?   7   s   ø„ Ù+õ!òòCö/r   r?   c                   óD   ‡ — e Zd ZdZdˆ fd„	Zd„ Zd„ Z e«       fd„Zˆ xZ	S )Ú	KneserNeyañ  Kneser-Ney Smoothing.

    This is an extension of smoothing with a discount.

    Resources:
    - https://pages.ucsd.edu/~rlevy/lign256/winter2008/kneser_ney_mini_example.pdf
    - https://www.youtube.com/watch?v=ody1ysUTD7o
    - https://medium.com/@dennyc/a-simple-numerical-example-for-kneser-ney-smoothing-nlp-4600addf38b8
    - https://www.cl.uni-heidelberg.de/courses/ss15/smt/scribe6.pdf
    - https://www-i6.informatik.rwth-aachen.de/publications/download/951/Kneser-ICASSP-1995.pdf
    c                 óD   •— t        ‰| �  ||fi |¤Ž || _        || _        y r	   )r   r   rA   Ú_order)r   r    r!   ÚorderrA   r"   r#   s         €r   r   zKneserNey.__init__[   s%   ø€ Ü‰Ñ˜ WÑ7°Ò7Ø ˆŒØˆ�r   c                 ó4   — | j                  |«      \  }}||z  S r	   )Ú_continuation_counts)r   r)   Úword_continuation_countÚtotal_counts       r   r5   zKneserNey.unigram_score`   s#   € Ø/3×/HÑ/HÈÓ/NÑ,Ð Ø&¨Ñ4Ð4r   c                 ó   — | j                   |   }t        |«      dz   | j                  k(  r||   |j                  «       fn| j	                  ||«      \  }}t        || j                  z
  d«      |z  }| j                  t        |«      z  |z  }||fS )Nr   g        )r%   ÚlenrI   r   rL   rC   rA   r   )r   r)   r*   Úprefix_countsrM   rN   r+   r,   s           r   r-   zKneserNey.alpha_gammad   s—   € ØŸ™ GÑ,ˆô �7‹|˜aÑ 4§;¡;Ò.ð ˜4Ñ  -§/¡/Ó"3Ñ4à×*Ñ*¨4°Ó9ñ 	-Ð ô
 Ð+¨d¯m©mÑ;¸SÓAÀKÑOˆØ—‘Ô 5°mÓ DÑDÀ{ÑRˆØ�eˆ|Ðr   c                 óÌ   ‡— ˆfd„| j                   t        ‰«      dz      j                  «       D «       }d\  }}|D ]$  }|t        ||   dkD  «      z  }|t	        |«      z  }Œ& ||fS )a  Count continuations that end with context and word.

        Continuations track unique ngram "types", regardless of how many
        instances were observed for each "type".
        This is different than raw ngram counts which track number of instances.
        c              3   ó8   •K  — | ]  \  }}|d d ‰k(  r|–— Œ y­w)r   Nr
   )r   Úprefix_ngramr%   r*   s      €r   r   z1KneserNey._continuation_counts.<locals>.<genexpr>v   s,   øè ø€ ò ,
á$�˜fØ˜A˜BÐ 7Ò*ô ñ,
ùs   ƒé   )r   r   r   )r%   rP   ÚitemsÚintr   )r   r)   r*   Ú higher_order_ngrams_with_contextÚ#higher_order_ngrams_with_word_countÚtotalr%   s     `    r   rL   zKneserNey._continuation_countso   s€   ø€ ó,
à(,¯©´C¸³LÀ1Ñ4DÑ(E×(KÑ(KÓ(Mô,
Ð(ð
 6:Ñ2Ð+¨UØ6ò 	3ˆFØ/´3°v¸d±|ÀaÑ7GÓ3HÑHÐ/ØÔ*¨6Ó2Ñ2‰Eð	3ð 3°EÐ9Ð9r   )gš™™™™™¹?)
r8   r9   r:   r;   r   r5   r-   ÚtuplerL   r<   r=   s   @r   rG   rG   N   s#   ø„ ñ
õò
5ò	ñ 27³÷ :r   rG   N)r;   Úoperatorr   Únltk.lm.apir   Únltk.probabilityr   r   r   r?   rG   r
   r   r   ú<module>r_      s>   ðñõ
 "å !Ý 0òô"/�ô /ô&/˜)ô /ô.1:�	õ 1:r   