Ë
    çÍ:jÀ/  ã                   ól   — d Z ddlZddlmZ ddlmZmZmZmZ ddl	m
Z
  G d„ de«      Z G d„ d	e
«      Zy)
a¸  
Lexical translation model that considers word order.

IBM Model 2 improves on Model 1 by accounting for word order.
An alignment probability is introduced, a(i | j,l,m), which predicts
a source word position, given its aligned target word's position.

The EM algorithm used in Model 2 is:

:E step: In the training data, collect counts, weighted by prior
         probabilities.

         - (a) count how many times a source language word is translated
               into a target language word
         - (b) count how many times a particular position in the source
               sentence is aligned to a particular position in the target
               sentence

:M step: Estimate new probabilities based on the counts from the E step

Notations
---------

:i: Position in the source sentence
     Valid values are 0 (for NULL), 1, 2, ..., length of source sentence
:j: Position in the target sentence
     Valid values are 1, 2, ..., length of target sentence
:l: Number of words in the source sentence, excluding NULL
:m: Number of words in the target sentence
:s: A word in the source language
:t: A word in the target language

References
----------

Philipp Koehn. 2010. Statistical Machine Translation.
Cambridge University Press, New York.

Peter E Brown, Stephen A. Della Pietra, Vincent J. Della Pietra, and
Robert L. Mercer. 1993. The Mathematics of Statistical Machine
Translation: Parameter Estimation. Computational Linguistics, 19 (2),
263-311.
é    N©Údefaultdict)ÚAlignedSentÚ	AlignmentÚIBMModelÚ	IBMModel1)ÚCountsc                   óT   ‡ — e Zd ZdZdˆ fd„	Zd„ Zd„ Zd„ Zd„ Zd„ Z	d„ Z
d	„ Zd
„ Zˆ xZS )Ú	IBMModel2u`  
    Lexical translation model that considers word order

    >>> bitext = []
    >>> bitext.append(AlignedSent(['klein', 'ist', 'das', 'haus'], ['the', 'house', 'is', 'small']))
    >>> bitext.append(AlignedSent(['das', 'haus', 'ist', 'ja', 'groÃŸ'], ['the', 'house', 'is', 'big']))
    >>> bitext.append(AlignedSent(['das', 'buch', 'ist', 'ja', 'klein'], ['the', 'book', 'is', 'small']))
    >>> bitext.append(AlignedSent(['das', 'haus'], ['the', 'house']))
    >>> bitext.append(AlignedSent(['das', 'buch'], ['the', 'book']))
    >>> bitext.append(AlignedSent(['ein', 'buch'], ['a', 'book']))

    >>> ibm2 = IBMModel2(bitext, 5)

    >>> print(round(ibm2.translation_table['buch']['book'], 3))
    1.0
    >>> print(round(ibm2.translation_table['das']['book'], 3))
    0.0
    >>> print(round(ibm2.translation_table['buch'][None], 3))
    0.0
    >>> print(round(ibm2.translation_table['ja'][None], 3))
    0.0

    >>> print(round(ibm2.alignment_table[1][1][2][2], 3))
    0.939
    >>> print(round(ibm2.alignment_table[1][2][2][2], 3))
    0.0
    >>> print(round(ibm2.alignment_table[2][2][4][5], 3))
    1.0

    >>> test_sentence = bitext[2]
    >>> test_sentence.words
    ['das', 'buch', 'ist', 'ja', 'klein']
    >>> test_sentence.mots
    ['the', 'book', 'is', 'small']
    >>> test_sentence.alignment
    Alignment([(0, 0), (1, 1), (2, 2), (3, 2), (4, 3)])

    c                 ó  •— t         ‰| �  |«       |€2t        |d|z  «      }|j                  | _        | j	                  |«       n|d   | _        |d   | _        t        d|«      D ]  }| j                  |«       Œ | j                  |«       y)a™  
        Train on ``sentence_aligned_corpus`` and create a lexical
        translation model and an alignment model.

        Translation direction is from ``AlignedSent.mots`` to
        ``AlignedSent.words``.

        :param sentence_aligned_corpus: Sentence-aligned parallel corpus
        :type sentence_aligned_corpus: list(AlignedSent)

        :param iterations: Number of iterations to run training algorithm
        :type iterations: int

        :param probability_tables: Optional. Use this to pass in custom
            probability values. If not specified, probabilities will be
            set to a uniform distribution, or some other sensible value.
            If specified, all the following entries must be present:
            ``translation_table``, ``alignment_table``.
            See ``IBMModel`` for the type and purpose of these tables.
        :type probability_tables: dict[str]: object
        Né   Útranslation_tableÚalignment_tabler   )	ÚsuperÚ__init__r   r   Úset_uniform_probabilitiesr   ÚrangeÚtrainÚ	align_all)ÚselfÚsentence_aligned_corpusÚ
iterationsÚprobability_tablesÚibm1ÚnÚ	__class__s         €úh/home/mcse/projects/srt_converter/srt-converter-venv/lib/python3.12/site-packages/nltk/translate/ibm2.pyr   zIBMModel2.__init__c   s™   ø€ ô, 	‰ÑÐ0Ô1àÐ%ô Ð4°a¸*±nÓEˆDØ%)×%;Ñ%;ˆDÔ"Ø×*Ñ*Ð+BÕCð &8Ð8KÑ%LˆDÔ"Ø#5Ð6GÑ#HˆDÔ ä�q˜*Ó%ò 	0ˆAØ�J‰JÐ.Õ/ð	0ð 	�‰Ð.Õ/ó    c                 ó¬  — t        «       }|D ]Å  }t        |j                  «      }t        |j                  «      }||f|vsŒ4|j	                  ||f«       d|dz   z  }|t
        j                  k  r$t        j                  dt        |«      z   dz   «       t        d|dz   «      D ].  }t        d|dz   «      D ]  }|| j                  |   |   |   |<   Œ Œ0 ŒÇ y )Né   zA source sentence is too long (z& words). Results may be less accurate.r   )ÚsetÚlenÚmotsÚwordsÚaddr   ÚMIN_PROBÚwarningsÚwarnÚstrr   r   )	r   r   Úl_m_combinationsÚaligned_sentenceÚlÚmÚinitial_probÚiÚjs	            r   r   z#IBMModel2.set_uniform_probabilitiesŒ   só   € ä›5ÐØ 7ò 	HÐÜÐ$×)Ñ)Ó*ˆAÜÐ$×*Ñ*Ó+ˆAØ�1ˆvÐ-Ò-Ø ×$Ñ$ a¨ VÔ,Ø  A¨¡E™{�Ø¤(×"3Ñ"3Ò3Ü—M‘MØ9Ü˜a›&ñ!àBñCôô ˜q ! a¡%›ò H�AÜ" 1 a¨!¡e›_ò H˜Ø;G˜×,Ñ,¨QÑ/°Ñ2°1Ñ5°aÒ8ñHñHñ	Hr   c           
      ó  — t        «       }|D ]Ô  }d g|j                  z   }dg|j                  z   }t        |j                  «      }t        |j                  «      }| j	                  ||«      }t        d|dz   «      D ]d  }	||	   }
t        d|dz   «      D ]K  }||   }| j                  ||	||«      }|||
   z  }|j                  |||
«       |j                  |||	||«       ŒM Œf ŒÖ | j                  |«       | j                  |«       y )NÚUNUSEDr    r   )ÚModel2Countsr#   r$   r"   Úprob_all_alignmentsr   Úprob_alignment_pointÚupdate_lexical_translationÚupdate_alignmentÚ*maximize_lexical_translation_probabilitiesÚ maximize_alignment_probabilities)r   Úparallel_corpusÚcountsr+   Úsrc_sentenceÚtrg_sentencer,   r-   Útotal_countr0   Útr/   ÚsÚcountÚnormalized_counts                  r   r   zIBMModel2.train    s.  € Ü“ˆØ /ò 	JÐØ ˜6Ð$4×$9Ñ$9Ñ9ˆLØ$˜:Ð(8×(>Ñ(>Ñ>ˆLÜÐ$×)Ñ)Ó*ˆAÜÐ$×*Ñ*Ó+ˆAð ×2Ñ2°<ÀÓNˆKô ˜1˜a !™e“_ò J�Ø  ‘O�Ü˜q ! a¡%›ò J�AØ$ Q™�AØ ×5Ñ5°a¸¸LÈ,ÓW�EØ',¨{¸1©~Ñ'=Ð$à×5Ñ5Ð6FÈÈ1ÔMØ×+Ñ+Ð,<¸aÀÀAÀqÕIñJñJð	Jð* 	×7Ñ7¸Ô?Ø×-Ñ-¨fÕ5r   c                 óv  — t         j                  }|j                  j                  «       D ]Œ  \  }}|j                  «       D ]t  \  }}|j                  «       D ]\  \  }}|D ]R  }	|j                  |   |   |   |	   |j                  |   |   |	   z  }
t        |
|«      | j                  |   |   |   |	<   ŒT Œ^ Œv ŒŽ y ©N)r   r&   Ú	alignmentÚitemsÚalignment_for_any_iÚmaxr   )r   r;   r&   r/   Új_sr0   Úsrc_sentence_lengthsr,   Útrg_sentence_lengthsr-   Úestimates              r   r9   z*IBMModel2.maximize_alignment_probabilitiesº   sç   € Ü×$Ñ$ˆØ×&Ñ&×,Ñ,Ó.ò 	S‰FˆAˆsØ+.¯9©9«;ò SÑ'�Ð'Ø/C×/IÑ/IÓ/Kò SÑ+�AÐ+Ø1ò S˜à"×,Ñ,¨QÑ/°Ñ2°1Ñ5°aÑ8Ø$×8Ñ8¸Ñ;¸AÑ>¸qÑAñBð !ô <?¸xÈÓ;R˜×,Ñ,¨QÑ/°Ñ2°1Ñ5°aÒ8ñSñSñSñ	Sr   c                 óÔ   — t        t        «      }t        dt        |«      «      D ]@  }||   }t        dt        |«      «      D ]!  }||xx   | j	                  ||||«      z  cc<   Œ# ŒB |S )aï  
        Computes the probability of all possible word alignments,
        expressed as a marginal distribution over target words t

        Each entry in the return value represents the contribution to
        the total alignment probability by the target word t.

        To obtain probability(alignment | src_sentence, trg_sentence),
        simply sum the entries in the return value.

        :return: Probability of t for all s in ``src_sentence``
        :rtype: dict(str): float
        r    r   )r   Úfloatr   r"   r5   )r   r<   r=   Úalignment_prob_for_tr0   r?   r/   s          r   r4   zIBMModel2.prob_all_alignmentsÆ   s|   € ô  +¬5Ó1ÐÜ�qœ#˜lÓ+Ó,ò 	ˆAØ˜Q‘ˆAÜ˜1œc ,Ó/Ó0ò �Ø$ QÓ'¨4×+DÑ+DØ�q˜,¨ó,ñ Ô'ñð	ð $Ð#r   c                 ó¤   — t        |«      dz
  }t        |«      dz
  }||   }||   }| j                  |   |   | j                  |   |   |   |   z  S )zz
        Probability that position j in ``trg_sentence`` is aligned to
        position i in the ``src_sentence``
        r    )r"   r   r   )	r   r/   r0   r<   r=   r,   r-   r@   r?   s	            r   r5   zIBMModel2.prob_alignment_pointÝ   si   € ô
 �Ó Ñ!ˆÜ�Ó Ñ!ˆØ˜‰OˆØ˜‰OˆØ×%Ñ% aÑ(¨Ñ+¨d×.BÑ.BÀ1Ñ.EÀaÑ.HÈÑ.KÈAÑ.NÑNÐNr   c                 óx  — d}t        |j                  «      dz
  }t        |j                  «      dz
  }t        |j                  «      D ]W  \  }}|dk(  rŒ|j                  |   }|j                  |   }|| j
                  |   |   | j                  |   |   |   |   z  z  }ŒY t        |t        j                  «      S )zc
        Probability of target sentence and an alignment given the
        source sentence
        g      ð?r    r   )
r"   r<   r=   Ú	enumeraterE   r   r   rH   r   r&   )	r   Úalignment_infoÚprobr,   r-   r0   r/   Útrg_wordÚsrc_words	            r   Úprob_t_a_given_szIBMModel2.prob_t_a_given_sè   sÏ   € ð
 ˆÜ�×+Ñ+Ó,¨qÑ0ˆÜ�×+Ñ+Ó,¨qÑ0ˆä˜n×6Ñ6Ó7ò 	‰DˆAˆqØ�AŠvØØ%×2Ñ2°1Ñ5ˆHØ%×2Ñ2°1Ñ5ˆHØØ×&Ñ& xÑ0°Ñ:Ø×&Ñ& qÑ)¨!Ñ,¨QÑ/°Ñ2ñ3ñ‰Dð	ô �4œ×*Ñ*Ó+Ð+r   c                 ó4   — |D ]  }| j                  |«       Œ y rD   )Úalign)r   r:   Úsentence_pairs      r   r   zIBMModel2.align_allý   s   € Ø,ò 	&ˆMØ�J‰J�}Õ%ñ	&r   c                 ó   — g }t        |j                  «      }t        |j                  «      }t        |j                  «      D ]º  \  }}| j                  |   d   | j
                  d   |dz      |   |   z  }t        |t        j                  «      }d}t        |j                  «      D ]@  \  }	}
| j                  |   |
   | j
                  |	dz      |dz      |   |   z  }||k\  sŒ=|}|	}ŒB |j                  ||f«       Œ¼ t        |«      |_        y)a  
        Determines the best word alignment for one sentence pair from
        the corpus that the model was trained on.

        The best alignment will be set in ``sentence_pair`` when the
        method returns. In contrast with the internal implementation of
        IBM models, the word indices in the ``Alignment`` are zero-
        indexed, not one-indexed.

        :param sentence_pair: A sentence in the source language and its
            counterpart sentence in the target language
        :type sentence_pair: AlignedSent
        Nr   r    )r"   r#   r$   rR   r   r   rH   r   r&   Úappendr   rE   )r   rZ   Úbest_alignmentr,   r-   r0   rU   Ú	best_probÚbest_alignment_pointr/   rV   Ú
align_probs               r   rY   zIBMModel2.align  s;  € ð ˆä�×"Ñ"Ó#ˆÜ�×#Ñ#Ó$ˆä$ ]×%8Ñ%8Ó9ò 	=‰KˆAˆxð ×&Ñ& xÑ0°Ñ6Ø×&Ñ& qÑ)¨!¨a©%Ñ0°Ñ3°AÑ6ñ7ð ô ˜I¤x×'8Ñ'8Ó9ˆIØ#'Ð Ü(¨×);Ñ);Ó<ò -‘��8à×*Ñ*¨8Ñ4°XÑ>Ø×*Ñ*¨1¨q©5Ñ1°!°a±%Ñ8¸Ñ;¸AÑ>ñ?ð ð  Ó*Ø *�IØ+,Ñ(ð-ð ×!Ñ! 1Ð&:Ð";Õ<ð#	=ô& #,¨NÓ";ˆÕr   rD   )Ú__name__Ú
__module__Ú__qualname__Ú__doc__r   r   r   r9   r4   r5   rW   r   rY   Ú__classcell__©r   s   @r   r   r   ;   s:   ø„ ñ%õN'0òRHò(6ò4
Sò$ò.	Oò,ò*&ö&<r   r   c                   ó.   ‡ — e Zd ZdZˆ fd„Zd„ Zd„ Zˆ xZS )r3   zo
    Data object to store counts of various parameters during training.
    Includes counts for alignment.
    c                 óf   •— t         ‰| �  «        t        d„ «      | _        t        d„ «      | _        y )Nc                  ó   — t        d„ «      S )Nc                  ó   — t        d„ «      S )Nc                  ó    — t        t        «      S rD   ©r   rN   © r   r   ú<lambda>zKModel2Counts.__init__.<locals>.<lambda>.<locals>.<lambda>.<locals>.<lambda>3  s   € ¼KÌÓ<N€ r   r   rm   r   r   rn   z9Model2Counts.__init__.<locals>.<lambda>.<locals>.<lambda>3  s   € ¬Ñ4NÓ(O€ r   r   rm   r   r   rn   z'Model2Counts.__init__.<locals>.<lambda>3  s   € ”KÑ OÓP€ r   c                  ó   — t        d„ «      S )Nc                  ó    — t        t        «      S rD   rl   rm   r   r   rn   z9Model2Counts.__init__.<locals>.<lambda>.<locals>.<lambda>6  s   € ¬´EÓ(:€ r   r   rm   r   r   rn   z'Model2Counts.__init__.<locals>.<lambda>6  s   € ”KÑ :Ó;€ r   )r   r   r   rE   rG   )r   r   s    €r   r   zModel2Counts.__init__0  s/   ø€ Ü‰ÑÔÜ$ÙPó
ˆŒô $/Ù;ó$
ˆÕ r   c                 óf   — | j                   |   |xx   |z  cc<   | j                  |xx   |z  cc<   y rD   )Ú	t_given_sÚany_t_given_s)r   rA   r@   r?   s       r   r6   z'Model2Counts.update_lexical_translation9  s1   € Ø�‰�qÑ˜!Ó Ñ%ÓØ×Ñ˜1Ó Ñ&Ôr   c                 ó~   — | j                   |   |   |   |xx   |z  cc<   | j                  |   |   |xx   |z  cc<   y rD   )rE   rG   )r   rA   r/   r0   r,   r-   s         r   r7   zModel2Counts.update_alignment=  sE   € Ø�‰�qÑ˜!Ñ˜QÑ Ó" eÑ+Ó"Ø× Ñ  Ñ# AÑ& qÓ)¨UÑ2Ô)r   )ra   rb   rc   rd   r   r6   r7   re   rf   s   @r   r3   r3   *  s   ø„ ñô

ò'ö3r   r3   )rd   r'   Úcollectionsr   Únltk.translater   r   r   r   Únltk.translate.ibm_modelr	   r   r3   rm   r   r   ú<module>rx      s7   ðñ*óX Ý #ç FÓ FÝ +ôl<�ô l<ô^3�6õ 3r   