<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE ep-patent-document PUBLIC "-//EPO//EP PATENT DOCUMENT 1.5//EN" "ep-patent-document-v1-5.dtd">
<ep-patent-document id="EP09725780B1" file="EP09725780NWB1.xml" lang="en" country="EP" doc-number="2269148" kind="B1" date-publ="20180905" status="n" dtd-version="ep-patent-document-v1-5">
<SDOBI lang="en"><B000><eptags><B001EP>ATBECHDEDKESFRGBGRITLILUNLSEMCPTIESILTLVFIROMKCY..TRBGCZEEHUPLSK..HRIS..MTNO........................</B001EP><B003EP>*</B003EP><B005EP>J</B005EP><B007EP>BDM Ver 0.1.63 (23 May 2017) -  2100000/0</B007EP></eptags></B000><B100><B110>2269148</B110><B120><B121>EUROPEAN PATENT SPECIFICATION</B121></B120><B130>B1</B130><B140><date>20180905</date></B140><B190>EP</B190></B100><B200><B210>09725780.2</B210><B220><date>20090227</date></B220><B240><B241><date>20101021</date></B241><B242><date>20170209</date></B242></B240><B250>en</B250><B251EP>en</B251EP><B260>en</B260></B200><B300><B310>58328</B310><B320><date>20080328</date></B320><B330><ctry>US</ctry></B330></B300><B400><B405><date>20180905</date><bnum>201836</bnum></B405><B430><date>20110105</date><bnum>201101</bnum></B430><B450><date>20180905</date><bnum>201836</bnum></B450><B452EP><date>20180329</date></B452EP></B400><B500><B510EP><classification-ipcr sequence="1"><text>G06F  17/28        20060101AFI20091019BHEP        </text></classification-ipcr><classification-ipcr sequence="2"><text>G06F  17/30        20060101ALN20110523BHEP        </text></classification-ipcr></B510EP><B540><B541>de</B541><B542>SPRACHENINTERNE STATISTISCHE MASCHINENÜBERSETZUNG</B542><B541>en</B541><B542>INTRA-LANGUAGE STATISTICAL MACHINE TRANSLATION</B542><B541>fr</B541><B542>TRADUCTION AUTOMATIQUE STATISTIQUE INTRA-LANGUES</B542></B540><B560><B561><text>EP-A2- 1 217 534</text></B561><B561><text>US-A- 5 991 710</text></B561><B561><text>US-A1- 2006 217 963</text></B561><B561><text>US-A1- 2007 083 359</text></B561><B562><text>STEFAN RIEZLER ET AL: "Statistical Machine Translation for Query Expansion in Answer Retrieval", ALC 2007 PROCEEDINGS,, [Online] 23 June 2007 (2007-06-23), XP008126878, Retrieved from the Internet: URL:http://www.stefanriezler.com/PAPERS/AC L07.pdf&gt;</text></B562><B562><text>ADAM BERGER ET AL: "Information retrieval as statistical translation", PROC. SIGIR 1999; [ANNUAL INTERNATIONAL ACM-SIGIR CONFERENCE ON RESEARCH AND DEVELOPMENT IN INFORMATION RETRIEVAL], UNIV. OF CALIFORNIA, BERKELEY, USA, [Online] 15 August 1999 (1999-08-15), pages 222-229, XP002495566, ISBN: 978-1-58113-096-6 Retrieved from the Internet: URL:http://portal.acm.org/citation.cfm?id= 312681&amp;dl=&gt; [retrieved on 2008-09-12]</text></B562><B562><text>QUIRK C ET AL: "Monolingual Machine Translation for Paraphrase Generation", PROCEEDINGS OF THE CONFERENCE ON EMPIRICAL METHODS IN NATURALLANGUAGE PROCESSING, XX, XX, 25 July 2004 (2004-07-25), pages 142-149, XP002372807,</text></B562><B562><text>XIAO LI, Y.-C. JU, GEOFFREY ZWEIG, AND ALEX ACERO: "LANGUAGE MODELING FOR VOICE SEARCH: A MACHINE TRANSLATION APPROACH", IN PROCEEDINGS OF ICASSP 2008, 31 March 2008 (2008-03-31), - 4 April 2008 (2008-04-04), pages 4913-4916, XP002637698, Las Vegas, Nevada, U.S.A.</text></B562><B562><text>JOSEP M. CREGO, ADRIA DE GISPERT, JOSE B. MARINO.: "THE TALP NGRAM-BASED SMT SYSTEM FOR IWSLT'05", IWSLT 2005 INTERNATIONAL WORKSHOP ON SPOKEN LANGUAGE TRANSLATION, 24 October 2005 (2005-10-24), - 25 October 2005 (2005-10-25), XP002637699, Pittsburgh, PA, USA</text></B562><B562><text>POPOVICI C ET AL: "Learning of User Formulations for Business Listings in Automatic Directory Assistance", IEEE WORKSHOP ON AUTOMATIC SPEECH RECOGNITION AND UNDERSTANDING, 2001. ASRU '01, vol. 4, 9 December 2001 (2001-12-09), - 13 December 2001 (2001-12-13), page 2325, XP007004855,</text></B562><B565EP><date>20110531</date></B565EP></B560></B500><B700><B720><B721><snm>LI, Xiao</snm><adr><str>c/o Microsoft Corporation
One Microsoft Way</str><city>Redmond, WA 98052-6399</city><ctry>US</ctry></adr></B721><B721><snm>JU, Yun-cheng</snm><adr><str>c/o Microsoft Corporation
One Microsoft Way</str><city>Redmond, WA 98052-6399</city><ctry>US</ctry></adr></B721><B721><snm>ZWEIG, Geoffrey</snm><adr><str>c/o Microsoft Corporation
One Microsoft Way</str><city>Redmond, WA 98052-6399</city><ctry>US</ctry></adr></B721><B721><snm>ACERO, Alex</snm><adr><str>c/o Microsoft Corporation
One Microsoft Way</str><city>Redmond, WA 98052-6399</city><ctry>US</ctry></adr></B721></B720><B730><B731><snm>Microsoft Technology Licensing, LLC</snm><iid>101490473</iid><irf>EP72104RK900kap</irf><adr><str>One Microsoft Way</str><city>Redmond, WA 98052</city><ctry>US</ctry></adr></B731></B730><B740><B741><snm>Grünecker Patent- und Rechtsanwälte 
PartG mbB</snm><iid>100060488</iid><adr><str>Leopoldstraße 4</str><city>80802 München</city><ctry>DE</ctry></adr></B741></B740></B700><B800><B840><ctry>AT</ctry><ctry>BE</ctry><ctry>BG</ctry><ctry>CH</ctry><ctry>CY</ctry><ctry>CZ</ctry><ctry>DE</ctry><ctry>DK</ctry><ctry>EE</ctry><ctry>ES</ctry><ctry>FI</ctry><ctry>FR</ctry><ctry>GB</ctry><ctry>GR</ctry><ctry>HR</ctry><ctry>HU</ctry><ctry>IE</ctry><ctry>IS</ctry><ctry>IT</ctry><ctry>LI</ctry><ctry>LT</ctry><ctry>LU</ctry><ctry>LV</ctry><ctry>MC</ctry><ctry>MK</ctry><ctry>MT</ctry><ctry>NL</ctry><ctry>NO</ctry><ctry>PL</ctry><ctry>PT</ctry><ctry>RO</ctry><ctry>SE</ctry><ctry>SI</ctry><ctry>SK</ctry><ctry>TR</ctry></B840><B860><B861><dnum><anum>US2009035389</anum></dnum><date>20090227</date></B861><B862>en</B862></B860><B870><B871><dnum><pnum>WO2009120449</pnum></dnum><date>20091001</date><bnum>200940</bnum></B871></B870></B800></SDOBI>
<description id="desc" lang="en"><!-- EPO <DP n="1"> -->
<heading id="h0001"><b>BACKGROUND</b></heading>
<p id="p0001" num="0001">Network based search services, Internet search engines, voice search, local search, and various other technologies for searching and retrieving information have become increasingly important for helping people find information. Voice search involves a coupling of voice recognition and information retrieval. An uttered phrase is automatically recognized as text, and the text is submitted as a query to a search service. For example, a person may use a mobile phone equipped with a voice search application to find a restaurant by speaking the name of the restaurant into the mobile device, and the mobile device may recognize the spoken restaurant name (i.e., convert it to text) and transmit the text of the restaurant name to a remote search service such as a business directory. Local search is a special case of search where listings of business establishments, firms, organizations, or other entities have been used to enable mobile devices to search same. Consider the following example.</p>
<p id="p0002" num="0002">A user may be interested in finding information about a business listed in a directory as "Kung Ho Cuisine of China". However, the user formulates a query as "Kung Ho Restaurant". Currently, a search for this listing will not take advantage of statistical parallels between parts of the query and listing forms. Furthermore, erroneous listings, e.g. "Kung Ho Grocery" may be returned as a relevant match.</p>
<p id="p0003" num="0003">Discussed below are techniques related to statistical intra-language machine translation, and applications thereof to speech recognition, search, and other technologies.<!-- EPO <DP n="2"> --></p>
<p id="p0004" num="0004"><nplcit id="ncit0001" npl-type="s"><text>Stefan Riezler et al.: "Statistical Machine Translation for Query Expansion in Answer Retrieval", ALC 2007 Proceedings, June 23, 2007</text></nplcit>, relates to system and methods for query expansion in answer retrieval that uses Statistical Machine Translation techniques to bridge the lexical gap between questions and answers. Statistical Machine Translation (SMT)-based query expansion is done by i) using a full-sentence paraphraser to introduce synonyms in context of the entire query, and ii) by translating query terms into answer terms using a full-sentence SMT model trained on question-answer pairs. We evaluate these global, context-aware query expansion techniques on <i>tfidf</i> retrieval from 10 million question-answer pairs extracted from FAQ pages. Experimental results show that SMTbased expansion improves retrieval performance over local expansion and over retrieval without expansion.</p>
<p id="p0005" num="0005"><nplcit id="ncit0002" npl-type="b"><text>Adam Berger et al.: "Information Retrieval as Statistical Translation", PROC. SIGIR 1999 (Annual International ACM-SIGIR conference on research and development in information retrieval), University of California, Berkeley, USA, August 15, 1999, pages 222-229</text></nplcit>, describes a new probabilistic approach to information retrieval based upon the ideas and methods of statistical machine translation. The central ingredient in this approach is a statistical model of how a user might distill or "translate" a given document into a query. To assess the relevance of a document to a user's query, the probability that the query would have been generated as a translation of the document is estimated, and factor in the user's general preferences in the form of a prior distribution over documents. A simple, well motivated model of the document-to-query translation process is proposed, and it is also described an algorithm for learning the parameters of this model in an unsupervised manner from a collection of documents.</p>
<p id="p0006" num="0006"><nplcit id="ncit0003" npl-type="s"><text>C. Quirk et al.: "Monolingual Machine Translation for Paraphrase Generation", Proceedings of the Conference on Empirical Methods in Natural Language Processing, July 25, 2004, pages 142-149</text></nplcit>. In this document, statistical machine translation tools are applied to generate novel paraphrases of input sentences in the same language. The system is trained on large volumes of sentence pairs automatically extracted from clustered news articles available on the World Wide Web. Alignment Error Rate is measured to gauge the quality of the resulting corpus. A monotone phrasal decoder generates contextual replacements. Human evaluation shows that this system outperforms baseline paraphrase generation techniques and, in a departure from previous work, offers better coverage and scalability than the current best-of-breed paraphrasing approaches.</p>
<p id="p0007" num="0007"><nplcit id="ncit0004" npl-type="s"><text>Josep M. Crego, Adria de Gispert, Jose B. Marino.: "The TALP NGRAM-Based SMT System for IWSLT'05" October 24, Pittsburgh, PA, USA</text></nplcit>, provides a description of TALP-Ngram, the tuple-based statistical machine translation system developed at the TALP Research Center of the UPC (Universitat Politecnica de Catalunya). Briefly, the system performs a log-linear combination of a translation model and additional feature functions. The translation model is estimated as an N-gram of bilingual units called<!-- EPO <DP n="3"> --> tuples, and the feature functions include a target language model, a word penalty, and lexical features, depending on the language pair and task.</p>
<heading id="h0002"><b>SUMMARY</b></heading>
<p id="p0008" num="0008">It is the object of the present invention to provide a system and a method for automatically translating between spoken queries and listings.</p>
<p id="p0009" num="0009">This object is solved by the subject matter of the independent claims.</p>
<p id="p0010" num="0010">Preferred embodiments are defined by the dependent claims.<!-- EPO <DP n="4"> --></p>
<p id="p0011" num="0011">The following summary is included only to introduce some concepts discussed in the Detailed Description below. This summary is not comprehensive and is not intended to delineate the scope of the claimed subject matter, which is set forth by the claims presented at the end.<!-- EPO <DP n="5"> --></p>
<p id="p0012" num="0012">Training data may be provided. The training data may include pairs of source phrases and target phrases. The pairs may be used to train an intra-language statistical machine translation model, where the intra-language statistical machine translation model, when given an input phrase of text in the human language, can compute probabilities of semantic equivalence of the input phrase to possible translations of the input phrase in the human language. The statistical machine translation model may be used to translate between queries and listings. The queries may be text strings in the human language submitted to a search engine. The listing strings may be text strings of formal names of real world entities that are to be searched by the search engine to find matches for the query strings.</p>
<p id="p0013" num="0013">Many of the attendant features will be explained below with reference to the following detailed description considered in connection with the accompanying drawings.</p>
<heading id="h0003"><b>BRIEF DESCRIPTION OF THE DRAWINGS</b></heading>
<p id="p0014" num="0014">The present description will be better understood from the following detailed description read in light of the accompanying drawings, wherein like reference numerals are used to designate like parts in the accompanying description.
<ul id="ul0001" list-style="none" compact="compact">
<li><figref idref="f0001">Figure 1</figref> shows a general process for intra-language statistical machine translation.</li>
<li><figref idref="f0002">Figure 2</figref> shows a process for building an n-gram based model.</li>
<li><figref idref="f0003">Figure 3</figref> shows an arrangement for using a statistical translation model to improve a search system and/or the language model of a voice recognition system.</li>
</ul></p>
<heading id="h0004"><b>DETAILED DESCRIPTION</b></heading>
<heading id="h0005"><b>OVERVIEW</b></heading>
<p id="p0015" num="0015">The description below covers embodiments related to using a statistical machine translation model to translate between sentences or phrases of a same human language. The description begins with discussion of how a relatively small set of training sentences or phrases are used to train a statistical translation model. Applications of the<!-- EPO <DP n="6"> --> intra-language machine translation model are then described, including applications to search, automatic speech recognition (ASR), and display of speech recognition results</p>
<heading id="h0006"><b>INTRALANGUAGE STATISTICAL MACHINE TRANSLATION MODEL</b></heading>
<p id="p0016" num="0016">Statistical models have been used to translate sentences from one language to another language. However, they have not been trained or used for translating between phrases or sentences of a same language. That is, statistical modeling has not previously been used to translate phrases in English, for example, to other semantically similar phrases also in English.</p>
<p id="p0017" num="0017">A statistical translation model is a generalization of some sample of text, which may be parallel phrases such as a query strings and corresponding directory listings. Some types of statistical translation models give probabilities that a target sentence or phrase is a translation of a source sentence or phrase, and the probabilities reflect the statistical patterns derived from the training text. In effect, the model is a probabilistic generalization of characteristics or trends reflected from statistical measurements of training sentences. Note that throughout this description, the terms "sentence" and "phrase" will be used interchangeably to refer to relatively short arrangements of words. Formal and informal names of businesses, query strings inputted by users, grammatical sentences, clauses, and the like are examples of sentences or phrases. Note also that while this description discusses intra-language statistical machine translation as applied to phrase-based search (in particular voice and/or geographically localized search), the concepts are not limited to search applications. Furthermore, searching listings of short phrases is also applicable to other types of search besides local search, including product search, job search, etc.</p>
<p id="p0018" num="0018"><figref idref="f0001">Figure 1</figref> shows a general process for intra-language statistical machine translation. Initially, a statistical machine translation model is trained 100. Training 100 will be described in detail later. The training 100 is performed using a sample of training data, which may come from a variety of sources. The training data will include parallel (paired) phrases in a same human language. The training 100 informs the translation model<!-- EPO <DP n="7"> --> with statistics (e.g., n-grams) that can be used to compute probabilities or likelihoods of candidate translations of a phrase. Specific training 100 for an n-gram based model will be described below.</p>
<p id="p0019" num="0019">After the model is trained 100, the model is used to translate 102 a source phrase into a target phrase. Translation 102 involves starting with a source phrase and obtaining a semantically similar or equivalent target phrase. For example, a source phrase "Kung Ho Cuisine of China" might be translated to a target phrase "Kung Ho Chinese Restaurant" or "Kung Ho Restaurant" Different forms of candidate target phrases are obtained. The statistical translation model is used to find one or more of the most probable candidate target phrases. Consider the following overview of voice based search and how it relates to intra-language machine translation.</p>
<p id="p0020" num="0020">A voice search system may involve two components: a voice recognition component and an information retrieval (search) component. A spoken utterance <b>o</b> is converted into a text query <b>q</b> using automatic speech recognition (ASR), i.e., <maths id="math0001" num="(1)"><math display="block"><msup><mi>q</mi><mo>*</mo></msup><mo>=</mo><msub><mi>argmax</mi><mi>q</mi></msub><mi>p</mi><mfenced><mrow><mrow><mi>o</mi><mo>|</mo></mrow><mi>q</mi></mrow></mfenced><mi>p</mi><mfenced><mi>q</mi></mfenced></math><img id="ib0001" file="imgb0001.tif" wi="108" he="5" img-content="math" img-format="tif"/></maths> where p(o|q) and p(q) represent an acoustic model and a language model (LM), respectively. Statistical LMs, e.g. n-gram models, are often used to allow flexibility in what users can say. That is, they allow a variety of sayings to be recognized by the ASR component. Next, the best (or n-best) <b>q</b> is passed to a search engine to retrieve the most relevant document d, i.e. <maths id="math0002" num="(2)"><math display="block"><mi>d</mi><mo>*</mo><mo>=</mo><msub><mi>argmax</mi><mi>d</mi></msub><mi>p</mi><mfenced><mrow><mrow><mi>d</mi><mo>|</mo></mrow><mi>q</mi></mrow></mfenced></math><img id="ib0002" file="imgb0002.tif" wi="108" he="5" img-content="math" img-format="tif"/></maths></p>
<p id="p0021" num="0021">In the context of local search, documents <b>d</b> may be in the form of business listings (names of business, organizations, or other entities), which are typically short, e.g. "Kung Ho Cuisine of China".</p>
<p id="p0022" num="0022">Given this framework for voice based search, listings and queries, because they are both relatively short, are treated as pairs akin to "sentence pairs" found in bilingual translation training. A bilingual statistical translation model, adapted for intra-language translation, may be used to automatically convert the original form of a listing to its query forms (i.e., the forms that a user might be expected to input when searching for the listing),<!-- EPO <DP n="8"> --> which in turn may be used for building more robust LMs for voice search, grammar checking, or other applications. Conveniently, the statistical translation model may be trained using a small number of transcribed or artificially produced queries, without necessarily having to acquire matching listings. While a variety of types of statistical models can be used for machine translation, an n-gram based model will be described next.</p>
<p id="p0023" num="0023">Although a query phrase and its intended listing phrase may differ in form, there is usually a semantic correspondence, at the word level, between the two phrases. In other words, words in the query can be mapped to words in the listing or to a null word, and vice versa. A machine translation approach may be used to predict possible query forms of a listing, and then to utilize the predicted query forms to improve language modeling. Specifically, as discussed next, n-grams may be used on word pairs to model the joint (conditional) probability of a listing and a query.</p>
<p id="p0024" num="0024"><figref idref="f0002">Figure 2</figref> shows a process for building an n-gram based model. A pair of source and target sentences in a same human language are received 120. An alignment between the source and target sentences is obtained 122 by computing an edit distance between the two sentences. Words and/or phrases of the aligned sentences are then paired 124 and treated as semantic units. Pairings may be formed by finding semantically/literally similar/equivalent words or phrases. The pairings are then used to train 126 an n-gram model. The steps of this process may be repeated for different source and target sentences. While a small set of training sentences may suffice for some applications, using more training data will create a more robust model. Note also that the alignment and the n-gram model may be iteratively updated and refined in the maximum likelihood sense.</p>
<p id="p0025" num="0025">Details of generating an n-gram based model will now be described. For training 100 an n-gram based model, initial training data is provided. This data may be a body of parallel text (d, q), where listings <b>d</b> and queries <b>q</b> serve as source and target sentences respectively. The sentences <b>d</b> and <b>q</b> may be monotonically aligned, where null words are added, if necessary, to account for insertions or deletions that occur in the alignment. The monotonic alignment will be denoted as <b>a.</b> Note that in another embodiment, a non-monotonic alignment may be used.<!-- EPO <DP n="9"> --></p>
<p id="p0026" num="0026">Once aligned, a sequence pairs of words from d and q is generated, which is denoted as (d, q, a) = ((d<sub>1</sub>, q<sub>1</sub>), (d<sub>2</sub>, q<sub>2</sub>), ..., (d<sub>L</sub>, q<sub>L</sub>)), where each (d<sub>i</sub>, q<sub>i</sub>) is treated as a single semantic unit. Consecutive word pairs can be merged to form phrase pairs if necessary.</p>
<p id="p0027" num="0027">The sequence of word pairs can then be used to train an n-gram model. Consequently, the probability of an aligned sentence pair is computed as <maths id="math0003" num="(3)"><math display="block"><mtable columnalign="left"><mtr><mtd><mfenced><mrow><mi>d</mi><mo>,</mo><mi>q</mi><mo>,</mo><mi>a</mi></mrow></mfenced><mo>=</mo></mtd></mtr><mtr><mtd><mstyle displaystyle="true"><msubsup><mo>∏</mo><mi>i</mi><msub><mi>p</mi><mi>M</mi></msub></msubsup><mrow><mi>p</mi><mfenced><mrow><mrow><mfenced><mrow><msub><mi>d</mi><mi>i</mi></msub><mo>,</mo><msub><mi>q</mi><mi>i</mi></msub></mrow></mfenced><mo>|</mo></mrow><mfenced><mrow><msub><mi>d</mi><mrow><mi>i</mi><mo>−</mo><mi>n</mi><mo>+</mo><mn>1</mn></mrow></msub><mo>,</mo><msub><mi>q</mi><mrow><mi>i</mi><mo>−</mo><mi>n</mi><mo>+</mo><mn>1</mn></mrow></msub></mrow></mfenced><mo>,</mo><mo>…</mo><mo>,</mo><mfenced><mrow><msub><mi>d</mi><mrow><mi>i</mi><mo>−</mo><mn>1</mn></mrow></msub><mo>,</mo><msub><mi>q</mi><mrow><mi>i</mi><mo>−</mo><mn>1</mn></mrow></msub></mrow></mfenced></mrow></mfenced></mrow></mstyle></mtd></mtr></mtable></math><img id="ib0003" file="imgb0003.tif" wi="108" he="10" img-content="math" img-format="tif"/></maths> where <b>M</b> denotes the monotonic condition. Note that the initial alignment <b>a</b> may be computed using the Levenshtein distance between <b>d</b> and <b>q.</b> The alignment and the n-gram model's parameters may be updated in the maximum likelihood sense. Re-alignment can be based on pairing frequencies, for example.</p>
<p id="p0028" num="0028">Given the trained n-gram model, a listing-to-query translation may be performed. Given a listing form <b>d,</b> and given query forms <b>q</b> (from a decoder, discussed later), the query forms are searched to find those that have the highest conditional probability: <maths id="math0004" num="(4)"><math display="block"><msup><mi>q</mi><mo>*</mo></msup><mo>≈</mo><msub><mi>max</mi><mi>q</mi></msub><msub><mi>max</mi><mi>a</mi></msub><msub><mi>p</mi><mi>M</mi></msub><mfenced><mrow><mi>d</mi><mo>,</mo><mi>q</mi><mo>,</mo><mi>a</mi></mrow></mfenced></math><img id="ib0004" file="imgb0004.tif" wi="108" he="5" img-content="math" img-format="tif"/></maths> where p(d, q, a) is evaluated using equation (3).</p>
<p id="p0029" num="0029">The translation not only exploits word-level semantic correspondence as modeled by unigrams, but it also takes into account word context by using higher order n-grams. The search for the best or n-best query forms can be achieved efficiently by applying the best-first search algorithm, which is described by <nplcit id="ncit0005" npl-type="b"><text>Russell and Norvig in Artificial Intelligence: A Modern Approach (Prentice Hall, second edition, 2003</text></nplcit>). Using this type of search, pruning techniques may be applied to reduce computational complexity. Returning to the language model (LM) for speech recognition, once the n-best query forms are obtained for the listings, they may be used as training sentences for LM estimation.</p>
<p id="p0030" num="0030">There are two implementation details to be considered. First, allowing the use of null words in <b>d</b> raises a potential problem at decode time -- the search space is significantly expanded because null can be present or absent at any position of the source<!-- EPO <DP n="10"> --> sentence. To avoid this problem, it is preferable to eliminate the use of (d<sub>i</sub> = null, q<sub>i</sub>) as semantic units for values of q<sub>i</sub>. Specifically, in training, (d<sub>i</sub> = null, q<sub>i</sub>) may be merged with its proceeding or following semantic unit, depending on which of the phrases, q<sub>i-1</sub>q<sub>i</sub> or q<sub>i</sub>q<sub>i+1</sub>, have more occurrences in the training data. Then, (d<sub>i+1</sub>, q<sub>i-1</sub>q<sub>i</sub>) or (d<sub>i+1</sub>, q<sub>i</sub>q<sub>i+1</sub>) may be treated as a single semantic unit. At decode time, null is not explicitly inserted in d, because using semantic units (d<sub>i-1</sub>, q<sub>i-1</sub>q<sub>i</sub>) or (d<sub>i+1</sub>, q<sub>i</sub>q<sub>i+1</sub>) is equivalent to adding null in the source sentence.</p>
<p id="p0031" num="0031">The second implementation detail concerns out-of-vocabulary (OOV) words in <b>d.</b> When OOV occurs, it might not be possible to produce any query forms, since p(d<sub>i</sub> = OOV, q<sub>i</sub>) = 0 for any value of q<sub>i</sub>. To deal with such cases, a positive probability may be assigned to unigrams (d<sub>i</sub> , q<sub>i</sub> = d<sub>i</sub>) whenever d<sub>i</sub> = OOV. This implies that a listing word, if never seen in the training data, will be translated to itself.</p>
<p id="p0032" num="0032">It should be noted that embodiments with non-monotonic alignment are also possible. Furthermore, a re-ordering strategy may be used. This may be implemented before a monotonic alignment is applied by reordering <b>d</b> while keeping the order of <b>q</b>. When training the translation model, the best way to reorder the words in the source form is determined by computing the resulting joint n-gram model likelihood. Only orders that are shifts of the original order are considered, and a maximum entropy classifier for these orders is built, where the input of the classifier is the source form, and the output is an order. Prior to translation, this classifier is applied to reorder a source form.</p>
<heading id="h0007"><b>APPLICATIONS OF INTRALANGUAGE STATISTICAL TRANSLATION MODEL</b></heading>
<p id="p0033" num="0033"><figref idref="f0003">Figure 3</figref> shows an arrangement for using a statistical translation model to improve a search system and/or the language model of a voice recognition system. A search engine 152 is configured to search listings 154, for example business listings. The search engine 152 receives text queries or transcribed spoken queries 156 that are generated by users and submitted to the search engine 152. Corresponding relevant listings 158 are retrieved by the search engine 152. Note that training pairs can also be obtained algorithmically using TF-IDF (term frequency-inverse document frequency).<!-- EPO <DP n="11"> --></p>
<p id="p0034" num="0034">The text or transcribed queries 156 and corresponding search-engine retrieved listings 158 are passed to a training component 160 that trains a statistical translation model 162, which may be n-gram based or another type of model. As discussed above, the training component 160 iterates through source-target pairs of the transcribed queries 156 and listings 158. In the case of an n-gram based model, given a (source, target) pair, an initial monotonic alignment is obtained between the source form and target form by computing an edit distance. Given the alignment, the training component 160 discovers word-level pairs and builds an n-gram translation model 162 based on the word-level pairs. The alignment and n-gram model parameters of the translation model 162 may be iteratively refined to improve the translation model 162. Furthermore, training may implement a backoff strategy which assumes that a word can be translated to itself, as is possible with intra-language translation. In other words, the aligned units WORD-WORD, where WORD can be a word or a phrase, will have a positive probability.</p>
<p id="p0035" num="0035">A translation module 164 uses the translation model 162 to test decoded candidates (potential translations). Given the trained translation model 162 and a source form, a best-first search algorithm may be used to obtain the top n-best target forms (the n decoded target forms with the highest probability according to the translation model 162). The weight of each target form is determined by p(target|source) produced by the translation model. Unlikely word-level pairs may be pruned to speed up translation.</p>
<p id="p0036" num="0036">Given the translation model 162 and the translation module 164, subsequent searches may be improved as follows. Given a user's query <b>q</b> and a listing <b>d</b> found by the search engine 152, translated query forms <b>x</b> of the listing <b>d</b> are considered when measuring the listing <b>d</b>'s relevancy to a user's query. Letting s(_,_) be a function or measure of relevancy (or similarity), the measurement of relevancy may be s(q, d) = sum_x {p(x|d) s(q, x)}. Alternatively, relevancy may be measured directly from the translation probability, in which case s(q, d) = p(q,d). In one embodiment, potential translations can be filtered out if their similarity measure is below a specified threshold.</p>
<p id="p0037" num="0037">Furthermore, not only may searching be improved as described above, but a language model 168 can also be built or augmented using intra-language translation.<!-- EPO <DP n="12"> --> Language models are used in many natural language processing applications such as ASR, machine translation, and parsing. The intra-language translation provided by the translation model 162 and translation module 164 may be used in language modeling by translating listings into query forms and using the same-language translated query forms when estimating a language model 168. When estimating the language model 168, the count of a translated query form may be set to its posterior probabilities multiplied by the count of its original listing.</p>
<p id="p0038" num="0038">In one embodiment, a server- or client-based voice recognizer may be provided with the language model 168, which will allow the voice recognizer to perform more accurate and comprehensive speech recognition with respect to utterances directed to the listings 154 or to listings. The translation model 162 may also be used at a server or at a mobile client to translate a string inputted at the mobile device (whether by ASR or otherwise) to a display form.</p>
<heading id="h0008"><b>CONCLUSION</b></heading>
<p id="p0039" num="0039">Embodiments and features discussed above can be realized in the form of information stored in volatile or non-volatile computer or device readable media. This is deemed to include at least media such as optical storage (e.g., CD-ROM), magnetic media, flash ROM, or any current or future means of storing digital information. The stored information can be in the form of machine executable instructions (e.g., compiled executable binary code), source code, bytecode, or any other information that can be used to enable or configure computing devices to perform the various embodiments discussed above. This is also deemed to include at least volatile memory such as RAM and/or virtual memory storing information such as CPU instructions during execution of a program carrying out an embodiment, as well as non-volatile media storing information that allows a program or executable to be loaded and executed. The embodiments and featured can be performed on any type of computing device, including portable devices, workstations, servers, mobile wireless devices, and so on. The modules, components, processes, and<!-- EPO <DP n="13"> --> search engine 152 discussed above may by realized on one computing device or multiple cooperating computing devices.</p>
</description>
<claims id="claims01" lang="en"><!-- EPO <DP n="14"> -->
<claim id="c-en-01-0001" num="0001">
<claim-text>A computer implemented method for intra-language machine translation of phrases in a human language, the method comprising:
<claim-text>receiving training data, the training data comprising pairings of source phrases and target phrases (120, 156, 158) in a same human language, wherein the pairings of source phrases and target phrases comprise query strings submitted by users paired with listings that a search engine matched with the query strings;</claim-text>
<claim-text>obtaining an initial monotonic alignment between a source form and a target form of a source phrase and a target phrase pair by computing an edit distance (122);</claim-text>
<claim-text>pairing words and/or phrases of the aligned sentences (124);</claim-text>
<claim-text>using the pairs of training data to train (100, 126) an n-gram intra-language statistical machine translation model (160, 162), where the n-gram intra-language statistical machine translation model, when given an input phrase of text in the human language, can compute probabilities of semantic equivalence of the input phrase to possible translations of the input phrase in the human language (102), wherein the training comprises rearranging at least one of the source and the target phrase of a training pair so that semantically equivalent words of the source and target phrases are aligned (122);</claim-text>
<claim-text>iteratively updating the alignment and parameters of the n-gram intra-language statistical machine translation model (160, 162); and</claim-text>
<claim-text>using the n-gram statistical machine translation model to translate between queries and listings (164), where the queries comprise text strings in the human language submitted to a search engine (156), where the listing strings comprise text strings of formal names of real world entities that are to be searched by the search engine to find matches for the query strings (158).</claim-text></claim-text></claim>
<claim id="c-en-01-0002" num="0002">
<claim-text>The method according to claim 1, wherein the using the n-gram intra-language statistical translation model comprises:<!-- EPO <DP n="15"> -->
<claim-text>receiving from the search engine listings that the search engine matched to the user's query (158);</claim-text>
<claim-text>generating query forms of one of the listings by using the n-gram translation model to translate the one of the listings to the query forms (164, 166);</claim-text>
<claim-text>using the n-gram translation model to compute similarities of the query forms to the user's query, and determining that the listing does not match the user's query based on the computed similarities.</claim-text></claim-text></claim>
<claim id="c-en-01-0003" num="0003">
<claim-text>The method according to claim 1, wherein the using the n-gram intra-language statistical translation model comprises:
<claim-text>receiving from the search engine a listing that the search engine matched to the user's query (158);</claim-text>
<claim-text>using the n-gram model (162) to find a probability that the listing is a translation of the user's query; and</claim-text>
<claim-text>determining whether the listing matches the user's query based on the probability.</claim-text></claim-text></claim>
<claim id="c-en-01-0004" num="0004">
<claim-text>The method according to claim 1, further comprising using the n-gram intra-language statistical translation model to generate a language model of the human language (168), wherein the using the n-gram intra-language statistical translation model to generate the language model comprises including with the language model translations from the n-gram intra-language statistical translation model, the language model being capable of determining a likelihood of strings in the human language, the method<br/>
further comprising performing automatic speech recognition with the language model.</claim-text></claim>
<claim id="c-en-01-0005" num="0005">
<claim-text>The method according to claim 1, wherein the listing forms comprise formal names, in the human language, of organizations and/or businesses searchable by the search engine.</claim-text></claim>
<claim id="c-en-01-0006" num="0006">
<claim-text>The method according to claim 5, wherein the using the n-gram statistical machine translation model comprises computing similarity between a query form and a listing form.</claim-text></claim>
<claim id="c-en-01-0007" num="0007">
<claim-text>The method according to claim 5, wherein, given a user query inputted by a user in the human language (156), given a corresponding listing in the human language that was found by the search engine (152), and given a set of candidate translations of the listing,<!-- EPO <DP n="16"> --> the candidate translations also in the human language, the using the n-gram statistical machine translation model comprises computing probabilities of the candidate translations, and/or<br/>
further comprising generating a search result for the given user query based on the computed probabilities.</claim-text></claim>
<claim id="c-en-01-0008" num="0008">
<claim-text>The method according to claim 5, further comprising generating or modifying a search result of the search engine based on probabilities computed by the n-gram statistical machine translation model, the search result corresponding to a user-inputted query form, and<br/>
further comprising using the probabilities to rank or eliminate the search result.</claim-text></claim>
<claim id="c-en-01-0009" num="0009">
<claim-text>One or more computer readable media storing information to enable a computing device to perform a process for translating phrases of a human language to other phrases of the language, the process comprising:
<claim-text>accessing training data, the training data comprising pairings of source phrases and target phrases in the human language (120, 156, 158), wherein the pairings of source phrases and target phrases comprise query strings submitted by users paired with listings that a search engine matched with the query strings;</claim-text>
<claim-text>given a source phrase and target phrase pair, obtaining an initial monotonic alignment between the source form and the target form by computing an edit distance;</claim-text>
<claim-text>pairing words and/or phrases of the aligned forms;</claim-text>
<claim-text>training an n-gram statistical machine translation model (160,162) with the training pairs (160), wherein the n-gram statistical machine translation model, when given an input phrase of text in the human language, is capable of computing probabilities of semantic equivalence of the input phrase to possible translations of the input phrase in the human language (102) and wherein the training comprises rearranging at least one of the source and the target phrase of a training pair so that semantically equivalent words of the source and target phrases are aligned (122);</claim-text>
<claim-text>iteratively updating the alignment and parameters of the n-gram intra-language statistical machine translation model;<!-- EPO <DP n="17"> --></claim-text>
<claim-text>receiving a text phrase in the human language corresponding to the query, decoding the text phrase to different candidate translations of the text phrase (102), and using the n-gram statistical machine translation model to compute probabilities that the candidate translations are translations of the query; and</claim-text>
<claim-text>based on the probabilities, storing and displaying, by computer, one or more of the candidate translations.</claim-text></claim-text></claim>
<claim id="c-en-01-0010" num="0010">
<claim-text>The one or more computer readable media according to claim 9, wherein the received text phrase comprises a query string inputted by a user, the query string comprising text in the human language, and the process further comprises using the n-gram statistical machine translation model to identify a plurality of probable potential translations of the query string, the potential translations comprising text in the human language.</claim-text></claim>
<claim id="c-en-01-0011" num="0011">
<claim-text>The one or more computer readable media according to claim 9, wherein the received text phrase comprises a name of an organization or business entity obtained from a search engine for searching listings of business/organization names, the name having been obtained from the search engine according to a user-inputted query, and wherein the process further comprises using the n-gram statistical machine translation model to determine a probability that the name is a valid translation of the query string and determining relevancy of the listing to the query based on the probability.</claim-text></claim>
<claim id="c-en-01-0012" num="0012">
<claim-text>The one or more computer readable media according to claim 9, further comprising using the n-gram statistical machine translation model to build a statistical language model of the human language, where the statistical language model provides probabilities of phrases in the human language.</claim-text></claim>
<claim id="c-en-01-0013" num="0013">
<claim-text>The one or more computer readable media according to claim 9, the process further comprising using the n-gram statistical machine translation model to translate text queries recognized by a speech recognizer into display forms.</claim-text></claim>
</claims>
<claims id="claims02" lang="de"><!-- EPO <DP n="18"> -->
<claim id="c-de-01-0001" num="0001">
<claim-text>Computerimplementiertes Verfahren für sprachinterne Maschinenübersetzung von Phrasen in einer menschlichen Sprache, wobei das Verfahren umfasst:
<claim-text>Empfangen von Trainingsdaten, wobei die Trainingsdaten Paare aus Quellphrasen und Zielphrasen (120, 156, 158) in derselben menschlichen Sprache umfassen, wobei die Paare aus Quellphrasen und Zielphrasen Abfragezeichenfolgen umfassen, die von Benutzern eingesendet werden, welche mit Listeneinträgen gepaart sind, die eine Suchmaschine mit den Abfragezeichenfolgen zusammengeführt hat;</claim-text>
<claim-text>Erhalten einer monotonen Anfangsausrichtung zwischen einer Quellform und einer Zielform eines Paares aus einer Quellphrase und einer Zielphrase durch Berechnen einer Bearbeitungsdistanz (122);</claim-text>
<claim-text>Bilden von Paaren aus Worten und/oder Phrasen der ausgerichteten Sätze (124);</claim-text>
<claim-text>Verwenden der Paare aus Trainingsdaten, um ein sprachinternes statistisches N-Gramm-Maschinenübersetzungsmodell (160, 162) zu trainieren, wobei das sprachinterne statistische N-Gramm-Maschinenübersetzungsmodell, wenn es eine Eingangsphrase aus Text in der menschlichen Sprache erhält, Wahrscheinlichkeiten semantischer Äquivalenz der Eingangsphrase mit möglichen Übersetzungen der Eingangsphrase in der menschlichen Sprache (102) berechnen kann, wobei das Training das Neuanordnen von mindestens einer aus der Quell- und der Zielphrase eines Trainingspaares umfasst, so dass semantisch äquivalente Worte der Quell- und Zielphrasen ausgerichtet werden (122);</claim-text>
<claim-text>interaktives Aktualisieren der Ausrichtung und der Parameter des sprachinternen statistischen N-Gramm-Maschinenübersetzungsmodells (160, 162); und</claim-text>
<claim-text>Verwenden des statistischen N-Gramm-Maschinenübersetzungsmodells für Übersetzungen zwischen Abfragen und Listeneinträgen (164), wobei die Abfragen Textzeichenfolgen in der menschlichen Sprache umfassen, die in eine Suchmaschine (156)<!-- EPO <DP n="19"> --> eingegeben werden, wobei die Listeneintragszeichenfolgen Textzeichenfolgen von formalen Namen realer Entitäten umfassen, die von der Suchmaschine gesucht werden sollen, um Übereinstimmungen für die Abfragezeichenfolgen (158) zu finden.</claim-text></claim-text></claim>
<claim id="c-de-01-0002" num="0002">
<claim-text>Verfahren nach Anspruch 1, wobei das Verwenden des sprachinternen statistischen N-Gramm-Übersetzungsmodells umfasst:
<claim-text>Empfangen, von der Suchmaschine, von Listeneinträgen, welche die Suchmaschine mit der Abfrage des Benutzers zusammengeführt hat (158);</claim-text>
<claim-text>Erzeugen von Abfrageformen von einem der Listeneinträge durch Verwenden des N-Gramm-Übersetzungsmodells zur Übersetzung des einen der Listeneinträge in die Abfrageformen (164, 166);</claim-text>
<claim-text>Verwenden des N-Gramm-Übersetzungsmodells zur Berechnung von Ähnlichkeiten der Abfrageformen mit der Abfrage des Benutzers und zum Bestimmen, dass der Listeneintrag nicht mit der Abfrage des Benutzers übereinstimmt, basierend auf den berechneten Ähnlichkeiten.</claim-text></claim-text></claim>
<claim id="c-de-01-0003" num="0003">
<claim-text>Verfahren nach Anspruch 1, wobei das Verwenden des sprachinternen statistischen N-Gramm-Übersetzungsmodells umfasst:
<claim-text>Empfangen, von der Suchmaschine, eines Listeneintrags, den die Suchmaschine mit der Abfrage des Benutzers zusammengeführt hat (158);</claim-text>
<claim-text>Verwenden das N-Gramm-Modells (162) zum Finden einer Wahrscheinlichkeit, dass der Listeneintrag eine Übersetzung der Abfrage des Benutzers ist; und</claim-text>
<claim-text>Bestimmen, ob der Listeneintrag mit der Abfrage des Benutzers übereinstimmt, basierend auf der Wahrscheinlichkeit.</claim-text></claim-text></claim>
<claim id="c-de-01-0004" num="0004">
<claim-text>Verfahren nach Anspruch 1, des Weiteren umfassend das Verwenden des sprachinternen statistischen N-Gramm-Übersetzungsmodells, um ein Sprachmodell der menschlichen Sprache (168) zu erzeugen, wobei das Verwenden des sprachinternen statistischen N-Gramm-Übersetzungsmodells zum Erzeugen des Sprachmodells das Einschließen, in das Sprachmodell, von Übersetzungen aus dem sprachinternen statistischen<!-- EPO <DP n="20"> --> N-Gramm-Übersetzungsmodell umfasst, wobei das Sprachmodell dazu in der Lage ist, eine Wahrscheinlichkeit von Zeichenfolgen in der menschlichen Sprache zu bestimmen, wobei das Verfahren des Weiteren umfasst<br/>
Durchführen von automatischer Spracherkennung mit dem Sprachmodell.</claim-text></claim>
<claim id="c-de-01-0005" num="0005">
<claim-text>Verfahren nach Anspruch 1, wobei die Listeneintragsformen formale Namen, in der menschlichen Sprache, von Organisationen und/oder Geschäften umfassen, nach welchen von der Suchmaschine gesucht werden kann.</claim-text></claim>
<claim id="c-de-01-0006" num="0006">
<claim-text>Verfahren nach Anspruch 5, wobei das Verwenden des statistischen N-Gramm-Maschinenübersetzungsmodells das Berechnen der Ähnlichkeit zwischen einer Abfrageform und einer Listeneintragsform umfasst.</claim-text></claim>
<claim id="c-de-01-0007" num="0007">
<claim-text>Verfahren nach Anspruch 5, wobei, angesichts einer Benutzerabfrage, die von einem Benutzer in der menschlichen Sprache (156) eingegeben wird, angesichts eines entsprechenden Listeneintrags in der menschlichen Sprache, der von der Suchmaschine (152) gefunden wurde, und angesichts eines Satzes aus potentiellen Übersetzungen des Listeneintrags, wobei die potentiellen Übersetzungen ebenfalls in der menschlichen Sprache vorhanden sind, das Verwenden des statistischen N-Gramm-Maschinenübersetzungsmodells das Berechnen von Wahrscheinlichkeiten der potentiellen Übersetzungen umfasst, und/oder<br/>
des Weiteren umfassend das Erzeugen eines Suchergebnisses für die vorgegebene Benutzerabfrage basierend auf den berechneten Wahrscheinlichkeiten.</claim-text></claim>
<claim id="c-de-01-0008" num="0008">
<claim-text>Verfahren nach Anspruch 5, des Weiteren umfassend das Erzeugen oder Modifizieren eines Suchergebnisses der Suchmaschine basierend auf Wahrscheinlichkeiten, die von dem statistischen N-Gramm-Maschinenübersetzungsmodell berechnet werden, wobei das Suchergebnis einer von einem Benutzer eingegebenen Abfrageform entspricht, und<br/>
des Weiteren umfassend das Verwenden der Wahrscheinlichkeiten, um die Suchergebnisse einzustufen oder zu eliminieren.</claim-text></claim>
<claim id="c-de-01-0009" num="0009">
<claim-text>Ein oder mehrere computerlesbare Medien zum Speichern von Informationen, um eine<!-- EPO <DP n="21"> --> Computervorrichtung dazu zu aktivieren, einen Prozess zum Übersetzen von Phrasen einer menschlichen Sprache in andere Phrasen der Sprache durchzuführen, wobei der Prozess umfasst:
<claim-text>Zugreifen auf Trainingsdaten, wobei die Trainingsdaten Paare aus Quellphrasen und Zielphrasen in der menschlichen Sprache (120, 156, 158) umfassen, wobei die Paare aus Quellphrasen und Zielphrasen Abfragezeichenfolgen umfassen, die von Benutzern eingesendet werden, welche mit Listeneinträgen gepaart sind, die eine Suchmaschine mit den Abfragezeichenfolgen zusammengeführt hat;</claim-text>
<claim-text>für ein vorgegebenes Paar aus einer Quellphrase und einer Zielphrase, Erhalten einer monotonen Anfangsausrichtung zwischen der Quellform und der Zielform durch Berechnen einer Bearbeitungsdistanz;</claim-text>
<claim-text>Bilden von Paaren aus Worten und/oder Phrasen der ausgerichteten Formen;</claim-text>
<claim-text>Trainieren eines sprachinternen statistischen N-Gramm-Maschinenübersetzungsmodells (160, 162) mit den Trainingspaaren (160), wobei das statistische N-Gramm-Maschinenübersetzungsmodell, wenn es eine Eingangsphrase aus Text in der menschlichen Sprache erhält, dazu in der Lage ist, Wahrscheinlichkeiten semantischer Äquivalenz der Eingangsphrase mit möglichen Übersetzungen der Eingangsphrase in der menschlichen Sprache (102) zu berechnen, und wobei das Training das Neuanordnen von mindestens einer aus der Quell- und der Zielphrase eines Trainingspaares umfasst, so dass semantisch äquivalente Worte der Quell- und Zielphrasen ausgerichtet werden (122);</claim-text>
<claim-text>interaktives Aktualisieren der Ausrichtung und der Parameter des sprachinternen statistischen N-Gramm-Maschinenübersetzungsmodells;</claim-text>
<claim-text>Empfangen einer Textphrase in der menschlichen Sprache entsprechend der Abfrage, Decodieren der Textphrase in unterschiedliche potentielle Übersetzungen der Textphrase (102), und Verwenden des statistischen N-Gramm-Maschinenübersetzungsmodells, um Wahrscheinlichkeiten zu berechnen, dass die potentiellen Übersetzungen Übersetzungen der Abfrage sind; und</claim-text>
<claim-text>basierend auf den Wahrscheinlichkeiten, Speichern und Anzeigen, durch Computer,<!-- EPO <DP n="22"> --> von einer oder mehreren der potentiellen Übersetzungen.</claim-text></claim-text></claim>
<claim id="c-de-01-0010" num="0010">
<claim-text>Ein oder mehrere computerlesbare Medien nach Anspruch 9, wobei die empfangene Textphrase eine Abfragezeichenfolge umfasst, die von einem Benutzer eingegeben wird, wobei die Abfragezeichenfolge Text in der menschlichen Sprache umfasst, und wobei der Prozess des Weiteren das Verwenden des statistischen N-Gramm-Maschinenübersetzungsmodells zum Identifizieren einer Vielzahl von wahrscheinlichen potentiellen Übersetzungen der Abfragezeichenfolge umfasst, wobei die potentiellen Übersetzungen Text in der menschlichen Sprache umfassen.</claim-text></claim>
<claim id="c-de-01-0011" num="0011">
<claim-text>Ein oder mehrere computerlesbare Medien nach Anspruch 9, wobei die empfangene Textphrase einen Namen einer Organisation oder einer Geschäftseinheit umfasst, die von einer Suchmaschine zum Durchsuchen von Listeneinträgen von Geschäfts-/Organisationsnamen erhalten wird, wobei der Name von der Suchmaschine entsprechend einer durch einen Benutzer eingegebenen Abfrage erhalten wurde, und wobei der Prozess des Weiteren das Verwenden des statistischen N-Gramm-Maschinenübersetzungsmodells zum Bestimmen einer Wahrscheinlichkeit umfasst, dass der Name eine gültige Übersetzung der Abfragezeichenfolge ist und zum Bestimmen der Relevanz des Listeneintrags für die Abfrage basierend auf der Wahrscheinlichkeit.</claim-text></claim>
<claim id="c-de-01-0012" num="0012">
<claim-text>Ein oder mehrere computerlesbare Medien nach Anspruch 9, des Weiteren umfassend das Verwenden des statistischen N-Gramm-Maschinenübersetzungsmodells zum Erstellen eines statistischen Sprachmodells der menschlichen Sprache, wobei das statistische Sprachmodell Wahrscheinlichkeiten von Phrasen in der menschlichen Sprache bietet.</claim-text></claim>
<claim id="c-de-01-0013" num="0013">
<claim-text>Ein oder mehrere computerlesbare Medien nach Anspruch 9, wobei der Prozess des Weiteren das Verwenden des statistischen N-Gramm-Maschinenübersetzungsmodells zum Übersetzen von Textabfragen, die von einer Spracherkennung erkannt werden, in Anzeigeformen umfasst.</claim-text></claim>
</claims>
<claims id="claims03" lang="fr"><!-- EPO <DP n="23"> -->
<claim id="c-fr-01-0001" num="0001">
<claim-text>Procédé mis en oeuvre sur un ordinateur pour une traduction par machine intra-langue de phrases dans une langue humaine, le procédé comprenant :<br/>
réception de données d'entraînement, les données d'entraînement comprenant des appariements de phrases source et phrases cible (120, 156, 158) dans une même langue humaine, dans lequel les appariements de phrases source et de phrases cible comprennent des chaînes de requête soumises par des utilisateurs associés à des listes qu'un moteur de recherche a mises en correspondance avec les chaînes de requête :
<claim-text>obtention d'un alignement monotone initial entre un formulaire source et un formulaire cible d'une paire de phrase source et de phrase cible en calculant une distance d'édition (122) ;</claim-text>
<claim-text>appariement de mots et/ou de phrases des phrases alignées (124) ;</claim-text>
<claim-text>utilisation des paires de données d'entraînement pour entraîner (100, 126) un modèle de traduction par machine statistique intra-langue à n-grammes (160, 162), où le modèle de traduction par machine statistique intra-langue à n-grammes, lorqu'est fournie une phrase<!-- EPO <DP n="24"> --> d'entrée de texte dans la langue humaine, peut calculer des probabilités d'équivalence sémantique de la phrase d'entrée en traductions possibles de la phrase d'entrée dans la langue humaine (102), dans lequel l'entraînement comprend un réagencement d'au moins une de la phrase source et de la phrase cible d'une paire d'entraînement de sorte que des mots équivalents sémantiquement des phrases source et cible sont alignés (122) ;</claim-text>
<claim-text>mise à jour itérative de l'alignement et des paramètres du modèle de traduction par machine statistique intra-langue à n-grammes (160, 162) ; et</claim-text>
<claim-text>utilisation du modèle de traduction par machine statistique intra-langue à n-grammes pour traduire entre des requêtes et des listes (164), où les requêtes comprennent des chaînes de texte dans la langue humaine soumise à un moteur de recherche (156), où les chaînes de liste comprennent des chaînes de texte de noms formels d'entités réelles qui sont à rechercher par le moteur de recherche pour trouver des correspondances pour les chaînes de requêtes (158).</claim-text></claim-text></claim>
<claim id="c-fr-01-0002" num="0002">
<claim-text>Le procédé de la revendication 1, dans lequel l'utilisation du modèle de traduction par machine statistique intra-langue à n-grammes comprend :
<claim-text>réception du moteur de recherche de listes que le moteur de recherche a mises en correspondance avec la requête de l'utilisateur (158) ;<!-- EPO <DP n="25"> --></claim-text>
<claim-text>génération de formulaires de requête d'une des listes en utilisant le modèle de traduction à n-grammes pour traduire l'une des listes en les formulaires de requête (164, 166) ;</claim-text>
<claim-text>utilisation du modèle de traduction à n-grammes pour calculer des similarités des formulaires de requête avec la requête de l'utilisateur, et détermination que la liste ne correspond pas à la requête de l'utilisateur en fonction des similarités calculées.</claim-text></claim-text></claim>
<claim id="c-fr-01-0003" num="0003">
<claim-text>Le procédé de la revendication 1, dans lequel l'utilisation du modèle de traduction statistique intra-langue à n-grammes comprend :
<claim-text>réception du moteur de recherche d'une liste que le moteur de recherche a mise en correspondance avec la requête de l'utilisateur (158) ;</claim-text>
<claim-text>utilisation du modèle à n-grammes (162) pour rechercher une probabilité que la liste est une traduction de la requête de l'utilisateur ; et</claim-text>
<claim-text>détermination que la liste correspond ou non à la requête de l'utilisation selon la probabilité.</claim-text></claim-text></claim>
<claim id="c-fr-01-0004" num="0004">
<claim-text>Le procédé selon la revendication 1 comprenant en outre une utilisation du modèle de traduction statistique intra-langue à n-grammes pour générer un modèle de langue de la langue humaine (168), dans lequel l'utilisation du modèle de traduction statistique intra-langue à n-grammes pour générer le modèle de langue<!-- EPO <DP n="26"> --> comprend une inclusion avec les traductions de modèle de langue issues du modèle de traduction statistique intra-langue à n-grammes, le modèle de langue étant capable de déterminer une probabilité de chaînes dans la langue humaine,<br/>
le procédé comprenant en outre l'exécution d'une reconnaissance vocale automatique avec le modèle de langue.</claim-text></claim>
<claim id="c-fr-01-0005" num="0005">
<claim-text>Le procédé selon la revendication 1, dans lequel les formulaires de liste comprennent des noms formels, dans la langue humaine, d'organismes et/ou d'entreprises que le moteur de recherche peut rechercher.</claim-text></claim>
<claim id="c-fr-01-0006" num="0006">
<claim-text>Le procédé selon la revendication 5, dans lequel l'utilisation du modèle de traduction par machine statistique intra-langue à n-grammes comprend un calcul de similarité entre un formulaire de requête et un formulaire de liste.</claim-text></claim>
<claim id="c-fr-01-0007" num="0007">
<claim-text>Le procédé conformément à la revendication 5, dans lequel, étant donné une requête d'utilisateur saisie par un utilisateur dans la langue humaine (156), étant donné une liste correspondante dans la langue humaine qui a été trouvée par le moteur de recherche (152), et étant donné un ensemble de traductions candidates de la liste, les traductions candidates également dans la langue humaine, l'utilisation du modèle de traduction par machine statistique intra-langue à n-grammes comprend un calcul de probabilités des traductions candidates, et/ou<!-- EPO <DP n="27"> --> comprenant en outre une génération d'un résultat de recherche pour la requête d'utilisateur donnée selon les probabilités calculées.</claim-text></claim>
<claim id="c-fr-01-0008" num="0008">
<claim-text>Le procédé selon la revendication 5, comprenant en outre une génération ou une modification d'un résultat de recherche du moteur de recherche selon des probabilités calculées par le modèle de traduction par machine statistique à n-grammes, le résultat de recherche correspondant à un formulaire de requête renseigné par un utilisateur, et<br/>
comprenant en outre une utilisation des probabilités pour classer ou éliminer le résultat de recherche.</claim-text></claim>
<claim id="c-fr-01-0009" num="0009">
<claim-text>Un ou plusieurs supports lisibles par un ordinateur stockant des informations pour permettre à un dispositif informatique d'exécuter un processus pour une traduction de phrases d'une langue humaine en d'autres phrases de la langue, le processus comprenant :<br/>
accès à des données d'entraînement, les données d'entraînement comprenant des appariements de phrases source et phrases cible dans la langue humaine (120, 156, 158), dans lequel les appariements de phrases source et de phrases cible comprennent des chaînes de requête soumises par des utilisateurs associés à des listes qu'un moteur de recherche a associées aux chaîne de requête :
<claim-text>étant donnée une paire de phrase source et de phrase cible, obtention d'un alignement monotone initial entre<!-- EPO <DP n="28"> --> le formulaire source et le formulaire cible en calculant une distance d'édition ;</claim-text>
<claim-text>appariement de mots et/ou de phrases des formulaires alignés ;</claim-text>
<claim-text>entraînement d'un modèle de traduction par machine statistique à n-grammes (160, 162) avec des paires d'entraînement (160), dans lequel le modèle de traduction par machine statistique à n-grammes, lorqu'est fournie une phrase d'entrée de texte dans la langue humaine, est capable de calculer des probabilités d'équivalence sémantique de la phrase d'entrée en traductions possibles de la phrase d'entrée dans la langue humaine (102), dans lequel l'entraînement comprend un réagencement d'au moins une de la phrase source et de la phrase cible d'une paire d'entraînement de sorte que des mots équivalents sémantiquement des phrases source et cible sont alignés (122) ;</claim-text>
<claim-text>mise à jour itérative de l'alignement et des paramètres du modèle de traduction par machine statistique intra-langue à n-grammes ;</claim-text>
<claim-text>réception d'une phrase de texte dans la langue humaine correspondant à la requête, décodage de la phrase de texte en différentes traductions candidates de la phrase de texte (102), et utilisation du modèle de traduction par machine statistique à n-grammes pour calculer des probabilités que les traductions candidates sont des traductions de la requête ; et<!-- EPO <DP n="29"> --></claim-text>
<claim-text>selon les probabilités, stockage et affichage, par un ordinateur, d'une ou plusieurs des traductions candidates.</claim-text></claim-text></claim>
<claim id="c-fr-01-0010" num="0010">
<claim-text>L'un ou plusieurs supports lisibles par un ordinateur selon la revendication 9, dans lesquels la phrase de texte reçue comprend une chaîne de requête saisie par un utilisateur, la chaîne de requête comprenant du texte dans la langue humaine, et le processus comprend en outre une utilisation du modèle de traduction par machine statistique à n-grammes pour identifier une pluralité de traductions potentielles probables de la chaîne de requête, les traductions potentielles comprenant du texte dans la langue humaine.</claim-text></claim>
<claim id="c-fr-01-0011" num="0011">
<claim-text>L'un ou plusieurs supports lisibles par un ordinateur selon la revendication 9, dans lesquels la phrase de texte reçue comprend un nom d'une entité d'organisme ou d'entreprise obtenu à partir d'un moteur de recherche pour une recherche dans des listes de noms d'organismes/d'entreprises, le nom ayant été obtenu à partir du moteur de recherche selon une requête saisie par un utilisateur, et dans lesquels le processus comprend en outre une utilisation du modèle de traduction par machine statistique à n-grammes pour déterminer une probabilité que le nom est une traduction valide de la chaîne de requête et une détermination d'une pertinence de la liste par rapport à la requête selon la probabilité.</claim-text></claim>
<claim id="c-fr-01-0012" num="0012">
<claim-text>L'un ou plusieurs supports lisibles par un ordinateur selon la revendication 9, comprenant en outre<!-- EPO <DP n="30"> --> une utilisation du modèle de traduction par machine statistique à n-grammes pour construire un modèle de langue statistique de la langue humaine, où le modèle de langue statistique fournit des probabilités de phrases dans la langue humaine.</claim-text></claim>
<claim id="c-fr-01-0013" num="0013">
<claim-text>L'un ou plusieurs supports lisibles par un ordinateur selon la revendication 9, le processus comprenant en outre une utilisation du modèle de traduction par machine statistique à n-grammes pour traduire des requêtes textuelles reconnues par un identificateur vocal en des formulaires d'affichage.</claim-text></claim>
</claims>
<drawings id="draw" lang="en"><!-- EPO <DP n="31"> -->
<figure id="f0001" num="1"><img id="if0001" file="imgf0001.tif" wi="77" he="108" img-content="drawing" img-format="tif"/></figure><!-- EPO <DP n="32"> -->
<figure id="f0002" num="2"><img id="if0002" file="imgf0002.tif" wi="77" he="211" img-content="drawing" img-format="tif"/></figure><!-- EPO <DP n="33"> -->
<figure id="f0003" num="3"><img id="if0003" file="imgf0003.tif" wi="156" he="155" img-content="drawing" img-format="tif"/></figure>
</drawings>
<ep-reference-list id="ref-list">
<heading id="ref-h0001"><b>REFERENCES CITED IN THE DESCRIPTION</b></heading>
<p id="ref-p0001" num=""><i>This list of references cited by the applicant is for the reader's convenience only. It does not form part of the European patent document. Even though great care has been taken in compiling the references, errors or omissions cannot be excluded and the EPO disclaims all liability in this regard.</i></p>
<heading id="ref-h0002"><b>Non-patent literature cited in the description</b></heading>
<p id="ref-p0002" num="">
<ul id="ref-ul0001" list-style="bullet">
<li><nplcit id="ref-ncit0001" npl-type="s"><article><author><name>STEFAN RIEZLER et al.</name></author><atl>Statistical Machine Translation for Query Expansion in Answer Retrieval</atl><serial><sertitle>ALC 2007 Proceedings</sertitle><pubdate><sdate>20070623</sdate><edate/></pubdate></serial></article></nplcit><crossref idref="ncit0001">[0004]</crossref></li>
<li><nplcit id="ref-ncit0002" npl-type="b"><article><atl>Information Retrieval as Statistical Translation</atl><book><author><name>ADAM BERGER et al.</name></author><book-title>PROC. SIGIR 1999 (Annual International ACM-SIGIR conference on research and development in information retrieval)</book-title><imprint><name>University of California</name><pubdate>19990815</pubdate></imprint><location><pp><ppf>222</ppf><ppl>229</ppl></pp></location></book></article></nplcit><crossref idref="ncit0002">[0005]</crossref></li>
<li><nplcit id="ref-ncit0003" npl-type="s"><article><author><name>C. QUIRK et al.</name></author><atl>Monolingual Machine Translation for Paraphrase Generation</atl><serial><sertitle>Proceedings of the Conference on Empirical Methods in Natural Language Processing</sertitle><pubdate><sdate>20040725</sdate><edate/></pubdate></serial><location><pp><ppf>142</ppf><ppl>149</ppl></pp></location></article></nplcit><crossref idref="ncit0003">[0006]</crossref></li>
<li><nplcit id="ref-ncit0004" npl-type="s"><article><author><name>JOSEP M. CREGO</name></author><author><name>ADRIA DE GISPERT</name></author><author><name>JOSE B. MARINO</name></author><atl/><serial><sertitle>The TALP NGRAM-Based SMT System for IWSLT'05</sertitle></serial></article></nplcit><crossref idref="ncit0004">[0007]</crossref></li>
<li><nplcit id="ref-ncit0005" npl-type="b"><article><atl/><book><author><name>RUSSELL</name></author><author><name>NORVIG</name></author><book-title>Artificial Intelligence: A Modern Approach</book-title><imprint><name>Prentice Hall</name><pubdate>20030000</pubdate></imprint></book></article></nplcit><crossref idref="ncit0005">[0029]</crossref></li>
</ul></p>
</ep-reference-list>
</ep-patent-document>
