<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE ep-patent-document PUBLIC "-//EPO//EP PATENT DOCUMENT 1.5.1//EN" "ep-patent-document-v1-5-1.dtd">
<!-- This XML data has been generated under the supervision of the European Patent Office -->
<ep-patent-document id="EP16784452B1" file="EP16784452NWB1.xml" lang="en" country="EP" doc-number="3335216" kind="B1" date-publ="20220126" status="n" dtd-version="ep-patent-document-v1-5-1">
<SDOBI lang="en"><B000><eptags><B001EP>ATBECHDEDKESFRGBGRITLILUNLSEMCPTIESILTLVFIROMKCYALTRBGCZEEHUPLSK..HRIS..MTNORS..SM..................</B001EP><B003EP>*</B003EP><B005EP>J</B005EP><B007EP>BDM Ver 2.0.14 (4th of August) -  2100000/0</B007EP></eptags></B000><B100><B110>3335216</B110><B120><B121>EUROPEAN PATENT SPECIFICATION</B121></B120><B130>B1</B130><B140><date>20220126</date></B140><B190>EP</B190></B100><B200><B210>16784452.1</B210><B220><date>20161014</date></B220><B240><B241><date>20180312</date></B241><B242><date>20190722</date></B242></B240><B250>en</B250><B251EP>en</B251EP><B260>en</B260></B200><B300><B310>PCT/EP2015/189865</B310><B320><date>20151015</date></B320><B330><ctry>WO</ctry></B330></B300><B400><B405><date>20220126</date><bnum>202204</bnum></B405><B430><date>20180620</date><bnum>201825</bnum></B430><B450><date>20220126</date><bnum>202204</bnum></B450><B452EP><date>20210930</date></B452EP></B400><B500><B510EP><classification-ipcr sequence="1"><text>G10L  19/02        20130101AFI20170503BHEP        </text></classification-ipcr><classification-ipcr sequence="2"><text>G10L  19/008       20130101ALI20170503BHEP        </text></classification-ipcr></B510EP><B520EP><classifications-cpc><classification-cpc sequence="1"><text>G10L  19/008       20130101 FI20161219BHEP        </text></classification-cpc><classification-cpc sequence="2"><text>G10L  19/02        20130101 LI20161219BHEP        </text></classification-cpc></classifications-cpc></B520EP><B540><B541>de</B541><B542>VERFAHREN UND VORRICHTUNG ZUR SINUSCODIERUNG UND -DECODIERUNG</B542><B541>en</B541><B542>METHOD AND APPARATUS FOR SINUSOIDAL ENCODING AND DECODING</B542><B541>fr</B541><B542>PROCÉDÉ ET APPAREIL DE CODAGE ET DE DÉCODAGE SINUSOÏDAL</B542></B540><B560><B561><text>AU-A1- 2011 205 144</text></B561><B561><text>US-A1- 2003 083 886</text></B561><B561><text>US-A1- 2005 078 832</text></B561><B561><text>US-A1- 2005 174 269</text></B561><B561><text>US-A1- 2007 238 415</text></B561><B562><text>TOMASZ ZERNICKI ET AL: "Updated MPEG-H 3D Audio Phase 2 Core Experiment Proposal on tonal component coding", 112. MPEG MEETING; 22-6-2015 - 26-6-2015; WARSAW; (MOTION PICTURE EXPERT GROUP OR ISO/IEC JTC1/SC29/WG11),, no. m36538, 18 June 2015 (2015-06-18), XP030064906,</text></B562><B562><text>ZERNICKI TOMASZ ET AL: "Application of Sinusoidal Coding for Enhanced Bandwidth Extension in MPEG-H USAC", AES CONVENTION 138; MAY, AES, 60 EAST 42ND STREET, ROOM 2520 NEW YORK 10165-2520, USA, 6 May 2015 (2015-05-06), pages 300-309, XP040670868,</text></B562><B562><text>Sascha Disch ET AL: "CHEAP BEEPS -EFFICIENT SYNTHESIS OF SINUSOIDS AND SWEEPS IN THE MDCT DOMAIN", Proceedings of ICASSP 2013, 26 May 2013 (2013-05-26), XP055091657, Retrieved from the Internet: URL:http://ieeexplore.ieee.org/ielx7/66195 49/6637585/06637701.pdf?tp=&amp;arnumber=66377 01&amp;isnumber=6637585 [retrieved on 2013-12-04]</text></B562><B562><text>PURNHAGEN H: "Advances in parametric audio coding", APPLICATIONS OF SIGNAL PROCESSING TO AUDIO AND ACOUSTICS, 1999 IEEE WO RKSHOP ON NEW PALTZ, NY, USA 17-20 OCT. 1999, PISCATAWAY, NJ, USA,IEEE, US, 17 October 1999 (1999-10-17), pages 31-34, XP010365061, DOI: 10.1109/ASPAA.1999.810842 ISBN: 978-0-7803-5612-2</text></B562></B560></B500><B700><B720><B721><snm>ZERNICKI, Tomasz</snm><adr><str>c/o Zylia Sp. z o.o.
Umultowska 85</str><city>61-614 Poznan</city><ctry>PL</ctry></adr></B721><B721><snm>JANUSZKIEWICZ, Lukasz</snm><adr><str>c/o Zylia Sp. z o.o.
Umultowska 85</str><city>61-614 Poznan</city><ctry>PL</ctry></adr></B721><B721><snm>SETIAWAN, Panji</snm><adr><str>c/o Huawei Technologies Duesseldorf GmbH
Riesstr. 25</str><city>80992 Munich</city><ctry>DE</ctry></adr></B721></B720><B730><B731><snm>Huawei Technologies Co., Ltd.</snm><iid>100970540</iid><irf>SAH11799EP</irf><adr><str>Huawei Administration Building 
Bantian</str><city>Longgang District
Shenzhen, Guangdong 518129</city><ctry>CN</ctry></adr></B731><B731><snm>Zylia SP. Z O.O.</snm><iid>101664536</iid><irf>SAH11799EP</irf><adr><str>Umultowska 85</str><city>61-614 Poznan</city><ctry>PL</ctry></adr></B731></B730><B740><B741><snm>Gill Jennings &amp; Every LLP</snm><iid>101574570</iid><adr><str>The Broadgate Tower 
20 Primrose Street</str><city>London EC2A 2ES</city><ctry>GB</ctry></adr></B741></B740></B700><B800><B840><ctry>AL</ctry><ctry>AT</ctry><ctry>BE</ctry><ctry>BG</ctry><ctry>CH</ctry><ctry>CY</ctry><ctry>CZ</ctry><ctry>DE</ctry><ctry>DK</ctry><ctry>EE</ctry><ctry>ES</ctry><ctry>FI</ctry><ctry>FR</ctry><ctry>GB</ctry><ctry>GR</ctry><ctry>HR</ctry><ctry>HU</ctry><ctry>IE</ctry><ctry>IS</ctry><ctry>IT</ctry><ctry>LI</ctry><ctry>LT</ctry><ctry>LU</ctry><ctry>LV</ctry><ctry>MC</ctry><ctry>MK</ctry><ctry>MT</ctry><ctry>NL</ctry><ctry>NO</ctry><ctry>PL</ctry><ctry>PT</ctry><ctry>RO</ctry><ctry>RS</ctry><ctry>SE</ctry><ctry>SI</ctry><ctry>SK</ctry><ctry>SM</ctry><ctry>TR</ctry></B840><B860><B861><dnum><anum>EP2016074742</anum></dnum><date>20161014</date></B861><B862>en</B862></B860><B870><B871><dnum><pnum>WO2017064264</pnum></dnum><date>20170420</date><bnum>201716</bnum></B871></B870></B800></SDOBI>
<description id="desc" lang="en"><!-- EPO <DP n="1"> -->
<p id="p0001" num="0001">This application relates to the field of audio coding, and in particular to the field of sinusoidal coding of audio signals.</p>
<heading id="h0001">BACKGROUND</heading>
<p id="p0002" num="0002">For the MPEG-H 3D Audio Core Coder a High Frequency Sinusoidal Coding (HFSC) enhancement has been proposed. The respective HFSC tool was already presented in 111th MPEG meeting in Geneva [1] and in 112th meeting in Warsaw [2].</p>
<p id="p0003" num="0003"><nplcit id="ncit0001" npl-type="s"><text>TOMASZ ZERNICKI et al.: "Updated MPEG-H 3D Audio Phase 2 Core Experiment Proposal on tonal component coding</text></nplcit>", describes high frequency sinusoidal coding. Segments are represented by a limited set of quantized DCT coefficients, including the obligatory DC coefficient and a variable number of AC coefficients, whose indices and values are transmitted using 8th order Golomb codes. Huffman codes are described as resulting in about 10% better coding efficiency, but requiring more memory for Huffman tables storage. <patcit id="pcit0001" dnum="AU2011205144A1"><text>AU 2011205144 A1</text></patcit> describes a scalable compressed audio bit stream and codec using a hierarchical filterbank and multichannel joint coding, including a primary channel and a secondary channel for each tonal component.</p>
<p id="p0004" num="0004"><patcit id="pcit0002" dnum="US2005078832A1"><text>US 2005/078832 A1</text></patcit> describes parametric audio coding, where common components in various signal channels can be represented by a single, common frequency and the respective amplitudes and phases of the respective components in the respective channels may differ. PURNHAGEN H: "Advances in parametric audio coding" describes extended sinusoidal models for parametric audio coding, and teaches taking source and perception models utilized in a parametric coder into account as a "joint model".</p>
<p id="p0005" num="0005"><patcit id="pcit0003" dnum="US2007238415A1"><text>US 2007/238415 A1</text></patcit> describes bandwidth extension to allow information to be encoded and decoded using a fractal self similarity model or an accurate spectral replacement model, where these tonal components are analyzed to determine if these fit into a harmonic structure.</p>
<heading id="h0002">SUMMARY</heading>
<p id="p0006" num="0006">It is an object of the present invention to provide improvements for, for example, the MPEG-H 3D Audio Codec, and in particular for the respective HFSC tool. However, embodiments of the present invention may also be used in and for other audio codecs using sinusoidal<!-- EPO <DP n="2"> --> coding. The term "codec" refers to or defines the functionalities of the audio encoder/encoding and audio decoder/decoding to implement the respective audio codec. More specifically, the present invention provides encoders, encoding methods, decoders and decoding methods as set out in the attached claims.</p>
<p id="p0007" num="0007">Embodiments of the invention can be implemented in Hardware or in Software or in any combination thereof.</p>
<heading id="h0003">SHORT DESCRIPTION OF THE FIGURES</heading>
<p id="p0008" num="0008">
<ul id="ul0001" list-style="none" compact="compact">
<li><figref idref="f0001">Figure 1</figref> shows an embodiment of the invention, in particular the general location of the proposed tool within the MPEG-H 3D Audio Core Encoder.</li>
<li><figref idref="f0002">Figure 2</figref> shows partitioning of sinusoidal trajectories into segments and their relation to GOS according to an embodiment of the invention.</li>
<li><figref idref="f0003">Figure 3</figref> shows a scheme of linking trajectory segments according to an embodiment of the invention.</li>
<li><figref idref="f0004">Figure 4a</figref> shows an illustration of independent encoding for each channel according to an example useful for understanding the invention which was originally filed but which does not represent an embodiment of the presently claimed invention.</li>
<li><figref idref="f0005">Figure 4b</figref> shows illustration of sending additional information related to trajectory panning according to an embodiment of the invention.</li>
<li><figref idref="f0006">Fig. 5</figref> shows the motivation for embodiments of the present invention.<!-- EPO <DP n="3"> --></li>
<li><figref idref="f0007">Fig. 6</figref> shows exemplary MPEG-H 3D Audio artifacts above fSBR.</li>
<li><figref idref="f0008">Fig. 7</figref> shows a comparison for 20kbps (∼2kbps of HESC), fSBR=4kHz, between "Original", "MPEG 3DA" and "MPEG 3DA+ HESC".</li>
<li><figref idref="f0009">Figure 8</figref> shows a flow-chart of an exemplary decoding method.</li>
<li><figref idref="f0010">Figure 9</figref> shows a block-diagram of an exemplary decoder.</li>
<li><figref idref="f0011">Fig. 10</figref> shows an example analysis of sinusoidal trajectories showing sparse DCT spectra according to prior art.</li>
<li><figref idref="f0012">Fig. 11</figref> shows a flow-chart of an exemplary decoding method.</li>
<li><figref idref="f0013">Fig. 12</figref> shows a block diagram of a corresponding exemplary decoder.</li>
<li><figref idref="f0014">Figure 13a</figref>) shows another embodiment of the invention, in particular the general location of the proposed tool within the MPEG-H 3D Audio Core Encoder .</li>
<li><figref idref="f0014">Figure 13b</figref>) shows a part of <figref idref="f0012">Fig. 11</figref>.</li>
<li><figref idref="f0014">Figure 13c</figref>) shows an embodiment of the present invention, wherein the steps depicted therein replace the respective steps in <figref idref="f0014">Fig. 13b</figref>).</li>
<li><figref idref="f0015">Figure 14a</figref>) shows an example for multichannel coding, useful for understanding the invention, which was originally filed but which does not represent an embodiment of the presently claimed invention.</li>
<li><figref idref="f0015">Figure 14b</figref>) shows an alternative embodiment of the invention for multichannel coding. Identical reference signs refer to identical or at least functionally equivalent features.</li>
</ul></p>
<heading id="h0004">DETAILED DESCRIPTION</heading>
<p id="p0009" num="0009">In the following certain embodiments are described in relation to an MPEG-H 3D Audio Phase 2 Core Experiment Proposal on tonal component coding.</p>
<heading id="h0005">1. Executive Summary</heading>
<p id="p0010" num="0010">This document provides a full technical description of the High Frequency Sinusoidal Coding (HFSC) for MPEG-H 3D Audio Core Coder. The HFSC tool was already presented in 111th MPEG meeting in Geneva [1] and in 112th meeting in Warsaw [2]. This document supplements the previous descriptions and clarifies all the issues concerning the target bit rate range of the tool, decoding process, sinusoidal synthesis, bit stream syntax and computational complexity and memory requirements of the decoder.</p>
<p id="p0011" num="0011">The proposed scheme consists of parametric coding of selected high frequency tonal components using an approach based on sinusoidal modeling. The HFSC tool acts as a preprocessor to MPS in Core Encoder (<figref idref="f0001">Figure 1</figref>). It generates an additional bit stream in the<!-- EPO <DP n="4"> --> range of 0 kbps to 1 kbps only in cases of signals exhibiting a strong tonal character in the high frequency range. The HFSC technique was tested as an extension to USAC Reference Quality Encoder. Verification tests were conducted to assess the subjective quality of proposed extension [3].</p>
<heading id="h0006">2. Technical Description of proposed tool</heading>
<heading id="h0007">2.1. Functions</heading>
<p id="p0012" num="0012">The purpose of the HFSC tool is to improve the representation of prominent tonal components in the operating range of the eSBR tool. In general, eSBR reconstructs high frequency components by employing the patching algorithm. Thus, its efficiency strongly depends on the availability of corresponding tonal components in the lower part of the spectrum. In certain situations, described below, the patching algorithm will not be able to reconstruct some important tonal components.
<ul id="ul0002" list-style="bullet" compact="compact">
<li>If the signal has a prominent components with fundamental frequency near or above the f_SBR_start frequency. This includes highly pitched sounds, like orchestral bells, and other percussive instruments. In this case, no shifting or scaling is able to recreate such components in the SBR range. The eSBR tool may use an additional technique called "sinusoidal coding" to inject a fixed sinusoidal component into a certain subband of the QMF filterbank. This component has a low frequency resolution and causes a significant discrepancy of timbre due to added inharmonicity.</li>
<li>If the signal has a significantly varying frequency (e.g. vibrato modulation), its energy in the lower band is spread over a range of transform coefficients which are subsequently distorted by quantization. For very low bit rates the local SNR becomes very low, and a partial that was originally purely tonal may not be considered as tonal any more. In such case, different patching variants lead to different additional artifacts:
<ul id="ul0003" list-style="none" compact="compact">
<li>∘ With harmonic patching mode based on phase vocoder, the quantization noise is further spread in frequency, and affects also the cross-terms</li>
<li>o With non-harmonic mode (spectral shifting), the frequency modulations are not properly scaled (modulation depth does not increase with partial order).</li>
</ul></li>
</ul><!-- EPO <DP n="5"> --></p>
<p id="p0013" num="0013">In our proposal, the HFSC tool is used occasionally, when sounds rich with prominent high frequency tonal partials are encountered. In such situations, prominent tonal components in the range from 3360Hz to 24000 Hz are detected, their potential distortion by the eSBR tool is analyzed, and the sinusoidal representation of selected components is encoded by the HFSC tool. The additional HFSC data represents a sum of sinusoidal partials with continuously varying frequencies and amplitudes. These partials are encoded in the form of sinusoidal trajectories, i.e. data vectors representing varying amplitude and frequency [4].</p>
<p id="p0014" num="0014">HFSC tool is active only when the strong tonal components are detected by dedicated classification tools. It additionally uses Signal Classifier embedded in Core Coder. There might be also an optional pre-processing done at the input of the MPS (MPEG Surround) block in core encoder, in order to minimize the further processing of selected components by the eSBR tool (<figref idref="f0001">Figure 1</figref>).</p>
<p id="p0015" num="0015"><figref idref="f0001">Figure 1</figref> shows the general location of the proposed tool within the MPEG-H 3D Audio Core Encoder.</p>
<heading id="h0008">2.2. HFSC Decoding Process</heading>
<heading id="h0009">2.2.1. Segmentation of sinusoidal trajectories</heading>
<p id="p0016" num="0016">Each individually encoded sinusoidal component is uniquely represented by its parameters: frequency and amplitude, one pair of values per component per each output data frame containing H = 256 samples. The parameters describing one tonal component are linked into so called sinusoidal trajectories. The original sinusoidal trajectories build in the encoder may have an arbitrary length. For the purpose of coding, these trajectories are partitioned into segments. Finally, segments of different trajectories starting within particular time are grouped into Groups of Segments (GOS). In our proposal GOS_LENGTH was limited to 8 trajectory data frames, which results in reduced coding delay and higher bit stream granularity.</p>
<p id="p0017" num="0017">Data values within each segment are encoded jointly. All segments of a trajectory can have lengths in the range from HFSC_MIN_SEG_LENGTH=GOS_LENGTH to HFSC_MAX_SEG_LENGTH = 32 and they are always multiple of 8, so the possible segment length values are: 8, 16, 24, and 32. During encoding process the segments length is adjusted by extrapolation process. Thanks to this the partitioning of trajectory into segments<!-- EPO <DP n="6"> --> is synchronized with the endpoints of GOS structure, i.e. each segment always starts and ends at the endpoints of GOS structure.</p>
<p id="p0018" num="0018">Upon decoding, this segment may continue to the next GOS (or even further), as shown in <figref idref="f0002">Figure 2</figref>. After decoding, the segmented trajectories are joined together in the trajectory buffer, as described in section 2.2.2. Decoding process of GOS structure is detailed in Annex A.</p>
<p id="p0019" num="0019"><figref idref="f0002">Figure 2</figref> shows partitioning of sinusoidal trajectories into segments and their relation to GOS according to an embodiment of the invention.</p>
<p id="p0020" num="0020">Encoding algorithm has also an ability to jointly encode clusters of segments belonging to harmonic structure of the sound source, i.e. clusters represent fundamental frequency of each harmonic structure and its integer multiplications. It can exploit the fact that each segment is characterized with a very similar FM and AM modulations.</p>
<heading id="h0010">2.2.2. Ordering and linking of corresponding trajectory segments</heading>
<p id="p0021" num="0021">Each decoded segment contains information about its length and if there will be any further corresponding continuation segment transmitted. The decoder uses this information to determine when (i.e. in which of the following GOS) the continuation segment will be received. Linking of segments relies on the particular order the trajectories are transmitted. The order of decoding and linking segments is presented and explained in <figref idref="f0003">Figure 3</figref>.</p>
<p id="p0022" num="0022"><figref idref="f0003">Figure 3</figref> shows a scheme of linking trajectory segments according to an embodiment of the invention. Segments decoded within one GOS are marked with the same color. Each segment is marked with a number (e.g. SEG #5) which determines the order of decoding (i.e. order of receiving the segment data from bitstream). In above example SEG#1 has length of 32 data points and is marked to be continued (isCont = 1). Therefore, SEG #1 is going to be continued in GOS #5, where there are two new segments received (SEG #5 and SEG #6). The order of decoding this segments determines that the continuation for SEG #1 is SEG #5.</p>
<heading id="h0011">2.2.3. Sinusoidal synthesis and output signal</heading>
<p id="p0023" num="0023">The currently decoded trajectories amplitude and frequency data are stored in the trajectory buffers segAmpl and segFreq. The length of each of the buffers is HFSC_BUFF_LENGTH is<!-- EPO <DP n="7"> --> equal to HFSC_MAX_SEGMENT_LENGTH = 32 trajectory data points. In order to keep high audio quality the decoder employs classic oscillator-based additive synthesis performed in sample domain. For this purpose, the trajectory data are to be interpolated on a sample basis, taking into account the synthesis frame length H = 256. In order to reduce the memory requirements the output signal is synthesized only from trajectory data points corresponding to currently decoded USAC frame and HFSC_BUFFER_LENGTH is equal to 2048. Once the synthesis is finished the buffer is shifted and appended with new HFSC data. There is no delay added during the synthesis process.</p>
<p id="p0024" num="0024">The operation of the HFSC tool is strictly synchronized with the USAC frame structure. The HFSC data frame (GOS) is sent once per 1 USAC frame. It describes up to 8 trajectory data values corresponding to 8 synthesis frames. In other words, there are 8 synthesis frames of sinusoidal trajectory data per each USAC frame and each synthesis frame is 256 samples long at the sampling rate of the USAC codec.</p>
<p id="p0025" num="0025">If Core Decoder output is carried in sample domain, the group of 2048 HFSC samples are passed to the output, where the data is mixed with the contents produced by the USAC decoder with appropriate scaling.</p>
<p id="p0026" num="0026">If output of the Core Decoder needs to be carried in frequency domain an additional QMF analysis is required. The QMF analysis introduces delay of 384 samples, however it holds within the delay introduced by eSBR decoder. Another option might be direct synthesis of sinusoidal partials to QMF domain.</p>
<heading id="h0012">3. Bitstream Syntax and Specification Text</heading>
<p id="p0027" num="0027">The necessary changes to the standard text containing bit stream syntax, semantics and a description of the decoding process can be found in Annex A of the document as a diff-text.</p>
<heading id="h0013">4. Coding delay</heading>
<p id="p0028" num="0028">The maximum coding delay is related to HFSC_MAX_SEGMENT_LENGTH, GOS_LENGTH, sinusoidal analysis frame length SINAN_LENGTH=2048 and synthesis frame length H = 256. Sinusoidal analysis requires zero-padding with 768 samples and overlapping with 1024 samples. The resulting maximum coding delay of HFSC tool is: (HFSC_MAX_SEGMENT_LENGTH + GOS _LENGTH - 1)<sup>∗</sup>H + SINAN_LENGTH - H =<!-- EPO <DP n="8"> --> (32+8-1)<sup>∗</sup>256+2048-256 = 11776 samples. The delay is not added at the front of other Core Coder tools.</p>
<heading id="h0014">5. Stereo and multichannel signals coding</heading>
<p id="p0029" num="0029">For stereo and multichannel signals, the trajectories of the channels are grouped and only the presence of the trajectories is signaled in a header. The HFSC tool is active only for part of audio channels. The HFSC payload is transmitted in USAC Extension Element. It is recommended to send additional information related to trajectory panning as illustrated in the <figref idref="f0005">Figures 4b</figref> below to further save some bits. However, due to low bitrate overhead introduced by HFSC each channel can also be encoded independently as illustrated in <figref idref="f0004">Figure 4a. Figure 4a</figref> shows an illustration of independent encoding for each channel according to an example useful for understanding the invention which was originally filed but which does not represent an embodiment of the presently claimed invention. <figref idref="f0005">Figure 4b</figref> shows an illustration of sending additional information related to trajectory panning according to an embodiment of the invention.</p>
<heading id="h0015">6. Complexity and memory requirements</heading>
<heading id="h0016">6.1. Computational complexity</heading>
<p id="p0030" num="0030">The computational complexity of the proposed tool depends on the number of currently transmitted trajectories which in every HFSC frame is limited to HFSC_MAX_TRJ=8. The dominant component of the computational complexity is related to the sinusoidal synthesis. Time domain synthesis assumptions are as follows:
<ul id="ul0004" list-style="bullet" compact="compact">
<li>Taylor series expansions employed for calculating of cos() and exp() functions</li>
<li>16-bit output resolution</li>
</ul></p>
<p id="p0031" num="0031">The computational complexity of DCT based segment decoding is negligibly small when compared to the synthesis. The HFSC tool generates in average is 0.6 sinusoidal trajectory, thus the total number of operations per sample is 18<sup>∗</sup>0.6 = 10.8. Assuming the output sampling frequency is 44100 Hz, the total number of MOPS per one channel active is 0.48. When 8 audio channels would be enhanced by HFSC tool, the total number of MOPS is 3.84.<!-- EPO <DP n="9"> -->
<ul id="ul0005" list-style="bullet" compact="compact">
<li>Comparison to the total computational complexity of Core decoder with 22 channels (11 CPE's used):Reference Model Core coder: 118 MOPS</li>
<li>HFSC: 8<sup>∗</sup>0.48 = 3.48</li>
<li>RM+HFSC = 121.48</li>
<li>(RM+HFSC/RM) = 1,02</li>
<li>2% increase of computational complexity, when no additional QMF analysis is needed</li>
</ul></p>
<heading id="h0017">6.2. Memory requirements</heading>
<p id="p0032" num="0032">For online operation, the trajectory decoding algorithm requires a number of matrices of size:
<ul id="ul0006" list-style="bullet" compact="compact">
<li>32 × 8 = 256 elements for <i>amplCoeff</i></li>
<li>32 × 8 = 256 elements for <i>freqCoeff</i></li>
<li>33 × 8 = 256 elements for <i>segAmpl</i></li>
<li>33 × 8 = 256 elements for <i>segFreq</i></li>
<li>32 elements for DCT decoding</li>
</ul></p>
<p id="p0033" num="0033">The synthesis requires vectors of size:
<ul id="ul0007" list-style="bullet" compact="compact">
<li>256<sup>∗</sup>8 = 2048 elements for amplitude output buffer</li>
<li>256<sup>∗</sup>8 = 2048 elements for frequency and phase output buffer</li>
</ul></p>
<p id="p0034" num="0034">Since these elements are used to store a 4-byte floating point values, the estimated amount of memory required for computations is around 20kB RAM.</p>
<p id="p0035" num="0035">The Huffman tables require approximately 250B ROM.</p>
<heading id="h0018">7. Evidence of merit</heading>
<p id="p0036" num="0036">According to workplan [5], the listening tests were conducted for stereo signals with total bitrate of 20kbps. The listening test report is presented in [3].</p>
<heading id="h0019">8. Summary and conclusions</heading>
<p id="p0037" num="0037">In the current document a complete CE proposal of HFSC tool was presented which improves high frequency tonal component coding in MPEG-H Core Coder. Embodiments of the<!-- EPO <DP n="10"> --> presented CE technology may be integrated into the MPEG-H audio standard as part of Phase 2.</p>
<heading id="h0020">Annex A: Proposed changes to the specification text</heading>
<p id="p0038" num="0038">The following bit stream syntax is based on ISO/IEC 23008-3:2015 where we propose the following modifications.</p>
<heading id="h0021"><i>Add table entry ID_EXT_ELE_HFSC to Table 50:</i></heading>
<p id="p0039" num="0039">
<tables id="tabl0001" num="0001">
<table frame="all">
<title>Table 50 - Value of usacExtElementType</title>
<tgroup cols="2">
<colspec colnum="1" colname="col1" colwidth="45mm"/>
<colspec colnum="2" colname="col2" colwidth="45mm"/>
<thead valign="top">
<row>
<entry>usacExtElementType Value</entry>
<entry>usacExtElementType Value</entry></row></thead>
<tbody>
<row>
<entry>···</entry>
<entry>···</entry></row>
<row>
<entry>ID_EXT_ELE_HFSC 10</entry>
<entry>10</entry></row>
<row>
<entry>···</entry>
<entry>···</entry></row></tbody></tgroup>
</table>
</tables></p>
<heading id="h0022"><i>Add table entry ID_EXT_ELE_HFSC to Table 51:</i></heading>
<p id="p0040" num="0040">
<tables id="tabl0002" num="0002">
<table frame="all">
<title>Table 51 - Interpretation of data blocks for extension payload decoding</title>
<tgroup cols="2">
<colspec colnum="1" colname="col1" colwidth="45mm"/>
<colspec colnum="2" colname="col2" colwidth="89mm"/>
<thead valign="top">
<row>
<entry>usacExtElementType</entry>
<entry>The concatenated usacExtElementSegmentData represents:</entry></row></thead>
<tbody>
<row>
<entry>···</entry>
<entry>···</entry></row>
<row>
<entry>ID_EXT_ELE_HFSC</entry>
<entry>HfscGroupOfSegments()</entry></row>
<row>
<entry>···</entry>
<entry>···</entry></row></tbody></tgroup>
</table>
</tables></p>
<heading id="h0023"><i>Add case ID_EXT_ELE_HFSC to syntax of mpegh3daExtElementConfig():</i></heading>
<p id="p0041" num="0041">
<tables id="tabl0003" num="0003">
<table frame="all">
<title>Table XX - Syntax of mpegh3daExtElementConfig()</title>
<tgroup cols="3">
<colspec colnum="1" colname="col1" colwidth="80mm"/>
<colspec colnum="2" colname="col2" colwidth="19mm"/>
<colspec colnum="3" colname="col3" colwidth="20mm"/>
<thead valign="top">
<row>
<entry>Syntax</entry>
<entry>No. of bits</entry>
<entry>Mnemonic</entry></row></thead>
<tbody>
<row rowsep="0">
<entry>mpegh3daExtElementConfig()</entry>
<entry/>
<entry/></row>
<row rowsep="0">
<entry>{</entry>
<entry/>
<entry/></row>
<row rowsep="0">
<entry> ···</entry>
<entry/>
<entry/></row>
<row rowsep="0">
<entry> case ID_EXT<i>_</i>ELE_HFSC: /<sup>∗</sup> high freq. sin. coding<sup>∗</sup>/</entry>
<entry/>
<entry/></row>
<row rowsep="0">
<entry>  HFSCConfig();</entry>
<entry/>
<entry/></row>
<row rowsep="0">
<entry>  break;</entry>
<entry/>
<entry/></row>
<row rowsep="0">
<entry> ···</entry>
<entry/>
<entry/></row>
<row>
<entry>}</entry>
<entry/>
<entry/></row></tbody></tgroup>
</table>
</tables><!-- EPO <DP n="11"> --></p>
<heading id="h0024"><i>Add Table XX - Syntax of HFSCConfig():</i></heading>
<p id="p0042" num="0042">
<tables id="tabl0004" num="0004">
<table frame="all">
<title>Table XX - Syntax of HFSCConfig()</title>
<tgroup cols="3">
<colspec colnum="1" colname="col1" colwidth="65mm"/>
<colspec colnum="2" colname="col2" colwidth="22mm"/>
<colspec colnum="3" colname="col3" colwidth="30mm"/>
<thead valign="top">
<row>
<entry>Syntax</entry>
<entry>No. of bits</entry>
<entry>Mnemonic</entry></row></thead>
<tbody>
<row rowsep="0">
<entry>HFSCConfig()</entry>
<entry/>
<entry/></row>
<row rowsep="0">
<entry>{</entry>
<entry/>
<entry/></row>
<row rowsep="0">
<entry>  for(elm=0;elm &lt; numElements; elm++) {</entry>
<entry/>
<entry/></row>
<row rowsep="0">
<entry>   <b>hfscFlag[elm];</b></entry>
<entry><b>1</b></entry>
<entry>uimsbf</entry></row>
<row rowsep="0">
<entry>  }</entry>
<entry/>
<entry/></row>
<row>
<entry>}</entry>
<entry/>
<entry/></row>
<row>
<entry namest="col1" nameend="col3" align="left">NOTE: numElements corresponds only to SCE, CPE and QCE channel elements.</entry></row></tbody></tgroup>
</table>
</tables></p>
<heading id="h0025"><i>Add Table XX - Syntax of HfscGroupOfSegments():</i></heading>
<p id="p0043" num="0043">
<tables id="tabl0005" num="0005">
<table frame="all">
<title>Table XX - Syntax of HfscGroupOfSegments()</title>
<tgroup cols="3">
<colspec colnum="1" colname="col1" colwidth="91mm"/>
<colspec colnum="2" colname="col2" colwidth="19mm"/>
<colspec colnum="3" colname="col3" colwidth="20mm"/>
<thead valign="top">
<row>
<entry>Syntax</entry>
<entry>No. of bits</entry>
<entry>Mnemonic</entry></row></thead>
<tbody>
<row rowsep="0">
<entry>HfscGroupOfSegments()</entry>
<entry/>
<entry/></row>
<row rowsep="0">
<entry>{</entry>
<entry/>
<entry/></row>
<row rowsep="0">
<entry><b> if(hfscDataPresent){</b></entry>
<entry><b>1</b></entry>
<entry><b>uimsbf</b></entry></row>
<row rowsep="0">
<entry><b>  numTrajectories;</b></entry>
<entry><b>3</b></entry>
<entry><b>uimsbf</b></entry></row>
<row rowsep="0">
<entry>  for(k=0;k&lt;numTrajectories;k++){</entry>
<entry/>
<entry/></row>
<row rowsep="0">
<entry><b>   isContinued[k];</b></entry>
<entry><b>1</b></entry>
<entry><b>uimsbf</b></entry></row>
<row rowsep="0">
<entry><b>   segLength[k];</b></entry>
<entry><b>2</b></entry>
<entry><b>uimsbf</b></entry></row>
<row rowsep="0">
<entry><b>   amplQuant[k];</b></entry>
<entry><b>1</b></entry>
<entry><b>uimsbf</b></entry></row>
<row rowsep="0">
<entry><b>   amplTransformCoeffDC[k];</b></entry>
<entry><b>8</b></entry>
<entry><b>uimsbf</b></entry></row>
<row rowsep="0">
<entry>   j = 0;</entry>
<entry>NOTE 1)</entry>
<entry/></row>
<row rowsep="0">
<entry>   while(amplTransformIndex[k][j] = huff_dec(<b>huffWord</b>)){</entry>
<entry><b>1..12</b></entry>
<entry/></row>
<row rowsep="0">
<entry>    if(amplTransformIndex[k][j] == 0) {</entry>
<entry/>
<entry/></row>
<row rowsep="0">
<entry>     numAmplCoeffs = j;</entry>
<entry/>
<entry/></row>
<row rowsep="0">
<entry>     break;</entry>
<entry/>
<entry/></row>
<row rowsep="0">
<entry>   }</entry>
<entry/>
<entry/></row>
<row rowsep="0">
<entry>   j++;</entry>
<entry/>
<entry/></row>
<row rowsep="0">
<entry>  }</entry>
<entry/>
<entry/></row>
<row rowsep="0">
<entry>  for(j=0; j &lt; numAmplCoeffs; j++)</entry>
<entry>NOTE 2)</entry>
<entry/></row>
<row rowsep="0">
<entry>   amplTransformCoeffAC[k][j]= huff_dec(<b>huffWord</b>);</entry>
<entry><b>1..15</b></entry>
<entry/></row>
<row rowsep="0">
<entry><b>  freqQuant[k];</b></entry>
<entry><b>1</b></entry>
<entry><b>uimsbf</b></entry></row>
<row rowsep="0">
<entry><b>  freqTransformCoeffDC[k];</b></entry>
<entry><b>11</b></entry>
<entry><b>uimsbf</b></entry></row>
<row rowsep="0">
<entry>  j = 0;</entry>
<entry>NOTE 1)</entry>
<entry/></row>
<row rowsep="0">
<entry>  while(freqTransformlndex[k][j] = huff_dec(<b>huffWord</b>)){</entry>
<entry><b>1..12</b></entry>
<entry/></row>
<row rowsep="0">
<entry>   if(freqTransformlndex[k][j] = =0) {</entry>
<entry/>
<entry/></row>
<row rowsep="0">
<entry>    numFreqCoeffs = j;</entry>
<entry/>
<entry/></row>
<row rowsep="0">
<entry>    break;</entry>
<entry/>
<entry/></row>
<row rowsep="0">
<entry>  }</entry>
<entry/>
<entry/></row><!-- EPO <DP n="12"> -->
<row rowsep="0">
<entry>   j++;</entry>
<entry/>
<entry/></row>
<row rowsep="0">
<entry>  }</entry>
<entry/>
<entry/></row>
<row rowsep="0">
<entry>  for(j=0; j &lt; numFreqCoeffs; j++)</entry>
<entry>NOTE 2)</entry>
<entry/></row>
<row rowsep="0">
<entry>   freqTransformCoeffAC[k][j]= huff_dec(<b>huffWord</b>);</entry>
<entry><b>1..15</b></entry>
<entry/></row>
<row rowsep="0">
<entry>  }</entry>
<entry/>
<entry/></row>
<row rowsep="0">
<entry> }</entry>
<entry/>
<entry/></row>
<row>
<entry>}</entry>
<entry/>
<entry/></row>
<row rowsep="0">
<entry namest="col1" nameend="col3" align="left">NOTE 1: Huffman codes table: Table XX</entry></row>
<row>
<entry namest="col1" nameend="col3" align="left">NOTE 2: Huffman codes table: Table XX</entry></row></tbody></tgroup>
</table>
</tables></p>
<p id="p0044" num="0044"><i>It is proposed to append the following descriptive text to a new section "5.5.X High Frequency Sinusoidal Coding Tool" with the following content:</i></p>
<heading id="h0026">5.5.X High Frequency Sinusoidal Coding Tool</heading>
<heading id="h0027">5.5.X.1 Tool description</heading>
<p id="p0045" num="0045">The High Frequency Sinusoidal Coding Tool (HFSC) is a method for coding of selected high frequency tonal components using an approach based on sinusoidal modeling. Tonal components are represented as sinusoidal trajectories - data vectors with varying amplitude and frequency values. The trajectories are divided into segments and encoded with technique based on Discreet Cosine Transform.</p>
<heading id="h0028">5.5.X.2 Terms and Definitions</heading>
<heading id="h0029"><b>Help elements:</b></heading>
<p id="p0046" num="0046"><br/>
hfscFlag[elm]  Indicates the use of the tool for a certain group of signals:
<tables id="tabl0006" num="0006">
<table frame="all">
<title>Table XX - hfscFlag</title>
<tgroup cols="2">
<colspec colnum="1" colname="col1" colwidth="19mm"/>
<colspec colnum="2" colname="col2" colwidth="35mm"/>
<thead valign="top">
<row>
<entry><b>hfscFlag</b></entry>
<entry><b>Meaning</b></entry></row></thead>
<tbody>
<row>
<entry>0</entry>
<entry>HFSC tool not applied</entry></row>
<row>
<entry>1</entry>
<entry>HFSC tool applied</entry></row></tbody></tgroup>
</table>
</tables><br/>
HfscGroupOfSegments ()  Syntactic element that contains HFSC Group Of Segment data<br/>
hfscDataPresent  Indicates if HFSC data are there any segments transmitted in current Group Of Segments (GOS)<br/>
numTrajectories  Indicates the number of trajectory segments transmitted in current GOS<br/>
isContinued  Indicates whether this particular segment will have its continuation in next GOS<!-- EPO <DP n="13"> -->
<tables id="tabl0007" num="0007">
<table frame="all">
<title>Table XX - isContinued</title>
<tgroup cols="2">
<colspec colnum="1" colname="col1" colwidth="24mm"/>
<colspec colnum="2" colname="col2" colwidth="47mm"/>
<thead valign="top">
<row>
<entry><b>isContinued</b></entry>
<entry>Meaning</entry></row></thead>
<tbody>
<row>
<entry>0</entry>
<entry>Segment will not be continued</entry></row>
<row>
<entry>1</entry>
<entry>Segment will be continued</entry></row></tbody></tgroup>
</table>
</tables><br/>
segLength  Indicates the length of the currently decoded segment
<tables id="tabl0008" num="0008">
<table frame="all">
<title>Table XX - segLength</title>
<tgroup cols="2">
<colspec colnum="1" colname="col1" colwidth="22mm"/>
<colspec colnum="2" colname="col2" colwidth="42mm"/>
<thead valign="top">
<row>
<entry><b>segLength</b></entry>
<entry>Trajectory segment length</entry></row></thead>
<tbody>
<row>
<entry>00</entry>
<entry>8</entry></row>
<row>
<entry>01</entry>
<entry>16</entry></row>
<row>
<entry>10</entry>
<entry>24</entry></row>
<row>
<entry>11</entry>
<entry>32</entry></row></tbody></tgroup>
</table>
</tables><br/>
amplQuant  Quantization step for amplitude coefficients
<tables id="tabl0009" num="0009">
<table frame="all">
<title>Table XX - amplQuant</title>
<tgroup cols="2">
<colspec colnum="1" colname="col1" colwidth="22mm"/>
<colspec colnum="2" colname="col2" colwidth="50mm"/>
<thead valign="top">
<row>
<entry><b>amplQuant</b></entry>
<entry>Amplitude quantization step in dB</entry></row></thead>
<tbody>
<row>
<entry>0</entry>
<entry>0.5</entry></row>
<row>
<entry>1</entry>
<entry>1</entry></row></tbody></tgroup>
</table>
</tables><br/>
freqQuant  Quantization step for frequency coefficients
<tables id="tabl0010" num="0010">
<table frame="all">
<tgroup cols="2">
<colspec colnum="1" colname="col1" colwidth="21mm"/>
<colspec colnum="2" colname="col2" colwidth="55mm"/>
<thead valign="top">
<row>
<entry><b>freqQuant</b></entry>
<entry>Frequency quantization step in cents</entry></row></thead>
<tbody>
<row>
<entry>0</entry>
<entry>2</entry></row>
<row>
<entry>1</entry>
<entry>4</entry></row></tbody></tgroup>
</table>
</tables><br/>
huffWord  Huffman codeword<br/>
amplTransformCoeffDC  Amplitude DCT transform DC coefficient<br/>
freqTransformCoeffDC  Frequency DCT transform DC coefficient<br/>
numAmplCoeffs  Number of decoded amplitude AC coefficients<!-- EPO <DP n="14"> -->
<dl id="dl0001" compact="compact">
<dt>numFreqCoeffs</dt><dd>Number of decoded frequency AC coefficients</dd>
<dt>amplTransformCoeffAC</dt><dd>Array with amplitude DCT transform AC coefficients</dd>
<dt>freqTransformCoeffAC</dt><dd>Array with frequency DCT transform AC coefficients</dd>
<dt>amplTransformIndex</dt><dd>Array with amplitude DCT transform AC indices</dd>
<dt>freqTransformIndex</dt><dd>Array with frequency DCT transform AC indices</dd>
<dt>amplOffsetDC</dt><dd>Constant integer added to each decoded amplitude DC coefficient, equal to 32</dd>
<dt>freqOffsetDC</dt><dd>Constant integer added to each decoded frequency DC coefficient, equal to 600</dd>
<dt>offsetAC</dt><dd>Constant integer added to each decoded amplitude and frequency AC coefficient, equal to 1</dd>
<dt>sgnAC</dt><dd>Bit indicating the sign of decoded AC coefficient, 1 indicates negative value.</dd>
<dt>MAX_NUM_TRJ</dt><dd>Maximum number of processed trajectories, equal to 8</dd>
<dt>HFSC_BUFFER_LENGTH</dt><dd>Length of buffer for storing decoded trajectory amplitude and frequency data</dd>
<dt>HFSC_SYNTH_LENGTH</dt><dd>Length of buffer for storing synthesized HFSC samples, equal to 2048</dd>
<dt>HFSC_FS</dt><dd>Nominal sampling frequency for HFSC sinusoidal trajectory data, equal to 48000 Hz</dd>
</dl></p>
<heading id="h0030">5.5.X.3 Decoding process</heading>
<heading id="h0031">5.5.X.3.1 General</heading>
<p id="p0047" num="0047">Element usacExtElementType ID_EXT_ELE_HFSC according to hfscFlag[] contains HFSC data (HFSC Groups of Segments - GOS) corresponding to the currently processed channel elements i.e. SCE (Single Channel Element), CPE (Channel Pair Element), QCE (Quad Channel Element). The number of transmitted GOS structures for particular type of channel element is defined as follows:
<tables id="tabl0011" num="0011">
<table frame="all">
<title>Table XX - Number of transmitted GOS structures</title>
<tgroup cols="2">
<colspec colnum="1" colname="col1" colwidth="35mm"/>
<colspec colnum="2" colname="col2" colwidth="42mm"/>
<thead valign="top">
<row>
<entry><b>USAC element type</b></entry>
<entry>Number of GOS structures</entry></row></thead>
<tbody>
<row>
<entry>SCE</entry>
<entry>1</entry></row>
<row>
<entry>CPE</entry>
<entry>2</entry></row>
<row>
<entry>QCE</entry>
<entry>4</entry></row></tbody></tgroup>
</table>
</tables><!-- EPO <DP n="15"> --></p>
<p id="p0048" num="0048">The decoding of each GOS starts with decoding the number of transmitted segments by reading the field numSegments and increasing it by 1. Then decoding of particular k-th segment starts from decoding its length segLength[k] and isContinued[k] flag. The decoding of other segment data is performed in multiple steps as follows:</p>
<heading id="h0032">5.5.X.3.2 Decoding of segment amplitude data</heading>
<p id="p0049" num="0049">The following procedures are performed for k-th segment amplitude data decoding:
<ol id="ol0001" compact="compact" ol-style="">
<li>1. The amplitude quantization stepA step is calculated according to formula: <maths id="math0001" num=""><math display="block"><mi mathvariant="italic">stepA</mi><mfenced open="[" close="]"><mi>k</mi></mfenced><mo>=</mo><mi mathvariant="italic">log</mi><mfenced><msup><mn mathvariant="italic">10</mn><mfrac><mrow><mi mathvariant="italic">amplQuant</mi><mfenced open="[" close="]"><mi>k</mi></mfenced></mrow><mn>20</mn></mfrac></msup></mfenced><mo>,</mo></math><img id="ib0001" file="imgb0001.tif" wi="66" he="14" img-content="math" img-format="tif"/></maths><br/>
where <i>amplQuant[k]</i> is expressed in dB.</li>
<li>2. The <i>amplTransformCoeffDC[k]</i> is decoded according to formula: <maths id="math0002" num=""><math display="block"><mi mathvariant="italic">amplDC</mi><mfenced open="[" close="]"><mi>k</mi></mfenced><mo>=</mo><mo>−</mo><mi mathvariant="italic">amplTransformCoeffDC</mi><mfenced open="[" close="]"><mi>k</mi></mfenced><mo>×</mo><mi mathvariant="italic">stepA</mi><mfenced open="[" close="]"><mi>k</mi></mfenced><mo>+</mo><mi mathvariant="italic">amplOffsetDC</mi></math><img id="ib0002" file="imgb0002.tif" wi="123" he="5" img-content="math" img-format="tif"/></maths></li>
<li>3. The amplitude AC indices <i>amplIndex[k][j]</i> are decoded by starting with <i>j</i>=<i>0</i> and decoding consecutive <i>amplTransformIndex[k][j]</i> Huffman code words and incrementing <i>j</i>, until a codeword representing 0 is encountered. The Huffman code words are listed in <i>huff_idxTab[]</i> table. Number of decoded indices indicates number of further transmitted coefficients - <i>numCoeff[k].</i> After decoding, each index should be incremented by o<i>ffsetAC.</i></li>
<li>4. The amplitude AC coefficients are also decoded by means of Huffman code words specified in <i>huff_acTab[]</i> table. The AC coefficients are signed values, so additional 1 sign bit <i>sgnAC[k][j]</i> after each Huffman code word is transmitted, where 1 indicates negative value. Finally, the value of AC coefficient is decoded according to formula: <i>amplAC[k][j]</i> = <i>sgnAC[k][j]</i> ( <i>amplTransformCoeffAC[k][j] -0.25)</i> ×<i>stepA[k]</i></li>
<li>5. Decoded amplitude transform DC and AC coefficients are placed into vector <i>amplCoeff</i> of length equal to <i>segLength[k].</i> The <i>amplDC[k]</i> coefficient is placed at index 0 and <i>amplAC[k][j]</i> coefficients are placed according to decoded <i>amplIndex[k][j]</i> indices.</li>
<li>6. The sequence of trajectory amplitude data in logarithmic scale is reconstructed from the inverse discrete cosine transform and moved into <i>segAmpl[k][i]</i> buffer according to:<!-- EPO <DP n="16"> --> <maths id="math0003" num=""><math display="block"><msub><mi mathvariant="italic">segAmpl</mi><mi mathvariant="italic">log</mi></msub><mfenced open="[" close="]"><mi>k</mi></mfenced><mfenced open="[" close="]"><mi>i</mi></mfenced><mo>=</mo><mstyle displaystyle="true"><munderover><mo>∑</mo><mrow><mi>r</mi><mo>=</mo><mn>0</mn></mrow><mrow><mi mathvariant="italic">segLength</mi><mfenced open="[" close="]"><mi>k</mi></mfenced></mrow></munderover><mrow><mi mathvariant="italic">amplCoeff</mi><mfenced open="[" close="]"><mi>k</mi></mfenced><mfenced open="[" close="]"><mi>r</mi></mfenced><mi mathvariant="normal"> </mi><mi mathvariant="normal"> </mi><mi>w</mi><mfenced open="[" close="]"><mi>r</mi></mfenced><mi mathvariant="normal"> </mi><mi mathvariant="normal"> </mi><mi mathvariant="italic">cos</mi></mrow></mstyle><mfenced separators=""><mfrac><mi>π</mi><mrow><mn>2</mn><mi mathvariant="normal"> </mi><mi mathvariant="italic">segLength</mi><mfenced open="[" close="]"><mi>k</mi></mfenced></mrow></mfrac><mfenced separators=""><mi>h</mi><mo>+</mo><mn>1</mn></mfenced><mi>r</mi></mfenced><mo>,</mo></math><img id="ib0003" file="imgb0003.tif" wi="119" he="12" img-content="math" img-format="tif"/></maths> where: <maths id="math0004" num=""><math display="block"><mi>w</mi><mfenced open="[" close="]"><mi>r</mi></mfenced><mo>=</mo><mrow><mo>{</mo><mtable><mtr><mtd><msup><mfenced separators=""><mi mathvariant="italic">segLength</mi><mfenced open="[" close="]"><mi>k</mi></mfenced></mfenced><mrow><mo>−</mo><mn>0.5</mn></mrow></msup></mtd><mtd><mrow><mi>f</mi><mi>o</mi><mi>r</mi><mi mathvariant="normal"> </mi><mi>r</mi><mo>=</mo><mn>0</mn></mrow></mtd></mtr><mtr><mtd><mrow><msqrt><mn>2</mn></msqrt><msup><mfenced separators=""><mi mathvariant="italic">segLength</mi><mfenced open="[" close="]"><mi>k</mi></mfenced></mfenced><mrow><mo>−</mo><mn>0.5</mn></mrow></msup></mrow></mtd><mtd><mrow><mi mathvariant="normal"> </mi><mi mathvariant="normal"> </mi><mi mathvariant="normal"> </mi><mi mathvariant="normal"> </mi><mi mathvariant="normal"> </mi><mi mathvariant="normal"> </mi><mi>f</mi><mi>o</mi><mi>r</mi><mi mathvariant="normal"> </mi><mi>r</mi><mo>&gt;</mo><mn>0</mn></mrow></mtd></mtr></mtable></mrow></math><img id="ib0004" file="imgb0004.tif" wi="71" he="12" img-content="math" img-format="tif"/></maths> The amplitude data are placed in <i>segAmpl</i> buffer of length equal to <i>HFSC_BUFFER_LENGTH</i>, beginning with the index <i>i</i> = 1. The value under index <i>i =</i> 0 is set to 0.</li>
<li>7. The linear values of amplitudes in <i>segAmpl[k][i]</i> are calculated by: <i>segAmpl[k][i] exp</i> (<i>segAmpllog[k][i]</i>)</li>
</ol></p>
<heading id="h0033">5.5.X.3.3 Decoding of segment frequency data</heading>
<p id="p0050" num="0050">The following procedures are performed for <i>k-th</i> segment frequency data decoding:
<ol id="ol0002" compact="compact" ol-style="">
<li>1. The frequency quantization <i>stepF[k]</i> is calculated according to formula: <maths id="math0005" num=""><math display="block"><mi mathvariant="italic">stepF</mi><mfenced open="[" close="]"><mi>k</mi></mfenced><mo>=</mo><mi mathvariant="italic">freqQuant</mi><mfenced open="[" close="]"><mi>k</mi></mfenced><mo>×</mo><mi mathvariant="italic">log</mi><mfenced><msup><mn>2</mn><mfrac><mn>1</mn><mn>1200</mn></mfrac></msup></mfenced><mo>,</mo></math><img id="ib0005" file="imgb0005.tif" wi="83" he="15" img-content="math" img-format="tif"/></maths> where <i>freqQuant[k]</i> is expressed in cents.</li>
<li>2. The <i>freqTransformCoeffDC[k]</i> is decoded according to formula: <maths id="math0006" num=""><math display="block"><mi mathvariant="italic">freqDC</mi><mfenced open="[" close="]"><mi>k</mi></mfenced><mo>=</mo><mo>−</mo><mi mathvariant="italic">freqTransformCoeffDC</mi><mfenced open="[" close="]"><mi>k</mi></mfenced><mo>×</mo><mi mathvariant="italic">stepF</mi><mfenced open="[" close="]"><mi>k</mi></mfenced><mo>+</mo><mi mathvariant="italic">freqOffsetDC</mi></math><img id="ib0006" file="imgb0006.tif" wi="119" he="5" img-content="math" img-format="tif"/></maths></li>
<li>3. Decoding process of frequency AC indices is the same as for amplitude AC indices. The resulting data vector is <i>freqIndex[k][j].</i></li>
<li>4. Decoding process of frequency AC coefficients is the same as for amplitude AC coefficients. The resulting data vector is <i>freqAC[k][j].</i></li>
<li>5. Decoded frequency transform DC and AC coefficients are placed into vector <i>freqCoeff</i> of length equal to <i>segLength[k].</i> The <i>freqDC[k]</i> coefficient is placed in position <i>j</i>=<i>0</i> and <i>freqAC[k][j]</i> coefficients are placed according to decoded <i>freqIndex[k][j]</i> indices.</li>
<li>6. The reconstruction of sequence of trajectory frequency data in logarithmic scale and further transformation to linear scale is performed in the same manner as for amplitude data. The resulting vector is <i>segFreq[k][i].</i> The linear values of frequency data are stored in the range from 0.07 - 0.5. In order to obtain frequency in Hz, decoded frequency values should be multiplied by <i>HFSC_FS.</i></li>
</ol></p>
<heading id="h0034">5.5.X.3.4 Ordering and linking of trajectory segments</heading><!-- EPO <DP n="17"> -->
<p id="p0051" num="0051">The original sinusoidal trajectories build in the encoder are partitioned into an arbitrary number of segments. The length of currently processed segment segLength[k] and continuation flag isContinued[k] is used to determine when (i.e. in which of the following GOS) the continuation segment will be received. Linking of segments relies on the particular order the trajectories are transmitted. The order of decoding and linking segments is presented and explained in <figref idref="f0003">Figure 3</figref>.</p>
<heading id="h0035">5.5.X.3.5 Synthesis of decoded trajectories</heading>
<p id="p0052" num="0052">The received representation of trajectory segments is temporarily stored in data buffers <i>segAmpl[k][i]</i> and <i>segFreq[k][i],</i> where <i>k</i> represents the index of segment not greater than <i>MAX_NUM_TRJ =</i> 8, and <i>i</i> represents the trajectory data index within a segment, <i>0</i>&lt;= <i>i &lt; HFSC_BUFFER_LENGTH.</i> The index <i>i=0</i> of buffers <i>segAmpl</i> and <i>segFreq</i> is filled with data depending on the one of two possible scenarios for further processing of particular segments:
<ol id="ol0003" compact="compact" ol-style="">
<li>1. The received segment is starting a new trajectory, then the <i>i=0</i> index amplitude and frequency data are provided by simple extrapolation process: <maths id="math0007" num=""><math display="block"><mtable><mtr><mtd><mrow><mi mathvariant="italic">segFreq</mi><mfenced open="[" close="]"><mi>k</mi></mfenced><mfenced open="[" close="]"><mn>0</mn></mfenced><mo>=</mo><mi mathvariant="italic">segFreq</mi><mfenced open="[" close="]"><mi>k</mi></mfenced><mfenced open="[" close="]"><mn>1</mn></mfenced><mo>,</mo></mrow></mtd></mtr><mtr><mtd><mrow><mi mathvariant="italic">segAmpl</mi><mfenced open="[" close="]"><mi>k</mi></mfenced><mfenced open="[" close="]"><mn>0</mn></mfenced><mo>=</mo><mn>0</mn><mo>.</mo></mrow></mtd></mtr></mtable></math><img id="ib0007" file="imgb0007.tif" wi="57" he="13" img-content="math" img-format="tif"/></maths></li>
<li>2. The received segment is recognized as a continuation for the segment processed in the previously received GOS structure, then the <i>i=0</i> index amplitude and frequency data are copy of the last data points from the segment being continued.</li>
</ol></p>
<p id="p0053" num="0053">The output signal is synthesized from sinusoidal trajectory data stored in the synthesis region of <i>segAmpl[k][l]</i> and <i>segFreq[k][l],</i> where each column corresponds to one synthesis frame and <i>l</i>=<i>0, 1,</i> ...,8. For the purpose of synthesis, these data are to be interpolated on a sample basis, taking into account the synthesis frame length H = 256. The samples of the output signal are calculated according to <maths id="math0008" num=""><math display="block"><msub><mi>y</mi><mi mathvariant="italic">HFSC</mi></msub><mfenced open="[" close="]"><mi>n</mi></mfenced><mo>=</mo><mstyle displaystyle="true"><munderover><mo>∑</mo><mrow><mi>k</mi><mo>=</mo><mn>1</mn></mrow><mrow><mi>K</mi><mfenced open="[" close="]"><mi>n</mi></mfenced></mrow></munderover><mrow><msub><mi>A</mi><mi>k</mi></msub><mfenced open="[" close="]"><mi>n</mi></mfenced><mi mathvariant="normal"> </mi><mi mathvariant="normal"> </mi><mi mathvariant="italic">cos</mi><mfenced separators=""><msub><mi mathvariant="normal">φ</mi><mi>k</mi></msub><mfenced open="[" close="]"><mi>n</mi></mfenced></mfenced></mrow></mstyle></math><img id="ib0008" file="imgb0008.tif" wi="64" he="14" img-content="math" img-format="tif"/></maths> where:
<ul id="ul0008" list-style="none" compact="compact">
<li><i>n = 0... HFSC SYNTH LENGTH-1,</i></li>
</ul><!-- EPO <DP n="18"> --></p>
<p id="p0054" num="0054"><i>K[n]</i> denotes the number of currently active trajectories, i.e. the number of rows synthesis region of <i>segAmpl[k][l]</i> and <i>segFreq[k][l]</i> which have valid data in the frame <i>l</i> = <i>floor(n</i>/<i>H)</i> and <i>l</i> = floor(<i>n</i>/H)+1.
<ul id="ul0009" list-style="none" compact="compact">
<li><i>Ak[n]</i> denotes the interpolated instantaneous amplitude of <i>k</i>-th partial,</li>
<li>ϕ<i>k[n]</i> denotes the interpolated instantaneous phase of <i>k</i>-th partial.</li>
</ul></p>
<p id="p0055" num="0055">The instantaneous phase ϕ<i><sub>k</sub>[n]</i> is calculated from the instantaneous frequency <i>Fk[n]</i> according to: <maths id="math0009" num=""><math display="block"><msub><mi>φ</mi><mi>k</mi></msub><mfenced open="[" close="]"><mi>n</mi></mfenced><mo>=</mo><msub><mi>φ</mi><mi>k</mi></msub><mfenced open="[" close="]" separators=""><msub><mi>n</mi><mi mathvariant="italic">start</mi></msub><mfenced open="[" close="]"><mi>k</mi></mfenced></mfenced><mo>+</mo><mn>2</mn><mi>π</mi><mstyle displaystyle="true"><munderover><mo>∑</mo><mrow><mi>m</mi><mo>=</mo><msub><mi>n</mi><mi mathvariant="italic">start</mi></msub><mfenced open="[" close="]"><mi>k</mi></mfenced><mo>+</mo><mn>1</mn></mrow><mi>n</mi></munderover><mrow><msub><mi>F</mi><mi>k</mi></msub><mfenced open="[" close="]"><mi>n</mi></mfenced><mo>,</mo></mrow></mstyle></math><img id="ib0009" file="imgb0009.tif" wi="99" he="17" img-content="math" img-format="tif"/></maths> where <i>n<sub>start</sub>[k]</i> denotes the initial sample, at which the current segment is started. This initial value of phase is not transmitted and should be stored between consecutive buffers, so that the evolution of phase is continuous. For this purpose the final value of ϕ<i>k[HFSC_SYNTH_LENGTH-1]</i> is written to a vector <i>segPhase[k].</i> This value is used as ϕ<i><sub>k</sub>[n<sub>start</sub>[k]]</i> during the synthesis in the next buffer. At the beginning of each trajectory, ϕ<i><sub>k</sub>[n<sub>start</sub>[k]]</i> = 0 is set.</p>
<p id="p0056" num="0056">The instantaneous parameters <i>Ak[n]</i> and <i>Fk[n]</i> are interpolated on a sample basis from trajectory data stored in trajectory buffer. These parameters are calculated by linear interpolation: <maths id="math0010" num=""><math display="block"><msub><mi>A</mi><mi>k</mi></msub><mfenced open="[" close="]"><msup><mi>n</mi><mo>′</mo></msup></mfenced><mo>=</mo><mi mathvariant="italic">segAmpl</mi><mfenced open="[" close="]"><mi>k</mi></mfenced><mfenced open="[" close="]" separators=""><mi>h</mi><mo>−</mo><mn>1</mn></mfenced><mo>+</mo><mfenced separators=""><mi mathvariant="italic">segAmpl</mi><mfenced open="[" close="]"><mi>k</mi></mfenced><mfenced open="[" close="]"><mi>h</mi></mfenced><mo>−</mo><mi mathvariant="italic">segAmpl</mi><mfenced open="[" close="]"><mi>k</mi></mfenced><mfenced open="[" close="]" separators=""><mi>h</mi><mo>−</mo><mn>1</mn></mfenced></mfenced><mo>*</mo><mfrac><mrow><msup><mi>n</mi><mo>′</mo></msup><mo>−</mo><mi mathvariant="italic">Hh</mi></mrow><mi>H</mi></mfrac></math><img id="ib0010" file="imgb0010.tif" wi="118" he="10" img-content="math" img-format="tif"/></maths> <maths id="math0011" num=""><math display="block"><msub><mi>F</mi><mi>k</mi></msub><mfenced open="[" close="]"><msup><mi>n</mi><mo>′</mo></msup></mfenced><mo>=</mo><mi mathvariant="italic">segFreq</mi><mfenced open="[" close="]"><mi>k</mi></mfenced><mfenced open="[" close="]" separators=""><mi>h</mi><mo>−</mo><mn>1</mn></mfenced><mo>+</mo><mfenced separators=""><mi mathvariant="italic">segFreq</mi><mfenced open="[" close="]"><mi>k</mi></mfenced><mfenced open="[" close="]"><mi>h</mi></mfenced><mo>−</mo><mi mathvariant="italic">segFreq</mi><mfenced open="[" close="]"><mi>k</mi></mfenced><mfenced open="[" close="]" separators=""><mi>h</mi><mo>−</mo><mn>1</mn></mfenced></mfenced><mo>*</mo><mfrac><mrow><msup><mi>n</mi><mo>′</mo></msup><mo>−</mo><mi mathvariant="italic">Hh</mi></mrow><mi>H</mi></mfrac><mo>,</mo></math><img id="ib0011" file="imgb0011.tif" wi="119" he="10" img-content="math" img-format="tif"/></maths> where:
<ul id="ul0010" list-style="none" compact="compact">
<li><i>n'</i> = <i>n-n<sub>start</sub></i></li>
<li><i>h</i> = <i>n' mod H</i></li>
</ul></p>
<p id="p0057" num="0057">Once the group of HFSC_SYNTH_LENGTH samples is synthesized, it is passed to the output, where the data is mixed with the contents produced by the Core Decoder with appropriate scaling to the output data range through multiplication by 215. After the synthesis, the content of <i>segAmpl[k][l]</i> and <i>segFreq[k][l]</i> is shifted by 8 trajectory data points and updated with new data from incoming GOS.</p>
<heading id="h0036">5.5.X.3.6 Additional transform of output signal to QMF domain</heading><!-- EPO <DP n="19"> -->
<p id="p0058" num="0058">Depending on the Core Decoder output signal domain, an additional QMF analysis of the HFSC output signal should be performed according to ISO/IEC 14496-3:2009, subclause 4.6.18.4.</p>
<heading id="h0037">5.5.X.3.7 Huffman Tables for AC indices</heading>
<p id="p0059" num="0059">The following Huffman table huff_idxTab[] shall be used for decoding the DCT AC indices:
<img id="ib0012" file="imgb0012.tif" wi="94" he="119" img-content="program-listing" img-format="tif"/></p>
<heading id="h0038">5.5.X.3.8 Huffman Tables for AC coefficients</heading>
<p id="p0060" num="0060">The following Huffman table huff_acTab[] shall be used for decoding the DCT AC values. Each code word in the bitstream is followed by a 1 bit indicating the sign of decoded AC value.</p>
<p id="p0061" num="0061">The decoded AC values need to be increased by adding the offsetAC value.
<img id="ib0013" file="imgb0013.tif" wi="95" he="15" img-content="program-listing" img-format="tif"/><!-- EPO <DP n="20"> -->
<img id="ib0014" file="imgb0014.tif" wi="93" he="163" img-content="program-listing" img-format="tif"/></p>
<p id="p0062" num="0062">In the following further information about embodiments of the invention is provided.</p>
<p id="p0063" num="0063">Subject of the application:<br/>
High Efficiency Sinusoidal Coding
<ul id="ul0011" list-style="bullet" compact="compact">
<li>low bitrate coding technique for audio signals
<ul id="ul0012" list-style="dash" compact="compact">
<li>based on hiqh quality sinusoidal model</li>
<li>extended with transient and noise coding</li>
<li>bridge between speech and general audio coding techniques</li>
<li>deals with high frequency artifacts introduced by Spectral Band Replication</li>
</ul></li>
<li>MPEG-H 3D Audio and Unified Speech and Audio Coding extension<!-- EPO <DP n="21"> --></li>
<li>MPEG-H 3D Audio / USAC has known problems with high frequency tonal<br/>
components</li>
</ul></p>
<p id="p0064" num="0064"><figref idref="f0006">Fig. 5</figref> shows the motivation for embodiments of the present invention.</p>
<p id="p0065" num="0065"><figref idref="f0007">Fig. 6</figref> shows exemplary MPEG-H 3D Audio artifacts above fSBR, and in particular that the SBR tool is not capable of proper reconstruction of high frequency tonal components (over fSBR band)</p>
<p id="p0066" num="0066"><figref idref="f0008">Fig. 7</figref> shows a comparison for 20kbps (~2kbps of HESC), fSBR=4kHz, between "Original", "MPEG 3DA" and "MPEG 3DA+ HESC".</p>
<p id="p0067" num="0067"><figref idref="f0009">Fig. 8</figref> shows a flow-chart of an exemplary encoding method, comprising the following steps and/or content:
<ul id="ul0013" list-style="none" compact="compact">
<li>114: audio signal samples per frame</li>
<li>312: determining sinusoidal components</li>
<li>313: estimation of frequencies of the components for each frame</li>
<li>314: estimation of amplitudes of the components for each frame</li>
<li>315: splitting particular trajectories into segments</li>
<li>---: merging thus obtained pairs into sinusoidal trajectories</li>
<li>316 &amp; 317: transform the values into the logarithmic scale</li>
<li>320 &amp; 321: quantization</li>
<li>318 &amp; 319: transforming particular trajectories to the frequency</li>
<li>domain by means of a digital transform performed on segments</li>
<li>longer than the frame duration</li>
<li>320 &amp; 321: quantization</li>
<li>322 &amp; 323: selection of transform coefficients in the segments</li>
<li>324 &amp; 326: array of indices of selected coefficients</li>
<li>325 &amp; 327: array of values of selected coefficients</li>
<li>328: entropy encoding</li>
<li>115: outputting the quantized coefficients as output data</li>
</ul></p>
<p id="p0068" num="0068">Thus, <figref idref="f0009">Fig. 8</figref> illustrates the schematic flow of an exemplary audio signal encoding method comprising the steps of: collecting the audio signal samples (114), determining sinusoidal<!-- EPO <DP n="22"> --> components (312) in subsequent frames, estimation of amplitudes (314) and frequencies (313) of the components for each frame, merging thus obtained pairs into sinusoidal trajectories, splitting particular trajectories into segments, transforming (318, 319) particular trajectories to the frequency domain by means of a digital transform performed on segments longer than the frame duration, quantization (320, 321) and selection (322, 323) of transform coefficients in the segments, entropy encoding (328), outputting the quantized coefficients as output data (115), wherein the length of the segments into which each trajectory is split is individually adjusted in time for each trajectory.</p>
<p id="p0069" num="0069"><figref idref="f0010">Figure 9</figref> shows a block-diagram of an exemplary encoder, comprising the following features:
<ul id="ul0014" list-style="none" compact="compact">
<li>110: audio signal encoder</li>
<li>111: analog-to-digital converter</li>
<li>112: processing unit</li>
<li>115: compressed data sequence</li>
<li>113: audio signal</li>
<li>114: audio signal samples</li>
</ul></p>
<p id="p0070" num="0070">Thus, <figref idref="f0010">Fig. 9</figref> illustrates the schematic structure of an exemplary audio signal encoder (110) comprising an analog-to-digital converter (111) and a processing unit (112) provided with: an audio signal samples collecting unit, a determining unit receiving the audio signal samples from the audio signal samples collecting unit and converting them into sinusoidal components in subsequent frames, an estimation unit receiving the sinusoidal components' samples from the determining unit and returning amplitudes and frequencies of the sinusoidal components in each frame, a synthesis unit, generating sinusoidal trajectories on a basis of values of amplitudes and frequencies, a splitting unit, receiving the trajectories from the synthesis unit and splitting them into segments, a transforming unit, transforming trajectories' segments to the frequency domain by means of a digital transform, a quantization and selection unit, converting selected transform coefficients into values resulting from selected quantization levels and discarding remaining coefficients, an entropy encoding unit, encoding quantized coefficients outputted by the quantization and selection unit, and a data outputting unit, wherein the splitting unit is adapted to set the length of the segment individually for each trajectory and to adjust this length over time.<!-- EPO <DP n="23"> --></p>
<p id="p0071" num="0071"><figref idref="f0011">Fig. 10</figref> shows an example analysis of sinusoidal trajectories showing sparse DCT spectra according to prior art.</p>
<p id="p0072" num="0072"><figref idref="f0012">Fig. 11</figref> shows a flow-chart of an exemplary decoding method, comprising the following steps and/or content:
<ul id="ul0015" list-style="none" compact="compact">
<li>115: transferred compressed data</li>
<li>411: entropy code decoder</li>
<li>324 &amp; 326: reconstructed array of indices of the quantized transform coeff.</li>
<li>325 &amp; 327: reconstructed array of values of the quantized transform coeff.</li>
<li>412 &amp; 413: reconstruction blocks, vectors' elements of transform coeff. are filled with the decoded values corresponding to the decoded indices</li>
<li>414 &amp; 415: dequantization, not-encoded coeff. are reconstructed using "ACEnergy" and/or "ACEnvelope"</li>
<li>416 &amp; 417: inverse transform to obtain the reconstructed logarithmic values of frequency and amplitude</li>
<li>418 &amp; 419: convert to linear scale by means of antilogarithm</li>
<li>420 &amp; 421: merging the reconstructed trajectories' segments with the already decoded segments</li>
<li>422: synthesis based on a sinusoidal representation</li>
<li>214: synthesized signal</li>
</ul></p>
<p id="p0073" num="0073">Thus, <figref idref="f0012">Fig. 11</figref> illustrates the schematic flow of an exemplary audio signal decoding method comprising the steps of: retrieving encoded data, reconstruction (411, 412, 413, 414, 415) from the encoded data digital transform coefficients of trajectories' segments, subjecting the coefficients to an inverse transform (416, 417) and performing reconstruction of the trajectories' segments, generation (420, 421) of sinusoidal components, each having amplitude and frequency corresponding to the particular trajectory, reconstruction of the audio signal by summation of the sinusoidal components, wherein missing, not encoded transform coefficients of the sinusoidal components' trajectories are replaced with noise samples generated on a basis of at least one parameter introduced to the encoded data instead of the missing coefficients.</p>
<p id="p0074" num="0074"><figref idref="f0013">Fig. 12</figref> shows a block diagram of an exemplary decoder comprising the following features: 210: audio signal decoder<!-- EPO <DP n="24"> -->
<ul id="ul0016" list-style="none" compact="compact">
<li>213: compressed data</li>
<li>215: analog signal</li>
<li>212: digital-to-analog converter</li>
<li>211: processing unit</li>
<li>214: synthesized digital samples</li>
</ul></p>
<p id="p0075" num="0075">Thus, <figref idref="f0013">Fig. 12</figref> illustrates the schematic structure of an audio signal decoder 210, comprising a digital-to-analog converter 212 and a processing unit 211 provided with: an encoded data retrieving unit, a reconstruction unit, receiving the encoded data and returning digital transform coefficients of trajectories' segments, an inverse transform unit, receiving the transform coefficients and returning reconstructed trajectories' segments, a sinusoidal components generation unit, receiving the reconstructed trajectories' segments and returning sinusoidal components, each having amplitude and frequency corresponding to the particular trajectory, an audio signal reconstruction unit, receiving the sinusoidal components and returning their sum, the decoder comprising a unit adapted to randomly generate not encoded coefficients on a basis of at least one parameter, the parameter being retrieved from the input data, and transferring the generated coefficients to the inverse transform unit.</p>
<p id="p0076" num="0076">In the following, specific aspects of embodiments of the inventions are described.</p>
<heading id="h0039">Aspect 1: QMF and/or MDCT synthesis</heading>
<p id="p0077" num="0077"><figref idref="f0014">Figure 13a</figref>) shows another embodiment of the invention, in particular the general location of the proposed tool within the MPEG-H 3D Audio Core Encoder.</p>
<p id="p0078" num="0078"><figref idref="f0014">Figure 13b</figref>) shows a part of <figref idref="f0012">Fig. 11</figref>. The problem of such implementations: due to complexity issue, the amplitudes and frequencies may not always be synthesized directly into the time domain representation.</p>
<p id="p0079" num="0079"><figref idref="f0014">Figure 13c</figref>) shows an embodiment of the present invention, wherein the steps depicted therein replace the respective steps in <figref idref="f0014">Fig. 13b</figref>), i.e. provide a solution: depending on the system configuration, the decoder shall perform the processing accordingly.</p>
<heading id="h0040">Aspect 2: Extension of Trajectory Length</heading><!-- EPO <DP n="25"> -->
<p id="p0080" num="0080">In some implementations, the length of the segments into which each trajectory is split is individually adjusted in time for each trajectory.</p>
<p id="p0081" num="0081">Such implementations have the problem that the actual trajectory length is arbitrary at the encoder side. This means that a segment may start and end arbitrarily within the group of segments (GOS) structure. Additional signaling is required.</p>
<p id="p0082" num="0082">In an embodiment, the partitioning of trajectory into segments is instead synchronized with the endpoints of the Group of Segments (GOS) structure.</p>
<p id="p0083" num="0083">Thus, there is no need for additional signaling since it will always be guaranteed that the beginning and end of a segment is aligned with the GOS structure.</p>
<heading id="h0041">Aspect 3: Information about trajectory panning</heading>
<p id="p0084" num="0084">
<ul id="ul0017" list-style="none" compact="compact">
<li>Problem: In the context of multichannel coding, it has been found out that the information regarding sinusoidal trajectories is redundant since it may be shared between several channels.</li>
<li>Solution:<br/>
According to an embodiment, instead of coding these trajectories independently for each channel (as shown in <figref idref="f0015">Fig. 14a</figref>)), they can be grouped and only signal their presence with fewer bits (as shown in <figref idref="f0015">Fig. 14b)</figref>), e.g. in headers. Therefore, it is recommended to send additional information related to trajectory panning.</li>
</ul></p>
<heading id="h0042">Aspect 4: Encoding of trajectory groups</heading>
<p id="p0085" num="0085">
<ul id="ul0018" list-style="none" compact="compact">
<li>Problem: Some trajectories may have redundancies such as the presence of harmonics.</li>
<li>Solution: The trajectories can be compressed by signaling only the presence of harmonics in the bitstream as described below as an example.</li>
</ul></p>
<p id="p0086" num="0086">Encoding algorithm has also an ability to jointly encode clusters of segments belonging to harmonic structure of the sound source, i.e. clusters represent fundamental frequency of each<!-- EPO <DP n="26"> --> harmonic structure and its integer multiplications. It can exploit the fact that each segment is characterized with a very similar FM and AM modulations.</p>
<p id="p0087" num="0087">Combination of the Aspects
<ul id="ul0019" list-style="bullet" compact="compact">
<li>The aspectss mentioned above can be applied independently or combined</li>
<li>The benefit of the combination is mostly cumulative. For example, Aspects 2, 3 and 4 can be combined resulting in a total reduced bitrate.</li>
</ul></p>
<heading id="h0043">9. References</heading>
<p id="p0088" num="0088">
<ul id="ul0020" list-style="none">
<li>[1] <nplcit id="ncit0002" npl-type="s"><text>ISO/IEC JTC1/SC29/WG11/M35934, "MPEG-H 3D Audio Phase 2 Core Experiment Proposal on tonal component coding," 111th MPEG Meeting, February 2015, Geneva, Switzerl</text></nplcit>and.</li>
<li>[2] <nplcit id="ncit0003" npl-type="s"><text>ISO/IEC JTC1/SC29/WG11/M36538, "Updated MPEG-H 3D Audio Phase 2 Core Experiment Proposal on tonal component coding," 112th MPEG Meeting, June 2015, Warsaw, Pol</text></nplcit>and.</li>
<li>[3] <nplcit id="ncit0004" npl-type="s"><text>ISO/IEC JTC1/SC29/WG11/M37215, "Zylia Listening Test Report on High Frequency Tonal Component Coding CE," 113th MPEG Meeting, October 2015, Geneva, Switzerl</text></nplcit>and.</li>
<li>[4] <nplcit id="ncit0005" npl-type="s"><text>Zernicki T., Bartkowiak M., Januszkiewicz L., Chryszczanowicz M., "Application of sinusoidal coding for enhanced bandwidth extension in MPEG-D USAC," Convention paper presented at the 138th AES Convention, Warsaw</text></nplcit>.</li>
<li>[5] <nplcit id="ncit0006" npl-type="s"><text>ISO/IEC JTC1/SC29/WG11/N15582, "Workplan on 3D Audio," 112th MPEG Meeting, June 2015, Warsaw, Pol</text></nplcit>and.</li>
<li>[Zernicki et al., 2011] <nplcit id="ncit0007" npl-type="s"><text>Tomasz Zernicki, Maciej Bartkowiak, Marek Domanski, "Enhanced coding of high-frequency tonal components in MPEG-D USAC through joint application of eSBR and sinusoidal modeling," in ICASSP 2011, pp. 501-504, 2011</text></nplcit>.</li>
<li>[Zernicki et al., 2015]<nplcit id="ncit0008" npl-type="s"><text> Tomasz Zernicki, Maciej Bartkowiak, Lukasz Januszkiewicz, Marcin Chryszczanowicz, "Application of sinusoidal coding for enhanced bandwidth extension in<!-- EPO <DP n="27"> --> MPEG-D USAC," in Audio Engineering Society 138th Convention, Warsaw, Poland, May 2015</text></nplcit>.</li>
</ul></p>
</description>
<claims id="claims01" lang="en"><!-- EPO <DP n="28"> -->
<claim id="c-en-01-0001" num="0001">
<claim-text>An audio signal encoding method, the method comprising the steps of:
<claim-text>- collecting the audio signal samples (114),</claim-text>
<claim-text>- determining sinusoidal components (312) in subsequent frames,</claim-text>
<claim-text>- estimation of amplitudes (314) and frequencies (313) of the components for each frame,</claim-text>
<claim-text>- merging thus obtained pairs into sinusoidal trajectories,</claim-text>
<claim-text>- splitting trajectories into segments,</claim-text>
<claim-text>- transforming (318, 319) the trajectories to the frequency domain by means of a digital transform performed on segments longer than the frame duration,</claim-text>
<claim-text>- quantization (320, 321) and selection (322, 323) of transform coefficients in the segments,</claim-text>
<claim-text>- entropy encoding (328), and</claim-text>
<claim-text>- outputting the quantized coefficients as output data (115),<br/>
wherein:</claim-text>
<claim-text>- when the method is an audio signal encoding method for stereo or multichannel encoding, the trajectories of the channels are grouped and only the presence of the trajectories is signaled in a header.</claim-text></claim-text></claim>
<claim id="c-en-01-0002" num="0002">
<claim-text>The audio signal encoding method according to claim 1,<br/>
wherein
<claim-text>- segments of different trajectories starting within a particular time are grouped into Groups of Segments, GOS, and</claim-text>
<claim-text>- the partitioning of trajectories into segments is synchronized with the endpoints of a Group of Segments, GOS.</claim-text></claim-text></claim>
<claim id="c-en-01-0003" num="0003">
<claim-text>The audio signal encoding method according to claim 2, wherein the segments length is adjusted by extrapolation to synchronize the partitioning of trajectories with the endpoints of the GOS.</claim-text></claim>
<claim id="c-en-01-0004" num="0004">
<claim-text>The audio signal encoding method according to claim 2 or 3, wherein the length of a group of segments is limited to eight frames.<!-- EPO <DP n="29"> --></claim-text></claim>
<claim id="c-en-01-0005" num="0005">
<claim-text>The audio signal encoding method according to any one of claims 2 to 4, wherein the audio signal encoding method is used for high frequency sinusoidal coding, HFSC, for example for HFSC according to the MPEG-H 3D codec.</claim-text></claim>
<claim id="c-en-01-0006" num="0006">
<claim-text>The audio signal encoding method according to any one of claims 1 to 5, wherein clusters of segments belonging to harmonic structures of a sound source are jointly encoded, clusters representing a fundamental frequency of each harmonic structure and its integer multiplications.</claim-text></claim>
<claim id="c-en-01-0007" num="0007">
<claim-text>An audio signal encoding apparatus configured to:
<claim-text>- collect the audio signal samples (114),</claim-text>
<claim-text>- determine sinusoidal components (312) in subsequent frames,</claim-text>
<claim-text>- estimate amplitudes (314) and frequencies (313) of the components for each frame,</claim-text>
<claim-text>- merge thus obtained pairs into sinusoidal trajectories,</claim-text>
<claim-text>- split trajectories into segments,</claim-text>
<claim-text>- transform (318, 319) the trajectories to the frequency domain by means of a digital transform performed on segments longer than the frame duration,</claim-text>
<claim-text>- quantize (320, 321) and select (322, 323) transform coefficients in the segments,</claim-text>
<claim-text>- entropy encode (328), and</claim-text>
<claim-text>- output the quantized coefficients as output data (115),<br/>
wherein:</claim-text>
<claim-text>- when the apparatus is an audio signal encoding apparatus for stereo or multichannel encoding, the trajectories of the channels are grouped and only the presence of the trajectories is signaled in a header.</claim-text></claim-text></claim>
<claim id="c-en-01-0008" num="0008">
<claim-text>The audio signal encoding apparatus according to claim 7,<br/>
wherein
<claim-text>- segments of different trajectories starting within a particular time are grouped into Groups of Segments, GOS, and</claim-text>
<claim-text>- the partitioning of trajectories into segments is synchronized with the endpoints of a Group of Segments, GOS.</claim-text></claim-text></claim>
<claim id="c-en-01-0009" num="0009">
<claim-text>The audio signal encoding apparatus according to claim 7 or claim 8, wherein clusters of segments belonging to harmonic structures of a sound source are jointly encoded, clusters<!-- EPO <DP n="30"> --> representing a fundamental frequency of each harmonic structure and its integer multiplications.</claim-text></claim>
<claim id="c-en-01-0010" num="0010">
<claim-text>An audio signal decoding method comprising the steps of:
<claim-text>- retrieving encoded data,</claim-text>
<claim-text>- reconstruction (411, 412, 413, 414, 415) from the encoded data digital transform coefficients of trajectories' segments,</claim-text>
<claim-text>- subjecting the coefficients to an inverse transform (416, 417) and performing reconstruction of the trajectories' segments,</claim-text>
<claim-text>- for each trajectory, generation (420, 421) of a sinusoidal components having an amplitude and a frequency corresponding to the trajectory,</claim-text>
<claim-text>- reconstruction of the audio signal by summation of the sinusoidal components,<br/>
wherein</claim-text>
<claim-text>- when the method is an audio signal decoding method for stereo or multichannel decoding, the trajectories of the channels are grouped and only the presence of the trajectories is signaled in a header in the retrieved encoded data.</claim-text></claim-text></claim>
<claim id="c-en-01-0011" num="0011">
<claim-text>The audio signal decoding method according to claim 10, wherein:
<claim-text>- segments of different trajectories starting within a particular time are grouped into Groups of Segments, GOS, and</claim-text>
<claim-text>- the partitioning of trajectories into segments is synchronized with the endpoints of a Group of Segments, GOS.</claim-text></claim-text></claim>
<claim id="c-en-01-0012" num="0012">
<claim-text>The audio signal decoding method according to claim 10 or claim 11, wherein clusters of segments belonging to harmonic structures of a sound source are jointly encoded in the retrieved encoded data, clusters representing a fundamental frequency of each harmonic structure and its integer multiplications.</claim-text></claim>
<claim id="c-en-01-0013" num="0013">
<claim-text>An audio signal decoding apparatus configured to:
<claim-text>- retrieve encoded data,</claim-text>
<claim-text>- reconstruct (411, 412, 413, 414, 415) from the encoded data digital transform coefficients of trajectories' segments,</claim-text>
<claim-text>- subject the coefficients to an inverse transform (416, 417) and perform reconstruction of the trajectories' segments,<!-- EPO <DP n="31"> --></claim-text>
<claim-text>- for each trajectory, generate (420, 421) a sinusoidal component having an amplitude and a frequency corresponding to the trajectory,</claim-text>
<claim-text>- reconstruct the audio signal by summation of the sinusoidal components,<br/>
wherein</claim-text>
<claim-text>- when the method is an audio signal decoding method for stereo or multichannel decoding, the trajectories of the channels are grouped and only the presence of the trajectories is signaled in a header in the retrieved encoded data.</claim-text></claim-text></claim>
<claim id="c-en-01-0014" num="0014">
<claim-text>The audio signal decoding apparatus according to claim 13, wherein:
<claim-text>- segments of different trajectories starting within a particular time are grouped into Groups of Segments (GOS), and</claim-text>
<claim-text>- the partitioning of trajectories into segments is synchronized with the endpoints of a Group of Segments (GOS).</claim-text></claim-text></claim>
<claim id="c-en-01-0015" num="0015">
<claim-text>The audio signal decoding apparatus according to claim 13 or claim 14, wherein clusters of segments belonging to harmonic structures of a sound source are jointly encoded in the retrieved encoded data, clusters representing a fundamental frequency of each harmonic structure and its integer multiplications.</claim-text></claim>
</claims>
<claims id="claims02" lang="de"><!-- EPO <DP n="32"> -->
<claim id="c-de-01-0001" num="0001">
<claim-text>Audiosignalcodierverfahren, wobei das Verfahren die folgenden Schritte umfasst:
<claim-text>- Sammeln der Audiosignalproben (114),</claim-text>
<claim-text>- Bestimmen von sinusförmigen Komponenten (312) in den nachfolgenden "Frames",</claim-text>
<claim-text>- Schätzen der Amplituden (314) und Frequenzen (313) der Komponenten für jedes "Frame",</claim-text>
<claim-text>- Zusammenfügen der so erhaltenen Paare zu sinusförmigen Bewegungsbahnen,</claim-text>
<claim-text>- Teilen der Bewegungsbahnen in Segmente,</claim-text>
<claim-text>- Transformieren (318, 319) der Bewegungsbahnen in den Frequenzbereich mittels einer digitalen Transformation, die an Segmenten durchgeführt wird, die länger als die "Frame"-Dauer sind,</claim-text>
<claim-text>- Quantisierung (320, 321) und Auswahl (322, 323) der Transformationskoeffizienten in den Segmenten,</claim-text>
<claim-text>- Entropiecodierung (328), und</claim-text>
<claim-text>- Ausgeben der quantisierten Koeffizienten als Ausgangsdaten (115),<br/>
wobei:</claim-text>
<claim-text>- wenn das Verfahren ein Audiosignalcodierverfahren für Stereo- oder Mehrkanalcodierung ist, die Bewegungsbahnen der Kanäle gruppiert werden und nur das Vorhandensein der Bewegungsbahnen in einem Kopfabschnitt signalisiert wird.</claim-text></claim-text></claim>
<claim id="c-de-01-0002" num="0002">
<claim-text>Audiosignalcodierverfahren gemäß Anspruch 1,<br/>
wobei
<claim-text>- Segmente verschiedener Bewegungsbahnen, die innerhalb eines bestimmten Zeitraums beginnen, in Segmentgruppen, GOS, gruppiert werden und</claim-text>
<claim-text>- das Unterteilen der Bewegungsbahnen in Segmente mit den Endpunkten einer Segmentgruppe, GOS, synchronisiert wird.</claim-text></claim-text></claim>
<claim id="c-de-01-0003" num="0003">
<claim-text>Audiosignalcodierverfahren gemäß Anspruch 2, wobei die Segmentlänge durch Extrapolation angepasst wird, um das Unterteilen der Bewegungsbahnen mit den Endpunkten der GOS zu synchronisieren.</claim-text></claim>
<claim id="c-de-01-0004" num="0004">
<claim-text>Audiosignalcodierverfahren gemäß Anspruch 2 oder 3, wobei die Länge einer Segmentgruppe auf acht "Frames" begrenzt ist.<!-- EPO <DP n="33"> --></claim-text></claim>
<claim id="c-de-01-0005" num="0005">
<claim-text>Audiosignalcodierverfahren gemäß einem der Ansprüche 2 bis 4, wobei das Audiosignalcodierverfahren für die Hochfrequenz-Sinus-Codierung, HFSC, verwendet wird, zum Beispiel für die HFSC gemäß dem MPEG-H 3D-Codec.</claim-text></claim>
<claim id="c-de-01-0006" num="0006">
<claim-text>Audiosignalcodierverfahren gemäß einem der Ansprüche 1 bis 5, wobei Cluster von Segmenten, die zu harmonischen Strukturen einer Schallquelle gehören, gemeinsam codiert werden, wobei die "Cluster" eine Grundfrequenz jeder harmonischen Struktur und ihre ganzzahligen Multiplikationen darstellen.</claim-text></claim>
<claim id="c-de-01-0007" num="0007">
<claim-text>Audiosignalcodiervorrichtung, die dafür konfiguriert ist:
<claim-text>- die Audiosignalproben zu sammeln (114),</claim-text>
<claim-text>- die sinusförmigen Komponenten (312) in den nachfolgenden "Frames" zu bestimmen,</claim-text>
<claim-text>- die Amplituden (314) und Frequenzen (313) der Komponenten für jedes "Frame" zu schätzen,</claim-text>
<claim-text>- die so erhaltenen Paare zu sinusförmigen Bewegungsbahnen zusammenzufügen,</claim-text>
<claim-text>- Bewegungsbahnen in Segmente zu unterteilen,</claim-text>
<claim-text>- die Bewegungsbahnen in den Frequenzbereich mittels einer digitalen Transformation zu transformieren (318, 319), die an Segmenten durchgeführt wird, die länger als die "Frame"-Dauer sind,</claim-text>
<claim-text>- Transformationskoeffizienten in den Segmenten zu quantisieren (320, 321) und auszuwählen (322, 323),</claim-text>
<claim-text>- Entropiecodierung vorzunehmen (328), und</claim-text>
<claim-text>- die quantisierten Koeffizienten als Ausgangsdaten auszugeben (115),<br/>
wobei:</claim-text>
<claim-text>- wenn die Vorrichtung eine Audiosignalcodiervorrichtung für Stereo- oder Mehrkanalcodierung ist, die Bewegungsbahnen der Kanäle gruppiert werden und nur das Vorhandensein der Bewegungsbahnen in einem Kopfabschnitt signalisiert wird.</claim-text></claim-text></claim>
<claim id="c-de-01-0008" num="0008">
<claim-text>Audiosignalcodiervorrichtung gemäß Anspruch 7,<br/>
wobei
<claim-text>- Segmente verschiedener Bewegungsbahnen, die innerhalb eines bestimmten Zeitraums beginnen, in Segmentgruppen, GOS, gruppiert werden und</claim-text>
<claim-text>- das Unterteilen der Bewegungsbahnen in Segmente mit den Endpunkten einer Segmentgruppe, GOS, synchronisiert wird.</claim-text><!-- EPO <DP n="34"> --></claim-text></claim>
<claim id="c-de-01-0009" num="0009">
<claim-text>Audiosignalcodiervorrichtung gemäß Anspruch 7 oder 8, wobei "Cluster" von Segmenten, die zu harmonischen Strukturen einer Schallquelle gehören, gemeinsam codiert werden, wobei die "Cluster" eine Grundfrequenz jeder harmonischen Struktur und ihre ganzzahligen Multiplikationen darstellen.</claim-text></claim>
<claim id="c-de-01-0010" num="0010">
<claim-text>Audiosignaldecodierverfahren, umfassend die folgenden Schritte:
<claim-text>- Abrufen codierter Daten,</claim-text>
<claim-text>- Rekonstruktion (411, 412, 413, 414, 415) der digitalen Transformationskoeffizienten der Segmente der Bewegungsbahn aus den codierten Daten,</claim-text>
<claim-text>- Unterziehen der Koeffizienten einer inversen Transformation (416, 417) und Durchführen einer Rekonstruktion der Segmente der Bewegungsbahnen,</claim-text>
<claim-text>- für jede Bewegungsbahn Erzeugung (420, 421) einer sinusförmigen Komponente mit einer Amplitude und einer Frequenz, die der Bewegungsbahn entsprechen,</claim-text>
<claim-text>- Rekonstruktion des Audiosignals durch Summierung der sinusförmigen Komponenten,<br/>
wobei</claim-text>
<claim-text>- wenn das Verfahren ein Audiosignaldecodierverfahren für Stereo- oder Mehrkanaldecodierung ist, die Bewegungsbahnen der Kanäle gruppiert werden und nur das Vorhandensein der Bewegungsbahnen in einem Kopfabschnitt der abgerufenen codierten Daten signalisiert wird.</claim-text></claim-text></claim>
<claim id="c-de-01-0011" num="0011">
<claim-text>Audiosignaldecodierverfahren gemäß Anspruch 10, wobei:
<claim-text>- Segmente verschiedener Bewegungsbahnen, die innerhalb eines bestimmten Zeitraums beginnen, in Segmentgruppen, GOS, gruppiert werden und</claim-text>
<claim-text>- das Unterteilen der Bewegungsbahnen in Segmente mit den Endpunkten einer Segmentgruppe, GOS, synchronisiert wird.</claim-text></claim-text></claim>
<claim id="c-de-01-0012" num="0012">
<claim-text>Audiosignaldecodierverfahren gemäß Anspruch 10 oder 11, wobei "Cluster" von Segmenten, die zu harmonischen Strukturen einer Schallquelle gehören, gemeinsam in den abgerufenen codierten Daten codiert werden, wobei die "Cluster" eine Grundfrequenz jeder harmonischen Struktur und ihre ganzzahligen Multiplikationen darstellen.</claim-text></claim>
<claim id="c-de-01-0013" num="0013">
<claim-text>Audiosignaldecodiervorrichtung, die dafür konfiguriert ist:
<claim-text>- codierte Daten abzurufen,<!-- EPO <DP n="35"> --></claim-text>
<claim-text>- digitale Transformationskoeffizienten der Bewegungsbahnsegmente aus den codierten Daten zu rekonstruieren (411, 412, 413, 414, 415),</claim-text>
<claim-text>- die Koeffizienten einer inversen Transformation (416, 417) zu unterziehen und eine Rekonstruktion der Segmente der Bewegungsbahnen vorzunehmen,</claim-text>
<claim-text>- für jede Bewegungsbahn eine sinusförmige Komponente mit einer Amplitude und einer Frequenz entsprechend der Bewegungsbahn zu erzeugen (420, 421),</claim-text>
<claim-text>- das Audiosignal durch Summierung der sinusförmigen Komponenten zu rekonstruieren,<br/>
wobei</claim-text>
<claim-text>- wenn das Verfahren ein Audiosignaldecodierverfahren für Stereo- oder Mehrkanaldecodierung ist, die Bewegungsbahnen der Kanäle gruppiert werden und nur das Vorhandensein der Bewegungsbahnen in einem Kopfabschnitt der abgerufenen codierten Daten signalisiert wird.</claim-text></claim-text></claim>
<claim id="c-de-01-0014" num="0014">
<claim-text>Audiosignaldecodiervorrichtung gemäß Anspruch 13, wobei:
<claim-text>- Segmente verschiedener Bewegungsbahnen, die innerhalb eines bestimmten Zeitraums beginnen, in Segmentgruppen (GOS) gruppiert werden und</claim-text>
<claim-text>- das Unterteilen der Bewegungsbahnen in Segmente mit den Endpunkten einer Segmentgruppe (GOS) synchronisiert wird.</claim-text></claim-text></claim>
<claim id="c-de-01-0015" num="0015">
<claim-text>Audiosignaldecodiervorrichtung gemäß Anspruch 13 oder 14, wobei "Cluster" von Segmenten, die zu harmonischen Strukturen einer Schallquelle gehören, gemeinsam in den abgerufenen codierten Daten codiert werden, wobei die "Cluster" eine Grundfrequenz jeder harmonischen Struktur und ihre ganzzahligen Multiplikationen darstellen.</claim-text></claim>
</claims>
<claims id="claims03" lang="fr"><!-- EPO <DP n="36"> -->
<claim id="c-fr-01-0001" num="0001">
<claim-text>Procédé d'encodage de signal audio, le procédé comprenant les étapes de :
<claim-text>- la collecte des échantillons de signal audio (114),</claim-text>
<claim-text>- la détermination de composantes sinusoïdales (312) dans des trames subséquentes,</claim-text>
<claim-text>- l'estimation d'amplitudes (314) et de fréquences (313) des composantes pour chaque trame,</claim-text>
<claim-text>- la fusion de paires ainsi obtenues en trajectoires sinusoïdales,</claim-text>
<claim-text>- la division de trajectoires en segments,</claim-text>
<claim-text>- la transformation (318, 319) des trajectoires en le domaine de fréquence au moyen d'une transformée numérique réalisée sur des segments plus longs que la durée de trame,</claim-text>
<claim-text>- la quantification (320, 321) et la sélection (322, 323) de coefficients de transformée dans les segments,</claim-text>
<claim-text>- l'encodage entropique (328), et</claim-text>
<claim-text>- la sortie des coefficients quantifiés sous forme de données de sortie (115),<br/>
dans lequel :</claim-text>
<claim-text>- lorsque le procédé est un procédé d'encodage de signal audio pour encodage stéréo ou multicanal, les trajectoires des canaux sont groupées et seulement la présence des trajectoires est signalée dans un en-tête.</claim-text></claim-text></claim>
<claim id="c-fr-01-0002" num="0002">
<claim-text>Procédé d'encodage de signal audio selon la revendication 1, dans lequel
<claim-text>- des segments de différentes trajectoires commençant au sein d'un temps particulier sont groupés en Groupes de Segments, GOS, et</claim-text>
<claim-text>- le partitionnement de trajectoires en segments est synchronisé avec les points terminaux d'un Groupe de Segments, GOS.</claim-text></claim-text></claim>
<claim id="c-fr-01-0003" num="0003">
<claim-text>Procédé d'encodage de signal audio selon la revendication 2, dans lequel la longueur des segments est ajustée par extrapolation pour synchroniser le partitionnement de trajectoires avec les points terminaux du GOS.</claim-text></claim>
<claim id="c-fr-01-0004" num="0004">
<claim-text>Procédé d'encodage de signal audio selon la revendication 2 ou 3, dans lequel la longueur d'un groupe de segments est limitées à huit trames.<!-- EPO <DP n="37"> --></claim-text></claim>
<claim id="c-fr-01-0005" num="0005">
<claim-text>Procédé d'encodage de signal audio selon l'une quelconque des revendications 2 à 4, dans lequel le procédé d'encodage de signal audio est utilisé pour un codage sinusoïdal haute fréquence, HFSC, par exemple pour HFSC selon le codec MPEG-H 3D.</claim-text></claim>
<claim id="c-fr-01-0006" num="0006">
<claim-text>Procédé d'encodage de signal audio selon l'une quelconque des revendications 1 à 5, dans lequel des groupements de segments appartenant à des structures harmoniques d'une source de son sont encodés conjointement, des groupements représentant une fréquence fondamentale de chaque structure harmonique et ses multiplications entières.</claim-text></claim>
<claim id="c-fr-01-0007" num="0007">
<claim-text>Appareil d'encodage de signal audio, configuré pour :
<claim-text>- collecter les échantillons de signal audio (114),</claim-text>
<claim-text>- déterminer des composantes sinusoïdales (312) dans des trames subséquentes,</claim-text>
<claim-text>- estimer des amplitudes (314) et des fréquences (313) des composantes pour chaque trame,</claim-text>
<claim-text>- fusionner des paires ainsi obtenues en trajectoires sinusoïdales,</claim-text>
<claim-text>- diviser des trajectoires en segments,</claim-text>
<claim-text>- transformer (318, 319) les trajectoires en le domaine de fréquence au moyen d'une transformée numérique réalisée sur des segments plus longs que la durée de trame,</claim-text>
<claim-text>- quantifier (320, 321) et sélectionner (322, 323) des coefficients de transformée dans les segments,</claim-text>
<claim-text>- encoder entropiquement (328), et</claim-text>
<claim-text>- sortir les coefficients quantifiés sous forme de données de sortie (115),<br/>
dans lequel :</claim-text>
<claim-text>- lorsque l'appareil est un appareil d'encodage de signal audio pour encodage stéréo ou multicanal, les trajectoires des canaux sont groupées et seulement la présence des trajectoires est signalée dans un en-tête.</claim-text></claim-text></claim>
<claim id="c-fr-01-0008" num="0008">
<claim-text>Appareil d'encodage de signal audio selon la revendication 7, dans lequel
<claim-text>- des segments de différentes trajectoires commençant au sein d'un temps particulier sont groupés en Groupes de Segments, GOS, et</claim-text>
<claim-text>- le partitionnement de trajectoires en segments est synchronisé avec les points terminaux d'un Groupe de Segments, GOS.</claim-text></claim-text></claim>
<claim id="c-fr-01-0009" num="0009">
<claim-text>Appareil d'encodage de signal audio selon la revendication 7 ou la revendication 8, dans lequel des groupements de segments appartenant à des structures harmoniques<!-- EPO <DP n="38"> --> d'une source de son sont encodés conjointement, des groupements représentant une fréquence fondamentale de chaque structure harmonique et ses multiplications entières.</claim-text></claim>
<claim id="c-fr-01-0010" num="0010">
<claim-text>Procédé de décodage de signal audio, comprenant les étapes de :
<claim-text>- la récupération de données encodées,</claim-text>
<claim-text>- la reconstruction (411, 412, 413, 414, 415) à partir des données encodées, des coefficients de transformée numérique de segments de trajectoires,</claim-text>
<claim-text>- la soumission des coefficients à une transformée inverse (416, 417) et la réalisation d'une reconstruction des segments de trajectoires,</claim-text>
<claim-text>- pour chaque trajectoire, la génération (420,421) d'une composante sinusoïdale ayant une amplitude et une fréquence correspondant à la trajectoire,</claim-text>
<claim-text>- la reconstruction du signal audio par sommation des composantes sinusoïdales, dans lequel</claim-text>
<claim-text>- lorsque le procédé est un procédé de décodage de signal audio pour décodage stéréo ou multicanal, les trajectoires des canaux sont groupées et seulement la présence des trajectoires est signalée dans un en-tête dans les données encodées récupérées.</claim-text></claim-text></claim>
<claim id="c-fr-01-0011" num="0011">
<claim-text>Procédé de décodage de signal audio selon la revendication 10, dans lequel :
<claim-text>- des segments de différentes trajectoires commençant au sein d'un temps particulier sont groupés en Groupes de Segments, GOS, et</claim-text>
<claim-text>- le partitionnement de trajectoires en segments est synchronisé avec les points terminaux d'un Groupe de Segments, GOS.</claim-text></claim-text></claim>
<claim id="c-fr-01-0012" num="0012">
<claim-text>Procédé de décodage de signal audio selon la revendication 10 ou la revendication 11, dans lequel des groupements de segments appartenant à des structures harmoniques d'une source de son sont encodés conjointement dans les données encodées récupérées, des groupements représentant une fréquence fondamentale de chaque structure harmonique et ses multiplications entières.</claim-text></claim>
<claim id="c-fr-01-0013" num="0013">
<claim-text>Appareil de décodage de signal audio, configuré pour :
<claim-text>- récupérer des données encodées,</claim-text>
<claim-text>- reconstruire (411, 412, 413, 414, 415), à partir des données encodées, des coefficients de transformée numérique de segments de trajectoires,</claim-text>
<claim-text>- soumettre les coefficients à une transformée inverse (416, 417) et réaliser une reconstruction des segments de trajectoires,<!-- EPO <DP n="39"> --></claim-text>
<claim-text>- pour chaque trajectoire, générer (420, 421) une composante sinusoïdale ayant une amplitude et une fréquence correspondant à la trajectoire,</claim-text>
<claim-text>- reconstruire le signal audio par sommation des composantes sinusoïdales, dans lequel</claim-text>
<claim-text>- lorsque le procédé est un procédé de décodage de signal audio pour décodage stéréo ou multicanal, les trajectoires des canaux sont groupées et seulement la présence des trajectoires est signalée dans un en-tête dans les données encodées récupérées.</claim-text></claim-text></claim>
<claim id="c-fr-01-0014" num="0014">
<claim-text>Appareil de décodage de signal audio selon la revendication 13, dans lequel :
<claim-text>- des segments de différentes trajectoires commençant au sein d'un temps particulier sont groupés en Groupes de Segments (GOS), et</claim-text>
<claim-text>- le partitionnement de trajectoires en segments est synchronisé avec les points terminaux d'un Groupe de Segments (GOS).</claim-text></claim-text></claim>
<claim id="c-fr-01-0015" num="0015">
<claim-text>Appareil de décodage de signal audio selon la revendication 13 ou la revendication 14, dans lequel des groupements de segments appartenant à des structures harmoniques d'une source de son sont encodés conjointement dans les données encodées récupérées, des groupements représentant une fréquence fondamentale de chaque structure harmonique et ses multiplications entières.</claim-text></claim>
</claims>
<drawings id="draw" lang="en"><!-- EPO <DP n="40"> -->
<figure id="f0001" num="1"><img id="if0001" file="imgf0001.tif" wi="133" he="233" img-content="drawing" img-format="tif"/></figure><!-- EPO <DP n="41"> -->
<figure id="f0002" num="2"><img id="if0002" file="imgf0002.tif" wi="144" he="233" img-content="drawing" img-format="tif"/></figure><!-- EPO <DP n="42"> -->
<figure id="f0003" num="3"><img id="if0003" file="imgf0003.tif" wi="130" he="233" img-content="drawing" img-format="tif"/></figure><!-- EPO <DP n="43"> -->
<figure id="f0004" num="4a"><img id="if0004" file="imgf0004.tif" wi="138" he="225" img-content="drawing" img-format="tif"/></figure><!-- EPO <DP n="44"> -->
<figure id="f0005" num="4b"><img id="if0005" file="imgf0005.tif" wi="127" he="233" img-content="drawing" img-format="tif"/></figure><!-- EPO <DP n="45"> -->
<figure id="f0006" num="5"><img id="if0006" file="imgf0006.tif" wi="150" he="233" img-content="drawing" img-format="tif"/></figure><!-- EPO <DP n="46"> -->
<figure id="f0007" num="6"><img id="if0007" file="imgf0007.tif" wi="150" he="233" img-content="drawing" img-format="tif"/></figure><!-- EPO <DP n="47"> -->
<figure id="f0008" num="7"><img id="if0008" file="imgf0008.tif" wi="157" he="233" img-content="drawing" img-format="tif"/></figure><!-- EPO <DP n="48"> -->
<figure id="f0009" num="8"><img id="if0009" file="imgf0009.tif" wi="138" he="232" img-content="drawing" img-format="tif"/></figure><!-- EPO <DP n="49"> -->
<figure id="f0010" num="9"><img id="if0010" file="imgf0010.tif" wi="136" he="199" img-content="drawing" img-format="tif"/></figure><!-- EPO <DP n="50"> -->
<figure id="f0011" num="10"><img id="if0011" file="imgf0011.tif" wi="165" he="219" img-content="drawing" img-format="tif"/></figure><!-- EPO <DP n="51"> -->
<figure id="f0012" num="11"><img id="if0012" file="imgf0012.tif" wi="145" he="217" img-content="drawing" img-format="tif"/></figure><!-- EPO <DP n="52"> -->
<figure id="f0013" num="12"><img id="if0013" file="imgf0013.tif" wi="133" he="208" img-content="drawing" img-format="tif"/></figure><!-- EPO <DP n="53"> -->
<figure id="f0014" num="13a,13b,13c"><img id="if0014" file="imgf0014.tif" wi="165" he="233" img-content="drawing" img-format="tif"/></figure><!-- EPO <DP n="54"> -->
<figure id="f0015" num="14a,14b"><img id="if0015" file="imgf0015.tif" wi="158" he="215" img-content="drawing" img-format="tif"/></figure>
</drawings>
<ep-reference-list id="ref-list">
<heading id="ref-h0001"><b>REFERENCES CITED IN THE DESCRIPTION</b></heading>
<p id="ref-p0001" num=""><i>This list of references cited by the applicant is for the reader's convenience only. It does not form part of the European patent document. Even though great care has been taken in compiling the references, errors or omissions cannot be excluded and the EPO disclaims all liability in this regard.</i></p>
<heading id="ref-h0002"><b>Patent documents cited in the description</b></heading>
<p id="ref-p0002" num="">
<ul id="ref-ul0001" list-style="bullet">
<li><patcit id="ref-pcit0001" dnum="AU2011205144A1"><document-id><country>AU</country><doc-number>2011205144</doc-number><kind>A1</kind></document-id></patcit><crossref idref="pcit0001">[0003]</crossref></li>
<li><patcit id="ref-pcit0002" dnum="US2005078832A1"><document-id><country>US</country><doc-number>2005078832</doc-number><kind>A1</kind></document-id></patcit><crossref idref="pcit0002">[0004]</crossref></li>
<li><patcit id="ref-pcit0003" dnum="US2007238415A1"><document-id><country>US</country><doc-number>2007238415</doc-number><kind>A1</kind></document-id></patcit><crossref idref="pcit0003">[0005]</crossref></li>
</ul></p>
<heading id="ref-h0003"><b>Non-patent literature cited in the description</b></heading>
<p id="ref-p0003" num="">
<ul id="ref-ul0002" list-style="bullet">
<li><nplcit id="ref-ncit0001" npl-type="s"><article><author><name>TOMASZ ZERNICKI et al.</name></author><atl/><serial><sertitle>Updated MPEG-H 3D Audio Phase 2 Core Experiment Proposal on tonal component coding</sertitle></serial></article></nplcit><crossref idref="ncit0001">[0003]</crossref></li>
<li><nplcit id="ref-ncit0002" npl-type="s"><article><atl>MPEG-H 3D Audio Phase 2 Core Experiment Proposal on tonal component coding</atl><serial><sertitle>111th MPEG Meeting</sertitle><pubdate><sdate>20150200</sdate><edate/></pubdate></serial></article></nplcit><crossref idref="ncit0002">[0088]</crossref></li>
<li><nplcit id="ref-ncit0003" npl-type="s"><article><atl>Updated MPEG-H 3D Audio Phase 2 Core Experiment Proposal on tonal component coding</atl><serial><sertitle>112th MPEG Meeting</sertitle><pubdate><sdate>20150600</sdate><edate/></pubdate></serial></article></nplcit><crossref idref="ncit0003">[0088]</crossref></li>
<li><nplcit id="ref-ncit0004" npl-type="s"><article><atl>Zylia Listening Test Report on High Frequency Tonal Component Coding CE</atl><serial><sertitle>113th MPEG Meeting</sertitle><pubdate><sdate>20151000</sdate><edate/></pubdate></serial></article></nplcit><crossref idref="ncit0004">[0088]</crossref></li>
<li><nplcit id="ref-ncit0005" npl-type="s"><article><author><name>ZERNICKI T.</name></author><author><name>BARTKOWIAK M.</name></author><author><name>JANUSZKIEWICZ L.</name></author><author><name>CHRYSZCZANOWICZ M.</name></author><atl>Application of sinusoidal coding for enhanced bandwidth extension in MPEG-D USAC</atl><serial><sertitle>Convention paper presented at the 138th AES Convention, Warsaw</sertitle></serial></article></nplcit><crossref idref="ncit0005">[0088]</crossref></li>
<li><nplcit id="ref-ncit0006" npl-type="s"><article><atl>Workplan on 3D Audio</atl><serial><sertitle>112th MPEG Meeting</sertitle><pubdate><sdate>20150600</sdate><edate/></pubdate></serial></article></nplcit><crossref idref="ncit0006">[0088]</crossref></li>
<li><nplcit id="ref-ncit0007" npl-type="s"><article><author><name>TOMASZ ZERNICKI</name></author><author><name>MACIEJ BARTKOWIAK</name></author><author><name>MAREK DOMANSKI</name></author><atl>Enhanced coding of high-frequency tonal components in MPEG-D USAC through joint application of eSBR and sinusoidal modeling</atl><serial><sertitle>ICASSP 2011</sertitle><pubdate><sdate>20110000</sdate><edate/></pubdate></serial><location><pp><ppf>501</ppf><ppl>504</ppl></pp></location></article></nplcit><crossref idref="ncit0007">[0088]</crossref></li>
<li><nplcit id="ref-ncit0008" npl-type="s"><article><author><name>TOMASZ ZERNICKI</name></author><author><name>MACIEJ BARTKOWIAK</name></author><author><name>LUKASZ JANUSZKIEWICZ</name></author><author><name>MARCIN CHRYSZCZANOWICZ</name></author><atl>Application of sinusoidal coding for enhanced bandwidth extension in MPEG-D USAC</atl><serial><sertitle>Audio Engineering Society 138th Convention, Warsaw, Poland</sertitle><pubdate><sdate>20150500</sdate><edate/></pubdate></serial></article></nplcit><crossref idref="ncit0008">[0088]</crossref></li>
</ul></p>
</ep-reference-list>
</ep-patent-document>
